diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml index 6430c3f4793..afcaf3c066e 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml @@ -91,7 +91,11 @@ jobs: test -n "${CAPACITY_SERVICE_ACCOUNT}" test -n "${DIRECTOR_RUNTIME_SERVICE_ACCOUNT}" + # Full history: the monitor evidence this job verifies is sealed at an ancestor commit, + # and the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: pnpm/action-setup@v4 with: { package_json_file: cloud/package.json } diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap.yml b/.github/workflows/cloud-deploy-relay-production-same-cap.yml index 1994966d083..fba5df0dcb9 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap.yml @@ -87,13 +87,18 @@ jobs: gate: if: ${{ vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true' && (github.ref == 'refs/heads/main') }} runs-on: blacksmith-2vcpu-ubuntu-2204 - timeout-minutes: 10 + # Headroom for the full-history checkout the canary provenance check needs. + timeout-minutes: 15 environment: production outputs: cells: ${{ steps.wave.outputs.cells }} job-mode: ${{ steps.wave.outputs.job-mode }} steps: + # Full history: the canary authority a batch verifies is sealed at an ancestor commit, and + # the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: actions/setup-node@v4 with: { node-version: 24 } diff --git a/.github/workflows/cloud-operate-relay-production-rehome-job.yml b/.github/workflows/cloud-operate-relay-production-rehome-job.yml index 682953af5e7..fdb1aca45e0 100644 --- a/.github/workflows/cloud-operate-relay-production-rehome-job.yml +++ b/.github/workflows/cloud-operate-relay-production-rehome-job.yml @@ -95,7 +95,11 @@ jobs: ;; esac + # Full history: the monitor evidence this job verifies is sealed at an ancestor commit, + # and the provenance check fails closed on a commit a shallow clone left out. - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: actions/setup-node@v4 with: diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index ca1651325c9..a749214e232 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -807,6 +807,7 @@ jobs: src/main/agent-hooks/windows-hook-payload-delivery.test.ts src/main/windows/windows-pty-job.win32.test.ts src/main/windows/windows-host-job.win32.test.ts + src/main/windows-live-tree-kill.win32.test.ts src/main/wsl/wsl-runner.test.ts src/main/wsl/wsl-guest-environment.test.ts src/main/wsl/wsl-invocation-boundary.test.ts diff --git a/cloud/apps/relay-ops/src/resource-inventory.test.ts b/cloud/apps/relay-ops/src/resource-inventory.test.ts index 6d8b3070c98..e2cfa13dccb 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.test.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.test.ts @@ -13,6 +13,41 @@ const runService = { latestReadyRevision: 'projects/project/revisions/revision-one' } +const sleepingStagingGcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } + +// Staging's Cloud SQL is stopped, so this inventory reads REST only and probes no endpoint. +type MigOutcome = 'ok' | 'throw' | 'missing' +const sleepingStagingFetch = (migOutcome: (migName: string) => MigOutcome): typeof fetch => + async (input) => { + const url = new URL(String(input)) + if (url.hostname === 'run.googleapis.com') return Response.json(runService) + if (url.hostname === 'sqladmin.googleapis.com') return Response.json({ + state: 'STOPPED', + databaseVersion: 'POSTGRES_17', + settings: { activationPolicy: 'NEVER', availabilityType: 'ZONAL', tier: 'db-custom-1-3840' } + }) + if (url.hostname === 'certificatemanager.googleapis.com') return Response.json({ + managed: { domains: ['*.relay-staging.onorca.dev'], state: 'ACTIVE' } + }) + if (url.pathname.includes('/instanceGroupManagers/')) { + const name = url.pathname.split('/').at(-1)! + const outcome = migOutcome(name) + if (outcome === 'throw') throw new TypeError('fetch failed') + if (outcome === 'missing') return new Response(null, { status: 404 }) + return Response.json({ + name, + targetSize: 0, + size: '0', + instanceGroup: `projects/project/zones/zone/instanceGroups/${name}`, + instanceTemplate: `projects/project/global/instanceTemplates/template-${name}`, + status: { isStable: true } + }) + } + if (url.pathname.includes('/instanceTemplates/')) return Response.json({ properties: {} }) + if (url.pathname.endsWith('/getHealth')) return Response.json([]) + throw new Error(`Unexpected request to ${url.hostname}${url.pathname}`) + } + describe('readResourceInventory', () => { it('does not delay a healthy endpoint sample', async () => { let calls = 0 @@ -23,8 +58,10 @@ describe('readResourceInventory', () => { calls += 1 return new Response(null, { status: 200 }) }, - async () => { - waits += 1 + { + wait: async () => { + waits += 1 + } } ) @@ -45,8 +82,10 @@ describe('readResourceInventory', () => { calls.set(path, call) return new Response(null, { status: path === '/ready' && call === 1 ? 503 : 200 }) }, - async (ms) => { - waits.push(ms) + { + wait: async (ms) => { + waits.push(ms) + } } ) @@ -65,17 +104,130 @@ describe('readResourceInventory', () => { calls += 1 return new Response(null, { status: 503 }) }, - async (ms) => { - waits.push(ms) + { + wait: async (ms) => { + waits.push(ms) + } } ) expect(result.health).toBe(false) expect(result.ready).toBe(false) expect(calls).toBe(4) + // A refusing endpoint is a reading, so only the independent retry runs. expect(waits).toEqual([11_000]) }) + it('treats a thrown fetch as no reading and re-asks that path once', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + if (path === '/health' && calls.filter((call) => call === '/health').length === 1) { + throw new TypeError('fetch failed') + } + return new Response(null, { status: 200 }) + }, + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(true) + expect(calls.filter((call) => call === '/health')).toEqual(['/health', '/health']) + expect(waits).toEqual([1_000]) + }) + + it('fails closed when both attempts of a path throw', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + if (path === '/health') throw new TypeError('fetch failed') + return new Response(null, { status: 200 }) + }, + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(false) + expect(calls.filter((call) => call === '/health')).toHaveLength(4) + expect(waits).toEqual([1_000, 11_000, 1_000]) + }) + + it('accepts an auth-shaped endpoint that serves no readiness path', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://login.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + return new Response(null, { status: path === '/ready' ? 404 : 200 }) + }, + { + requiresReady: false, + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBeNull() + expect(calls).toEqual(['/health']) + expect(waits).toEqual([]) + }) + + it('still requires readiness for the director and cells', async () => { + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://relay.onorca.dev', + async (input) => new Response(null, { + status: new URL(String(input)).pathname === '/ready' ? 503 : 200 + }), + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(false) + expect(waits).toEqual([11_000]) + }) + + it('measures latency as the answering round trip, not the retry delay', async () => { + let healthCalls = 0 + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + if (new URL(String(input)).pathname !== '/health') return new Response(null, { status: 200 }) + healthCalls += 1 + if (healthCalls === 1) throw new TypeError('fetch failed') + return new Response(null, { status: 200 }) + }, + { wait: async (ms) => await new Promise((resolve) => setTimeout(resolve, Math.min(ms, 60))) } + ) + + expect(result.health).toBe(true) + expect(result.latencyMs).not.toBeNull() + expect(result.latencyMs!).toBeLessThan(60) + }) + it('uses aggregate REST inventory without probing sleeping staging endpoints', async () => { const gcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } let publicProbeCalls = 0 @@ -132,6 +284,74 @@ describe('readResourceInventory', () => { expect(JSON.stringify(result)).not.toContain('SECRET_TEXT') }) + it('re-asks a MIG read that failed once before calling a cell powered-unknown', async () => { + const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let parkedMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(parkedCell.hostname)) return 'ok' + parkedMigCalls += 1 + return parkedMigCalls === 1 ? 'throw' : 'ok' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)! + // The MIG was fine and parked at zero; one transient read must not erase that reading. + expect(parked.targetSize).toBe(0) + expect(parkedMigCalls).toBe(2) + expect(waits).toEqual([1_000]) + expect(result.warnings).toEqual([]) + }) + + it('reports a MIG unavailable only when the retry fails too', async () => { + const parkedCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let parkedMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(parkedCell.hostname)) return 'ok' + parkedMigCalls += 1 + return 'throw' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + const parked = result.cells.find((cell) => cell.cellId === parkedCell.cellId)! + expect(parked.targetSize).toBeNull() + expect(parked.backendHealth).toBe('unknown') + expect(parkedMigCalls).toBe(2) + expect(waits).toEqual([1_000]) + expect(result.warnings).toEqual([ + `${parkedCell.hostname.toUpperCase()} MIG inventory is unavailable.` + ]) + }) + + it('does not re-ask a MIG read the API answered with 404', async () => { + const missingCell = RELAY_OPS_ENVIRONMENTS.staging.cells[0]! + const waits: number[] = [] + let missingMigCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + sleepingStagingGcloud, + sleepingStagingFetch((migName) => { + if (!migName.endsWith(missingCell.hostname)) return 'ok' + missingMigCalls += 1 + return 'missing' + }), + { wait: async (ms) => { waits.push(ms) } } + ) + + expect(result.cells.find((cell) => cell.cellId === missingCell.cellId)!.targetSize).toBeNull() + expect(missingMigCalls).toBe(1) + expect(waits).toEqual([]) + }) + it('represents missing credentials as unknown inventory, never sleeping', async () => { const gcloud: GcloudClient = { accessToken: async () => { throw new Error('sensitive context') } diff --git a/cloud/apps/relay-ops/src/resource-inventory.ts b/cloud/apps/relay-ops/src/resource-inventory.ts index 62ed3fd862b..da490685198 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.ts @@ -102,6 +102,9 @@ export type ResourceInventory = { const unavailableEndpoint = (): EndpointHealth => ({ health: null, ready: null, latencyMs: null }) const independentEndpointRetryDelayMs = 11_000 +const transientProbeRetryDelayMs = 1_000 +const sleep = async (ms: number): Promise => + await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)) function finalSegment(value: string): string { return value.split('/').at(-1) ?? value @@ -120,6 +123,12 @@ function parseService(value: unknown): ServiceInventory { } } +class GoogleApiError extends Error { + constructor(readonly status: number) { + super(`Google API returned ${status}`) + } +} + async function googleRequest( fetchImpl: typeof fetch, token: string, @@ -134,45 +143,97 @@ async function googleRequest( }, signal: AbortSignal.timeout(30_000) }) - if (!response.ok) throw new Error(`Google API returned ${response.status}`) + if (!response.ok) throw new GoogleApiError(response.status) return await response.json() } -async function endpointProbe(origin: string, fetchImpl: typeof fetch): Promise { - const startedAt = performance.now() - const check = async (path: '/health' | '/ready'): Promise => { +// A 404 is the API's answer about the resource; anything else is the absence of a reading, so re-ask. +async function readOnceMore( + read: () => Promise, + wait: (ms: number) => Promise +): Promise { + try { + return await read() + } catch (error) { + if (error instanceof GoogleApiError && error.status === 404) throw error + await wait(transientProbeRetryDelayMs) + return await read() + } +} + +// A reading the endpoint actually produced: ok is its answer, latencyMs is that answer's round trip. +type PathReading = { ok: boolean; latencyMs: number | null } + +async function probePath( + origin: string, + path: '/health' | '/ready', + fetchImpl: typeof fetch, + wait: (ms: number) => Promise +): Promise { + // null means the request never produced an answer (DNS/TCP/TLS failure or the 8s abort). + const attempt = async (): Promise => { + const startedAt = performance.now() try { const response = await fetchImpl(`${origin}${path}`, { redirect: 'error', signal: AbortSignal.timeout(8_000) }) - return response.ok + return { ok: response.ok, latencyMs: Math.round(performance.now() - startedAt) } } catch { - return false + return null } } - const [health, ready] = await Promise.all([check('/health'), check('/ready')]) - return { health, ready, latencyMs: Math.round(performance.now() - startedAt) } + const first = await attempt() + if (first) return first + // A thrown fetch is the absence of a reading, not an unhealthy answer, so re-ask before concluding. + await wait(transientProbeRetryDelayMs) + return (await attempt()) ?? { ok: false, latencyMs: null } +} + +async function endpointProbe( + origin: string, + fetchImpl: typeof fetch, + requiresReady: boolean, + wait: (ms: number) => Promise +): Promise { + const [health, ready] = await Promise.all([ + probePath(origin, '/health', fetchImpl, wait), + requiresReady ? probePath(origin, '/ready', fetchImpl, wait) : null + ]) + // Latency is the slowest answering round trip in this probe; retry delays are not serving latency. + const latencies = [health.latencyMs, ready?.latencyMs ?? null].filter( + (value): value is number => value !== null + ) + return { + health: health.ok, + ready: ready ? ready.ok : null, + latencyMs: latencies.length > 0 ? Math.max(...latencies) : null + } +} + +export type EndpointProbeOptions = { + // Auth serves no /ready by design, so it is judged on /health and latency alone. + requiresReady?: boolean + wait?: (ms: number) => Promise } export async function probeEndpointHealth( origin: string, fetchImpl: typeof fetch, - wait: (ms: number) => Promise = async (ms) => - await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)) + options: EndpointProbeOptions = {} ): Promise { - const first = await endpointProbe(origin, fetchImpl) - if ( - first.health && - first.ready && - first.latencyMs !== null && - first.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs - ) { - return first - } + const requiresReady = options.requiresReady ?? true + const wait = options.wait ?? sleep + const accepted = (probe: EndpointHealth): boolean => + probe.health === true && + (!requiresReady || probe.ready === true) && + probe.latencyMs !== null && + probe.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs + const first = await endpointProbe(origin, fetchImpl, requiresReady, wait) + if (accepted(first)) return first // Outwait Relay's ten-second readiness cache before treating the retry as independent. await wait(independentEndpointRetryDelayMs) - return await endpointProbe(origin, fetchImpl) + return await endpointProbe(origin, fetchImpl, requiresReady, wait) } function imageDigest(template: z.infer): string | null { @@ -285,11 +346,17 @@ function unavailableInventory(environment: RelayOpsEnvironment, warning: string) } } +export type ResourceInventoryOptions = { + wait?: (ms: number) => Promise +} + export async function readResourceInventory( environment: RelayOpsEnvironment, gcloud: GcloudClient, - fetchImpl: typeof fetch = fetch + fetchImpl: typeof fetch = fetch, + options: ResourceInventoryOptions = {} ): Promise { + const wait = options.wait ?? sleep let token: string try { token = await gcloud.accessToken() @@ -316,7 +383,10 @@ export async function readResourceInventory( token, `https://certificatemanager.googleapis.com/v1/projects/${environment.project}/locations/global/certificates/${environment.certificateName}` ), - ...environment.cells.map((cell) => googleRequest(fetchImpl, token, migUrl(cell))) + // One transient Compute read must never become a verdict on a cell's power state. + ...environment.cells.map((cell) => + readOnceMore(async () => await googleRequest(fetchImpl, token, migUrl(cell)), wait) + ) ]) const warnings: string[] = [] const directorValue = parsed(settled[0]!, RunServiceSchema, 'Director service inventory is unavailable.', warnings) @@ -338,7 +408,8 @@ export async function readResourceInventory( ? [unavailableEndpoint(), unavailableEndpoint()] : await Promise.all([ probeEndpointHealth(environment.directorOrigin, fetchImpl), - probeEndpointHealth(environment.authOrigin, fetchImpl) + // The auth service exposes no /ready, so requiring it would fail every first probe. + probeEndpointHealth(environment.authOrigin, fetchImpl, { requiresReady: false }) ]) const cells = await Promise.all(environment.cells.map((cell, index) => readCell(environment, cell, migValues[index] ?? null, token, fetchImpl) diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts index 9d240e304a7..226df9b3984 100644 --- a/cloud/apps/relay/src/assignment-store.ts +++ b/cloud/apps/relay/src/assignment-store.ts @@ -645,9 +645,17 @@ export class RelayAssignmentStore { ): Promise { const now = this.now() return await this.database.transaction(async (transaction) => { - const lockedCells = inventoryFirst - ? await this.lockCellInventory(transaction, lockMode) + // Why: the retry exists to take a cell row before the assignment row, the + // order placement uses. It only ever needs the one cell this host is + // pinned to, so read the pin unlocked and lock that row alone; taking all + // 23 queued every sticky refresh in the fleet behind every other one. + const pinnedCellId = inventoryFirst + ? await this.pinnedCellId(transaction, identity) : undefined + const lockedCells = + pinnedCellId === undefined + ? undefined + : await this.lockCellRows(transaction, [pinnedCellId], lockMode) const existing = await this.assignmentRow(transaction, identity, inventoryFirst) if (!existing) return null const activityLeases = await this.lockAssignmentActivities(transaction, identity, true) @@ -661,6 +669,11 @@ export class RelayAssignmentStore { } const currentCellId = text(existing, 'cell_id') + // The pin moved between the unlocked read and the assignment lock, so the + // row held is the wrong one. Same recovery as losing the lock: retry. + if (pinnedCellId !== undefined && pinnedCellId !== currentCellId) { + throw new Error('database_lock_unavailable') + } const hadControl = holdsControlLease( activityLeases, currentCellId, @@ -701,14 +714,9 @@ export class RelayAssignmentStore { if (hadControl) { await this.touchAssignment(transaction, identity, leaseExpiresAt, now) } else { - const nextReservation = integer(currentRow, 'reserved_requests') + 1 - if (nextReservation > integer(currentRow, 'capacity_requests')) { - throw new Error('relay_capacity_exhausted') - } - await transaction.query( - `UPDATE relay_cells SET reserved_requests = ?, updated_at = ? WHERE cell_id = ?`, - [nextReservation, now, currentCellId] - ) + // Delta, not the value read from the snapshot: an absolute write here + // would clobber any concurrent movement of the same counter. + await this.adjustCellReservationAtomically(transaction, currentCellId, 1) await this.adjustActivityCount(transaction, identity, 'control', 1, leaseExpiresAt, now) await this.insertPendingControlLease( transaction, @@ -6864,13 +6872,24 @@ export class RelayAssignmentStore { targetCellId ] ) - const cells = await this.lockCellInventory(transaction, 'pool-default') + // Only the two cells this repairs need holding. The id set below is an + // existence check against a table that only reconcileCells writes, so it + // reads unlocked instead of dragging the other 21 rows into the section. + const cellIds = new Set( + (await transaction.query(`SELECT cell_id FROM relay_cells`)).map((row) => + text(row, 'cell_id') + ) + ) + const cells = await this.lockCellRows( + transaction, + [sourceCellId, targetCellId], + 'pool-default' + ) const assignmentKeys = new Set( assignments.map((row) => assignmentKey(text(row, 'user_id'), text(row, 'relay_host_id')) ) ) - const cellIds = new Set(cells.map((row) => text(row, 'cell_id'))) const assignmentCounts = new Map< string, { counts: Record; leaseExpiresAt: number } @@ -6924,9 +6943,7 @@ export class RelayAssignmentStore { ) } - for (const row of cells.filter((cell) => - [sourceCellId, targetCellId].includes(text(cell, 'cell_id')) - )) { + for (const row of cells) { const cellId = text(row, 'cell_id') const expected = cellUnits.get(cellId) ?? 0 if (expected > integer(row, 'capacity_requests')) { @@ -6958,16 +6975,40 @@ export class RelayAssignmentStore { // Per-connection paths touch one or two cells. Locking exactly those rows, // in the same ascending order the inventory lock uses (ORDER BY fixes the // row-lock order), keeps them off the fleet-wide lock without a cycle. - private async lockCellRows(database: RelayDatabase, cellIds: string[]): Promise { + // The wait policy follows the caller for the same reason the inventory lock's + // does: a sweep must not fail terminally on ordinary contention. Hold time is + // deliberately not sampled here — the metric tracks the fleet-wide lock these + // rows replace, and mixing in short single-row holds would flatter it. + private async lockCellRows( + database: RelayDatabase, + cellIds: string[], + mode: CellInventoryLockMode = 'request' + ): Promise { const distinct = [...new Set(cellIds)] + const { measureHoldMs: _sampled, ...wait } = cellInventoryLockOptions(mode) return await database.queryLocked( `SELECT * FROM relay_cells WHERE cell_id IN (${distinct.map(() => '?').join(', ')}) ORDER BY cell_id ASC`, distinct, - { lockTimeoutMs: CELL_INVENTORY_LOCK_TIMEOUT_MS } + wait ) } + // Unlocked on purpose: this only names the row to lock next, and the caller + // re-checks the pin once the assignment row is held. + private async pinnedCellId( + database: RelayDatabase, + identity: AssignmentIdentity + ): Promise { + const row = ( + await database.query( + `SELECT cell_id FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + )[0] + return row ? text(row, 'cell_id') : undefined + } + private async lockGeneralCellInventory( database: RelayDatabase, mode: CellInventoryLockMode @@ -6986,10 +7027,11 @@ export class RelayAssignmentStore { private async leastLoadedCell( database: RelayDatabase, - lockedCells: SqlRow[] | undefined, + // Required: the one caller has already locked the inventory it selects from, + // and an optional parameter left a second fleet-wide lock reachable here. + rows: SqlRow[], preferredRegion: RelayRegion ): Promise { - const rows = lockedCells ?? (await this.lockCellInventory(database, 'pool-default')) const regions = new Map( (await database.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ text(row, 'cell_id'), diff --git a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts index a26e15f8e1d..0ac4c8225e3 100644 --- a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts +++ b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts @@ -18,7 +18,9 @@ type CensusEntry = { method: string; mode: CensusMode; reach: Reachability } // assignment-store.ts, in source order. A new site fails this test until it is // classified here, which is the point. const CENSUS: CensusEntry[] = [ - { method: 'assignStickyOnce', mode: 'caller', reach: 'both' }, + // assignStickyOnce is gone from this list: its retry now locks only the row + // the host is pinned to (lockCellRows), which is what a sticky refresh + // touches. Placement below is the one genuinely fleet-wide decision left. { method: 'assignOnce', mode: 'caller', reach: 'both' }, { method: 'assignOnce', mode: 'caller', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, @@ -49,8 +51,9 @@ const CENSUS: CensusEntry[] = [ { method: 'abortExpiredEvacuations', mode: 'nowait', reach: 'sweep' }, { method: 'releaseExpiredActivityLeases', mode: 'nowait', reach: 'sweep' }, { method: 'releaseExpiredActivity', mode: 'nowait', reach: 'sweep' }, - { method: 'reconcileReservationAccounting', mode: 'pool-default', reach: 'both' }, - { method: 'leastLoadedCell', mode: 'pool-default', reach: 'both' } + // reconcileReservationAccounting and leastLoadedCell are gone too: the first + // repairs exactly two cells' counters and now holds only those rows, and the + // second selects from the inventory its single caller has already locked. ] // Every inline `FROM relay_cells ... FOR UPDATE` outside the named lock helpers, diff --git a/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts b/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts new file mode 100644 index 00000000000..8c0ebd73f32 --- /dev/null +++ b/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts @@ -0,0 +1,206 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +// Sorted ascending, and the host is pinned to the LAST id on purpose: the +// fleet-wide lock is one ordered scan, so it holds every earlier row while it +// waits on the pinned one. Pinning to the first id would make the two locking +// models indistinguishable. +const cells = ['a', 'b', 'c'].map((suffix) => ({ + id: `percell-postgres-${suffix}`, + url: `https://percell-postgres-${suffix}.example.com`, + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 +})) +const [cellA, cellB, cellC] = cells as [(typeof cells)[0], (typeof cells)[0], (typeof cells)[0]] +const identity = { userId: 'percell-postgres-user', relayHostId: 'percellhost00001' } + +function heartbeat(cell: (typeof cells)[number]) { + return { + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 1, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +} + +describePostgres('PostgreSQL per-cell inventory locking', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + for (let index = 0; index < 3; index++) { + databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' })) + } + }) + + async function removeTestRows(database: RelayDatabase): Promise { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id LIKE 'percell-postgres-%'` + ) + for (const table of [ + 'relay_assignment_activity_leases', + 'relay_post_drain_migration_pins', + 'relay_assignment_migration_incarnations', + 'relay_assignment_migrations', + 'relay_assignment_region_preferences', + 'relay_assignments' + ]) { + await database.query(`DELETE FROM ${table} WHERE user_id LIKE 'percell-postgres-%'`) + } + for (const cell of cells) { + for (const table of [ + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_connection_limits', + 'relay_cell_runtime', + 'relay_cells' + ]) { + await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id]) + } + } + } + + afterAll(async () => { + if (databases[0]) await removeTestRows(databases[0]) + for (const connection of databases) await connection.close() + }) + + async function pinHostToLastCell(store: RelayAssignmentStore): Promise { + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cellC.id) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + } + + async function lockWaiterAppeared(database: RelayDatabase): Promise { + const deadline = Date.now() + 4_000 + while (Date.now() < deadline) { + const rows = await database.query( + `SELECT count(*) AS waiting FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock'` + ) + if (Number(rows[0]!.waiting) > 0) return true + await new Promise((resolve) => setTimeout(resolve, 10)) + } + return false + } + + // Why: a sticky refresh whose first NOWAIT probe loses retries by taking a + // cell row before the assignment row. That retry used to take the whole + // inventory, so one busy cell stalled every other cell's reconnects. + it('waits only on the pinned cell row while refreshing a sticky assignment', async () => { + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await pinHostToLastCell(store) + // A host whose control lease was already reaped still holds its pin; that + // is the shape that reaches the cell-row probe instead of touchAssignment. + await databases[0]!.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + [identity.userId] + ) + + let releaseRow!: () => void + const rowReleased = new Promise((resolve) => { + releaseRow = resolve + }) + let rowHeld!: () => void + const rowHeldPromise = new Promise((resolve) => { + rowHeld = resolve + }) + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellC.id]) + rowHeld() + await rowReleased + }) + await rowHeldPromise + + const refresh = store.assign(identity) + expect(await lockWaiterAppeared(databases[2]!)).toBe(true) + // The refresh is blocked on cell C. Every earlier row must still be free: + // the ordered fleet-wide scan would be holding both of them by now. + const heldWhileRefreshWaits: string[] = [] + await databases[2]!.transaction(async (transaction) => { + for (const cell of [cellA, cellB]) { + try { + await transaction.queryLocked( + `SELECT * FROM relay_cells WHERE cell_id = ?`, + [cell.id], + { failIfUnavailable: true } + ) + } catch { + heldWhileRefreshWaits.push(cell.id) + } + } + }) + releaseRow() + await holder + + expect(heldWhileRefreshWaits).toEqual([]) + expect((await refresh).cellId).toBe(cellC.id) + }, 15_000) + + // Why: the counter moves by a delta now instead of an absolute value read + // from a snapshot, so concurrent movement on the same cell must still sum. + it('keeps a cell reservation exact under concurrent same-cell activity', async () => { + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + + const hosts = Array.from({ length: 6 }, (_, index) => ({ + userId: `percell-postgres-user-${index}`, + relayHostId: `percellhost0000${index}` + })) + const stores = databases.map((database) => new RelayAssignmentStore(database, () => 100)) + await Promise.all(hosts.map((host, index) => stores[index % stores.length]!.assign(host))) + + // One splice each (2 units) on the same cell, from three connections at once. + await Promise.all( + hosts.map((host, index) => + stores[index % stores.length]!.acquireActivity(host, { + activityId: `splice:percell-${index}`, + kind: 'splice', + cellId: cellC.id + }) + ) + ) + const afterAcquire = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cellC.id] + ) + // 6 pending control grants + 6 splices at 2 units each. + expect(Number(afterAcquire[0]!.reserved_requests)).toBe(6 + 12) + + await Promise.all( + hosts.map((host, index) => + stores[index % stores.length]!.releaseActivity(host, `splice:percell-${index}`) + ) + ) + const afterRelease = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cellC.id] + ) + expect(Number(afterRelease[0]!.reserved_requests)).toBe(6) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/database-postgres-timeout.test.ts b/cloud/apps/relay/src/database-postgres-timeout.test.ts index 7aba1234f7f..c9021a9ef18 100644 --- a/cloud/apps/relay/src/database-postgres-timeout.test.ts +++ b/cloud/apps/relay/src/database-postgres-timeout.test.ts @@ -2,7 +2,10 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const fakes = vi.hoisted(() => ({ configs: [] as Array>, - query: vi.fn(async () => ({ rows: [], rowCount: 0 })), + // Pool construction and pool shutdown interleaved, so "the schema pool is + // gone before the serving pool opens" is checkable rather than assumed. + lifecycle: [] as string[], + query: vi.fn(async (_sql: string) => ({ rows: [], rowCount: 0 })), release: vi.fn(), end: vi.fn(async () => undefined) })) @@ -13,20 +16,37 @@ vi.mock('pg', () => ({ totalCount = 1 idleCount = 1 waitingCount = 0 - end = fakes.end on = vi.fn() connect = vi.fn(async () => ({ query: fakes.query, release: fakes.release })) + private readonly label: string constructor(config: Record) { fakes.configs.push(config) + this.label = `max=${String(config.max)} statement_timeout=${String(config.statement_timeout)}` + fakes.lifecycle.push(`open ${this.label}`) + } + + async end(): Promise { + fakes.lifecycle.push(`end ${this.label}`) + await fakes.end() } } } })) -import { openRelayDatabase } from './database.js' +import { openRelayDatabase, relayPostgresStatementTimeoutMs } from './database.js' import { applyPostgresSchema } from './postgres-schema-startup.js' +const SCHEMA_POOL = { + max: 1, + application_name: 'orca-relay/director/director/schema', + connectionTimeoutMillis: 2_000, + // Why: DDL must not inherit the request deadline. + statement_timeout: 0, + lock_timeout: 1_000, + idle_in_transaction_session_timeout: 5_000 +} + afterEach(() => { vi.restoreAllMocks() }) @@ -34,9 +54,11 @@ afterEach(() => { describe('PostgreSQL relay deadlines', () => { beforeEach(() => { fakes.configs.length = 0 + fakes.lifecycle.length = 0 fakes.query.mockClear() fakes.release.mockClear() fakes.end.mockClear() + delete process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS }) it('bounds pool acquisition, statements, locks, and abandoned transactions', async () => { @@ -48,6 +70,7 @@ describe('PostgreSQL relay deadlines', () => { }) expect(fakes.configs).toEqual([ + expect.objectContaining(SCHEMA_POOL), expect.objectContaining({ max: 3, application_name: 'orca-relay/director/director', @@ -59,6 +82,110 @@ describe('PostgreSQL relay deadlines', () => { ]) await database.close() }) + + // Why: an untimed session left open would be a standing way for request work + // to escape the deadline this whole pool config exists to enforce. + it('closes the untimed schema pool before the serving pool opens', async () => { + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused', + poolMax: 3, + applicationName: 'orca-relay/director/director' + }) + + expect(fakes.lifecycle).toEqual([ + 'open max=1 statement_timeout=0', + 'end max=1 statement_timeout=0', + 'open max=3 statement_timeout=5000' + ]) + await database.close() + }) + + it('applies the schema on the untimed pool, never on the serving one', async () => { + fakes.query.mockClear() + const ddl: string[] = [] + fakes.query.mockImplementation(async (sql: string) => { + // Every statement issued before the serving pool exists is schema work. + if (fakes.lifecycle.length === 1) ddl.push(sql) + return { rows: [], rowCount: 0 } + }) + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + + expect(ddl.length).toBeGreaterThan(0) + // Statements can open with a leading `--` rationale comment. + const body = (statement: string): string => + statement.replace(/^(?:\s*--[^\n]*\n)*\s*/, '') + expect(ddl.every((statement) => /^CREATE\b/i.test(body(statement)))).toBe(true) + // The backfill is DML, so it stays on the deadline-bearing serving pool. + expect(ddl.some((statement) => statement.includes('INSERT INTO'))).toBe(false) + await database.close() + }) + + it('takes the serving statement deadline from the environment', async () => { + process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS = '2500' + + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + + expect(fakes.configs).toEqual([ + expect.objectContaining({ statement_timeout: 0 }), + expect.objectContaining({ statement_timeout: 2_500 }) + ]) + await database.close() + }) + + it.each(['0', '-1', '2.5', 'soon', ' '])( + 'refuses %s as a statement deadline instead of running unbounded', + (value) => { + expect(() => + relayPostgresStatementTimeoutMs({ ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value }) + ).toThrow('invalid_statement_timeout') + } + ) + + it.each([undefined, ''])('defaults to 5s when the environment says %s', (value) => { + expect( + relayPostgresStatementTimeoutMs( + value === undefined ? {} : { ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value } + ) + ).toBe(5_000) + }) + + // Why: a statement deadline that reaches the caller as a crash converts a + // transient stall into a failed assignment. It aborts the transaction exactly + // as a lock timeout does, so it belongs on the same bounded retry. + it('retries a statement timeout on a fresh client', async () => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + let attempts = 0 + + const result = await database.transaction(async (transaction) => { + attempts += 1 + if (attempts === 1) { + await transaction.query('SELECT 1') + throw Object.assign(new Error('canceling statement due to statement timeout'), { + code: '57014' + }) + } + return 'committed' + }) + + expect(result).toBe('committed') + expect(attempts).toBe(2) + expect(console.warn).toHaveBeenCalledWith( + expect.stringContaining('"event":"orca_relay_postgres_transaction_retry"') + ) + expect(console.warn).toHaveBeenCalledWith(expect.stringContaining('"code":"57014"')) + await database.close() + }) }) describe('PostgreSQL schema startup', () => { diff --git a/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts b/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts new file mode 100644 index 00000000000..6b7ebb0334e --- /dev/null +++ b/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts @@ -0,0 +1,98 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const applicationName = 'orca-relay/statement-timeout-postgres' + +describePostgres('PostgreSQL statement deadline', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' })) + }) + + afterAll(async () => { + for (const database of databases) await database.close() + }) + + it('serves requests under the configured deadline', async () => { + const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 }) + databases.push(database) + + expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([ + { statement_timeout: '300ms' } + ]) + }) + + // Why: a real 57014 aborts the transaction exactly as a lock timeout does. If + // it escapes the bounded retry it becomes a failed assignment instead of a + // slow one. + it('retries a real statement timeout on a fresh client', async () => { + const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 }) + databases.push(database) + let attempts = 0 + + const result = await database.transaction(async (transaction) => { + attempts += 1 + if (attempts === 1) await transaction.query(`SELECT pg_sleep(2)`) + return attempts + }) + + expect(result).toBe(2) + }, 15_000) + + // Why: DDL runs on its own untimed connection. relay_invites carries a + // CREATE INDEX IF NOT EXISTS, which (unlike CREATE TABLE IF NOT EXISTS) + // really does queue behind an ACCESS EXCLUSIVE lock on the table. + it('applies the schema behind a held ACCESS EXCLUSIVE lock', async () => { + let releaseTable!: () => void + const tableReleased = new Promise((resolve) => { + releaseTable = resolve + }) + let tableHeld!: () => void + const tableHeldPromise = new Promise((resolve) => { + tableHeld = resolve + }) + const holder = databases[0]!.transaction(async (transaction) => { + await transaction.query(`LOCK TABLE relay_invites IN ACCESS EXCLUSIVE MODE`) + tableHeld() + await tableReleased + }) + await tableHeldPromise + + const opening = openRelayDatabase({ + databaseUrl, + dataDir: '', + applicationName, + // Far too short for a blocked DDL; the serving pool wears it, the schema + // connection must not. + statementTimeoutMs: 200 + }) + const blockedOnSchemaConnection = async (): Promise => { + const deadline = Date.now() + 4_000 + while (Date.now() < deadline) { + const rows = await databases[0]!.query( + `SELECT count(*) AS waiting FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock' + AND application_name = ?`, + [`${applicationName}/schema`] + ) + if (Number(rows[0]!.waiting) > 0) return true + await new Promise((resolve) => setTimeout(resolve, 10)) + } + return false + } + const blocked = await blockedOnSchemaConnection() + releaseTable() + await holder + + const database = await opening + databases.push(database) + expect(blocked).toBe(true) + // The serving pool still carries the short deadline it was opened with. + expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([ + { statement_timeout: '200ms' } + ]) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/database.ts b/cloud/apps/relay/src/database.ts index 326ab010ccb..2558831ca64 100644 --- a/cloud/apps/relay/src/database.ts +++ b/cloud/apps/relay/src/database.ts @@ -796,12 +796,34 @@ class PostgresTransaction implements RelayDatabase { const POSTGRES_TRANSACTION_ATTEMPTS = 3 const POSTGRES_RETRY_MAX_DELAY_MS = 25 const POSTGRES_CONNECTION_TIMEOUT_MS = 2_000 -const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 +// Derivation: a control renewal must land inside its own 30s tick +// (RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2), and a transaction gets +// POSTGRES_TRANSACTION_ATTEMPTS tries, so the worst case a renewal can spend in +// Postgres is attempts * timeout. 5s keeps that at 15s, half the tick, and still +// leaves room for the connect timeout above. +export const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 const POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 5_000 +export function relayPostgresStatementTimeoutMs( + env: NodeJS.ProcessEnv = process.env +): number { + const configured = env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS + if (configured === undefined || configured === '') return POSTGRES_STATEMENT_TIMEOUT_MS + const milliseconds = Number(configured) + // 0 is PostgreSQL's "no timeout"; refusing it keeps the deadline this exists + // to enforce from being disabled by a typo in an environment variable. + if (!Number.isInteger(milliseconds) || milliseconds < 1) { + throw new Error('invalid_statement_timeout') + } + return milliseconds +} + function retryablePostgresTransactionError(error: unknown): boolean { const code = String((error as { code?: unknown }).code) - return code === '40P01' || code === '40001' || code === '55P03' + // 57014 is the pool statement_timeout firing. It aborts the transaction the + // same way a lock timeout does, so it belongs on the bounded retry path + // rather than surfacing as a terminal failure to the caller. + return code === '40P01' || code === '40001' || code === '55P03' || code === '57014' } export function isRelayDatabaseTransientError(error: unknown): boolean { @@ -963,11 +985,36 @@ async function applySchema(database: RelayDatabase): Promise { } } -async function applySchemaWithPostgresRetries(database: RelayDatabase): Promise { - await applyPostgresSchema( - SCHEMA.split(';').filter((statement) => statement.trim()), - async (statement) => await database.query(statement) - ) +// Why: DDL is not a request. A CREATE INDEX on a grown table legitimately runs +// longer than the request statement_timeout, and inheriting that timeout would +// make every startup fail at the same statement instead of finishing once. One +// short-lived connection of its own, ended before the serving pool opens, keeps +// the untimed session off the request path entirely. +async function applySchemaOnUntimedPool( + databaseUrl: string, + applicationName: string | undefined +): Promise { + const pool = new pg.Pool({ + connectionString: databaseUrl, + max: 1, + application_name: applicationName ? `${applicationName}/schema` : undefined, + connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, + statement_timeout: 0, + // Kept: a DDL blocked behind another director's ACCESS EXCLUSIVE lock must + // yield to the bounded schema retry instead of holding the connection. + lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, + idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS + }) + absorbPostgresIdleClientErrors(pool) + const database = new PostgresDatabase(pool) + try { + await applyPostgresSchema( + SCHEMA.split(';').filter((statement) => statement.trim()), + async (statement) => await database.query(statement) + ) + } finally { + await database.close().catch(() => undefined) + } } async function backfillRelayCellRegions(database: RelayDatabase): Promise { @@ -983,15 +1030,17 @@ export async function openRelayDatabase(input: { dataDir: string poolMax?: number applicationName?: string + statementTimeoutMs?: number }): Promise { let database: RelayDatabase if (input.databaseUrl) { + await applySchemaOnUntimedPool(input.databaseUrl, input.applicationName) const pool = new pg.Pool({ connectionString: input.databaseUrl, max: input.poolMax ?? 10, application_name: input.applicationName, connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, - statement_timeout: POSTGRES_STATEMENT_TIMEOUT_MS, + statement_timeout: input.statementTimeoutMs ?? relayPostgresStatementTimeoutMs(), lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS }) @@ -1004,8 +1053,7 @@ export async function openRelayDatabase(input: { database = new SqliteDatabase(sqlite) } try { - if (input.databaseUrl) await applySchemaWithPostgresRetries(database) - else await applySchema(database) + if (!input.databaseUrl) await applySchema(database) await backfillRelayCellRegions(database) return database } catch (error) { diff --git a/cloud/apps/relay/src/host-close-reason-memory.test.ts b/cloud/apps/relay/src/host-close-reason-memory.test.ts new file mode 100644 index 00000000000..2985e6f1a1d --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.test.ts @@ -0,0 +1,82 @@ +import { ASSIGNMENT_LIMITS, RELAY_HOST_CLOSE_REASON } from '@orca-cloud/relay-contract' +import { describe, expect, it } from 'vitest' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' + +function memoryAt(clock: { now: number }): HostCloseReasonMemory { + return new HostCloseReasonMemory(() => clock.now) +} + +describe('HostCloseReasonMemory', () => { + it('remembers only reasons it knows', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', 'quitting') + memory.record('c', Buffer.alloc(0)) + memory.record('d', undefined) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + expect(memory.read('b')).toBeNull() + expect(memory.read('c')).toBeNull() + expect(memory.read('d')).toBeNull() + }) + + it('accepts the reason as the Buffer a ws close delivers', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', Buffer.from(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('expires an entry once its host may have been rebalanced away', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += ASSIGNMENT_LIMITS.dormantTtlMs - 1 + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += 1 + expect(memory.read('a')).toBeNull() + expect(memory.size()).toBe(0) + }) + + it('forgets on demand', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + memory.forget('a') + + expect(memory.read('a')).toBeNull() + }) + + it('drops the oldest survivors rather than growing without bound', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + for (let index = 0; index < 50_050; index++) { + memory.record(`host-${index}`, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + } + + expect(memory.size()).toBe(50_000) + expect(memory.read('host-0')).toBeNull() + expect(memory.read('host-50049')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('re-recording refreshes recency so a live host is not evicted first', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + expect([...['a', 'b'].map((key) => memory.read(key))]).toEqual([ + RELAY_HOST_CLOSE_REASON.SIGNED_OUT, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ]) + expect(memory.size()).toBe(2) + }) +}) diff --git a/cloud/apps/relay/src/host-close-reason-memory.ts b/cloud/apps/relay/src/host-close-reason-memory.ts new file mode 100644 index 00000000000..ed01aacd666 --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.ts @@ -0,0 +1,72 @@ +import { + ASSIGNMENT_LIMITS, + relayHostCloseReasonFrom, + type RelayHostCloseReason +} from '@orca-cloud/relay-contract' + +// Retention matches the dormant assignment TTL: past it the host may have been +// rebalanced onto another cell, so this cell is no longer the one a phone asks. +const RETENTION_MS = ASSIGNMENT_LIMITS.dormantTtlMs +// A fleet-wide auth outage signs out every host at once; the cap bounds that +// burst well above any single cell's host count without becoming a leak. +const MAX_ENTRIES = 50_000 + +// Why in-memory and not Postgres: a phone reaches the cell its host's assignment +// row already names, which is the same cell that watched the control socket +// close. Losing this on a cell restart degrades to the pre-existing generic +// verdict, so the failure mode is the old behaviour rather than a wrong one. +export class HostCloseReasonMemory { + private readonly entries = new Map() + + constructor(private readonly now: () => number = Date.now) {} + + // Silently ignores anything that is not a known reason, which is every close + // from a host that predates the field and every abrupt 1006. + record(key: string, reason: unknown): void { + const parsed = relayHostCloseReasonFrom(reason) + if (!parsed) { + return + } + this.entries.delete(key) + this.entries.set(key, { reason: parsed, expiresAt: this.now() + RETENTION_MS }) + this.evict() + } + + forget(key: string): void { + this.entries.delete(key) + } + + read(key: string): RelayHostCloseReason | null { + const entry = this.entries.get(key) + if (!entry) { + return null + } + if (entry.expiresAt <= this.now()) { + this.entries.delete(key) + return null + } + return entry.reason + } + + size(): number { + return this.entries.size + } + + private evict(): void { + const now = this.now() + for (const [key, entry] of this.entries) { + if (entry.expiresAt > now) { + break + } + this.entries.delete(key) + } + // Insertion order is recency order (record deletes before setting), so the + // head is always the oldest survivor. + for (const key of this.entries.keys()) { + if (this.entries.size <= MAX_ENTRIES) { + break + } + this.entries.delete(key) + } + } +} diff --git a/cloud/apps/relay/src/host-session-registry.ts b/cloud/apps/relay/src/host-session-registry.ts index 11b7d1de030..5c53041e7ff 100644 --- a/cloud/apps/relay/src/host-session-registry.ts +++ b/cloud/apps/relay/src/host-session-registry.ts @@ -14,7 +14,8 @@ import { HostHelloSchema, InviteCreateSchema, RELAY_PROTOCOL_LIMITS, - RELAY_CLOSE_CODE + RELAY_CLOSE_CODE, + type RelayHostCloseReason } from '@orca-cloud/relay-contract' import nacl from 'tweetnacl' import type WebSocket from 'ws' @@ -25,6 +26,7 @@ import { RelayCredentialStore, type CredentialReservation } from './credential-store.js' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' import { relayHostLogDigest } from './relay-host-log-digest.js' import type { RelayTokenClaims } from './relay-token-verifier.js' import type { RelayRuntimeObserver } from './relay-observability.js' @@ -130,6 +132,10 @@ const ACTIVATION_QUEUE_WAIT_MS = 30_000 export class HostSessionRegistry { private readonly sessions = new Map() private readonly activationQueues = new Map>() + // Why it outlives `sessions`: the orphan grace deletes the session within 30s, + // but a signed-out desktop never comes back, so the phone that asks minutes + // later would otherwise find nothing to explain its rejection with. + private readonly hostCloseReasons = new HostCloseReasonMemory(() => this.now()) private draining = false constructor( @@ -175,7 +181,8 @@ export class HostSessionRegistry { return } this.observer.recordAuth(true) - const session = this.sessions.get(this.key(reservation.userId, hostId)) + const sessionKey = this.key(reservation.userId, hostId) + const session = this.sessions.get(sessionKey) if ( !session || session.state !== 'active' || @@ -184,7 +191,13 @@ export class HostSessionRegistry { ) { capacityReservation?.release() await this.store.failReservation(reservation) - this.rejectClient(socket, RELAY_CLOSE_CODE.HOST_OFFLINE) + // The only rejection that can name a cause: the host is genuinely absent. + // The attach-deadline 4404 below fires while control is still connected. + this.rejectClient( + socket, + RELAY_CLOSE_CODE.HOST_OFFLINE, + this.hostCloseReasons.read(sessionKey) + ) return } if (session.activeConnIds.size + session.pendingConns.size >= 8) { @@ -793,7 +806,10 @@ export class HostSessionRegistry { regionalDrainTimer: null, regionalDrainExpiresAt: null } - this.sessions.set(this.key(identity.sub, identity.relayHostId), session) + const sessionKey = this.key(identity.sub, identity.relayHostId) + // A host that proved itself again is not signed out, whatever it said last. + this.hostCloseReasons.forget(sessionKey) + this.sessions.set(sessionKey, session) this.wireActiveControl(session) this.sendHelloAck(session) } @@ -813,6 +829,11 @@ export class HostSessionRegistry { }) socket.once('close', (code, reason) => { this.observer.recordControlClose?.(code) + // Guarded on identity: a predecessor retired by a rebind must not stamp a + // cause onto the live session that replaced it. + if (session.socket === socket) { + this.hostCloseReasons.record(this.key(session.identity.sub, session.relayHostId), reason) + } // One line per control close makes reconnect churners attributable by // host digest without exposing the raw relay host id. console.warn( @@ -1187,9 +1208,16 @@ export class HostSessionRegistry { if (session.socket) send(session.socket, 'control-error', { ...(reqId ? { reqId } : {}), code }) } - private rejectClient(socket: WebSocket, code: number): void { + // hostCloseReason rides the WebSocket close reason, never relay-hello: every + // shipped phone parses relay-hello with a strict schema that rejects an + // unknown key, and none of them read the close reason at all. + private rejectClient( + socket: WebSocket, + code: number, + hostCloseReason?: RelayHostCloseReason | null + ): void { send(socket, 'relay-hello', { ok: false, code }) - closeRelayWebSocket(socket, code, 'relay connection rejected') + closeRelayWebSocket(socket, code, hostCloseReason ?? 'relay connection rejected') } private releaseControlActivity(session: HostSession): void { diff --git a/cloud/apps/relay/src/host-signed-out-rejection.test.ts b/cloud/apps/relay/src/host-signed-out-rejection.test.ts new file mode 100644 index 00000000000..0f8542c960f --- /dev/null +++ b/cloud/apps/relay/src/host-signed-out-rejection.test.ts @@ -0,0 +1,206 @@ +import { EventEmitter } from 'node:events' +import { + CONTROL_CONTINUITY_LIMITS, + RELAY_CLOSE_CODE, + RELAY_HOST_CLOSE_REASON +} from '@orca-cloud/relay-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { RelayCredentialStore } from './credential-store.js' +import { HostSessionRegistry } from './host-session-registry.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close', 1006, Buffer.alloc(0)) + }) +} + +const config = { + port: 8080, + publicUrl: 'https://relay-c3.example.com', + cellUrl: 'https://relay-c3.example.com', + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [] +} as unknown as RelayConfig + +const identity = { + sub: 'user-1', + prof: 'profile-1', + org: 'org-1', + relayHostId: 'AbCdEf0123_-xyZ9' +} as unknown as RelayTokenClaims + +const reservation = { + userId: identity.sub, + relayHostId: identity.relayHostId, + credentialKind: 'resume', + relayDeviceId: 'device-1', + leaseExpiresAt: Date.now() + 60_000 +} + +function createRegistry() { + const store = { + resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }), + reserveCredential: vi.fn().mockResolvedValue(reservation), + failReservation: vi.fn().mockResolvedValue(undefined) + } + const assignments = { + activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'), + markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined), + resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }), + acquireActivity: vi.fn().mockResolvedValue(undefined), + renewControlActivity: vi.fn().mockResolvedValue(undefined), + releaseActivity: vi.fn().mockResolvedValue(true) + } as unknown as RelayAssignmentStore + const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn(), + recordControlClose: vi.fn(), + recordSpliceClose: vi.fn() + } satisfies RelayRuntimeObserver + const registry = new HostSessionRegistry( + config, + vi.fn(), + store as unknown as RelayCredentialStore, + assignments, + new ProcessQueuedByteBudget(), + observer + ) + const activate = (socket: WebSocket, generation: number): Promise => + ( + registry as unknown as { + activate: ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string + ) => Promise + } + ).activate(socket, identity, null, generation, false, 1, '1.4.173') + return { registry, activate } +} + +async function dialPhone(registry: HostSessionRegistry): Promise { + const phone = new FakeSocket() + await registry.acceptClient(phone as unknown as WebSocket, identity.relayHostId, 'credential') + return phone +} + +// The 4404 hello body is unchanged: every shipped phone parses it with a strict +// schema, so the cause has to ride the close frame instead. +const HOST_OFFLINE_HELLO = JSON.stringify({ + type: 'relay-hello', + ok: false, + code: RELAY_CLOSE_CODE.HOST_OFFLINE +}) + +describe('host sign-out reason on phone rejection', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('names the sign-out to a phone that arrives after the host is gone', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.send).toHaveBeenCalledWith(HOST_OFFLINE_HELLO) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ) + }) + + it('says nothing when the host died without naming a cause', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('ignores a close reason the host invented', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, 'signed-out-ish') + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('forgets the sign-out once the host proves itself again', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const reconnected = new FakeSocket() + await activate(reconnected as unknown as WebSocket, 2) + // Drop it abruptly, as a network death would, so only the stale memory + // could still name a cause. + reconnected.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + // A live host is present: the 4404 there is an attach deadline, not absence. + it('never names a cause while the host control is connected', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + const phone = await dialPhone(registry) + expect(phone.close).not.toHaveBeenCalled() + expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"')) + }) +}) diff --git a/cloud/apps/relay/src/postgres-transaction-recovery.test.ts b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts index a49c07d2e7d..ae9d52a7c86 100644 --- a/cloud/apps/relay/src/postgres-transaction-recovery.test.ts +++ b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts @@ -503,12 +503,19 @@ describePostgres('PostgreSQL transaction recovery', () => { const directorLockOrder: string[] = [] const assignmentDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { if (phase === 'before') { - if (sql.includes('FROM relay_assignments WHERE user_id = ?')) { + // Only locked statements reach this hook, so classifying the pin read + // is what proves it stays unlocked: if it ever grows a FOR UPDATE it + // shows up in the order below instead of silently joining the queue. + if (sql.includes('SELECT cell_id FROM relay_assignments')) { + directorLockOrder.push('pin-read') + } else if (sql.includes('FROM relay_assignments WHERE user_id = ?')) { directorLockOrder.push('assignment') } else if (sql.includes('FROM relay_assignment_activity_leases')) { directorLockOrder.push('activity') } else if (sql.includes('FROM relay_cells ORDER BY')) { directorLockOrder.push('cell-inventory') + } else if (sql.includes('FROM relay_cells WHERE cell_id IN')) { + directorLockOrder.push('cell-rows') } else if (sql.includes('FROM relay_cells WHERE cell_id = ?')) { directorLockOrder.push('cell') } @@ -530,11 +537,14 @@ describePostgres('PostgreSQL transaction recovery', () => { }) await expect(legacyTransaction).resolves.toBeUndefined() expect(assignmentDatabase.attempts).toBe(2) + // The retry still takes a cell row before the assignment row — the order + // that avoids the legacy cycle — but only the pinned row, never the + // inventory. expect(directorLockOrder).toEqual([ 'assignment', 'activity', 'cell', - 'cell-inventory', + 'cell-rows', 'assignment', 'activity' ]) diff --git a/cloud/dev/fixtures/terraform-root-partition/families.json b/cloud/dev/fixtures/terraform-root-partition/families.json index 9664c1eb299..dfe100fd2dd 100644 --- a/cloud/dev/fixtures/terraform-root-partition/families.json +++ b/cloud/dev/fixtures/terraform-root-partition/families.json @@ -133,10 +133,15 @@ "google_logging_metric.relay_snapshot", "google_monitoring_alert_policy.relay_assignment_5xx", "google_monitoring_alert_policy.relay_assignment_edge_429", + "google_monitoring_alert_policy.relay_cell_process_exit", + "google_monitoring_alert_policy.relay_cloud_nat_port_drops", "google_monitoring_alert_policy.relay_cloud_sql_backends", + "google_monitoring_alert_policy.relay_cloud_sql_checkpoint_loop", + "google_monitoring_alert_policy.relay_cloud_sql_disk", "google_monitoring_alert_policy.relay_custom", "google_monitoring_alert_policy.relay_gce_connection_headroom", "google_monitoring_alert_policy.relay_postgres_retry_exhausted", + "google_monitoring_dashboard.relay_incident", "google_project_iam_custom_role.github_production_relay_capacity_mutation", "google_project_iam_custom_role.github_relay_asia_topology_mutation", "google_project_iam_custom_role.github_relay_asia_topology_read", diff --git a/cloud/dev/scripts/relay-evidence-code-provenance.mjs b/cloud/dev/scripts/relay-evidence-code-provenance.mjs new file mode 100644 index 00000000000..233a8139b85 --- /dev/null +++ b/cloud/dev/scripts/relay-evidence-code-provenance.mjs @@ -0,0 +1,94 @@ +import { spawnSync } from 'node:child_process' +import { fileURLToPath } from 'node:url' +import { + RELAY_REPOSITORY_ROOT, + relayTreePath, + relayWorkflowPath +} from './relay-repository.mjs' + +const SHA = /^[a-f0-9]{40}$/ + +// Every file that decides how relay evidence is produced, sealed, verified, and then spent against +// production; identical content across two commits is what makes the older commit's verdict binding. +export const TRUSTED_EVIDENCE_CODE_PATHS = [ + // Produces and seals the 15-minute dry-run evidence. + relayWorkflowPath('monitor-relay-production.yml'), + relayWorkflowPath('monitor-relay-production-job.yml'), + // Download it, verify its authority, and mutate production on it. + relayWorkflowPath('deploy-relay-production-same-cap.yml'), + relayWorkflowPath('deploy-relay-production-same-cap-job.yml'), + relayWorkflowPath('operate-relay-production-rehome.yml'), + relayWorkflowPath('operate-relay-production-rehome-job.yml'), + // Sealing, verification, the wave/canary authority, and the path constants below. + relayTreePath('dev/scripts/relay-evidence-code-provenance.mjs'), + relayTreePath('dev/scripts/relay-monitor-evidence.mjs'), + relayTreePath('dev/scripts/relay-production-same-cap-wave.mjs'), + relayTreePath('dev/scripts/relay-repository.mjs'), + // Every other script those jobs run against live production. + relayTreePath('dev/scripts/infra.mjs'), + relayTreePath('dev/scripts/operate-relay-regional-rehome.mjs'), + relayTreePath('dev/scripts/prepare-relay-production-capacity-canary.mjs'), + relayTreePath('dev/scripts/probe-relay-rehome-trust.mjs'), + relayTreePath('dev/scripts/validate-relay-capacity-plan.mjs'), + relayTreePath('dev/scripts/verify-relay-capacity-transition.mjs'), + // The monitor itself and the live preflight recheck, plus anything that changes their behaviour. + relayTreePath('apps/relay-ops'), + relayTreePath('package.json'), + relayTreePath('pnpm-lock.yaml'), + relayTreePath('pnpm-workspace.yaml'), + // The Cloud SQL rollout lease every mutation job takes and releases. + '.github/actions/cloud-sql-rollout-lease' +] + +function git(root, args) { + const result = spawnSync('git', ['-C', root, ...args], { encoding: 'utf8' }) + if (result.error) throw new Error('relay evidence provenance cannot run git') + return result +} + +/** + * Accepts evidence sealed at a different commit only when the current commit descends from it and + * every trusted path is byte-identical, so the verdict provably came from this exact code. Anything + * git cannot answer (no checkout, unknown commit, shallow clone) fails closed. + */ +export function requireSameEvidenceCode({ + sealedSha, + currentSha, + label, + repositoryRoot = fileURLToPath(RELAY_REPOSITORY_ROOT) +}) { + if (!SHA.test(sealedSha ?? '') || !SHA.test(currentSha ?? '')) { + throw new Error(`${label} commit is invalid`) + } + if (sealedSha === currentSha) return + if (git(repositoryRoot, ['rev-parse', '--git-dir']).status !== 0) { + throw new Error(`${label} commit cannot be compared without a git checkout`) + } + for (const sha of [sealedSha, currentSha]) { + if (git(repositoryRoot, ['rev-parse', '--verify', '--quiet', `${sha}^{commit}`]).status !== 0) { + throw new Error( + `${label} commit ${sha} is unknown to this checkout; check out with fetch-depth: 0` + ) + } + } + const ancestry = git(repositoryRoot, ['merge-base', '--is-ancestor', sealedSha, currentSha]) + if (ancestry.status === 1) { + throw new Error(`${label} commit ${sealedSha} is not an ancestor of ${currentSha}`) + } + if (ancestry.status !== 0) { + throw new Error(`${label} commit ancestry could not be determined`) + } + const diff = git(repositoryRoot, [ + 'diff', + '--name-only', + sealedSha, + currentSha, + '--', + ...TRUSTED_EVIDENCE_CODE_PATHS + ]) + if (diff.status !== 0) throw new Error(`${label} commit comparison failed`) + const changed = diff.stdout.split('\n').filter(Boolean) + if (changed.length > 0) { + throw new Error(`${label} code changed after it was sealed: ${changed.join(',')}`) + } +} diff --git a/cloud/dev/scripts/relay-monitor-evidence.mjs b/cloud/dev/scripts/relay-monitor-evidence.mjs index 7f387663f60..26eb37d0d4d 100644 --- a/cloud/dev/scripts/relay-monitor-evidence.mjs +++ b/cloud/dev/scripts/relay-monitor-evidence.mjs @@ -2,6 +2,7 @@ import { createHash } from 'node:crypto' import { chmod, readFile, readdir, stat, writeFile } from 'node:fs/promises' import { basename, join, resolve } from 'node:path' import { pathToFileURL } from 'node:url' +import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs' const SAFE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{1,127}$/ const SHA = /^[a-f0-9]{40}$/ @@ -102,7 +103,7 @@ export async function createEvidenceManifest(argv) { return manifest } -async function readAndVerifyManifest(directory, expected) { +async function readAndVerifyManifest(directory, expected, sameCodeCommit) { const manifest = JSON.parse( await readFile(join(directory, 'evidence-manifest.json'), 'utf8') ) @@ -111,11 +112,23 @@ async function readAndVerifyManifest(directory, expected) { manifest.incidentId !== expected.incidentId || manifest.runId !== expected.runId || manifest.runAttempt !== expected.runAttempt || - manifest.commitSha !== expected.commitSha || - manifest.mode !== expected.mode + !SHA.test(manifest.commitSha ?? '') || + manifest.mode !== expected.mode || + (!sameCodeCommit && manifest.commitSha !== expected.commitSha) ) { throw new Error('relay monitor evidence provenance does not match') } + // Unrelated merges land on main every few minutes, so the deployer resolves a newer commit than + // the monitor it must trust; identical monitor and mutation code is the property the SHA stood in + // for. Restore and mutation keep the exact-SHA bind: both run at the commit that sealed them. + if (sameCodeCommit) { + requireSameEvidenceCode({ + sealedSha: manifest.commitSha, + currentSha: expected.commitSha, + label: 'relay monitor evidence', + ...sameCodeCommit + }) + } const names = Object.keys(manifest.files ?? {}) if (!names.includes(`${expected.incidentId}.state.json`)) { throw new Error('relay monitor evidence has no durable state') @@ -209,12 +222,12 @@ function validCompletedDryRunState(state, expected, nowMs, maxAgeMs) { ) } -export async function verifyDryRunAuthority(argv, now = Date.now) { +export async function verifyDryRunAuthority(argv, now = Date.now, repositoryRoot) { const values = argumentsByName(argv) const directory = resolve(values.directory ?? '') const expected = provenance(values) if (expected.mode !== 'dry-run') throw new Error('relay mutation requires dry-run evidence') - const manifest = await readAndVerifyManifest(directory, expected) + const manifest = await readAndVerifyManifest(directory, expected, { repositoryRoot }) const state = JSON.parse( await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8') ) diff --git a/cloud/dev/scripts/relay-monitor-evidence.test.mjs b/cloud/dev/scripts/relay-monitor-evidence.test.mjs index 45116761119..43d2ac02763 100644 --- a/cloud/dev/scripts/relay-monitor-evidence.test.mjs +++ b/cloud/dev/scripts/relay-monitor-evidence.test.mjs @@ -1,9 +1,15 @@ import assert from 'node:assert/strict' -import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { execFileSync } from 'node:child_process' +import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' -import { join } from 'node:path' +import { dirname, join } from 'node:path' import test from 'node:test' -import { relayWorkflowPath, relayWorkflowUrl } from './relay-repository.mjs' +import { TRUSTED_EVIDENCE_CODE_PATHS } from './relay-evidence-code-provenance.mjs' +import { + RELAY_REPOSITORY_ROOT, + relayWorkflowPath, + relayWorkflowUrl +} from './relay-repository.mjs' import { createEvidenceManifest, verifyDryRunAuthority, @@ -12,7 +18,7 @@ import { } from './relay-monitor-evidence.mjs' const now = Date.parse('2026-07-28T12:00:00.000Z') -const provenance = [ +const provenanceFor = (commitSha) => [ '--incident-id', 'relay-123', '--run-id', @@ -20,10 +26,11 @@ const provenance = [ '--run-attempt', '1', '--commit-sha', - 'a'.repeat(40), + commitSha, '--mode', 'dry-run' ] +const provenance = provenanceFor('a'.repeat(40)) const selector = { generation: 2, membership: { @@ -513,3 +520,157 @@ test('monitor uses a reusable job so exact job_workflow_ref is present', async ( assert.match(job, /workflow_call:/) assert.match(job, /environment: production/) }) + +function gitIn(root, ...args) { + return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim() +} + +// A real repository shaped like main under unrelated merge traffic: one sealed commit, a +// descendant that only touched untrusted files, a descendant that touched the monitor, and a +// sibling that never descended from the seal. +async function trustedCodeRepository() { + const root = await mkdtemp(join(tmpdir(), 'relay-evidence-repository-')) + gitIn(root, 'init', '--quiet') + gitIn(root, 'config', 'user.email', 'relay@example.test') + gitIn(root, 'config', 'user.name', 'Relay Evidence Test') + gitIn(root, 'config', 'commit.gpgsign', 'false') + const commit = async (path, body, message) => { + await mkdir(dirname(join(root, path)), { recursive: true }) + await writeFile(join(root, path), body) + gitIn(root, 'add', '--all') + gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message) + return gitIn(root, 'rev-parse', 'HEAD') + } + const base = await commit( + 'cloud/apps/relay-ops/src/incident-monitor.ts', + 'export const v = 1\n', + 'monitor' + ) + const sealed = await commit('README.md', 'base\n', 'base') + const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated') + const changedCode = await commit( + 'cloud/apps/relay-ops/src/incident-monitor.ts', + 'export const v = 2\n', + 'monitor change' + ) + // Branches before the seal, so the seal is not in its history even though its code matches. + gitIn(root, 'checkout', '--quiet', '--detach', base) + const sibling = await commit('README.md', 'a divergent line\n', 'divergent') + return { root, sealed, sameCode, changedCode, sibling } +} + +const authorityAt = (directory, commitSha, repositoryRoot) => verifyDryRunAuthority( + [ + '--directory', + directory, + ...provenanceFor(commitSha), + '--required-migration-policy', + 'strict' + ], + () => now, + repositoryRoot +) + +test('accepts dry-run evidence sealed by identical code at an ancestor commit', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + // An exact match never consults git: a root with no checkout at all still verifies. + await assert.doesNotReject(authorityAt(directory, repository.sealed, directory)) + await assert.doesNotReject(authorityAt(directory, repository.sameCode, repository.root)) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +test('rejects dry-run evidence whose monitor code or lineage differs', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + await assert.rejects( + authorityAt(directory, repository.changedCode, repository.root), + /code changed after it was sealed: cloud\/apps\/relay-ops\/src\/incident-monitor\.ts/ + ) + await assert.rejects( + authorityAt(directory, repository.sibling, repository.root), + /is not an ancestor of/ + ) + // Fails closed: a shallow clone that never fetched the sealed commit proves nothing. + await assert.rejects( + authorityAt(directory, 'f'.repeat(40), repository.root), + /unknown to this checkout/ + ) + // Fails closed: no checkout to compare against. + await assert.rejects( + authorityAt(directory, repository.sameCode, directory), + /cannot be compared without a git checkout/ + ) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +test('keeps restore and mutation bound to the exact sealing commit', async () => { + const repository = await trustedCodeRepository() + const directory = await evidenceDirectory() + try { + await createEvidenceManifest([ + '--directory', + directory, + ...provenanceFor(repository.sealed) + ]) + await assert.rejects( + verifyRestoredEvidence([ + '--directory', + directory, + ...provenanceFor(repository.sameCode) + ]), + /provenance does not match/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + directory, + ...provenanceFor(repository.sameCode), + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + async () => Response.json({ selector }), + () => now + ), + /provenance does not match/ + ) + } finally { + await rm(repository.root, { recursive: true, force: true }) + await rm(directory, { recursive: true, force: true }) + } +}) + +// A trusted path that no longer exists silently stops being compared, so the same-code rule would +// pass over code it was written to pin. +test('every trusted provenance path exists in this checkout', async () => { + for (const path of TRUSTED_EVIDENCE_CODE_PATHS) { + await assert.doesNotReject( + stat(new URL(path, RELAY_REPOSITORY_ROOT)), + `${path} is missing` + ) + } +}) diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.mjs index e391c4c4381..6e84c1c9104 100644 --- a/cloud/dev/scripts/relay-production-same-cap-wave.mjs +++ b/cloud/dev/scripts/relay-production-same-cap-wave.mjs @@ -1,5 +1,6 @@ import { readFileSync } from 'node:fs' import { pathToFileURL } from 'node:url' +import { requireSameEvidenceCode } from './relay-evidence-code-provenance.mjs' export const SAME_CAP_CELLS = [ 'production-gce-c7', 'production-gce-c8', 'production-gce-c9', 'production-gce-c10', @@ -85,10 +86,10 @@ export function canaryAuthority(input) { } } -export function verifyCanaryAuthority(authority, expected) { +export function verifyCanaryAuthority(authority, expected, repositoryRoot) { if ( authority?.v !== 1 || - authority.commitSha !== expected.commitSha || + !/^[0-9a-f]{40}$/.test(authority.commitSha ?? '') || authority.runId !== expected.runId || authority.targetDigest !== expected.targetDigest || authority.rollbackDigest !== expected.rollbackDigest || @@ -96,6 +97,14 @@ export function verifyCanaryAuthority(authority, expected) { authority.rehomeGeneration !== Number(expected.rehomeGeneration) || !SAME_CAP_CELLS.includes(authority.cellId) ) throw new Error('canary authority does not match this batch') + // The batch dispatch resolves main after the canary sealed, so bind to the same code, not the + // same SHA; every field above still pins this batch to that exact canary. + requireSameEvidenceCode({ + sealedSha: authority.commitSha, + currentSha: expected.commitSha, + label: 'relay same-cap canary authority', + repositoryRoot + }) return authority } diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs index 0b45ae85a99..d636c324b33 100644 --- a/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs +++ b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs @@ -1,4 +1,8 @@ import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' import { test } from 'node:test' import { canaryAuthority, @@ -104,3 +108,66 @@ test('seals and verifies canary authority for later batches', () => { rehomeGeneration: '4' }), /does not match/) }) + +function gitIn(root, ...args) { + return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8' }).trim() +} + +async function canaryRepository() { + const root = await mkdtemp(join(tmpdir(), 'relay-same-cap-canary-')) + gitIn(root, 'init', '--quiet') + gitIn(root, 'config', 'user.email', 'relay@example.test') + gitIn(root, 'config', 'user.name', 'Relay Wave Test') + gitIn(root, 'config', 'commit.gpgsign', 'false') + const commit = async (path, body, message) => { + await mkdir(dirname(join(root, path)), { recursive: true }) + await writeFile(join(root, path), body) + gitIn(root, 'add', '--all') + gitIn(root, 'commit', '--quiet', '--no-verify', '--message', message) + return gitIn(root, 'rev-parse', 'HEAD') + } + const sealed = await commit( + 'cloud/dev/scripts/relay-production-same-cap-wave.mjs', + 'export const v = 1\n', + 'wave' + ) + const sameCode = await commit('README.md', 'an unrelated merge\n', 'unrelated') + const changedCode = await commit( + 'cloud/dev/scripts/relay-production-same-cap-wave.mjs', + 'export const v = 2\n', + 'wave change' + ) + return { root, sealed, sameCode, changedCode } +} + +test('a batch trusts a canary sealed by identical code at an ancestor commit', async () => { + const repository = await canaryRepository() + try { + const authority = canaryAuthority({ + cellIds: 'production-gce-c7', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c7`, + commitSha: repository.sealed, + runId: '42', + selectorGeneration: '11', + rehomeGeneration: '4' + }) + const verifyAt = (commitSha, repositoryRoot) => verifyCanaryAuthority(authority, { + commitSha, + runId: '42', + targetDigest, + rollbackDigest, + selectorGeneration: '13', + rehomeGeneration: '4' + }, repositoryRoot) + assert.equal(verifyAt(repository.sameCode, repository.root).cellId, 'production-gce-c7') + assert.throws( + () => verifyAt(repository.changedCode, repository.root), + /code changed after it was sealed/ + ) + assert.throws(() => verifyAt('f'.repeat(40), repository.root), /unknown to this checkout/) + } finally { + await rm(repository.root, { recursive: true, force: true }) + } +}) diff --git a/cloud/dev/scripts/relay-repository.mjs b/cloud/dev/scripts/relay-repository.mjs index 7e8b01e4799..040acef5fe5 100644 --- a/cloud/dev/scripts/relay-repository.mjs +++ b/cloud/dev/scripts/relay-repository.mjs @@ -1,4 +1,6 @@ import { readFileSync } from 'node:fs' +import { relative } from 'node:path' +import { fileURLToPath } from 'node:url' // Single place naming the repository the Relay workflows live in and where their files sit. The // public-repo copy moves this tree under cloud/, prefixes every workflow filename, and changes the @@ -11,6 +13,19 @@ export const RELAY_WORKFLOW_FILE_PREFIX = 'cloud-' // this tree moves under cloud/, so the depth changes at the copy even though the layout does not. export const RELAY_WORKFLOW_DIRECTORY = new URL('../../../.github/workflows/', import.meta.url) +// Repository root, derived from the one directory above that already tracks the copy's depth. +export const RELAY_REPOSITORY_ROOT = new URL('../../', RELAY_WORKFLOW_DIRECTORY) + +// Repository-relative path for a file in this tree. The prefix is 'cloud/' here and empty where +// the tree is the repository root, so callers naming git paths never restate the layout. +export function relayTreePath(suffix) { + const prefix = relative( + fileURLToPath(RELAY_REPOSITORY_ROOT), + fileURLToPath(new URL('../../', import.meta.url)) + ).split(/[\\/]/).filter(Boolean) + return [...prefix, suffix].join('/') +} + export function relayWorkflowFile(name) { return `${RELAY_WORKFLOW_FILE_PREFIX}${name}` } diff --git a/cloud/infra/terraform/relay-gce-cells.tf b/cloud/infra/terraform/relay-gce-cells.tf index d6b7f3351f9..a4505ba2e37 100644 --- a/cloud/infra/terraform/relay-gce-cells.tf +++ b/cloud/infra/terraform/relay-gce-cells.tf @@ -242,6 +242,7 @@ resource "google_compute_instance_template" "relay_gce_cell" { artifact_registry_host = "${var.region}-docker.pkg.dev" relay_image = each.value.image cloud_sql_proxy_image = var.relay_gce_cloud_sql_proxy_image + cloud_sql_private_ip = var.relay_cloud_sql_private_ip # Keep cell-only plans independent from unrelated database configuration drift. cloud_sql_connection_name = local.relay_database_connection_name }) diff --git a/cloud/infra/terraform/relay-gce-foundation.tf b/cloud/infra/terraform/relay-gce-foundation.tf index aab3b4579eb..a8d64b3fcea 100644 --- a/cloud/infra/terraform/relay-gce-foundation.tf +++ b/cloud/infra/terraform/relay-gce-foundation.tf @@ -42,6 +42,12 @@ resource "google_compute_router_nat" "relay_gce" { router = google_compute_router.relay_gce[0].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce[0].id @@ -85,6 +91,12 @@ resource "google_compute_router_nat" "relay_gce_additional" { router = google_compute_router.relay_gce_additional[each.key].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce_additional[each.key].id diff --git a/cloud/infra/terraform/relay-gce-startup.sh.tftpl b/cloud/infra/terraform/relay-gce-startup.sh.tftpl index f593d94e9e5..a77466f2169 100644 --- a/cloud/infra/terraform/relay-gce-startup.sh.tftpl +++ b/cloud/infra/terraform/relay-gce-startup.sh.tftpl @@ -109,6 +109,9 @@ docker run --detach \ --user 0:0 \ --volume "$${cloudsql_dir}:/cloudsql" \ '${cloud_sql_proxy_image}' \ +%{ if cloud_sql_private_ip ~} + --private-ip \ +%{ endif ~} --unix-socket=/cloudsql \ '${cloud_sql_connection_name}' diff --git a/cloud/infra/terraform/relay-observability.tf b/cloud/infra/terraform/relay-observability.tf index a8fa4276eb2..6bc938100c6 100644 --- a/cloud/infra/terraform/relay-observability.tf +++ b/cloud/infra/terraform/relay-observability.tf @@ -37,6 +37,16 @@ locals { description = "Relay PostgreSQL transactions that exhausted bounded retry." filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR resource.type=\"gce_instance\") AND jsonPayload.event=\"orca_relay_postgres_transaction_exhausted\"" } + cell_process_exit = { + # The docker event stream is the only per-exit line: the relay's own crash footer only + # appears for unhandled rejections, and `container start` also counts healthy first boots. + description = "Relay cell container exits, one Docker `container die` event per process exit." + filter = "resource.type=\"gce_instance\" AND logName=\"projects/${var.project_id}/logs/cos_system\" AND jsonPayload.SYSLOG_IDENTIFIER=\"docker\" AND jsonPayload.MESSAGE:\"container die\" AND jsonPayload.MESSAGE:\"name=orca-relay)\"" + } + cloud_sql_wal_checkpoint = { + description = "Cloud SQL checkpoints triggered by WAL volume instead of the timed schedule; a sustained run is the fsync loop that stalled every relay process at once on 2026-09-04." + filter = "resource.type=\"cloudsql_database\" AND resource.labels.database_id=\"${var.project_id}:${local.relay_database_instance_name}\" AND textPayload:\"checkpoint starting: wal\"" + } } relay_runtime_metrics = { @@ -201,7 +211,8 @@ resource "google_logging_metric" "relay_snapshot" { label_extractors = { role = "EXTRACT(jsonPayload.role)" cell_id = "EXTRACT(jsonPayload.cellId)" - region = "EXTRACT(jsonPayload.region)" + # No region label: adding one replaces all 21 live metrics (label change = delete+create), + # which resets history and blanks the relay alert policies during the swap. } metric_descriptor { @@ -220,12 +231,6 @@ resource "google_logging_metric" "relay_snapshot" { value_type = "STRING" description = "Durable relay cell identifier." } - - labels { - key = "region" - value_type = "STRING" - description = "Coarse Relay region." - } } bucket_options { @@ -523,3 +528,280 @@ resource "google_monitoring_alert_policy" "relay_cloud_sql_backends" { mime_type = "text/markdown" } } + +resource "google_monitoring_alert_policy" "relay_cloud_sql_checkpoint_loop" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL checkpoint loop" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "WAL-triggered checkpoints above 3 in 5 minutes" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\"" + comparison = "COMPARISON_GT" + threshold_value = 3 + duration = "300s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Healthy operation is one timed checkpoint every 5 minutes. Repeated `checkpoint starting: wal` lines mean WAL is outrunning `max_wal_size` and every checkpoint fsync stalls all relay SQL for seconds. Check `checkpoint complete` sync= times and disk write throughput against the PD-SSD ceiling; the fix is disk size and `max_wal_size` in the Terraform root that owns the instance (orca-cloud `infra/terraform-foundation`)." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +resource "google_monitoring_alert_policy" "relay_cloud_sql_disk" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL disk utilization" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cloud SQL disk above 70%" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND resource.label.\"database_id\"=\"${var.project_id}:${local.relay_database_instance_name}\" AND metric.type=\"cloudsql.googleapis.com/database/disk/utilization\"" + comparison = "COMPARISON_GT" + threshold_value = 0.7 + duration = "600s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_MAX" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "The shared auth/relay Cloud SQL disk is filling. `refresh_tokens` is the largest table and grows without pruning; grow the disk (IOPS scale with size) before it reaches the WAL checkpoint loop, and prune revoked token rows." + mime_type = "text/markdown" + } +} + +resource "google_monitoring_alert_policy" "relay_cloud_nat_port_drops" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + display_name = "Orca Relay: Cloud NAT port exhaustion" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "NAT packets dropped for lack of ports" + + condition_threshold { + filter = "resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\") AND metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND metric.label.\"reason\"=\"OUT_OF_RESOURCES\"" + comparison = "COMPARISON_GT" + threshold_value = 0 + duration = "120s" + + aggregations { + alignment_period = "60s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"gateway_name\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Relay cells reach Cloud SQL's public IP through this NAT. Port exhaustion makes every cell's Cloud SQL Auth Proxy dial time out at once, which reads as a fleet-wide SQL stall with a healthy database. Check `nat/port_usage` per VM and raise `max_ports_per_vm` in `relay-gce-foundation.tf`, or move the database to a private IP." + mime_type = "text/markdown" + } +} + +resource "google_monitoring_alert_policy" "relay_cell_process_exit" { + project = var.project_id + display_name = "Orca Relay: cell process exits" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cell container exits above 3 in 15 minutes" + + condition_threshold { + filter = "resource.type=\"gce_instance\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cell_process_exit\"" + comparison = "COMPARISON_GT" + threshold_value = 3 + duration = "0s" + + aggregations { + alignment_period = "900s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"instance_id\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "A Relay GCE cell restarted its container more than three times in 15 minutes. Each exit drops every host and phone on that cell, and 201 exits went unpaged over 48 h on 2026-09-04. The instance hostname is `relay--`; read `jsonPayload.MESSAGE` on `cos_system` for the exit code and the container's own stderr for the stack before blaming MIG autoheal or load. A same-capacity roll is the remedy when the running image is behind." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +# Why: the four signals that had to be assembled by hand during the 2026-09-04 incident. +resource "google_monitoring_dashboard" "relay_incident" { + project = var.project_id + + dashboard_json = jsonencode({ + displayName = "Orca Relay: incident overview" + mosaicLayout = { + columns = 12 + tiles = [ + { + xPos = 0 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Cloud SQL WAL checkpoints" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\" AND resource.type=\"cloudsql_database\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "checkpoints" + scale = "LINEAR" + } + } + } + }, + { + xPos = 3 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Cloud NAT dropped packets" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\")" + aggregation = { + alignmentPeriod = "60s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + groupByFields = ["resource.label.\"gateway_name\"", "metric.label.\"reason\""] + } + } + } + }] + yAxis = { + label = "packets" + scale = "LINEAR" + } + } + } + }, + { + xPos = 6 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Auth refresh 401s" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"logging.googleapis.com/user/orca_auth_refresh_401\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "rejections" + scale = "LINEAR" + } + } + } + }, + { + xPos = 9 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Standing desktop controls (fleet sum)" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + # ALIGN_MEAN, not ALIGN_SUM: each process reports its standing control count once per interval. + filter = "metric.type=\"logging.googleapis.com/user/orca_relay_controls\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_MEAN" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "controls" + scale = "LINEAR" + } + } + } + } + ] + } + }) + + depends_on = [google_logging_metric.relay_incident, google_logging_metric.relay_snapshot] +} diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf index 68f6c555bd3..91f67e8ebe0 100644 --- a/cloud/infra/terraform/variables.tf +++ b/cloud/infra/terraform/variables.tf @@ -468,6 +468,12 @@ variable "relay_gce_fenced_cells" { default = [] } +variable "relay_cloud_sql_private_ip" { + type = bool + description = "Dial Cloud SQL over its private IP inside this VPC instead of its public IP through Cloud NAT. Requires the foundation root's private services access peering to be applied first; a cell that cannot reach the private IP never becomes ready." + default = false +} + variable "relay_gce_cloud_sql_proxy_image" { type = string description = "Digest-pinned Cloud SQL Auth Proxy image used by private relay workers." diff --git a/cloud/packages/relay-contract/src/host-close-reason.ts b/cloud/packages/relay-contract/src/host-close-reason.ts new file mode 100644 index 00000000000..3a5abde3f00 --- /dev/null +++ b/cloud/packages/relay-contract/src/host-close-reason.ts @@ -0,0 +1,18 @@ +// Mirror of src/shared/relay-host-close-reason.ts in the Orca app repo half. +// A host control socket may close with one of these as its WebSocket close +// reason; the cell records it so a later phone rejection can name the cause. +// Anything else (including the empty reason of an abrupt 1006) means "unknown", +// which is what every peer that predates this file sends. +export const RELAY_HOST_CLOSE_REASON = { + SIGNED_OUT: 'signed-out' +} as const + +export type RelayHostCloseReason = + (typeof RELAY_HOST_CLOSE_REASON)[keyof typeof RELAY_HOST_CLOSE_REASON] + +const REASONS: readonly string[] = Object.values(RELAY_HOST_CLOSE_REASON) + +export function relayHostCloseReasonFrom(value: unknown): RelayHostCloseReason | null { + const text = typeof value === 'string' ? value : (value?.toString() ?? '') + return REASONS.includes(text) ? (text as RelayHostCloseReason) : null +} diff --git a/cloud/packages/relay-contract/src/index.ts b/cloud/packages/relay-contract/src/index.ts index 2a7d7d0feda..aab3b53b5f3 100644 --- a/cloud/packages/relay-contract/src/index.ts +++ b/cloud/packages/relay-contract/src/index.ts @@ -5,6 +5,7 @@ export * from './control-messages.js' export * from './control-continuity.js' export * from './credential-messages.js' export * from './director-messages.js' +export * from './host-close-reason.js' export * from './host-proof-transcript.js' export * from './persistence-invariants.js' export * from './protocol-limits.js' diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index f5a73f6239f..15ded915c67 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -219,6 +219,7 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/agent-hooks/windows-hook-payload-delivery.test.ts', 'src/main/windows/windows-pty-job.win32.test.ts', 'src/main/windows/windows-host-job.win32.test.ts', + 'src/main/windows-live-tree-kill.win32.test.ts', 'src/main/wsl/wsl-runner.test.ts', 'src/main/wsl/wsl-guest-environment.test.ts', 'src/main/wsl/wsl-invocation-boundary.test.ts', diff --git a/config/scripts/verify-localization-catalog.mjs b/config/scripts/verify-localization-catalog.mjs index 002a84360f6..a73e9d5e3cc 100644 --- a/config/scripts/verify-localization-catalog.mjs +++ b/config/scripts/verify-localization-catalog.mjs @@ -11,7 +11,12 @@ import { repairTranslatedValue } from './locale-translation-policy.mjs' const SOURCE_EXTENSIONS = new Set(['.ts', '.tsx', '.js', '.jsx', '.mts', '.cts']) const SKIP_PATH_PARTS = new Set(['.git', 'dist', 'node_modules', 'out', '__snapshots__', 'assets']) -const LOCALIZATION_FUNCTION_NAMES = new Set(['t', 'translate', 'translateMain', 'translateSearchKeyword']) +const LOCALIZATION_FUNCTION_NAMES = new Set([ + 't', + 'translate', + 'translateMain', + 'translateSearchKeyword' +]) const PLACEHOLDER_RE = /\{\{[^}]+\}\}/g const LOCALES_RELATIVE_DIR = path.join('src', 'renderer', 'src', 'i18n', 'locales') export const LOCALIZATION_SOURCE_ROOTS = [ diff --git a/mobile/src/home/MobileHomeHostList.tsx b/mobile/src/home/MobileHomeHostList.tsx index 3907df16f03..41d1f07156d 100644 --- a/mobile/src/home/MobileHomeHostList.tsx +++ b/mobile/src/home/MobileHomeHostList.tsx @@ -19,6 +19,7 @@ type MobileHomeHostListProps = { hostAttempts: Record hostLastConnected: Record hostPairingRejected: Record + hostSignedOut: Record hostPaths: Record hostPendingPaths: Record hosts: HostCatalogEntry[] @@ -40,6 +41,7 @@ export function MobileHomeHostList(props: MobileHomeHostListProps) { hostAttempts={props.hostAttempts} hostLastConnected={props.hostLastConnected} hostPairingRejected={props.hostPairingRejected} + hostSignedOut={props.hostSignedOut} hostPaths={props.hostPaths} hostPendingPaths={props.hostPendingPaths} hostStates={props.hostStates} @@ -54,6 +56,7 @@ export function MobileHomeHostList(props: MobileHomeHostListProps) { props.hostAttempts, props.hostLastConnected, props.hostPairingRejected, + props.hostSignedOut, props.hostPaths, props.hostPendingPaths, props.hostStates, @@ -91,6 +94,7 @@ type MobileHomeHostRowProps = Pick< | 'hostAttempts' | 'hostLastConnected' | 'hostPairingRejected' + | 'hostSignedOut' | 'hostPaths' | 'hostPendingPaths' | 'hostStates' @@ -113,7 +117,8 @@ const MobileHomeHostRow = memo(function MobileHomeHostRow(props: MobileHomeHostR lastConnectedAt: props.hostLastConnected[item.id] ?? null, endpoint: item.endpoint, pendingPath: props.hostPendingPaths[item.id] ?? null, - pairingRejected: props.hostPairingRejected[item.id] ?? false + pairingRejected: props.hostPairingRejected[item.id] ?? false, + hostSignedOut: props.hostSignedOut[item.id] ?? false }) const open = useCallback(() => onOpen(item), [item, onOpen]) const longPress = useCallback(() => onLongPress(item), [item, onLongPress]) diff --git a/mobile/src/home/MobileHomeScreen.tsx b/mobile/src/home/MobileHomeScreen.tsx index 83cf3de4b8a..7772f5152c6 100644 --- a/mobile/src/home/MobileHomeScreen.tsx +++ b/mobile/src/home/MobileHomeScreen.tsx @@ -143,6 +143,7 @@ export function MobileHomeScreen() { hostAttempts={data.hostAttempts} hostLastConnected={data.hostLastConnected} hostPairingRejected={data.hostPairingRejected} + hostSignedOut={data.hostSignedOut} hostPaths={data.hostPaths} hostPendingPaths={data.hostPendingPaths} hosts={data.sortedHostCatalog} diff --git a/mobile/src/home/home-host-connection-projection.ts b/mobile/src/home/home-host-connection-projection.ts index f8fdfd4bcdf..9a49f6186c4 100644 --- a/mobile/src/home/home-host-connection-projection.ts +++ b/mobile/src/home/home-host-connection-projection.ts @@ -5,12 +5,14 @@ export type HomeHostConnectionProjectionEntry = { path: MobileConnectionPath pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean } export type HomeHostConnectionProjection = { hostPaths: Record hostPendingPaths: Record hostPairingRejected: Record + hostSignedOut: Record } /** Build all host lookup maps while reading each connection entry once. */ @@ -22,16 +24,19 @@ export function projectHomeHostConnections( const hostPaths = Object.create(null) as Record const hostPendingPaths = Object.create(null) as Record const hostPairingRejected = Object.create(null) as Record + const hostSignedOut = Object.create(null) as Record - for (const { hostId, path, pendingPath, pairingRejected } of entries) { + for (const { hostId, path, pendingPath, pairingRejected, hostSignedOut: signedOut } of entries) { hostPaths[hostId] = path hostPendingPaths[hostId] = pendingPath hostPairingRejected[hostId] = pairingRejected + hostSignedOut[hostId] = signedOut } Object.setPrototypeOf(hostPaths, Object.prototype) Object.setPrototypeOf(hostPendingPaths, Object.prototype) Object.setPrototypeOf(hostPairingRejected, Object.prototype) + Object.setPrototypeOf(hostSignedOut, Object.prototype) - return { hostPaths, hostPendingPaths, hostPairingRejected } + return { hostPaths, hostPendingPaths, hostPairingRejected, hostSignedOut } } diff --git a/mobile/src/home/use-mobile-home-data.ts b/mobile/src/home/use-mobile-home-data.ts index c77d024158e..6b28354a86d 100644 --- a/mobile/src/home/use-mobile-home-data.ts +++ b/mobile/src/home/use-mobile-home-data.ts @@ -179,6 +179,7 @@ export function useMobileHomeData() { connectedHosts, hostCatalog, hostPairingRejected: hostConnectionProjection.hostPairingRejected, + hostSignedOut: hostConnectionProjection.hostSignedOut, hostPaths: hostConnectionProjection.hostPaths, hostPendingPaths: hostConnectionProjection.hostPendingPaths, primaryHost, diff --git a/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts b/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts index 4430c60115a..5a959b870bf 100644 --- a/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts +++ b/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts @@ -41,6 +41,7 @@ const MODEL_DESCRIPTOR: SessionOptionDescriptor = { ] }, valueSource: 'reported', + transport: 'catalog', settable: true } @@ -57,6 +58,7 @@ const EFFORT_DESCRIPTOR: SessionOptionDescriptor = { ] }, valueSource: 'dispatched', + transport: 'catalog', settable: true } @@ -66,6 +68,7 @@ const FAST_MODE_DESCRIPTOR: SessionOptionDescriptor = { category: 'mode', kind: { type: 'boolean', currentValue: false }, valueSource: 'reported', + transport: 'catalog', settable: true } @@ -227,6 +230,7 @@ describe('MobileNativeChatSessionOptionPickers', () => { ...MODEL_DESCRIPTOR, kind: { type: 'select', choices: [] }, valueSource: 'unknown', + transport: 'catalog', action: { type: 'agent-picker' } } ]) @@ -236,6 +240,46 @@ describe('MobileNativeChatSessionOptionPickers', () => { expect(invokeAction).toHaveBeenCalledWith('model') }) + // The terminal transport can only learn the outcome by parsing the screen back, + // so the sheet admits the value is unconfirmed; the structured transport reports + // it every turn, which makes the same caption noise there. + it.each([ + { transport: 'catalog' as const, caption: true }, + { transport: 'agent-session' as const, caption: false } + ])('captions a dispatched value only on the terminal transport', async (scenario) => { + mount([ + MODEL_DESCRIPTOR, + { ...EFFORT_DESCRIPTOR, valueSource: 'dispatched', transport: scenario.transport } + ]) + await act(async () => pill('Model').props.onPress()) + await act(async () => rowByText('Effort').props.onPress()) + const captions = renderer!.root + .findAll((node) => node.type === 'Text') + .filter( + (node) => + (node.props as { children?: unknown }).children === 'Sent to the agent — not confirmed' + ) + expect(captions.length > 0).toBe(scenario.caption) + }) + + it.each(['catalog', 'agent-session'] as const)( + 'does not caption a reported value on the %s transport', + async (transport) => { + mount([MODEL_DESCRIPTOR, { ...EFFORT_DESCRIPTOR, valueSource: 'reported', transport }]) + await act(async () => pill('Model').props.onPress()) + await act(async () => rowByText('Effort').props.onPress()) + expect( + renderer!.root + .findAll((node) => node.type === 'Text') + .some( + (node) => + (node.props as { children?: unknown }).children === + 'Sent to the agent — not confirmed' + ) + ).toBe(false) + } + ) + it('locks the pills while the agent is working', () => { mount([MODEL_DESCRIPTOR, EFFORT_DESCRIPTOR], true) expect(pill('Model').props).toMatchObject({ disabled: true }) diff --git a/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx b/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx index bfa5244a398..c9d641f74ec 100644 --- a/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx +++ b/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx @@ -3,9 +3,10 @@ import { ActivityIndicator, Keyboard, Pressable, StyleSheet, Text, View } from ' import { ChevronLeft, X } from 'lucide-react-native' import { BottomDrawer } from '../components/BottomDrawer' import { colors, radii, spacing, typography } from '../theme/mobile-theme' -import type { - SessionOptionDescriptor, - SessionOptionValue +import { + sessionOptionDispatchUnconfirmed, + type SessionOptionDescriptor, + type SessionOptionValue } from '../../../src/shared/native-chat-session-options' import { mobileModelPillLabel, @@ -119,7 +120,7 @@ export function MobileNativeChatSessionOptionPickers({ ) : null} - {activeDescriptor.valueSource === 'dispatched' ? ( + {sessionOptionDispatchUnconfirmed(activeDescriptor) ? ( Sent to the agent — not confirmed ) : null} {reason ? {reason} : null} diff --git a/mobile/src/session/use-mobile-native-chat-session-options.ts b/mobile/src/session/use-mobile-native-chat-session-options.ts index 66a929aeb56..6acdf8b3750 100644 --- a/mobile/src/session/use-mobile-native-chat-session-options.ts +++ b/mobile/src/session/use-mobile-native-chat-session-options.ts @@ -168,7 +168,8 @@ export function useMobileNativeChatSessionOptions(args: { models: activeModels(catalog, record), record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) }, [agent, catalog, scopeKey, version]) diff --git a/mobile/src/transport/client-context-connection-metrics.ts b/mobile/src/transport/client-context-connection-metrics.ts index 1e2ff6f7919..1fe72687bd2 100644 --- a/mobile/src/transport/client-context-connection-metrics.ts +++ b/mobile/src/transport/client-context-connection-metrics.ts @@ -30,14 +30,16 @@ export function useConnectionPathStatus(hostId: string | undefined): { export function useRelayRecoveryStatus(hostId: string | undefined): { pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean } { return useHostMetric( hostId, (context, id) => ({ pendingPath: context.getPendingPath(id), - pairingRejected: context.isPairingRejected(id) + pairingRejected: context.isPairingRejected(id), + hostSignedOut: context.isHostSignedOut(id) }), - { pendingPath: null, pairingRejected: false } + { pendingPath: null, pairingRejected: false, hostSignedOut: false } ) } diff --git a/mobile/src/transport/client-context.test.ts b/mobile/src/transport/client-context.test.ts index 4a7d5d7b0e8..56b227bdff7 100644 --- a/mobile/src/transport/client-context.test.ts +++ b/mobile/src/transport/client-context.test.ts @@ -533,12 +533,12 @@ describe('useAllHostClients', () => { await Promise.resolve() }) act(() => client.emitPendingPath('relay')) - expect(status).toEqual({ pendingPath: 'relay', pairingRejected: false }) + expect(status).toEqual({ pendingPath: 'relay', pairingRejected: false, hostSignedOut: false }) // Why: the desktop refusing the credential is a status-only change — no // transport state moves, so only the connection-path signal can carry it. act(() => client.emitPairingRejected(true)) - expect(status).toEqual({ pendingPath: 'relay', pairingRejected: true }) + expect(status).toEqual({ pendingPath: 'relay', pairingRejected: true, hostSignedOut: false }) act(() => renderer.unmount()) }) diff --git a/mobile/src/transport/connection-health.ts b/mobile/src/transport/connection-health.ts index 858b13a8b24..1a9e282047f 100644 --- a/mobile/src/transport/connection-health.ts +++ b/mobile/src/transport/connection-health.ts @@ -29,6 +29,10 @@ const STALE_SINCE_LAST_CONNECT_MS = 60_000 // instead of leaving the user staring at a generic "Can't connect". const TAILSCALE_HINT = 'check Tailscale' +// No hint field: the remedy is the label, and appending "— check Tailscale" to +// it would be wrong advice for a desktop that is reachable but signed out. +const SIGNED_OUT_LABEL = 'Desktop signed out — sign in to Orca on your desktop to reconnect' + export type ConnectionVerdict = | { kind: 'normal'; label: string } | { kind: 'warning'; label: string; hint?: string } // "Can't connect" @@ -54,6 +58,10 @@ export function classifyConnection(args: { // The desktop has repeatedly refused this device's relay credential — retrying // cannot fix it, so it outranks any "still connecting" reading (STA-4681). pairingRejected?: boolean + // The relay says the desktop's last control close named its own Orca Cloud + // sign-out. Retrying is still correct and still happens on the same cadence, + // but only the desktop's owner can end it, so the label has to say so. + hostSignedOut?: boolean nowMs?: number }): ConnectionVerdict { const { state, reconnectAttempts, lastConnectedAt } = args @@ -70,6 +78,17 @@ export function classifyConnection(args: { return { kind: 'normal', label: 'Connected' } } + // Ahead of the attempt thresholds: this is evidence, not an inference from a + // failure streak, and waiting twelve dials to show it wastes the whole point. + // Below auth-failed because a revoked pairing cannot be fixed by signing in. + if (args.hostSignedOut) { + return { + kind: 'unreachable', + label: SIGNED_OUT_LABEL, + reason: lastConnectedAt == null ? 'never-connected' : 'stale' + } + } + // A disconnected pending path can survive a cleared retry timer during a // lifecycle race. Only narrate Relay while dialing or after a retry has // recorded progress; otherwise the idle transport must read Disconnected. diff --git a/mobile/src/transport/host-client-context-state.ts b/mobile/src/transport/host-client-context-state.ts index 859e5d847f9..db4c7908b41 100644 --- a/mobile/src/transport/host-client-context-state.ts +++ b/mobile/src/transport/host-client-context-state.ts @@ -89,10 +89,16 @@ export function createHostClientSelectors( getPendingPath: (hostId: string): MobileConnectionPath | null => clientPendingPath(entries.get(hostId)?.client), isPairingRejected: (hostId: string): boolean => - clientPairingRejected(entries.get(hostId)?.client) + clientPairingRejected(entries.get(hostId)?.client), + isHostSignedOut: (hostId: string): boolean => clientHostSignedOut(entries.get(hostId)?.client) } } +export function clientHostSignedOut(client: RpcClient | undefined): boolean { + const logical = client as Partial | undefined + return logical?.isHostSignedOut?.() ?? false +} + export function clientPairingRejected(client: RpcClient | undefined): boolean { const logical = client as Partial | undefined return logical?.isPairingRejected?.() ?? false diff --git a/mobile/src/transport/logical-client-connection-path.ts b/mobile/src/transport/logical-client-connection-path.ts index 55c6d1d28f3..b7f02b40b88 100644 --- a/mobile/src/transport/logical-client-connection-path.ts +++ b/mobile/src/transport/logical-client-connection-path.ts @@ -5,6 +5,7 @@ export class LogicalClientConnectionPath { private recovery: MobileConnectionPath | null = null private recoveryAttempt = 0 private pairingRejected = false + private hostSignedOut = false private readonly listeners = new Set<() => void>() constructor(private readonly isConnected: () => boolean) {} @@ -35,12 +36,23 @@ export class LogicalClientConnectionPath { }) } + isHostSignedOut(): boolean { + return this.hostSignedOut + } + + setHostSignedOut(signedOut: boolean): void { + this.update(() => { + this.hostSignedOut = signedOut + }) + } + clearAfterConnected(): void { this.migration = null this.recovery = null this.recoveryAttempt = 0 // Why: an authenticated session is the desktop accepting this device. this.pairingRejected = false + this.hostSignedOut = false } setRecovery(path: MobileConnectionPath | null, attempt?: number): void { @@ -69,11 +81,13 @@ export class LogicalClientConnectionPath { const previousPath = this.pending() const previousAttempt = this.reconnectAttempt(0) const previousRejected = this.pairingRejected + const previousSignedOut = this.hostSignedOut apply() if ( previousPath === this.pending() && previousAttempt === this.reconnectAttempt(0) && - previousRejected === this.pairingRejected + previousRejected === this.pairingRejected && + previousSignedOut === this.hostSignedOut ) { return } diff --git a/mobile/src/transport/mobile-endpoint-lifecycle.ts b/mobile/src/transport/mobile-endpoint-lifecycle.ts index 8ee8df6948e..7ec5f28b945 100644 --- a/mobile/src/transport/mobile-endpoint-lifecycle.ts +++ b/mobile/src/transport/mobile-endpoint-lifecycle.ts @@ -86,7 +86,7 @@ function createSupervisor( ): MobileEndpointSupervisor { return new MobileEndpointSupervisor(logical, host, { openDirect: (endpoint) => connect(endpoint, host.deviceToken, host.publicKeyB64, { onLog }), - openRelay: (relay, credential, confirmReqId) => + openRelay: (relay, credential, confirmReqId, onHostCloseReason) => connectMobileRelayRpcSession({ relay, resumeToken: credential.token, @@ -94,6 +94,7 @@ function createSupervisor( resumeConfirmReqId: confirmReqId, deviceToken: host.deviceToken, desktopPublicKeyB64: host.publicKeyB64, + onHostCloseReason, onLog }), resolveRelay: resolveMobileRelayEndpoint, diff --git a/mobile/src/transport/mobile-endpoint-supervisor-contract.ts b/mobile/src/transport/mobile-endpoint-supervisor-contract.ts index 0098de6e079..2a784fd8895 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-contract.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-contract.ts @@ -1,4 +1,5 @@ import type { MobileRelayEndpoint } from '../../../src/shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../src/shared/relay-host-close-reason' import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bundle' import type { MobileRelayRpcSession } from './mobile-relay-rpc-session' import type { resolveMobileRelayEndpoint } from './mobile-relay-resume-director' @@ -10,7 +11,8 @@ export type MobileEndpointSupervisorDependencies = { openRelay: ( relay: MobileRelayEndpoint, credential: { token: string; version: number }, - confirmReqId: string + confirmReqId: string, + onHostCloseReason?: (reason: RelayHostCloseReason) => void ) => MobileRelayRpcSession resolveRelay: typeof resolveMobileRelayEndpoint readBundle: (hostId: string) => Promise diff --git a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts index 1dc1473d9db..80f4438c160 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts @@ -134,10 +134,22 @@ export class FakeLogicalClient extends FakeSession implements StableLogicalRpcCl } }) isPairingRejected = () => this.pairingRejected + private hostSignedOut = false + setHostSignedOut = vi.fn((signedOut: boolean) => { + if (this.hostSignedOut === signedOut) { + return + } + this.hostSignedOut = signedOut + for (const listener of this.pathListeners) { + listener() + } + }) + isHostSignedOut = () => this.hostSignedOut // Mirrors LogicalClientConnectionPath.clearAfterConnected. publishState(state: ConnectionState): void { if (state === 'connected') { this.pairingRejected = false + this.hostSignedOut = false } super.publishState(state) } diff --git a/mobile/src/transport/mobile-endpoint-supervisor.test.ts b/mobile/src/transport/mobile-endpoint-supervisor.test.ts index aeb9cddef63..10ef892a479 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor.test.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor.test.ts @@ -185,7 +185,12 @@ describe('mobile endpoint supervisor', () => { await supervisor.start() expect(deps.resolveRelay).toHaveBeenCalledOnce() - expect(openRelay).toHaveBeenLastCalledWith(resolved, expect.any(Object), expect.any(String)) + expect(openRelay).toHaveBeenLastCalledWith( + resolved, + expect.any(Object), + expect.any(String), + expect.any(Function) + ) expect(deps.saveHost).toHaveBeenCalledWith( expect.objectContaining({ relay: resolved, endpoint: host.endpoint }) ) @@ -556,7 +561,8 @@ describe('mobile endpoint supervisor', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) supervisor.stop() }) @@ -603,7 +609,8 @@ describe('mobile endpoint supervisor', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) supervisor.stop() }) diff --git a/mobile/src/transport/mobile-relay-e2ee-link.ts b/mobile/src/transport/mobile-relay-e2ee-link.ts index 7743deb23dd..9b1f7a9a355 100644 --- a/mobile/src/transport/mobile-relay-e2ee-link.ts +++ b/mobile/src/transport/mobile-relay-e2ee-link.ts @@ -2,6 +2,10 @@ import { RelayPhoneHelloSchema, type RelayPhoneHello } from '../../../src/shared/mobile-relay-phone-protocol' +import { + relayHostCloseReasonFrom, + type RelayHostCloseReason +} from '../../../src/shared/relay-host-close-reason' import { MobileE2EEV2ClientSession } from './mobile-e2ee-v2-client-session' import { MobileE2EEV2PhysicalChannel } from './mobile-e2ee-v2-physical-channel' import { websocketPayloadToUint8 } from './websocket-payload-bytes' @@ -26,6 +30,12 @@ type MobileRelayE2eeLinkOptions = { onText: (plaintext: string) => void onBinary: (plaintext: Uint8Array) => void onHello?: (hello: Extract) => void + // The cell's account of why the desktop is absent, read off the close frame. + // Reported separately from onError because a rejection is delivered as both a + // relay-hello and a close, and which one the runtime dispatches first is not + // ordered — only the close carries the reason, and it must not be lost to + // that race. + onHostCloseReason?: (reason: RelayHostCloseReason) => void // Fired once relay-auth is on the wire: from here the cell owns the wait. onOpen?: () => void onError: (error: Error) => void @@ -129,6 +139,11 @@ export class MobileRelayE2eeLink { clearTimeout(this.transportErrorTimer) this.transportErrorTimer = null } + // Ahead of fail(), which no-ops once the hello already reported this close. + const hostCloseReason = relayHostCloseReasonFrom(event.reason) + if (hostCloseReason) { + this.options.onHostCloseReason?.(hostCloseReason) + } this.fail(new RelayOuterError(event.code || 1006)) } } diff --git a/mobile/src/transport/mobile-relay-rpc-session.ts b/mobile/src/transport/mobile-relay-rpc-session.ts index 947a1d23ce8..203a0329192 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.ts @@ -13,6 +13,7 @@ import { RelayDialStageTracker, type RelayDialStageSource } from './relay-dial-s import { RelayPendingRequests } from './relay-pending-requests' import { RpcSessionLivenessWatchdog } from './rpc-session-liveness-watchdog' import { settleMobileRuntimeCapabilities } from './mobile-runtime-capability-negotiation' +import type { RelayHostCloseReason } from '../../../src/shared/relay-host-close-reason' import type { RpcClient } from './rpc-client' import type { ConnectionLogSink, ConnectionState, RpcResponse } from './types' @@ -40,6 +41,7 @@ export function connectMobileRelayRpcSession(args: { desktopPublicKeyB64: string requestTimeoutMs?: number createSocket?: (url: string) => WebSocket + onHostCloseReason?: (reason: RelayHostCloseReason) => void onLog?: ConnectionLogSink }): MobileRelayRpcSession { const requestTimeoutMs = args.requestTimeoutMs ?? 30_000 @@ -69,6 +71,7 @@ export function connectMobileRelayRpcSession(args: { deviceToken: args.deviceToken, desktopPublicKeyB64: args.desktopPublicKeyB64, createSocket: args.createSocket, + onHostCloseReason: args.onHostCloseReason, onOpen: () => dialStage.advance('awaiting-hello'), onHello: (hello) => { if ( diff --git a/mobile/src/transport/mobile-relay-runtime-failover.test.ts b/mobile/src/transport/mobile-relay-runtime-failover.test.ts index 01f4d45feb0..ce7cca3fd9f 100644 --- a/mobile/src/transport/mobile-relay-runtime-failover.test.ts +++ b/mobile/src/transport/mobile-relay-runtime-failover.test.ts @@ -145,10 +145,22 @@ class FakeLogicalClient extends FakeSession implements StableLogicalRpcClient { } }) isPairingRejected = () => this.pairingRejected + private hostSignedOut = false + setHostSignedOut = vi.fn((signedOut: boolean) => { + if (this.hostSignedOut === signedOut) { + return + } + this.hostSignedOut = signedOut + for (const listener of this.pathListeners) { + listener() + } + }) + isHostSignedOut = () => this.hostSignedOut // Mirrors LogicalClientConnectionPath.clearAfterConnected. publishState(state: ConnectionState): void { if (state === 'connected') { this.pairingRejected = false + this.hostSignedOut = false } super.publishState(state) } @@ -264,7 +276,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() @@ -353,7 +366,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(deps.openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 2 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() @@ -382,7 +396,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 1 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() diff --git a/mobile/src/transport/mobile-relay-session-establisher.ts b/mobile/src/transport/mobile-relay-session-establisher.ts index 7a8ce372155..9a04ae44137 100644 --- a/mobile/src/transport/mobile-relay-session-establisher.ts +++ b/mobile/src/transport/mobile-relay-session-establisher.ts @@ -10,6 +10,7 @@ import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bund import type { RelayReconnectController } from './mobile-relay-reconnect-controller' import type { StableLogicalRpcClient } from './stable-logical-rpc-client' import type { MobileRelayEndpoint } from '../../../src/shared/mobile-relay-credential-contract' +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' import type { HostProfile } from './types' type EstablishResult = { ok: true } | { ok: false; error: Error } @@ -100,7 +101,16 @@ export class MobileRelaySessionEstablisher { const session = args.openRelay( relay, credential, - `confirm-${encodeBase64Url(args.randomBytes(16))}` + `confirm-${encodeBase64Url(args.randomBytes(16))}`, + // Latched on the logical client, not on the dial result: the close that + // carries the reason can land after this dial has already reported its + // failure. Clearing is clearAfterConnected's job, so any path that + // reaches connected retires it. + (reason) => { + if (reason === RELAY_HOST_CLOSE_REASON.SIGNED_OUT) { + args.logical.setHostSignedOut(true) + } + } ) try { // Why: backgrounding or a direct winner withdraws this dial before cutover. diff --git a/mobile/src/transport/relay-host-signed-out-supervisor.test.ts b/mobile/src/transport/relay-host-signed-out-supervisor.test.ts new file mode 100644 index 00000000000..cc7a9ac6c35 --- /dev/null +++ b/mobile/src/transport/relay-host-signed-out-supervisor.test.ts @@ -0,0 +1,60 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { MOBILE_RELAY_CLOSE_CODE } from '../../../src/shared/mobile-relay-close-codes' +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' +import { RelayOuterError } from './mobile-relay-e2ee-link' +import { + dependencies, + FakeLogicalClient, + FakeRelaySession, + host +} from './mobile-endpoint-supervisor-test-fakes' +import { MobileEndpointSupervisor } from './mobile-endpoint-supervisor' + +vi.mock('react-native', () => ({ Platform: { OS: 'ios' } })) +vi.mock('expo-secure-store', () => ({ WHEN_UNLOCKED_THIS_DEVICE_ONLY: 'when-unlocked' })) +vi.mock('expo-crypto', () => ({ getRandomBytes: (length: number) => new Uint8Array(length) })) + +// The reason travels from the cell's close frame to the screens. This covers +// the production wiring between them: the supervisor's own openRelay callback. +describe('a signed-out desktop reaches the phone verdict', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-07-13T12:00:00Z')) + }) + afterEach(() => vi.useRealTimers()) + + function supervisorOver(closeReason: string | null) { + const logical = new FakeLogicalClient('disconnected', 'lan') + const deps = dependencies({ + openDirect: vi.fn(() => new FakeRelaySession('disconnected')), + openRelay: vi.fn((_relay, _credential, _confirmReqId, onHostCloseReason) => { + if (closeReason) { + onHostCloseReason?.(closeReason as never) + } + return new FakeRelaySession( + 'disconnected', + new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE) + ) + }) + }) + return { logical, supervisor: new MobileEndpointSupervisor(logical, host, deps) } + } + + it('latches the sign-out the cell reported', async () => { + const { logical, supervisor } = supervisorOver(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + await supervisor.start() + await vi.waitFor(() => expect(logical.isHostSignedOut()).toBe(true)) + + supervisor.stop() + }) + + it('stays quiet for an ordinary host-offline rejection', async () => { + const { logical, supervisor } = supervisorOver(null) + + await supervisor.start() + + expect(logical.isHostSignedOut()).toBe(false) + supervisor.stop() + }) +}) diff --git a/mobile/src/transport/relay-host-signed-out-verdict.test.ts b/mobile/src/transport/relay-host-signed-out-verdict.test.ts new file mode 100644 index 00000000000..2607b922b58 --- /dev/null +++ b/mobile/src/transport/relay-host-signed-out-verdict.test.ts @@ -0,0 +1,216 @@ +import { describe, expect, it, vi } from 'vitest' + +vi.mock('./mobile-e2ee-v2-client-session', () => ({ + MobileE2EEV2ClientSession: { create: () => ({}) } +})) + +vi.mock('./mobile-e2ee-v2-physical-channel', () => ({ + MobileE2EEAuthenticationError: class extends Error {}, + MobileE2EEV2PhysicalChannel: class { + start = vi.fn() + handleMessage = vi.fn(async () => {}) + sendText = vi.fn(() => true) + sendBinary = vi.fn(() => true) + dispose = vi.fn() + } +})) + +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' +import { MOBILE_RELAY_CLOSE_CODE } from '../../../src/shared/mobile-relay-close-codes' +import { classifyConnection, verdictDisplayLabel } from './connection-health' +import { MobileRelayE2eeLink, RelayOuterError } from './mobile-relay-e2ee-link' +import { LogicalClientConnectionPath } from './logical-client-connection-path' +import { RelayReconnectController } from './mobile-relay-reconnect-controller' + +const SIGNED_OUT_LABEL = 'Desktop signed out — sign in to Orca on your desktop to reconnect' + +class FakeSocket { + static readonly OPEN = 1 + readonly OPEN = FakeSocket.OPEN + readyState = FakeSocket.OPEN + bufferedAmount = 0 + onopen: (() => void) | null = null + onmessage: ((event: { data: unknown }) => void) | null = null + onerror: (() => void) | null = null + onclose: ((event: { code: number; reason: string }) => void) | null = null + send = vi.fn() + close = vi.fn() +} + +function linkOver( + socket: FakeSocket, + onHostCloseReason: (reason: string) => void, + onError: (error: Error) => void +): MobileRelayE2eeLink { + return new MobileRelayE2eeLink({ + endpoint: { cellUrl: 'https://relay-c1.onorca.dev', relayHostId: 'AbCdEf0123_-xyZ9' }, + credential: 'credential', + expectedCredentialKind: 'resume', + deviceToken: 'device-token', + desktopPublicKeyB64: 'desktop-key', + onAuthenticated: vi.fn(), + onText: vi.fn(), + onBinary: vi.fn(), + onHostCloseReason, + onError, + createSocket: () => socket as unknown as WebSocket + }) +} + +describe('relay close reason on the phone', () => { + it('reports the cell close reason and still fails with 4404', () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + const onError = vi.fn() + linkOver(socket, onHostCloseReason, onError) + + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + + expect(onHostCloseReason).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + expect(onError).toHaveBeenCalledWith(new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE)) + }) + + // An old cell sends its constant, and every other close sends nothing. + it('reports nothing for a reason it does not know', () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + linkOver(socket, onHostCloseReason, vi.fn()) + + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: 'relay connection rejected' + }) + + expect(onHostCloseReason).not.toHaveBeenCalled() + }) + + // The rejection arrives as a relay-hello AND a close, in an unordered pair. + // Whichever lands first, the reason must survive. + it('still reports the reason when the hello already failed the link', async () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + linkOver(socket, onHostCloseReason, vi.fn()) + + socket.onmessage?.({ + data: JSON.stringify({ + type: 'relay-hello', + ok: false, + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE + }) + }) + await Promise.resolve() + await Promise.resolve() + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + + expect(onHostCloseReason).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) +}) + +describe('the signed-out signal on the logical client', () => { + it('publishes on change and retires when any path reaches connected', () => { + const path = new LogicalClientConnectionPath(() => false) + const changes = vi.fn() + path.subscribe(changes) + + path.setHostSignedOut(true) + path.setHostSignedOut(true) + expect(path.isHostSignedOut()).toBe(true) + expect(changes).toHaveBeenCalledTimes(1) + + path.clearAfterConnected() + expect(path.isHostSignedOut()).toBe(false) + }) +}) + +describe('RelayReconnectController cadence', () => { + // The reason changes no recovery decision; 4404 keeps the host-offline + // backoff it has always had, so a phone on this build retries exactly as + // often as one that never hears the reason. + it('keeps the host-offline retry delay for a 4404', () => { + const delays: number[] = [] + const controller = new RelayReconnectController( + { + now: () => 0, + randomBytes: () => new Uint8Array([0, 0]), + setTimer: ((callback: () => void, delay: number) => { + delays.push(delay) + return 1 as unknown as ReturnType + }) as unknown as typeof setTimeout, + clearTimer: (() => {}) as unknown as typeof clearTimeout + }, + vi.fn() + ) + + controller.registerFailure(new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE)) + + // hostOfflineDelayMs' 5s floor, not the 250ms transport-backoff floor. + expect(delays.at(-1)).toBe(5_000) + }) +}) + +describe('classifyConnection with a signed-out desktop', () => { + const base = { reconnectAttempts: 0, lastConnectedAt: null, hostSignedOut: true } + + it('says so from the first failed dial instead of "Connecting via Relay…"', () => { + const verdict = classifyConnection({ + ...base, + state: 'connecting', + pendingPath: 'relay' + }) + + expect(verdict).toEqual({ + kind: 'unreachable', + label: SIGNED_OUT_LABEL, + reason: 'never-connected' + }) + expect(verdictDisplayLabel(verdict)).toBe(SIGNED_OUT_LABEL) + }) + + it('replaces "Can\'t reach desktop" on the direct path too', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', reconnectAttempts: 20 }).label + ).toBe(SIGNED_OUT_LABEL) + }) + + it('reads as stale once this session had been connected', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', lastConnectedAt: 1, nowMs: 2 }).reason + ).toBe('stale') + }) + + // A Tailscale endpoint cannot make "sign in on your desktop" better advice. + it('never appends the Tailscale hint', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', endpoint: '100.64.0.1' }) + ).not.toHaveProperty('hint') + }) + + it('never outranks a connected session', () => { + expect(classifyConnection({ ...base, state: 'connected' }).label).toBe('Connected') + }) + + // Re-pairing, not signing in, is the remedy when the pairing itself is dead. + it('never outranks a revoked pairing', () => { + expect(classifyConnection({ ...base, state: 'reconnecting', pairingRejected: true }).kind).toBe( + 'auth-failed' + ) + }) + + it('leaves every other verdict alone when the desktop is not signed out', () => { + expect( + classifyConnection({ + state: 'connecting', + reconnectAttempts: 0, + lastConnectedAt: null, + pendingPath: 'relay', + hostSignedOut: false + }).label + ).toBe('Connecting via Relay…') + }) +}) diff --git a/mobile/src/transport/rpc-client-context-contract.ts b/mobile/src/transport/rpc-client-context-contract.ts index 54e25973c7f..65262a6fc15 100644 --- a/mobile/src/transport/rpc-client-context-contract.ts +++ b/mobile/src/transport/rpc-client-context-contract.ts @@ -24,6 +24,7 @@ export type RpcClientContextValue = { getActivePath: (hostId: string) => MobileConnectionPath getPendingPath: (hostId: string) => MobileConnectionPath | null isPairingRejected: (hostId: string) => boolean + isHostSignedOut: (hostId: string) => boolean subscribeHostState: (hostId: string, listener: (state: ConnectionState) => void) => () => void getAllClients: () => { hostId: string; client: RpcClient }[] subscribeAllHosts: (listener: () => void) => () => void diff --git a/mobile/src/transport/stable-logical-rpc-client.ts b/mobile/src/transport/stable-logical-rpc-client.ts index d1f1701aed9..fb514128382 100644 --- a/mobile/src/transport/stable-logical-rpc-client.ts +++ b/mobile/src/transport/stable-logical-rpc-client.ts @@ -55,6 +55,9 @@ export type StableLogicalRpcClient = RpcClient & { // Latched when the desktop has repeatedly refused this device's relay credential. setPairingRejected(rejected: boolean): void isPairingRejected(): boolean + // Latched when the relay named the desktop's own sign-out as the reason it is absent. + setHostSignedOut(signedOut: boolean): void + isHostSignedOut(): boolean // Recovery attempts share this signal so status-only changes rerender. onConnectionPathChange(listener: () => void): () => void getGeneration(): number @@ -282,6 +285,8 @@ export function createStableLogicalRpcClient( setRecoveryAttempt: (attempt) => connectionPath.setRecoveryAttempt(attempt), setPairingRejected: (rejected) => connectionPath.setPairingRejected(rejected), isPairingRejected: () => connectionPath.isPairingRejected(), + setHostSignedOut: (signedOut) => connectionPath.setHostSignedOut(signedOut), + isHostSignedOut: () => connectionPath.isHostSignedOut(), onConnectionPathChange: (listener) => connectionPath.subscribe(listener), getGeneration: () => generation } diff --git a/mobile/src/transport/use-all-host-clients.ts b/mobile/src/transport/use-all-host-clients.ts index 03ac5890015..70c709d965f 100644 --- a/mobile/src/transport/use-all-host-clients.ts +++ b/mobile/src/transport/use-all-host-clients.ts @@ -138,6 +138,7 @@ export function useAllHostClients(hostIds: string[], options?: UseAllHostClients path: MobileConnectionPath pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean }>((hostId) => { const client = clientsByHostId.get(hostId) return client @@ -148,7 +149,8 @@ export function useAllHostClients(hostIds: string[], options?: UseAllHostClients state: ctx.getState(hostId), path: ctx.getActivePath(hostId), pendingPath: ctx.getPendingPath(hostId), - pairingRejected: ctx.isPairingRejected(hostId) + pairingRejected: ctx.isPairingRejected(hostId), + hostSignedOut: ctx.isHostSignedOut(hostId) } ] : [] diff --git a/package.json b/package.json index 58f4805d937..c519d7ea6b1 100644 --- a/package.json +++ b/package.json @@ -153,6 +153,7 @@ "repro:live-remote-realistic-freeze": "node config/scripts/live-remote-realistic-freeze-repro.mjs" }, "dependencies": { + "@anthropic-ai/claude-agent-sdk": "0.3.251", "@electron-toolkit/preload": "^3.0.2", "@electron-toolkit/utils": "^4.0.0", "@floating-ui/dom": "1.7.6", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index d7481f2e556..6b59d23e026 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -122,6 +122,9 @@ importers: .: dependencies: + '@anthropic-ai/claude-agent-sdk': + specifier: 0.3.251 + version: 0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4) '@electron-toolkit/preload': specifier: ^3.0.2 version: 3.0.2(electron@43.4.1(supports-color@7.2.0)) @@ -535,6 +538,23 @@ packages: '@antfu/install-pkg@1.1.0': resolution: {integrity: sha512-MGQsmw10ZyI+EJo45CdSER4zEb+p31LpDAFp2Z3gkSd1yqVZGi0Ebx++YTEMonJy4oChEMLsxZ64j8FH6sSqtQ==} + '@anthropic-ai/claude-agent-sdk@0.3.251': + resolution: {integrity: sha512-DqSi8mH2tQYRlVV0G+lJnQ/WbjJZ/a+8cJ3vPuYoqh8esIIvXHm1ZOXV1UPGsFYRnbBytEoiSGitguEXd+sQ+Q==} + engines: {node: '>=18.0.0'} + peerDependencies: + '@anthropic-ai/sdk': '>=0.93.0' + '@modelcontextprotocol/sdk': ^1.29.0 + zod: ^4.0.0 + + '@anthropic-ai/sdk@0.122.0': + resolution: {integrity: sha512-GGPNftt0caaz9MDlmNQGHX8855Ojaduyy5pm9Sm1h7HalCn0cWNb5/bweadJF+4yzbal+QL6ztBa09WAAOzLmQ==} + hasBin: true + peerDependencies: + zod: ^3.25.0 || ^4.0.0 + peerDependenciesMeta: + zod: + optional: true + '@babel/code-frame@7.29.7': resolution: {integrity: sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw==} engines: {node: '>=6.9.0'} @@ -2628,6 +2648,9 @@ packages: resolution: {integrity: sha512-tlqY9xq5ukxTUZBmoOp+m61cqwQD5pHJtFY3Mn8CA8ps6yghLH/Hw8UPdqg4OLmFW3IFlcXnQNmo/dh8HzXYIQ==} engines: {node: '>=18'} + '@stablelib/base64@1.0.1': + resolution: {integrity: sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ==} + '@stablyai/playwright-base@2.1.14': resolution: {integrity: sha512-/iAgMW5tC0ETDo3mFyTzszRrD7rGFIT4fgDgtZxqa9vPhiTLix/1+GeOOBNY0uS+XRLFY0Uc/irsC3XProL47g==} engines: {node: '>=18'} @@ -4493,6 +4516,9 @@ packages: resolution: {integrity: sha512-7MptL8U0cqcFdzIzwOTHoilX9x5BrNqye7Z/LuC7kCMRio1EMSyqRK3BEAUD7sXRq4iT4AzTVuZdhgQ2TCvYLg==} engines: {node: '>=8.6.0'} + fast-sha256@1.3.0: + resolution: {integrity: sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ==} + fast-string-truncated-width@3.0.3: resolution: {integrity: sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g==} @@ -5039,6 +5065,10 @@ packages: json-parse-even-better-errors@2.3.1: resolution: {integrity: sha512-xyFwyhro/JEof6Ghe2iz2NcXoj2sloNsWr/XsERDK/oiPCfaNhl5ONfp+jQdAZRQQ0IJWNzH9zIZF7li91kh2w==} + json-schema-to-ts@3.1.1: + resolution: {integrity: sha512-+DWg8jCJG2TEnpy7kOm/7/AxaYoaRbjVB4LFZLySZlWn8exGs3A4OLJR966cVvU26N7X9TWxl+Jsw7dzAqKT6g==} + engines: {node: '>=16'} + json-schema-traverse@1.0.0: resolution: {integrity: sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==} @@ -6425,6 +6455,9 @@ packages: stackback@0.0.2: resolution: {integrity: sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw==} + standardwebhooks@1.1.1: + resolution: {integrity: sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ==} + stat-mode@1.0.0: resolution: {integrity: sha512-jH9EhtKIjuXZ2cWxmXS8ZP80XyC3iasQxMDV8jzhNJpfDb7VbQLVW4Wvsxz9QZvzV+G4YoSfBUVKDOyxLzi/sg==} engines: {node: '>= 6'} @@ -6608,6 +6641,9 @@ packages: truncate-utf8-bytes@1.0.2: resolution: {integrity: sha512-95Pu1QXQvruGEhv62XCMO3Mm90GscOCClvrIUwCM0PYOXK3kaF3l3sIHxx71ThJfcbM2O5Au6SO3AWCSEfW4mQ==} + ts-algebra@2.0.0: + resolution: {integrity: sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw==} + ts-dedent@2.2.0: resolution: {integrity: sha512-q5W7tVM71e2xjHZTlgfTDoPF/SmqKG5hddq9SzR49CH2hayqRKJtQ4mtRlSxKaJlR/+9rEM+mnBHf7I2/BQcpQ==} engines: {node: '>=6.10'} @@ -7007,6 +7043,16 @@ packages: zwitch@2.0.4: resolution: {integrity: sha512-bXE4cR/kVZhKZX/RjPEflHaKVhUVl85noU3v6b8apfQEc1x4A+zBxjZ4lN8LqGd6WZ3dl98pY4o717VFmoPp+A==} +ignoredOptionalDependencies: + - '@anthropic-ai/claude-agent-sdk-darwin-arm64' + - '@anthropic-ai/claude-agent-sdk-darwin-x64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64-musl' + - '@anthropic-ai/claude-agent-sdk-linux-x64' + - '@anthropic-ai/claude-agent-sdk-linux-x64-musl' + - '@anthropic-ai/claude-agent-sdk-win32-arm64' + - '@anthropic-ai/claude-agent-sdk-win32-x64' + snapshots: '@adobe/css-tools@4.5.0': {} @@ -7016,6 +7062,19 @@ snapshots: package-manager-detector: 1.6.0 tinyexec: 1.1.2 + '@anthropic-ai/claude-agent-sdk@0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4)': + dependencies: + '@anthropic-ai/sdk': 0.122.0(zod@4.5.4) + '@modelcontextprotocol/sdk': 1.30.0(supports-color@7.2.0)(zod@4.5.4) + zod: 4.5.4 + + '@anthropic-ai/sdk@0.122.0(zod@4.5.4)': + dependencies: + json-schema-to-ts: 3.1.1 + standardwebhooks: 1.1.1 + optionalDependencies: + zod: 4.5.4 + '@babel/code-frame@7.29.7': dependencies: '@babel/helper-validator-identifier': 7.29.7 @@ -7669,6 +7728,28 @@ snapshots: dependencies: '@chevrotain/types': 11.1.2 + '@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4)': + dependencies: + '@hono/node-server': 2.1.0(hono@4.13.0) + ajv: 8.20.0 + ajv-formats: 3.0.1(ajv@8.20.0) + content-type: 1.0.5 + cors: 2.8.6 + cross-spawn: 7.0.6 + eventsource: 3.0.7 + eventsource-parser: 3.0.8 + express: 5.2.1(supports-color@7.2.0) + express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0)) + hono: 4.13.0 + jose: 6.2.3 + json-schema-typed: 8.0.2 + pkce-challenge: 5.0.1 + raw-body: 3.0.2 + zod: 4.5.4 + zod-to-json-schema: 3.25.2(zod@4.5.4) + transitivePeerDependencies: + - supports-color + '@modelcontextprotocol/sdk@1.30.0(zod@3.25.76)': dependencies: '@hono/node-server': 2.1.0(hono@4.13.0) @@ -7679,8 +7760,8 @@ snapshots: cross-spawn: 7.0.6 eventsource: 3.0.7 eventsource-parser: 3.0.8 - express: 5.2.1 - express-rate-limit: 8.5.2(express@5.2.1) + express: 5.2.1(supports-color@7.2.0) + express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0)) hono: 4.13.0 jose: 6.2.3 json-schema-typed: 8.0.2 @@ -8917,6 +8998,8 @@ snapshots: '@sindresorhus/merge-streams@4.0.0': {} + '@stablelib/base64@1.0.1': {} + '@stablyai/playwright-base@2.1.14(@playwright/test@1.59.1)(zod@4.5.4)': dependencies: '@playwright/test': 1.59.1 @@ -9923,7 +10006,7 @@ snapshots: bluebird@3.7.2: {} - body-parser@2.3.0: + body-parser@2.3.0(supports-color@7.2.0): dependencies: bytes: 3.1.2 content-type: 2.0.0 @@ -10779,15 +10862,15 @@ snapshots: exponential-backoff@3.1.3: {} - express-rate-limit@8.5.2(express@5.2.1): + express-rate-limit@8.5.2(express@5.2.1(supports-color@7.2.0)): dependencies: - express: 5.2.1 + express: 5.2.1(supports-color@7.2.0) ip-address: 10.4.0 - express@5.2.1: + express@5.2.1(supports-color@7.2.0): dependencies: accepts: 2.0.0 - body-parser: 2.3.0 + body-parser: 2.3.0(supports-color@7.2.0) content-disposition: 1.1.0 content-type: 1.0.5 cookie: 0.7.2 @@ -10797,7 +10880,7 @@ snapshots: encodeurl: 2.0.0 escape-html: 1.0.3 etag: 1.8.1 - finalhandler: 2.1.1 + finalhandler: 2.1.1(supports-color@7.2.0) fresh: 2.0.0 http-errors: 2.0.1 merge-descriptors: 2.0.0 @@ -10808,9 +10891,9 @@ snapshots: proxy-addr: 2.0.7 qs: 6.15.2 range-parser: 1.2.1 - router: 2.2.0 - send: 1.2.1 - serve-static: 2.2.1 + router: 2.2.0(supports-color@7.2.0) + send: 1.2.1(supports-color@7.2.0) + serve-static: 2.2.1(supports-color@7.2.0) statuses: 2.0.2 type-is: 2.1.0 vary: 1.1.2 @@ -10831,6 +10914,8 @@ snapshots: merge2: 1.4.1 micromatch: 4.0.8 + fast-sha256@1.3.0: {} + fast-string-truncated-width@3.0.3: {} fast-string-width@3.0.2: @@ -10871,7 +10956,7 @@ snapshots: dependencies: to-regex-range: 5.0.1 - finalhandler@2.1.1: + finalhandler@2.1.1(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) encodeurl: 2.0.0 @@ -11445,6 +11530,11 @@ snapshots: json-parse-even-better-errors@2.3.1: {} + json-schema-to-ts@3.1.1: + dependencies: + '@babel/runtime': 7.29.7 + ts-algebra: 2.0.0 + json-schema-traverse@1.0.0: {} json-schema-typed@8.0.2: {} @@ -13037,7 +13127,7 @@ snapshots: points-on-curve: 0.2.0 points-on-path: 0.2.1 - router@2.2.0: + router@2.2.0(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) depd: 2.0.0 @@ -13080,7 +13170,7 @@ snapshots: semver@7.8.1: {} - send@1.2.1: + send@1.2.1(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) encodeurl: 2.0.0 @@ -13107,12 +13197,12 @@ snapshots: transitivePeerDependencies: - typescript - serve-static@2.2.1: + serve-static@2.2.1(supports-color@7.2.0): dependencies: encodeurl: 2.0.0 escape-html: 1.0.3 parseurl: 1.3.3 - send: 1.2.1 + send: 1.2.1(supports-color@7.2.0) transitivePeerDependencies: - supports-color @@ -13267,6 +13357,11 @@ snapshots: stackback@0.0.2: {} + standardwebhooks@1.1.1: + dependencies: + '@stablelib/base64': 1.0.1 + fast-sha256: 1.3.0 + stat-mode@1.0.0: {} state-local@1.0.7: {} @@ -13437,6 +13532,8 @@ snapshots: dependencies: utf8-byte-length: 1.0.5 + ts-algebra@2.0.0: {} + ts-dedent@2.2.0: {} ts-morph@26.0.0: @@ -13786,6 +13883,10 @@ snapshots: dependencies: zod: 3.25.76 + zod-to-json-schema@3.25.2(zod@4.5.4): + dependencies: + zod: 4.5.4 + zod@3.25.76: {} zod@4.5.4: {} diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index d241459f884..97920f087c5 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -12,6 +12,20 @@ minimumReleaseAgeExclude: - zod@4.5.4 shamefullyHoist: true +# Orca always launches the user's own resolved Claude CLI via +# pathToClaudeCodeExecutable, so the SDK's bundled ~95 MB-per-platform CLI +# binaries must never be installed. Excluding them is what makes the path +# override mandatory rather than merely preferred. +ignoredOptionalDependencies: + - '@anthropic-ai/claude-agent-sdk-darwin-arm64' + - '@anthropic-ai/claude-agent-sdk-darwin-x64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64-musl' + - '@anthropic-ai/claude-agent-sdk-linux-x64' + - '@anthropic-ai/claude-agent-sdk-linux-x64-musl' + - '@anthropic-ai/claude-agent-sdk-win32-arm64' + - '@anthropic-ai/claude-agent-sdk-win32-x64' + supportedArchitectures: os: - current diff --git a/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt b/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt index 54837bf4d65..b79ed543494 100644 --- a/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt +++ b/src/main/__fixtures__/shell-wrapper-snapshots/daemon-bash-rcfile.txt @@ -127,10 +127,6 @@ __orca_osc133_precmd() { unset __orca_in_command fi printf "\033]133;A\007" - # Why: emit the shell-ready marker here (not a trailing PROMPT_COMMAND entry) - # so a framework that must be last in PROMPT_COMMAND — bash-preexec — is not - # displaced by one of Orca's own hooks. - [[ -n "$__orca_ready_marker" ]] && printf "\033]777;orca-shell-ready\007" return "$exit_code" } __orca_osc133_preexec() { @@ -188,6 +184,11 @@ __orca_osc133_epilogue() { unset __orca_in_prompt_command __orca_adopt_outer_debug_trap trap '__orca_osc133_preexec' DEBUG + # Readline renders PS1 after entering raw mode; prompt hooks still run in cooked mode. + if [[ -n "$__orca_ready_marker" ]]; then + PS1="${PS1-}"'\[\e]777;orca-shell-ready\a\]' + __orca_ready_marker="" + fi } __orca_normalize_prompt_command_part() { local __orca_value="$1" __orca_output_name="$2" __orca_character __orca_chunk diff --git a/src/main/agent-awake-service-platform-assertions.test.ts b/src/main/agent-awake-service-platform-assertions.test.ts index cd1d7fb1adc..7b3566b322f 100644 --- a/src/main/agent-awake-service-platform-assertions.test.ts +++ b/src/main/agent-awake-service-platform-assertions.test.ts @@ -22,6 +22,20 @@ function workingStatus(): AgentAwakeStatus { } } +describe('AgentAwakeService status array ownership', () => { + it('does not observe rows appended to the caller array after setStatuses', () => { + const service = new AgentAwakeService() + service.setMode('auto') + const statuses: AgentAwakeStatus[] = [workingStatus()] + + service.setStatuses(statuses) + const before = service.getWorkingAgentCount() + statuses.push(workingStatus(), workingStatus()) + + expect(service.getWorkingAgentCount()).toBe(before) + }) +}) + function createBlocker() { const startedIds = new Set() let nextId = 1 diff --git a/src/main/agent-awake-service.ts b/src/main/agent-awake-service.ts index b79612e2b9c..6be27e9d0e6 100644 --- a/src/main/agent-awake-service.ts +++ b/src/main/agent-awake-service.ts @@ -105,7 +105,8 @@ export class AgentAwakeService { } setStatuses(statuses: AgentAwakeStatus[]): void { - this.statuses = statuses.map((status) => ({ ...status })) + // Copy the array, not every row: the hook server allocates each row fresh per event. + this.statuses = [...statuses] this.refresh('status-change') } @@ -171,7 +172,8 @@ export class AgentAwakeService { private getEligibleRunningStatusCount(): number { const now = this.now() - return this.statuses.filter((status) => this.isWakeEligible(status, now)).length + // Counted in place: the filtered array was only ever measured, and this runs per hook event. + return this.statuses.reduce((count, s) => count + (this.isWakeEligible(s, now) ? 1 : 0), 0) } private isWakeEligible(status: AgentAwakeStatus, now: number): boolean { diff --git a/src/main/automations/precheck-runner.ts b/src/main/automations/precheck-runner.ts index 753bd784b06..ab38fd42355 100644 --- a/src/main/automations/precheck-runner.ts +++ b/src/main/automations/precheck-runner.ts @@ -4,6 +4,7 @@ import type { AutomationPrecheck, AutomationPrecheckResult } from '../../shared/ import { MAX_AUTOMATION_PRECHECK_OUTPUT_CHARS } from '../../shared/automation-precheck' import { getSshConnectionManager } from '../ipc/ssh' import { shellEscape } from '../ssh/ssh-connection-utils' +import { admitSelfInitiatedTreeKill } from '../own-chromium-tree-kill-guard' type AutomationPrecheckExecutionTarget = | { @@ -73,7 +74,10 @@ function failedPrecheckResult( }) } -function killLocalPrecheckProcessTree(child: ChildProcess): ReturnType | null { +/** Exported for the refusal-fallback test; the timeout path is otherwise unreachable. */ +export function killLocalPrecheckProcessTree( + child: ChildProcess +): ReturnType | null { const pid = child.pid if (!pid) { child.kill() @@ -81,6 +85,18 @@ function killLocalPrecheckProcessTree(child: ChildProcess): ReturnType + > +): Parameters[0] { + return { + claudeManagedAccounts: [HOST_ACCOUNT, WSL_ACCOUNT, LEGACY_ACCOUNT], + activeClaudeManagedAccountId: null, + ...overrides + } as Parameters[0] +} + +// The predicate now backs BOTH transports (runtime-auth-preparation.ts and the +// structured wiring), so it needs a test of its own: forcing it to a constant used +// to leave ~1000 tests green. +describe('shouldStripClaudeAuthEnvForAccount', () => { + it('does not strip when no managed account is selected', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], null)).toBe(false) + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], undefined)).toBe(false) + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], '')).toBe(false) + }) + + it('strips for a host-managed account', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT, WSL_ACCOUNT], 'host-a')).toBe(true) + }) + + it('strips for an account with no explicit runtime (the legacy host shape)', () => { + expect(shouldStripClaudeAuthEnvForAccount([LEGACY_ACCOUNT], 'legacy-c')).toBe(true) + }) + + it('does not strip for a WSL-managed account, matching runtime-auth-preparation', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT, WSL_ACCOUNT], 'wsl-b')).toBe(false) + }) + + it('strips for a selected id no account list explains', () => { + // Fail-safe: an id we cannot resolve is treated as a pinned account, never as + // "no account", so an unreadable settings blob cannot open the strip. + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], 'deleted-d')).toBe(true) + expect(shouldStripClaudeAuthEnvForAccount(undefined, 'deleted-d')).toBe(true) + expect(shouldStripClaudeAuthEnvForAccount([], 'deleted-d')).toBe(true) + }) +}) + +describe('claudeStructuredAuthPolicyForSettings', () => { + it('reads the host runtime selection, not the legacy flat field alone', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ + activeClaudeManagedAccountId: 'host-a', + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } + }) + ) + ).toEqual({ stripAuthEnv: true }) + }) + + it('strips when a host account is pinned by runtime selection', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ activeClaudeManagedAccountIdsByRuntime: { host: 'host-a', wsl: {} } }) + ) + ).toEqual({ stripAuthEnv: true }) + }) + + it('does not strip for system auth, so an API-key-only user keeps their sign-in', () => { + expect(claudeStructuredAuthPolicyForSettings(settings({}))).toEqual({ stripAuthEnv: false }) + }) + + it('ignores a WSL-only selection: the structured child is always a native host process', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-b' } } + }) + ) + ).toEqual({ stripAuthEnv: false }) + }) +}) + +describe('the strip vocabulary the policy governs', () => { + it('covers every Anthropic auth variable the terminal path knows about', () => { + // A new auth var added to the list without a matching refusal/strip path is the + // shape of the leak this lane already shipped once. + expect([...CLAUDE_AUTH_ENV_VARS]).toEqual([ + 'ANTHROPIC_API_KEY', + 'ANTHROPIC_AUTH_TOKEN', + 'CLAUDE_CODE_OAUTH_TOKEN', + 'AWS_BEARER_TOKEN_BEDROCK' + ]) + }) +}) + +// The refusal has to cover exactly what the strip removes. Anything narrower lets an +// override reach the child that applyClaudeEnvPatch would have deleted. +describe('hasClaudeAuthEnvConflict matches the strip it guards', () => { + it('refuses each Anthropic auth variable', () => { + for (const key of CLAUDE_AUTH_ENV_VARS) { + expect(hasClaudeAuthEnvConflict({ [key]: 'v' }, 'linux')).toBe(true) + } + }) + + // `ANTHROPIC_API_KEY=` in the agent env box is how a user blanks a variable, and the + // settings pipeline preserves the empty value (agent-default-env-draft.ts assigns + // everything after the `=`; normalizeTuiAgentEnvRecord drops empty KEYS only). An + // empty value cannot beat the pinned account and the strip removes the name anyway, + // so refusing it would break a terminal launch that works today for no security gain. + it('admits an override whose value is empty, the documented way to blank a variable', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_API_KEY: '' }, 'linux')).toBe(false) + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: '' }, 'win32')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_CUSTOM_HEADERS: '' }, 'linux')).toBe(false) + }) + + it('still refuses the same names once they carry a value', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_API_KEY: 'sk-ant' }, 'linux')).toBe(true) + }) + + // The end-to-end shape the regression actually took: settings text -> normalized + // record -> launch env -> the predicate the terminal preflight gates on. + it('admits a blanked variable all the way from the settings record', () => { + const configured = normalizeTuiAgentEnvRecord({ claude: { ANTHROPIC_API_KEY: '' } }) + const launchEnv = resolveTuiAgentLaunchEnv('claude', configured) + + expect(launchEnv).toEqual({ ANTHROPIC_API_KEY: '' }) + expect(hasClaudeAuthEnvConflict(launchEnv, 'linux')).toBe(false) + }) + + it('folds case on win32, where the OS does', () => { + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: 'sk-lower' }, 'win32')).toBe(true) + expect(hasClaudeAuthEnvConflict({ Anthropic_Custom_Headers: 'x-api-key: v' }, 'win32')).toBe( + true + ) + }) + + it('keeps env names case-sensitive off win32', () => { + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: 'sk-lower' }, 'linux')).toBe(false) + }) + + it('admits non-auth Anthropic settings on both platforms', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_BASE_URL: 'https://gw.test' }, 'linux')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_BASE_URL: 'https://gw.test' }, 'win32')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_CUSTOM_HEADERS: 'X-Trace: 1' }, 'linux')).toBe( + false + ) + expect(hasClaudeAuthEnvConflict(undefined, 'linux')).toBe(false) + }) +}) diff --git a/src/main/claude-accounts/claude-structured-auth-policy.ts b/src/main/claude-accounts/claude-structured-auth-policy.ts new file mode 100644 index 00000000000..c30cd69b827 --- /dev/null +++ b/src/main/claude-accounts/claude-structured-auth-policy.ts @@ -0,0 +1,37 @@ +import type { GlobalSettings } from '../../shared/global-settings-types' +import { shouldStripClaudeAuthEnvForAccount } from './environment' +import { getSelectedClaudeAccountIdForTarget } from './runtime-selection' + +/** The structured mirror of the terminal preflight's `prepareClaudeAuth` result: + * the one field a launch resolution needs from the managed-account state. */ +export type ClaudeStructuredAuthPolicy = { + stripAuthEnv: boolean +} + +/** + * The only supported way to build a structured launch's auth policy. + * + * It exists as a named function rather than an inline object at the wiring site so + * that the settings-to-policy mapping is testable on its own: the one production + * wiring lives in a `@ts-nocheck` file, where neither the compiler nor a type test + * can see a dropped field. + * + * Structured Claude always spawns a native local-host child — the launch resolver + * refuses any record with a remote execution host or a WSL distro — so the host + * selection, not the platform default target, owns its auth. + */ +export function claudeStructuredAuthPolicyForSettings( + settings: Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' + > +): ClaudeStructuredAuthPolicy { + return { + stripAuthEnv: shouldStripClaudeAuthEnvForAccount( + settings.claudeManagedAccounts, + getSelectedClaudeAccountIdForTarget(settings, { runtime: 'host' }) + ) + } +} diff --git a/src/main/claude-accounts/environment.ts b/src/main/claude-accounts/environment.ts index 83fe3b40209..b85dd60a854 100644 --- a/src/main/claude-accounts/environment.ts +++ b/src/main/claude-accounts/environment.ts @@ -1,3 +1,5 @@ +import type { ClaudeManagedAccount } from '../../shared/managed-account-types' + export const CLAUDE_AUTH_ENV_VARS = [ 'ANTHROPIC_API_KEY', 'ANTHROPIC_AUTH_TOKEN', @@ -13,14 +15,21 @@ export type ClaudeEnvPatch = { export function applyClaudeEnvPatch( baseEnv: Record, patch: ClaudeEnvPatch, - options?: { stripAuthEnv?: boolean } + options?: { stripAuthEnv?: boolean; platform?: NodeJS.Platform } ): Record { if (options?.stripAuthEnv) { for (const key of CLAUDE_AUTH_ENV_VARS) { delete baseEnv[key] } - if (isAuthLikeCustomHeaders(baseEnv.ANTHROPIC_CUSTOM_HEADERS)) { - delete baseEnv.ANTHROPIC_CUSTOM_HEADERS + const platform = options.platform ?? process.platform + for (const key of Object.keys(baseEnv)) { + const normalized = platform === 'win32' ? key.toUpperCase() : key + if ( + (platform === 'win32' && CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized)) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(baseEnv[key])) + ) { + delete baseEnv[key] + } } } @@ -34,16 +43,94 @@ export function applyClaudeEnvPatch( return baseEnv } -export function hasClaudeAuthEnvConflict(env: Record | undefined): boolean { - if (!env) { +/** One string for every transport, so a terminal launch and a structured launch + * cannot drift into telling the user two different things about one refusal. */ +export const CLAUDE_AUTH_ENV_CONFLICT_MESSAGE = + 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' + +export const CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE = + 'A Claude account switch is in progress. Try again after it finishes.' + +/** + * Whether a launch on the host runtime must drop inherited Anthropic auth. + * + * Only a pinned host-managed account owns the credential, so only it may strip: + * with no managed account the user's own `ANTHROPIC_*` is their sign-in, and + * removing it signs them out of a CLI that would otherwise have worked. + */ +export function shouldStripClaudeAuthEnvForAccount( + accounts: readonly ClaudeManagedAccount[] | undefined, + activeAccountId: string | null | undefined +): boolean { + if (!activeAccountId) { return false } return ( - CLAUDE_AUTH_ENV_VARS.some((key) => Boolean(env[key])) || - isAuthLikeCustomHeaders(env.ANTHROPIC_CUSTOM_HEADERS) + (accounts ?? []).find((account) => account.id === activeAccountId)?.managedAuthRuntime !== 'wsl' ) } +/** + * Whether a launch's explicit env carries Anthropic auth a managed account must own. + * + * The key comparison mirrors applyClaudeEnvPatch's strip exactly: case-insensitive on + * win32, where the OS folds env names so `anthropic_api_key` is an effective + * `ANTHROPIC_API_KEY`, and case-sensitive elsewhere. A refusal narrower than the strip + * lets an override through that the strip would have removed. + * + * A non-empty value is what makes it a conflict. `ANTHROPIC_API_KEY=` in the agent env + * box is how a user blanks a variable — the settings pipeline preserves that empty value + * (normalizeTuiAgentEnvRecord drops empty KEYS only) — and an empty override can neither + * authenticate nor beat the pinned account, while the strip removes the name regardless. + * Refusing it would break a terminal launch that works today for no security gain. + */ +/** + * The inherited Anthropic auth a non-stripping launch has to carry forward explicitly. + * + * applyClaudeEnvPatch always strips the inherited half of a child env, and the + * configured half is what overrides it — so a system-auth user's own key only survives + * if the caller puts it back deliberately. Returns the exact keys present, so a + * win32 `anthropic_api_key` is carried under the name the OS actually has. + */ +export function claudeAuthEnvCarriedForward( + inherited: NodeJS.ProcessEnv, + platform: NodeJS.Platform = process.platform +): Record { + const carried: Record = {} + for (const [key, value] of Object.entries(inherited)) { + if (value === undefined) { + continue + } + const normalized = platform === 'win32' ? key.toUpperCase() : key + if ( + CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(value)) + ) { + carried[key] = value + } + } + return carried +} + +export function hasClaudeAuthEnvConflict( + env: Record | undefined, + platform: NodeJS.Platform = process.platform +): boolean { + if (!env) { + return false + } + for (const [key, value] of Object.entries(env)) { + const normalized = platform === 'win32' ? key.toUpperCase() : key + if (value && CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized)) { + return true + } + if (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(value)) { + return true + } + } + return false +} + function isAuthLikeCustomHeaders(value: string | undefined): boolean { if (!value) { return false diff --git a/src/main/claude-accounts/live-pty-gate.ts b/src/main/claude-accounts/live-pty-gate.ts index 9e30b621924..caab66c3430 100644 --- a/src/main/claude-accounts/live-pty-gate.ts +++ b/src/main/claude-accounts/live-pty-gate.ts @@ -5,6 +5,13 @@ const liveClaudePtyIds = new Set() // survived the app restart inside the daemon. const seededUnconfirmedPtyIds = new Set() let switchInProgress = false +// Woken by endClaudeAuthSwitch so a caller past the point of no return can wait the +// swap out instead of refusing. See whenClaudeAuthSwitchSettles. +const switchSettledListeners = new Set<() => void>() + +/** A managed account swap is a credential-file rewrite, not a network round trip; + * anything past this is a wedged switch, and refusing beats waiting forever. */ +export const CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS = 15_000 export type ClaudeLivePtyPersistence = { addClaudeLivePtySessionId(sessionId: string): void @@ -81,6 +88,35 @@ export function markClaudePtyExited(ptyId: string): void { notifyDrainedOnTransition(hadLivePtys) } +/** + * Register a structured Claude child with the same gate the terminal path uses. + * + * The gate is what makes the managed OAuth refresh defer instead of rotating a + * single-use refresh token out from under a running Claude (runtime-auth-sync.ts). + * A structured session's child is as much a live Claude as a PTY's is, so it has to + * hold the gate too — otherwise a refresh mid-turn breaks its next API call while an + * identical terminal session is protected. + * + * Deliberately not persisted, unlike markClaudePtySpawned: these children are direct + * children of this process and cannot survive a restart, so seeding them back on the + * next launch would hold the gate closed for a process that is provably gone. + */ +export function markClaudeStructuredChildSpawned(childKey: string): void { + liveClaudePtyIds.add(structuredChildGateId(childKey)) +} + +export function markClaudeStructuredChildExited(childKey: string): void { + const hadLivePtys = liveClaudePtyIds.size > 0 + liveClaudePtyIds.delete(structuredChildGateId(childKey)) + notifyDrainedOnTransition(hadLivePtys) +} + +// Namespaced so a structured child can never collide with a daemon PTY session id, +// which confirmSeededClaudeLivePtys reconciles against the daemon's own list. +function structuredChildGateId(childKey: string): string { + return `claude-structured:${childKey}` +} + export function hasLiveClaudePtys(): boolean { return liveClaudePtyIds.size > 0 } @@ -93,7 +129,44 @@ export function beginClaudeAuthSwitch(): void { } export function endClaudeAuthSwitch(): void { + const wasInProgress = switchInProgress switchInProgress = false + if (!wasInProgress) { + return + } + // Each listener removes itself as it settles; Set iteration is defined over that. + for (const listener of switchSettledListeners) { + listener() + } +} + +/** + * Resolves `true` once no account switch is running, `false` if one is still running + * at the deadline. + * + * Exists for callers that have already done irreversible work — a structured acquire + * has closed the old child by the time it resolves its launch, so turning a switch + * into a refusal there strands the user with a dead session and no replacement. + * Waiting for the swap and then launching against it is the recoverable answer; + * refusing is only correct when nothing has been torn down yet. + */ +export function whenClaudeAuthSwitchSettles( + timeoutMs = CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS +): Promise { + if (!switchInProgress) { + return Promise.resolve(true) + } + return new Promise((resolve) => { + const settle = (settled: boolean): void => { + switchSettledListeners.delete(listener) + clearTimeout(timer) + resolve(settled) + } + const listener = (): void => settle(true) + switchSettledListeners.add(listener) + const timer = setTimeout(() => settle(false), timeoutMs) + timer.unref?.() + }) } export function isClaudeAuthSwitchInProgress(): boolean { diff --git a/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts b/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts index ae79c4c7bbb..dabcd9d472f 100644 --- a/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts +++ b/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts @@ -2,6 +2,7 @@ import { join } from 'node:path' import type { ClaudeManagedAccount } from '../../../shared/managed-account-types' import { resolveLocalAccountRuntimeTarget } from '../../../shared/local-account-runtime' import { parseWslUncPath } from '../../../shared/wsl-paths' +import { shouldStripClaudeAuthEnvForAccount } from '../environment' import { getDefaultWslDistro, getWslHome } from '../../wsl' import { getSelectedClaudeAccountIdForTarget, @@ -69,7 +70,10 @@ export class ClaudeRuntimeAuthPreparationService extends ClaudeRuntimeAuthSnapsh wslDistro: null, wslLinuxConfigDir: null, envPatch: paths.envPatch, - stripAuthEnv: Boolean(activeAccountId && activeAccount?.managedAuthRuntime !== 'wsl'), + stripAuthEnv: shouldStripClaudeAuthEnvForAccount( + settings.claudeManagedAccounts, + activeAccountId + ), managedRefreshDeferredByLivePty: Boolean( activeAccountId && activeAccount?.managedAuthRuntime !== 'wsl' && diff --git a/src/main/claude-usage/transcript-record-parser-prefilter.test.ts b/src/main/claude-usage/transcript-record-parser-prefilter.test.ts new file mode 100644 index 00000000000..992b4ba8fd3 --- /dev/null +++ b/src/main/claude-usage/transcript-record-parser-prefilter.test.ts @@ -0,0 +1,40 @@ +import { describe, expect, it } from 'vitest' +import { parseClaudeUsageRecord } from './transcript-record-parser' + +function assistantLine(overrides: Record = {}): string { + return JSON.stringify({ + type: 'assistant', + sessionId: 'session-1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { usage: { input_tokens: 3, output_tokens: 5 } }, + ...overrides + }) +} + +describe('assistant-record prefilter', () => { + it('still parses an ordinary assistant record', () => { + expect(parseClaudeUsageRecord(assistantLine())?.inputTokens).toBe(3) + }) + + it('rejects a user record that never mentions assistant', () => { + const userLine = JSON.stringify({ + type: 'user', + sessionId: 'session-1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { content: 'x'.repeat(200) } + }) + + expect(parseClaudeUsageRecord(userLine)).toBeNull() + }) + + it('rejects a non-assistant record that happens to contain the word assistant', () => { + const userLine = JSON.stringify({ + type: 'user', + sessionId: 'session-1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { content: 'ask the assistant about this' } + }) + + expect(parseClaudeUsageRecord(userLine)).toBeNull() + }) +}) diff --git a/src/main/claude-usage/transcript-record-parser.ts b/src/main/claude-usage/transcript-record-parser.ts index 59b0e75be39..2ad51f3ee90 100644 --- a/src/main/claude-usage/transcript-record-parser.ts +++ b/src/main/claude-usage/transcript-record-parser.ts @@ -84,10 +84,30 @@ function dedupeClaudeUsageTurns( return deduped } +/** + * Necessary condition for `JSON.parse(line).type === 'assistant'`, checked before the parse. + * + * Sound for any transcript written by a standard JSON serializer: `JSON.stringify` (which writes + * these files) escapes only quotes, backslashes and control characters, never ASCII letters, so + * the decoded value can only be `assistant` if the line spells it literally. The gate over-admits + * freely — the `parsed.type` check below stays authoritative. + * + * A `\u`-escape fallback was measured and rejected: it costs a second full-line scan and made + * transcripts whose tool results contain control characters 1.43x slower overall. + */ +function mayEncodeAssistantType(line: string): boolean { + return line.includes('assistant') +} + function parseClaudeUsageSourceRecord( line: string, fallbackSessionId: string | null = null ): ClaudeUsageParsedSourceTurn | null { + // Only assistant records carry usage, but transcripts interleave user/tool-result lines that + // routinely embed whole files. Reject those before paying for a full parse. + if (!mayEncodeAssistantType(line)) { + return null + } let parsed: ClaudeUsageSourceRecord try { parsed = JSON.parse(line) as ClaudeUsageSourceRecord diff --git a/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs b/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs new file mode 100644 index 00000000000..4f2a09425fd --- /dev/null +++ b/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs @@ -0,0 +1,144 @@ +// Scripted stand-in for the Claude Code CLI, driven by the SDK contract-pin +// tests. It speaks just enough stream-json to satisfy the SDK: it answers every +// inbound control_request with a success control_response, records everything it +// observes to a report file, and plays back the steps listed in a scenario file. +// +// Env contract (set by the test): +// ORCA_SDK_CONTRACT_SCENARIO_PATH — JSON file +// { steps: Step[], controlResponses?: { [subtype]: } } where a Step is +// { emit: } | { awaitUserMessage: true } | { stderr: } | +// { awaitControlResponse: } | { delayMs: } | { exit: } +// ORCA_SDK_CONTRACT_REPORT_PATH — where argv/env observations are written +// ORCA_SDK_CONTRACT_IGNORE_SIGTERM — trap SIGTERM/SIGINT and outlive stdin close +// ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS — record control requests but never answer +// ORCA_SDK_CONTRACT_DESCENDANT — fork an idle grandchild and report its pid +import { spawn } from 'node:child_process' +import { readFileSync, writeFileSync } from 'node:fs' +import { createInterface } from 'node:readline' + +const scenarioPath = process.env.ORCA_SDK_CONTRACT_SCENARIO_PATH +const reportPath = process.env.ORCA_SDK_CONTRACT_REPORT_PATH + +const report = { + argv: process.argv.slice(1), + execPath: process.execPath, + controlRequests: [], + controlResponses: [], + userMessages: [], + descendantPid: null +} +const writeReport = () => { + if (reportPath) { + writeFileSync(reportPath, JSON.stringify(report)) + } +} +// Written immediately so a test can prove which script the SDK executed even if +// the session dies before the scenario completes. +writeReport() + +const scenario = scenarioPath ? JSON.parse(readFileSync(scenarioPath, 'utf8')) : { steps: [] } + +if (process.env.ORCA_SDK_CONTRACT_IGNORE_SIGTERM) { + process.on('SIGTERM', () => {}) + process.on('SIGINT', () => {}) + setInterval(() => {}, 1_000_000) +} +if (process.env.ORCA_SDK_CONTRACT_DESCENDANT) { + const descendant = spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000000)'], { + stdio: 'ignore' + }) + descendant.unref() + report.descendantPid = descendant.pid ?? null + writeReport() +} + +const emit = (frame) => process.stdout.write(`${JSON.stringify(frame)}\n`) + +const waiters = [] +const settle = (kind, requestId) => { + for (let i = waiters.length - 1; i >= 0; i--) { + const waiter = waiters[i] + if ( + waiter.kind === kind && + (waiter.requestId === undefined || waiter.requestId === requestId) + ) { + waiters.splice(i, 1) + waiter.resolve() + } + } +} +const waitFor = (kind, requestId) => { + if (kind === 'user' && report.userMessages.length > 0) { + return Promise.resolve() + } + if ( + kind === 'control_response' && + report.controlResponses.some((frame) => frame.response?.request_id === requestId) + ) { + return Promise.resolve() + } + return new Promise((resolve) => waiters.push({ kind, requestId, resolve })) +} + +createInterface({ input: process.stdin }).on('line', (line) => { + let frame + try { + frame = JSON.parse(line) + } catch { + return + } + if (frame.type === 'control_request') { + report.controlRequests.push(frame) + writeReport() + if (process.env.ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS) { + return + } + emit({ + type: 'control_response', + response: { + subtype: 'success', + request_id: frame.request_id, + response: scenario.controlResponses?.[frame.request?.subtype] ?? { + commands: [], + models: [] + } + } + }) + return + } + if (frame.type === 'control_response') { + report.controlResponses.push(frame) + writeReport() + settle('control_response', frame.response?.request_id) + return + } + if (frame.type === 'user') { + report.userMessages.push(frame) + writeReport() + settle('user') + } +}) + +// Never outlive a wedged test: the readline subscription would otherwise hold +// this process open forever if the SDK side stops driving the scenario. +setTimeout(() => process.exit(3), 20_000).unref() + +for (const step of scenario.steps) { + if (step.emit) { + emit(step.emit) + } else if (step.stderr !== undefined) { + process.stderr.write(step.stderr) + } else if (step.awaitUserMessage) { + await waitFor('user') + } else if (step.awaitControlResponse !== undefined) { + await waitFor('control_response', step.awaitControlResponse) + } else if (step.delayMs) { + await new Promise((resolve) => setTimeout(resolve, step.delayMs)) + } else if (step.exit !== undefined) { + // A CLI that refuses to start: leave with its own status, stderr already written. + writeReport() + process.exit(step.exit) + } +} +writeReport() +process.exit(0) diff --git a/src/main/claude/claude-agent-sdk-contract-pins.test.ts b/src/main/claude/claude-agent-sdk-contract-pins.test.ts new file mode 100644 index 00000000000..46bcb7219d3 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-contract-pins.test.ts @@ -0,0 +1,519 @@ +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import { + query, + type CanUseTool, + type Options, + type SDKUserMessage, + type SpawnedProcess as SdkSpawnedProcess, + type SpawnOptions as SdkSpawnOptions +} from '@anthropic-ai/claude-agent-sdk' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { spawnProcess } from '../../shared/child-process/run-process' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { claudeQuerySettingsReader } from './claude-agent-sdk-control-requests' +import { createClaudeStructuredLaunchResolver } from './claude-structured-launch-resolution' + +// Contract pins for @anthropic-ai/claude-agent-sdk, run against the real SDK +// driving a scripted fake CLI (never the real Claude binary). These tests exist +// to catch a future SDK version drifting under Orca: unknown-frame pass-through, +// spawner env fidelity, argument parity with the pre-SDK argv, +// permission-callback semantics, and executable-path override. + +const FAKE_CLI = join(__dirname, '__fixtures__', 'claude-agent-sdk-scripted-cli.mjs') +const SESSION_ID = '5348c19f-6a54-4c2e-9c68-9c2b1a3d4e5f' +const LEAF_UUID = 'ad0f7c9e-1b2c-4d3e-8f90-abc123def456' +const PINNED_SDK_VERSION = '0.3.251' +const SDK_PLATFORM_PACKAGE_BASENAMES = [ + 'claude-agent-sdk-darwin-arm64', + 'claude-agent-sdk-darwin-x64', + 'claude-agent-sdk-linux-arm64', + 'claude-agent-sdk-linux-arm64-musl', + 'claude-agent-sdk-linux-x64', + 'claude-agent-sdk-linux-x64-musl', + 'claude-agent-sdk-win32-arm64', + 'claude-agent-sdk-win32-x64' +] + +/** + * The exact argv the hand-rolled transport built before the SDK swap. Frozen here + * as the parity oracle: CLAUDE_STRUCTURED_BASE_OPTIONS has to keep producing it. + */ +const PRE_SDK_ARGV = [ + '-p', + '--input-format', + 'stream-json', + '--output-format', + 'stream-json', + '--include-partial-messages', + '--verbose', + '--replay-user-messages', + '--permission-prompt-tool', + 'stdio', + '--setting-sources', + 'user,project,local' +] + +const RESULT_FRAME = { + type: 'result', + subtype: 'success', + is_error: false, + duration_ms: 1, + duration_api_ms: 1, + num_turns: 1, + result: 'ok', + session_id: SESSION_ID, + total_cost_usd: 0, + usage: { input_tokens: 1, output_tokens: 1 }, + uuid: 'uuid-result-1' +} + +type ScenarioStep = Record +type SpawnSeen = { + command: string + args: string[] + cwd: string | undefined + env: Record +} +type ScriptedCliReport = { + argv: string[] + execPath: string + controlRequests: { request_id: string; request: { subtype: string } }[] + controlResponses: { response: { request_id: string; response?: Record } }[] + userMessages: Record[] +} + +const scratchDirs: string[] = [] +afterEach(() => { + vi.unstubAllEnvs() + for (const dir of scratchDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +function scriptScenario( + steps: ScenarioStep[], + controlResponses: Record = {} +): { + scenarioPath: string + reportPath: string + cwd: string + readReport: () => ScriptedCliReport +} { + const dir = mkdtempSync(join(tmpdir(), 'claude-sdk-contract-')) + scratchDirs.push(dir) + const scenarioPath = join(dir, 'scenario.json') + const reportPath = join(dir, 'report.json') + writeFileSync(scenarioPath, JSON.stringify({ steps, controlResponses })) + return { + scenarioPath, + reportPath, + cwd: dir, + readReport: () => JSON.parse(readFileSync(reportPath, 'utf8')) as ScriptedCliReport + } +} + +function scenarioEnv(scenario: { scenarioPath: string; reportPath: string }) { + return { + PATH: process.env.PATH, + ORCA_SDK_CONTRACT_SCENARIO_PATH: scenario.scenarioPath, + ORCA_SDK_CONTRACT_REPORT_PATH: scenario.reportPath + } +} + +function recordingSpawner(spawns: SpawnSeen[]) { + return (opts: SdkSpawnOptions): SdkSpawnedProcess => { + spawns.push({ + command: opts.command, + args: [...opts.args], + cwd: opts.cwd, + env: { ...opts.env } + }) + return spawnProcess({ + program: opts.command, + args: opts.args, + cwd: opts.cwd, + env: opts.env as NodeJS.ProcessEnv, + signal: opts.signal + }) as unknown as SdkSpawnedProcess + } +} + +function resolvedLaunch(launchArgs: string[]) { + const record = { + sessionId: 'contract-pin-session', + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [], + launchArgs + } as unknown as AgentSessionRecord + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => FAKE_CLI, + resolveAuthPolicy: () => ({ stripAuthEnv: true }) + })({ identity: { sessionId: record.sessionId } as never }) +} + +function singleUserTurn(): AsyncIterable { + return (async function* () { + yield { + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + } as SDKUserMessage + // Hold input open; the stream ends when the scripted CLI exits, and an + // unresolved bare promise does not keep the event loop alive. + await new Promise(() => {}) + })() +} + +async function drainQuery(options: Options): Promise[]> { + const messages: Record[] = [] + for await (const message of query({ prompt: singleUserTurn(), options })) { + messages.push(message as unknown as Record) + } + return messages +} + +/** Expand `--flag=value` argv entries so both SDK spellings compare equal. */ +function normalizeArgv(args: string[]): string[] { + return args.flatMap((arg) => { + if (!arg.startsWith('--')) { + return [arg] + } + const eq = arg.indexOf('=') + return eq === -1 ? [arg] : [arg.slice(0, eq), arg.slice(eq + 1)] + }) +} + +/** Group the pre-SDK argv into flag/value pairs. */ +function flagTable(args: readonly string[]): { flag: string; value: string | null }[] { + const table: { flag: string; value: string | null }[] = [] + for (let i = 0; i < args.length; i++) { + const flag = args[i]! + const next = args[i + 1] + if (next !== undefined && !next.startsWith('-')) { + table.push({ flag, value: next }) + i++ + } else { + table.push({ flag, value: null }) + } + } + return table +} + +describe('Claude Agent SDK contract pins', () => { + it('yields unknown types, unknown fields and unknown content blocks verbatim, and consumes keep_alive', async () => { + const unknownTopLevel = { + type: 'message_kind_from_the_future', + session_id: SESSION_ID, + uuid: 'uuid-unknown-1', + payload: { alpha: 1, nested: { flags: ['a', 'b'] } } + } + const assistantWithUnknowns = { + type: 'assistant', + message: { + id: 'msg-1', + type: 'message', + role: 'assistant', + model: 'claude-x', + content: [ + { type: 'text', text: 'hello back' }, + { type: 'content_block_from_the_future', payload: { depth: 3 } } + ], + stop_reason: null, + stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 2 } + }, + parent_tool_use_id: null, + uuid: 'uuid-assistant-1', + session_id: SESSION_ID, + field_from_the_future: 'preserved' + } + const scenario = scriptScenario([ + { awaitUserMessage: true }, + { emit: { type: 'keep_alive' } }, + { emit: unknownTopLevel }, + { emit: assistantWithUnknowns }, + { emit: RESULT_FRAME } + ]) + const spawns: SpawnSeen[] = [] + const messages = await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(messages.find((m) => m.uuid === 'uuid-unknown-1')).toEqual(unknownTopLevel) + expect(messages.find((m) => m.uuid === 'uuid-assistant-1')).toEqual(assistantWithUnknowns) + // The SDK intercepts keep_alive internally — a liveness signal must never + // be derived from it reaching the consumer, because it does not. + expect(messages.some((m) => m.type === 'keep_alive')).toBe(false) + expect(messages.some((m) => m.type === 'result')).toBe(true) + }) + + it('hands the custom spawner exactly the caller-supplied env, plus the two pinned SDK mutations', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'ambient-key-must-not-leak') + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: { + ...scenarioEnv(scenario), + CLAUDE_CONFIG_DIR: '/pinned/claude-config', + ORCA_AGENT_SESSION_SPAWN_TOKEN: 'spawn-token-1', + NODE_OPTIONS: '--max-old-space-size=64' + }, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + const env = spawns[0]!.env + // Supplied values arrive verbatim: the config-dir pin and spawn token are + // observable at this boundary, so Orca's auth scrubbing stays assertable. + expect(env.CLAUDE_CONFIG_DIR).toBe('/pinned/claude-config') + expect(env.ORCA_AGENT_SESSION_SPAWN_TOKEN).toBe('spawn-token-1') + // Ambient process.env is NOT merged in when env is supplied. + expect(env.ANTHROPIC_API_KEY).toBeUndefined() + // The SDK's two documented mutations, pinned so a change is noticed. + expect(env.CLAUDE_CODE_ENTRYPOINT).toBe('sdk-ts') + expect('NODE_OPTIONS' in env).toBe(false) + }) + + it('inherits process.env into the child when env is omitted — the ambient-auth sharp edge', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + vi.stubEnv('ORCA_SDK_CONTRACT_SCENARIO_PATH', scenario.scenarioPath) + vi.stubEnv('ORCA_SDK_CONTRACT_REPORT_PATH', scenario.reportPath) + vi.stubEnv('ORCA_SDK_CONTRACT_AMBIENT_CANARY', 'inherited-from-process-env') + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + // Omitting env reproduces the ambient-auth-leak failure mode: the child + // sees everything in process.env. Orca must therefore always pass an + // explicit, fully-constructed env. + expect(spawns[0]!.env.ORCA_SDK_CONTRACT_AMBIENT_CANARY).toBe('inherited-from-process-env') + }) + + it('emits --replay-user-messages only through extraArgs, never on its own', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const bareSpawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + spawnClaudeCodeProcess: recordingSpawner(bareSpawns) + }) + expect(bareSpawns[0]!.args).not.toContain('--replay-user-messages') + + const replayScenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const replaySpawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: replayScenario.cwd, + env: scenarioEnv(replayScenario), + extraArgs: { 'replay-user-messages': null }, + spawnClaudeCodeProcess: recordingSpawner(replaySpawns) + }) + const replayArgs = replaySpawns[0]!.args + expect(replayArgs.filter((arg) => arg === '--replay-user-messages')).toHaveLength(1) + }) + + it('produces a matching CLI flag for every pre-SDK argv entry', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + // Driven by the real resolver, so the argv walk covers the durable-launchArgs + // translation and its merge order, not a hand-written options literal. + const launch = await resolvedLaunch(['--model', 'claude-sonnet-4-5', '--effort', 'high']) + await drainQuery({ + ...launch.options, + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + canUseTool: (async () => ({ behavior: 'deny', message: 'unused' })) as CanUseTool, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(spawns).toHaveLength(1) + const argv = normalizeArgv(spawns[0]!.args) + // Typed-first translation must not also spell the flag through extraArgs. + for (const flag of ['--model', '--effort']) { + expect( + argv.filter((arg) => arg === flag), + `${flag} occurrences` + ).toHaveLength(1) + } + expect(argv[argv.indexOf('--model') + 1]).toBe('claude-sonnet-4-5') + expect(argv[argv.indexOf('--effort') + 1]).toBe('high') + // Headless print mode is the SDK's only mode; `query()` never passes `-p`, + // and if the SDK ever started passing it this pin would notice. + const impliedByHeadlessQuery = new Set(['-p']) + for (const entry of flagTable(PRE_SDK_ARGV)) { + if (impliedByHeadlessQuery.has(entry.flag)) { + expect(argv, `${entry.flag} is implied, never spelled`).not.toContain(entry.flag) + continue + } + const at = argv.indexOf(entry.flag) + expect(at, `SDK argv is missing ${entry.flag}`).toBeGreaterThanOrEqual(0) + if (entry.value !== null) { + expect(argv[at + 1], `value of ${entry.flag}`).toBe(entry.value) + } + } + // The launch resolver always carries one of --session-id / --resume. + const sessionAt = argv.indexOf('--session-id') + expect(sessionAt).toBeGreaterThanOrEqual(0) + expect(argv[sessionAt + 1]).toBe(launch.providerSessionId) + }) + + it('still exposes the runtime get_settings reader the auth diagnostic depends on', async () => { + // 0.3.251 ships getSettings() but redacts it from the Query declaration. This pin + // is the drift alarm: if a bump drops or reshapes it, the diagnostic degrades and + // this test says so instead of the degradation shipping silently. + const settings = { env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } } + const scenario = scriptScenario([{ delayMs: 3_000 }], { get_settings: settings }) + const session = query({ + prompt: singleUserTurn(), + options: { + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario) + } + }) + try { + const read = claudeQuerySettingsReader(session) + expect(read, 'the SDK no longer exposes get_settings at runtime').not.toBeNull() + await expect(read?.()).resolves.toEqual(settings) + } finally { + await session.return(undefined) + } + }) + + it('maps resume identity to --resume and --resume-session-at', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + resume: SESSION_ID, + resumeSessionAt: LEAF_UUID, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + const argv = normalizeArgv(spawns[0]!.args) + const resumeAt = argv.indexOf('--resume') + expect(resumeAt).toBeGreaterThanOrEqual(0) + expect(argv[resumeAt + 1]).toBe(SESSION_ID) + const leafAt = argv.indexOf('--resume-session-at') + expect(leafAt).toBeGreaterThanOrEqual(0) + expect(argv[leafAt + 1]).toBe(LEAF_UUID) + }) + + it('gives canUseTool the wire request_id and fires its abort signal on control_cancel_request', async () => { + const scenario = scriptScenario([ + { awaitUserMessage: true }, + { + emit: { + type: 'control_request', + request_id: 'perm-421', + request: { + subtype: 'can_use_tool', + tool_name: 'Bash', + input: { command: 'echo hi' }, + tool_use_id: 'tool-use-9' + } + } + }, + { delayMs: 120 }, + { emit: { type: 'control_cancel_request', request_id: 'perm-421' } }, + { awaitControlResponse: 'perm-421' }, + { emit: RESULT_FRAME } + ]) + const seen: { toolName: string; requestId: string; toolUseID: string }[] = [] + let abortFired = false + const canUseTool: CanUseTool = (toolName, _input, { signal, requestId, toolUseID }) => { + seen.push({ toolName, requestId, toolUseID }) + return new Promise((resolve) => { + signal.addEventListener('abort', () => { + abortFired = true + resolve({ behavior: 'deny', message: 'cancelled by test' }) + }) + }) + } + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + canUseTool, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(seen).toEqual([{ toolName: 'Bash', requestId: 'perm-421', toolUseID: 'tool-use-9' }]) + expect(abortFired).toBe(true) + // The callback's settlement is written back onto the wire against the same id. + const settled = scenario + .readReport() + .controlResponses.find((frame) => frame.response.request_id === 'perm-421') + expect(settled?.response.response?.behavior).toBe('deny') + // Exactly one process spawn per query, control traffic included. + expect(spawns).toHaveLength(1) + }) + + it('runs the executable given via pathToClaudeCodeExecutable under the default spawner', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const messages = await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario) + }) + + expect(messages.some((m) => m.type === 'result')).toBe(true) + const report = scenario.readReport() + // The SDK executed exactly the script we pointed it at — no bundled binary. + expect(report.argv[0]).toBe(FAKE_CLI) + expect(report.execPath).toContain('node') + // And the streaming handshake went to it: the SDK sent its initialize + // control request to our script. + expect(report.controlRequests.some((frame) => frame.request.subtype === 'initialize')).toBe( + true + ) + }) + + it('pins the SDK version the contract was verified against', () => { + const sdkEntry = createRequire(__filename).resolve('@anthropic-ai/claude-agent-sdk') + const manifest = JSON.parse(readFileSync(join(dirname(sdkEntry), 'package.json'), 'utf8')) as { + version: string + } + expect(manifest.version).toBe(PINNED_SDK_VERSION) + }) + + it('keeps the eight bundled CLI platform binaries out of the install', () => { + const sdkEntry = createRequire(__filename).resolve('@anthropic-ai/claude-agent-sdk') + // The SDK's own scoped directory is where pnpm would link its optional + // platform packages; ignoredOptionalDependencies must keep them all absent. + const scopeDir = dirname(dirname(sdkEntry)) + for (const basename of SDK_PLATFORM_PACKAGE_BASENAMES) { + expect( + existsSync(join(scopeDir, basename, 'package.json')), + `${basename} must not be installed` + ).toBe(false) + } + }) +}) diff --git a/src/main/claude/claude-agent-sdk-control-requests.ts b/src/main/claude/claude-agent-sdk-control-requests.ts new file mode 100644 index 00000000000..6bd396413fa --- /dev/null +++ b/src/main/claude/claude-agent-sdk-control-requests.ts @@ -0,0 +1,154 @@ +import type { + PermissionMode, + Query, + SDKControlInterruptResponse +} from '@anthropic-ai/claude-agent-sdk' + +export class ClaudeControlRequestError extends Error { + constructor( + readonly subtype: string, + message: string + ) { + super(message) + this.name = 'ClaudeControlRequestError' + } +} + +export const CLAUDE_DEFAULT_REQUEST_TIMEOUT_MS = 30_000 + +/** The SDK closes a query out from under an in-flight control request with this exact message. */ +const QUERY_CLOSED_MESSAGE = 'Query closed before response received' + +/** 0.3.251 ships getSettings() but redacts it from the Query declaration; the typeof guard below is its degradation path. */ +type ClaudeQuerySettingsReader = { getSettings?: () => Promise } + +export function claudeQuerySettingsReader(query: Query): (() => Promise) | null { + const reader = (query as unknown as ClaudeQuerySettingsReader).getSettings + return typeof reader === 'function' ? reader.bind(query) : null +} + +/** + * cancel_async_message is a runtime Query method the shipped 0.3.251 declaration omits; + * it withdraws a single still-queued async user message by uuid so an interrupted turn + * cannot spawn a later unexpected turn. The typeof guard is its degradation path. + */ +type ClaudeQueryAsyncCanceller = { cancelAsyncMessage?: (uuid: string) => Promise } + +export function claudeQueryAsyncCanceller( + query: Query +): ((uuid: string) => Promise) | null { + const cancel = (query as unknown as ClaudeQueryAsyncCanceller).cancelAsyncMessage + return typeof cancel === 'function' ? cancel.bind(query) : null +} + +export type ClaudeControlOptions = { timeoutMs?: number } + +/** + * Run one native Query control method under Orca's deadline and error classification. + * + * The SDK owns correlation but applies no deadline, so the timeout stays here — and its + * message is load-bearing: the init proof matches on `claude initialize request timed out`. + * A closed query is a transport failure, not the CLI rejecting the request, so only the + * latter is re-thrown as a `ClaudeControlRequestError` a caller may surface as a rejection. + */ +export function runClaudeControl( + subtype: string, + run: () => Promise, + timeoutMs: number = CLAUDE_DEFAULT_REQUEST_TIMEOUT_MS +): Promise { + let timer: ReturnType | null = null + const deadline = new Promise((_resolve, reject) => { + timer = setTimeout(() => reject(new Error(`claude ${subtype} request timed out`)), timeoutMs) + timer.unref?.() + }) + return Promise.race([ + Promise.resolve() + .then(run) + .catch((error: unknown) => { + const message = error instanceof Error ? error.message : String(error) + if (error instanceof ClaudeControlRequestError || message === QUERY_CLOSED_MESSAGE) { + throw error + } + throw new ClaudeControlRequestError(subtype, message) + }), + deadline + ]).finally(() => { + if (timer) { + clearTimeout(timer) + } + }) +} + +/** The native control surface Orca drives, one method per Query control request. */ +export type ClaudeControlSurface = { + interrupt: ( + options?: ClaudeControlOptions & { cancelQueued?: boolean } + ) => Promise + cancelAsyncMessage: (uuid: string, options?: ClaudeControlOptions) => Promise + setModel: (model: string | undefined, options?: ClaudeControlOptions) => Promise + setPermissionMode: (mode: PermissionMode, options?: ClaudeControlOptions) => Promise + applyFlagSettings: ( + settings: Parameters[0], + options?: ClaudeControlOptions + ) => Promise + supportedModels: (options?: ClaudeControlOptions) => Promise + initializationResult: (options?: ClaudeControlOptions) => Promise + getSettings: (options?: ClaudeControlOptions) => Promise +} + +type InterruptingQuery = { + interrupt: (options?: { + cancelQueued?: boolean + }) => Promise +} + +export function createClaudeControlSurface(query: Query): ClaudeControlSurface { + return { + interrupt: (options) => + runClaudeControl( + 'interrupt', + () => + (query as unknown as InterruptingQuery).interrupt( + options?.cancelQueued ? { cancelQueued: true } : undefined + ), + options?.timeoutMs + ), + cancelAsyncMessage: (uuid, options) => { + const cancel = claudeQueryAsyncCanceller(query) + return cancel + ? runClaudeControl('cancel_async_message', () => cancel(uuid), options?.timeoutMs).then( + () => {} + ) + : Promise.resolve() + }, + setModel: (model, options) => + runClaudeControl('set_model', () => query.setModel(model), options?.timeoutMs).then(() => {}), + setPermissionMode: (mode, options) => + runClaudeControl( + 'set_permission_mode', + () => query.setPermissionMode(mode), + options?.timeoutMs + ).then(() => {}), + applyFlagSettings: (settings, options) => + runClaudeControl( + 'apply_flag_settings', + () => query.applyFlagSettings(settings), + options?.timeoutMs + ).then(() => {}), + supportedModels: (options) => + runClaudeControl('list_models', () => query.supportedModels(), options?.timeoutMs), + initializationResult: (options) => + runClaudeControl('initialize', () => query.initializationResult(), options?.timeoutMs), + getSettings: (options) => { + const read = claudeQuerySettingsReader(query) + return read + ? runClaudeControl('get_settings', read, options?.timeoutMs) + : Promise.reject( + new ClaudeControlRequestError( + 'get_settings', + 'this SDK exposes no get_settings request' + ) + ) + } + } +} diff --git a/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts b/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts new file mode 100644 index 00000000000..04d58067353 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts @@ -0,0 +1,104 @@ +import { describe, expect, it, vi } from 'vitest' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { collectDescendantRows } from '../pty-descendant-termination' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' +import { mergeClaudeCapturedTrees } from './claude-child-tree-snapshot' + +function posixSnapshot(capturedAtMs: number): DescendantSnapshot { + return { + root: { pid: 100, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 100, + descendants: [{ pid: 200, ppid: 100, pgid: 100, startedAt: 'Mon Jan 1 00:00:01 2026' }], + capturedAtMs + } +} + +function windowsSnapshot(): WindowsDescendantSnapshot { + return { + root: { pid: 100, creationTimeMs: 5 }, + descendants: [{ pid: 200, creationTimeMs: 7 }], + unidentifiedCount: 0, + capturedAtMs: 1 + } +} + +describe('Claude child root identity', () => { + it('keeps a retained row boundary when a refresh observes no new descendants', () => { + const previous = posixSnapshot(1_700_000_000_900) + const next = posixSnapshot(1_700_000_002_100) + + expect( + mergeClaudeCapturedTrees( + { platform: 'posix', tree: previous }, + { platform: 'posix', tree: next } + ) + ).toEqual({ + platform: 'posix', + tree: { ...next, capturedAtMsByPid: { '200': previous.capturedAtMs } } + }) + }) + + it('keeps the descendant verdict when a POSIX root probe is unavailable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => posixSnapshot(1)), + terminateDescendants, + verifyRootIdentity: vi.fn(async () => false) + }) + + // POSIX runs no bare-pid root operation, so a declined probe withholds + // nothing: the handle kill still lands and the verification still speaks. + await expect(tree.reap()).resolves.toBe('exited') + expect(terminateDescendants).toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('rejects mixed old and recycled root rows instead of making the tree killable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => + collectDescendantRows( + 100, + [ + { pid: 100, ppid: 1, pgid: 100, startedAt: 'Mon Jan 1 00:00:00 2026' }, + { pid: 100, ppid: 1, pgid: 101, startedAt: 'Mon Jan 1 00:00:01 2026' }, + { pid: 200, ppid: 100, pgid: 200, startedAt: 'Mon Jan 1 00:00:00 2026' } + ], + 1 + ) + ), + terminateDescendants, + verifyRootIdentity: vi.fn(async () => true) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // No admissible snapshot means no row may be signalled from its number, but + // the root still leaves through the handle Node owns. + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('fails closed when Windows root identity revalidation is unavailable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshot()), + terminateWindowsTree, + terminateWindowsDescendants, + verifyRootIdentity: vi.fn(async () => false) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // taskkill /T /F addresses a bare pid and stays gated; the handle does not. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-exit-proof.test.ts b/src/main/claude/claude-agent-sdk-exit-proof.test.ts new file mode 100644 index 00000000000..3f15f8e8fca --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof.test.ts @@ -0,0 +1,934 @@ +import { execFileSync } from 'node:child_process' +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import { spawnProcess, type SpawnedProcess } from '../../shared/child-process/run-process' +import type { DescendantTreeVerdict } from '../pty-descendant-exit-verification' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { + createClaudeChildTreeReaper as createClaudeChildTreeReaperImpl, + proveClaudeChildExit, + type ClaudeChildTreeReaper +} from './claude-agent-sdk-exit-proof' + +// The descendant models an MCP server: it either cooperates or, when it traps +// SIGTERM, only a forced, verified sweep can reach it. The root either traps +// SIGTERM too, or leaves promptly on stdin end the way a healthy CLI does — +// which is the path that used to skip descendant proof entirely. +function childWithDescendantScript(input: { + rootTrapsSigterm: boolean + descendantTrapsSigterm: boolean +}): string { + const descendantScript = `${input.descendantTrapsSigterm ? 'process.on("SIGTERM", () => {}); ' : ''}setInterval(() => {}, 1000000)` + const rootBehaviour = input.rootTrapsSigterm + ? `process.on('SIGTERM', () => {}) +process.on('SIGINT', () => {}) +setInterval(() => {}, 1000000)` + : `process.stdin.on('end', () => process.exit(0)) +process.stdin.resume()` + return ` +const descendant = require('node:child_process').spawn( + process.execPath, + ['-e', ${JSON.stringify(descendantScript)}], + { stdio: 'ignore' } +) +descendant.unref() +process.stdout.write(JSON.stringify({ descendantPid: descendant.pid }) + '\\n') +${rootBehaviour} +` +} + +const COOPERATIVE_CHILD = ` +process.stdin.on('end', () => process.exit(0)) +process.stdin.resume() +process.stdout.write('ready\\n') +` + +/** + * Sampled synchronously so it reads the exact moment the close boundary is + * crossed. A zombie has exited (its parent just has not reaped it yet), so a + * kill(pid, 0) probe would misreport it as running. + */ +function descendantState(pid: number): 'running' | 'exited' { + let state: string + try { + state = execFileSync('ps', ['-o', 'state=', '-p', String(pid)], { + encoding: 'utf8', + env: { ...process.env, LANG: 'C', LC_ALL: 'C' } + }).trim() + } catch (error) { + // ps exits 1 when no process matches; anything else is a failed probe, not an answer. + if ((error as { status?: number }).status !== 1) { + throw error + } + return 'exited' + } + return state.startsWith('Z') ? 'exited' : 'running' +} + +/** + * ps lstart is second-resolution, so the identity-safe sweep only SIGKILLs a row + * born strictly before the second the snapshot was captured in. The snapshot is + * armed the moment close begins, so a descendant born in that same second can + * only be asked, never forced — the same bound an MCP server spawned within a + * second of the user closing the chat would hit. + */ +function ageDescendantPastTheCaptureSecond(): Promise { + return new Promise((resolve) => setTimeout(resolve, 1_000 - (Date.now() % 1_000) + 20)) +} + +/** + * The close ladder as production drives it: `closeProcessRegistry` retries an + * unproven close, and each retry re-verifies the retained snapshot. A loaded + * host can spend one attempt's whole window inside `ps`, and reporting false + * there is the honest verdict — the requirement is that TRUE never outruns the + * observation, which the caller asserts at whichever boundary returns it. + */ +async function proveExitWithRetries( + input: Parameters[0], + attempts = 3 +): Promise { + for (let attempt = 1; attempt < attempts; attempt += 1) { + if (await proveClaudeChildExit(input)) { + return true + } + } + return proveClaudeChildExit(input) +} + +function spawnScript(script: string): ReturnType { + return spawnProcess({ + program: process.execPath, + args: ['-e', script], + stdio: ['pipe', 'pipe', 'pipe'] + }) +} + +function firstStdoutLine(child: ReturnType): Promise { + return new Promise((resolve) => { + child.stdout.setEncoding('utf8').once('data', (chunk: string) => resolve(chunk.trim())) + }) +} + +function observeExit(child: EventEmitter): { exitPromise: Promise; exited: () => boolean } { + let exited = false + const exitPromise = new Promise((resolve) => { + child.once('exit', () => { + exited = true + resolve() + }) + }) + return { exitPromise, exited: () => exited } +} + +/** `null` models a spawn that failed before a pid existed. */ +function mockChild( + pid: number | null = 424242 +): EventEmitter & + Pick & { kill: ReturnType } { + const child = new EventEmitter() + return Object.assign(child, { + pid: pid ?? undefined, + stdin: new PassThrough(), + kill: vi.fn(() => true) + }) as never +} + +/** A tree whose verdict is scripted per reap, recording when it was armed. */ +function mockTree(verdicts: DescendantTreeVerdict[]): ClaudeChildTreeReaper & { + capture: ReturnType + reap: ReturnType +} { + let treeVerdict: DescendantTreeVerdict = 'unverifiable' + return { + capture: vi.fn(async () => {}), + reap: vi.fn(async () => { + treeVerdict = verdicts.shift() ?? treeVerdict + return treeVerdict + }), + get treeVerdict() { + return treeVerdict + } + } +} + +function windowsSnapshotOf(descendantPid: number): WindowsDescendantSnapshot { + return { + root: { pid: 424242, creationTimeMs: 1_700_000_000_001 }, + descendants: [{ pid: descendantPid, creationTimeMs: 1_700_000_000_000 }], + unidentifiedCount: 0, + capturedAtMs: 1 + } +} + +function snapshotOf(descendantPid: number): DescendantSnapshot { + return { + root: { pid: 424242, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 1, + descendants: [ + { pid: descendantPid, ppid: 424242, pgid: 1, startedAt: 'Mon Jan 1 00:00:00 2026' } + ], + capturedAtMs: 1 + } +} + +// Unit tests use synthetic process ids; production always supplies the fresh +// identity probe, so the harness explicitly models a matching probe. +function createClaudeChildTreeReaper( + child: Parameters[0], + deps: Parameters[1] = {} +): ReturnType { + return createClaudeChildTreeReaperImpl(child, { + verifyRootIdentity: async () => true, + ...deps + }) +} + +describe('claude child exit proof', () => { + it.runIf(process.platform !== 'win32')( + 'reports a proven exit only once a SIGTERM-resistant descendant is gone at the close boundary', + async () => { + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: true, descendantTrapsSigterm: true }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + expect(descendantState(descendantPid)).toBe('running') + await ageDescendantPastTheCaptureSecond() + + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + // Evaluated AT the boundary, not by polling until a deferred sweep timer + // wins: true releases the lease, so a descendant still running here is + // exactly the orphan the proof exists to prevent. False would be the + // honest verdict for a tree that outlived the bounded ladder. + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + // Failure-safe only: the assertion above owns the requirement, this just + // stops a failing run from leaking a process. + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it.runIf(process.platform !== 'win32')( + 'proves a promptly exiting root only once its stubborn descendant is gone too', + async () => { + // The ordinary healthy close: the root leaves on stdin end within the graceful + // window. Its descendant must still be proven gone, not assumed gone with it. + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: false, descendantTrapsSigterm: true }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + expect(descendantState(descendantPid)).toBe('running') + await ageDescendantPastTheCaptureSecond() + + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it.runIf(process.platform !== 'win32')( + 'still proves a stubborn child whose descendant honours SIGTERM', + async () => { + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: true, descendantTrapsSigterm: false }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it('arms the snapshot before stdin closes and verifies it after a clean exit', async () => { + const child = spawnScript(COOPERATIVE_CHILD) + expect(await firstStdoutLine(child)).toBe('ready') + const exit = observeExit(child) + const tree = mockTree(['exited']) + let exitedWhenArmed: boolean | null = null + tree.capture.mockImplementation(async () => { + exitedWhenArmed = exit.exited() + }) + + await expect(proveClaudeChildExit({ child, ...exit, tree })).resolves.toBe(true) + // The snapshot is the only proof that survives the root: taken while it lived, + // verified once it left. A reap before the exit would have been the forced ladder. + expect(exitedWhenArmed).toBe(false) + expect(tree.reap).toHaveBeenCalledTimes(1) + expect(exit.exited()).toBe(true) + }, 20_000) + + it('proves a clean close of a childless root with one snapshot and no signal', async () => { + const child = spawnScript(COOPERATIVE_CHILD) + expect(await firstStdoutLine(child)).toBe('ready') + + await expect(proveClaudeChildExit({ child, ...observeExit(child) })).resolves.toBe(true) + }, 20_000) + + it('reports an unprovable exit as false rather than assuming the child died', async () => { + const child = mockChild() + const tree = mockTree(['exited']) + + await expect( + proveClaudeChildExit({ + child, + exitPromise: new Promise(() => {}), + exited: () => false, + tree + }) + ).resolves.toBe(false) + expect(tree.reap).toHaveBeenCalledTimes(1) + }, 20_000) + + it('reports false when the root exit was observed but a descendant was seen alive', async () => { + const child = mockChild() + const exit = observeExit(child) + const tree = mockTree(['live']) + tree.reap.mockImplementation(async () => { + child.emit('exit', null, 'SIGKILL') + return 'live' + }) + + await expect(proveClaudeChildExit({ child, ...exit, tree })).resolves.toBe(false) + expect(exit.exited()).toBe(true) + // One verification per attempt: the retried close re-verifies, this one does not. + expect(tree.reap).toHaveBeenCalledTimes(1) + }, 20_000) + + it('re-verifies an unproven tree on a retried close instead of trusting the dead root', async () => { + const child = mockChild() + const tree = mockTree(['exited']) + + await expect( + proveClaudeChildExit({ child, exitPromise: Promise.resolve(), exited: () => true, tree }) + ).resolves.toBe(true) + expect(tree.reap).toHaveBeenCalledTimes(1) + }) + + it('stays unproven for a root that left before any snapshot could be armed', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const terminateDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'darwin', + exited: () => true, + captureDescendants, + terminateDescendants + }) + + await expect( + proveClaudeChildExit({ child, exitPromise: Promise.resolve(), exited: () => true, tree }) + ).resolves.toBe(false) + // A dead root's descendants have reparented: walking its pid now could only + // sweep a stranger, so no walk is attempted and nothing is proven. + expect(captureDescendants).not.toHaveBeenCalled() + expect(terminateDescendants).not.toHaveBeenCalled() + expect(tree.treeVerdict).toBe('unverifiable') + }) +}) + +describe('claude child tree reaper', () => { + it('kills the root while verification runs and never stops it first', async () => { + const child = mockChild() + const release = Promise.withResolvers() + const terminateDescendants = vi.fn(() => release.promise) + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const tree = createClaudeChildTreeReaper(child, { + platform: 'darwin', + captureDescendants, + terminateDescendants + }) + + const first = tree.reap() + const second = tree.reap() + await vi.waitFor(() => expect(terminateDescendants).toHaveBeenCalledTimes(1)) + // A stopped root cannot verify: its killed children stay zombie rows in ps. + expect(child.kill.mock.calls).toEqual([['SIGKILL']]) + expect(tree.treeVerdict).toBe('unverifiable') + + release.resolve('exited') + await expect(Promise.all([first, second])).resolves.toEqual(['exited', 'exited']) + expect(captureDescendants).toHaveBeenCalledTimes(1) + expect(tree.treeVerdict).toBe('exited') + }) + + it('re-verifies the retained snapshot on a later reap rather than re-walking a dead root', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('live') + .mockResolvedValueOnce('exited') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(1) + expect(terminateDescendants).toHaveBeenNthCalledWith(2, snapshotOf(4243)) + expect(tree.treeVerdict).toBe('exited') + }) + + it('keeps an observed exit when a later re-read cannot see the table', async () => { + const child = mockChild() + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('exited') + .mockResolvedValueOnce('unverifiable') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => snapshotOf(4243)), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('exited') + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(tree.treeVerdict).toBe('exited') + }) + + it('keeps an observed live descendant when a later re-read cannot see the table', async () => { + const child = mockChild() + // Reap #1 completed and saw a descendant alive at its deadline; the root then + // left on its own and the re-verification on a loaded host could not read the + // table. "Could not look" must not erase "was seen alive": the lease release + // gate is exactly the pair this distinguishes. + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('live') + .mockResolvedValueOnce('unverifiable') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => snapshotOf(4243)), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('live') + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(tree.treeVerdict).toBe('live') + }) + + it('treats an unreadable process table as unproven and re-walks the live root', async () => { + const child = mockChild() + // A loaded host can miss the table's deadline; while the root still lives + // that is a retryable read, not evidence that it has no descendants. + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(snapshotOf(4243)) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateDescendants).not.toHaveBeenCalled() + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(2) + }) + + it('does not latch a missing root while it is still live', async () => { + const child = mockChild() + const captureDescendants = vi + .fn() + .mockResolvedValueOnce({ rootPgid: null, descendants: [], capturedAtMs: 1 }) + .mockResolvedValueOnce(snapshotOf(4243)) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(snapshotOf(4243)) + }) + + it('refreshes the live snapshot at close time so late descendants are included', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const second = { + ...first, + descendants: [...first.descendants, { ...first.descendants[0], pid: 4244 }] + } + const captureDescendants = vi.fn().mockResolvedValueOnce(first).mockResolvedValueOnce(second) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(second) + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('keeps the original capture boundary for retained POSIX rows', async () => { + const child = mockChild() + const first = { + ...snapshotOf(4243), + capturedAtMs: 1_700_000_000_900 + } + const refreshed = { + ...first, + capturedAtMs: 1_700_000_002_100, + descendants: [ + ...first.descendants, + { + pid: 4244, + ppid: 424242, + pgid: 1, + startedAt: 'Tue Jan 2 00:00:00 2026' + } + ] + } + const captureDescendants = vi.fn().mockResolvedValueOnce(first).mockResolvedValueOnce(refreshed) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(terminateDescendants).toHaveBeenCalledWith({ + ...refreshed, + // The retained 4243 row was first observed in the earlier displayed + // second. Its per-row boundary must not advance with the refresh. + capturedAtMsByPid: { + '4243': first.capturedAtMs, + '4244': refreshed.capturedAtMs + } + }) + }) + + it('fails closed when a POSIX refresh reuses a PID with a new identity', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const replacement = { + ...first, + descendants: [ + { + ...first.descendants[0], + pgid: 9, + startedAt: 'Tue Jan 2 00:00:00 2026' + } + ] + } + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + + await expect(tree.reap()).resolves.toBe('unverifiable') + // The descendant evidence is discarded; the root's identity never was in doubt. + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('fails closed when a Windows refresh reuses a PID with a new creation time', async () => { + const child = mockChild() + const first = windowsSnapshotOf(4243) + const replacement = { + ...first, + descendants: [{ pid: 4243, creationTimeMs: first.descendants[0].creationTimeMs + 1 }] + } + const captureWindowsDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants, + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + await tree.refresh?.() + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('queues a fresh boundary behind an output-triggered capture already in flight', async () => { + const child = mockChild() + const firstDone = Promise.withResolvers() + const first = snapshotOf(4243) + const second = { + ...first, + descendants: [...first.descendants, { ...first.descendants[0], pid: 4244 }] + } + const captureDescendants = vi + .fn() + .mockImplementationOnce(async () => { + await firstDone.promise + return first + }) + .mockResolvedValueOnce(second) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + const outputCapture = tree.refresh!() + await vi.waitFor(() => expect(captureDescendants).toHaveBeenCalledTimes(1)) + const closeCapture = tree.refresh!() + await Promise.resolve() + expect(captureDescendants).toHaveBeenCalledTimes(1) + + firstDone.resolve() + await closeCapture + await tree.reap() + + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(second) + await outputCapture + }) + + it('retains a replacement descendant when the prior identity exited', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const replacement = snapshotOf(4244) + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateDescendants = vi.fn(async (snapshot: DescendantSnapshot) => + snapshot.descendants.some((row) => row.pid === 4244) ? ('live' as const) : ('exited' as const) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await expect(tree.reap()).resolves.toBe('live') + + expect(terminateDescendants).toHaveBeenCalledWith({ + ...replacement, + descendants: [...first.descendants, ...replacement.descendants] + }) + }) + + it('retains a Windows replacement descendant while preserving unidentified rows', async () => { + const child = mockChild() + const first = windowsSnapshotOf(4243) + const replacement = { + ...windowsSnapshotOf(4244), + unidentifiedCount: 0 + } + const captureWindowsDescendants = vi + .fn() + .mockResolvedValueOnce({ ...first, unidentifiedCount: 1 }) + .mockResolvedValueOnce(replacement) + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async (snapshot: WindowsDescendantSnapshot) => + snapshot.descendants.some((row) => row.pid === 4244) ? ('live' as const) : ('exited' as const) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants, + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + await tree.refresh?.() + await expect(tree.reap()).resolves.toBe('live') + + expect(terminateWindowsDescendants).toHaveBeenCalledWith({ + ...replacement, + descendants: [...first.descendants, ...replacement.descendants], + unidentifiedCount: 1 + }) + }) + + it('retains the prior identity-safe snapshot when a refresh is partial', async () => { + const child = mockChild() + const first = { + ...snapshotOf(4243), + descendants: [ + ...snapshotOf(4243).descendants, + { ...snapshotOf(4243).descendants[0], pid: 4244 } + ] + } + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce({ + ...first, + descendants: first.descendants.slice(0, 1) + }) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(terminateDescendants).toHaveBeenCalledWith(first) + }) + + it('stops re-walking once the root is gone, however the table behaved', async () => { + const child = mockChild() + let exited = false + const captureDescendants = vi.fn(async () => null) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => exited, + captureDescendants, + terminateDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // An unreadable table costs the snapshot, never the kill on the live root. + expect(child.kill).toHaveBeenCalledTimes(1) + exited = true + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(captureDescendants).toHaveBeenCalledTimes(1) + // The second attempt observes a dead root: Node has dropped the handle, so + // there is nothing left to signal and no recycled pid to reach. + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('discards a walk that found no root instead of proving an empty tree', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => ({ + rootPgid: null, + descendants: [], + capturedAtMs: 1 + })) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants: vi.fn() + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('unverifiable') + // A vacuous walk remains retryable while the root is live; no empty-tree + // verdict is latched from a missing root row. + expect(captureDescendants).toHaveBeenCalledTimes(2) + }) + + it('discards a walk that raced the root exit instead of proving an empty tree', async () => { + const child = mockChild() + let exited = false + const captureDescendants = vi.fn(async () => { + exited = true + return { rootPgid: 1, descendants: [], capturedAtMs: 1 } + }) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => exited, + captureDescendants, + terminateDescendants: vi.fn() + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('unverifiable') + }) + + it('proves a childless snapshot without signalling anything', async () => { + const child = mockChild() + const terminateDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => ({ + root: { pid: 424242, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 1, + descendants: [], + capturedAtMs: 1 + })), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('waits for the Windows tree kill before releasing the root', async () => { + const child = mockChild() + const release = Promise.withResolvers() + const terminateWindowsTree = vi.fn(() => release.promise) + const captureDescendants = vi.fn() + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureDescendants, + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants + }) + + const reap = tree.reap() + await vi.waitFor(() => + expect(terminateWindowsTree).toHaveBeenCalledWith({ + pid: 424242, + creationTimeMs: 1_700_000_000_001 + }) + ) + expect(child.kill).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + release.resolve() + await expect(reap).resolves.toBe('exited') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + expect(terminateWindowsDescendants).toHaveBeenCalledWith(windowsSnapshotOf(4243)) + expect(captureDescendants).not.toHaveBeenCalled() + }) + + it('stays unproven on Windows when taskkill fails and a descendant is still observed', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree: vi.fn(async () => { + throw new Error('taskkill: access denied') + }), + terminateWindowsDescendants: vi.fn(async () => 'live' as const) + }) + + // taskkill's own outcome is not the proof; the table read after it is. + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('stays unproven on Windows when taskkill resolves but a descendant survives it', async () => { + const child = mockChild() + const terminateWindowsTree = vi.fn(async () => {}) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants: vi.fn(async () => 'live' as const) + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(terminateWindowsTree).toHaveBeenCalledTimes(1) + expect(tree.treeVerdict).toBe('live') + }) + + it('never taskkills a Windows root that already exited, but still verifies its snapshot', async () => { + const child = mockChild() + let exited = false + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => exited, + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + exited = true + await expect(tree.reap()).resolves.toBe('exited') + // A dead root's pid may already belong to a stranger: taskkill /T /F on it + // would take down an unrelated tree. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).toHaveBeenCalledWith(windowsSnapshotOf(4243)) + }) + + it('treats an unreadable Windows table as unproven', async () => { + const child = mockChild() + const terminateWindowsDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => null), + terminateWindowsTree: vi.fn(async () => {}), + terminateWindowsDescendants + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + // A host that cannot supply creation times blocks taskkill, not the root kill. + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('has nothing to reap for a child that never spawned', async () => { + const child = mockChild(null) + const captureDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { platform: 'linux', captureDescendants }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/claude/claude-agent-sdk-exit-proof.ts b/src/main/claude/claude-agent-sdk-exit-proof.ts new file mode 100644 index 00000000000..17533f87a70 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof.ts @@ -0,0 +1,366 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import { + terminateDescendantSnapshotWithVerdict, + type DescendantTreeVerdict +} from '../pty-descendant-exit-verification' +import { + captureDescendantSnapshot, + type DescendantSnapshot, + type PosixProcessIdentity +} from '../pty-descendant-termination' +import { + captureWindowsDescendantSnapshot, + terminateIdentifiedWindowsProcessTree, + verifyWindowsDescendantSnapshotExit, + verifyWindowsProcessIdentity, + type WindowsDescendantSnapshot, + type WindowsProcessIdentity +} from '../windows-descendant-exit-verification' +import { mergeClaudeCapturedTrees, type ClaudeCapturedTree } from './claude-child-tree-snapshot' +import { terminateClaudeRoot, terminateClaudeWindowsRoot } from './claude-child-root-termination' +import { + proveClaudeChildExitWithReaper, + type ClaudeChildExitProofInput +} from './claude-child-exit-proof-ladder' + +/** + * A later reap may only raise the latched verdict. An observed exit is final, and + * a descendant seen alive at a deadline is never forgotten by a later look that + * could not read the table: the lease gate discriminates on exactly that pair. + */ +const TREE_VERDICT_TRUST: Record = { + unverifiable: 0, + live: 1, + exited: 2 +} + +type ReapableChild = Pick + +/** + * A walk is only admissible while the root it walked was alive. A POSIX walk + * that found no root says so with a null pgid; either platform's walk can also + * have raced the root's death. Both can only have missed descendants that + * already reparented away, so neither is evidence about the tree. + */ +function admissibleTree( + captured: DescendantSnapshot | WindowsDescendantSnapshot | null, + platform: NodeJS.Platform, + exited: boolean +): ClaudeCapturedTree | null { + if (!captured || exited) { + return null + } + if (platform === 'win32') { + return { platform: 'win32', tree: captured as WindowsDescendantSnapshot } + } + const tree = captured as DescendantSnapshot + return tree.rootPgid === null ? null : { platform: 'posix', tree } +} + +export type ClaudeChildTreeReaperDeps = { + platform?: NodeJS.Platform + /** Whether the root's exit has been observed; only a live root can be walked. */ + exited?: () => boolean + captureDescendants?: (rootPid: number) => Promise + terminateDescendants?: (snapshot: DescendantSnapshot) => Promise + terminateWindowsTree?: (root: WindowsProcessIdentity) => Promise + captureWindowsDescendants?: (rootPid: number) => Promise + terminateWindowsDescendants?: ( + snapshot: WindowsDescendantSnapshot + ) => Promise + /** Identity probe for the bare-pid tree kill; only Windows has one to gate. */ + verifyRootIdentity?: (root: PosixProcessIdentity | WindowsProcessIdentity) => Promise +} + +export type ClaudeChildTreeReaper = { + /** + * Snapshot the root's live descendants. The moment the root dies they reparent + * and no table walk can find them again, so this has to run before anything + * gives the root a reason to leave. Held once; later calls are no-ops. + */ + capture(): Promise + /** Refresh a live root's snapshot at the close boundary; a failed refresh keeps the prior proof. */ + refresh?: () => Promise + /** + * Kill the child's whole tree and report what the bounded verification + * observed. Concurrent calls share one reap, and a later call re-verifies the + * same snapshot rather than trusting a root that has since died on its own. + */ + reap(): Promise + /** + * `unverifiable` until a reap observes otherwise. `exited` is the only verdict + * that lets a close release the lease; `live` names a descendant that was seen + * still running, which no later caller may collapse into "unknown". + */ + readonly treeVerdict: DescendantTreeVerdict +} + +/** + * The same shared primitives the Codex structured provider composes: a raw + * pipe child owns no PTY job, so there is nothing for the PTY job sweep to + * terminate on Windows and no unref'd timer is allowed to outlive the proof. + * + * The proof is unproven by default. `treeVerdict` is assigned in exactly one + * place, from the verdict of `judgeTree`, so a code path that never reaches a + * verification cannot report the tree gone by omission. + */ +export function createClaudeChildTreeReaper( + child: ReapableChild, + deps: ClaudeChildTreeReaperDeps = {} +): ClaudeChildTreeReaper { + const platform = deps.platform ?? process.platform + const exited = deps.exited ?? (() => false) + // Undefined until captured; null when no admissible snapshot exists — the root + // was already gone, or the table could not be read while it was alive — which + // no later read can make up for. + let snapshot: ClaudeCapturedTree | null | undefined + let capturing: Promise | null = null + let refreshing: Promise | null = null + let queuedRefresh: Promise | null = null + let inFlight: Promise | null = null + let treeVerdict: DescendantTreeVerdict = 'unverifiable' + + // Consulted only on win32: POSIX signals descendants by revalidated identity + // and reaches the root solely through Node's handle, so neither needs a probe. + const verifyRoot = + deps.verifyRootIdentity ?? + ((root: PosixProcessIdentity | WindowsProcessIdentity) => + verifyWindowsProcessIdentity(root as WindowsProcessIdentity)) + + function captureOnce(): Promise { + if (refreshing) { + const pending = refreshing + return pending.then(() => queuedRefresh ?? undefined) + } + if (snapshot !== undefined) { + return Promise.resolve() + } + if (capturing) { + const pending = capturing + return pending.then(() => queuedRefresh ?? undefined) + } + const rootPid = child.pid + if (!rootPid || exited()) { + // Only the root's death makes a missing snapshot final: its descendants + // have reparented, and no later walk can reach them. + snapshot = exited() ? null : snapshot + return Promise.resolve() + } + const capture = + platform === 'win32' + ? (deps.captureWindowsDescendants ?? captureWindowsDescendantSnapshot) + : (deps.captureDescendants ?? captureDescendantSnapshot) + capturing = capture(rootPid) + .catch(() => null) + .then((captured) => { + // A walk that found no root, or that raced the root's death, can only + // have missed descendants that already reparented away. A table that + // could not be read in time is not an answer at all: while the root + // still lives the walk is simply retried, rather than latching a failed + // read as proof that there was nothing to find. + const rootExited = exited() + const tree = admissibleTree(captured, platform, rootExited) + if (tree) { + snapshot = tree + } else if (rootExited) { + // Once the root has exited its descendants may have reparented; no + // later table read can make an absent snapshot safe to signal. + snapshot = null + } else { + // A failed read or a walk that did not observe the live root is + // retryable while the root remains alive. Never latch a vacuous null. + snapshot = undefined + } + }) + .finally(() => { + capturing = null + }) + return capturing + } + + function startRefresh(): Promise { + if (exited()) { + return Promise.resolve() + } + const rootPid = child.pid + if (!rootPid) { + return Promise.resolve() + } + const capture = + platform === 'win32' + ? (deps.captureWindowsDescendants ?? captureWindowsDescendantSnapshot) + : (deps.captureDescendants ?? captureDescendantSnapshot) + const operation = (async () => { + const captured = await capture(rootPid).catch(() => null) + if (exited()) { + return + } + const tree = admissibleTree(captured, platform, false) + if (!tree) { + return + } + if (snapshot === undefined) { + snapshot = tree + return + } + if (snapshot !== null) { + // A merge that returns null saw a same-PID identity change: a + // recycle/replace decision, not an absent descendant, so no row here may + // be signalled from its number. Only the descendant evidence is lost — + // the root still leaves through the handle no recycled pid can reach. + snapshot = mergeClaudeCapturedTrees(snapshot, tree) + } + // Keep an earlier admissible snapshot when this close-boundary read fails; + // it remains the only identity-safe evidence after root exit. + })() + refreshing = operation + const clearRefreshing = (): void => { + if (refreshing === operation) { + refreshing = null + } + } + void operation.then(clearRefreshing, clearRefreshing) + return operation + } + + function queueRefreshAfter(pending: Promise): Promise { + if (queuedRefresh) { + return queuedRefresh + } + const operation = pending.then(() => { + if (exited()) { + return + } + return startRefresh() + }) + queuedRefresh = operation + const clearQueuedRefresh = (): void => { + if (queuedRefresh === operation) { + queuedRefresh = null + } + } + void operation.then(clearQueuedRefresh, clearQueuedRefresh) + return operation + } + + async function refresh(): Promise { + const pending = capturing ?? refreshing + if (pending) { + await queueRefreshAfter(pending) + return + } + if (queuedRefresh) { + await queuedRefresh + return + } + try { + await startRefresh() + } catch { + // A refresh is advisory; capture failures leave the prior proof intact. + } + } + + /** The only source of a tree verdict: every `exited` here is an observation. */ + async function judgeTree(): Promise { + const killRoot = (): boolean => terminateClaudeRoot({ child, exited }) + const rootPid = child.pid + if (!rootPid) { + // Never spawned, so the OS never created a tree to orphan. + return 'exited' + } + await captureOnce() + if (platform === 'win32') { + // Why taskkill's own outcome is never the verdict: it resolves identically + // on a timeout, an access denial, a recycled root and a real kill. + const { rootVerified } = await terminateClaudeWindowsRoot({ + snapshot: snapshot?.platform === 'win32' ? snapshot.tree : null, + exited, + verifyRoot: (root) => verifyRoot(root), + terminateTree: (root) => + deps.terminateWindowsTree + ? deps.terminateWindowsTree(root) + : terminateIdentifiedWindowsProcessTree(root, { + ownsRoot: () => !exited() + }).then(() => undefined), + killRoot + }) + if (!rootVerified && !exited()) { + return 'unverifiable' + } + return snapshot?.platform === 'win32' + ? await (deps.terminateWindowsDescendants ?? verifyWindowsDescendantSnapshotExit)( + snapshot.tree + ) + : 'unverifiable' + } + if (snapshot?.platform !== 'posix') { + killRoot() + return 'unverifiable' + } + if (snapshot.tree.descendants.length === 0) { + // Read while the root was alive and childless: a later table read has no + // row it could match, so it would add nothing to this observation. + killRoot() + return 'exited' + } + // Why the root is killed while verification is already running, and never + // SIGSTOPped first the way the Codex non-group path does: measured on macOS, a + // killed child of a stopped parent stays a zombie row in ps with its lstart + // and pgid intact, so verification cannot pass until the root is dead. The + // descendants are signalled by the verifier as soon as it revalidates their + // identities; the root's death then reparents any zombies to init, which + // reaps them. After a root exit the kill is a no-op: Node drops the handle + // on exit and never signals a possibly recycled pid. + const verdictPromise = deps.terminateDescendants + ? deps.terminateDescendants(snapshot.tree) + : terminateDescendantSnapshotWithVerdict(snapshot.tree, { + requireIdentityBeforeSignal: true + }) + killRoot() + // What the verification observed is the verdict: a kill that reports no + // signal means the handle was already gone, never that the tree survived. + return verdictPromise + } + + return { + capture: captureOnce, + refresh, + reap() { + if (inFlight) { + return inFlight + } + const attempt = judgeTree() + .catch((): DescendantTreeVerdict => 'unverifiable') + .then((verdict) => { + treeVerdict = + TREE_VERDICT_TRUST[verdict] > TREE_VERDICT_TRUST[treeVerdict] ? verdict : treeVerdict + return verdict + }) + inFlight = attempt + void attempt.finally(() => { + if (inFlight === attempt) { + inFlight = null + } + }) + return attempt + }, + get treeVerdict() { + return treeVerdict + } + } +} + +/** + * Orca's own shutdown ladder on the child it spawned, kept because the SDK's + * close path returns no proof and Orca never releases a lease on an assumed exit. + * + * Resolves true only after the child actually emitted exit and its snapshotted + * descendants were observed gone; false is unproven. A root that left on its + * own before a snapshot could be armed stays unproven: its descendants had + * already reparented out of reach when the ladder first looked. + */ +export function proveClaudeChildExit(input: ClaudeChildExitProofInput): Promise { + return proveClaudeChildExitWithReaper(input, () => + createClaudeChildTreeReaper(input.child, { exited: input.exited }) + ) +} diff --git a/src/main/claude/claude-agent-sdk-import-boundary.test.ts b/src/main/claude/claude-agent-sdk-import-boundary.test.ts new file mode 100644 index 00000000000..f1a38466d41 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-import-boundary.test.ts @@ -0,0 +1,154 @@ +import { existsSync, readFileSync, statSync } from 'node:fs' +import { dirname, join, relative, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { spawnProcess } from '../../shared/child-process/run-process' + +/** + * Keep the agent SDK on the structured-Claude side of the toggle. + * + * A user who never leaves the terminal/TUI Claude path must not pay for the SDK: + * importing it evaluates a package that rewrites + * `process.env.NoDefaultCurrentDirectoryInExePath`, changing how Windows resolves + * executables for every later subprocess, and a missing or incompatible install + * would take normal runtime startup down with it. The ordinary + * `OrcaRuntimeService` graph reaches the Claude transport module, so only a + * deferred import keeps that boundary — and only a walk of the real import graph + * keeps the next static import from quietly restoring it. + */ +const SDK_PACKAGE = '@anthropic-ai/claude-agent-sdk' +const REPO_ROOT = resolve(__dirname, '..', '..', '..') + +/** The Electron main entry: everything the app loads before any session exists. */ +const ROOT = 'src/main/index.ts' +/** Proof the walk goes all the way into the Claude transport rather than stopping short. */ +const TRANSPORT_MODULE = 'src/main/claude/claude-stream-json-connection.ts' + +/** + * Static, value-carrying specifiers only, read statement by statement so a + * multi-line `import { ... } from '...'` counts. `import type` is erased before + * the module ever loads and a bare `import(...)` is the deferral this guards, so + * neither is an edge the runtime traverses at load time. + */ +const STATEMENT_START = /^\s*(?:import|export)\b/ +const TYPE_ONLY = /^\s*(?:import|export)\s+type\b/ +const FROM_SPECIFIER = /(?:^|\s)from\s*['"]([^'"]+)['"]/ +const SIDE_EFFECT_IMPORT = /^\s*import\s*['"]([^'"]+)['"]/ +/** An import statement never spans more lines than its longest specifier list. */ +const MAX_STATEMENT_LINES = 60 + +function readSpecifiers(source: string): string[] { + const lines = source.split('\n') + const found: string[] = [] + for (let index = 0; index < lines.length; index += 1) { + const first = lines[index] as string + if (!STATEMENT_START.test(first) || TYPE_ONLY.test(first)) { + continue + } + const sideEffect = SIDE_EFFECT_IMPORT.exec(first) + if (sideEffect) { + found.push(sideEffect[1] as string) + continue + } + for (let scan = index; scan < Math.min(lines.length, index + MAX_STATEMENT_LINES); scan += 1) { + if (scan > index && STATEMENT_START.test(lines[scan] as string)) { + break + } + const specifier = FROM_SPECIFIER.exec(lines[scan] as string) + if (specifier) { + found.push(specifier[1] as string) + break + } + } + } + return found +} + +/** Resolve a relative specifier the way the bundler does; unresolvable means not a module. */ +function resolveRelative(fromFile: string, specifier: string): string | null { + const base = join(dirname(fromFile), specifier) + for (const candidate of [base, `${base}.ts`, `${base}.tsx`, join(base, 'index.ts')]) { + if (existsSync(candidate) && statSync(candidate).isFile()) { + return candidate + } + } + return null +} + +function walkStaticImports(rootFile: string): { visited: Set; sdkImporters: string[] } { + const visited = new Set() + const sdkImporters: string[] = [] + const queue = [resolve(REPO_ROOT, rootFile)] + while (queue.length > 0) { + const file = queue.pop() as string + const key = relative(REPO_ROOT, file).split('\\').join('/') + if (visited.has(key)) { + continue + } + visited.add(key) + for (const specifier of readSpecifiers(readFileSync(file, 'utf8'))) { + if (specifier === SDK_PACKAGE || specifier.startsWith(`${SDK_PACKAGE}/`)) { + sdkImporters.push(key) + continue + } + if (!specifier.startsWith('.')) { + continue + } + const target = resolveRelative(file, specifier) + if (target) { + queue.push(target) + } + } + } + return { visited, sdkImporters } +} + +describe('claude agent SDK import boundary', () => { + const walk = walkStaticImports(ROOT) + + it('walks a graph deep enough to reach the Claude transport', () => { + // Without this the guard passes for the wrong reason the moment the walk breaks. + expect(walk.visited.size).toBeGreaterThan(500) + expect([...walk.visited]).toContain(TRANSPORT_MODULE) + }) + + it('never reaches the SDK through a static import from the main entry', () => { + expect( + walk.sdkImporters, + `${SDK_PACKAGE} must stay behind the structured-Claude boundary. Load it with a deferred import inside the session path instead.` + ).toEqual([]) + }) + + it('leaves the Windows executable-search environment alone when the runtime loads', async () => { + // A vitest file runs in its own fork, so this is a clean process; the ambient + // value is cleared first because the developer's own shell may carry one. + delete process.env.NoDefaultCurrentDirectoryInExePath + await import('../runtime/structured-agent-session-runtime') + + expect(process.env.NoDefaultCurrentDirectoryInExePath).toBeUndefined() + }) + + it('still lets the SDK set it, so the guard above is not measuring nothing', async () => { + // A separate process, not this fork: the assertion has to be about a first + // evaluation of the package, which a cached module registry cannot give. + const { NoDefaultCurrentDirectoryInExePath: _cleared, ...env } = process.env + const probe = spawnProcess({ + program: process.execPath, + args: [ + '-e', + `import(${JSON.stringify(SDK_PACKAGE)}).then(() => console.log(String(process.env.NoDefaultCurrentDirectoryInExePath)))` + ], + cwd: REPO_ROOT, + env: env as Record, + stdio: ['ignore', 'pipe', 'ignore'] + }) + const observed = await new Promise((settle) => { + let output = '' + probe.stdout?.setEncoding('utf8').on('data', (chunk: string) => { + output += chunk + }) + probe.once('close', () => settle(output.trim())) + }) + + expect(observed).toBe('1') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-process-spawn.test.ts b/src/main/claude/claude-agent-sdk-process-spawn.test.ts new file mode 100644 index 00000000000..cd3520cf6d5 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-process-spawn.test.ts @@ -0,0 +1,107 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import type { SpawnOptions as SdkSpawnOptions } from '@anthropic-ai/claude-agent-sdk' +import { resolveSpawn, type spawnProcess } from '../../shared/child-process/run-process' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import { createClaudeCodeProcessSpawn } from './claude-agent-sdk-process-spawn' + +type FakeChild = EventEmitter & { + pid: number + stdin: PassThrough + stdout: PassThrough + stderr: PassThrough + kill: ReturnType +} + +function fakeSpawn() { + const child = new EventEmitter() as FakeChild + child.pid = 4321 + child.stdin = new PassThrough() + child.stdout = new PassThrough() + child.stderr = new PassThrough() + child.kill = vi.fn(() => true) + const specs: ProcessSpec[] = [] + const spawnImpl = ((spec: ProcessSpec) => { + specs.push(spec) + return child + }) as unknown as typeof spawnProcess + return { child, spawnImpl, specs } +} + +function sdkOptions(overrides: Partial = {}): SdkSpawnOptions { + return { + command: '/usr/local/bin/claude', + args: ['--output-format', 'stream-json'], + cwd: '/work/repo', + env: { PATH: '/usr/bin', CLAUDE_CONFIG_DIR: '/accounts/one', UNSET: undefined }, + signal: new AbortController().signal, + ...overrides + } +} + +describe('claude agent SDK process spawn', () => { + it('routes the SDK spawn through Orca and retains the pid the lease adjudicates on', () => { + const process = fakeSpawn() + const spawn = createClaudeCodeProcessSpawn(process.spawnImpl) + + expect(spawn.pid).toBeUndefined() + expect(spawn.child).toBeNull() + const child = spawn.spawn(sdkOptions()) + + expect(child).toBe(process.child) + expect(spawn.child).toBe(process.child) + expect(spawn.pid).toBe(4321) + expect(process.specs[0]).toEqual({ + program: '/usr/local/bin/claude', + args: ['--output-format', 'stream-json'], + cwd: '/work/repo', + env: { PATH: '/usr/bin', CLAUDE_CONFIG_DIR: '/accounts/one' }, + stdio: ['pipe', 'pipe', 'pipe'] + }) + }) + + it('keeps the child out of the SDK abort path so exit proof stays Orca-owned', () => { + const process = fakeSpawn() + const controller = new AbortController() + createClaudeCodeProcessSpawn(process.spawnImpl).spawn(sdkOptions({ signal: controller.signal })) + + // Node's spawn({signal}) kills the child on abort; Orca's ladder must be the + // only thing that can end this process, or close() would report an assumed exit. + expect(process.specs[0]).not.toHaveProperty('signal') + }) + + it('drains stderr into a bounded tail so an exit error still carries it', async () => { + const process = fakeSpawn() + const spawn = createClaudeCodeProcessSpawn(process.spawnImpl) + spawn.spawn(sdkOptions()) + + process.child.stderr.write('x'.repeat(9000)) + process.child.stderr.write('claude: not signed in') + await new Promise((resolve) => setImmediate(resolve)) + + expect(spawn.stderrTail).toMatch(/claude: not signed in$/) + expect(spawn.stderrTail.length).toBe(8192) + }) + + it('hands a Windows .cmd shim to Orca\u2019s argument encoder', () => { + const process = fakeSpawn() + createClaudeCodeProcessSpawn(process.spawnImpl).spawn( + sdkOptions({ + command: 'C:\\Users\\dev\\AppData\\npm\\claude.cmd', + args: ['--setting-sources=user,project,local', '--session-id', 'a b&c'] + }) + ) + + // The spec the spawner builds is what Orca's Windows branch encodes; the SDK's + // own spawn would hand `.cmd` straight to Node and mangle the argument. + const resolved = resolveSpawn(process.specs[0] as ProcessSpec, 'win32') + expect(resolved.file.toLowerCase()).toContain('cmd.exe') + expect(resolved.options.windowsVerbatimArguments).toBe(true) + expect(resolved.args).toHaveLength(1) + // `/v:off` plus the quoted argument is what keeps `&` from splitting the line. + expect(resolved.args[0]).toContain('/v:off') + expect(resolved.args[0]).toContain('"a b&c"') + expect(resolved.args[0]).toContain('"--setting-sources=user,project,local"') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-process-spawn.ts b/src/main/claude/claude-agent-sdk-process-spawn.ts new file mode 100644 index 00000000000..a2b1ad7f158 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-process-spawn.ts @@ -0,0 +1,69 @@ +import type { SpawnOptions as ClaudeAgentSdkSpawnOptions } from '@anthropic-ai/claude-agent-sdk' +import { spawnProcess } from '../../shared/child-process/run-process' + +/** Derived rather than imported: only src/shared/child-process may name node:child_process. */ +type ClaudeCodeChild = ReturnType + +const STDERR_TAIL_MAX_BYTES = 8192 + +export type ClaudeCodeProcessSpawn = { + /** Pass as the SDK's `spawnClaudeCodeProcess`; the SDK never learns the pid because it never owns it. */ + spawn: (options: ClaudeAgentSdkSpawnOptions) => ClaudeCodeChild + /** The retained child, so Orca keeps its own tree-kill and exit-proof ladder. Null until the SDK spawns. */ + readonly child: ClaudeCodeChild | null + /** Ownership proof: the durable lease adjudicates on this pid plus start time plus the spawn token. */ + readonly pid: number | undefined + readonly stderrTail: string +} + +function definedEnv(env: Record): Record { + const next: Record = {} + for (const [key, value] of Object.entries(env)) { + if (value !== undefined) { + next[key] = value + } + } + return next +} + +/** + * Orca supplies the Claude Code child rather than letting the SDK spawn it. + * + * Two independent reasons: the SDK's `SpawnedProcess` has no pid, and Orca's + * spawner is the only path that encodes `.cmd` arguments safely on Windows. + */ +export function createClaudeCodeProcessSpawn( + spawnImpl: typeof spawnProcess = spawnProcess +): ClaudeCodeProcessSpawn { + let child: ClaudeCodeChild | null = null + let stderrTail = '' + return { + spawn: (options) => { + // Why `options.signal` is dropped: it would let the SDK kill the child outside + // Orca's ladder, and close() may never report an exit it did not observe. + const spawned = spawnImpl({ + program: options.command, + args: [...options.args], + ...(options.cwd === undefined ? {} : { cwd: options.cwd }), + env: definedEnv(options.env), + stdio: ['pipe', 'pipe', 'pipe'] + }) + child = spawned + // The SDK drains stderr only for its own local spawn, so a custom spawner must: + // otherwise the child blocks on a full pipe and exit errors lose their tail. + spawned.stderr.setEncoding('utf8').on('data', (chunk: string) => { + stderrTail = (stderrTail + chunk).slice(-STDERR_TAIL_MAX_BYTES) + }) + return spawned + }, + get child() { + return child + }, + get pid() { + return child?.pid + }, + get stderrTail() { + return stderrTail + } + } +} diff --git a/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts b/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts new file mode 100644 index 00000000000..9f843527e30 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts @@ -0,0 +1,190 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' +import { mergeClaudeCapturedTrees } from './claude-child-tree-snapshot' + +const ROOT_PID = 424242 +const ROOT_STARTED_AT = 'Mon Jan 1 00:00:00 2026' +const ROOT_FORK_MS = Date.parse(ROOT_STARTED_AT) + +function mockChild(): EventEmitter & + Pick & { kill: ReturnType } { + return Object.assign(new EventEmitter(), { + pid: ROOT_PID, + stdin: new PassThrough(), + kill: vi.fn(() => true) + }) as never +} + +function posixSnapshot(input: { + capturedAtMs: number + descendants?: DescendantSnapshot['descendants'] +}): DescendantSnapshot { + return { + root: { pid: ROOT_PID, startedAt: ROOT_STARTED_AT }, + rootPgid: ROOT_PID, + descendants: input.descendants ?? [], + capturedAtMs: input.capturedAtMs + } +} + +function windowsSnapshot(capturedAtMs = 1): WindowsDescendantSnapshot { + return { + root: { pid: ROOT_PID, creationTimeMs: 1_700_000_000_001 }, + descendants: [{ pid: 4243, creationTimeMs: 1_700_000_000_000 }], + unidentifiedCount: 0, + capturedAtMs + } +} + +describe('Claude root kill fallback', () => { + it('kills the root when the first capture landed in the fork second', async () => { + // The production POSIX verifier declines a root born in its capture second, + // and that verdict must not cost the tree the kill on Node's own handle. + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => posixSnapshot({ capturedAtMs: ROOT_FORK_MS + 300 })) + }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the root after a recycled descendant pid voided the snapshot', async () => { + const child = mockChild() + const captureDescendants = vi + .fn() + .mockResolvedValueOnce( + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 5_000, + descendants: [{ pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: ROOT_STARTED_AT }] + }) + ) + .mockResolvedValueOnce( + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 6_000, + descendants: [ + { pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: 'Mon Jan 1 00:00:30 2026' } + ] + }) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants, + terminateDescendants: vi.fn(async () => 'exited' as const), + verifyRootIdentity: vi.fn(async () => true) + }) + + await tree.capture() + await tree.refresh?.() + // The descendant evidence is rightly discarded; the root's never was in doubt. + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('keeps an observed live descendant when the root identity probe declined', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 5_000, + descendants: [{ pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: ROOT_STARTED_AT }] + }) + ), + terminateDescendants: vi.fn(async () => 'live' as const), + verifyRootIdentity: vi.fn(async () => false) + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('reports a Windows taskkill that worked as exited, not unverifiable', async () => { + const child = mockChild() + // Probe 1 gates taskkill; a later probe correctly finds the root already dead. + const verifyRootIdentity = vi.fn().mockResolvedValueOnce(true).mockResolvedValue(false) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => false, + captureWindowsDescendants: vi.fn(async () => windowsSnapshot()), + terminateWindowsTree: vi.fn(async () => {}), + terminateWindowsDescendants: vi.fn(async () => 'exited' as const), + verifyRootIdentity + }) + + await expect(tree.reap()).resolves.toBe('exited') + }) + + it('kills the root when no POSIX snapshot could be read', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => null), + terminateDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the root when the Windows process table is unreadable', async () => { + const child = mockChild() + const terminateWindowsTree = vi.fn(async () => {}) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => false, + captureWindowsDescendants: vi.fn(async () => null), + terminateWindowsTree, + terminateWindowsDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // No identity means no bare-pid tree kill, but the owned handle is still ours. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('never signals a root the reaper already saw exit', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => true, + captureDescendants: vi.fn(async () => null) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).not.toHaveBeenCalled() + }) + + it('chains per-pid Windows boundaries across a second merge', async () => { + const first = windowsSnapshot(1_000) + const second: WindowsDescendantSnapshot = { + ...windowsSnapshot(2_000), + descendants: [ + { pid: 4243, creationTimeMs: 1_700_000_000_000 }, + { pid: 4244, creationTimeMs: 1_700_000_000_002 } + ] + } + const third: WindowsDescendantSnapshot = { ...second, capturedAtMs: 3_000 } + + const merged = mergeClaudeCapturedTrees( + { platform: 'win32', tree: first }, + { platform: 'win32', tree: second } + ) + expect(merged?.tree.capturedAtMsByPid).toEqual({ '4243': 1_000, '4244': 2_000 }) + const rechained = mergeClaudeCapturedTrees(merged!, { platform: 'win32', tree: third }) + + expect(rechained?.tree.capturedAtMsByPid).toEqual({ '4243': 1_000, '4244': 2_000 }) + }) +}) diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.test.ts b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts new file mode 100644 index 00000000000..62fbd8cf203 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts @@ -0,0 +1,65 @@ +import type { SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' +import { describe, expect, it } from 'vitest' +import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' + +/** + * The SDK's input pump is `for await (const frame of prompt) { await transport.write(frame) }`. + * A rejected write — or an abort — ends that loop abruptly, which calls the + * generator's `return()`. Everything below drives that exact shape, because the + * frame the pump already pulled is the one nothing else can reach. + */ +const frame = (text: string): SDKUserMessage => + ({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text }] } + }) as unknown as SDKUserMessage + +const settled = (promise: Promise): Promise<'settled' | 'pending'> => + Promise.race([ + promise.then( + () => 'settled' as const, + () => 'settled' as const + ), + new Promise<'pending'>((resolve) => setTimeout(() => resolve('pending'), 100)) + ]) + +describe('claude user message queue', () => { + it('rejects the frame the SDK pulled but abandoned without writing', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + await pump.next() + await pump.return?.(undefined) + + await expect(settled(sent)).resolves.toBe('settled') + await expect(sent).rejects.toThrow( + 'claude stream-json input ended before the frame was written' + ) + }) + + it('rejects an in-flight frame from fail() when the SDK never resumes the pump', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + await pump.next() + queue.fail(new Error('claude stream-json exited: child died')) + + await expect(settled(sent)).resolves.toBe('settled') + await expect(sent).rejects.toThrow('claude stream-json exited: child died') + }) + + it('still settles a written frame only once the pump asks for the next one', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + const pulled = await pump.next() + expect(pulled.value).toMatchObject({ type: 'user' }) + // The write proof is the pump coming back for more, exactly as before. + await expect(settled(sent)).resolves.toBe('pending') + void pump.next() + await expect(sent).resolves.toBeUndefined() + }) +}) diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.ts b/src/main/claude/claude-agent-sdk-user-message-queue.ts new file mode 100644 index 00000000000..87fa6660159 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-user-message-queue.ts @@ -0,0 +1,100 @@ +import type { SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' + +type QueuedMessage = { + message: SDKUserMessage + resolve: () => void + reject: (error: Error) => void +} + +export type ClaudeUserMessageQueue = { + /** The SDK's streaming-input prompt; it stays open until `end`. */ + messages: AsyncIterable + /** Resolves once the SDK has finished writing the frame to the child. */ + push: (message: SDKUserMessage) => Promise + /** Reject every unwritten frame, in-flight included; a caller waiting on a send must not hang past the exit. */ + fail: (error: Error) => void + end: () => void +} + +/** The rejection an abandoned frame carries when nothing else has named a cause yet. */ +const UNWRITTEN_FRAME_MESSAGE = 'claude stream-json input ended before the frame was written' + +export function createClaudeUserMessageQueue(): ClaudeUserMessageQueue { + const queued: QueuedMessage[] = [] + // The frame the SDK has taken but not yet acknowledged. It is out of `queued`, + // so it is unreachable from anywhere else and would otherwise never settle. + let inFlight: QueuedMessage | null = null + let wake: (() => void) | null = null + let ended = false + let failure: Error | null = null + const notify = (): void => { + wake?.() + wake = null + } + const rejectInFlight = (error: Error): void => { + const abandoned = inFlight + inFlight = null + abandoned?.reject(error) + } + + async function* drain(): AsyncGenerator { + for (;;) { + const next = queued.shift() + if (next) { + inFlight = next + let written = false + try { + yield next.message + written = true + } finally { + // The SDK's input pump abandons this iterator when its + // `await transport.write(...)` rejects or the query aborts, and the code + // after a `yield` never runs on that path. Settling here is the only + // place a frame it already took can be reached. + if (written) { + inFlight = null + // Resumed only after the SDK's `await transport.write(...)` settled, so this + // is the same "the frame reached the child" proof the hand-rolled write gave. + next.resolve() + } else { + rejectInFlight(failure ?? new Error(UNWRITTEN_FRAME_MESSAGE)) + } + } + continue + } + if (ended || failure) { + return + } + await new Promise((resolve) => { + wake = resolve + }) + } + } + + return { + messages: drain(), + push: (message) => + new Promise((resolve, reject) => { + if (failure) { + reject(failure) + return + } + queued.push({ message, resolve, reject }) + notify() + }), + fail: (error) => { + failure ??= error + for (const entry of queued.splice(0)) { + entry.reject(error) + } + // A pump that never resumes cannot run the generator's cleanup, so the + // exit path has to reach the in-flight frame itself. + rejectInFlight(error) + notify() + }, + end: () => { + ended = true + notify() + } + } +} diff --git a/src/main/claude/claude-child-exit-proof-ladder.ts b/src/main/claude/claude-child-exit-proof-ladder.ts new file mode 100644 index 00000000000..85ed629f1b9 --- /dev/null +++ b/src/main/claude/claude-child-exit-proof-ladder.ts @@ -0,0 +1,41 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import { waitForProcessExitUntil } from '../codex/codex-process-exit-deadline' +import type { ClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' + +const GRACEFUL_EXIT_MS = 1_500 +const FORCED_EXIT_MS = 1_000 + +export type ClaudeChildExitProofInput = { + child: Pick + exitPromise: Promise + exited: () => boolean + tree?: ClaudeChildTreeReaper +} + +export async function proveClaudeChildExitWithReaper( + input: ClaudeChildExitProofInput, + createTree: () => ClaudeChildTreeReaper +): Promise { + const tree = input.tree ?? createTree() + // Arm before stdin closes: only a live root can identify its descendants. + await tree.capture() + try { + input.child.stdin?.end() + } catch { + // The reap below still owns the process. + } + let reaped = false + if (!input.exited()) { + await waitForProcessExitUntil(input.exitPromise, GRACEFUL_EXIT_MS) + if (!input.exited()) { + reaped = true + await tree.refresh?.() + await tree.reap() + await waitForProcessExitUntil(input.exitPromise, FORCED_EXIT_MS) + } + } + if (!reaped && input.exited() && tree.treeVerdict !== 'exited') { + await tree.reap() + } + return input.exited() && tree.treeVerdict === 'exited' +} diff --git a/src/main/claude/claude-child-process-environment.test.ts b/src/main/claude/claude-child-process-environment.test.ts new file mode 100644 index 00000000000..8299fdcdd61 --- /dev/null +++ b/src/main/claude/claude-child-process-environment.test.ts @@ -0,0 +1,63 @@ +import { describe, expect, it } from 'vitest' +import { applyClaudeEnvPatch } from '../claude-accounts/environment' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' + +describe('Claude child process environment', () => { + it('strips case-insensitive auth headers through the shared env patch on Windows', () => { + expect( + applyClaudeEnvPatch( + { + anthropic_api_key: 'inherited-key', + Anthropic_Custom_Headers: 'Authorization: inherited', + SAFE_VALUE: 'preserved' + }, + {}, + { stripAuthEnv: true, platform: 'win32' } + ) + ).toEqual({ SAFE_VALUE: 'preserved' }) + }) + + it('strips case-insensitive inherited auth and session stamps on Windows', () => { + const env = buildClaudeChildProcessEnv( + { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + Claude_Code_Session_Id: 'configured-session' + }, + { + platform: 'win32', + inheritedEnv: { + anthropic_api_key: 'inherited-key', + Anthropic_Custom_Headers: 'Authorization: inherited', + claude_code_child_session: '1', + CLAUDE_CODE_SESSION_ID: 'inherited-session', + SAFE_VALUE: 'preserved' + } + } + ) + + expect(env).toEqual({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + Claude_Code_Session_Id: 'configured-session', + SAFE_VALUE: 'preserved' + }) + }) + + it('can strip child-session stamps reintroduced by a full SDK launch overlay', () => { + expect( + buildClaudeChildProcessEnv( + { + CLAUDE_CODE_CHILD_SESSION: 'configured-child-session', + CLAUDE_CODE_SESSION_ID: 'configured-session', + CLAUDE_CODE_BRIDGE_SESSION_ID: 'configured-bridge-session' + }, + { + scrubConfiguredChildSessionStamps: true, + inheritedEnv: { + CLAUDE_CODE_CHILD_SESSION: 'inherited-child-session', + SAFE_VALUE: 'preserved' + } + } + ) + ).toEqual({ SAFE_VALUE: 'preserved' }) + }) +}) diff --git a/src/main/claude/claude-child-process-environment.ts b/src/main/claude/claude-child-process-environment.ts new file mode 100644 index 00000000000..58f00c5c1b8 --- /dev/null +++ b/src/main/claude/claude-child-process-environment.ts @@ -0,0 +1,69 @@ +import { CLAUDE_AUTH_ENV_VARS, applyClaudeEnvPatch } from '../claude-accounts/environment' + +const CLAUDE_CHILD_SESSION_STAMP_ENV_KEYS = [ + 'CLAUDE_CODE_CHILD_SESSION', + 'CLAUDE_CODE_SESSION_ID', + 'CLAUDE_CODE_BRIDGE_SESSION_ID' +] as const + +function cloneProcessEnv(source: NodeJS.ProcessEnv): Record { + const env: Record = {} + for (const [key, value] of Object.entries(source)) { + if (value !== undefined) { + env[key] = value + } + } + return env +} + +function stripClaudeChildSessionStamps( + env: Record, + platform: NodeJS.Platform +): Record { + for (const key of CLAUDE_CHILD_SESSION_STAMP_ENV_KEYS) { + for (const envKey of Object.keys(env)) { + if (envKey === key || (platform === 'win32' && envKey.toUpperCase() === key)) { + delete env[envKey] + } + } + } + return env +} + +export function buildClaudeChildProcessEnv( + configuredEnv: Record = {}, + options: { + inheritedEnv?: NodeJS.ProcessEnv + platform?: NodeJS.Platform + scrubConfiguredChildSessionStamps?: boolean + } = {} +): Record { + const inheritedEnv = options.inheritedEnv ?? process.env + const platform = options.platform ?? process.platform + const env = applyClaudeEnvPatch( + cloneProcessEnv(inheritedEnv), + {}, + { + stripAuthEnv: true, + platform + } + ) + if (platform === 'win32') { + const authKeys = new Set(CLAUDE_AUTH_ENV_VARS.map((key) => key.toUpperCase())) + for (const [key, value] of Object.entries(env)) { + const normalized = key.toUpperCase() + if ( + authKeys.has(normalized) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && + /authorization|x-api-key|api-key|bearer/i.test(value)) + ) { + delete env[key] + } + } + } + if (options.scrubConfiguredChildSessionStamps) { + return stripClaudeChildSessionStamps({ ...env, ...configuredEnv }, platform) + } + stripClaudeChildSessionStamps(env, platform) + return { ...env, ...configuredEnv } +} diff --git a/src/main/claude/claude-child-root-termination.ts b/src/main/claude/claude-child-root-termination.ts new file mode 100644 index 00000000000..bed422532e1 --- /dev/null +++ b/src/main/claude/claude-child-root-termination.ts @@ -0,0 +1,54 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import type { PosixProcessIdentity } from '../pty-descendant-termination' +import type { + WindowsDescendantSnapshot, + WindowsProcessIdentity +} from '../windows-descendant-exit-verification' + +export type ClaudeRootIdentity = PosixProcessIdentity | WindowsProcessIdentity + +type RootTerminationInput = { + child: Pick + exited: () => boolean +} + +/** + * Kills the root through the handle Node owns rather than through its pid, which + * is why no identity probe gates it: libuv drops that handle in the same turn it + * reaps, so the signal either reaches the process Orca spawned or reaches + * nothing. A probe here could only let an unreadable process table cost the tree + * the one fallback that still works once every table read has failed. + * + * False means no signal was sent, because the root had already left. + */ +export function terminateClaudeRoot(input: RootTerminationInput): boolean { + return input.exited() ? false : input.child.kill('SIGKILL') +} + +type WindowsRootTerminationInput = { + snapshot: WindowsDescendantSnapshot | null + exited: () => boolean + verifyRoot: (root: WindowsProcessIdentity) => Promise + terminateTree: (root: WindowsProcessIdentity) => Promise + killRoot: () => boolean +} + +/** + * `taskkill /T /F` addresses a bare pid, so a dead root's pid may already belong + * to a stranger whose whole tree it would take down: that one is identity-gated. + * The direct root kill after it runs however the probe decided. + */ +export async function terminateClaudeWindowsRoot( + input: WindowsRootTerminationInput +): Promise<{ rootVerified: boolean }> { + const { snapshot, exited, verifyRoot, terminateTree, killRoot } = input + let rootVerified = false + if (!exited() && snapshot) { + rootVerified = await verifyRoot(snapshot.root).catch(() => false) + if (rootVerified && !exited()) { + await terminateTree(snapshot.root).catch(() => {}) + } + } + killRoot() + return { rootVerified } +} diff --git a/src/main/claude/claude-child-tree-snapshot.ts b/src/main/claude/claude-child-tree-snapshot.ts new file mode 100644 index 00000000000..e0955648b02 --- /dev/null +++ b/src/main/claude/claude-child-tree-snapshot.ts @@ -0,0 +1,128 @@ +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' + +/** One platform's descendant tree, tagged so neither verifier can be handed the other's rows. */ +export type ClaudeCapturedTree = + | { platform: 'posix'; tree: DescendantSnapshot } + | { platform: 'win32'; tree: WindowsDescendantSnapshot } + +/** + * Process-table reads are not atomic: a refresh can omit a still-live row, but + * it can also observe a new process after the old row exited. Retain rows absent + * from the refresh, but reject a PID whose identity changed between reads. + */ +function mergeRowsByPid( + previous: readonly Row[], + next: readonly Row[], + sameIdentity: (previous: Row, next: Row) => boolean, + previousBoundary: (row: Row) => number, + nextBoundary: (row: Row) => number, + refreshBoundary: number +): { rows: Row[]; capturedAtMsByPid?: Readonly> } | null { + const merged = new Map() + const capturedAtMsByPid: Record = {} + for (const row of previous) { + const prior = merged.get(row.pid) + if (prior && !sameIdentity(prior, row)) { + return null + } + merged.set(row.pid, row) + capturedAtMsByPid[String(row.pid)] = previousBoundary(row) + } + for (const row of next) { + const prior = merged.get(row.pid) + if (prior && !sameIdentity(prior, row)) { + return null + } + if (!prior) { + capturedAtMsByPid[String(row.pid)] = nextBoundary(row) + } + merged.set(row.pid, row) + } + const boundaries = Object.values(capturedAtMsByPid) + const needsBoundaryMap = + new Set(boundaries).size > 1 || boundaries.some((boundary) => boundary !== refreshBoundary) + return { + rows: [...merged.values()], + ...(needsBoundaryMap ? { capturedAtMsByPid } : {}) + } +} + +export function mergeClaudeCapturedTrees( + previous: ClaudeCapturedTree, + next: ClaudeCapturedTree +): ClaudeCapturedTree | null { + if (previous.platform !== next.platform) { + return null + } + if (previous.platform === 'posix' && next.platform === 'posix') { + if (previous.tree.rootPgid !== next.tree.rootPgid) { + return null + } + // A refresh cannot repair an earlier capture that lacked root identity; + // retaining those rows would permit a later numeric-pid kill without proof. + if (!previous.tree.root || !next.tree.root) { + return null + } + if ( + previous.tree.root.pid !== next.tree.root.pid || + previous.tree.root.startedAt !== next.tree.root.startedAt + ) { + return null + } + const descendants = mergeRowsByPid( + previous.tree.descendants, + next.tree.descendants, + (left, right) => left.pgid === right.pgid && left.startedAt === right.startedAt, + (row) => previous.tree.capturedAtMsByPid?.[String(row.pid)] ?? previous.tree.capturedAtMs, + (row) => next.tree.capturedAtMsByPid?.[String(row.pid)] ?? next.tree.capturedAtMs, + next.tree.capturedAtMs + ) + if (!descendants) { + return null + } + return { + platform: 'posix', + tree: { + ...next.tree, + // Retained rows keep their earlier boundary; new rows use the refresh + // boundary. The scalar remains the latest scan for legacy consumers. + descendants: descendants.rows, + ...(descendants.capturedAtMsByPid + ? { capturedAtMsByPid: descendants.capturedAtMsByPid } + : {}) + } + } + } + if (previous.platform === 'win32' && next.platform === 'win32') { + if ( + previous.tree.root.pid !== next.tree.root.pid || + previous.tree.root.creationTimeMs !== next.tree.root.creationTimeMs + ) { + return null + } + const descendants = mergeRowsByPid( + previous.tree.descendants, + next.tree.descendants, + (left, right) => left.creationTimeMs === right.creationTimeMs, + (row) => previous.tree.capturedAtMsByPid?.[String(row.pid)] ?? previous.tree.capturedAtMs, + (row) => next.tree.capturedAtMsByPid?.[String(row.pid)] ?? next.tree.capturedAtMs, + next.tree.capturedAtMs + ) + if (!descendants) { + return null + } + return { + platform: 'win32', + tree: { + ...next.tree, + descendants: descendants.rows, + ...(descendants.capturedAtMsByPid + ? { capturedAtMsByPid: descendants.capturedAtMsByPid } + : {}), + unidentifiedCount: Math.max(previous.tree.unidentifiedCount, next.tree.unidentifiedCount) + } + } + } + return null +} diff --git a/src/main/claude/claude-command-lifecycle-frames.test.ts b/src/main/claude/claude-command-lifecycle-frames.test.ts new file mode 100644 index 00000000000..8126a546ed8 --- /dev/null +++ b/src/main/claude/claude-command-lifecycle-frames.test.ts @@ -0,0 +1,115 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: () => {}, + publish: vi.fn() + } + return { sink, items } +} + +function providerFrameKinds(items: { body: AgentJournalItemBody }[]): string[] { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) +} + +/** + * The queue-bookkeeping frame Claude Code 2.1.258 emits for every uuid-stamped + * command: `command_uuid` plus a state, and no content of its own. Shape and + * states taken from the CLI's own emission sites. + */ +function commandLifecycle(state: 'started' | 'completed' | 'cancelled', uuid: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'command_lifecycle', + command_uuid: 'command-1', + state, + uuid, + session_id: 'claude-session' + } + } +} + +function userTurn(uuid: string, text: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + startsTurn: true as const, + message: { + type: 'user', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: text } + } + } +} + +function assistantReply(uuid: string, text: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'assistant', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { role: 'assistant', content: [{ type: 'text', text }] } + } + } +} + +describe('Claude command_lifecycle frames', () => { + it('keeps queue bookkeeping off the transcript for a whole turn', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 'Reply with exactly PROBE_OK and nothing else.')) + translator.handle(commandLifecycle('started', 'lifecycle-1')) + translator.handle(assistantReply('assistant-1', 'PROBE_OK')) + translator.handle(commandLifecycle('completed', 'lifecycle-2')) + translator.handle(commandLifecycle('completed', 'lifecycle-3')) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { + type: 'result', + subtype: 'success', + uuid: 'result-1', + session_id: 'claude-session', + is_error: false, + result: 'PROBE_OK', + terminal_reason: 'completed' + } + }) + + expect(providerFrameKinds(state.items)).toEqual([]) + // The turn's real content is untouched. + expect( + state.items.flatMap((item) => + item.body.kind === 'message' && item.body.role === 'assistant' ? [item.body.blocks] : [] + ) + ).toEqual([[{ type: 'text', text: 'PROBE_OK' }]]) + }) + + it('keeps a cancelled command off the transcript too', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(commandLifecycle('cancelled', 'lifecycle-4')) + + expect(providerFrameKinds(state.items)).toEqual([]) + }) +}) diff --git a/src/main/claude/claude-config-dir-pin.test.ts b/src/main/claude/claude-config-dir-pin.test.ts new file mode 100644 index 00000000000..c7be40a6f38 --- /dev/null +++ b/src/main/claude/claude-config-dir-pin.test.ts @@ -0,0 +1,34 @@ +import { homedir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { claudeConfigDirEnvPatch, defaultClaudeConfigDir } from './claude-config-dir-pin' + +describe('claude config dir pin', () => { + it('does not pin the CLI default home, so the macOS Keychain stays reachable', () => { + expect(claudeConfigDirEnvPatch(join(homedir(), '.claude'), { env: {} })).toEqual({}) + expect(claudeConfigDirEnvPatch(`${join(homedir(), '.claude')}/`, { env: {} })).toEqual({}) + expect(claudeConfigDirEnvPatch(' ', { env: {} })).toEqual({}) + }) + + it('pins a managed account home the CLI would not find on its own', () => { + expect(claudeConfigDirEnvPatch('/accounts/claude/managed', { env: {} })).toEqual({ + CLAUDE_CONFIG_DIR: '/accounts/claude/managed' + }) + }) + + it('treats an inherited CLAUDE_CONFIG_DIR as the default the CLI already resolves', () => { + const env = { CLAUDE_CONFIG_DIR: '/inherited/home' } + expect(defaultClaudeConfigDir(env)).toBe('/inherited/home') + expect(claudeConfigDirEnvPatch('/inherited/home', { env })).toEqual({}) + expect(claudeConfigDirEnvPatch('/other/home', { env })).toEqual({ + CLAUDE_CONFIG_DIR: '/other/home' + }) + }) + + it('compares Windows homes case-insensitively', () => { + const env = { CLAUDE_CONFIG_DIR: 'C:\\Users\\Work\\.claude' } + expect(claudeConfigDirEnvPatch('c:\\users\\work\\.claude', { env, platform: 'win32' })).toEqual( + {} + ) + }) +}) diff --git a/src/main/claude/claude-config-dir-pin.ts b/src/main/claude/claude-config-dir-pin.ts new file mode 100644 index 00000000000..d5cc0b8b186 --- /dev/null +++ b/src/main/claude/claude-config-dir-pin.ts @@ -0,0 +1,37 @@ +import { homedir } from 'node:os' +import { join, resolve } from 'node:path' + +/** The config dir the Claude CLI resolves for itself when nothing pins one. */ +export function defaultClaudeConfigDir(env: NodeJS.ProcessEnv = process.env): string { + return env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') +} + +function samePath(a: string, b: string, platform: NodeJS.Platform): boolean { + const left = resolve(a) + const right = resolve(b) + return platform === 'win32' ? left.toLowerCase() === right.toLowerCase() : left === right +} + +/** + * An explicit CLAUDE_CONFIG_DIR moves the Claude CLI off the default Keychain item onto + * one derived from the pinned path, so a claude.ai OAuth login stops working even when + * the pin names the CLI's own default. Pin only a home the CLI would not find on its + * own — the same rule the legacy PTY path applies via `ClaudeRuntimePathResolver`. + * + * The pinned value is the account home verbatim: the CLI keys its credential lookup on + * the literal string, so re-spelling an equivalent path (absolute vs `~`, trailing + * separator) selects a different identity. Normalization here is for the equality test + * only and must never reach the env. + */ +export function claudeConfigDirEnvPatch( + accountHome: string, + options: { env?: NodeJS.ProcessEnv; platform?: NodeJS.Platform } = {} +): { CLAUDE_CONFIG_DIR?: string } { + const env = options.env ?? process.env + const platform = options.platform ?? process.platform + const resolved = accountHome.trim() + if (!resolved || samePath(resolved, defaultClaudeConfigDir(env), platform)) { + return {} + } + return { CLAUDE_CONFIG_DIR: resolved } +} diff --git a/src/main/claude/claude-descendant-escalation-boundary.test.ts b/src/main/claude/claude-descendant-escalation-boundary.test.ts new file mode 100644 index 00000000000..55459df0c7f --- /dev/null +++ b/src/main/claude/claude-descendant-escalation-boundary.test.ts @@ -0,0 +1,124 @@ +import { describe, expect, it, vi } from 'vitest' +import { terminateDescendantSnapshotWithVerdict } from '../pty-descendant-exit-verification' +import { + collectDescendantRows, + type DescendantSnapshot, + type ProcessTableRow +} from '../pty-descendant-termination' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' + +const ROOT_PID = 500 +const ORCA_PGID = 400 +const ROOT_STARTED_AT = 'Thu Sep 3 18:04:50 2026' +/** The second both close-time walks land in. */ +const WALK_SECOND = 'Thu Sep 3 18:05:04 2026' +const WALK_MS = Date.parse(WALK_SECOND) +const EARLIER_SECOND = 'Thu Sep 3 18:05:03 2026' + +/** The measured split: `s20` at :03.946 died, `s21` at :04.042 leaked. */ +const EARLIER_BORN = [700, 701, 702] +const WALK_SECOND_BORN = [721, 722, 723, 724] + +type Cohort = { pids: number[]; startedAt: string } + +const LIVE_TREE: Cohort[] = [ + { pids: EARLIER_BORN, startedAt: EARLIER_SECOND }, + { pids: WALK_SECOND_BORN, startedAt: WALK_SECOND } +] + +function rowsFor(cohorts: Cohort[]): ProcessTableRow[] { + return [ + { pid: ROOT_PID, ppid: 1, pgid: ORCA_PGID, startedAt: ROOT_STARTED_AT }, + ...cohorts.flatMap((cohort) => + cohort.pids.map((pid) => ({ + pid, + ppid: ROOT_PID, + pgid: ORCA_PGID, + startedAt: cohort.startedAt + })) + ) + ] +} + +/** A real ppid walk from the root, exactly as production captures one. */ +function walk(capturedAtMs: number, cohorts: Cohort[] = LIVE_TREE): DescendantSnapshot { + return collectDescendantRows(ROOT_PID, rowsFor(cohorts), capturedAtMs) +} + +function killedPids(calls: [number, NodeJS.Signals][]): number[] { + return calls.flatMap(([pid, signal]) => (signal === 'SIGKILL' ? [pid] : [])).sort((a, b) => a - b) +} + +function signalledPids(calls: [number, NodeJS.Signals][]): number[] { + return calls.flatMap(([pid, signal]) => (signal === 'SIGTERM' ? [pid] : [])).sort((a, b) => a - b) +} + +/** + * Drives the real reaper and the real verifier against a process table where + * every descendant traps SIGTERM, so only a forced sweep can end them. The root + * is alive for both walks and gone by the sweep, which is the measured teardown. + */ +async function sweep( + captures: DescendantSnapshot[], + liveTree: Cohort[] = LIVE_TREE +): Promise<[number, NodeJS.Signals][]> { + const calls: [number, NodeJS.Signals][] = [] + const captureDescendants = vi.fn() + for (const capture of captures) { + captureDescendants.mockResolvedValueOnce(capture) + } + const tree = createClaudeChildTreeReaper( + { pid: ROOT_PID, kill: vi.fn(() => true) }, + { + platform: 'linux', + exited: () => false, + captureDescendants, + terminateDescendants: (snapshot) => + terminateDescendantSnapshotWithVerdict(snapshot, { + requireIdentityBeforeSignal: true, + graceMs: 0, + verifyMs: 120, + sendSignal: (pid, signal) => calls.push([pid, signal]), + readTable: async () => ({ rows: rowsFor(liveTree), capturedAtMs: Date.now() }) + }) + } + ) + // The close ladder's shape: arm, then re-walk the live root at the boundary. + await tree.capture() + await tree.refresh?.() + await tree.reap() + return calls +} + +describe('Claude descendant forced-sweep fence', () => { + it('escalates a descendant forked in the same second as both close walks', async () => { + // Both walks land inside second :04, one ps duration apart, and the root is + // gone before a third could run. A descendant born at :04.042 is no less + // ours than its sibling born 96ms earlier at :03.946. + const calls = await sweep([walk(WALK_MS + 42), walk(WALK_MS + 140)]) + + expect(signalledPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + expect(killedPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + }) + + it('still escalates descendants born before the walk that first saw them', async () => { + const onlyEarlier = [{ pids: EARLIER_BORN, startedAt: EARLIER_SECOND }] + const calls = await sweep([walk(WALK_MS + 42, onlyEarlier)], onlyEarlier) + + expect(killedPids(calls)).toEqual(EARLIER_BORN) + }) + + it('withholds the sweep from a row no walk re-derived, on its start second alone', async () => { + // 900 was seen once, in its own birth second, and the refresh did not find + // it. The merge retains the row, but nothing re-proved it belongs to us, so + // the second-resolution fence is all there is and it still says no. + const retained = { pids: [900], startedAt: WALK_SECOND } + const firstWalk = walk(WALK_MS + 42, [...LIVE_TREE, retained]) + const refresh = walk(WALK_MS + 140) + + const calls = await sweep([firstWalk, refresh], [...LIVE_TREE, retained]) + + expect(signalledPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN, 900]) + expect(killedPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + }) +}) diff --git a/src/main/claude/claude-stream-json-connection-close.test.ts b/src/main/claude/claude-stream-json-connection-close.test.ts new file mode 100644 index 00000000000..e824139a776 --- /dev/null +++ b/src/main/claude/claude-stream-json-connection-close.test.ts @@ -0,0 +1,126 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import type { ChildProcessWithoutNullStreams } from 'node:child_process' +import { describe, expect, it, vi } from 'vitest' +import type { query } from '@anthropic-ai/claude-agent-sdk' +import { + openClaudeStreamJsonConnection, + type ClaudeStreamJsonLaunch +} from './claude-stream-json-connection' + +const mocks = vi.hoisted(() => { + const refresh = vi.fn() + const proveClaudeChildExit = vi.fn() + const tree = { + capture: vi.fn(async () => {}), + refresh: (...args: unknown[]) => refresh(...args), + reap: vi.fn(async () => 'exited' as const), + treeVerdict: 'unverifiable' as const + } + return { proveClaudeChildExit, refresh, tree } +}) + +vi.mock('./claude-agent-sdk-exit-proof', () => ({ + createClaudeChildTreeReaper: vi.fn(() => mocks.tree), + proveClaudeChildExit: (...args: unknown[]) => mocks.proveClaudeChildExit(...args) +})) + +function fakeChild(): ChildProcessWithoutNullStreams { + const child = new EventEmitter() + return Object.assign(child, { + pid: 424242, + stdin: new PassThrough(), + stdout: new PassThrough(), + stderr: new PassThrough(), + kill: vi.fn() + }) as unknown as ChildProcessWithoutNullStreams +} + +describe('Claude stream-json close ordering', () => { + it('waits for the live tree refresh before ending stdin', async () => { + const refreshDone = Promise.withResolvers() + mocks.refresh.mockReturnValueOnce(refreshDone.promise) + mocks.proveClaudeChildExit.mockResolvedValueOnce(true) + const child = fakeChild() + const launch: ClaudeStreamJsonLaunch = { + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo' + } + const queryImpl = ((params: Parameters[0]) => { + if (!params.options) { + throw new Error('missing SDK options') + } + params.options.spawnClaudeCodeProcess?.({ + command: 'claude', + args: [], + env: {}, + signal: new AbortController().signal + }) + void (async () => { + for await (const _message of params.prompt) { + // The SDK owns the transport write; the close test only needs its EOF boundary. + } + child.stdin.end() + })() + return (async function* () {})() + }) as typeof query + const connection = await openClaudeStreamJsonConnection(launch, {}, () => child, queryImpl) + + const closing = connection.close() + await new Promise((resolve) => setImmediate(resolve)) + expect(child.stdin.writableEnded).toBe(false) + + refreshDone.resolve() + await expect(closing).resolves.toBe(true) + expect(child.stdin.writableEnded).toBe(true) + }) + + it('requests a fresh close boundary after an output capture starts', async () => { + mocks.refresh.mockReset() + mocks.proveClaudeChildExit.mockReset() + const outputCapture = Promise.withResolvers() + const closeCapture = Promise.withResolvers() + mocks.refresh + .mockReturnValueOnce(outputCapture.promise) + .mockReturnValueOnce(closeCapture.promise) + mocks.proveClaudeChildExit.mockResolvedValueOnce(true) + const child = fakeChild() + const launch: ClaudeStreamJsonLaunch = { + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo' + } + const queryImpl = ((params: Parameters[0]) => { + params.options?.spawnClaudeCodeProcess?.({ + command: 'claude', + args: [], + env: {}, + signal: new AbortController().signal + }) + void (async () => { + for await (const _message of params.prompt) { + // The SDK owns the transport write; the close test only needs its EOF boundary. + } + child.stdin.end() + })() + return (async function* () {})() + }) as typeof query + const connection = await openClaudeStreamJsonConnection(launch, {}, () => child, queryImpl) + + child.stderr.emit('data', 'output') + await vi.waitFor(() => expect(mocks.refresh).toHaveBeenCalledTimes(1)) + const closing = connection.close() + await Promise.resolve() + + expect(mocks.refresh).toHaveBeenCalledTimes(2) + expect(child.stdin.writableEnded).toBe(false) + + outputCapture.resolve() + await Promise.resolve() + expect(child.stdin.writableEnded).toBe(false) + closeCapture.resolve() + await expect(closing).resolves.toBe(true) + expect(child.stdin.writableEnded).toBe(true) + }) +}) diff --git a/src/main/claude/claude-stream-json-connection.test.ts b/src/main/claude/claude-stream-json-connection.test.ts new file mode 100644 index 00000000000..c4f1f9a6fca --- /dev/null +++ b/src/main/claude/claude-stream-json-connection.test.ts @@ -0,0 +1,768 @@ +import { execFileSync } from 'node:child_process' +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { spawnProcess, type SpawnedProcess } from '../../shared/child-process/run-process' +import { hasLiveClaudePtys } from '../claude-accounts/live-pty-gate' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import { query, type CanUseTool, type Options } from '@anthropic-ai/claude-agent-sdk' +import { + openClaudeStreamJsonConnection, + type ClaudeStreamJsonConnection, + type ClaudeStreamJsonLaunch +} from './claude-stream-json-connection' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createDeferredStructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { claudeAuthDiagnostic } from './claude-structured-init-proof' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import { readClaudeStructuredSessionOptions } from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' + +// These drive the real SDK against the scripted fake CLI, so every assertion is +// about the environment, argv and frames a real child actually saw. +const FAKE_CLI = join(__dirname, '__fixtures__', 'claude-agent-sdk-scripted-cli.mjs') +const SESSION_ID = '5348c19f-6a54-4c2e-9c68-9c2b1a3d4e5f' +const HOLD_OPEN = { delayMs: 10_000 } + +type ScriptedCliReport = { + argv: string[] + controlRequests: { request_id: string; request: { subtype: string } }[] + controlResponses: { response: { request_id: string; response?: unknown } }[] + userMessages: Record[] + descendantPid: number | null +} + +const scratchDirs: string[] = [] +const openConnections: ClaudeStreamJsonConnection[] = [] + +afterEach(async () => { + for (const connection of openConnections.splice(0)) { + await connection.close() + } + for (const dir of scratchDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } + spawned.splice(0) + spawnedChildren.splice(0) + vi.unstubAllEnvs() +}) + +function scriptScenario( + steps: Record[], + controlResponses: Record = {} +) { + const dir = mkdtempSync(join(tmpdir(), 'claude-sdk-connection-')) + scratchDirs.push(dir) + const scenarioPath = join(dir, 'scenario.json') + const reportPath = join(dir, 'report.json') + writeFileSync(scenarioPath, JSON.stringify({ steps, controlResponses })) + return { + cwd: dir, + env: { + PATH: process.env.PATH ?? '', + ORCA_SDK_CONTRACT_SCENARIO_PATH: scenarioPath, + ORCA_SDK_CONTRACT_REPORT_PATH: reportPath + }, + readReport: () => JSON.parse(readFileSync(reportPath, 'utf8')) as ScriptedCliReport + } +} + +function launchFor( + scenario: { cwd: string; env: Record }, + env: Record = {} +): ClaudeStreamJsonLaunch { + return { + pathToClaudeCodeExecutable: FAKE_CLI, + options: { ...CLAUDE_STRUCTURED_BASE_OPTIONS, sessionId: SESSION_ID }, + cwd: scenario.cwd, + env: { ...scenario.env, ...env } + } +} + +/** The derived child environment, captured where Orca actually hands it to the OS. */ +const spawned: ProcessSpec[] = [] +/** The retained child, so a test can end it the way a crashing CLI would. */ +const spawnedChildren: SpawnedProcess[] = [] + +async function open( + launch: ClaudeStreamJsonLaunch, + handlers: Parameters[1] = {}, + queryImpl?: typeof query +): Promise { + const connection = await openClaudeStreamJsonConnection( + launch, + handlers, + (spec) => { + spawned.push(spec) + const child = spawnProcess(spec) + spawnedChildren.push(child) + return child + }, + queryImpl + ) + openConnections.push(connection) + return connection +} + +function childEnv(): Record { + return (spawned.at(-1)?.env ?? {}) as Record +} + +async function until(read: () => T | null | undefined, label: string): Promise { + for (let attempt = 0; attempt < 400; attempt++) { + const value = read() + if (value !== null && value !== undefined) { + return value + } + await new Promise((resolve) => setTimeout(resolve, 25)) + } + throw new Error(`timed out waiting for ${label}`) +} + +function readReportSafely(scenario: { readReport: () => ScriptedCliReport }) { + try { + return scenario.readReport() + } catch { + return null + } +} + +function processState(pid: number): 'running' | 'exited' { + try { + const state = execFileSync('ps', ['-o', 'state=', '-p', String(pid)], { + encoding: 'utf8', + env: { ...process.env, LANG: 'C', LC_ALL: 'C' } + }).trim() + return state.startsWith('Z') ? 'exited' : 'running' + } catch (error) { + if ((error as { status?: number }).status === 1) { + return 'exited' + } + throw error + } +} + +describe('Claude stream-json connection', () => { + it('passes the Claude Code system-prompt preset through to SDK query', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + let captured: Options | undefined + await open(launchFor(scenario), {}, (params) => { + captured = params.options + return query(params) + }) + + expect(captured?.systemPrompt).toEqual({ type: 'preset', preset: 'claude_code' }) + }) + + it('hands the child a derived environment, the resolved CLI path, and keeps the pid', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + vi.stubEnv('CLAUDE_CODE_CHILD_SESSION', '1') + vi.stubEnv('NODE_OPTIONS', '--require=/tmp/inject.js') + // An inherited value wins over the SDK's default, so clear it to pin the default. + vi.stubEnv('CLAUDE_CODE_ENTRYPOINT', undefined) + vi.stubEnv('ORCA_CONNECTION_MARKER', 'inherited') + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open( + launchFor(scenario, { + CLAUDE_CONFIG_DIR: '/accounts/managed/home', + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ORCA_AGENT_SESSION_SPAWN_TOKEN: 'spawn-9', + CLAUDE_CODE_CHILD_SESSION: 'configured-child-session', + CLAUDE_CODE_SESSION_ID: 'configured-session', + CLAUDE_CODE_BRIDGE_SESSION_ID: 'configured-bridge-session' + }) + ) + + // Ownership proof: the pid is a real live process, not a value the SDK reported. + expect(connection.pid).toEqual(expect.any(Number)) + expect(() => process.kill(connection.pid as number, 0)).not.toThrow() + const env = childEnv() + // The managed home is pinned verbatim: the CLI keys credential lookup on the literal string. + expect(env.CLAUDE_CONFIG_DIR).toBe('/accounts/managed/home') + expect(env.ANTHROPIC_AUTH_TOKEN).toBe('configured-token') + expect(env.ORCA_AGENT_SESSION_SPAWN_TOKEN).toBe('spawn-9') + expect(env.ORCA_CONNECTION_MARKER).toBe('inherited') + expect(env.ANTHROPIC_API_KEY).toBeUndefined() + expect(env.CLAUDE_CODE_CHILD_SESSION).toBeUndefined() + expect(env.CLAUDE_CODE_SESSION_ID).toBeUndefined() + expect(env.CLAUDE_CODE_BRIDGE_SESSION_ID).toBeUndefined() + // Two SDK mutations of the child env, pinned so a bump cannot change them unseen. + expect(env.CLAUDE_CODE_ENTRYPOINT).toBe('sdk-ts') + expect(env.NODE_OPTIONS).toBeUndefined() + // The bundled binary is excluded from the install, so the resolved path is mandatory. + const report = await until(() => readReportSafely(scenario), 'the scripted CLI report') + expect(report.argv[0]).toBe(FAKE_CLI) + // The .mjs fixture makes the SDK run it under node; a real CLI path is the program + // itself. Either way the resolved path is what Orca's spawner is asked to execute. + expect([spawned.at(-1)?.program, ...(spawned.at(-1)?.args ?? [])]).toContain(FAKE_CLI) + expect(report.argv).toContain('--replay-user-messages') + expect(report.argv).toContain(`--session-id=${SESSION_ID}`) + }) + + it('leaves the default CLI home unpinned so macOS Keychain OAuth keeps working', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + await open(launchFor(scenario)) + + await until(() => readReportSafely(scenario), 'the scripted CLI report') + expect(childEnv().CLAUDE_CONFIG_DIR).toBeUndefined() + }) + + it('settles a send only once the frame reached the child, and replays reach onMessage', async () => { + const replay = { + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + isReplay: true, + session_id: SESSION_ID, + uuid: 'uuid-replay-1' + } + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: replay }, HOLD_OPEN]) + const messages: Record[] = [] + const connection = await open(launchFor(scenario), { + onMessage: (message) => messages.push(message) + }) + + await connection.send({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + }) + // The report exists from the child's first line of work, so poll for the frame + // itself: `send` settles on the SDK's completed write, and the child still has + // to read that line before it can record it. + const report = await until( + () => (readReportSafely(scenario)?.userMessages.length ? readReportSafely(scenario) : null), + 'the user frame recorded by the child' + ) + expect(report.userMessages).toHaveLength(1) + + await until(() => messages.find((message) => message.uuid === 'uuid-replay-1'), 'the replay') + // The replay is delivered verbatim, so the dispatch acknowledgement still binds on it. + expect(messages.find((message) => message.uuid === 'uuid-replay-1')).toEqual(replay) + }) + + it('rejects a send the SDK pulled but could not write to a terminated child', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, HOLD_OPEN]) + const connection = await open(launchFor(scenario)) + const child = spawnedChildren.at(-1) + + // Same tick as the send, so the liveness guard still passes and the frame + // reaches the SDK's input pump: its `transport.write` is what fails, which is + // the window a child crashing mid-send actually opens. + child?.kill('SIGKILL') + const sent = connection.send({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + }) + + await expect(sent).rejects.toThrow() + expect(readReportSafely(scenario)?.userMessages ?? []).toHaveLength(0) + }) + + it('delivers an unmodeled frame verbatim so the provider-fallback row survives', async () => { + const unknown = { + type: 'frame_kind_from_the_future', + session_id: SESSION_ID, + uuid: 'uuid-unknown-1', + payload: { nested: { flags: ['a', 'b'] } } + } + const scenario = scriptScenario([{ emit: unknown }, HOLD_OPEN]) + const messages: Record[] = [] + await open(launchFor(scenario), { onMessage: (message) => messages.push(message) }) + + await until(() => messages.find((message) => message.uuid === 'uuid-unknown-1'), 'the frame') + expect(messages.find((message) => message.uuid === 'uuid-unknown-1')).toEqual(unknown) + }) + + it('commits the real partial-message cadence as one assistant item through the translator', async () => { + // The frame order and per-frame uuids are the ones Claude Code 2.1.258 emits + // under --include-partial-messages: every stream_event and the block's final + // assistant frame each carry their own uuid; only message.id ties them. + const stream = (uuid: string, event: Record) => ({ + type: 'stream_event', + uuid, + session_id: SESSION_ID, + parent_tool_use_id: null, + event + }) + const frames = [ + stream('uuid-message-start', { + type: 'message_start', + message: { id: 'msg_01', role: 'assistant', content: [] } + }), + stream('uuid-block-start', { + type: 'content_block_start', + index: 0, + content_block: { type: 'text', text: '' } + }), + stream('uuid-delta-1', { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: 'ST' } + }), + stream('uuid-delta-2', { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: 'REAMOK_ELEC_64E632' } + }), + { + type: 'assistant', + uuid: 'uuid-assistant-final', + session_id: SESSION_ID, + parent_tool_use_id: null, + message: { + id: 'msg_01', + role: 'assistant', + content: [{ type: 'text', text: 'STREAMOK_ELEC_64E632' }], + stop_reason: null + } + }, + stream('uuid-block-stop', { type: 'content_block_stop', index: 0 }), + stream('uuid-message-delta', { type: 'message_delta', delta: { stop_reason: 'end_turn' } }), + stream('uuid-message-stop', { type: 'message_stop' }), + { + type: 'result', + subtype: 'success', + is_error: false, + duration_ms: 1, + duration_api_ms: 1, + num_turns: 1, + result: 'STREAMOK_ELEC_64E632', + stop_reason: 'end_turn', + session_id: SESSION_ID, + uuid: 'uuid-result' + } + ] + const scenario = scriptScenario([...frames.map((frame) => ({ emit: frame })), HOLD_OPEN]) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: SESSION_ID, leafUuid: 'leaf-1' } + }, + journalDir: join(scenario.cwd, 'journal'), + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ journal, fence: 1, publish: vi.fn() }) + const translator = createClaudeJournalTranslator({ sink: deferred.sink }) + let settled = false + await open(launchFor(scenario), { + onMessage: (message) => { + translator.handle({ type: 'message', sessionId: 'session-1', message }) + settled ||= message.type === 'result' + } + }) + + await until(() => (settled ? true : null), 'the result frame') + await deferred.drained() + const items = journal.snapshot().items + const assistant = items.filter( + (item) => item.body.kind === 'message' && item.body.role === 'assistant' + ) + expect(assistant.map((item) => item.body)).toEqual([ + { + kind: 'message', + role: 'assistant', + blocks: [{ type: 'text', text: 'STREAMOK_ELEC_64E632' }] + } + ]) + expect(assistant.map((item) => item.itemId)).toEqual([`claude:${SESSION_ID}:uuid-block-start`]) + expect( + items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) + ).toEqual([]) + // The journal owns a SQLite connection now; afterEach removes this temp root and an open + // handle blocks that on Windows. + await journal.close() + }) + + it('feeds an inbound permission request to canUseTool and writes its answer back on the same id', async () => { + const scenario = scriptScenario([ + { + emit: { + type: 'control_request', + request_id: 'perm-421', + request: { + subtype: 'can_use_tool', + tool_name: 'Bash', + input: { command: 'ls' }, + tool_use_id: 'toolu_1', + permission_suggestions: [{ type: 'addRules' }] + } + } + }, + { awaitControlResponse: 'perm-421' }, + HOLD_OPEN + ]) + const seen: { toolName: string; requestId: string; toolUseID: string; suggestions: unknown }[] = + [] + const canUseTool: CanUseTool = (toolName, _input, options) => { + seen.push({ + toolName, + requestId: options.requestId, + toolUseID: options.toolUseID, + suggestions: options.suggestions + }) + return Promise.resolve({ behavior: 'deny', message: 'No', toolUseID: options.toolUseID }) + } + await open(launchFor(scenario), { canUseTool }) + + await until(() => (seen.length > 0 ? seen : null), 'the inbound permission request') + expect(seen).toEqual([ + { + toolName: 'Bash', + requestId: 'perm-421', + toolUseID: 'toolu_1', + suggestions: [{ type: 'addRules' }] + } + ]) + const written = await until( + () => + readReportSafely(scenario)?.controlResponses.find( + (frame) => frame.response.request_id === 'perm-421' + ), + 'the permission answer' + ) + expect(written.response.response).toMatchObject({ behavior: 'deny', message: 'No' }) + }) + + it('drives Orca control methods onto the SDK and times out with the init proof message', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { models: [{ value: 'sonnet' }], account: { tokenSource: 'oauth' } }, + get_settings: { env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } } + }) + const connection = await open(launchFor(scenario)) + + await expect(connection.initializationResult()).resolves.toMatchObject({ + models: [{ value: 'sonnet' }] + }) + await expect(connection.getSettings()).resolves.toEqual({ + env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } + }) + await expect(connection.setModel('opus')).resolves.toBeUndefined() + const requests = await until( + () => + readReportSafely(scenario)?.controlRequests.find( + (frame) => frame.request.subtype === 'set_model' + ), + 'the set_model control request' + ) + expect(requests.request.subtype).toBe('set_model') + }) + + it('reads supportedModels from the catalog the running CLI reported', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { + models: [ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { + value: 'opus', + displayName: 'Opus 5', + description: 'The live row, not the seed', + resolvedModel: 'claude-opus-5', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + } + ] + } + }) + const connection = await open(launchFor(scenario)) + + await expect(connection.supportedModels()).resolves.toMatchObject([ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { value: 'opus', displayName: 'Opus 5', supportedEffortLevels: ['low', 'high'] } + ]) + }) + + it('serves the picker the live catalog rather than falling back to the static seed', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { + models: [ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { + value: 'opus', + displayName: 'Opus 5', + description: 'The live row, not the seed', + resolvedModel: 'claude-opus-5', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + } + ] + } + }) + const connection = await open(launchFor(scenario)) + const session = { + connection, + options: new Map(), + reportedOptions: {} + } as unknown as ClaudeSession + + const options = await readClaudeStructuredSessionOptions(session, 5_000) + + // The seed carries neither this description nor a two-level effort list, so + // both can only have come from the child. + expect(options.models).toContainEqual({ + id: 'opus', + label: 'Opus 5', + description: 'The live row, not the seed', + isDefault: true, + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }) + expect(options.current.model).toBe('opus') + }) + + it('feeds the auth diagnostic from the settings the running child reports', async () => { + for (const key of ['ANTHROPIC_BASE_URL', 'ANTHROPIC_AUTH_TOKEN', 'ANTHROPIC_API_KEY']) { + vi.stubEnv(key, undefined) + } + const scenario = scriptScenario([HOLD_OPEN], { + get_settings: { + env: { + ANTHROPIC_BASE_URL: 'https://settings.example.test', + ANTHROPIC_AUTH_TOKEN: 'secret' + } + } + }) + const connection = await open(launchFor(scenario)) + const init = { providerSessionId: SESSION_ID, uuid: null, model: null, message: {} } + + // With no ambient auth, every true below can only have come from the CLI's settings. + expect(claudeAuthDiagnostic(init, null)).toMatchObject({ + baseUrlConfigured: false, + authTokenConfigured: false + }) + const diagnostic = claudeAuthDiagnostic(init, await connection.getSettings()) + expect(diagnostic).toMatchObject({ + baseUrlConfigured: true, + authTokenConfigured: true, + apiKeyConfigured: false + }) + expect(JSON.stringify(diagnostic)).not.toContain('secret') + }) + + it('reports an unauthenticated start through the init deadline instead of hanging', async () => { + // The scripted CLI never answers, which is the shape of a silently unauthenticated CLI. + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open({ + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS: '1' } + }) + + await expect(connection.initializationResult({ timeoutMs: 200 })).rejects.toThrow( + 'claude initialize request timed out' + ) + }) + + it('reports a self-exit with its status and stderr, and leaves its tree unverifiable', async () => { + const scenario = scriptScenario([{ stderr: 'claude: not signed in\n' }, { exit: 1 }]) + let exit: Error | null = null + const connection = await open(launchFor(scenario), { + onExit: (error) => { + exit = error + } + }) + + await until(() => exit, 'the exit error') + // The status and stderr are the only diagnostic a refused start leaves behind. + expect((exit as unknown as Error).message).toMatch(/exited \(code 1\): claude: not signed in/) + expect(connection.closed).toBe(true) + // The root's exit is first-hand, but it left before a descendant snapshot + // could be armed, so close() has no tree proof to offer and says so. + await expect(connection.close()).resolves.toBe(false) + expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'unverifiable' }) + }) + + it.runIf(process.platform !== 'win32')( + 'proves a natural SDK exit and cleans up its descendant before recovery', + async () => { + const scenario = scriptScenario([ + { stderr: 'claude: natural exit\n' }, + { delayMs: 500 }, + { exit: 1 } + ]) + let exit: Error | null = null + const connection = await open( + { + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_DESCENDANT: '1' } + }, + { onExit: (error) => (exit = error) } + ) + const report = await until(() => { + const current = readReportSafely(scenario) + return current?.descendantPid ? current : null + }, 'the descendant report') + await until(() => exit, 'the natural exit error') + try { + await expect(connection.close()).resolves.toBe(true) + expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'exited' }) + expect(processState(report.descendantPid as number)).toBe('exited') + } finally { + try { + process.kill(report.descendantPid as number, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it('settles a spawn error followed by close as processless and closes idempotently', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + const missingCli = join(scenario.cwd, 'claude-that-does-not-exist') + let fault: Error | null = null + let exit: Error | null = null + const connection = await open( + { ...launchFor(scenario), pathToClaudeCodeExecutable: missingCli }, + { + onFault: (error) => { + fault = error + }, + onExit: (error) => { + exit = error + } + } + ) + + await until( + () => (connection.exitVerdict.root === 'processless' ? connection.exitVerdict : null), + 'the processless spawn settlement' + ) + expect(connection.pid).toBeUndefined() + expect(fault).toBeInstanceOf(Error) + expect(exit).toBeNull() + await expect(Promise.all([connection.close(), connection.close()])).resolves.toEqual([ + true, + true + ]) + await expect(connection.close()).resolves.toBe(true) + expect(connection.exitVerdict).toEqual({ root: 'processless', tree: 'exited' }) + }) + + it('does not treat a child error event as first-hand root exit proof', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + let exit: Error | null = null + const connection = await open(launchFor(scenario), { + onExit: (error) => { + exit = error + } + }) + const child = spawnedChildren.at(-1) + expect(child).toBeDefined() + + child?.emit('error', new Error('child transport fault')) + + expect(exit).toBeNull() + expect(connection.exitVerdict.root).toBe('live') + await until(() => exit, 'the distinct child exit') + expect(connection.exitVerdict.root).toBe('exited') + }) + + it('proves the exit of a child that ignores a graceful shutdown', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open({ + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_IGNORE_SIGTERM: '1' } + }) + + // Keep the lstart capture boundary outside the child's displayed start second. + await new Promise((resolve) => setTimeout(resolve, 1_100)) + await expect(connection.close()).resolves.toBe(true) + }, 20_000) +}) + +// A structured Claude child owns the account's credentials while it runs, exactly as +// a Claude PTY does. The gate is what makes runtime-auth-sync defer the managed OAuth +// refresh instead of rotating the single-use token out from under a live session, and +// structured sessions used to be invisible to it. +describe('the managed-auth live gate', () => { + it('holds while a structured child runs and releases when it ends', async () => { + // The gate is a process-wide singleton and a sibling test's release lands on its + // child's 'close' event, which can settle after that test's close() resolved. + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + const connection = await open(launchFor(scenario)) + + expect(hasLiveClaudePtys()).toBe(true) + + await connection.close() + + await until(() => (hasLiveClaudePtys() ? null : true), 'the auth gate to drain') + expect(hasLiveClaudePtys()).toBe(false) + }, 30_000) + + it('releases when the child dies on its own rather than through close()', async () => { + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + await open(launchFor(scenario)) + expect(hasLiveClaudePtys()).toBe(true) + + spawnedChildren.at(-1)?.kill('SIGKILL') + + await until(() => (hasLiveClaudePtys() ? null : true), 'the auth gate to drain') + expect(hasLiveClaudePtys()).toBe(false) + }, 30_000) + + // The gate entry is deliberately unpersisted, so confirmSeededClaudeLivePtys can never + // reconcile a stray one: a leak here defers the managed OAuth refresh for the life of + // the process. Entering the gate only after the release handlers are attached makes + // that unreachable regardless of what the setup in between does. + it('leaks no gate entry when setup throws between spawn and handler attachment', async () => { + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + let started: SpawnedProcess | null = null + + try { + await expect( + openClaudeStreamJsonConnection(launchFor(scenario), {}, (spec) => { + const child = spawnProcess(spec) + started = child + const attach = child.stderr.on.bind(child.stderr) + // Measured attach order: the SDK binds stderr 'data' from inside query(), + // before the child is even assigned. The SECOND bind is this connection's own + // armTreeOnOutput — the first statement that runs after the child exists and + // before its 'exit'/'close' release handlers. Throwing on the first is + // vacuous: it escapes before any gate entry could have happened. + let dataAttaches = 0 + child.stderr.on = ((event: string, listener: (...args: unknown[]) => void) => { + if (event === 'data') { + dataAttaches += 1 + if (dataAttaches === 2) { + throw new Error('stderr listener attach failed') + } + } + return attach(event, listener) + }) as typeof child.stderr.on + return child + }) + ).rejects.toThrow('stderr listener attach failed') + + expect(hasLiveClaudePtys()).toBe(false) + } finally { + ;(started as SpawnedProcess | null)?.kill('SIGKILL') + } + }, 30_000) +}) diff --git a/src/main/claude/claude-stream-json-connection.ts b/src/main/claude/claude-stream-json-connection.ts new file mode 100644 index 00000000000..dd6bbc8a5eb --- /dev/null +++ b/src/main/claude/claude-stream-json-connection.ts @@ -0,0 +1,283 @@ +import { randomUUID } from 'node:crypto' +import type * as ClaudeAgentSdk from '@anthropic-ai/claude-agent-sdk' +import type { CanUseTool, OnUserDialog, SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' +import { spawnProcess } from '../../shared/child-process/run-process' +import { + markClaudeStructuredChildExited, + markClaudeStructuredChildSpawned +} from '../claude-accounts/live-pty-gate' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' +import { + ClaudeControlRequestError, + createClaudeControlSurface, + type ClaudeControlSurface +} from './claude-agent-sdk-control-requests' +import { createClaudeChildTreeReaper, proveClaudeChildExit } from './claude-agent-sdk-exit-proof' +import type { DescendantTreeVerdict } from '../pty-descendant-exit-verification' +import { createClaudeCodeProcessSpawn } from './claude-agent-sdk-process-spawn' +import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' +import type { ClaudeStructuredSdkOptions } from './claude-structured-launch-resolution' + +export { ClaudeControlRequestError } + +/** + * The SDK is loaded at the structured-Claude boundary rather than by this module's + * import. The ordinary runtime's class graph statically reaches this file, and the + * SDK sets `process.env.NoDefaultCurrentDirectoryInExePath` at import time — a + * Windows executable-search change that a user who never leaves the terminal/TUI + * path never opted into, and a missing SDK would fail runtime startup. Memoized, + * so a session pays the import once per process rather than once per connection. + */ +let claudeAgentSdk: Promise | null = null + +function loadClaudeAgentSdk(): Promise { + claudeAgentSdk ??= import('@anthropic-ai/claude-agent-sdk') + return claudeAgentSdk +} + +export type ClaudeStreamJsonLaunch = { + /** Orca's resolved user CLI; the SDK falls back to a bundled binary that is not installed. */ + pathToClaudeCodeExecutable: string + options: ClaudeStructuredSdkOptions + cwd: string + env?: Record +} + +export type ClaudeStreamJsonConnectionHandlers = { + onMessage?: (message: Record) => void + /** + * The SDK owns inbound permission control: it hands `can_use_tool` to this callback with + * a stable requestId and an abort signal, dedups duplicate delivery, and matches the + * response by request_id itself. Setting it makes the SDK pass `--permission-prompt-tool + * stdio` automatically; it must not be paired with `permissionPromptToolName`. + */ + canUseTool?: CanUseTool + /** `request_user_dialog` control; the CLI only emits kinds declared in `supportedDialogKinds`. */ + onUserDialog?: OnUserDialog + /** A transport/process fault that is not itself first-hand root exit proof. */ + onFault?: (error: Error) => void + onExit?: (error: Error) => void +} + +/** + * Two questions with their own evidence. The root's verdict is first-hand: Orca's + * own child handle reported exit, or reported error then close before it ever had + * a pid. The tree's comes from bounded descendant verification, and `unverifiable` + * is never collapsed into either neighbour. + */ +export type ClaudeChildExitVerdict = { + root: 'exited' | 'live' | 'processless' + tree: DescendantTreeVerdict +} + +export type ClaudeStreamJsonConnection = ClaudeControlSurface & { + readonly pid: number | undefined + readonly closed: boolean + /** What the ladder has observed so far; read after a `close()` that returned false. */ + readonly exitVerdict: ClaudeChildExitVerdict + send: (message: Record) => Promise + /** Resolves true after processless settlement, or root exit plus observed tree exit. */ + close: () => Promise +} + +type ExitStatus = { code: number | null; signal: NodeJS.Signals | null } + +function exitError(stderrTail: string, status: ExitStatus | null, cause?: Error): Error { + const detail = stderrTail.trim() + // The status is the diagnostic a signed-out or refused start leaves behind; + // it has to survive every wrapper between here and the user. + const how = + status?.signal !== null && status?.signal !== undefined + ? ` (signal ${status.signal})` + : status?.code !== null && status?.code !== undefined + ? ` (code ${status.code})` + : '' + const message = `claude stream-json exited${how}${detail ? `: ${detail}` : ''}` + return cause ? new Error(message, { cause }) : new Error(message) +} + +export async function openClaudeStreamJsonConnection( + launch: ClaudeStreamJsonLaunch, + handlers: ClaudeStreamJsonConnectionHandlers = {}, + spawnImpl: typeof spawnProcess = spawnProcess, + queryImpl?: typeof ClaudeAgentSdk.query +): Promise { + const { query } = await loadClaudeAgentSdk() + const spawner = createClaudeCodeProcessSpawn(spawnImpl) + const inbox = createClaudeUserMessageQueue() + const session = (queryImpl ?? query)({ + prompt: inbox.messages, + options: { + ...launch.options, + cwd: launch.cwd, + // Why env is never omitted: the SDK inherits process.env when it is, which is + // exactly the ambient ANTHROPIC_* auth leak this lane already shipped once. + env: buildClaudeChildProcessEnv(launch.env, { scrubConfiguredChildSessionStamps: true }), + pathToClaudeCodeExecutable: launch.pathToClaudeCodeExecutable, + spawnClaudeCodeProcess: spawner.spawn, + ...(handlers.canUseTool ? { canUseTool: handlers.canUseTool } : {}), + ...(handlers.onUserDialog ? { onUserDialog: handlers.onUserDialog } : {}) + } + }) + const child = spawner.child + if (!child) { + throw new Error('the claude agent SDK returned without spawning a child') + } + // This child owns the account's credentials for as long as it runs, exactly as a + // Claude PTY does — hold the OAuth-refresh gate so a managed refresh cannot rotate + // the single-use token out from under it mid-turn. Entered below, once a release + // path exists. + const authGateKey = randomUUID() + const releaseAuthGate = (): void => markClaudeStructuredChildExited(authGateKey) + let exited = false + let exitStatus: ExitStatus | null = null + let closing = false + let processless = false + let prePidSpawnError = false + let terminalError: Error | null = null + let faultReported = false + let exitReported = false + let closePromise: Promise | null = null + // One reaper per child: every close attempt and error-path reap shares its proof. + const rootSettled = (): boolean => exited || processless + const tree = createClaudeChildTreeReaper(child, { exited: rootSettled }) + + // Arm lazily on actual child output instead of issuing a process-table scan for + // every session at startup. A natural SDK exit can race a later close, while + // output-triggered observation still catches the usual live-child window. + let outputObservationArmed = false + const armTreeOnOutput = (): void => { + if (outputObservationArmed) { + return + } + outputObservationArmed = true + void (tree.refresh?.() ?? tree.capture()) + } + child.stderr.on('data', armTreeOnOutput) + // The SDK may synchronously spawn the CLI and consume an early stderr chunk + // before this connection can attach its listener; the bounded tail preserves + // that observation for the same lazy arm. + if (spawner.stderrTail.length > 0) { + armTreeOnOutput() + } + + let settleExit = (): void => {} + const exitPromise = new Promise((resolve) => { + settleExit = resolve + }) + const markExited = (): void => { + exited = true + releaseAuthGate() + settleExit() + } + child.on('exit', (code, signal) => { + exitStatus = { code, signal } + markExited() + handleUnexpectedEnd() + }) + + const handleUnexpectedEnd = (cause?: Error): void => { + terminalError ??= exitError(spawner.stderrTail, exitStatus, cause) + inbox.fail(terminalError) + if (!closing && !faultReported) { + faultReported = true + handlers.onFault?.(terminalError) + } + if (!closing && exited && !exitReported) { + exitReported = true + handlers.onExit?.(terminalError) + } + } + + void (async () => { + for await (const message of session) { + handlers.onMessage?.(message as unknown as Record) + } + })().catch((error: unknown) => { + // The SDK ends its generator in error when the child dies or the transport + // fails; a transport failure with a live child still has to reap the tree. + if (!closing && !exited) { + void tree.reap() + } + handleUnexpectedEnd(error instanceof Error ? error : new Error(String(error))) + }) + + child.on('error', (error) => { + if (spawner.pid === undefined) { + prePidSpawnError = true + } + if (!closing && !exited) { + void tree.reap() + } + handleUnexpectedEnd(error) + }) + child.on('close', () => { + // Covers the spawn-failure path too, where no 'exit' ever arrives. + releaseAuthGate() + if (prePidSpawnError && spawner.pid === undefined) { + processless = true + settleExit() + } + handleUnexpectedEnd() + }) + child.stdin.on('error', (error) => { + if (!closing) { + void tree.reap() + handleUnexpectedEnd(error) + } + }) + // Why here and not at spawn: a structured gate entry is deliberately unpersisted, so + // confirmSeededClaudeLivePtys can never reconcile a stray one and a leak defers the + // managed OAuth refresh for the life of the process. Entering only after 'exit' and + // 'close' are attached makes that unreachable — any later throw still leaves a + // listener that releases. Nothing between spawn and here can yield, so the child + // cannot end before the gate is entered. + markClaudeStructuredChildSpawned(authGateKey) + + const send = (message: Record): Promise => { + if (closing || exited || terminalError || child.stdin.destroyed || !child.stdin.writable) { + return Promise.reject(terminalError ?? new Error('claude stream-json connection is closed')) + } + return inbox.push(message as unknown as SDKUserMessage) + } + + const close = (): Promise => { + closePromise ??= (async () => { + closing = true + // Arm the descendant proof before ending stdin. The SDK may exit the root + // immediately; a post-exit walk cannot recover descendants that reparented. + await (tree.refresh?.() ?? tree.capture()) + inbox.end() + const proven = await proveClaudeChildExit({ + child, + exitPromise, + exited: rootSettled, + tree + }) + inbox.fail(new Error('claude stream-json connection closed')) + if (!proven) { + closePromise = null + } + return proven + })() + return closePromise + } + + return { + ...createClaudeControlSurface(session), + get pid() { + return spawner.pid + }, + get closed() { + return closing || exited || terminalError !== null + }, + get exitVerdict() { + return { + root: processless ? 'processless' : exited ? 'exited' : 'live', + tree: tree.treeVerdict + } as const + }, + send, + close + } +} diff --git a/src/main/claude/claude-streamed-block-identity.ts b/src/main/claude/claude-streamed-block-identity.ts new file mode 100644 index 00000000000..5cbf6674159 --- /dev/null +++ b/src/main/claude/claude-streamed-block-identity.ts @@ -0,0 +1,110 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { claudeRecord, claudeText } from './claude-structured-item-translation' + +// Under --include-partial-messages every stream_event frame carries its own +// uuid, and the block's final `assistant` frame carries yet another; only +// `message.id` ties them together. The block's first stream frame mints the +// journal identity, and the final frame lands on it in block order instead of +// appending a duplicate under its own uuid. + +export type ClaudeStreamedTextDelta = { identity: AgentJournalItemIdentity; text: string } + +type StreamedMessage = { + messageId: string | null + blocks: Map + /** Streamed text blocks whose final assistant frame has not arrived, in block order. */ + awaitingFinal: AgentJournalItemIdentity[] +} + +export type ClaudeStreamedBlockRegistry = { + /** Text a stream_event frame appends to its block, or null when it carries none. */ + observe: (frame: Record) => ClaudeStreamedTextDelta | null + /** The streamed identity a final assistant frame reconciles onto, if its block streamed. */ + reconcile: (frame: { + sessionId: string + parentToolUseId: string | null + messageId: string | null + }) => AgentJournalItemIdentity | null + clear: () => void +} + +function scopeKey(sessionId: string, parentToolUseId: string | null): string { + return `${sessionId}/${parentToolUseId ?? ''}` +} + +export function createClaudeStreamedBlockRegistry(): ClaudeStreamedBlockRegistry { + const messages = new Map() + + const messageFor = (scope: string): StreamedMessage => { + let streamed = messages.get(scope) + if (!streamed) { + streamed = { messageId: null, blocks: new Map(), awaitingFinal: [] } + messages.set(scope, streamed) + } + return streamed + } + + const mint = ( + streamed: StreamedMessage, + sessionId: string, + index: number, + uuid: string + ): AgentJournalItemIdentity => { + const identity: AgentJournalItemIdentity = { provider: 'claude', sessionId, uuid } + streamed.blocks.set(index, identity) + streamed.awaitingFinal.push(identity) + return identity + } + + return { + observe: (frame) => { + const event = claudeRecord(frame.event) + const sessionId = claudeText(frame.session_id) + const uuid = claudeText(frame.uuid) + if (frame.type !== 'stream_event' || !event || !sessionId || !uuid) { + return null + } + const scope = scopeKey(sessionId, claudeText(frame.parent_tool_use_id)) + if (event.type === 'message_start') { + messages.set(scope, { + messageId: claudeText(claudeRecord(event.message)?.id), + blocks: new Map(), + awaitingFinal: [] + }) + return null + } + const index = typeof event.index === 'number' ? event.index : 0 + if (event.type === 'content_block_start') { + const block = claudeRecord(event.content_block) + if (block?.type !== 'text') { + return null + } + const identity = mint(messageFor(scope), sessionId, index, uuid) + const text = claudeText(block.text) + return text ? { identity, text } : null + } + if (event.type !== 'content_block_delta') { + return null + } + const delta = claudeRecord(event.delta) + const text = delta?.type === 'text_delta' ? claudeText(delta.text) : null + if (!text) { + return null + } + const streamed = messageFor(scope) + const identity = streamed.blocks.get(index) ?? mint(streamed, sessionId, index, uuid) + return { identity, text } + }, + reconcile: (frame) => { + const streamed = messages.get(scopeKey(frame.sessionId, frame.parentToolUseId)) + if ( + !streamed || + (frame.messageId && streamed.messageId && frame.messageId !== streamed.messageId) + ) { + return null + } + return streamed.awaitingFinal.shift() ?? null + }, + clear: () => messages.clear() + } +} diff --git a/src/main/claude/claude-streamed-text-checkpoints.test.ts b/src/main/claude/claude-streamed-text-checkpoints.test.ts new file mode 100644 index 00000000000..00a0bc0edd6 --- /dev/null +++ b/src/main/claude/claude-streamed-text-checkpoints.test.ts @@ -0,0 +1,93 @@ +import { describe, expect, it } from 'vitest' +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' + +function identityOf(uuid: string): AgentJournalItemIdentity { + return { provider: 'claude', sessionId: 'claude-session', uuid } +} + +function checkpoints() { + const rows: { uuid: string; text: string }[] = [] + let scheduled: (() => void) | null = null + const store = createClaudeStreamedTextCheckpoints({ + persist: (identity, text) => { + rows.push({ uuid: 'uuid' in identity ? identity.uuid : '', text }) + }, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + return { + store, + rows, + runWindow: () => { + const run = scheduled as (() => void) | null + run?.() + } + } +} + +describe('claude streamed text checkpoints', () => { + it('rewrites a block row with the full text accumulated so far', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'hel') + store.append(identityOf('block-1'), 'lo') + runWindow() + + expect(rows).toEqual([{ uuid: 'block-1', text: 'hello' }]) + expect(store.pending).toBe(1) + }) + + it('drops every block still awaiting its final frame at settlement', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'partial answer') + runWindow() + store.settle() + + expect(store.pending).toBe(0) + // The row written before settlement stays; nothing is rewritten afterwards. + store.flush() + expect(rows).toEqual([{ uuid: 'block-1', text: 'partial answer' }]) + }) + + it('keeps a block whose final frame arrived out of the settlement sweep', () => { + const { store } = checkpoints() + + store.append(identityOf('block-1'), 'one') + store.append(identityOf('block-2'), 'two') + store.forget('claude:claude-session:block-1') + + expect(store.pending).toBe(1) + store.settle() + expect(store.pending).toBe(0) + }) + + it('flushes text the widening checkpoint interval has not written yet', () => { + const { store, rows } = checkpoints() + + store.append(identityOf('block-1'), 'x') + store.flush() + + expect(rows).toEqual([{ uuid: 'block-1', text: 'x' }]) + // Already at the row's length: a second flush has nothing to write. + store.flush() + expect(rows).toHaveLength(1) + }) + + it('stops persisting once disposed', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'text') + store.dispose() + runWindow() + store.flush() + + expect(rows).toEqual([]) + expect(store.pending).toBe(0) + }) +}) diff --git a/src/main/claude/claude-streamed-text-checkpoints.ts b/src/main/claude/claude-streamed-text-checkpoints.ts new file mode 100644 index 00000000000..348ecd99558 --- /dev/null +++ b/src/main/claude/claude-streamed-text-checkpoints.ts @@ -0,0 +1,105 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { + createAgentSessionDeltaCoalescer, + type AgentSessionDeltaCoalescerDeps +} from '../native-chat/agent-session-wire/agent-session-delta-coalescer' + +export type ClaudeStreamedTextCheckpointDeps = { + /** Rewrites the block's journal row with the text accumulated so far. */ + persist: (identity: AgentJournalItemIdentity, text: string) => void + coalesceMs?: number + schedule?: AgentSessionDeltaCoalescerDeps['schedule'] +} + +export type ClaudeStreamedTextCheckpoints = { + /** Accumulate a delta; the row is rewritten on the coalescer's own cadence. */ + append: (identity: AgentJournalItemIdentity, text: string) => void + /** Write every block whose row is behind the text received for it. */ + flush: () => void + /** Drop one block's state, for a block whose final frame has now landed. */ + forget: (key: string) => void + /** + * Drop every block still awaiting its final frame, at turn settlement. Their + * text is already journaled by the flush that precedes settlement; keeping it + * live would grow with every interrupted turn for the life of the session. + */ + settle: () => void + /** Blocks still awaiting a final frame. A settled turn must leave none. */ + readonly pending: number + dispose: () => void +} + +/** + * Growth of a streamed block's row between its deltas and its final frame. + * + * The row is rewritten on a widening interval rather than per delta: a 200-line + * reply would otherwise rewrite the same journal row once per token. + */ +export function createClaudeStreamedTextCheckpoints( + deps: ClaudeStreamedTextCheckpointDeps +): ClaudeStreamedTextCheckpoints { + const identities = new Map() + const latestText = new Map() + const checkpointLengths = new Map() + + const persist = (key: string, text: string, force: boolean): void => { + latestText.set(key, text) + const checkpointLength = checkpointLengths.get(key) ?? 0 + const nextLength = Math.max(checkpointLength + 32, Math.ceil(checkpointLength * 1.125)) + if (!force && checkpointLength > 0 && text.length < nextLength) { + return + } + const identity = identities.get(key) + if (!identity) { + return + } + checkpointLengths.set(key, text.length) + deps.persist(identity, text) + } + + const coalescer = createAgentSessionDeltaCoalescer({ + ...(deps.coalesceMs === undefined ? {} : { windowMs: deps.coalesceMs }), + ...(deps.schedule ? { schedule: deps.schedule } : {}), + emit: (key, text) => persist(key, text, false) + }) + + const drop = (key: string): void => { + coalescer.forget(key) + identities.delete(key) + latestText.delete(key) + checkpointLengths.delete(key) + } + + return { + append: (identity, text) => { + const key = agentJournalItemKey(identity) + identities.set(key, identity) + coalescer.append(key, text) + }, + flush: () => { + coalescer.flushAll() + for (const [key, text] of latestText) { + if (checkpointLengths.get(key) !== text.length) { + persist(key, text, true) + } + } + }, + forget: drop, + settle: () => { + // Map iteration tolerates deletion of the entry just visited. + for (const key of identities.keys()) { + drop(key) + } + }, + get pending() { + return identities.size + }, + dispose: () => { + coalescer.dispose() + identities.clear() + latestText.clear() + checkpointLengths.clear() + } + } +} diff --git a/src/main/claude/claude-structured-acquisition-release.ts b/src/main/claude/claude-structured-acquisition-release.ts new file mode 100644 index 00000000000..1b633553e89 --- /dev/null +++ b/src/main/claude/claude-structured-acquisition-release.ts @@ -0,0 +1,43 @@ +import { + closeClaudeSession, + claudeAcquisitionCleanupError +} from './claude-structured-session-close' +import type { + ClaudeAcquisitionRegistry, + ClaudeSession, + ClaudeSessionExit, + ClaudeStructuredSessionAdapterDeps +} from './claude-structured-session-state' + +/** + * Cleanup for an acquisition the host could not commit or prove. A session that + * a first-hand exit already removed is not an absence to report as proven: the + * ladder on its connection still answers, and that answer is classified exactly + * as a start-time failure would be. + */ +export async function releaseClaudeAcquisition(input: { + sessionId: string + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + onExitProven?: (sessionId: string, exit: ClaudeSessionExit) => Promise + persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'] + onEvent?: ClaudeStructuredSessionAdapterDeps['onEvent'] +}): Promise { + const exit = input.exits.get(input.sessionId) + if (!exit || input.sessions.has(input.sessionId) || input.acquisitions.get(input.sessionId)) { + return closeClaudeSession(input) + } + const firstProof = exit.closePromise ? await exit.closePromise : false + // A failed exit-path proof is retained as evidence, not as a terminal result; + // a release retry must drive a fresh tree verification on the same connection. + const retriedProof = firstProof || (await exit.connection.close()) + if (retriedProof) { + await input.onExitProven?.(input.sessionId, exit) + // Keep the first-hand exit evidence indexed until the tree proof succeeds; + // a failed close must be retryable and cannot look like an absent session. + input.exits.delete(input.sessionId) + return true + } + throw claudeAcquisitionCleanupError(exit.connection, exit.error) +} diff --git a/src/main/claude/claude-structured-auth-parity.test.ts b/src/main/claude/claude-structured-auth-parity.test.ts new file mode 100644 index 00000000000..ddc69366aad --- /dev/null +++ b/src/main/claude/claude-structured-auth-parity.test.ts @@ -0,0 +1,235 @@ +import { afterEach, describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { beginClaudeAuthSwitch, endClaudeAuthSwitch } from '../claude-accounts/live-pty-gate' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE +} from '../claude-accounts/environment' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { createClaudeStructuredLaunchResolver } from './claude-structured-launch-resolution' +import { ClaudeStructuredSessionAdapter } from './claude-structured-session-adapter' +import { + PROVIDER_SESSION_ID, + adapterFor, + fakeClaude, + identityFor +} from './claude-structured-session-test-support' + +const SESSION_ID = 'orca-session-auth' +const IDENTITY = { sessionId: SESSION_ID } as Parameters< + ReturnType +>[0]['identity'] + +function record(): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [] + } as unknown as AgentSessionRecord +} + +function resolverFor(options: { + stripAuthEnv: boolean + overlay?: Record + authSwitchSettleTimeoutMs?: number +}): ReturnType { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record() } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: options.stripAuthEnv }), + authSwitchSettleTimeoutMs: options.authSwitchSettleTimeoutMs ?? 20, + ...(options.overlay ? { resolveEnv: () => options.overlay as Record } : {}) + }) +} + +/** + * An adapter driven by the REAL launch resolver, not the stub in the shared test + * support — the stub has no auth guard at all, so a teardown-window test built on it + * would pass whatever the guard did. + */ +function realResolverAdapter( + claude: ReturnType, + authSwitchSettleTimeoutMs: number +): ClaudeStructuredSessionAdapter { + const resumable = { + ...record(), + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null } } + ] + } as unknown as AgentSessionRecord + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: createClaudeStructuredLaunchResolver({ + store: { getRecord: () => resumable } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + authSwitchSettleTimeoutMs + }), + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + persistHandle: async () => {} + }) +} + +function withAmbientAuth(value: string, run: () => Promise): Promise { + const restore = process.env.ANTHROPIC_API_KEY + process.env.ANTHROPIC_API_KEY = value + return run().finally(() => { + if (restore === undefined) { + delete process.env.ANTHROPIC_API_KEY + } else { + process.env.ANTHROPIC_API_KEY = restore + } + }) +} + +describe('claude structured auth parity with the terminal preflight', () => { + afterEach(() => { + endClaudeAuthSwitch() + }) + + // Task 1 — the terminal preflight refuses this at spawn-env.ts:25 and + // runtime/spawn-preflight.ts:139; the structured path used to let the override win. + it('refuses an explicit Anthropic auth override while a managed account is pinned', async () => { + await expect( + resolverFor({ stripAuthEnv: true, overlay: { ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' } })({ + identity: IDENTITY + }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) + + it('refuses an auth-like ANTHROPIC_CUSTOM_HEADERS override while a managed account is pinned', async () => { + await expect( + resolverFor({ + stripAuthEnv: true, + overlay: { ANTHROPIC_CUSTOM_HEADERS: 'Authorization: Bearer sk-ant-CONFIGURED' } + })({ identity: IDENTITY }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) + + it('still admits a non-auth env overlay under a managed account', async () => { + const launch = await resolverFor({ + stripAuthEnv: true, + overlay: { ANTHROPIC_BASE_URL: 'https://gateway.example.test' } + })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_BASE_URL).toBe('https://gateway.example.test') + }) + + // Task 2 — legacy computes stripAuthEnv at runtime-auth-preparation.ts:72, so a + // system-auth user's own shell key is their sign-in and must survive. + it('passes an ambient Anthropic key through when no managed account is active', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ stripAuthEnv: false })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-SHELL') + }) + }) + + it('lets an explicit overlay override the ambient key when no managed account is active', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ + stripAuthEnv: false, + overlay: { ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' } + })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-CONFIGURED') + }) + }) + + it('still strips the ambient Anthropic key when a managed account is pinned', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ stripAuthEnv: true })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBeUndefined() + }) + }) + + // Task 3 — the terminal preflight guards this at four sites; the structured path had none. + it('refuses launch resolution when an account switch never settles', async () => { + beginClaudeAuthSwitch() + + await expect( + resolverFor({ stripAuthEnv: true, authSwitchSettleTimeoutMs: 20 })({ identity: IDENTITY }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + }) + + it('waits a settling account switch out rather than refusing a resolved launch', async () => { + beginClaudeAuthSwitch() + setTimeout(() => endClaudeAuthSwitch(), 20) + + const launch = await resolverFor({ + stripAuthEnv: true, + authSwitchSettleTimeoutMs: 5_000 + })({ identity: IDENTITY }) + + expect(launch.claudeConfigDir).toBe('/home/work/.claude') + }) + + it('refuses an acquire before it tears the previous session down', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude) + beginClaudeAuthSwitch() + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + // Nothing was spawned, so the refusal must not have opened a connection. + expect(claude.connections).toHaveLength(0) + }) + + // The teardown between the entry guard and launch resolution closes the live child + // and proves its tree — seconds, not milliseconds. A switch that begins inside it + // has already cost the user their session, so refusing there produces exactly the + // outcome the entry guard advertises against: a dead chat and no replacement. + it('replaces the session when a switch begins inside the acquire teardown', async () => { + const claude = fakeClaude() + const adapter = realResolverAdapter(claude, 5_000) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const live = claude.connections[0]! + const closeWithSwitch = live.close + live.close = async () => { + beginClaudeAuthSwitch() + setTimeout(() => endClaudeAuthSwitch(), 20) + return closeWithSwitch() + } + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).resolves.toMatchObject({ process: { spawnToken: 'spawn-10' } }) + expect(live.closed).toBe(true) + // The replacement child exists: the user's chat came back. + expect(claude.connections).toHaveLength(2) + expect(claude.connections[1]!.closed).toBe(false) + await adapter.closeAll() + }) + + it('still refuses a mid-teardown switch that never settles, leaving nothing half-open', async () => { + const claude = fakeClaude() + const adapter = realResolverAdapter(claude, 20) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const live = claude.connections[0]! + const closeWithSwitch = live.close + live.close = async () => { + beginClaudeAuthSwitch() + return closeWithSwitch() + } + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + // No replacement child was opened, so nothing is left running unowned. + expect(claude.connections).toHaveLength(1) + await adapter.closeAll() + }) +}) diff --git a/src/main/claude/claude-structured-content-parts.test.ts b/src/main/claude/claude-structured-content-parts.test.ts new file mode 100644 index 00000000000..d2142150937 --- /dev/null +++ b/src/main/claude/claude-structured-content-parts.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: () => {}, + publish: vi.fn() + } + return { sink, items } +} + +function providerRows(items: { body: AgentJournalItemBody }[]) { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame + ? [{ kind: item.body.providerFrame.kind, text: item.body.text }] + : [] + ) +} + +function userMessageWith(part: unknown) { + return { + type: 'message' as const, + sessionId: 'orca-session', + startsTurn: true as const, + message: { + type: 'user', + uuid: 'user-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: [{ type: 'text', text: 'look at this' }, part] } + } + } +} + +/** Exactly what claudeDispatchMessageContent sends for a local attachment. */ +const BASE64_IMAGE = { + type: 'image', + source: { type: 'base64', media_type: 'image/png', data: 'iVBORw0KGgoAAAANSUhEUg==' } +} + +describe('Claude message content parts', () => { + it('does not leak a wire kind for a locally attached image', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userMessageWith(BASE64_IMAGE)) + + expect(providerRows(state.items)).toEqual([]) + }) + + it('still renders an image the CLI sends by url', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + userMessageWith({ type: 'image', source: { type: 'url', url: 'https://x.test/a.png' } }) + ) + + expect(providerRows(state.items)).toEqual([]) + expect( + state.items.flatMap((item) => (item.body.kind === 'message' ? item.body.blocks : [])) + ).toContainEqual({ type: 'image-ref', url: 'https://x.test/a.png' }) + }) + + it('says what is true for a content part it cannot render, not the wire kind', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userMessageWith({ type: 'some_future_part', payload: { a: 1 } })) + + const rows = providerRows(state.items) + expect(rows).toHaveLength(1) + // The kind stays on the row for debugging, behind the disclosure. + expect(rows[0].kind).toBe('message:user:content:some_future_part') + // ...but the visible text is a sentence, not the opcode. + expect(rows[0].text).not.toContain('message:user:content') + expect(rows[0].text.toLowerCase()).toContain('claude') + }) + + it('prefers a readable sentence the part carries over the placeholder', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + userMessageWith({ type: 'some_future_part', message: 'the server refused the upload' }) + ) + + expect(providerRows(state.items)[0].text).toBe('the server refused the upload') + }) +}) diff --git a/src/main/claude/claude-structured-control-actions.test.ts b/src/main/claude/claude-structured-control-actions.test.ts new file mode 100644 index 00000000000..c471a8aca80 --- /dev/null +++ b/src/main/claude/claude-structured-control-actions.test.ts @@ -0,0 +1,113 @@ +import { describe, expect, it, vi } from 'vitest' +import { cancelClaudeTurn, answerClaudePrompt } from './claude-structured-control-actions' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeSession } from './claude-structured-session-state' + +type InterruptResult = Awaited> + +function sessionWith(input: { + capabilities?: string[] + interrupt: (options?: { cancelQueued?: boolean; timeoutMs?: number }) => Promise + cancelAsyncMessage?: (uuid: string) => Promise + prompts?: ClaudePromptRegistry +}): { + session: ClaudeSession + interrupt: ReturnType + cancelAsyncMessage: ReturnType +} { + const interrupt = vi.fn(input.interrupt) + const cancelAsyncMessage = vi.fn(input.cancelAsyncMessage ?? (async () => {})) + const session = { + capabilities: input.capabilities ?? [], + prompts: input.prompts ?? new ClaudePromptRegistry(), + connection: { interrupt, cancelAsyncMessage } + } as unknown as ClaudeSession + return { session, interrupt, cancelAsyncMessage } +} + +describe('cancelClaudeTurn', () => { + it('interrupts without a receipt on an older CLI and reports the turn cancelled', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + interrupt: async () => undefined + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(interrupt).toHaveBeenCalledWith({ timeoutMs: 5_000 }) + expect(cancelAsyncMessage).not.toHaveBeenCalled() + }) + + it('withdraws every still-queued message a plain interrupt receipt reports', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + capabilities: ['interrupt_receipt_v1'], + interrupt: async () => ({ still_queued: ['queued-1', 'queued-2'] }) + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + // No cancel_queued capability, so the queue is swept one uuid at a time. + expect(interrupt).toHaveBeenCalledWith({ timeoutMs: 5_000 }) + expect(cancelAsyncMessage.mock.calls.map((call) => call[0])).toEqual(['queued-1', 'queued-2']) + }) + + it('sends cancel_queued and never sweeps when the CLI advertises the capability', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + capabilities: ['interrupt_receipt_v1', 'interrupt_cancel_queued_v1'], + interrupt: async () => ({ still_queued: [], cancelled: ['queued-1'] }) + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(interrupt).toHaveBeenCalledWith({ cancelQueued: true, timeoutMs: 5_000 }) + expect(cancelAsyncMessage).not.toHaveBeenCalled() + }) + + it('reports a not-running interrupt as not cancelled without throwing', async () => { + const { session } = sessionWith({ + interrupt: async () => { + throw new ClaudeControlRequestError('interrupt', 'not running') + } + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: false }) + }) + + it('propagates a transport failure such as an interrupt timeout', async () => { + const { session } = sessionWith({ + interrupt: async () => { + throw new Error('claude interrupt request timed out') + } + }) + + await expect(cancelClaudeTurn(session, 5_000)).rejects.toThrow('timed out') + }) +}) + +describe('answerClaudePrompt', () => { + it('settles the pending prompt callback and forgets it', async () => { + const prompts = new ClaudePromptRegistry() + const settle = vi.fn() + const prompt = prompts.register({ + requestId: 'perm-1', + toolName: 'Bash', + toolUseId: 'tool-1', + input: { command: 'ls' }, + suggestions: [], + settle + })! + prompts.bindJournalItemId('journal-1', prompt.promptKey) + const { session } = sessionWith({ interrupt: async () => undefined, prompts }) + + await answerClaudePrompt(session, { itemId: 'journal-1', kind: 'approval', optionId: 'allow' }) + + expect(settle).toHaveBeenCalledWith( + expect.objectContaining({ behavior: 'allow', toolUseID: 'tool-1' }) + ) + expect(prompts.find('journal-1')).toBeNull() + }) + + it('refuses an answer for a prompt Claude is no longer waiting on', async () => { + const { session } = sessionWith({ interrupt: async () => undefined }) + await expect( + answerClaudePrompt(session, { itemId: 'missing', kind: 'approval', optionId: 'allow' }) + ).rejects.toThrow(/no longer waiting/) + }) +}) diff --git a/src/main/claude/claude-structured-control-actions.ts b/src/main/claude/claude-structured-control-actions.ts new file mode 100644 index 00000000000..d4484963aae --- /dev/null +++ b/src/main/claude/claude-structured-control-actions.ts @@ -0,0 +1,60 @@ +import { applyClaudePromptAnswer } from './claude-structured-prompt-replies' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import type { ClaudeSession } from './claude-structured-session-state' + +const INTERRUPT_CANCEL_QUEUED_CAPABILITY = 'interrupt_cancel_queued_v1' + +export type ClaudeTurnCancellationGuard = () => boolean + +/** + * Interrupt the running turn, then make sure no queued async user message survives to spawn a + * later unexpected turn. On a CLI advertising `interrupt_cancel_queued_v1` one round trip + * cancels the queue alongside the abort; otherwise the interrupt receipt lists `still_queued` + * uuids, and each is withdrawn best-effort with `cancel_async_message`. Older CLIs resolve no + * receipt, so there is nothing to sweep. + */ +export async function cancelClaudeTurn( + session: ClaudeSession, + timeoutMs: number | undefined, + isCurrent: ClaudeTurnCancellationGuard = () => true +): Promise<{ cancelled: boolean }> { + // The SDK interrupt is session-scoped. Re-check the caller's turn/fence + // immediately before issuing it so a delayed request cannot stop a later turn. + if (!isCurrent()) { + return { cancelled: false } + } + const cancelQueued = session.capabilities.includes(INTERRUPT_CANCEL_QUEUED_CAPABILITY) + try { + const receipt = await session.connection.interrupt({ + ...(cancelQueued ? { cancelQueued: true } : {}), + timeoutMs + }) + if (!cancelQueued) { + for (const uuid of receipt?.still_queued ?? []) { + await session.connection.cancelAsyncMessage(uuid, { timeoutMs }).catch(() => {}) + } + } + return { cancelled: true } + } catch (error) { + if (error instanceof ClaudeControlRequestError) { + return { cancelled: false } + } + throw error + } +} + +export async function answerClaudePrompt( + session: ClaudeSession, + input: { itemId: string; kind: 'approval' | 'question'; optionId: string } +): Promise { + const found = session.prompts.find(input.itemId) + if (!found || found.prompt.kind !== input.kind) { + throw new Error(`claude is no longer waiting on ${input.itemId}`) + } + const response = applyClaudePromptAnswer(found, input.optionId) + if (response === null) { + return + } + session.prompts.forget(found.prompt) + found.prompt.settle(response) +} diff --git a/src/main/claude/claude-structured-dispatch-content.ts b/src/main/claude/claude-structured-dispatch-content.ts new file mode 100644 index 00000000000..71f180bc3ac --- /dev/null +++ b/src/main/claude/claude-structured-dispatch-content.ts @@ -0,0 +1,165 @@ +import { createHash } from 'node:crypto' +import { open } from 'node:fs/promises' +import { extname } from 'node:path' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import type { NativeChatBlock } from '../../shared/native-chat-types' + +const MAX_IMAGE_BYTES = 5 * 1024 * 1024 +const MAX_IMAGE_COUNT = 20 +const MAX_TOTAL_IMAGE_BYTES = 20 * 1024 * 1024 +const MAX_REPLAY_CONTENT_KEY_BYTES = 256 + +type ImageBudget = { + count: number + localBytes: number +} + +export async function readClaudeImage(path: string, openImpl: typeof open = open): Promise { + const file = await openImpl(path, 'r') + try { + const invalidImage = (): Error => + new Error(`Claude image must be a non-empty file no larger than ${MAX_IMAGE_BYTES} bytes`) + const info = await file.stat() + if (!info.isFile()) { + throw new Error('Claude image must be a file') + } + if (info.size > MAX_IMAGE_BYTES) { + throw invalidImage() + } + const buffer = Buffer.allocUnsafe(info.size + 1) + let bytesRead = 0 + while (bytesRead < buffer.length) { + const result = await file.read(buffer, bytesRead, buffer.length - bytesRead, bytesRead) + if (result.bytesRead === 0) { + break + } + bytesRead += result.bytesRead + } + // A file can grow after the initial stat and after the final read returns + // zero. Prove the descriptor's size matches what was copied before sending. + const finalInfo = await file.stat() + if (bytesRead === 0 || bytesRead > MAX_IMAGE_BYTES || finalInfo.size !== bytesRead) { + throw invalidImage() + } + return buffer.subarray(0, bytesRead) + } finally { + await file.close() + } +} + +const IMAGE_MIME_BY_EXTENSION: Record = { + '.gif': 'image/gif', + '.jpeg': 'image/jpeg', + '.jpg': 'image/jpeg', + '.png': 'image/png', + '.webp': 'image/webp' +} + +async function imageContent( + block: Extract, + budget: ImageBudget +): Promise { + budget.count += 1 + if (budget.count > MAX_IMAGE_COUNT) { + throw new Error(`Claude messages support at most ${MAX_IMAGE_COUNT} images`) + } + if (block.url) { + return { type: 'image', source: { type: 'url', url: block.url } } + } + if (!block.path) { + throw new Error('image reference has neither a path nor a URL') + } + const data = await readClaudeImage(block.path) + budget.localBytes += data.byteLength + if (budget.localBytes > MAX_TOTAL_IMAGE_BYTES) { + throw new Error(`Claude images must total no more than ${MAX_TOTAL_IMAGE_BYTES} bytes`) + } + const mediaType = IMAGE_MIME_BY_EXTENSION[extname(block.path).toLowerCase()] + if (!mediaType) { + throw new Error(`Claude does not support the image type ${extname(block.path)}`) + } + return { + type: 'image', + source: { + type: 'base64', + media_type: mediaType, + data: data.toString('base64') + } + } +} + +export async function claudeDispatchMessageContent( + body: AgentJournalMessageItem +): Promise { + if (body.role !== 'user') { + throw new Error('Claude dispatch accepts only user messages') + } + const content: unknown[] = [] + const imageBudget: ImageBudget = { count: 0, localBytes: 0 } + for (const block of body.blocks as NativeChatBlock[]) { + if (block.type === 'text' && block.text.length > 0) { + content.push({ type: 'text', text: block.text }) + } else if (block.type === 'image-ref') { + content.push(await imageContent(block, imageBudget)) + } + } + if (content.length === 0) { + throw new Error('Claude dispatch requires text or an image') + } + return content +} + +/** + * Keep waiter metadata bounded even when a dispatch contains large base64 images. + * The digest is only diagnostic: replay acknowledgement must use provider identity. + */ +export function claudeDispatchContentKey(content: readonly unknown[]): string { + const digest = createHash('sha256') + const summary = content + .map((part) => { + const record = + typeof part === 'object' && part !== null && !Array.isArray(part) + ? (part as Record) + : null + const type = typeof record?.type === 'string' ? record.type : 'unknown' + if (type === 'text') { + return `text:${typeof record?.text === 'string' ? record.text.length : 0}` + } + const source = + typeof record?.source === 'object' && record.source !== null + ? (record.source as Record) + : null + if (type === 'image' && source?.type === 'base64') { + return `image:${typeof source.media_type === 'string' ? source.media_type : ''}:${typeof source.data === 'string' ? source.data.length : 0}` + } + return type + }) + .join(',') + for (const [index, part] of content.entries()) { + const record = + typeof part === 'object' && part !== null && !Array.isArray(part) + ? (part as Record) + : null + const type = typeof record?.type === 'string' ? record.type : 'unknown' + digest.update(`${index}:${type}:`) + if (type === 'text' && typeof record?.text === 'string') { + digest.update(record.text) + continue + } + const source = + typeof record?.source === 'object' && record.source !== null + ? (record.source as Record) + : null + if (type === 'image' && source?.type === 'base64') { + digest.update(typeof source.media_type === 'string' ? source.media_type : '') + digest.update(':') + if (typeof source.data === 'string') { + digest.update(source.data) + } + continue + } + digest.update(JSON.stringify(part)) + } + const key = `v1:${summary.slice(0, 128)}:${digest.digest('hex')}` + return key.slice(0, MAX_REPLAY_CONTENT_KEY_BYTES) +} diff --git a/src/main/claude/claude-structured-dispatch.test.ts b/src/main/claude/claude-structured-dispatch.test.ts new file mode 100644 index 00000000000..4e8289a89e3 --- /dev/null +++ b/src/main/claude/claude-structured-dispatch.test.ts @@ -0,0 +1,598 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import { dispatchClaudeTurn, resolveClaudeReplayWaiter } from './claude-structured-dispatch' +import { readClaudeImage } from './claude-structured-dispatch-content' +import type { ClaudeSession } from './claude-structured-session-state' + +function sessionFor(send = vi.fn().mockResolvedValue(undefined)): ClaudeSession { + return { + connection: { send } as unknown as ClaudeSession['connection'], + providerSessionId: 'provider-session', + claudeConfigDir: '/accounts/claude', + leafUuid: null, + fence: 1, + acquisitionGeneration: 'generation-1', + prompts: {} as ClaudeSession['prompts'], + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(), + reportedOptions: {}, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + capabilities: [], + events: undefined, + translator: null + } +} + +function userMessage(blocks: AgentJournalMessageItem['blocks']): AgentJournalMessageItem { + return { kind: 'message', role: 'user', blocks } +} + +function userReplayFrame(uuid: string, text: string): Record { + return { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid, + message: { role: 'user', content: [{ type: 'text', text }] } + } +} + +describe('Claude structured dispatch image limits', () => { + it('recovers the active identity when a timed-out replay arrives late', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + await expect(dispatched).resolves.toMatchObject({ state: 'unknown' }) + + expect(resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid!, 'one'))).toBe(true) + expect(session.activeTurnId).toBe(sentUuid) + expect(session.activeTurnSequence).toBe(session.dispatchSequence) + }) + + it('never lets a late replay for dispatch A resolve dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const firstUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + + expect(resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one'))).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + expect(resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid!, 'two'))).toBe(true) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let an identical late replay for dispatch A resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect(resolveClaudeReplayWaiter(session, userReplayFrame('provider-a', 'same prompt'))).toBe( + false + ) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let a fresh-UUID replay for an evicted dispatch resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid + + const fillerDispatches = await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn( + session, + { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }, + 5 + ) + ) + ) + expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(session.retiredDispatchWaiters).toHaveLength(64) + expect(session.replayContentFallbackBlocked).toBe(true) + expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( + false + ) + + while (session.retiredDispatchWaiters.length > 0) { + const sentUuid = session.retiredDispatchWaiters[0]!.sentUuid + resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid, 'same prompt')) + } + expect(session.retiredDispatchWaiters).toHaveLength(0) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, userReplayFrame('provider-a-late', 'same prompt')) + ).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let a fresh-UUID result for an evicted slash dispatch resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid + + const fillerDispatches = await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn( + session, + { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: '/permissions' }]) + }, + 5 + ) + ) + ) + expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(session.retiredDispatchWaiters).toHaveLength(64) + expect(session.replayContentFallbackBlocked).toBe(true) + expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( + false + ) + + while (session.retiredDispatchWaiters.length > 0) { + const sentUuid = session.retiredDispatchWaiters[0]!.sentUuid + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: `result-${sentUuid}`, + user_message_uuid: sentUuid + }) + ).toBe(false) + } + expect(session.retiredDispatchWaiters).toHaveLength(0) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-a-late' + }) + ).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ + providerIdentity: { uuid: 'result-b' } + }) + }) + + it('does not let a legacy result for timed-out ordinary dispatch A resolve slash dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'ordinary' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'legacy-result-a' + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ state: 'unknown' }) + }) + + it('removes only its own waiter when a later send fails', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const firstWaiter = session.dispatchWaiters[0] + session.connection.send = vi.fn().mockRejectedValue(new Error('broken pipe')) + + await expect( + dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, + 100 + ) + ).resolves.toMatchObject({ state: 'unknown', reason: 'broken pipe' }) + expect(session.dispatchWaiters).toEqual([firstWaiter]) + + const firstUuid = (firstWaiter as { sentUuid?: string }).sentUuid + resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one')) + await expect(first).resolves.toMatchObject({ providerIdentity: { uuid: firstUuid } }) + }) + + it('keeps a replay accepted before its send reports failure', async () => { + let session!: ClaudeSession + const send = vi.fn(async (message: Record) => { + resolveClaudeReplayWaiter(session, { ...message, uuid: 'turn-race' }) + throw new Error('write raced provider acknowledgement') + }) + session = sessionFor(send) + + await expect( + dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 100 + ) + ).resolves.toMatchObject({ state: 'accepted', providerIdentity: { uuid: 'turn-race' } }) + expect(session.dispatchWaiters).toHaveLength(0) + }) + + it('accepts a slash command from its result receipt when Claude omits the user replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'command-result-uuid' + }) + ).toBe(false) + + await expect(dispatched).resolves.toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'command-result-uuid' + } + }) + }) + + it('correlates a later slash-command result by user_message_uuid despite a timed-out slash waiter', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'result-b' } + }) + }) + + it('does not mistake a normal turn result for its missing user replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'hello' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + session_id: 'provider-session', + uuid: 'unrelated-result-uuid' + }) + ).toBe(false) + expect(session.dispatchWaiters).toHaveLength(1) + expect( + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: 'hello' }] + } + }) + ).toBe(true) + + await expect(dispatched).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'user-replay-uuid' } + }) + }) + + it('ignores a top-level tool-result user frame while waiting for a slash command replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'tool-result-uuid', + message: { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'done' }] + } + }) + expect(session.dispatchWaiters).toHaveLength(1) + + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: '/permissions' }] + } + }) + + await expect(dispatched).resolves.toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'user-replay-uuid' + } + }) + }) + + it('rejects more than twenty URL images before sending', async () => { + const session = sessionFor() + const body = userMessage( + Array.from({ length: 21 }, (_, index) => ({ + type: 'image-ref' as const, + url: `https://example.test/${index}.png` + })) + ) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ state: 'rejected', reason: 'Claude messages support at most 20 images' }) + expect(session.connection.send).not.toHaveBeenCalled() + }) + + it('rejects local images whose aggregate size exceeds twenty MiB', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-images-')) + try { + const paths = await Promise.all( + Array.from({ length: 5 }, async (_, index) => { + const path = join(directory, `${index}.png`) + await writeFile(path, Buffer.alloc(5 * 1024 * 1024)) + return path + }) + ) + const session = sessionFor() + const body = userMessage(paths.map((path) => ({ type: 'image-ref' as const, path }))) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ + state: 'rejected', + reason: `Claude images must total no more than ${20 * 1024 * 1024} bytes` + }) + expect(session.connection.send).not.toHaveBeenCalled() + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('rejects a local image by actual bytes read beyond the per-image cap', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + try { + const path = join(directory, 'oversized.png') + await writeFile(path, Buffer.alloc(5 * 1024 * 1024 + 1)) + const session = sessionFor() + const body = userMessage([{ type: 'image-ref', path }]) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ + state: 'rejected', + reason: `Claude image must be a non-empty file no larger than ${5 * 1024 * 1024} bytes` + }) + expect(session.connection.send).not.toHaveBeenCalled() + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('allocates local image reads from the file size, not the maximum cap', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + const allocUnsafe = vi.spyOn(Buffer, 'allocUnsafe') + try { + const path = join(directory, 'small.png') + await writeFile(path, Buffer.alloc(64)) + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'image-ref', path }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + resolveClaudeReplayWaiter(session, { + ...userReplayFrame(sentUuid!, ''), + message: { + role: 'user', + content: [ + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: '' } } + ] + } + }) + await expect(dispatched).resolves.toMatchObject({ state: 'accepted' }) + expect(allocUnsafe).toHaveBeenCalled() + expect(allocUnsafe.mock.calls.some(([size]) => size === 64 + 1)).toBe(true) + expect(allocUnsafe.mock.calls.some(([size]) => size >= 5 * 1024 * 1024)).toBe(false) + } finally { + allocUnsafe.mockRestore() + await rm(directory, { recursive: true, force: true }) + } + }) + + it('bounds retained waiter identity bytes when image dispatches time out', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + try { + const path = join(directory, 'large.png') + await writeFile(path, Buffer.alloc(64 * 1024)) + const session = sessionFor() + const body = userMessage([{ type: 'image-ref', path }]) + await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn(session, { clientMessageId: `client-${index}`, body }, 1) + ) + ) + + expect(session.retiredDispatchWaiters).toHaveLength(64) + const retainedKeyBytes = session.retiredDispatchWaiters.reduce( + (total, waiter) => total + waiter.replayContentKey.length, + 0 + ) + expect(retainedKeyBytes).toBeLessThan(64 * 512) + expect( + session.retiredDispatchWaiters.every((waiter) => waiter.replayContentKey.length < 512) + ).toBe(true) + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('rejects a local image when it grows after the initial stat', async () => { + const stat = vi + .fn() + .mockResolvedValueOnce({ isFile: () => true, size: 64 }) + .mockResolvedValueOnce({ isFile: () => true, size: 128 }) + const read = vi.fn(async (buffer: Buffer, offset: number) => { + if (read.mock.calls.length === 1) { + buffer.fill(1, offset, offset + 64) + return { bytesRead: 64, buffer } + } + return { bytesRead: 0, buffer } + }) + const open = vi.fn().mockResolvedValue({ + stat, + read, + close: vi.fn().mockResolvedValue(undefined) + } as never) + await expect(readClaudeImage('/controlled/growing.png', open)).rejects.toThrow( + `Claude image must be a non-empty file no larger than ${5 * 1024 * 1024} bytes` + ) + }) +}) diff --git a/src/main/claude/claude-structured-dispatch.ts b/src/main/claude/claude-structured-dispatch.ts new file mode 100644 index 00000000000..96271e41d71 --- /dev/null +++ b/src/main/claude/claude-structured-dispatch.ts @@ -0,0 +1,264 @@ +import { randomUUID } from 'node:crypto' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import type { AgentSessionDispatchOutcome } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + claudeHasReplayContent, + readClaudeMessageEnvelope +} from './claude-structured-item-translation' +import type { ClaudeDispatchWaiter, ClaudeSession } from './claude-structured-session-state' +import { readClaudeFrameString } from './claude-structured-init-proof' +import { + claudeDispatchContentKey, + claudeDispatchMessageContent +} from './claude-structured-dispatch-content' + +const MAX_RETIRED_DISPATCH_WAITERS = 64 + +export function resolveClaudeReplayWaiter( + session: ClaudeSession, + message: Record +): boolean { + const envelope = readClaudeMessageEnvelope(message) + const isUserReplay = + envelope?.role === 'user' && + message.parent_tool_use_id === null && + claudeHasReplayContent(envelope) + const isCompletedCommand = message.type === 'result' + if ( + (!isUserReplay && !isCompletedCommand) || + readClaudeFrameString(message, 'session_id') !== session.providerSessionId + ) { + return false + } + const uuid = readClaudeFrameString(message, 'uuid') + if (!uuid) { + return false + } + + // Newer SDK frames carry the client uuid that caused a turn. A correlation + // value is authoritative: never fall back to queue order or content, since + // identical prompts may be in flight across a timeout boundary. + const userMessageUuid = readClaudeFrameString(message, 'user_message_uuid') + if (userMessageUuid) { + const exact = session.dispatchWaiters.find( + (candidate) => candidate.sentUuid === userMessageUuid + ) + if (exact) { + settleWaiter(session, exact, uuid) + return isUserReplay && exact.dispatchSequence === session.dispatchSequence + } + const retired = session.retiredDispatchWaiters.find( + (candidate) => candidate.sentUuid === userMessageUuid + ) + if (retired) { + forgetRetiredWaiter(session, retired) + return recoverLateIdentity(session, retired, uuid, isUserReplay) + } + return false + } + + const exact = session.dispatchWaiters.find((candidate) => candidate.sentUuid === uuid) + if (exact) { + settleWaiter(session, exact, uuid) + return isUserReplay && exact.dispatchSequence === session.dispatchSequence + } + const retired = session.retiredDispatchWaiters.find((candidate) => candidate.sentUuid === uuid) + if (retired) { + forgetRetiredWaiter(session, retired) + return recoverLateIdentity(session, retired, uuid, isUserReplay) + } + + if (isUserReplay) { + // Compatibility CLIs may mint a new replay uuid instead of echoing the + // client uuid. Content is an acceptable join only when it is the sole + // candidate on one side of the timeout boundary; with active and retired + // candidates present, identical prompts are intentionally left unknown. + const replayContentKey = claudeDispatchContentKey(envelope.content) + if (!session.replayContentFallbackBlocked && session.retiredDispatchWaiters.length === 0) { + const compatible = session.dispatchWaiters.filter( + (candidate) => candidate.replayContentKey === replayContentKey + ) + if (compatible.length === 1) { + settleWaiter(session, compatible[0]!, uuid) + return compatible[0]!.dispatchSequence === session.dispatchSequence + } + } else if (!session.replayContentFallbackBlocked && session.dispatchWaiters.length === 0) { + const lateCompatible = session.retiredDispatchWaiters.filter( + (candidate) => candidate.replayContentKey === replayContentKey + ) + if (lateCompatible.length === 1) { + const [candidate] = lateCompatible + forgetRetiredWaiter(session, candidate!) + return recoverLateIdentity(session, candidate!, uuid, true) + } + } + return false + } + const current = session.dispatchWaiters[0] + if (isCompletedCommand && !current?.acceptsResult) { + return false + } + // A legacy result has no dispatch correlation. Any retired waiter makes queue order ambiguous, + // even when the retired dispatch was an ordinary turn rather than a slash command. + if (isCompletedCommand && session.retiredDispatchWaiters.length > 0) { + return false + } + // Once an eviction occurred, a fresh result uuid cannot be joined to a waiter by queue order. + if (isCompletedCommand && session.replayContentFallbackBlocked) { + return false + } + const waiter = uuid ? session.dispatchWaiters.shift() : undefined + if (waiter && uuid) { + clearTimeout(waiter.timer) + waiter.settledUuid = uuid + waiter.resolve(uuid) + return isUserReplay + } + return false +} + +function settleWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter, uuid: string): void { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + clearTimeout(waiter.timer) + waiter.settledUuid = uuid + waiter.resolve(uuid) +} + +function forgetRetiredWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { + const index = session.retiredDispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.retiredDispatchWaiters.splice(index, 1) + } +} + +function recoverLateIdentity( + session: ClaudeSession, + waiter: ClaudeDispatchWaiter, + uuid: string, + isUserReplay: boolean +): boolean { + if (!isUserReplay && !waiter.acceptsResult) { + return false + } + if (waiter.dispatchSequence === session.dispatchSequence) { + session.activeTurnId = uuid + session.activeTurnSequence = waiter.dispatchSequence + } + return isUserReplay && waiter.dispatchSequence === session.dispatchSequence +} + +function waitForReplay( + session: ClaudeSession, + timeoutMs: number, + acceptsResult: boolean, + sentUuid: string, + replayContentKey: string +): { waiter: ClaudeDispatchWaiter; promise: Promise } { + let waiter!: ClaudeDispatchWaiter + const promise = new Promise((resolve) => { + waiter = { + acceptsResult, + sentUuid, + dispatchSequence: session.dispatchSequence, + replayContentKey, + resolve, + timer: setTimeout(() => { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + retireWaiter(session, waiter) + resolve(null) + }, timeoutMs) + } + waiter.timer.unref?.() + session.dispatchWaiters.push(waiter) + }) + return { waiter, promise } +} + +function retireWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + clearTimeout(waiter.timer) + if (!waiter.retired) { + waiter.retired = true + session.retiredDispatchWaiters.push(waiter) + if (session.retiredDispatchWaiters.length > MAX_RETIRED_DISPATCH_WAITERS) { + session.replayContentFallbackBlocked = true + session.retiredDispatchWaiters.splice( + 0, + session.retiredDispatchWaiters.length - MAX_RETIRED_DISPATCH_WAITERS + ) + } + } +} + +export async function dispatchClaudeTurn( + session: ClaudeSession, + input: { clientMessageId: string; body: AgentJournalMessageItem }, + timeoutMs: number +): Promise { + let content: unknown[] + try { + content = await claudeDispatchMessageContent(input.body) + } catch (error) { + return { state: 'rejected', reason: (error as Error).message } + } + const dispatchSequence = ++session.dispatchSequence + const acceptsResult = input.body.blocks.some( + (block) => block.type === 'text' && block.text.trimStart().startsWith('/') + ) + const sentUuid = randomUUID() + const replay = waitForReplay( + session, + timeoutMs, + acceptsResult, + sentUuid, + claudeDispatchContentKey(content) + ) + const replayed = replay.promise + try { + await session.connection.send({ + type: 'user', + uuid: sentUuid, + message: { role: 'user', content }, + parent_tool_use_id: null, + session_id: session.providerSessionId + }) + } catch (error) { + const waiter = replay.waiter + if (waiter.settledUuid) { + const uuid = await replayed + if (uuid) { + session.activeTurnId = uuid + session.activeTurnSequence = dispatchSequence + return { + state: 'accepted', + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + } + } + } + if (!waiter.retired) { + retireWaiter(session, waiter) + waiter.resolve(null) + } + return { state: 'unknown', reason: (error as Error).message } + } + const uuid = await replayed + if (uuid) { + session.activeTurnId = uuid + session.activeTurnSequence = dispatchSequence + } + return uuid + ? { + state: 'accepted', + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + } + : { state: 'unknown', reason: 'claude accepted a message but did not replay its uuid in time' } +} diff --git a/src/main/claude/claude-structured-effort-reporting.test.ts b/src/main/claude/claude-structured-effort-reporting.test.ts new file mode 100644 index 00000000000..be022d86956 --- /dev/null +++ b/src/main/claude/claude-structured-effort-reporting.test.ts @@ -0,0 +1,257 @@ +import { describe, expect, it } from 'vitest' +import { AgentSessionOptionRejectedError } from '../native-chat/agent-session-wire/structured-agent-session-option-error' +import { + restoreClaudeStructuredSessionOptions, + setClaudeStructuredOption +} from './claude-structured-options' +import { readClaudeSettingsEffort } from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-adapter' +import { acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim from Claude Code 2.1.258's get_settings response. */ +const REAL_SETTINGS = { + applied: { model: 'claude-opus-5[1m]', effort: 'high', advisor: null, ultracode: false }, + effective: { model: 'claude-opus-5[1m]', effortLevel: 'high', env: {} }, + sources: {} +} + +function sessionWith( + reported: string | null, + calls: string[] = [], + listed?: { model: string; catalog: readonly Record[] } +) { + return { + session: { + options: new Map(listed ? [['model', listed.model]] : []), + reportedOptions: {} as { model?: string; effort?: string }, + optionMutationSequence: 0, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + connection: { + supportedModels: async () => { + calls.push('list_models') + return [...(listed?.catalog ?? [])] + }, + setModel: async (model: string) => { + calls.push(`set_model:${model}`) + }, + applyFlagSettings: async (settings: { effortLevel?: string }) => { + // The measured behaviour: an unknown effort is accepted and ignored. + calls.push(`apply:${settings.effortLevel}`) + }, + getSettings: async () => { + calls.push('get_settings') + return reported === null + ? { applied: {}, effective: {}, sources: {} } + : { applied: { effort: reported }, effective: { effortLevel: reported }, sources: {} } + } + } + } as unknown as ClaudeSession, + calls + } +} + +describe('Claude effort reporting', () => { + it('reads the effort get_settings reports', () => { + expect(readClaudeSettingsEffort(REAL_SETTINGS)).toBe('high') + }) + + it.each([ + [ + 'the provider stops reporting it', + { applied: { effort: 'high' }, effective: {}, sources: {} } + ], + ['the payload carries no effective block', { applied: { effort: 'high' } }], + ['the request failed outright', null] + ])('reports no effort when %s', (_case, settings) => { + // Never defaulted: an effort nothing measured would be worse than a blank + // pill, and this is the assertion that goes red if the key is renamed. + expect(readClaudeSettingsEffort(settings)).toBeNull() + }) + + it('publishes the effort from get_settings, which system/init never carries', async () => { + const claude = fakeClaude({ settings: REAL_SETTINGS }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { effort: 'high' } + }) + }) + + it('leaves the effort unreported when the session never learns one', async () => { + const claude = fakeClaude({ settings: { applied: {}, effective: {}, sources: {} } }) + const adapter = await acquired(claude) + + const options = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(options.current.effort).toBeUndefined() + expect(options.current.model).toBeTruthy() + }) + + it('keeps the init fixture free of an effort the real frame never sends', async () => { + const events: ClaudeStructuredSessionEvent[] = [] + await acquired(fakeClaude(), {}, events) + const init = events.flatMap((event) => + event.type === 'message' && event.message.subtype === 'init' ? [event.message] : [] + ) + + expect(init).toHaveLength(1) + expect(init[0]).toHaveProperty('model') + // The regression that hid this defect: a fixture inventing `effortLevel` + // kept every gate green over a value that is always empty in production. + expect(Object.keys(init[0])).not.toContain('effortLevel') + }) +}) + +describe('Claude effort readback', () => { + it('records an effort the child did not adopt without vouching for it', async () => { + const { session, calls } = sessionWith('high') + + // The disagreement stops the confirmation, not the write: no other client + // vetoes here, and the pre-flight catalog guard already refuses the levels + // the model cannot run. + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'bogus-effort-xyz' }, undefined) + ).resolves.toEqual({ effort: 'bogus-effort-xyz' }) + expect(session.confirmedOptions.has('effort')).toBe(false) + // The child's own answer is kept rather than discarded with the refusal. + expect(session.reportedOptions.effort).toBe('high') + expect(calls).toEqual(['apply:bogus-effort-xyz', 'get_settings']) + }) + + it('records an effort the child confirms', async () => { + const { session } = sessionWith('low') + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + ).resolves.toEqual({ effort: 'low' }) + }) + + it('records the request when the readback is unavailable', async () => { + // No evidence of a refusal is not evidence of one; the apply itself succeeded. + const { session } = sessionWith(null) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + ).resolves.toEqual({ effort: 'low' }) + }) +}) + +describe('Claude effort against the model that must run it', () => { + const HAIKU = { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } + const SONNET = { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + } + + it('refuses an effort the current model advertises no control for', async () => { + const { session, calls } = sessionWith('high', [], { model: 'haiku', catalog: [HAIKU, SONNET] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + // Measured on Claude Code 2.1.260: apply_flag_settings stores `high` on a + // haiku session and get_settings reads it straight back, so a send here is + // never undone. The refusal has to land before the write. + expect(calls).toEqual(['list_models']) + expect(session.options.has('effort')).toBe(false) + }) + + it('refuses a level outside the ones the current model advertises', async () => { + const { session } = sessionWith('high', [], { + model: 'sonnet', + catalog: [{ ...SONNET, supportedEffortLevels: ['low', 'medium'] }] + }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + }) + + it('sends an effort the current model advertises', async () => { + const { session, calls } = sessionWith('high', [], { + model: 'sonnet', + catalog: [HAIKU, SONNET] + }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + expect(session.confirmedOptions.has('effort')).toBe(true) + }) + + it('sends `max`, which the readback cannot report, when the model advertises it', async () => { + // UNREPORTED_EFFORTS still governs: no get_settings, so no false disagreement. + const { session, calls } = sessionWith('high', [], { model: 'sonnet', catalog: [SONNET] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'max' }) + expect(calls).toEqual(['list_models', 'apply:max']) + expect(session.confirmedOptions.has('effort')).toBe(false) + }) + + it('sends the effort when the model is not in the catalog the CLI listed', async () => { + // An unlisted model is an unknown one, not one that refuses effort. + const { session, calls } = sessionWith('high', [], { model: 'sonnet', catalog: [HAIKU] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + }) + + it('sends the effort when list_models is unavailable', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [] }) + session.connection.supportedModels = async () => { + calls.push('list_models') + throw new Error('this CLI predates list_models') + } + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + }) + + it('matches the model the init frame reported, not just the id the user picked', async () => { + const { session } = sessionWith('high', [], { model: 'sonnet', catalog: [HAIKU, SONNET] }) + session.options.delete('model') + session.reportedOptions.model = 'claude-haiku-4-5-20251001' + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + }) + + it('keeps a disagreeing effort through restore instead of skipping it', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [SONNET] }) + session.options.set('effort', 'low') + + await restoreClaudeStructuredSessionOptions(session, undefined) + + expect(session.options.get('effort')).toBe('low') + expect(session.restoreSkippedOptions.has('effort')).toBe(false) + expect(session.confirmedOptions.has('effort')).toBe(false) + }) + + it('drops a stale effort on restore instead of replaying it onto the new model', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [HAIKU, SONNET] }) + session.options.set('model', 'haiku') + session.options.set('effort', 'high') + + await restoreClaudeStructuredSessionOptions(session, undefined) + + expect(session.options.has('effort')).toBe(false) + expect(session.restoreSkippedOptions.has('effort')).toBe(true) + expect(calls.filter((call) => call.startsWith('apply:'))).toEqual([]) + }) +}) diff --git a/src/main/claude/claude-structured-inbound-control.test.ts b/src/main/claude/claude-structured-inbound-control.test.ts new file mode 100644 index 00000000000..07be4bbb516 --- /dev/null +++ b/src/main/claude/claude-structured-inbound-control.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it, vi } from 'vitest' +import type { CanUseTool } from '@anthropic-ai/claude-agent-sdk' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { + buildClaudePermissionCallbacks, + CLAUDE_BLOCKING_CONTROL_CALLBACKS, + CLAUDE_CAN_USE_TOOL_SUBTYPE, + CLAUDE_REQUEST_USER_DIALOG_SUBTYPE +} from './claude-structured-inbound-control' + +type CanUseToolOptions = Parameters[2] + +function permissionOptions( + requestId: string, + toolUseID: string, + signal: AbortSignal, + suggestions?: unknown[] +): CanUseToolOptions { + return { + requestId, + toolUseID, + signal, + ...(suggestions ? { suggestions } : {}) + } as unknown as CanUseToolOptions +} + +function callbacksFor() { + const prompts = new ClaudePromptRegistry() + const emit = vi.fn() + const { canUseTool, onUserDialog } = buildClaudePermissionCallbacks({ + sessionId: 'session-1', + prompts, + emit + }) + return { prompts, emit, canUseTool, onUserDialog } +} + +describe('Claude permission callbacks', () => { + it('registers a decodable can_use_tool as a durable prompt and settles it from the registry', async () => { + const control = callbacksFor() + const answered = control.canUseTool( + 'Bash', + { command: 'git status' }, + permissionOptions('perm-1', 'tool-1', new AbortController().signal, [{ type: 'addRules' }]) + ) + + expect(control.emit).toHaveBeenCalledWith( + expect.objectContaining({ + type: 'prompt', + sessionId: 'session-1', + prompt: expect.objectContaining({ promptKey: 'perm-1', toolName: 'Bash', kind: 'approval' }) + }) + ) + const found = control.prompts.find('perm-1') + expect(found?.prompt.suggestions).toEqual([{ type: 'addRules' }]) + // The prompt's settle is the SDK callback's own resolve — answering resolves this promise. + found?.prompt.settle({ behavior: 'allow', toolUseID: 'tool-1' }) + await expect(answered).resolves.toEqual({ behavior: 'allow', toolUseID: 'tool-1' }) + }) + + it('denies a malformed permission request without registering a prompt', async () => { + const control = callbacksFor() + const answered = control.canUseTool( + '', + {}, + permissionOptions('perm-2', 'tool-2', new AbortController().signal) + ) + + await expect(answered).resolves.toEqual({ + behavior: 'deny', + message: 'Orca could not decode this permission request.', + toolUseID: 'tool-2' + }) + expect(control.prompts.find('perm-2')).toBeNull() + expect(control.emit).not.toHaveBeenCalled() + }) + + it('settles a pending prompt with null and forgets it when the abort signal fires', async () => { + const control = callbacksFor() + const controller = new AbortController() + const answered = control.canUseTool( + 'Bash', + { command: 'ls' }, + permissionOptions('perm-3', 'tool-3', controller.signal) + ) + expect(control.prompts.find('perm-3')).not.toBeNull() + + controller.abort() + + await expect(answered).resolves.toBeNull() + expect(control.emit).toHaveBeenLastCalledWith( + expect.objectContaining({ type: 'prompt-cancelled', promptKey: 'perm-3' }) + ) + // Forgotten: a late answer can no longer find the prompt to authorize the wrong tool. + expect(control.prompts.find('perm-3')).toBeNull() + }) + + it('cancels a request whose abort raced ahead of delivery without emitting a prompt', async () => { + const control = callbacksFor() + const controller = new AbortController() + controller.abort() + + const answered = control.canUseTool( + 'Bash', + { command: 'ls' }, + permissionOptions('perm-4', 'tool-4', controller.signal) + ) + + await expect(answered).resolves.toBeNull() + expect(control.prompts.find('perm-4')).toBeNull() + expect(control.emit).toHaveBeenCalledTimes(1) + expect(control.emit).toHaveBeenCalledWith( + expect.objectContaining({ type: 'prompt-cancelled', promptKey: 'perm-4' }) + ) + }) + + it('settles every in-flight prompt with null when the registry is cleared', async () => { + const control = callbacksFor() + const first = control.canUseTool( + 'Bash', + { command: 'a' }, + permissionOptions('perm-5', 'tool-5', new AbortController().signal) + ) + const second = control.canUseTool( + 'Bash', + { command: 'b' }, + permissionOptions('perm-6', 'tool-6', new AbortController().signal) + ) + + // What session close does: settle each pending callback so no promise dangles. + for (const prompt of control.prompts.clear()) { + prompt.settle(null) + } + + await expect(first).resolves.toBeNull() + await expect(second).resolves.toBeNull() + }) + + it('answers a user dialog deny-safe', async () => { + const control = callbacksFor() + await expect( + control.onUserDialog( + { dialogKind: 'refusal_fallback_prompt', payload: {} }, + { signal: new AbortController().signal, requestId: 'dialog-1' } + ) + ).resolves.toEqual({ behavior: 'cancelled' }) + }) + + it('enumerates every blocking control request and wires a callback for each', () => { + // The stable surface of controls a turn can block on. Adding one here without wiring its + // callback below fails this test rather than silently leaving a control unhandled. + expect(new Set(Object.keys(CLAUDE_BLOCKING_CONTROL_CALLBACKS))).toEqual( + new Set([CLAUDE_CAN_USE_TOOL_SUBTYPE, CLAUDE_REQUEST_USER_DIALOG_SUBTYPE]) + ) + const callbacks = buildClaudePermissionCallbacks({ + sessionId: 'session-1', + prompts: new ClaudePromptRegistry(), + emit: vi.fn() + }) as unknown as Record + for (const callbackName of Object.values(CLAUDE_BLOCKING_CONTROL_CALLBACKS)) { + expect(typeof callbacks[callbackName], `${callbackName} must be wired`).toBe('function') + } + }) +}) diff --git a/src/main/claude/claude-structured-inbound-control.ts b/src/main/claude/claude-structured-inbound-control.ts new file mode 100644 index 00000000000..343e76d4ea5 --- /dev/null +++ b/src/main/claude/claude-structured-inbound-control.ts @@ -0,0 +1,91 @@ +import type { CanUseTool, OnUserDialog, PermissionResult } from '@anthropic-ai/claude-agent-sdk' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' + +export const CLAUDE_CAN_USE_TOOL_SUBTYPE = 'can_use_tool' +export const CLAUDE_REQUEST_USER_DIALOG_SUBTYPE = 'request_user_dialog' + +/** + * The blocking control requests Orca answers, each mapped to the SDK consumer callback that + * answers it. This is the stable surface a real turn can block on: `can_use_tool` through + * `canUseTool` and `request_user_dialog` through `onUserDialog`. Every other control-request + * subtype the SDK routes (elicitation, oauth/host token refresh, mcp_message, hook_callback) + * is either not surfaced to this consumer or fails closed inside the SDK; adding a new + * blocking control Orca must answer means adding its callback here, and the catalog test + * fails if a named callback is missing. + */ +export const CLAUDE_BLOCKING_CONTROL_CALLBACKS = { + [CLAUDE_CAN_USE_TOOL_SUBTYPE]: 'canUseTool', + [CLAUDE_REQUEST_USER_DIALOG_SUBTYPE]: 'onUserDialog' +} as const + +export type ClaudeBlockingControlSubtype = keyof typeof CLAUDE_BLOCKING_CONTROL_CALLBACKS + +export type ClaudePermissionCallbackDeps = { + sessionId: string + prompts: ClaudePromptRegistry + emit: (event: ClaudeStructuredSessionEvent) => void +} + +function denySafeResult(toolUseId: string | undefined): PermissionResult { + return { + behavior: 'deny', + message: 'Orca could not decode this permission request.', + ...(toolUseId ? { toolUseID: toolUseId } : {}) + } +} + +/** + * Build the SDK permission callbacks from the durable prompt registry. + * + * A decodable `can_use_tool` becomes a durable prompt whose `settle` resolves this callback; + * a malformed one is denied without registering. The SDK's abort signal fires on + * `control_cancel_request` (a cancelled turn), which forgets the prompt and settles it with + * `null` — never authorizing a tool. A late answer after abort finds no prompt and is refused + * by `answerClaudePrompt`. `onUserDialog` is deny-safe; the CLI only emits dialog kinds Orca + * declares in `supportedDialogKinds`, which is empty. + */ +export function buildClaudePermissionCallbacks(deps: ClaudePermissionCallbackDeps): { + canUseTool: CanUseTool + onUserDialog: OnUserDialog +} { + const canUseTool: CanUseTool = (toolName, input, options) => + new Promise((resolve) => { + const prompt = deps.prompts.register({ + requestId: options.requestId, + toolName, + toolUseId: options.toolUseID, + input, + suggestions: options.suggestions ?? [], + settle: resolve as (response: Record | null) => void + }) + if (!prompt) { + resolve(denySafeResult(options.toolUseID)) + return + } + const cancel = (): void => { + if (deps.prompts.forgetIfPending(prompt)) { + deps.emit({ + type: 'prompt-cancelled', + sessionId: deps.sessionId, + promptKey: prompt.promptKey + }) + // Null is the SDK's "no response written" sentinel: a cancelled request must not + // be answered, only forgotten. + resolve(null) + } + } + if (options.signal.aborted) { + // No abort event can still fire, so registering a listener would park the callback + // forever behind a prompt nothing will answer. + cancel() + return + } + options.signal.addEventListener('abort', cancel, { once: true }) + deps.emit({ type: 'prompt', sessionId: deps.sessionId, prompt }) + }) + + const onUserDialog: OnUserDialog = () => Promise.resolve({ behavior: 'cancelled' }) + + return { canUseTool, onUserDialog } +} diff --git a/src/main/claude/claude-structured-init-deadline.ts b/src/main/claude/claude-structured-init-deadline.ts new file mode 100644 index 00000000000..f3acd6c3af9 --- /dev/null +++ b/src/main/claude/claude-structured-init-deadline.ts @@ -0,0 +1,68 @@ +import type { ClaudeInitObservation } from './claude-structured-init-proof' +import { claudeInitializationAuthError } from './claude-structured-init-proof' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { AgentSessionAcquisitionRefusal } from '../native-chat/agent-session-wire/structured-agent-session-adapter' + +export type ClaudeInitDeadline = { + promise: Promise + resolve: (init: ClaudeInitObservation) => void + reject: (error: Error) => void + start: () => void + clear: () => void +} + +export function claudeInitTimeoutError( + sessionId: string, + timeoutMs: number +): AgentSessionAcquisitionRefusal { + return new AgentSessionAcquisitionRefusal( + `Claude did not finish starting session ${sessionId} within ${Math.ceil(timeoutMs / 1000)} seconds. Verify the selected Claude account is signed in and CLAUDE_CONFIG_DIR contains valid credentials, then retry; no SessionStart or system/init proof arrived.` + ) +} + +export async function requestClaudeInitialization( + connection: ClaudeStreamJsonConnection, + sessionId: string, + timeoutMs: number +): Promise { + try { + const result = await connection.initializationResult({ timeoutMs }) + const authError = claudeInitializationAuthError(result) + if (authError) { + throw authError + } + return result + } catch (error) { + if (error instanceof Error && error.message === 'claude initialize request timed out') { + throw claudeInitTimeoutError(sessionId, timeoutMs) + } + throw error + } +} + +export function createClaudeInitDeadline(sessionId: string, timeoutMs: number): ClaudeInitDeadline { + let resolve = (_init: ClaudeInitObservation): void => {} + let reject = (_error: Error): void => {} + const promise = new Promise((resolvePromise, rejectPromise) => { + resolve = resolvePromise + reject = rejectPromise + }) + void promise.catch(() => {}) + let timer: ReturnType | null = null + + return { + promise, + resolve, + reject, + start: () => { + timer = setTimeout(() => reject(claudeInitTimeoutError(sessionId, timeoutMs)), timeoutMs) + timer.unref?.() + }, + clear: () => { + if (timer) { + clearTimeout(timer) + timer = null + } + } + } +} diff --git a/src/main/claude/claude-structured-init-proof.ts b/src/main/claude/claude-structured-init-proof.ts new file mode 100644 index 00000000000..c29cb2d4715 --- /dev/null +++ b/src/main/claude/claude-structured-init-proof.ts @@ -0,0 +1,88 @@ +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import type { ClaudeAuthDiagnostic } from './claude-structured-session-state' +import { AgentSessionAcquisitionRefusal } from '../native-chat/agent-session-wire/structured-agent-session-adapter' + +export type ClaudeInitObservation = { + providerSessionId: string + uuid: string | null + /** The resolved model id the CLI reports it is running; only `system/init` carries it. */ + model: string | null + message: Record +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +export function readClaudeFrameString(source: Record, key: string): string | null { + const value = source[key] + return typeof value === 'string' && value.length > 0 ? value : null +} + +export function readClaudeInit(message: Record): ClaudeInitObservation | null { + const hookName = readClaudeFrameString(message, 'hook_name') + const isInit = message.type === 'system' && message.subtype === 'init' + const isSessionStart = + message.type === 'system' && + (message.subtype === 'hook_started' || message.subtype === 'hook_response') && + hookName?.startsWith('SessionStart:') === true + if (!isInit && !isSessionStart) { + return null + } + const providerSessionId = readClaudeFrameString(message, 'session_id') + return providerSessionId + ? { + providerSessionId, + uuid: isInit ? readClaudeFrameString(message, 'uuid') : null, + model: isInit ? readClaudeFrameString(message, 'model') : null, + message + } + : null +} + +export function readClaudeModels(initialization: unknown): unknown[] { + return isRecord(initialization) && Array.isArray(initialization.models) + ? initialization.models + : [] +} + +/** CLI capabilities advertised on the initialize result or the yielded system/init frame. */ +export function readClaudeCapabilities( + init: ClaudeInitObservation, + initialization: unknown +): string[] { + const fromResult = isRecord(initialization) ? initialization.capabilities : undefined + const fromFrame = init.message.capabilities + const source = Array.isArray(fromResult) ? fromResult : Array.isArray(fromFrame) ? fromFrame : [] + return source.filter((value): value is string => typeof value === 'string') +} + +export function claudeInitializationAuthError( + initialization: unknown +): AgentSessionAcquisitionRefusal | null { + const account = + isRecord(initialization) && isRecord(initialization.account) ? initialization.account : null + return readClaudeFrameString(account ?? {}, 'tokenSource') === 'none' + ? new AgentSessionAcquisitionRefusal( + 'Claude is not signed in for the selected account. Sign in with the Claude CLI for this CLAUDE_CONFIG_DIR, then retry.' + ) + : null +} + +export function claudeAuthDiagnostic( + init: ClaudeInitObservation, + settings: unknown +): ClaudeAuthDiagnostic { + const env = isRecord(settings) && isRecord(settings.env) ? settings.env : {} + const apiKeySource = readClaudeFrameString(init.message, 'apiKeySource') + const configured = (key: string): boolean => + (typeof env[key] === 'string' && (env[key] as string).trim().length > 0) || + Boolean(process.env[key]?.trim()) + return { + apiKeySourceConfigured: apiKeySource !== null && apiKeySource !== 'none', + baseUrlConfigured: configured('ANTHROPIC_BASE_URL'), + authTokenConfigured: configured('ANTHROPIC_AUTH_TOKEN'), + apiKeyConfigured: configured('ANTHROPIC_API_KEY'), + settingSources: CLAUDE_DEFAULT_SETTING_SOURCES + } +} diff --git a/src/main/claude/claude-structured-item-translation.ts b/src/main/claude/claude-structured-item-translation.ts new file mode 100644 index 00000000000..d86093ee0a5 --- /dev/null +++ b/src/main/claude/claude-structured-item-translation.ts @@ -0,0 +1,179 @@ +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentJournalMessageItem +} from '../../shared/agent-session-journal-types' +import type { NativeChatBlock } from '../../shared/native-chat-types' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' + +export type ClaudeMessageEnvelope = { + sessionId: string + uuid: string + role: 'assistant' | 'user' + content: unknown[] + /** Messages API id shared by every frame of one streamed assistant message. */ + messageId: string | null + parentToolUseId: string | null +} + +export type ClaudeToolUse = { id: string; name: string; input: unknown } +export type ClaudeToolResult = { toolUseId: string; output: string; failed: boolean } + +export function claudeRecord(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +export function claudeText(value: unknown): string | null { + return typeof value === 'string' && value.length > 0 ? value : null +} + +export function readClaudeMessageEnvelope( + frame: Record +): ClaudeMessageEnvelope | null { + if (frame.type !== 'assistant' && frame.type !== 'user') { + return null + } + const message = claudeRecord(frame.message) + const sessionId = claudeText(frame.session_id) + const uuid = claudeText(frame.uuid) + const role = message?.role + return sessionId && uuid && (role === 'assistant' || role === 'user') + ? { + sessionId, + uuid, + role, + content: messageContent(message?.content), + messageId: claudeText(message?.id), + parentToolUseId: claudeText(frame.parent_tool_use_id) + } + : null +} + +// A user replay may carry its text as a bare string (MessageParam), not blocks. +function messageContent(content: unknown): unknown[] { + if (Array.isArray(content)) { + return content + } + const text = claudeText(content) + return text ? [{ type: 'text', text }] : [] +} + +export function claudeMessageIdentity( + envelope: Pick +): AgentJournalItemIdentity { + return { provider: 'claude', sessionId: envelope.sessionId, uuid: envelope.uuid } +} + +function messageBlocks(envelope: ClaudeMessageEnvelope): NativeChatBlock[] { + const blocks: NativeChatBlock[] = [] + for (const value of envelope.content) { + const part = claudeRecord(value) + const text = claudeText(part?.text) + if (part?.type === 'text' && text) { + blocks.push({ type: 'text', text }) + continue + } + const source = claudeRecord(part?.source) + const url = claudeText(source?.url) + if (part?.type === 'image' && source?.type === 'url' && url) { + blocks.push({ type: 'image-ref', url }) + } + } + return blocks +} + +export function claudeMessageBody(envelope: ClaudeMessageEnvelope): AgentJournalMessageItem | null { + const blocks = messageBlocks(envelope) + return blocks.length > 0 ? { kind: 'message', role: envelope.role, blocks } : null +} + +export function claudeHasReplayContent(envelope: ClaudeMessageEnvelope): boolean { + return envelope.content.some((value) => { + const part = claudeRecord(value) + return part !== null && part.type !== 'tool_result' + }) +} + +export function claudeToolUses(envelope: ClaudeMessageEnvelope): ClaudeToolUse[] { + return envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const id = claudeText(part?.id) + const name = claudeText(part?.name) + return part?.type === 'tool_use' && id && name ? [{ id, name, input: part.input ?? null }] : [] + }) +} + +function resultText(value: unknown): string { + if (typeof value === 'string') { + return value + } + if (!Array.isArray(value)) { + return value === undefined ? '' : JSON.stringify(value) + } + return value + .flatMap((entry) => { + if (typeof entry === 'string') { + return [entry] + } + const part = claudeRecord(entry) + return part?.type === 'text' && typeof part.text === 'string' ? [part.text] : [] + }) + .join('\n') +} + +export function claudeToolResults(envelope: ClaudeMessageEnvelope): ClaudeToolResult[] { + return envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const toolUseId = claudeText(part?.tool_use_id) + return part?.type === 'tool_result' && toolUseId + ? [ + { + toolUseId, + output: resultText(part.content), + failed: part.is_error === true + } + ] + : [] + }) +} + +export function claudeThinkingText(envelope: ClaudeMessageEnvelope): string | null { + const parts = envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const thinking = claudeText(part?.thinking) + return part?.type === 'thinking' && thinking ? [thinking] : [] + }) + return parts.length > 0 ? parts.join('\n') : null +} + +export function claudeToolBody(input: { + tool: ClaudeToolUse + result?: ClaudeToolResult +}): AgentJournalItemBody { + return { + kind: 'tool-call', + name: input.tool.name, + input: input.tool.input, + state: input.result ? (input.result.failed ? 'failed' : 'completed') : 'running', + ...(input.result + ? { output: boundInlineText(input.result.output, DEFAULT_JOURNAL_PAYLOAD_LIMITS).bounded } + : {}) + } +} + +export function claudeStreamingMessageBody(text: string): AgentJournalMessageItem { + return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text }] } +} + +export function claudeToolIdentity(sessionId: string, toolUseId: string): AgentJournalItemIdentity { + return { provider: 'orca', clientMessageId: `claude-tool:${sessionId}:${toolUseId}` } +} + +export function claudeThinkingIdentity(sessionId: string, uuid: string): AgentJournalItemIdentity { + return { provider: 'orca', clientMessageId: `claude-thinking:${sessionId}:${uuid}` } +} diff --git a/src/main/claude/claude-structured-journal-translation.test.ts b/src/main/claude/claude-structured-journal-translation.test.ts new file mode 100644 index 00000000000..f403313dae8 --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation.test.ts @@ -0,0 +1,811 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentJournalRenderItem, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { activeStructuredAgentSessionTurnId } from '../../shared/structured-agent-session-projection' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { + createDeferredStructuredAgentSessionEventSink, + type StructuredAgentSessionEventSink +} from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import type { ClaudePendingPrompt } from './claude-structured-prompt-replies' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const tombstones: AgentJournalItemIdentity[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: (identity) => tombstones.push(identity), + publish: vi.fn() + } + return { sink, items, tombstones } +} + +function message( + type: 'assistant' | 'user', + uuid: string, + content: unknown[], + parentToolUseId: string | null = null +) { + return { + type: 'message' as const, + sessionId: 'orca-session', + ...(type === 'user' && parentToolUseId === null ? { startsTurn: true as const } : {}), + message: { + type, + uuid, + session_id: 'claude-session', + parent_tool_use_id: parentToolUseId, + message: { role: type, content } + } + } +} + +// Frames below follow the Claude Code 2.1.258 / SDK 0.3.251 partial-message +// cadence captured from the real CLI: every stream_event carries its own uuid, +// the final assistant frame for a block carries yet another, and only +// message.id ties them together. +function streamEvent(uuid: string, event: Record) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'stream_event', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + event + } + } +} + +function resultFrame(subtype: string, fields: Record) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'result', + subtype, + duration_ms: 1200, + duration_api_ms: 1100, + num_turns: 1, + session_id: 'claude-session', + uuid: `result-${subtype}`, + ...fields + } + } +} + +/** One streamed text turn in wire order: message_start, the block's start frame, + * one delta per chunk, the block's final assistant frame, the stop frames and + * the success result. */ +function streamedTextTurn(input: { + messageId: string + startUuid: string + finalUuid: string + chunks: string[] +}) { + const text = input.chunks.join('') + return { + start: [ + streamEvent(`${input.messageId}-message-start`, { + type: 'message_start', + message: { id: input.messageId, role: 'assistant', content: [] } + }), + streamEvent(input.startUuid, { + type: 'content_block_start', + index: 0, + content_block: { type: 'text', text: '' } + }) + ], + deltas: input.chunks.map((chunk, index) => + streamEvent(`${input.messageId}-delta-${index}`, { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: chunk } + }) + ), + final: { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'assistant', + uuid: input.finalUuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { + id: input.messageId, + role: 'assistant', + content: [{ type: 'text', text }], + stop_reason: null + } + } + }, + stop: [ + streamEvent(`${input.messageId}-block-stop`, { type: 'content_block_stop', index: 0 }), + streamEvent(`${input.messageId}-message-delta`, { + type: 'message_delta', + delta: { stop_reason: 'end_turn' } + }), + streamEvent(`${input.messageId}-message-stop`, { type: 'message_stop' }), + resultFrame('success', { + is_error: false, + result: text, + stop_reason: 'end_turn', + terminal_reason: 'completed' + }) + ], + text + } +} + +function assistantMessages(items: T[]): T[] { + return items.filter((item) => item.body.kind === 'message' && item.body.role === 'assistant') +} + +function providerFrameKinds(items: { body: AgentJournalItemBody }[]): string[] { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) +} + +const JOURNAL_IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'claude-session', leafUuid: 'leaf-1' } +} + +let journalRoot = '' + +beforeEach(async () => { + journalRoot = await mkdtemp(join(tmpdir(), 'orca-claude-journal-translation-')) +}) + +afterEach(async () => { + await rm(journalRoot, { recursive: true, force: true }) +}) + +describe('Claude structured journal translation', () => { + it('coalesces partial deltas onto the block identity and reconciles the final frame onto it', () => { + const state = sinkState() + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: state.sink, + schedule: (run, delay) => { + expect(delay).toBe(60) + scheduled = run + return () => { + scheduled = null + } + } + }) + const turn = streamedTextTurn({ + messageId: 'msg_01', + startUuid: 'block-start-1', + finalUuid: 'assistant-final-1', + chunks: ['ST', 'REAMOK_ELEC_64E632'] + }) + const streamedIdentity = { + provider: 'claude', + sessionId: 'claude-session', + uuid: 'block-start-1' + } + + for (const event of turn.start) { + translator.handle(event) + } + for (const delta of turn.deltas) { + translator.handle(delta) + } + expect(state.items).toEqual([]) + + const run = scheduled as (() => void) | null + run?.() + expect(state.items.at(-1)).toEqual({ + identity: streamedIdentity, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: turn.text }] } + }) + + translator.handle(turn.final) + for (const event of turn.stop) { + translator.handle(event) + } + const assistant = assistantMessages(state.items) + expect(assistant.at(-1)).toEqual({ + identity: streamedIdentity, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: turn.text }] } + }) + expect(new Set(assistant.map((item) => agentJournalItemKey(item.identity))).size).toBe(1) + expect(providerFrameKinds(state.items)).toEqual([]) + }) + + it('journals a count-to-200 stream as one assistant item carrying the complete reply', async () => { + const journal = await openAgentSessionJournal({ + identity: JOURNAL_IDENTITY, + journalDir: journalRoot, + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ journal, fence: 1, publish: vi.fn() }) + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: deferred.sink, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + const numbers = Array.from({ length: 200 }, (_, index) => String(index + 1)) + // The chunk boundaries the real CLI produced for this prompt. + const boundaries = [0, 1, 45, 93, 141, 189, 200] + const chunks = boundaries.slice(1).map((end, index) => { + const slice = numbers.slice(boundaries[index], end).join('\n') + return index === 0 ? slice : `\n${slice}` + }) + const turn = streamedTextTurn({ + messageId: 'msg_count', + startUuid: 'count-start', + finalUuid: 'count-final', + chunks + }) + + for (const event of turn.start) { + translator.handle(event) + } + for (const delta of turn.deltas) { + translator.handle(delta) + // Each chunk lands in its own coalescing window, as it did on the wire. + const run = scheduled as (() => void) | null + run?.() + } + translator.handle(turn.final) + for (const event of turn.stop) { + translator.handle(event) + } + await deferred.drained() + + const items: AgentJournalRenderItem[] = journal.snapshot().items + const assistant = assistantMessages(items) + expect(assistant.map((item) => item.itemId)).toEqual(['claude:claude-session:count-start']) + expect(assistant[0]?.body).toEqual({ + kind: 'message', + role: 'assistant', + blocks: [{ type: 'text', text: numbers.join('\n') }] + }) + expect(providerFrameKinds(items)).toEqual([]) + }) + + it('settles result frames, empty thinking and string user replays without painting a row', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + startsTurn: true, + message: { + type: 'user', + uuid: 'user-replay-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + timestamp: '2026-09-01T00:00:00.000Z', + message: { role: 'user', content: 'Reply with exactly PROBE_OK_1 and nothing else.' } + } + }) + translator.handle( + message('assistant', 'assistant-thinking-empty', [ + { type: 'thinking', thinking: '', signature: 'CAQS6QcKEAgRGAI4AUIIdGhpbmtpbmc' } + ]) + ) + translator.handle( + resultFrame('success', { + is_error: false, + result: 'PROBE_OK_1', + stop_reason: 'end_turn', + terminal_reason: 'completed' + }) + ) + translator.handle( + message('user', 'user-interrupt', [{ type: 'text', text: '[Request interrupted by user]' }]) + ) + translator.handle( + resultFrame('error_during_execution', { + is_error: true, + errors: ['[ede_diagnostic] result_type=user last_content_type=n/a stop_reason=null'], + stop_reason: null, + terminal_reason: 'aborted_streaming', + permission_denials: [] + }) + ) + translator.handle(message('user', 'control-only', [])) + + expect(providerFrameKinds(state.items)).toEqual([]) + expect( + state.items.flatMap((item) => + item.body.kind === 'message' && item.body.role === 'user' ? [item.body.blocks] : [] + ) + ).toEqual([ + [{ type: 'text', text: 'Reply with exactly PROBE_OK_1 and nothing else.' }], + [{ type: 'text', text: '[Request interrupted by user]' }] + ]) + expect( + state.items.some((item) => item.body.kind === 'status' && !item.body.turnLifecycle) + ).toBe(false) + expect( + state.tombstones.flatMap((identity) => + identity.provider === 'legacy' ? [identity.recordId] : [] + ) + ).toEqual(['turn-lifecycle:user-replay-1', 'turn-lifecycle:user-interrupt']) + }) + + it('does not reopen a completed turn when the SDK replays its user row after restart', () => { + const live = sinkState() + const liveTranslator = createClaudeJournalTranslator({ sink: live.sink }) + const replay = { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'user', + uuid: 'picker-command-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: '/model' } + } + } + + liveTranslator.handle({ ...replay, startsTurn: true }) + liveTranslator.handle(resultFrame('success', { is_error: false, result: '' })) + expect(live.tombstones).toContainEqual({ + provider: 'legacy', + agent: 'claude', + sessionId: 'claude-session', + recordId: 'turn-lifecycle:picker-command-1' + }) + liveTranslator.dispose() + + const restarted = sinkState() + const restartedTranslator = createClaudeJournalTranslator({ sink: restarted.sink }) + restartedTranslator.handle(replay) + + expect( + activeStructuredAgentSessionTurnId( + restarted.items.map((item, sequence) => ({ + itemId: agentJournalItemKey(item.identity), + revision: 1, + body: item.body, + sequence, + observedAt: sequence + })) + ) + ).toBeNull() + }) + + it('surfaces an API error carried by a success-subtype result with no assistant frame', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'user-1', [{ type: 'text', text: 'summarize this' }])) + // The SDK models this as a SUCCESS-subtype result whose `result` string is the + // user-facing API error. Suppressing it as ordinary turn bookkeeping ends the + // turn with nothing shown at all. + translator.handle( + resultFrame('success', { + is_error: true, + result: 'API Error: 529 upstream overloaded', + stop_reason: null, + terminal_reason: 'api_error' + }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:success']) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'status', + text: 'API Error: 529 upstream overloaded' + }) + // The turn still settles: the error is an extra row, not a stuck lifecycle. + expect( + state.tombstones.flatMap((identity) => + identity.provider === 'legacy' ? [identity.recordId] : [] + ) + ).toEqual(['turn-lifecycle:user-1']) + }) + + it('drops the stream state of turns that ended without their final frame', () => { + const state = sinkState() + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: state.sink, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + for (let turn = 0; turn < 3; turn += 1) { + const aborted = streamedTextTurn({ + messageId: `msg_abort_${turn}`, + startUuid: `abort-start-${turn}`, + finalUuid: `abort-final-${turn}`, + chunks: ['x'.repeat(4_000)] + }) + for (const event of [...aborted.start, ...aborted.deltas]) { + translator.handle(event) + } + const run = scheduled as (() => void) | null + run?.() + // The user interrupts: the result arrives with no final assistant frame, + // so nothing ever reconciles these blocks. + translator.handle( + resultFrame('error_during_execution', { + is_error: true, + terminal_reason: 'aborted_streaming' + }) + ) + // The partial text is already journaled; only the live state is dropped. + expect(translator.pendingStreamedBlocks).toBe(0) + } + + expect(assistantMessages(state.items)).toHaveLength(3) + }) + + it('keeps an ordinary successful result off the timeline', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(resultFrame('success', { is_error: false, result: 'done', errors: [] })) + + expect(providerFrameKinds(state.items)).toEqual([]) + }) + + it('surfaces the reason an error-subtype result stopped the turn', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + resultFrame('error_max_turns', { is_error: true, errors: ['turn limit reached'] }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:error_max_turns']) + }) + + it('keeps an unmodeled result subtype on the bounded provider fallback', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + resultFrame('error_from_the_future', { is_error: true, errors: ['budget exhausted'] }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:error_from_the_future']) + }) + + it('journals turn lifecycle and updates one tool row through its result', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'user-1', [{ type: 'text', text: 'List files' }])) + translator.handle( + message('assistant', 'assistant-tool', [ + { type: 'tool_use', id: 'tool-1', name: 'Bash', input: { command: 'ls' } } + ]) + ) + translator.handle( + message( + 'user', + 'tool-result-1', + [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'a.ts\nb.ts' }], + 'tool-1' + ) + ) + + const keyed = new Map( + state.items.map((item) => [agentJournalItemKey(item.identity), item.body]) + ) + expect(keyed.get('claude:claude-session:user-1')).toMatchObject({ + kind: 'message', + role: 'user' + }) + expect(keyed.get('orca:claude-tool%3Aclaude-session%3Atool-1')).toMatchObject({ + kind: 'tool-call', + name: 'Bash', + state: 'completed', + output: { head: 'a.ts\nb.ts', truncated: false } + }) + expect( + state.items.some( + (item) => item.body.kind === 'status' && item.body.turnLifecycle?.turnId === 'user-1' + ) + ).toBe(true) + + translator.handle( + message( + 'user', + 'tool-result-2', + [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'done again' }], + 'tool-1' + ) + ) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'tool-call', + name: 'tool', + input: null, + output: { head: 'done again' } + }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'result', session_id: 'claude-session', uuid: 'result-1' } + }) + expect(state.tombstones.at(-1)).toMatchObject({ + provider: 'legacy', + agent: 'claude', + recordId: 'turn-lifecycle:user-1' + }) + }) + + it('bounds persisted thinking text to the shared journal payload limit', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + const thinking = 'considering '.repeat(20_000) + + translator.handle(message('assistant', 'assistant-thinking', [{ type: 'thinking', thinking }])) + + expect(state.items.at(-1)?.body).toEqual({ + kind: 'status', + text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + }) + }) + + it('starts a cancellable lifecycle for image-only root user replays', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + message('user', 'user-image', [ + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'AA==' } } + ]) + ) + + expect(state.items.at(-1)?.body).toEqual({ + kind: 'status', + text: 'Claude is working…', + turnLifecycle: { turnId: 'user-image', state: 'running' } + }) + }) + + it('does not start a lifecycle for a top-level user tool result', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + message('user', 'tool-result-only', [ + { type: 'tool_result', tool_use_id: 'tool-1', content: 'done' } + ]) + ) + + expect(state.items.map((item) => agentJournalItemKey(item.identity))).toEqual([ + 'orca:claude-tool%3Aclaude-session%3Atool-1' + ]) + expect(state.items[0]?.body).toMatchObject({ + kind: 'tool-call', + state: 'completed', + output: { head: 'done' } + }) + expect( + state.items.some( + (item) => item.body.kind === 'status' && item.body.turnLifecycle !== undefined + ) + ).toBe(false) + }) + + it('paints nothing for a user frame that carries no content', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'control-only', [])) + + expect(state.items).toEqual([]) + expect(state.tombstones).toEqual([]) + }) + + it('renders unmodeled substantive Claude frames as bounded provider rows', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'local_command_output', summary: 'x'.repeat(100_000) } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'hook_response', hook_name: 'PostToolUse' } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'command_started', command: '/compact' } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'result', usage: { input_tokens: 12 }, total_cost_usd: 0.01 } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'tool_progress', tool_use_id: 'tool-1', elapsed_time_seconds: 2 } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'prompt_suggestion', suggestion: '/compact' } + }) + translator.handle( + message('user', 'attachment-1', [ + { type: 'document', source: { type: 'base64', media_type: 'application/pdf' } } + ]) + ) + translator.handle({ + type: 'provider-frame', + sessionId: 'orca-session', + kind: 'control_request:future_control', + payload: { subtype: 'future_control' } + }) + + const frames = state.items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame] : [] + ) + expect(frames.map((frame) => frame.kind)).toEqual( + expect.arrayContaining([ + 'message:system:local_command_output', + 'message:system:command_started', + 'message:result', + 'message:user:content:document', + 'control_request:future_control' + ]) + ) + expect(frames.map((frame) => frame.kind)).not.toEqual( + expect.arrayContaining([ + 'message:system:hook_response', + 'message:tool_progress', + 'message:prompt_suggestion' + ]) + ) + expect( + frames.find((frame) => frame.kind === 'message:system:local_command_output')?.payload + ).toEqual(expect.objectContaining({ truncated: true, byteLength: expect.any(Number) })) + }) + + it('preserves a question group as one addressable prompt and cancels it durably', () => { + const state = sinkState() + const bindings: unknown[][] = [] + const translator = createClaudeJournalTranslator({ + sink: state.sink, + bindPromptItemId: (...args) => bindings.push(args) + }) + const approval = prompt({ + requestId: 'permission-1', + promptKey: 'permission-1', + toolUseId: 'tool-1', + toolName: 'Bash', + kind: 'approval', + input: { command: 'git status' }, + questionIds: [] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: approval }) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'approval', + title: 'Allow Bash?', + options: expect.arrayContaining([{ id: 'allow', label: 'Allow' }]) + }) + expect(bindings[0]).toEqual([ + 'orca:claude-prompt%3Aorca-session%3Apermission-1', + 'permission-1' + ]) + + const questions = prompt({ + requestId: 'questions-1', + promptKey: 'questions-1', + toolUseId: 'tool-q', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { question: 'Library?', options: [{ label: 'Luxon' }] }, + { question: 'Ship?', options: [{ label: 'Yes' }] } + ] + }, + questionIds: ['Library?', 'Ship?'] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: questions }) + expect(state.items.filter((item) => item.body.kind === 'question')).toHaveLength(1) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'question', + questions: [ + { id: 'q1', question: 'Library?', multiSelect: false }, + { id: 'q2', question: 'Ship?', multiSelect: false } + ] + }) + expect(bindings.at(-1)).toEqual([ + 'orca:claude-prompt%3Aorca-session%3Aquestions-1', + 'questions-1' + ]) + + const multiSelect = prompt({ + requestId: 'questions-multi', + promptKey: 'questions-multi', + toolUseId: 'tool-multi', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { + question: 'Libraries?', + multiSelect: true, + options: [{ label: 'Luxon' }, { label: 'Temporal' }] + } + ] + }, + questionIds: ['Libraries?'] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: multiSelect }) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'question', + question: '1 grouped question from Claude', + options: [], + questions: [ + { + id: 'q1', + question: 'Libraries?', + multiSelect: true, + options: [{ label: 'Luxon' }, { label: 'Temporal' }], + freeTextQuestionId: 'q1' + } + ] + }) + + translator.handle({ + type: 'prompt-cancelled', + sessionId: 'orca-session', + promptKey: 'questions-1' + }) + expect(state.tombstones).toHaveLength(1) + }) +}) + +function prompt( + input: Pick< + ClaudePendingPrompt, + 'requestId' | 'promptKey' | 'toolUseId' | 'toolName' | 'kind' | 'input' | 'questionIds' + > +): ClaudePendingPrompt { + return { + ...input, + suggestions: [], + answers: new Map(), + settle: () => {} + } +} diff --git a/src/main/claude/claude-structured-journal-translation.ts b/src/main/claude/claude-structured-journal-translation.ts new file mode 100644 index 00000000000..ffaad4da570 --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation.ts @@ -0,0 +1,287 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-session-wire/agent-session-delta-coalescer' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' +import { + claudeMessageBody, + claudeMessageIdentity, + claudeHasReplayContent, + claudeRecord, + claudeStreamingMessageBody, + claudeText, + claudeThinkingIdentity, + claudeThinkingText, + claudeToolBody, + claudeToolIdentity, + claudeToolResults, + claudeToolUses, + readClaudeMessageEnvelope, + type ClaudeToolUse +} from './claude-structured-item-translation' +import { + claudeApprovalItem, + claudePromptIdentity, + claudeQuestionItems +} from './claude-structured-prompt-items' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { readableProviderFrameText } from '../native-chat/agent-session-wire/unhandled-provider-frame' +import { + CLAUDE_UNRENDERABLE_CONTENT_TEXT, + claudeProviderFrameKind, + claudeResultFailure, + createClaudeProviderFrameFallback, + isModeledClaudeContent, + isSettledClaudeResultKind +} from './claude-structured-provider-fallback' +import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity' +import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' + +export type ClaudeJournalTranslatorDeps = { + sink: StructuredAgentSessionEventSink + bindPromptItemId?: (journalItemId: string, promptKey: string, questionId?: string) => void + coalesceMs?: number + schedule?: AgentSessionDeltaCoalescerDeps['schedule'] + fallbackIdPrefix?: string +} + +export type ClaudeJournalTranslator = { + handle: (event: ClaudeStructuredSessionEvent) => void + flush: () => void + /** Streamed blocks still awaiting a final frame. A settled turn leaves none. */ + readonly pendingStreamedBlocks: number + dispose: () => void +} + +export function createClaudeSessionJournalTranslator( + sink: StructuredAgentSessionEventSink | undefined, + prompts: ClaudePromptRegistry, + fallbackIdPrefix: string +): ClaudeJournalTranslator | null { + return sink + ? createClaudeJournalTranslator({ + sink, + fallbackIdPrefix, + bindPromptItemId: (itemId, promptKey, questionId) => + prompts.bindJournalItemId(itemId, promptKey, questionId) + }) + : null +} + +function lifecycleIdentity(sessionId: string, turnId: string): AgentJournalItemIdentity { + return { + provider: 'legacy', + agent: 'claude', + sessionId, + recordId: `turn-lifecycle:${turnId}` + } +} + +export function createClaudeJournalTranslator( + deps: ClaudeJournalTranslatorDeps +): ClaudeJournalTranslator { + const tools = new Map() + const promptItems = new Map() + const streamedBlocks = createClaudeStreamedBlockRegistry() + let currentTurn: { sessionId: string; turnId: string } | null = null + const providerFallback = createClaudeProviderFrameFallback( + deps.sink, + deps.fallbackIdPrefix ?? 'acquisition' + ) + const streamedText = createClaudeStreamedTextCheckpoints({ + ...(deps.coalesceMs === undefined ? {} : { coalesceMs: deps.coalesceMs }), + ...(deps.schedule ? { schedule: deps.schedule } : {}), + persist: (identity, text) => { + deps.sink.appendItem(identity, claudeStreamingMessageBody(text)) + deps.sink.publish() + } + }) + + const publishLifecycle = (sessionId: string, turnId: string, running: boolean): void => { + const identity = lifecycleIdentity(sessionId, turnId) + if (running) { + deps.sink.appendItem(identity, { + kind: 'status', + text: 'Claude is working…', + turnLifecycle: { turnId, state: 'running' } + }) + } else { + deps.sink.appendTombstone(identity) + } + deps.sink.publish() + } + + const handleStream = (message: Record): boolean => { + const delta = streamedBlocks.observe(message) + if (!delta) { + return false + } + streamedText.append(delta.identity, delta.text) + return true + } + + const handleMessage = (message: Record, startsTurn: boolean): boolean => { + const envelope = readClaudeMessageEnvelope(message) + if (!envelope) { + return false + } + let changed = false + const body = claudeMessageBody(envelope) + // The final frame of a streamed block lands on the block's identity, not its own uuid. + const identity = + (body && envelope.role === 'assistant' ? streamedBlocks.reconcile(envelope) : null) ?? + claudeMessageIdentity(envelope) + streamedText.forget(agentJournalItemKey(identity)) + if (body) { + deps.sink.appendItem(identity, body) + changed = true + } + for (const tool of claudeToolUses(envelope)) { + tools.set(tool.id, tool) + deps.sink.appendItem( + claudeToolIdentity(envelope.sessionId, tool.id), + claudeToolBody({ tool }) + ) + changed = true + } + for (const result of claudeToolResults(envelope)) { + const tool = tools.get(result.toolUseId) ?? { + id: result.toolUseId, + name: 'tool', + input: null + } + deps.sink.appendItem( + claudeToolIdentity(envelope.sessionId, result.toolUseId), + claudeToolBody({ tool, result }) + ) + // Tool inputs are only needed until their matching result arrives. + tools.delete(result.toolUseId) + changed = true + } + const thinking = claudeThinkingText(envelope) + if (thinking) { + deps.sink.appendItem(claudeThinkingIdentity(envelope.sessionId, envelope.uuid), { + kind: 'status', + text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + }) + changed = true + } + const unhandledContent = envelope.content.filter((part) => !isModeledClaudeContent(part)) + for (const part of unhandledContent) { + const partType = claudeText(claudeRecord(part)?.type) ?? 'unknown' + providerFallback.append( + `message:${envelope.role}:content:${partType}`, + part, + readableProviderFrameText(part) ?? CLAUDE_UNRENDERABLE_CONTENT_TEXT + ) + changed = true + } + // An empty user frame is a replay with nothing to show, not an unknown kind. + if (envelope.content.length === 0 && envelope.role === 'assistant') { + providerFallback.append(`message:${envelope.role}:empty`, message) + changed = true + } + if ( + envelope.role === 'user' && + startsTurn && + claudeHasReplayContent(envelope) && + message.parent_tool_use_id === null + ) { + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + } + currentTurn = { sessionId: envelope.sessionId, turnId: envelope.uuid } + publishLifecycle(envelope.sessionId, envelope.uuid, true) + } + if (changed) { + deps.sink.publish() + } + return true + } + + const handlePrompt = (event: Extract): void => { + const identities: AgentJournalItemIdentity[] = [] + if (event.prompt.kind === 'question') { + for (const question of claudeQuestionItems({ + sessionId: event.sessionId, + prompt: event.prompt + })) { + identities.push(question.identity) + deps.sink.appendItem(question.identity, question.body) + deps.bindPromptItemId?.(agentJournalItemKey(question.identity), event.prompt.promptKey) + } + } else { + const identity = claudePromptIdentity({ + sessionId: event.sessionId, + promptKey: event.prompt.promptKey + }) + identities.push(identity) + deps.sink.appendItem(identity, claudeApprovalItem(event.prompt)) + deps.bindPromptItemId?.(agentJournalItemKey(identity), event.prompt.promptKey) + } + promptItems.set(event.prompt.promptKey, identities) + deps.sink.publish() + } + + return { + handle: (event) => { + if (event.type === 'ended') { + streamedText.flush() + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + currentTurn = null + } + return + } + if (event.type === 'message' && handleStream(event.message)) { + return + } + streamedText.flush() + if (event.type === 'prompt') { + handlePrompt(event) + } else if (event.type === 'prompt-cancelled') { + for (const identity of promptItems.get(event.promptKey) ?? []) { + deps.sink.appendTombstone(identity) + } + promptItems.delete(event.promptKey) + deps.sink.publish() + } else if (event.type === 'message' && event.message.type === 'result') { + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + currentTurn = null + } + // The turn is over. A block still awaiting its final keeps the text the + // flush above journaled, but its live state goes: an interrupted turn + // would otherwise retain that text for the life of the session. + streamedBlocks.clear() + streamedText.settle() + const kind = claudeProviderFrameKind(event.message) + // Ordinary turn bookkeeping stays suppressed; a reported failure never does. + const failure = claudeResultFailure(event.message) + if (failure || !isSettledClaudeResultKind(kind)) { + providerFallback.append(kind, event.message, failure?.text) + } + } else if (event.type === 'message') { + if (!handleMessage(event.message, event.startsTurn === true)) { + providerFallback.append(claudeProviderFrameKind(event.message), event.message) + } + } else if (event.type === 'provider-frame') { + providerFallback.append(event.kind, event.payload) + } + }, + flush: streamedText.flush, + get pendingStreamedBlocks() { + return streamedText.pending + }, + dispose: () => { + streamedText.dispose() + tools.clear() + promptItems.clear() + streamedBlocks.clear() + } + } +} diff --git a/src/main/claude/claude-structured-launch-resolution.test.ts b/src/main/claude/claude-structured-launch-resolution.test.ts new file mode 100644 index 00000000000..650947cffa1 --- /dev/null +++ b/src/main/claude/claude-structured-launch-resolution.test.ts @@ -0,0 +1,392 @@ +import { chmodSync, mkdtempSync, mkdirSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { delimiter, join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { claudeStructuredAuthPolicyForSettings } from '../claude-accounts/claude-structured-auth-policy' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' +import { + CLAUDE_DEFAULT_SETTING_SOURCES, + CLAUDE_STRUCTURED_BASE_OPTIONS, + claudeSdkOptionsForLaunchArgs, + claudeSessionIdForOrcaSession, + createClaudeStructuredLaunchResolver +} from './claude-structured-launch-resolution' + +const SESSION_ID = 'orca-session-1' +const IDENTITY = { sessionId: SESSION_ID } as Parameters< + ReturnType +>[0]['identity'] + +function record(overrides: Partial = {}): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [], + ...overrides + } as AgentSessionRecord +} + +function identityAt(leafUuid: string | null): typeof IDENTITY { + return { + ...IDENTITY, + providerHandle: { kind: 'claude', sessionId: 'provider-current', leafUuid } + } +} + +function makeExecutable(path: string): void { + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, '') + if (process.platform !== 'win32') { + chmodSync(path, 0o755) + } +} + +function resolverFor( + value: AgentSessionRecord | null, + resolveEnv?: () => Record, + stripAuthEnv = false +) { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => value } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv }), + ...(resolveEnv ? { resolveEnv } : {}) + }) +} + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +/** The normalized steady state of a Windows user whose only Claude account is WSL-managed: the + * prune drops the WSL account out of the host slot and persists that. */ +const WSL_ONLY_NORMALIZED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +const RESUMABLE = record({ + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: 'provider-current', leafUuid: 'leaf-current' } } + ] as AgentSessionRecord['providerHandleChain'] +}) + +describe('claude structured launch resolution', () => { + it('pre-mints a stable provider id and pins interactive setting sources', async () => { + const first = await resolverFor(record())({ identity: IDENTITY }) + const second = await resolverFor(record())({ identity: IDENTITY }) + + expect(first.providerSessionId).toBe(claudeSessionIdForOrcaSession(SESSION_ID)) + expect(second.providerSessionId).toBe(first.providerSessionId) + expect(first).toMatchObject({ + pathToClaudeCodeExecutable: '/usr/local/bin/claude', + cwd: '/repos/workspace-1', + claudeConfigDir: '/home/work/.claude', + resumeLeafUuid: null, + resumed: false + }) + expect(first.options).toEqual({ + includePartialMessages: true, + settingSources: [...CLAUDE_DEFAULT_SETTING_SOURCES], + supportedDialogKinds: [], + extraArgs: { 'replay-user-messages': null }, + systemPrompt: { type: 'preset', preset: 'claude_code' }, + sessionId: first.providerSessionId + }) + expect(first.options.resume).toBeUndefined() + expect(CLAUDE_STRUCTURED_BASE_OPTIONS.includePartialMessages).toBe(true) + }) + + it('resumes the session and leaf at the durable chain head', async () => { + const launch = await resolverFor( + record({ + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: 'provider-old', leafUuid: 'leaf-old' } }, + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: 'leaf-current' + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + )({ identity: identityAt('leaf-current') }) + + expect(launch).toMatchObject({ + providerSessionId: 'provider-current', + resumeLeafUuid: 'leaf-current', + resumed: true + }) + expect(launch.options.resume).toBe('provider-current') + expect(launch.options.resumeSessionAt).toBe('leaf-current') + expect(launch.options.sessionId).toBeUndefined() + }) + + it('refuses a durable journal leaf that diverged before resume resolution', async () => { + const resolve = resolverFor( + record({ + providerHandleChain: [ + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: 'leaf-current' + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + ) + + await expect(resolve({ identity: identityAt('leaf-stale') })).rejects.toThrow( + 'durable resume identity changed before spawn' + ) + }) + + it('keeps session-only resume when the durable handle has no leaf', async () => { + const launch = await resolverFor( + record({ + providerHandleChain: [ + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: null + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + )({ identity: identityAt(null) }) + + expect(launch.options.resume).toBe('provider-current') + expect(launch.options.resumeSessionAt).toBeUndefined() + }) + + it('preserves durable Claude launch arguments as typed options and extraArgs', async () => { + const launch = await resolverFor( + record({ + launchArgs: [ + '--model', + 'claude-sonnet-4-5', + '--effort', + 'high', + '--dangerously-skip-permissions' + ] + }) + )({ identity: IDENTITY }) + + expect(launch.options.model).toBe('claude-sonnet-4-5') + expect(launch.options.effort).toBe('high') + expect(launch.options.extraArgs).toEqual({ + 'dangerously-skip-permissions': null, + 'replay-user-messages': null + }) + }) + + it('routes durable launch arguments to a typed option first and refuses what neither can carry', () => { + // The catalog's own output: each flag lands in exactly one place, so the SDK + // cannot emit it twice with two different values. + expect(claudeSdkOptionsForLaunchArgs(['--model', 'opus', '--effort', 'xhigh'])).toEqual({ + model: 'opus', + effort: 'xhigh' + }) + // An effort the SDK's union does not name still reaches the CLI, unchanged. + expect(claudeSdkOptionsForLaunchArgs(['--effort', 'ultra'])).toEqual({ + extraArgs: { effort: 'ultra' } + }) + expect(claudeSdkOptionsForLaunchArgs(['--settings=/tmp/s.json'])).toEqual({ + extraArgs: { settings: '/tmp/s.json' } + }) + expect(() => claudeSdkOptionsForLaunchArgs(['-m', 'opus'])).toThrow(/no SDK option/) + }) + + it('keeps the session launch environment pinned after account settings change', async () => { + const resolver = resolverFor(record(), () => ({ + ANTHROPIC_AUTH_TOKEN: 'rotated-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + })) + + expect((await resolver({ identity: IDENTITY })).env).toMatchObject({ + ANTHROPIC_AUTH_TOKEN: 'rotated-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + }) + expect((await resolver({ identity: IDENTITY })).env?.ANTHROPIC_AUTH_TOKEN).toBe('rotated-token') + }) + + // Stripping is the managed-account rule the terminal preflight computes at + // runtime-auth-preparation.ts:72; claude-structured-auth-parity.test.ts covers + // the system-auth half, where the user's own key has to survive. + it('strips ambient Anthropic auth under a managed account but keeps the rest of the env', async () => { + const restore = { + ANTHROPIC_API_KEY: process.env.ANTHROPIC_API_KEY, + ANTHROPIC_AUTH_TOKEN: process.env.ANTHROPIC_AUTH_TOKEN, + CLAUDE_CODE_OAUTH_TOKEN: process.env.CLAUDE_CODE_OAUTH_TOKEN, + ORCA_LAUNCH_RESOLUTION_MARKER: process.env.ORCA_LAUNCH_RESOLUTION_MARKER + } + process.env.ANTHROPIC_API_KEY = 'sk-ant-SHELL-LEAK' + process.env.ANTHROPIC_AUTH_TOKEN = 'tok-SHELL-LEAK' + process.env.CLAUDE_CODE_OAUTH_TOKEN = 'oauth-SHELL-LEAK' + process.env.ORCA_LAUNCH_RESOLUTION_MARKER = 'inherited' + try { + const launch = await resolverFor(record(), undefined, true)({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBeUndefined() + expect(launch.env?.ANTHROPIC_AUTH_TOKEN).toBeUndefined() + expect(launch.env?.CLAUDE_CODE_OAUTH_TOKEN).toBeUndefined() + // The inherited env is still the base — only auth is removed from it. + expect(launch.env?.ORCA_LAUNCH_RESOLUTION_MARKER).toBe('inherited') + expect(launch.env?.PATH ?? launch.env?.Path).toBeTruthy() + } finally { + for (const [key, value] of Object.entries(restore)) { + if (value === undefined) { + delete process.env[key] + } else { + process.env[key] = value + } + } + } + }) + + it('lets an explicit Claude env overlay override ambient auth under system auth', async () => { + const restore = process.env.ANTHROPIC_API_KEY + process.env.ANTHROPIC_API_KEY = 'sk-ant-SHELL-LEAK' + try { + const launch = await resolverFor(record(), () => ({ + ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' + }))({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-CONFIGURED') + } finally { + if (restore === undefined) { + delete process.env.ANTHROPIC_API_KEY + } else { + process.env.ANTHROPIC_API_KEY = restore + } + } + }) + + it('pairs a resolved Claude CLI with its sibling Node runtime', async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-claude-launch-')) + const binDir = join(root, 'bin') + const claudeCommand = join(binDir, process.platform === 'win32' ? 'claude.cmd' : 'claude') + const nodeCommand = join(binDir, process.platform === 'win32' ? 'node.cmd' : 'node') + makeExecutable(claudeCommand) + makeExecutable(nodeCommand) + + const launch = await createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record() } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => claudeCommand, + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ + PATH: '/usr/bin', + CLAUDE_CONFIG_DIR: '/accounts/selected/home' + }) + })({ identity: IDENTITY }) + + expect((launch.env?.PATH ?? launch.env?.Path)?.split(delimiter)[0]).toBe(binDir) + }) + + it('refuses other hosts, WSL, providers, and account-home variables', async () => { + await expect( + resolverFor(record({ location: { ...record().location, executionHostId: 'ssh:build' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/local host/) + await expect( + resolverFor(record({ location: { ...record().location, wslDistro: 'Ubuntu' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/local host/) + await expect( + resolverFor(record({ provider: 'codex' } as Partial))({ + identity: IDENTITY + }) + ).rejects.toThrow(/codex session/) + await expect( + resolverFor(record({ accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/CLAUDE_CONFIG_DIR/) + }) + + /** The account state can change while a session lives, and a reacquire after an unexpected child + * exit re-resolves the launch. Without the gate here, that reacquire spawns under whatever the + * account state has become. */ + describe('managed-account gate on every acquisition', () => { + function resolverWithGate(read: () => ClaudeManagedAccountGateSettings | null) { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => RESUMABLE } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + // Derived, not a literal: the gate and the policy must read the SAME account state, so a + // hardcoded value could assert a pairing production cannot produce. + resolveAuthPolicy: () => { + const settings = read() + if (!settings) { + throw new Error('the gate refuses before the auth policy is computed') + } + return claudeStructuredAuthPolicyForSettings(settings) + }, + readManagedAccountGate: read + }) + } + + it('refuses a reacquire once the account state becomes the refused shape', async () => { + let gate: ClaudeManagedAccountGateSettings | null = HOST_SELECTED + const resolve = resolverWithGate(() => gate) + + // Created while supported: the launch resolves and would spawn. + await expect(resolve({ identity: identityAt('leaf-current') })).resolves.toMatchObject({ + providerSessionId: 'provider-current' + }) + + gate = WSL_ONLY_NORMALIZED + + // Reacquire after the account state changed: refused before anything spawns. + await expect(resolve({ identity: identityAt('leaf-current') })).rejects.toBeInstanceOf( + AgentSessionPreSpawnError + ) + }) + + it('fails closed when the account state cannot be read', async () => { + await expect( + resolverWithGate(() => null)({ identity: identityAt('leaf-current') }) + ).rejects.toBeInstanceOf(AgentSessionPreSpawnError) + }) + + it('keeps resolving when no gate is wired, so other embedders are unaffected', async () => { + await expect( + resolverFor(RESUMABLE)({ identity: identityAt('leaf-current') }) + ).resolves.toMatchObject({ providerSessionId: 'provider-current' }) + }) + }) +}) diff --git a/src/main/claude/claude-structured-launch-resolution.ts b/src/main/claude/claude-structured-launch-resolution.ts new file mode 100644 index 00000000000..4f28f14ad65 --- /dev/null +++ b/src/main/claude/claude-structured-launch-resolution.ts @@ -0,0 +1,273 @@ +import { createHash } from 'node:crypto' +import type { EffortLevel, Options as ClaudeAgentSdkOptions } from '@anthropic-ai/claude-agent-sdk' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { agentSessionProviderHandleChainHead } from '../../shared/agent-session-provider-handle' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + applyClaudeEnvPatch, + hasClaudeAuthEnvConflict +} from '../claude-accounts/environment' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { + CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS, + whenClaudeAuthSwitchSettles +} from '../claude-accounts/live-pty-gate' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' +import { resolveClaudeCommand } from '../codex-cli/command' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' + +export const CLAUDE_DEFAULT_SETTING_SOURCES = ['user', 'project', 'local'] as const + +export type ClaudeStructuredSdkOptions = Pick< + ClaudeAgentSdkOptions, + | 'includePartialMessages' + | 'systemPrompt' + | 'settingSources' + | 'supportedDialogKinds' + | 'extraArgs' + | 'model' + | 'effort' + | 'sessionId' + | 'resume' + | 'resumeSessionAt' +> + +/** + * The options translation of the flags this transport used to build by hand. + * + * `-p`, `--input-format`, `--output-format` and `--verbose` are implied by + * `query()`; `--permission-prompt-tool stdio` is emitted because a `canUseTool` + * callback is supplied. `--replay-user-messages` has no option — the SDK never + * emits it — and Orca's send acknowledgement depends on the replay. + */ +export const CLAUDE_STRUCTURED_BASE_OPTIONS: ClaudeStructuredSdkOptions = { + includePartialMessages: true, + // Keep the SDK on Claude Code's own system-prompt contract. + systemPrompt: { type: 'preset', preset: 'claude_code' }, + settingSources: [...CLAUDE_DEFAULT_SETTING_SOURCES], + supportedDialogKinds: [], + extraArgs: { 'replay-user-messages': null } +} + +const EFFORT_LEVELS: readonly string[] = ['low', 'medium', 'high', 'xhigh', 'max'] + +function cloneDefinedEnv(env: NodeJS.ProcessEnv | Record): Record { + const next: Record = {} + for (const [key, value] of Object.entries(env)) { + if (value !== undefined) { + next[key] = value + } + } + return next +} + +/** + * Translate the record's durable launch arguments into SDK options. + * + * Typed option first so a flag is never emitted twice; `extraArgs` carries + * anything without one. A token expressible neither way is refused rather than + * dropped — a silent drop is how this lane loses launch flags. + */ +export function claudeSdkOptionsForLaunchArgs( + args: readonly string[] +): Pick { + let model: string | undefined + let effort: EffortLevel | undefined + const extraArgs: Record = {} + for (let index = 0; index < args.length; index += 1) { + const token = args[index] ?? '' + if (!token.startsWith('--') || token.length <= 2) { + throw new Error( + `claude launch argument ${token} has no SDK option; refusing rather than dropping it` + ) + } + const equals = token.indexOf('=') + const flag = equals === -1 ? token : token.slice(0, equals) + let value = equals === -1 ? null : token.slice(equals + 1) + if (value === null) { + const next = args[index + 1] + if (next !== undefined && !next.startsWith('-')) { + value = next + index += 1 + } + } + if (flag === '--model' && value !== null) { + model = value + } else if (flag === '--effort' && value !== null && EFFORT_LEVELS.includes(value)) { + effort = value as EffortLevel + } else { + extraArgs[flag.slice(2)] = value + } + } + return { + ...(model === undefined ? {} : { model }), + ...(effort === undefined ? {} : { effort }), + ...(Object.keys(extraArgs).length > 0 ? { extraArgs } : {}) + } +} + +export type ClaudeStructuredLaunch = { + /** Always Orca's resolved user CLI: the SDK's bundled binaries are excluded from the install. */ + pathToClaudeCodeExecutable: string + options: ClaudeStructuredSdkOptions + cwd: string + env?: Record + claudeConfigDir: string + providerSessionId: string + resumeLeafUuid: string | null + resumed: boolean +} + +export type ClaudeStructuredLaunchResolverDeps = { + store: AgentSessionRecordStore + resolveWorkspacePath: (workspaceId: string) => Promise + resolveCommand?: () => string + resolveEnv?: () => + | Promise | undefined> + | Record + | undefined + /** + * Required, and deliberately not defaulted. `stripAuthEnv` used to be a literal + * `true` here, so a missing dependency could not under-strip. Now it can, and the + * failure is silent — so every caller states the account's policy rather than + * inherit a guess. Build it with claudeStructuredAuthPolicyForSettings. + */ + resolveAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + /** How long an in-flight account switch may hold a launch before it is refused. */ + authSwitchSettleTimeoutMs?: number + /** Account state for the managed-account gate; null when it cannot be read, which refuses. */ + readManagedAccountGate?: () => ClaudeManagedAccountGateSettings | null +} + +/** + * Wait a running account switch out, and refuse only if it never settles. + * + * Launch resolution is reached from `acquireClaudeSession` *after* the old child has + * been closed and proved, so a plain refusal here would leave the user with a dead + * chat and no replacement — the very harm the acquire-entry guard exists to prevent. + * The entry guard still refuses outright, because nothing has been torn down yet. + */ +export async function assertClaudeAuthSwitchSettled( + timeoutMs = CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS +): Promise { + if (!(await whenClaudeAuthSwitchSettles(timeoutMs))) { + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + } +} + +export function claudeSessionIdForOrcaSession(sessionId: string): string { + const bytes = createHash('sha256').update(`orca-claude:${sessionId}`).digest().subarray(0, 16) + bytes[6] = ((bytes[6] ?? 0) & 0x0f) | 0x40 + bytes[8] = ((bytes[8] ?? 0) & 0x3f) | 0x80 + const hex = bytes.toString('hex') + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}` +} + +export function createClaudeStructuredLaunchResolver( + deps: ClaudeStructuredLaunchResolverDeps +): (input: { identity: AgentSessionJournalIdentity }) => Promise { + return async ({ identity }) => { + await assertClaudeAuthSwitchSettled(deps.authSwitchSettleTimeoutMs) + const record = deps.store.getRecord(identity.sessionId) + if (!record) { + throw new Error(`no durable agent-session record for ${identity.sessionId}`) + } + if (record.provider !== 'claude') { + throw new Error(`session ${identity.sessionId} is a ${record.provider} session`) + } + if ( + record.location.executionHostId !== LOCAL_EXECUTION_HOST_ID || + record.location.wslDistro !== null + ) { + throw new Error( + `claude structured sessions run on the local host, not ${record.location.executionHostId}` + ) + } + if (record.accountHome.variable !== 'CLAUDE_CONFIG_DIR') { + throw new Error(`claude sessions pin CLAUDE_CONFIG_DIR, not ${record.accountHome.variable}`) + } + // Every acquisition, not just the first: the account state can change under a live session, and + // a reacquire after an unexpected exit would otherwise spawn under whatever it has become. + // Codex has no gate here — it resolves its account on a different path. + if ( + deps.readManagedAccountGate && + !structuredClaudeMatchesActiveManagedAccount(deps.readManagedAccountGate()) + ) { + throw new AgentSessionPreSpawnError( + 'structured Claude is not offered under the active managed Claude account' + ) + } + const head = agentSessionProviderHandleChainHead(record.providerHandleChain) + if ( + head?.handle.provider === 'claude' && + (identity.providerHandle.kind !== 'claude' || + identity.providerHandle.sessionId !== head.handle.sessionId || + identity.providerHandle.leafUuid !== head.handle.leafUuid) + ) { + throw new Error('claude durable resume identity changed before spawn') + } + const providerSessionId = + head?.handle.provider === 'claude' + ? head.handle.sessionId + : claudeSessionIdForOrcaSession(identity.sessionId) + const durable = claudeSdkOptionsForLaunchArgs(record.launchArgs ?? []) + const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const auth = await deps.resolveAuthPolicy() + const overlay = await deps.resolveEnv?.() + // A switch can begin while the policy and overlay resolve, exactly as it can + // during the terminal preflight's prepareClaudeAuth — recheck after the awaits. + await assertClaudeAuthSwitchSettled(deps.authSwitchSettleTimeoutMs) + // Under a managed account the pinned credential is the only auth this launch may + // use, so an explicit override is refused rather than silently beating the pin. + if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(overlay)) { + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + } + // Why the overlay merges onto the inherited env rather than replacing it: the child + // still needs PATH and the rest of the shell environment, and withCliRuntimeOnPath + // derives PATH from what it is handed. Ambient Anthropic auth is stripped from the + // inherited half only when a managed account owns the credential; a system-auth + // user's own key is their sign-in and must reach the child. + const env = withCliRuntimeOnPath( + command, + { + ...applyClaudeEnvPatch( + cloneDefinedEnv(process.env), + {}, + { + stripAuthEnv: auth.stripAuthEnv, + platform: process.platform + } + ), + ...(overlay ? cloneDefinedEnv(overlay) : {}) + }, + { platform: process.platform } + ) + return { + pathToClaudeCodeExecutable: command, + options: { + ...durable, + ...CLAUDE_STRUCTURED_BASE_OPTIONS, + extraArgs: { ...durable.extraArgs, ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs }, + ...(head?.handle.provider === 'claude' + ? { + resume: providerSessionId, + ...(head.handle.leafUuid === null ? {} : { resumeSessionAt: head.handle.leafUuid }) + } + : { sessionId: providerSessionId }) + }, + cwd: await deps.resolveWorkspacePath(record.location.workspaceId), + env, + claudeConfigDir: record.accountHome.path, + providerSessionId, + resumeLeafUuid: head?.handle.provider === 'claude' ? head.handle.leafUuid : null, + resumed: head?.handle.provider === 'claude' + } + } +} diff --git a/src/main/claude/claude-structured-location-support.test.ts b/src/main/claude/claude-structured-location-support.test.ts new file mode 100644 index 00000000000..1106667d544 --- /dev/null +++ b/src/main/claude/claude-structured-location-support.test.ts @@ -0,0 +1,91 @@ +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + __setWindowsProcessTreeLoaderForTests, + resetWindowsProcessTableForTests +} from '../windows/windows-process-table' +import { supportsClaudeStructuredLocation } from './claude-structured-location-support' + +function setPlatform(platform: NodeJS.Platform): PropertyDescriptor | undefined { + const previous = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: platform }) + return previous +} + +describe('supportsClaudeStructuredLocation', () => { + let previousPlatform: PropertyDescriptor | undefined + + beforeEach(() => { + previousPlatform = setPlatform('darwin') + __setWindowsProcessTreeLoaderForTests() + }) + + afterEach(() => { + __setWindowsProcessTreeLoaderForTests() + resetWindowsProcessTableForTests() + if (previousPlatform) { + Object.defineProperty(process, 'platform', previousPlatform) + } + }) + + it('allows local non-WSL locations on macOS and Linux', () => { + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(true) + }) + + it('rejects Windows local locations until creation-time proof is available', () => { + previousPlatform = setPlatform('win32') + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2 }, + getAllProcesses: () => undefined + })) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + }) + + it('accepts Windows local locations once creation-time proof is available', () => { + previousPlatform = setPlatform('win32') + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2, CreationTime: 4 }, + getAllProcesses: () => undefined + })) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(true) + }) + + it('rejects WSL and remote locations', () => { + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: 'Ubuntu', + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'runtime:env-1', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + }) +}) diff --git a/src/main/claude/claude-structured-location-support.ts b/src/main/claude/claude-structured-location-support.ts new file mode 100644 index 00000000000..9c784a91325 --- /dev/null +++ b/src/main/claude/claude-structured-location-support.ts @@ -0,0 +1,11 @@ +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { isWindowsProcessStartTimeAvailable } from '../windows/windows-process-table' + +export function supportsClaudeStructuredLocation(location: AgentSessionExecutionLocation): boolean { + return ( + location.executionHostId === LOCAL_EXECUTION_HOST_ID && + location.wslDistro === null && + (process.platform !== 'win32' || isWindowsProcessStartTimeAvailable()) + ) +} diff --git a/src/main/claude/claude-structured-model-confirmation.test.ts b/src/main/claude/claude-structured-model-confirmation.test.ts new file mode 100644 index 00000000000..7bd8f629119 --- /dev/null +++ b/src/main/claude/claude-structured-model-confirmation.test.ts @@ -0,0 +1,209 @@ +import { describe, expect, it } from 'vitest' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { PROVIDER_SESSION_ID, acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim rows from Claude Code 2.1.258's list_models response. */ +const CATALOG = [ + { + value: 'default', + resolvedModel: 'claude-opus-5[1m]', + displayName: 'Default (recommended)', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } +] + +function initFrame(model: string): Record { + // Keys mirror the real per-turn system/init frame: it carries `model` as the + // resolved id, and no effort of any kind. + return { + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION_ID, + uuid: 'turn-init-uuid', + model, + apiKeySource: 'none' + } +} + +describe('Claude model confirmation', () => { + it('adopts the model a later turn reports when nothing was set since', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet' } + }) + + // The CLI's own report of what it is running — the only channel that carries + // it, since set_model answers success for a model it never resolves. + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-haiku-4-5-20251001')) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'haiku' } + }) + }) + + it('keeps a just-set model until the next turn reports one', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // No turn has run, so the acquisition-time report is older than the write and + // must not flip the pill back to the model the session started on. + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'haiku' } + }) + }) + + it('corrects the record when the turn runs a different model than was set', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet' } + }) + }) + + it('guards an effort against the model the turn reported, not the one that was set', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + // set_model answered success for a model it never resolved; the turn runs sonnet. + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + // The picker offers sonnet's levels, so refusing one under haiku — a model the + // pill does not show and the child is not running — is the false positive. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + ).resolves.toMatchObject({ effort: 'high' }) + }) + + it('keeps guarding against the reported model across a second effort write', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + + // The effort write bumps the option fence but does not change what the child + // runs, so the sonnet report is still current and still governs the guard. + // `max` skips the settings readback by contract, so only the catalog gates it: + // sonnet advertises it, haiku advertises no effort control at all. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'max', fence: 7 }) + ).resolves.toMatchObject({ effort: 'max' }) + }) + + it('guards an effort against a just-set model no turn has reported yet', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // The acquisition-time report predates the write, so haiku — which advertises + // no effort control — is still the model the guard must answer for. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + ).rejects.toThrow('claude model haiku does not accept effort high') + }) + + it('stops vouching for a confirmed effort once the model changes', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } + }) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // The readback was taken under sonnet; nothing has reported haiku holding it. + const options = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(options.current.effort).toBe('high') + expect(options.current.confirmed).toBeUndefined() + }) +}) + +describe('Claude effort the settings readback cannot report', () => { + function sessionWith( + reported: string, + calls: string[] = [] + ): { session: ClaudeSession; calls: string[] } { + return { + session: { + options: new Map([['model', 'sonnet']]), + reportedOptions: {}, + optionMutationSequence: 0, + confirmedOptions: new Set(), + connection: { + supportedModels: async () => { + calls.push('list_models') + return CATALOG + }, + applyFlagSettings: async (settings: { effortLevel?: string }) => { + calls.push(`apply:${settings.effortLevel}`) + }, + getSettings: async () => { + calls.push('get_settings') + return { + applied: { effort: reported }, + effective: { effortLevel: reported }, + sources: {} + } + } + } + } as unknown as ClaudeSession, + calls + } + } + + it('records a session-scoped effort the persisted settings never carry', async () => { + // `max` applies for the session and is deliberately excluded from the + // persisted effortLevel, so the readback reporting `high` is an absence of + // evidence, not a refusal — and the CLI offers `max` in its own catalog. + const { session, calls } = sessionWith('high') + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'max' }) + expect(calls).toEqual(['list_models', 'apply:max']) + }) +}) diff --git a/src/main/claude/claude-structured-option-confirmation.test.ts b/src/main/claude/claude-structured-option-confirmation.test.ts new file mode 100644 index 00000000000..ca7b8b70f1c --- /dev/null +++ b/src/main/claude/claude-structured-option-confirmation.test.ts @@ -0,0 +1,193 @@ +import { describe, expect, it } from 'vitest' +import { + applyStructuredAgentSessionOptions, + createStructuredAgentSessionOptionState, + structuredAgentSessionOptionSnapshot +} from '../../shared/structured-agent-session-options' +import { CLAUDE_SESSION_OPTION_CATALOG } from '../../shared/agent-session-option-catalog-claude-codex' +import type { AgentSessionOptionsResult } from '../../shared/agent-session-wire' +import type { SessionOptionDescriptor } from '../../shared/native-chat-session-options' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { PROVIDER_SESSION_ID, acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim rows from Claude Code 2.1.260's list_models response: `haiku` really + * does omit both effort keys, which is what makes an effort under it refusable. */ +const CATALOG = [ + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } +] + +function initFrame(model: string): Record { + return { + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION_ID, + uuid: 'turn-init-uuid', + model, + apiKeySource: 'none' + } +} + +function modelPill(result: AgentSessionOptionsResult): SessionOptionDescriptor | undefined { + const state = applyStructuredAgentSessionOptions( + createStructuredAgentSessionOptionState('claude'), + CLAUDE_SESSION_OPTION_CATALOG, + result + ) + return structuredAgentSessionOptionSnapshot(state).find((d) => d.category === 'model') +} + +/** Provenance the record keeps. Nothing renders it — the pill shows the value + * either way, and a report that disagrees is what corrects it. */ +function modelSource(result: AgentSessionOptionsResult): string | undefined { + return modelPill(result)?.valueSource +} + +function modelValue(result: AgentSessionOptionsResult): string | undefined { + const kind = modelPill(result)?.kind + return kind?.type === 'select' ? kind.currentValue : undefined +} + +describe('structured option confirmation reaches the pill', () => { + it('shows a just-set model before any turn reports it', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed ?? []).not.toContain('model') + expect(modelSource(result)).toBe('dispatched') + }) + + it('marks the model reported once the provider names it back', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-haiku-4-5-20251001')) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed).toContain('model') + expect(modelSource(result)).toBe('reported') + }) + + it('records an effort the readback could not take without confirming it', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + settings: { applied: {}, effective: {}, sources: {} }, + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + // `max` is session-scoped and absent from the persisted settings, so it records + // without a readback — recorded, never vouched for. + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'max', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.effort).toBe('max') + expect(result.current.confirmed ?? []).not.toContain('effort') + }) + + it('confirms an effort the readback agreed with', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + settings: { applied: { effort: 'low' }, effective: { effortLevel: 'low' }, sources: {} }, + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'low', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed).toContain('effort') + }) + + it('treats a host that reports no confirmation as unconfirmed', () => { + // Wire compatibility: an older host omits `confirmed` entirely. Absence must + // read as unconfirmed provenance, and the pill still shows the host's value. + const result = { + models: [{ id: 'haiku', label: 'Haiku', isDefault: false, efforts: [] }], + current: { model: 'haiku' } + } + expect(modelSource(result)).toBe('dispatched') + expect(modelValue(result)).toBe('haiku') + }) +}) + +describe('the provider report corrects the pill', () => { + it('moves the pill to the model the turn actually ran', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + expect(modelValue(await adapter.readOptions({ sessionId: 'session-1', fence: 7 }))).toBe( + 'haiku' + ) + + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + const corrected = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(modelValue(corrected)).toBe('sonnet') + expect(corrected.current.confirmed).toContain('model') + }) + + it('lets a newer write outrank the report it precedes', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(modelValue(result)).toBe('haiku') + expect(result.current.confirmed ?? []).not.toContain('model') + }) +}) + +describe('confirmation never outlives the write it belongs to', () => { + it('drops an earlier effort confirmation when the value changes', async () => { + const calls: string[] = [] + let reported = 'low' + const session = { + options: new Map([['model', 'sonnet']]), + reportedOptions: {}, + optionMutationSequence: 0, + confirmedOptions: new Set(), + connection: { + supportedModels: async () => CATALOG, + applyFlagSettings: async (s: { effortLevel?: string }) => { + calls.push(`apply:${s.effortLevel}`) + }, + getSettings: async () => ({ + applied: { effort: reported }, + effective: { effortLevel: reported }, + sources: {} + }) + } + } as unknown as ClaudeSession + + await setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + expect(session.confirmedOptions.has('effort')).toBe(true) + + // The provider now reports a level it cannot represent; the stale confirmation + // must not survive into the new value. + await setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + expect(session.options.get('effort')).toBe('max') + expect(session.confirmedOptions.has('effort')).toBe(false) + expect(calls).toEqual(['apply:low', 'apply:max']) + }) +}) diff --git a/src/main/claude/claude-structured-options.test.ts b/src/main/claude/claude-structured-options.test.ts new file mode 100644 index 00000000000..bc18a589e10 --- /dev/null +++ b/src/main/claude/claude-structured-options.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, it, vi } from 'vitest' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' + +function sessionFor(setModel: ClaudeSession['connection']['setModel']): ClaudeSession { + return { + connection: { setModel } as ClaudeSession['connection'], + providerSessionId: 'provider-session', + claudeConfigDir: '/accounts/claude', + leafUuid: null, + fence: 1, + acquisitionGeneration: 'generation-1', + prompts: {} as ClaudeSession['prompts'], + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(), + reportedOptions: {}, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + capabilities: [], + events: undefined, + translator: null + } +} + +describe('Claude structured option mutation fencing', () => { + it('does not let a delayed earlier apply overwrite a later option', async () => { + let releaseFirst!: () => void + const firstApply = new Promise((resolve) => { + releaseFirst = resolve + }) + const setModel = vi + .fn() + .mockReturnValueOnce(firstApply) + .mockResolvedValue(undefined) + const session = sessionFor(setModel) + + const first = setClaudeStructuredOption(session, { key: 'model', value: 'old' }, undefined) + await vi.waitFor(() => expect(setModel).toHaveBeenCalledTimes(1)) + const second = setClaudeStructuredOption(session, { key: 'model', value: 'new' }, undefined) + await expect(second).resolves.toEqual({ model: 'new' }) + + releaseFirst() + await expect(first).resolves.toEqual({ model: 'new' }) + expect(session.options).toEqual(new Map([['model', 'new']])) + }) +}) diff --git a/src/main/claude/claude-structured-options.ts b/src/main/claude/claude-structured-options.ts new file mode 100644 index 00000000000..3d1377b12c6 --- /dev/null +++ b/src/main/claude/claude-structured-options.ts @@ -0,0 +1,147 @@ +import type { EffortLevel, PermissionMode } from '@anthropic-ai/claude-agent-sdk' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { + AgentSessionOptionRejectedError, + isAgentSessionOptionRejectedError +} from '../native-chat/agent-session-wire/structured-agent-session-option-error' +import { + readClaudeCurrentModel, + readClaudeModelEffortLevels, + readClaudeSettingsEffort +} from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' + +const OPTION_ORDER = ['model', 'effort', 'permissionMode'] as const + +/** + * Efforts the settings readback cannot report. `max` applies for the rest of the + * session and is excluded from the persisted `effortLevel` by contract, so + * `get_settings` answers with the level underneath it — an absence of evidence + * that must not be read as the child refusing a level its own catalog offers. + */ +const UNREPORTED_EFFORTS: ReadonlySet = new Set(['max']) + +export function restoredClaudeStructuredSessionOptions( + options: Readonly> | undefined +): Map { + return new Map( + OPTION_ORDER.flatMap((key) => { + const value = options?.[key] + return value ? [[key, value] as const] : [] + }) + ) +} + +export async function setClaudeStructuredOption( + session: ClaudeSession, + input: { key: string; value: string }, + timeoutMs: number | undefined +): Promise>> { + const apply = + input.key === 'model' + ? () => session.connection.setModel(input.value, { timeoutMs }) + : input.key === 'permissionMode' + ? () => session.connection.setPermissionMode(input.value as PermissionMode, { timeoutMs }) + : input.key === 'effort' + ? () => + session.connection.applyFlagSettings( + { effortLevel: input.value as EffortLevel }, + { timeoutMs } + ) + : null + if (!apply) { + throw new AgentSessionOptionRejectedError( + `claude stream-json has no session option named ${input.key}` + ) + } + // The child stores an effort its model has no control for and keeps it across + // every later model switch and restore, so refuse before the write rather than + // read the acceptance back as adoption. Refused here, restore drops the stale + // value instead of replaying it onto a model that cannot use it. + if (input.key === 'effort') { + const { modelId, levels } = await readClaudeModelEffortLevels(session, timeoutMs) + if (levels && !levels.has(input.value)) { + throw new AgentSessionOptionRejectedError( + `claude model ${modelId} does not accept effort ${input.value}` + ) + } + } + const modelWasConfirmed = readClaudeCurrentModel(session).confirmed + const mutationSequence = ++session.optionMutationSequence + // Only a model write can stale the model report — an effort or permission-mode + // write does not change what the child is running. Leaving the stamp behind + // would drop the session back to the written model and refuse, on the next + // effort write, a level the model actually running advertises. + if (modelWasConfirmed && input.key !== 'model') { + session.reportedModelMutation = mutationSequence + } + try { + await apply() + } catch (error) { + if (error instanceof ClaudeControlRequestError) { + throw new AgentSessionOptionRejectedError(error) + } + throw error + } + // apply_flag_settings answers `success` for an effort it then ignores, so the + // absence of a throw proves nothing. Ask what the child actually holds. + const adopted = + input.key === 'effort' && !UNREPORTED_EFFORTS.has(input.value) + ? await session.connection + .getSettings({ timeoutMs }) + .then(readClaudeSettingsEffort) + .catch(() => null) + : null + if (mutationSequence !== session.optionMutationSequence) { + return Object.fromEntries(session.options) + } + // A disagreement stops main vouching for the value, it does not veto the write: + // the pre-flight guard already refuses levels the model advertises no control + // for, and no other client refuses on a readback. Keep the child's own answer so + // the disagreement survives as the level a later read falls back to. + if (adopted !== null && adopted !== input.value) { + session.reportedOptions.effort = adopted + } + session.options.set(input.key, input.value) + // Only a readback that agreed is adoption evidence; one that disagreed or could + // not be taken records the value but must not also claim the provider vouched for it. + if (adopted !== null && adopted === input.value) { + session.confirmedOptions.add(input.key) + } else { + session.confirmedOptions.delete(input.key) + } + // The effort readback was taken under the old model, so a model switch retires + // it: the child keeps the value but nothing has reported the new model holding + // it, and vouching for it would show a confirmed effort no readback covers. + if (input.key === 'model') { + session.confirmedOptions.delete('effort') + } + return Object.fromEntries(session.options) +} + +export async function restoreClaudeStructuredSessionOptions( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise { + // Any write that was already in flight belongs to the previous acquisition + // state and must not repopulate this map after restore starts. + session.optionMutationSequence += 1 + // The fence bump is not a write, so the report the session already holds is still + // current as of this instant; leaving the stamp behind would make every restored + // session read as unconfirmed until its next turn. + session.reportedModelMutation = session.optionMutationSequence + const options = [...session.options.entries()] + session.options.clear() + for (const [key, value] of options) { + try { + await setClaudeStructuredOption(session, { key, value }, timeoutMs) + } catch (error) { + if (!isAgentSessionOptionRejectedError(error)) { + throw error + } + // A stale or unavailable preference must not poison every future acquire; + // the provider's current value remains authoritative and is re-persisted. + session.restoreSkippedOptions.add(key) + } + } +} diff --git a/src/main/claude/claude-structured-owner-identity.test.ts b/src/main/claude/claude-structured-owner-identity.test.ts new file mode 100644 index 00000000000..592b61df338 --- /dev/null +++ b/src/main/claude/claude-structured-owner-identity.test.ts @@ -0,0 +1,39 @@ +import { describe, expect, it, vi } from 'vitest' +import { CLAUDE_SPAWN_TOKEN_ENV, claudeProcessIdentity } from './claude-structured-owner-identity' + +const IDENTITY = { + sessionId: 'session-identity', + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude' as const, + providerHandle: { kind: 'claude' as const, sessionId: 'session-1', leafUuid: 'leaf-1' } +} + +describe('claude structured owner identity', () => { + it('exports the spawn token env and records the observed process identity', async () => { + expect(CLAUDE_SPAWN_TOKEN_ENV).toBe('ORCA_AGENT_SESSION_SPAWN_TOKEN') + await expect( + claudeProcessIdentity( + { identity: IDENTITY, spawnToken: 'spawn-a', pid: 4242 }, + async () => 123 + ) + ).resolves.toEqual({ + hostId: 'local', + pid: 4242, + processStartTimeMs: 123, + spawnToken: 'spawn-a' + }) + }) + + it('retries a failed start-time read before giving up', async () => { + const readStartTime = vi + .fn<(pid: number) => Promise>() + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(456) + await expect( + claudeProcessIdentity({ identity: IDENTITY, spawnToken: 'spawn-a', pid: 4242 }, readStartTime) + ).resolves.toMatchObject({ processStartTimeMs: 456 }) + expect(readStartTime).toHaveBeenCalledTimes(3) + }) +}) diff --git a/src/main/claude/claude-structured-owner-identity.ts b/src/main/claude/claude-structured-owner-identity.ts index 1d13e6ec7c2..e8f251d41f9 100644 --- a/src/main/claude/claude-structured-owner-identity.ts +++ b/src/main/claude/claude-structured-owner-identity.ts @@ -1,4 +1,7 @@ +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' import type { AgentSessionProviderHandleLink } from '../../shared/agent-session-provider-handle' +import type { AgentSessionProcessIdentity } from '../../shared/agent-session-record' +import { readProcessStartTimeMs } from '../runtime/agent-session-process-identity-probe' export function claudeProviderHandleLink(input: { sessionId: string @@ -19,3 +22,41 @@ export function claudeProviderHandleLink(input: { observedAt: input.observedAt } } + +/** The child echoes its spawn token here so the owner probe can tell a live + * child of this reservation from a same-pid stranger. */ +export const CLAUDE_SPAWN_TOKEN_ENV = 'ORCA_AGENT_SESSION_SPAWN_TOKEN' + +const START_TIME_READ_ATTEMPTS = 3 + +export async function claudeProcessIdentity( + input: { + identity: AgentSessionJournalIdentity + spawnToken: string + pid: number | undefined + }, + readStartTime: (pid: number) => Promise = readProcessStartTimeMs +): Promise { + if (input.pid === undefined) { + throw new Error('claude app-server started without a pid') + } + let processStartTimeMs: number | null = null + for ( + let attempt = 0; + attempt < START_TIME_READ_ATTEMPTS && processStartTimeMs === null; + attempt += 1 + ) { + processStartTimeMs = await readStartTime(input.pid) + } + if (processStartTimeMs === null) { + // Why: recording null makes every later owner probe indeterminate — a durable latch. + // Failing here reaps the child and leaves a retryable refusal instead. + throw new Error(`claude app-server start time for pid ${input.pid} could not be read`) + } + return { + hostId: input.identity.hostId, + pid: input.pid, + processStartTimeMs, + spawnToken: input.spawnToken + } +} diff --git a/src/main/claude/claude-structured-prompt-items.test.ts b/src/main/claude/claude-structured-prompt-items.test.ts new file mode 100644 index 00000000000..79916d6a507 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-items.test.ts @@ -0,0 +1,110 @@ +import { describe, expect, it } from 'vitest' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { encodeAgentSessionQuestionAnswers } from '../../shared/agent-session-question-answer' +import { claudeQuestionItems } from './claude-structured-prompt-items' +import { + applyClaudePromptAnswer, + encodeClaudeQuestionOptionId, + type ClaudePendingPrompt +} from './claude-structured-prompt-replies' + +describe('Claude structured question addressing', () => { + it('keeps wire IDs bounded while returning the original question and choice', () => { + const questionId = 'Which option? '.repeat(100) + const label = 'A detailed choice '.repeat(100) + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { questions: [{ question: questionId, options: [{ label }] }] }, + suggestions: [], + questionIds: [questionId], + answers: new Map(), + settle: () => {} + } + + const item = claudeQuestionItems({ sessionId: 'session-1', prompt })[0]! + expect(agentJournalItemKey(item.identity).length).toBeLessThan(512) + expect(item.body.options[0]!.id.length).toBeLessThan(512) + expect(item.body.freeTextQuestionId).toBe('q1') + expect(applyClaudePromptAnswer({ prompt }, item.body.options[0]!.id)).toMatchObject({ + updatedInput: { answers: { [questionId]: label } } + }) + }) + + it('preserves colon-containing free-text answers', () => { + const questionId = 'Where should this run?' + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { questions: [{ question: questionId }] }, + suggestions: [], + questionIds: [questionId], + answers: new Map(), + settle: () => {} + } + const answer = 'https://example.test:8443/path' + + expect( + applyClaudePromptAnswer({ prompt }, encodeClaudeQuestionOptionId('q1', answer)) + ).toMatchObject({ + updatedInput: { answers: { [questionId]: answer } } + }) + }) + + it('returns arrays for multi-select and preserves mixed single and Other answers', () => { + const multiQuestion = 'Which targets?' + const singleQuestion = 'Which mode?' + const otherQuestion = 'Where should it run?' + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { + question: multiQuestion, + multiSelect: true, + options: [{ label: 'frontend' }, { label: 'backend' }] + }, + { + question: singleQuestion, + options: [{ label: 'fast' }, { label: 'safe' }] + }, + { question: otherQuestion, options: [] } + ] + }, + suggestions: [], + questionIds: [multiQuestion, singleQuestion, otherQuestion], + answers: new Map(), + settle: () => {} + } + const item = claudeQuestionItems({ sessionId: 'session-1', prompt })[0]! + const questions = item.body.questions! + const encoded = encodeAgentSessionQuestionAnswers([ + { + questionId: 'q1', + optionIds: [questions[0]!.options[0]!.id, questions[0]!.options[1]!.id] + }, + { questionId: 'q2', optionIds: [questions[1]!.options[1]!.id] }, + { questionId: 'q3', optionIds: [], other: 'remote host' } + ]) + + expect(applyClaudePromptAnswer({ prompt }, encoded)).toMatchObject({ + updatedInput: { + answers: { + [multiQuestion]: ['frontend', 'backend'], + [singleQuestion]: 'safe', + [otherQuestion]: 'remote host' + } + } + }) + }) +}) diff --git a/src/main/claude/claude-structured-prompt-items.ts b/src/main/claude/claude-structured-prompt-items.ts new file mode 100644 index 00000000000..3bdf8ab6091 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-items.ts @@ -0,0 +1,134 @@ +import type { + AgentJournalApprovalItem, + AgentJournalItemIdentity, + AgentJournalPromptOption, + AgentJournalQuestion, + AgentJournalQuestionItem +} from '../../shared/agent-session-journal-types' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import { claudeRecord, claudeText } from './claude-structured-item-translation' +import { + CLAUDE_APPROVAL_DECISIONS, + encodeClaudeQuestionOptionId, + type ClaudeApprovalDecision, + type ClaudePendingPrompt +} from './claude-structured-prompt-replies' + +const APPROVAL_LABELS: Record = { + allow: 'Allow', + allowForSession: 'Allow for this session', + deny: 'Deny', + cancel: 'Stop' +} + +const PENDING = { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null +} as const + +export function claudePromptIdentity(input: { + sessionId: string + promptKey: string + questionId?: string +}): AgentJournalItemIdentity { + const suffix = input.questionId ? `:${input.questionId}` : '' + return { + provider: 'orca', + clientMessageId: `claude-prompt:${input.sessionId}:${input.promptKey}${suffix}` + } +} + +export function claudeApprovalItem(prompt: ClaudePendingPrompt): AgentJournalApprovalItem { + const serialized = JSON.stringify(prompt.input) + return { + kind: 'approval', + title: `Allow ${prompt.toolName}?`, + detail: serialized ? boundInlineText(serialized, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text : null, + options: CLAUDE_APPROVAL_DECISIONS.map((decision) => ({ + id: decision, + label: APPROVAL_LABELS[decision] + })), + resolution: { ...PENDING } + } +} + +export type ClaudeQuestionItem = { + identity: AgentJournalItemIdentity + body: AgentJournalQuestionItem +} + +function questionOptions( + question: Record, + questionAddress: string +): AgentJournalPromptOption[] { + if (!Array.isArray(question.options)) { + return [] + } + return question.options.flatMap((value, index) => { + const option = claudeRecord(value) + const label = claudeText(option?.label) + const description = claudeText(option?.description) + return label + ? [ + { + id: encodeClaudeQuestionOptionId(questionAddress, `choice-${index + 1}`), + label, + ...(description ? { description } : {}) + } + ] + : [] + }) +} + +export function claudeQuestionItems(input: { + sessionId: string + prompt: ClaudePendingPrompt +}): ClaudeQuestionItem[] { + const values = Array.isArray(input.prompt.input.questions) ? input.prompt.input.questions : [] + const questions = values.flatMap((value, index): AgentJournalQuestion[] => { + const question = claudeRecord(value) + const questionAddress = `q${index + 1}` + const text = claudeText(question?.question) ?? claudeText(question?.header) + const header = claudeText(question?.header) + return question && input.prompt.questionIds[index] && text + ? [ + { + id: questionAddress, + question: text, + ...(header ? { header } : {}), + options: questionOptions(question, questionAddress), + multiSelect: question.multiSelect === true, + freeTextQuestionId: questionAddress + } + ] + : [] + }) + if (questions.length === 0) { + return [] + } + const legacyCompatible = questions.length === 1 && questions[0]?.multiSelect === false + const first = questions[0]! + return [ + { + identity: claudePromptIdentity({ + sessionId: input.sessionId, + promptKey: input.prompt.promptKey + }), + body: { + kind: 'question', + question: legacyCompatible + ? first.question + : `${questions.length} grouped question${questions.length === 1 ? '' : 's'} from Claude`, + options: legacyCompatible ? first.options : [], + ...(legacyCompatible ? { freeTextQuestionId: first.freeTextQuestionId } : {}), + questions, + resolution: { ...PENDING } + } + } + ] +} diff --git a/src/main/claude/claude-structured-prompt-replies.ts b/src/main/claude/claude-structured-prompt-replies.ts new file mode 100644 index 00000000000..deec74b7308 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-replies.ts @@ -0,0 +1,297 @@ +import { decodeAgentSessionQuestionAnswers } from '../../shared/agent-session-question-answer' + +export const CLAUDE_APPROVAL_DECISIONS = ['allow', 'allowForSession', 'deny', 'cancel'] as const +export type ClaudeApprovalDecision = (typeof CLAUDE_APPROVAL_DECISIONS)[number] + +/** Settles the SDK's `canUseTool` promise; `null` is the SDK's "no response written" sentinel. */ +export type ClaudePromptSettle = (response: Record | null) => void + +export type ClaudePendingPrompt = { + requestId: string + promptKey: string + toolUseId: string + toolName: string + kind: 'approval' | 'question' + input: Record + suggestions: unknown[] + questionIds: readonly string[] + answers: Map + settle: ClaudePromptSettle +} + +export type ClaudePromptRegistration = { + requestId: string + toolName: string + toolUseId: string + input: Record + suggestions: unknown[] + settle: ClaudePromptSettle +} + +type PromptBinding = { + address: string + questionId?: string +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +function readString(value: unknown): string | null { + return typeof value === 'string' && value.trim().length > 0 ? value : null +} + +function questionsFrom(input: Record): Record[] { + return Array.isArray(input.questions) ? input.questions.filter(isRecord) : [] +} + +function questionIdFromAddress(prompt: ClaudePendingPrompt, address: string): string | null { + const match = /^q([1-9]\d*)$/.exec(address) + const index = match ? Number(match[1]) - 1 : -1 + return index >= 0 ? (prompt.questionIds[index] ?? null) : null +} + +function questionAnswer(prompt: ClaudePendingPrompt, questionId: string, optionId: string): string { + const decoded = decodeClaudeQuestionOptionId(optionId) + if (!decoded) { + return optionId + } + const questionIndex = prompt.questionIds.indexOf(questionId) + if (questionIndex === -1) { + return optionId + } + const choice = /^choice-([1-9]\d*)$/.exec(decoded.answer) + const optionIndex = choice ? Number(choice[1]) - 1 : -1 + const question = questionsFrom(prompt.input)[questionIndex] + const options = Array.isArray(question?.options) ? question.options : [] + const option = options[optionIndex] + const label = isRecord(option) ? readString(option.label) : null + if (decoded.questionId === `q${questionIndex + 1}` && label) { + return label + } + if (decoded.questionId === `q${questionIndex + 1}`) { + return decoded.answer + } + const legacyChoice = options.some( + (candidate) => isRecord(candidate) && readString(candidate.label) === decoded.answer + ) + return decoded.questionId === questionId && (legacyChoice || decoded.answer.trim().length > 0) + ? decoded.answer + : optionId +} + +function questionId(question: Record, index: number): string { + return readString(question.question) ?? readString(question.header) ?? `question-${index + 1}` +} + +export function encodeClaudeQuestionOptionId(questionId: string, answer: string): string { + return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` +} + +export function decodeClaudeQuestionOptionId( + optionId: string +): { questionId: string; answer: string } | null { + const separator = optionId.indexOf(':') + if (separator <= 0) { + return null + } + try { + return { + questionId: decodeURIComponent(optionId.slice(0, separator)), + answer: decodeURIComponent(optionId.slice(separator + 1)) + } + } catch { + return null + } +} + +export class ClaudePromptRegistry { + private readonly prompts = new Map() + private readonly journalBindings = new Map() + + register(registration: ClaudePromptRegistration): ClaudePendingPrompt | null { + const toolUseId = readString(registration.toolUseId) + const toolName = readString(registration.toolName) + const input = isRecord(registration.input) ? registration.input : null + if (!toolUseId || !toolName || !input) { + return null + } + const questions = toolName === 'AskUserQuestion' ? questionsFrom(input) : [] + const prompt: ClaudePendingPrompt = { + requestId: registration.requestId, + promptKey: registration.requestId, + toolUseId, + toolName, + kind: questions.length > 0 ? 'question' : 'approval', + input, + suggestions: Array.isArray(registration.suggestions) ? registration.suggestions : [], + questionIds: questions.map(questionId), + answers: new Map(), + settle: registration.settle + } + this.prompts.set(prompt.promptKey, prompt) + return prompt + } + + /** True only if the prompt was still pending; lets an abort and an answer race settle once. */ + forgetIfPending(prompt: ClaudePendingPrompt): boolean { + if (!this.prompts.has(prompt.promptKey)) { + return false + } + this.forget(prompt) + return true + } + + bindJournalItemId(journalItemId: string, promptKey: string, questionIdForItem?: string): void { + this.journalBindings.set(journalItemId, { + address: promptKey, + ...(questionIdForItem ? { questionId: questionIdForItem } : {}) + }) + } + + find(itemId: string): { prompt: ClaudePendingPrompt; questionId?: string } | null { + const binding = this.journalBindings.get(itemId) + const prompt = this.prompts.get(binding?.address ?? itemId) + return prompt + ? { prompt, ...(binding?.questionId ? { questionId: binding.questionId } : {}) } + : null + } + + cancel(requestId: string): ClaudePendingPrompt | null { + const prompt = this.prompts.get(requestId) ?? null + if (prompt) { + this.forget(prompt) + } + return prompt + } + + forget(prompt: ClaudePendingPrompt): void { + this.prompts.delete(prompt.promptKey) + for (const [itemId, binding] of this.journalBindings) { + if (binding.address === prompt.promptKey) { + this.journalBindings.delete(itemId) + } + } + } + + clear(): ClaudePendingPrompt[] { + const pending = [...this.prompts.values()] + this.prompts.clear() + this.journalBindings.clear() + return pending + } +} + +function approvalResponse(prompt: ClaudePendingPrompt, optionId: string): Record { + if (!(CLAUDE_APPROVAL_DECISIONS as readonly string[]).includes(optionId)) { + throw new Error(`${optionId} is not a Claude approval decision`) + } + const decision = optionId as ClaudeApprovalDecision + if (decision === 'allow' || decision === 'allowForSession') { + return { + behavior: 'allow', + updatedInput: prompt.input, + ...(decision === 'allowForSession' && prompt.suggestions.length > 0 + ? { updatedPermissions: prompt.suggestions } + : {}), + toolUseID: prompt.toolUseId + } + } + return { + behavior: 'deny', + message: decision === 'cancel' ? 'User stopped this turn.' : 'User denied this action.', + ...(decision === 'cancel' ? { interrupt: true } : {}), + toolUseID: prompt.toolUseId + } +} + +function questionResponse( + prompt: ClaudePendingPrompt, + optionId: string, + boundQuestionId?: string +): Record | null { + const decoded = decodeClaudeQuestionOptionId(optionId) + const decodedQuestionId = decoded + ? (questionIdFromAddress(prompt, decoded.questionId) ?? + (prompt.questionIds.includes(decoded.questionId) ? decoded.questionId : null)) + : null + const selectedQuestionId = + boundQuestionId ?? + decodedQuestionId ?? + (prompt.questionIds.length === 1 ? prompt.questionIds[0] : null) + if (!selectedQuestionId || !prompt.questionIds.includes(selectedQuestionId)) { + throw new Error(`${optionId} does not name a question on Claude prompt ${prompt.promptKey}`) + } + const answer = questionAnswer(prompt, selectedQuestionId, optionId) + prompt.answers.set(selectedQuestionId, answer) + if (prompt.questionIds.some((id) => !prompt.answers.has(id))) { + return null + } + const answers: Record = {} + for (const id of prompt.questionIds) { + answers[id] = prompt.answers.get(id) as string + } + return { + behavior: 'allow', + updatedInput: { ...prompt.input, answers }, + toolUseID: prompt.toolUseId + } +} + +function groupedQuestionResponse( + prompt: ClaudePendingPrompt, + optionId: string +): Record | null { + const grouped = decodeAgentSessionQuestionAnswers(optionId) + if (!grouped) { + return null + } + const questions = questionsFrom(prompt.input) + if (grouped.length !== prompt.questionIds.length) { + throw new Error(`Grouped answer does not match Claude prompt ${prompt.promptKey}`) + } + const answers: Record = {} + for (let index = 0; index < questions.length; index += 1) { + const question = questions[index]! + const providerQuestionId = prompt.questionIds[index] + const answer = grouped.find((entry) => entry.questionId === `q${index + 1}`) + if (!providerQuestionId || !answer) { + throw new Error(`Grouped answer does not name question ${index + 1}`) + } + const selected = answer.optionIds.map((selectedId) => + questionAnswer(prompt, providerQuestionId, selectedId) + ) + const other = answer.other?.trim() + if (question.multiSelect === true) { + const values = [...selected, ...(other ? [other] : [])] + if (values.length === 0) { + throw new Error(`Grouped answer leaves question ${index + 1} empty`) + } + answers[providerQuestionId] = values + } else { + const value = other || selected[0] + if (!value || selected.length > 1) { + throw new Error(`Grouped answer is invalid for question ${index + 1}`) + } + answers[providerQuestionId] = value + } + } + return { + behavior: 'allow', + updatedInput: { ...prompt.input, answers }, + toolUseID: prompt.toolUseId + } +} + +export function applyClaudePromptAnswer( + found: { prompt: ClaudePendingPrompt; questionId?: string }, + optionId: string +): Record | null { + if (found.prompt.kind === 'approval') { + return approvalResponse(found.prompt, optionId) + } + return ( + groupedQuestionResponse(found.prompt, optionId) ?? + questionResponse(found.prompt, optionId, found.questionId) + ) +} diff --git a/src/main/claude/claude-structured-provider-fallback.test.ts b/src/main/claude/claude-structured-provider-fallback.test.ts new file mode 100644 index 00000000000..e8b27114da4 --- /dev/null +++ b/src/main/claude/claude-structured-provider-fallback.test.ts @@ -0,0 +1,117 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createDeferredStructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'provider-1', leafUuid: 'leaf-1' } +} + +let root = '' + +function message( + role: 'assistant' | 'user', + uuid: string, + content: unknown[] +): ClaudeStructuredSessionEvent { + return { + type: 'message', + sessionId: 'orca-session', + message: { + type: role, + uuid, + session_id: 'provider-1', + message: { role, content } + } + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-claude-provider-fallback-')) +}) + +afterEach(async () => { + await rm(root, { recursive: true, force: true }) +}) + +describe('Claude provider fallback', () => { + it('drops suppressed init frames instead of dereferencing a null translation', () => { + const items: { identity: unknown; body: AgentJournalItemBody }[] = [] + const sink = { + appendItem: (identity: unknown, body: AgentJournalItemBody) => { + items.push({ identity, body }) + }, + appendTombstone: vi.fn(), + publish: vi.fn() + } + const translator = createClaudeJournalTranslator({ sink }) + const initEvent: ClaudeStructuredSessionEvent = { + type: 'message', + sessionId: 'orca-session', + message: { + type: 'system', + subtype: 'init', + session_id: 'provider-1', + uuid: 'init-1' + } + } + + expect(() => translator.handle(initEvent)).not.toThrow() + expect(items).toEqual([]) + }) + + it('keeps provider-fallback rows distinct across acquisitions', async () => { + const journal = await openAgentSessionJournal({ + identity: IDENTITY, + journalDir: root, + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ + journal, + fence: 1, + publish: vi.fn() + }) + + const first = createClaudeJournalTranslator({ sink: deferred.sink, fallbackIdPrefix: '1' }) + const second = createClaudeJournalTranslator({ sink: deferred.sink, fallbackIdPrefix: '2' }) + + first.handle(message('assistant', 'assistant-1', [{ type: 'future_event', message: 'first' }])) + await deferred.drained() + second.handle( + message('assistant', 'assistant-2', [{ type: 'future_event', message: 'second' }]) + ) + await deferred.drained() + + const fallbackRows = journal + .snapshot() + .items.filter( + (item) => + item.body.kind === 'status' && + item.body.providerFrame?.kind === 'message:assistant:content:future_event' + ) + + expect(fallbackRows).toHaveLength(2) + expect(fallbackRows.map(statusText)).toEqual(['first', 'second']) + }) +}) + +function statusText(row: { body: AgentJournalItemBody }): string { + if (row.body.kind !== 'status') { + throw new Error('expected status row') + } + return row.body.text +} diff --git a/src/main/claude/claude-structured-provider-fallback.ts b/src/main/claude/claude-structured-provider-fallback.ts new file mode 100644 index 00000000000..2528ac027df --- /dev/null +++ b/src/main/claude/claude-structured-provider-fallback.ts @@ -0,0 +1,125 @@ +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import { CLAUDE_STREAM_JSON_FRAME_KINDS } from '../native-chat/agent-session-wire/claude-stream-json-frame-schema' +import { unhandledProviderFrameJournalItem } from '../native-chat/agent-session-wire/unhandled-provider-frame' +import { claudeRecord, claudeText } from './claude-structured-item-translation' + +export function claudeProviderFrameKind(message: Record): string { + const type = claudeText(message.type) ?? 'unknown' + const subtype = claudeText(message.subtype) + const eventType = claudeText(claudeRecord(message.event)?.type) + return ['message', type, subtype ?? eventType].filter(Boolean).join(':') +} + +const SETTLED_RESULT_KINDS: ReadonlySet = new Set( + CLAUDE_STREAM_JSON_FRAME_KINDS.filter((kind) => kind.startsWith('message:result:')) +) + +/** A catalogued result subtype is the turn-complete signal the translator settles + * itself; only an unmodeled subtype still needs the provider-fallback row. */ +export function isSettledClaudeResultKind(kind: string): boolean { + return SETTLED_RESULT_KINDS.has(kind) +} + +/** + * The failure a result frame carries that the turn's own frames never showed. + * + * Suppression is by meaning, not by kind. The SDK models an API failure as a + * SUCCESS-subtype result whose `result` string IS the error text and which has + * no assistant frame behind it, so keying on the subtype tombstones the turn and + * shows the user a completed, empty reply. A turn the user aborted is the + * opposite: its interrupt frame already says so, and the diagnostic in `errors` + * would only be noise. + */ +export function claudeResultFailure( + message: Record +): { text: string | null } | null { + if (message.is_error !== true) { + return null + } + const terminalReason = claudeText(message.terminal_reason) + if (terminalReason === 'aborted_streaming' || terminalReason === 'aborted_tools') { + return null + } + const result = claudeText(message.result)?.trim() + if (result) { + return { text: result } + } + const errors = Array.isArray(message.errors) + ? message.errors.flatMap((entry) => { + const text = claudeText(entry)?.trim() + return text ? [text] : [] + }) + : [] + // Nothing readable to lead with, but a reported failure still gets its row. + return { text: errors.length > 0 ? errors.join('\n') : null } +} + +/** + * What a message part that Orca cannot render says for itself. The kinds under + * `message::content:*` are synthesised from whatever `part.type` the CLI + * sends, so they can never be catalogued ahead of time; printing one is leaking + * wire vocabulary at a user who cannot act on it. The frame stays on the row's + * disclosure, so nothing is dropped and the next reader can still name it. + */ +export const CLAUDE_UNRENDERABLE_CONTENT_TEXT = 'Claude sent content Orca cannot display yet' + +export function isModeledClaudeContent(value: unknown): boolean { + const part = claudeRecord(value) + if (!part) { + return false + } + if (part.type === 'text') { + return claudeText(part.text) !== null + } + if (part.type === 'image') { + const source = claudeRecord(part.source) + if (source?.type === 'url') { + return claudeText(source.url) !== null + } + // A local attachment is replayed as the base64 (or file) source Orca itself + // sent, so it is content we recognise -- not an unknown part to surface. + return source?.type === 'base64' || source?.type === 'file' + } + if (part.type === 'tool_use') { + return claudeText(part.id) !== null && claudeText(part.name) !== null + } + if (part.type === 'tool_result') { + return claudeText(part.tool_use_id) !== null + } + // Redacted thinking arrives as an empty string plus a signature. + return part.type === 'thinking' || part.type === 'redacted_thinking' +} + +export function createClaudeProviderFrameFallback( + sink: StructuredAgentSessionEventSink, + acquisitionId: string +): { + /** `displayText` leads the row when Claude knows the sentence the frame itself does not name. */ + append: (kind: string, payload: unknown, displayText?: string | null) => void +} { + let sequence = 0 + return { + append: (kind, payload, displayText) => { + sequence += 1 + const translated = unhandledProviderFrameJournalItem('claude', kind, payload) + if (!translated) { + return + } + const bounded = displayText + ? boundInlineText(displayText, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + : null + sink.appendItem( + { + provider: 'orca', + clientMessageId: `provider-frame:claude:${acquisitionId}:${sequence}` + }, + bounded ? { ...translated.body, text: bounded } : translated.body + ) + sink.publish() + } + } +} diff --git a/src/main/claude/claude-structured-real-cli.test.ts b/src/main/claude/claude-structured-real-cli.test.ts new file mode 100644 index 00000000000..0f22c175cc6 --- /dev/null +++ b/src/main/claude/claude-structured-real-cli.test.ts @@ -0,0 +1,299 @@ +import { spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtemp, rm } from 'node:fs/promises' +import { homedir, tmpdir } from 'node:os' +import { basename, join, relative } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { resolveClaudeCommand } from '../codex-cli/command' +import { resolveSessionFilePath } from '../native-chat/session-file-resolver' +import { getSpawnArgsForWindows } from '../win32-utils' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' + +const command = resolveClaudeCommand() +const versionLaunch = getSpawnArgsForWindows(command, ['--version']) +const realClaudeAvailable = + spawnSync(versionLaunch.spawnCmd, versionLaunch.spawnArgs, { + stdio: 'ignore', + windowsHide: true, + timeout: 5_000 + }).status === 0 +const authStatusLaunch = getSpawnArgsForWindows(command, ['auth', 'status', '--json']) +/** The CLI's own account report — the only source of truth for where it writes that + * is not derived from Orca's own path expressions. */ +const realClaudeAuthStatus = (() => { + if (!realClaudeAvailable) { + return null + } + const result = spawnSync(authStatusLaunch.spawnCmd, authStatusLaunch.spawnArgs, { + encoding: 'utf8', + windowsHide: true, + timeout: 5_000 + }) + if (result.status !== 0) { + return null + } + try { + return JSON.parse(result.stdout) as { loggedIn?: boolean; projectsDirectory?: string } + } catch { + return null + } +})() +const realClaudeAuthenticated = realClaudeAuthStatus?.loggedIn === true + +function realAdapter( + providerSessionId: string, + claudeConfigDir: string, + events: ClaudeStructuredSessionEvent[] = [] +): ClaudeStructuredSessionAdapter { + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: command, + options: { ...CLAUDE_STRUCTURED_BASE_OPTIONS, sessionId: providerSessionId }, + cwd: process.cwd(), + claudeConfigDir, + providerSessionId, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1, + now: () => 2, + initTimeoutMs: 5_000 + }) +} + +function identity(providerSessionId: string): AgentSessionJournalIdentity { + return { + sessionId: 'real-cli-handshake', + workspaceId: 'real-cli-workspace', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: providerSessionId, leafUuid: null } + } +} + +/** The CLI flushes its transcript on its own schedule; poll rather than race it. */ +async function waitForResolvedTranscript( + providerSessionId: string, + timeoutMs = 15_000 +): Promise { + const deadline = Date.now() + timeoutMs + for (;;) { + // No options: the exact call transcript-read-cache.ts makes for mobile. + const resolved = await resolveSessionFilePath('claude', providerSessionId) + if (resolved || Date.now() >= deadline) { + return resolved + } + await new Promise((resolve) => setTimeout(resolve, 250)) + } +} + +describe.skipIf(!realClaudeAvailable)('Claude structured real CLI handshake', () => { + it.skipIf(!realClaudeAuthenticated)( + 'proves a pre-minted session before the first user message', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + const acquisition = await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli' + }) + const observedSubtypes = events.flatMap((event) => + event.type === 'message' ? [event.message.subtype] : [] + ) + + expect(acquisition.link.handle).toMatchObject({ + provider: 'claude', + sessionId: providerSessionId, + // Init/SessionStart UUIDs are protocol frames, not resumable + // main-transcript leaves; no cursor exists before the first user turn. + leafUuid: null + }) + expect(observedSubtypes).toContain('hook_started') + } finally { + await adapter.closeAll() + } + }, + 10_000 + ) + + // Unit tests can only pin the shape we read, which is exactly how the blank + // Effort pill survived every gate: the fixture invented an `effortLevel` on a + // frame the CLI does not send. This asserts both halves against the live + // binary — that get_settings reports the effort, and that init does not. + it.skipIf(!realClaudeAuthenticated)( + 'reports the current effort through get_settings and never on the init frame', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-effort' + }) + const published = events.flatMap((event) => + event.type === 'message' ? [event.message] : [] + ) + const options = await adapter.readOptions({ sessionId: 'real-cli-handshake', fence: 1 }) + + expect(published.length).toBeGreaterThan(0) + // Not just the init frame: no frame the CLI publishes carries an effort + // at all. Goes red the day one does, which is when the simpler fix + // becomes available. Which frame proves the session varies by host, so + // this asserts over all of them rather than picking one. + expect(published.filter((frame) => 'effortLevel' in frame)).toEqual([]) + // Goes red if `effective.effortLevel` is renamed or dropped, which no + // fixture-backed test can see. + expect(options.current.effort).toEqual(expect.any(String)) + } finally { + await adapter.closeAll() + } + }, + 15_000 + ) + + // Mobile native chat never reads the structured journal — it reads the CLI's own + // transcript through native-chat/session-file-resolver.ts. So this resolves the way + // transcript-read-cache.ts:104 does, with NO root override, and checks the answer + // against the root the CLI itself reports. Deriving the expected root from Orca's own + // `CLAUDE_CONFIG_DIR || ~/.claude` expression — the same one the code under test uses — + // would move both sides together and stay green in exactly the environment that + // blacks mobile out. + // The turn is what creates the file: an init-only handshake writes nothing. + it.skipIf(!realClaudeAuthenticated || !realClaudeAuthStatus?.projectsDirectory)( + 'writes its transcript where the mobile session-file resolver looks for it', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const adapter = realAdapter(providerSessionId, claudeConfigDir) + const cliProjectsDir = realClaudeAuthStatus?.projectsDirectory as string + + let transcriptPath: string | null = null + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-transcript' + }) + await adapter.dispatch({ + sessionId: 'real-cli-handshake', + clientMessageId: 'real-cli-transcript-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] }, + fence: 1 + }) + transcriptPath = await waitForResolvedTranscript(providerSessionId) + } finally { + await adapter.closeAll() + } + + expect(transcriptPath).not.toBeNull() + expect(basename(transcriptPath ?? '')).toBe(`${providerSessionId}.jsonl`) + // `//.jsonl` + expect(relative(cliProjectsDir, transcriptPath ?? '').split(/[\\/]/)).toHaveLength(2) + // And the pinned account home is that same root, so the host-side leaf recovery + // (structured-claude-runtime-adapter.ts:64) and mobile agree. + expect(join(claudeConfigDir, 'projects')).toBe(cliProjectsDir) + }, + 45_000 + ) + + // The model half of the same lesson: a fixture can only pin the shape we read. + // set_model answers success for a model it never resolves — a nonexistent id is + // accepted and only fails once a turn runs — so the CLI's own report is the only + // adoption evidence, and it arrives on the init frame that opens each turn. This + // asserts that frame carries the resolved model against the live binary; it goes + // red the day the CLI stops reporting it, which is the day the confirmation + // silently degrades to echoing back whatever Orca sent. + it.skipIf(!realClaudeAuthenticated)( + 'reports the model it adopted on the init frame that opens each turn', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-model' + }) + await adapter.setOption({ + sessionId: 'real-cli-handshake', + key: 'model', + value: 'haiku', + fence: 1 + }) + const before = events.length + await adapter.dispatch({ + sessionId: 'real-cli-handshake', + clientMessageId: 'real-cli-model-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'Say ok' }] }, + fence: 1 + }) + const deadline = Date.now() + 60_000 + let frames: Record[] = [] + for (;;) { + frames = events + .slice(before) + .flatMap((event) => + event.type === 'message' && + event.message.type === 'system' && + event.message.subtype === 'init' + ? [event.message] + : [] + ) + if (frames.length > 0 || Date.now() >= deadline) { + break + } + await new Promise((resolve) => setTimeout(resolve, 250)) + } + + expect(frames).not.toHaveLength(0) + // Both halves: the field exists, and it names the model the picker asked + // for in the catalog's resolved shape rather than the id Orca sent. + expect(frames[0]?.model).toEqual(expect.any(String)) + expect(frames[0]?.model).toBe('claude-haiku-4-5-20251001') + await expect( + adapter.readOptions({ sessionId: 'real-cli-handshake', fence: 1 }) + ).resolves.toMatchObject({ current: { model: 'haiku' } }) + } finally { + await adapter.closeAll() + } + }, + 90_000 + ) + + it('turns a real silent unauthenticated startup into sign-in guidance', async () => { + const claudeConfigDir = await mkdtemp(join(tmpdir(), 'orca-claude-no-auth-')) + const providerSessionId = randomUUID() + const adapter = realAdapter(providerSessionId, claudeConfigDir) + + try { + await expect( + adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-no-auth' + }) + ).rejects.toThrow(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + } finally { + await adapter.closeAll() + await rm(claudeConfigDir, { recursive: true, force: true }) + } + }, 10_000) +}) diff --git a/src/main/claude/claude-structured-session-acquisition-processless.test.ts b/src/main/claude/claude-structured-session-acquisition-processless.test.ts new file mode 100644 index 00000000000..e0996477e84 --- /dev/null +++ b/src/main/claude/claude-structured-session-acquisition-processless.test.ts @@ -0,0 +1,70 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { + ClaudeStreamJsonConnection, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import { ClaudeStructuredSessionAdapter } from './claude-structured-session-adapter' + +const PROVIDER_SESSION_ID = '819cf9f8-e43c-4ad7-b50f-54aa158a726a' +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-processless', + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'opaque', agent: 'claude', value: 'pending' } +} + +describe('Claude structured processless acquisition', () => { + it('classifies pre-pid error and close as processless with idempotent cleanup', async () => { + const fault = new Error('spawn claude ENOENT') + const close = vi.fn(async () => true) + const openConnection: typeof openClaudeStreamJsonConnection = async ( + _launch, + handlers = {} + ) => { + const connection: ClaudeStreamJsonConnection = { + pid: undefined, + closed: true, + exitVerdict: { root: 'processless', tree: 'exited' }, + initializationResult: async () => { + handlers.onFault?.(fault) + throw fault + }, + getSettings: async () => ({}), + supportedModels: async () => [], + interrupt: async () => undefined, + cancelAsyncMessage: async () => {}, + setModel: async () => {}, + setPermissionMode: async () => {}, + applyFlagSettings: async () => {}, + send: async () => {}, + close + } + return connection + } + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + }), + openConnection + }) + + const error = await adapter + .acquire({ identity: IDENTITY, fence: 7, spawnToken: 'spawn-9' }) + .catch((cause: unknown) => cause) + + expect(error).toBeInstanceOf(AgentSessionPreSpawnError) + expect(error).toMatchObject({ message: fault.message }) + expect(close).toHaveBeenCalledOnce() + await expect(adapter.releaseAcquisition({ sessionId: IDENTITY.sessionId })).resolves.toBe(true) + expect(close).toHaveBeenCalledOnce() + }) +}) diff --git a/src/main/claude/claude-structured-session-acquisition.ts b/src/main/claude/claude-structured-session-acquisition.ts new file mode 100644 index 00000000000..e8d09bd78d9 --- /dev/null +++ b/src/main/claude/claude-structured-session-acquisition.ts @@ -0,0 +1,298 @@ +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionPreSpawnError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { + AgentSessionAcquisition, + StructuredAgentSessionAcquireInput +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE } from '../claude-accounts/environment' +import { isClaudeAuthSwitchInProgress } from '../claude-accounts/live-pty-gate' +import { openClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { buildClaudePermissionCallbacks } from './claude-structured-inbound-control' +import { resolveClaudeReplayWaiter } from './claude-structured-dispatch' +import { + claudeAuthDiagnostic, + readClaudeCapabilities, + readClaudeFrameString, + readClaudeInit, + readClaudeModels +} from './claude-structured-init-proof' +import { + createClaudeInitDeadline, + requestClaudeInitialization +} from './claude-structured-init-deadline' +import { claudeConfigDirEnvPatch } from './claude-config-dir-pin' +import { CLAUDE_SPAWN_TOKEN_ENV, claudeProcessIdentity } from './claude-structured-owner-identity' +import { + restoreClaudeStructuredSessionOptions, + restoredClaudeStructuredSessionOptions +} from './claude-structured-options' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { createClaudeSessionJournalTranslator } from './claude-structured-journal-translation' +import { readClaudeSettingsEffort } from './claude-structured-session-options' +import { createClaudeSessionPublication } from './claude-structured-session-publication' +import { + cancelClaudeAcquisitionAttempt, + mintClaudeAcquisitionGeneration, + type ClaudeAcquisitionRegistry, + type ClaudeSession, + type ClaudeSessionExit, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeAcquireCallbacks +} from './claude-structured-session-state' +import { + closeClaudePublishedSessionForDeps, + claudeAcquisitionCleanupError +} from './claude-structured-session-close' +import { readClaudeTranscriptEntryUuid } from './claude-tui-exit' + +export const CLAUDE_STRUCTURED_INIT_TIMEOUT_MS = 10_000 + +export async function acquireClaudeSession({ + input, + deps, + sessions, + acquisitions, + exits, + callbacks +}: { + input: StructuredAgentSessionAcquireInput + deps: ClaudeStructuredSessionAdapterDeps + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + callbacks: ClaudeAcquireCallbacks +}): Promise { + // A managed-account switch is mid-swap of the pinned credential home; refuse here, + // before this acquisition cancels the previous attempt and closes the live session. + if (isClaudeAuthSwitchInProgress()) { + throw new AgentSessionPreSpawnError(new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE)) + } + const sessionId = input.identity.sessionId + const prompts = new ClaudePromptRegistry() + const translator = createClaudeSessionJournalTranslator( + input.events, + prompts, + String(input.fence) + ) + const { previous, attempt } = acquisitions.start(sessionId, prompts) + let liveSession: ClaudeSession | null = null + let observedLeafUuid: string | null = null, + expectedProviderSessionId: string | null = null + // Frames are admitted only after launch resolution proves the provider session + // this acquisition owns. Keep the check ahead of every stateful consumer. + const initTimeoutMs = deps.initTimeoutMs ?? CLAUDE_STRUCTURED_INIT_TIMEOUT_MS + const initDeadline = createClaudeInitDeadline(sessionId, initTimeoutMs) + + const onMessage = (message: Record): void => { + const init = readClaudeInit(message) + if (readClaudeFrameString(message, 'session_id') !== expectedProviderSessionId) { + // An init proof for another (or unnamed) provider must fail acquisition + // promptly, while ordinary foreign frames stay quarantined silently. + if (init || (message.type === 'system' && message.subtype === 'init')) { + initDeadline.reject(new Error('claude provider session expected')) + } + return + } + if (init) { + initDeadline.resolve(init) + // Every turn opens with an init frame naming the model the CLI is actually + // running; set_model answers success for a model it never resolves, so this + // report is the session's only adoption evidence. + if (liveSession && init.model) { + liveSession.reportedOptions.model = init.model + liveSession.reportedModelMutation = liveSession.optionMutationSequence + } + } + observedLeafUuid = readClaudeTranscriptEntryUuid(message) ?? observedLeafUuid + if (liveSession) { + liveSession.leafUuid = observedLeafUuid + } + const startsTurn = liveSession ? resolveClaudeReplayWaiter(liveSession, message) : false + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { + type: 'message', + sessionId, + message, + ...(startsTurn ? { startsTurn: true } : {}) + }) + ) + } + const { canUseTool, onUserDialog } = buildClaudePermissionCallbacks({ + sessionId, + prompts, + emit: (event) => + callbacks.deliver(attempt, sessionId, () => callbacks.emit(liveSession, input.events, event)) + }) + + try { + if (previous && !(await cancelClaudeAcquisitionAttempt(previous))) { + acquisitions.restoreIfCurrent(sessionId, attempt, previous) + throw new AgentSessionAcquisitionExitUnprovenError( + new Error(`claude acquisition for session ${sessionId} could not be stopped`) + ) + } + acquisitions.assertCurrent(sessionId, attempt) + let resumeSession = sessions.get(sessionId) + if (!(await closeClaudePublishedSessionForDeps(sessions, sessionId, deps))) { + throw new AgentSessionAcquisitionExitUnprovenError( + new Error(`claude session ${sessionId} could not be stopped`) + ) + } + // A first-hand exit that has not yet proved its full tree still owns a cleanup + // obligation; never let a new acquisition hide that evidence by omission. + const retainedExit = exits.get(sessionId) + if (retainedExit) { + const firstProof = retainedExit.closePromise ? await retainedExit.closePromise : false + const proven = firstProof || (await retainedExit.connection.close().catch(() => false)) + if (!proven) { + throw claudeAcquisitionCleanupError(retainedExit.connection, retainedExit.error) + } + // The old child is superseded by this acquisition. Settle its lifecycle + // before discarding the retained proof so its cursor and callbacks are + // cleaned up exactly once. + await callbacks.settleExit(sessionId, retainedExit) + resumeSession ??= retainedExit.session + } + acquisitions.assertCurrent(sessionId, attempt) + // Both close paths persist their final leaf, so launch validates that durable head. + const launchIdentity = resumeSession + ? { + ...input.identity, + providerHandle: { + kind: 'claude' as const, + sessionId: resumeSession.providerSessionId, + leafUuid: resumeSession.leafUuid + } + } + : input.identity + const launch = await deps + .resolveLaunch({ identity: launchIdentity }) + .catch((error: unknown) => { + throw error instanceof AgentSessionPreSpawnError + ? error + : new AgentSessionPreSpawnError(error) + }) + expectedProviderSessionId = launch.providerSessionId + observedLeafUuid = launch.resumeLeafUuid + acquisitions.assertCurrent(sessionId, attempt) + const open = deps.openConnection ?? openClaudeStreamJsonConnection + const connection = await open( + { + pathToClaudeCodeExecutable: launch.pathToClaudeCodeExecutable, + options: launch.options, + cwd: launch.cwd, + env: { + ...launch.env, + [CLAUDE_SPAWN_TOKEN_ENV]: input.spawnToken, + // Compared against what the child would otherwise inherit, so the record's + // account home still wins over a diverging overlay without a needless pin. + // (`process` is shadowed by a local later in this function, so it is not named here.) + ...claudeConfigDirEnvPatch(launch.claudeConfigDir, launch.env ? { env: launch.env } : {}) + } + }, + { + onMessage, + canUseTool, + onUserDialog, + onFault: (error) => { + if (!attempt.published) { + initDeadline.reject(error) + } + }, + onExit: (error) => { + if (!attempt.published) { + initDeadline.reject(error) + } + callbacks.handleExit(sessionId, attempt, error) + } + } + ) + attempt.connection = connection + acquisitions.assertCurrent(sessionId, attempt) + initDeadline.start() + const [initialization, init] = await Promise.all([ + requestClaudeInitialization(connection, sessionId, initTimeoutMs), + initDeadline.promise + ]) + const models = readClaudeModels(initialization) + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { type: 'options', sessionId, models }) + ) + initDeadline.clear() + acquisitions.assertCurrent(sessionId, attempt) + if (init.providerSessionId !== launch.providerSessionId) { + throw new Error( + `claude proved session ${init.providerSessionId}, expected ${launch.providerSessionId}` + ) + } + const settings = await connection + .getSettings({ timeoutMs: deps.requestTimeoutMs }) + .catch(() => null) + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { + type: 'auth-diagnostic', + sessionId, + diagnostic: claudeAuthDiagnostic(init, settings) + }) + ) + const process = await claudeProcessIdentity( + { ...input, pid: connection.pid }, + deps.readProcessStartTime + ) + acquisitions.assertCurrent(sessionId, attempt) + if (connection.closed) { + throw new Error(`claude stream-json for session ${sessionId} exited while being acquired`) + } + const publication = createClaudeSessionPublication({ + connection, + init, + claudeConfigDir: launch.claudeConfigDir, + leafUuid: observedLeafUuid, + fence: input.fence, + effort: readClaudeSettingsEffort(settings), + resumed: launch.resumed, + prompts, + translator, + events: input.events, + process, + acquisitionGeneration: mintClaudeAcquisitionGeneration(deps), + options: restoredClaudeStructuredSessionOptions(input.options), + capabilities: readClaudeCapabilities(init, initialization), + ...(deps.mintLinkId ? { linkId: deps.mintLinkId() } : {}), + observedAt: deps.now?.() ?? Date.now() + }) + const acquired: AgentSessionAcquisition = publication.acquisition + liveSession = publication.session + await restoreClaudeStructuredSessionOptions(liveSession, deps.requestTimeoutMs) + acquisitions.assertCurrent(sessionId, attempt) + acquisitions.deleteIfCurrent(sessionId, attempt) + sessions.set(sessionId, liveSession) + attempt.published = true + for (const event of attempt.buffered.splice(0)) { + event() + } + return acquired + } catch (error) { + initDeadline.clear() + let acquisitionError = error + if (sessions.get(sessionId)?.connection !== attempt.connection) { + translator?.dispose() + // Settle any callback that fired before the failure so no SDK promise dangles. + for (const prompt of prompts.clear()) { + prompt.settle(null) + } + const closed = (await attempt.connection?.close()) ?? true + if (attempt.connection?.exitVerdict.root === 'processless') { + acquisitionError = new AgentSessionPreSpawnError(error) + } else if (!closed) { + acquisitionError = claudeAcquisitionCleanupError(attempt.connection, error) + } + } + acquisitions.deleteIfCurrent(sessionId, attempt) + throw acquisitionError + } finally { + attempt.finish() + } +} diff --git a/src/main/claude/claude-structured-session-adapter.test.ts b/src/main/claude/claude-structured-session-adapter.test.ts new file mode 100644 index 00000000000..859ba9a8ed8 --- /dev/null +++ b/src/main/claude/claude-structured-session-adapter.test.ts @@ -0,0 +1,891 @@ +import { homedir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRefusal, + AgentSessionAcquisitionRootExitObservedError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' +import { encodeClaudeQuestionOptionId } from './claude-structured-prompt-replies' +import { + CLAUDE_STRUCTURED_INIT_TIMEOUT_MS, + type ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { + acquired, + adapterFor, + fakeClaude, + identityFor, + invokeCanUseTool, + PROVIDER_SESSION_ID, + tick, + USER_MESSAGE, + type FakeConnection +} from './claude-structured-session-test-support' + +describe('ClaudeStructuredSessionAdapter.acquire', () => { + it('finishes its startup deadline before the paired mobile request deadline', () => { + expect(CLAUDE_STRUCTURED_INIT_TIMEOUT_MS).toBeLessThan(30_000) + }) + + it('pins the account and proves init without treating the system-frame uuid as a chain leaf', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + + const acquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9' + }) + + expect(claude.connections[0].launch).toMatchObject({ + cwd: '/work/repo', + env: { + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9', + CLAUDE_CONFIG_DIR: '/accounts/claude' + } + }) + // supportedDialogKinds is now a query() launch option, not an initialize request param. + expect(claude.connections[0].calls.slice(0, 2)).toEqual([ + { subtype: 'initialize' }, + { subtype: 'get_settings' } + ]) + expect(acquisition.process).toEqual({ + hostId: 'host-1', + pid: 4321, + processStartTimeMs: 1_700_000_000_000, + spawnToken: 'spawn-9' + }) + expect(acquisition.link).toEqual({ + linkId: `claude-7-${PROVIDER_SESSION_ID}-empty`, + handle: { provider: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null }, + origin: 'created', + mintedAtFence: 7, + observedAt: 1_700_000_000_500 + }) + expect(events[0]).toMatchObject({ type: 'message', message: { subtype: 'init' } }) + }) + + it('restores persisted model and effort before publishing a reacquired session', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { resumed: true }) + + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options: { model: 'opus', effort: 'high' } + }) + + expect(claude.connections[0].calls.slice(-4)).toEqual([ + { subtype: 'set_model', params: { model: 'opus' } }, + // The restored model's advertised levels gate the replay, so a stale effort + // is dropped rather than re-applied to a model with no effort control. + { subtype: 'list_models' }, + { subtype: 'apply_flag_settings', params: { settings: { effortLevel: 'high' } } }, + // The effort is only recorded once the child reports having adopted it. + { subtype: 'get_settings' } + ]) + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'opus', effort: 'high' } + }) + }) + + it.each([ + ['model', 'set_model', { model: 'retired-model' }], + ['effort', 'apply_flag_settings', { effort: 'retired-effort' }], + ['permissionMode', 'set_permission_mode', { permissionMode: 'retired-mode' }] + ] as const)( + 'self-heals a persisted %s rejected during restore', + async (key, subtype, options) => { + const claude = fakeClaude({ + routes: { + [subtype]: () => { + throw new ClaudeControlRequestError(subtype, 'value is no longer available') + } + } + }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options + }) + ).resolves.toBeDefined() + expect(adapter.readOptionRestoreFailures('session-1')).toEqual([key]) + } + ) + + it('does not treat a transport timeout while restoring an option as recoverable', async () => { + const claude = fakeClaude({ + routes: { + set_model: () => { + throw new Error('claude set_model request timed out') + } + } + }) + const adapter = adapterFor(claude) + const input = { + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options: { model: 'temporarily-unavailable' } + } + + await expect(adapter.acquire(input)).rejects.toThrow('claude set_model request timed out') + expect(claude.connections[0]?.closeCount).toBe(1) + }) + + it('recovers a cancellable lifecycle when a timed-out replay arrives late', async () => { + const claude = fakeClaude({ replayUuid: null }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + const sent = claude.connections[0]!.sent[0]! + claude.connections[0]!.handlers.onMessage?.({ + ...sent, + uuid: 'late-turn-1' + }) + + expect(events).toContainEqual( + expect.objectContaining({ + type: 'message', + startsTurn: true, + message: expect.objectContaining({ uuid: 'late-turn-1' }) + }) + ) + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'late-turn-1', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + }) + + it('quarantines SDK frames without the acquired session identity', async () => { + const claude = fakeClaude({ replayUuid: null }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const connection = claude.connections[0]! + + connection.handlers.onMessage?.({ + type: 'assistant', + uuid: 'foreign-leaf', + session_id: 'foreign-provider-session', + message: { role: 'assistant', content: [{ type: 'text', text: 'do not admit' }] } + }) + connection.handlers.onMessage?.({ + type: 'assistant', + uuid: 'missing-session-leaf', + message: { role: 'assistant', content: [{ type: 'text', text: 'do not admit' }] } + }) + + const dispatch = adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + await Promise.resolve() + expect(connection.sent).toHaveLength(1) + connection.handlers.onMessage?.({ + ...connection.sent[0], + uuid: 'foreign-replay', + session_id: 'foreign-provider-session' + }) + await Promise.resolve() + expect(events.filter((event) => event.type === 'message')).toHaveLength(1) + + connection.handlers.onMessage?.({ + ...connection.sent[0], + session_id: PROVIDER_SESSION_ID + }) + await expect(dispatch).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: connection.sent[0]!.uuid } + }) + }) + + it('forwards configured launch environment while keeping ownership pins authoritative', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { + env: { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: '/wrong/account', + [CLAUDE_SPAWN_TOKEN_ENV]: 'wrong-token' + } + }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + expect(claude.connections[0].launch.env).toEqual({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: '/accounts/claude', + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' + }) + }) + + it('leaves CLAUDE_CONFIG_DIR unset when the account home is the CLI default', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { claudeConfigDir: join(homedir(), '.claude'), env: {} }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + // Pinning the CLI's own default suppresses the macOS Keychain and breaks claude.ai login. + expect(claude.connections[0].launch.env).toEqual({ [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' }) + }) + + it('re-pins the account home when the launch env would send the child elsewhere', async () => { + const claude = fakeClaude() + const accountHome = join(homedir(), '.claude') + const adapter = adapterFor(claude, { + claudeConfigDir: accountHome, + env: { CLAUDE_CONFIG_DIR: '/other/account' } + }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + expect(claude.connections[0].launch.env).toEqual({ + CLAUDE_CONFIG_DIR: accountHome, + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' + }) + }) + + it('accepts SessionStart as pre-turn proof without treating its system uuid as a leaf', async () => { + const claude = fakeClaude({ initProof: 'session-start', initUuid: 'session-start-uuid' }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + + const acquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9' + }) + + expect(acquisition.link.handle).toEqual({ + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: null + }) + expect(events[0]).toMatchObject({ + type: 'message', + message: { subtype: 'hook_started', hook_name: 'SessionStart:startup' } + }) + }) + + it('records only non-secret effective auth-lane diagnostics', async () => { + const claude = fakeClaude({ + settings: { + env: { + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + ANTHROPIC_AUTH_TOKEN: 'secret' + } + } + }) + const events: ClaudeStructuredSessionEvent[] = [] + await acquired(claude, {}, events) + + const diagnostic = events.find((event) => event.type === 'auth-diagnostic') + expect(diagnostic).toEqual({ + type: 'auth-diagnostic', + sessionId: 'session-1', + diagnostic: { + apiKeySourceConfigured: false, + baseUrlConfigured: true, + authTokenConfigured: true, + apiKeyConfigured: false, + settingSources: ['user', 'project', 'local'] + } + }) + expect(JSON.stringify(diagnostic)).not.toContain('secret') + expect(JSON.stringify(diagnostic)).not.toContain('gateway.example.test') + }) + + it('resumes the same provider id and refuses an init proof for another session', async () => { + const resumedClaude = fakeClaude() + const resumed = adapterFor(resumedClaude, { + resumed: true, + resumeLeafUuid: 'leaf-before' + }) + const acquisition = await resumed.acquire({ + identity: identityFor(), + fence: 9, + spawnToken: 'spawn-9' + }) + expect(acquisition.link.origin).toBe('resumed') + expect(acquisition.link.handle).toEqual({ + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'leaf-before' + }) + + const wrongClaude = fakeClaude({ initSessionId: 'different-session' }) + const wrong = adapterFor(wrongClaude) + await expect( + wrong.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(/expected/) + expect(wrongClaude.connections[0].closeCount).toBe(1) + }) + + it('surfaces a CLI startup failure instead of waiting for the init deadline', async () => { + const claude = fakeClaude({ exitBeforeInit: 'Claude login required' }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow('Claude login required') + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('closes a silent unauthenticated startup with actionable account guidance', async () => { + const claude = fakeClaude({ initProof: 'none' }) + const adapter = adapterFor(claude, {}, [], [], 20) + + const error = await adapter + .acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + .catch((cause: unknown) => cause) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRefusal) + expect(error).toMatchObject({ + message: expect.stringMatching(/selected Claude account is signed in.*CLAUDE_CONFIG_DIR/s) + }) + expect(claude.connections[0].calls[0]).toEqual({ subtype: 'initialize' }) + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('refuses an unauthenticated initialize response even when SessionStart runs', async () => { + const claude = fakeClaude({ + initProof: 'session-start', + initAccount: { apiProvider: 'firstParty', tokenSource: 'none' } + }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + expect(claude.connections[0].closeCount).toBe(1) + }) +}) + +describe('ClaudeStructuredSessionAdapter turns and controls', () => { + it('accepts a dispatch only after Claude replays its provider uuid', async () => { + const claude = fakeClaude({ replayUuid: 'user-provider-uuid' }) + const adapter = await acquired(claude) + + const result = await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + + expect(result).toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + uuid: 'user-provider-uuid' + } + }) + expect(claude.connections[0].sent[0]).toMatchObject({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'ship it' }] }, + session_id: PROVIDER_SESSION_ID + }) + }) + + it('leaves delivery unconfirmed when no replay uuid arrives', async () => { + const adapter = await acquired(fakeClaude({ replayUuid: null })) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + }) + + it('requires an acknowledged interrupt and supports controlled options', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-1', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'sonnet', fence: 7 }) + ).resolves.toEqual({ model: 'sonnet' }) + expect(claude.connections[0].calls.slice(-2)).toEqual([ + { subtype: 'interrupt', params: {} }, + { subtype: 'set_model', params: { model: 'sonnet' } } + ]) + + claude.routes.interrupt = () => { + throw new ClaudeControlRequestError('interrupt', 'not running') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-2', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + + claude.routes.interrupt = () => { + throw new Error('claude interrupt request timed out') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-3', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('does not let a delayed cancellation for an earlier turn interrupt the later turn', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', 'turn-U'] }) + const adapter = await acquired(claude) + + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 6 }) + ).resolves.toEqual({ cancelled: false }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 1 + ) + }) + + it('does not cancel an acknowledged turn after a later dispatch returns unknown', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', null] }) + const adapter = await acquired(claude) + + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'turn-T' } + }) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + expect(claude.connections[0].sent).toHaveLength(2) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + }) + + it('classifies provider-declined options without treating timeouts as settled', async () => { + const claude = fakeClaude({ + routes: { + set_model: () => { + throw new ClaudeControlRequestError('set_model', 'model unavailable') + } + } + }) + const adapter = await acquired(claude) + + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'fable', fence: 7 }) + ).rejects.toMatchObject({ name: 'AgentSessionOptionRejectedError' }) + claude.routes.set_model = () => { + throw new Error('claude set_model request timed out') + } + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'opus', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('hydrates live model choices and maps the resolved current model to its CLI id', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { + list_models: () => [ + { value: 'default', resolvedModel: 'claude-opus-5', displayName: 'Default' }, + { + value: 'opus', + resolvedModel: 'claude-opus-5', + displayName: 'Opus', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + }, + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet' + } + ] + } + }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toEqual({ + models: [ + { + id: 'opus', + label: 'Opus', + isDefault: true, + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }, + { id: 'sonnet', label: 'Sonnet', isDefault: false, efforts: [] } + ], + current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } + }) + }) + + it('keeps the shared Claude seed when live model discovery is unavailable', async () => { + const claude = fakeClaude({ + initModel: 'custom-model', + routes: { + list_models: () => { + throw new Error('unsupported') + } + } + }) + const adapter = await acquired(claude) + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + + expect(result.models.map((model) => model.id)).toEqual([ + 'fable', + 'opus', + 'sonnet', + 'haiku', + 'custom-model' + ]) + expect(result.current).toEqual({ + model: 'custom-model', + effort: 'high', + confirmed: ['model', 'effort'] + }) + }) +}) + +describe('ClaudeStructuredSessionAdapter acquisition cleanup', () => { + /** A start that fails after the child self-exited, with its close verdict scripted. */ + function failedStart( + unprovenCloseVerdict: ClaudeStreamJsonConnection['exitVerdict'] + ): Promise { + const claude = fakeClaude({ + exitBeforeInit: 'claude stream-json exited (code 1): not logged in', + unprovenCloseVerdict + }) + return adapterFor(claude) + .acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + .catch((error: unknown) => error) + } + + it('releases on a first-hand root exit while still carrying the CLI diagnostic', async () => { + // The root's pid and start time are the lease's identity, and they are + // provably dead: latching the session would strand a signed-out user. + const error = await failedStart({ root: 'exited', tree: 'unverifiable' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): not logged in') + }) + + it('never releases while a descendant was observed alive', async () => { + const error = await failedStart({ root: 'exited', tree: 'live' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + it('never releases for a root Orca never saw leave', async () => { + const error = await failedStart({ root: 'live', tree: 'unverifiable' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + /** A published session whose CLI then exits first-hand, with the verdict its ladder holds. */ + async function exitedAfterPublish( + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + ): Promise<{ adapter: ClaudeStructuredSessionAdapter; connection: FakeConnection }> { + const claude = fakeClaude({ unprovenCloseVerdict: exitVerdict }) + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + return { adapter, connection } + } + + it('classifies cleanup after a first-hand exit removed the session as a root exit, never as proven', async () => { + // The host may still be committing or proving the lease when the child dies; + // its cleanup must find the exit the ladder observed, not an absence. + const { adapter, connection } = await exitedAfterPublish({ + root: 'exited', + tree: 'unverifiable' + }) + const error = await adapter.releaseAcquisition({ sessionId: 'session-1' }).catch((e) => e) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): crashed') + expect(connection.closeCount).toBe(2) + }) + + it('never releases after an exit that left a descendant observed alive', async () => { + const { adapter } = await exitedAfterPublish({ root: 'exited', tree: 'live' }) + const error = await adapter.releaseAcquisition({ sessionId: 'session-1' }).catch((e) => e) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + it('forgets a retained exit once the session is acquired again', async () => { + const options: Parameters[0] = {} + const claude = fakeClaude(options) + const adapter = await acquired(claude) + const first = claude.connections[0] + first.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + first.exitVerdict = { root: 'exited', tree: 'unverifiable' } + first.close = async () => false + options.exitBeforeInit = 'claude stream-json exited (code 1): not logged in' + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).rejects.toThrow('not logged in') + // The second start's own proven close is the answer; the first exit is stale. + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + expect(first.closeCount).toBe(1) + }) + + it('reports unproven published-session cleanup so callers can retry safely', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as FakeConnection['close'] + + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(false) + expect(await adapter.readOptions({ sessionId: 'session-1', fence: 7 })).toMatchObject({ + current: { model: 'claude-sonnet-5' } + }) + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + expect(() => adapter.readOptions({ sessionId: 'session-1', fence: 7 })).toThrow( + 'no live claude stream-json session' + ) + }) + + it('does not report a second release as successful while retained exit evidence is unproven', async () => { + const claude = fakeClaude({ unprovenCloseVerdict: { root: 'exited', tree: 'unverifiable' } }) + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + connection.close = vi.fn().mockResolvedValue(false) as unknown as FakeConnection['close'] + + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBeInstanceOf( + AgentSessionAcquisitionRootExitObservedError + ) + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBeInstanceOf( + AgentSessionAcquisitionRootExitObservedError + ) + expect(connection.close).toHaveBeenCalledTimes(2) + }) + + it('keeps shutdown pending until a retained unexpected-exit proof settles', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const connection = claude.connections[0] + const proof = Promise.withResolvers() + connection.close = vi + .fn<() => Promise>() + .mockImplementationOnce(() => proof.promise) + .mockResolvedValueOnce(true) as unknown as FakeConnection['close'] + + connection.handlers.onExit?.(new Error('crashed')) + await tick() + let settled = false + const closing = adapter.closeAll().then(() => { + settled = true + }) + await tick() + expect(settled).toBe(false) + + proof.resolve(false) + await expect(closing).resolves.toBeUndefined() + expect(connection.close).toHaveBeenCalledTimes(2) + }) + + it('does not claim shutdown success for a retained false exit proof', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const connection = claude.connections[0] + connection.close = vi + .fn<() => Promise>() + .mockResolvedValue(false) as unknown as FakeConnection['close'] + + connection.handlers.onExit?.(new Error('crashed')) + await tick() + + await expect(adapter.closeAll()).rejects.toThrow( + 'claude structured session shutdown could not prove every child stopped' + ) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + expect(connection.close).toHaveBeenCalledTimes(4) + }) +}) + +describe('ClaudeStructuredSessionAdapter prompts', () => { + it('turns can_use_tool into an addressable durable approval that settles the SDK callback', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-1', 'tool-1', { + input: { command: 'git status' }, + suggestions: [{ type: 'addRules' }] + }) + expect(events.at(-1)).toMatchObject({ + type: 'prompt', + prompt: { kind: 'approval', toolName: 'Bash', promptKey: 'permission-1' } + }) + + adapter.bindPromptItemId('session-1', 'journal-approval', 'permission-1') + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-approval', + kind: 'approval', + optionId: 'allowForSession', + fence: 7 + }) + // The answer resolves the SDK's own callback promise; the SDK writes the wire response. + await expect(answered.promise).resolves.toEqual({ + behavior: 'allow', + updatedInput: { command: 'git status' }, + updatedPermissions: [{ type: 'addRules' }], + toolUseID: 'tool-1' + }) + }) + + it('collects every AskUserQuestion card before settling the one callback', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const answered = invokeCanUseTool( + claude.connections[0], + 'AskUserQuestion', + 'question-1', + 'tool-question', + { + input: { + questions: [ + { question: 'Library?', options: [{ label: 'Luxon' }] }, + { question: 'Ship now?', options: [{ label: 'Yes' }] } + ] + } + } + ) + adapter.bindPromptItemId('session-1', 'journal-q1', 'question-1', 'Library?') + adapter.bindPromptItemId('session-1', 'journal-q2', 'question-1', 'Ship now?') + + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-q1', + kind: 'question', + optionId: encodeClaudeQuestionOptionId('Library?', 'Luxon'), + fence: 7 + }) + await tick() + expect(answered.settled()).toBe(false) + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-q2', + kind: 'question', + optionId: encodeClaudeQuestionOptionId('Ship now?', 'Yes'), + fence: 7 + }) + await expect(answered.promise).resolves.toMatchObject({ + behavior: 'allow', + updatedInput: { answers: { 'Library?': 'Luxon', 'Ship now?': 'Yes' } }, + toolUseID: 'tool-question' + }) + }) + + it('leaves a prompt cancelled and unanswerable once the SDK abort signal fires', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const controller = new AbortController() + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-9', 'tool-9', { + input: { command: 'rm -rf /' }, + signal: controller.signal + }) + adapter.bindPromptItemId('session-1', 'journal-9', 'permission-9') + + controller.abort() + // A cancelled request is forgotten and settled with null — never an authorization. + await expect(answered.promise).resolves.toBeNull() + expect(events.at(-1)).toMatchObject({ type: 'prompt-cancelled', promptKey: 'permission-9' }) + // A late answer after the abort must not authorize the wrong tool. + await expect( + adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-9', + kind: 'approval', + optionId: 'allow', + fence: 7 + }) + ).rejects.toThrow(/no longer waiting/) + }) + + it('settles an in-flight permission callback when the session closes, leaving no dangling promise', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-close', 'tool-c', { + input: { command: 'ls' } + }) + await tick() + expect(answered.settled()).toBe(false) + + await adapter.closeSession('session-1') + + await expect(answered.promise).resolves.toBeNull() + }) +}) diff --git a/src/main/claude/claude-structured-session-adapter.ts b/src/main/claude/claude-structured-session-adapter.ts new file mode 100644 index 00000000000..f28b6e37f8f --- /dev/null +++ b/src/main/claude/claude-structured-session-adapter.ts @@ -0,0 +1,246 @@ +import type { + AgentSessionAcquisition, + StructuredAgentSessionAcquireInput, + StructuredAgentSessionAdapter +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { answerClaudePrompt, cancelClaudeTurn } from './claude-structured-control-actions' +import { dispatchClaudeTurn } from './claude-structured-dispatch' +import { releaseClaudeAcquisition } from './claude-structured-acquisition-release' +import { acquireClaudeSession } from './claude-structured-session-acquisition' +export { CLAUDE_STRUCTURED_INIT_TIMEOUT_MS } from './claude-structured-session-acquisition' +import { supportsClaudeStructuredLocation } from './claude-structured-location-support' +import { setClaudeStructuredOption } from './claude-structured-options' +import { readClaudeStructuredSessionOptions } from './claude-structured-session-options' +import { + ClaudeAcquisitionRegistry, + type ClaudeAcquisitionAttempt, + type ClaudeSession, + type ClaudeSessionExit, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-state' +import { + closeAllClaudeSessions, + closeClaudeSession, + settleClaudeExitedSession +} from './claude-structured-session-close' +import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' + +export type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution' +export type { + ClaudeAuthDiagnostic, + ClaudeStructuredSessionAdapterDeps, + ClaudeStructuredSessionEvent +} from './claude-structured-session-state' + +const DISPATCH_ACK_TIMEOUT_MS = 10_000 + +export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAdapter { + private readonly sessions = new Map() + private readonly acquisitions = new ClaudeAcquisitionRegistry() + private readonly exits = new Map() + + constructor(private readonly deps: ClaudeStructuredSessionAdapterDeps) {} + + supportsLocation = supportsClaudeStructuredLocation + + acquire = (input: StructuredAgentSessionAcquireInput): Promise => + acquireClaudeSession({ + input, + deps: this.deps, + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + callbacks: { + deliver: (attempt, sessionId, event) => this.deliver(attempt, sessionId, event), + emit: (session, events, event) => this.emit(session, events, event), + handleExit: (sessionId, attempt, error) => this.handleExit(sessionId, attempt, error), + settleExit: (sessionId, exit) => this.settleUnexpectedExit(sessionId, exit) + } + }) + + private deliver(attempt: ClaudeAcquisitionAttempt, sessionId: string, event: () => void): void { + if (!attempt.published) { + attempt.buffered.push(event) + return + } + if (this.sessions.get(sessionId)?.connection === attempt.connection) { + event() + } + } + + private handleExit(sessionId: string, attempt: ClaudeAcquisitionAttempt, error: Error): void { + const session = this.sessions.get(sessionId) + if (!session || session.connection !== attempt.connection) { + return + } + this.sessions.delete(sessionId) + // Re-enter the provider's close ladder before publishing lifecycle recovery. + // An exit callback is root evidence only; the retained tree proof must run + // before the host releases and reacquires this exact child. + const closePromise = session.connection.close().catch(() => false) + const exit: ClaudeSessionExit = { + connection: session.connection, + session, + error, + closePromise + } + this.exits.set(sessionId, exit) + void closePromise + .then((proven) => (proven ? this.settleUnexpectedExit(sessionId, exit) : undefined)) + .catch(() => undefined) + } + + /** Lifecycle recovery is published only after the child tree proof is true. */ + private settleUnexpectedExit(sessionId: string, exit: ClaudeSessionExit): Promise { + exit.settlementPromise ??= (async () => { + if (this.exits.get(sessionId) !== exit) { + settleClaudeExitedSession(exit.session) + return + } + // Persist the transcript-derived cursor before publishing the lifecycle + // event that lets the host release and reacquire this exact child. + await this.persistSessionHandle(sessionId, exit.session).catch(() => undefined) + if (this.exits.get(sessionId) !== exit) { + settleClaudeExitedSession(exit.session) + return + } + this.exits.delete(sessionId) + const ended: ClaudeStructuredSessionEvent = { + type: 'ended', + sessionId, + reason: exit.error.message, + cause: 'unexpected-exit', + fence: exit.session.fence, + acquisitionGeneration: exit.session.acquisitionGeneration + } + try { + this.emit(exit.session, exit.session.events, ended) + } finally { + settleClaudeExitedSession(exit.session) + } + })() + return exit.settlementPromise + } + + private async persistSessionHandle(sessionId: string, session: ClaudeSession): Promise { + try { + const transcriptLeaf = this.deps.readTranscriptLeaf + ? await readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf: this.deps.readTranscriptLeaf, + providerSessionId: session.providerSessionId, + previousLeafUuid: session.leafUuid, + claudeConfigDir: session.claudeConfigDir + }) + : null + if (transcriptLeaf) { + session.leafUuid = transcriptLeaf + } + } catch { + // A stale or unavailable tail must not overwrite the last observed leaf. + } + await this.deps.persistHandle?.({ + sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + } + + private emit( + _session: ClaudeSession | null, + _events: StructuredAgentSessionEventSink | undefined, + event: ClaudeStructuredSessionEvent + ): void { + _session?.translator?.handle(event) + this.deps.onEvent?.(event) + } + + bindPromptItemId( + sessionId: string, + journalItemId: string, + promptKey: string, + questionId?: string + ): void { + this.sessions.get(sessionId)?.prompts.bindJournalItemId(journalItemId, promptKey, questionId) + } + + dispatch: StructuredAgentSessionAdapter['dispatch'] = (input) => + dispatchClaudeTurn( + this.session(input.sessionId), + input, + this.deps.dispatchAckTimeoutMs ?? DISPATCH_ACK_TIMEOUT_MS + ) + + cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => { + const session = this.session(input.sessionId) + const acquisitionGeneration = session.acquisitionGeneration + return cancelClaudeTurn(session, this.deps.requestTimeoutMs, () => { + // Keep every ownership check adjacent to the provider interrupt. The + // session map check fences a replaced child; the turn check fences a + // delayed cancel after a newer turn was admitted on the same child. + return ( + this.sessions.get(input.sessionId) === session && + session.fence === input.fence && + session.acquisitionGeneration === acquisitionGeneration && + (session.activeTurnId === undefined + ? session.dispatchSequence === 0 + : session.activeTurnId === input.turnId && + session.activeTurnSequence === session.dispatchSequence) + ) + }) + } + answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => + answerClaudePrompt(this.session(input.sessionId), input) + setOption: StructuredAgentSessionAdapter['setOption'] = (input) => + setClaudeStructuredOption(this.session(input.sessionId), input, this.deps.requestTimeoutMs) + readOptions = (input: { sessionId: string; fence: number }) => + readClaudeStructuredSessionOptions(this.session(input.sessionId), this.deps.requestTimeoutMs) + + readOptionRestoreFailures = (sessionId: string): readonly string[] => [ + ...(this.sessions.get(sessionId)?.restoreSkippedOptions ?? []) + ] + + releaseAcquisition = (input: { sessionId: string }): Promise => + releaseClaudeAcquisition({ + sessionId: input.sessionId, + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + onExitProven: (sessionId, exit) => this.settleUnexpectedExit(sessionId, exit), + ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), + ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) + }) + + closeSession = (sessionId: string): Promise => { + if (this.exits.has(sessionId)) { + return this.releaseAcquisition({ sessionId }) + } + return closeClaudeSession({ + sessionId, + sessions: this.sessions, + acquisitions: this.acquisitions, + ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), + ...(this.deps.readTranscriptLeaf ? { readTranscriptLeaf: this.deps.readTranscriptLeaf } : {}), + ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) + }) + } + + closeAll = (): Promise => + closeAllClaudeSessions({ + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + closeSession: this.closeSession, + closeExit: (sessionId) => this.releaseAcquisition({ sessionId }) + }) + + private session(sessionId: string): ClaudeSession { + const session = this.sessions.get(sessionId) + if (!session) { + throw new Error(`no live claude stream-json session for ${sessionId}`) + } + return session + } +} diff --git a/src/main/claude/claude-structured-session-close.test.ts b/src/main/claude/claude-structured-session-close.test.ts new file mode 100644 index 00000000000..f92975e9ef4 --- /dev/null +++ b/src/main/claude/claude-structured-session-close.test.ts @@ -0,0 +1,50 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import type { + ClaudeStructuredSessionAdapterDeps, + ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { adapterFor, fakeClaude, identityFor } from './claude-structured-session-test-support' + +describe('Claude published session close lifecycle', () => { + it('ends the session even when the durable handle write rejects', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistenceError = new Error('store unavailable') + const persistHandle = vi + .fn>() + .mockRejectedValueOnce(persistenceError) + .mockResolvedValueOnce(undefined) + const adapter = adapterFor(claude, {}, events, [], undefined, undefined, persistHandle) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + await expect(adapter.closeSession('session-1')).rejects.toBe(persistenceError) + // The child is provably dead; a failed cursor write may not suppress the end. + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(0) + expect(disposeTranslator).toHaveBeenCalledOnce() + + await expect(adapter.closeSession('session-1')).resolves.toBe(true) + expect(persistHandle).toHaveBeenCalledTimes(2) + // The retry persists the same cursor without a second lifecycle end. + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) +}) diff --git a/src/main/claude/claude-structured-session-close.ts b/src/main/claude/claude-structured-session-close.ts new file mode 100644 index 00000000000..d37d3917796 --- /dev/null +++ b/src/main/claude/claude-structured-session-close.ts @@ -0,0 +1,256 @@ +import type { + ClaudeAcquisitionRegistry, + ClaudeSession, + ClaudeSessionExit, + ClaudeStructuredSessionEvent +} from './claude-structured-session-state' +import { cancelClaudeAcquisitionAttempt } from './claude-structured-session-state' +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, + AgentSessionPreSpawnError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { closeProcessRegistry } from '../../shared/child-process/close-process-registry' +import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' + +export function claudeAcquisitionCleanupError( + connection: ClaudeStreamJsonConnection | null | undefined, + cause: unknown +): Error { + const verdict = connection?.exitVerdict + if (verdict?.root === 'processless') { + return new AgentSessionPreSpawnError(cause) + } + return verdict?.root === 'exited' && verdict.tree === 'unverifiable' + ? new AgentSessionAcquisitionRootExitObservedError(cause) + : new AgentSessionAcquisitionExitUnprovenError(cause) +} + +export function settleClaudeDispatchWaiters(session: ClaudeSession): void { + for (const waiter of session.dispatchWaiters.splice(0)) { + clearTimeout(waiter.timer) + waiter.resolve(null) + } +} + +export function settleClaudeExitedSession(session: ClaudeSession): void { + settleClaudeDispatchWaiters(session) + for (const prompt of session.prompts.clear()) { + prompt.settle(null) + } + session.translator?.dispose() +} + +type CloseClaudePublishedSessionInput = { + sessions: Map + sessionId: string + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise +} + +async function finalizeClaudePublishedSession( + input: CloseClaudePublishedSessionInput, + session: ClaudeSession +): Promise { + settleClaudeDispatchWaiters(session) + // Settle every in-flight permission callback so closing leaves no dangling promise; `null` + // writes no response, and the SDK ignores any post-cleanup answer regardless. + for (const prompt of session.prompts.clear()) { + prompt.settle(null) + } + if ((await session.connection.close()) !== true) { + return false + } + try { + const transcriptLeaf = input.readTranscriptLeaf + ? await readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf: input.readTranscriptLeaf, + providerSessionId: session.providerSessionId, + previousLeafUuid: session.leafUuid, + claudeConfigDir: session.claudeConfigDir + }) + : null + if (transcriptLeaf) { + session.leafUuid = transcriptLeaf + } + } catch { + // Keep the last observed main-transcript frame when the durable tail is + // unavailable or proves a stale/divergent branch. + } + const persistence = + session.closePersistence ?? + (session.closePersistence = (async () => { + await input.persistHandle?.({ + sessionId: input.sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + })()) + const ended = { + type: 'ended', + sessionId: input.sessionId, + reason: 'claude session closed' + } as const + let callbackError: unknown + let callbackThrew = false + const deliver = (event: ClaudeStructuredSessionEvent): void => { + try { + input.onEvent?.(event) + } catch (error) { + callbackThrew = true + callbackError ??= error + } + } + let persistenceError: unknown + try { + await persistence + session.closeFinalized = true + input.sessions.delete(input.sessionId) + deliver({ + type: 'handle', + sessionId: input.sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + } catch (error) { + // Keep the closed session indexed so a retry can persist the same cursor. + // Removing it first would turn a durable-write failure into a no-op retry. + if (session.closePersistence === persistence) { + session.closePersistence = undefined + } + persistenceError = error + } + // The connection already proved the child dead, so the session has ended + // whatever the durable write did: withholding it would strand the renderer on + // a session nothing re-drives. Emitted once, so a retry only re-persists. + if (!session.closeEnded) { + session.closeEnded = true + try { + try { + session.translator?.handle(ended) + } catch (error) { + callbackThrew = true + callbackError ??= error + } + deliver(ended) + } finally { + session.translator?.dispose() + } + } + if (persistenceError) { + throw persistenceError + } + if (callbackThrew) { + throw callbackError + } + return true +} + +export async function closeClaudePublishedSession( + input: CloseClaudePublishedSessionInput +): Promise { + const session = input.sessions.get(input.sessionId) + if (!session) { + return true + } + if (session.closeFinalized) { + return true + } + if (session.closeFinalization) { + return session.closeFinalization + } + const finalization = finalizeClaudePublishedSession(input, session) + session.closeFinalization = finalization + try { + return await finalization + } finally { + if (session.closeFinalization === finalization && !session.closeFinalized) { + session.closeFinalization = undefined + } + } +} + +export function closeClaudePublishedSessionForDeps( + sessions: Map, + sessionId: string, + deps: { + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise + } +): Promise { + return closeClaudePublishedSession({ sessions, sessionId, ...deps }) +} + +export async function closeClaudeSession(input: { + sessionId: string + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise +}): Promise { + const attempt = input.acquisitions.get(input.sessionId) + if (!(await cancelClaudeAcquisitionAttempt(attempt))) { + return false + } + if (attempt) { + input.acquisitions.deleteIfCurrent(input.sessionId, attempt) + } + return closeClaudePublishedSession(input) +} + +export async function closeAllClaudeSessions(input: { + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + closeSession: (sessionId: string) => Promise + closeExit: (sessionId: string) => Promise +}): Promise { + input.acquisitions.close() + await closeProcessRegistry({ + attempts: 3, + hasEntries: () => + input.sessions.size > 0 || input.acquisitions.size > 0 || input.exits.size > 0, + entryIds: () => + new Set([ + ...input.sessions.keys(), + ...input.acquisitions.sessionIds(), + ...input.exits.keys() + ]), + closeEntry: async (sessionId) => + input.exits.has(sessionId) ? input.closeExit(sessionId) : input.closeSession(sessionId), + failureMessage: 'claude structured session shutdown could not prove every child stopped' + }) +} diff --git a/src/main/claude/claude-structured-session-options.ts b/src/main/claude/claude-structured-session-options.ts new file mode 100644 index 00000000000..afb4fd65076 --- /dev/null +++ b/src/main/claude/claude-structured-session-options.ts @@ -0,0 +1,183 @@ +import type { + AgentSessionModelOption, + AgentSessionOptionChoice, + AgentSessionOptionsResult +} from '../../shared/agent-session-wire' +import { CLAUDE_SESSION_OPTION_CATALOG } from '../../shared/agent-session-option-catalog-claude-codex' +import type { CatalogModel } from '../../shared/agent-session-option-catalog-types' +import type { ClaudeSession } from './claude-structured-session-state' + +type ListedModel = AgentSessionModelOption & { resolvedModel: string | null } + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +function text(value: unknown): string | null { + return typeof value === 'string' && value.trim() ? value : null +} + +/** + * The session's current effort, which only `get_settings` reports: the + * `system/init` frame carries `model` but has never carried an effort of any + * kind. Null when the provider stops reporting it, so the pill goes empty + * rather than showing an effort nothing measured. + */ +export function readClaudeSettingsEffort(settings: unknown): string | null { + return text(record(record(settings)?.effective)?.effortLevel) +} + +function effortLabel(value: string): string { + return value === 'xhigh' ? 'Extra high' : `${value.charAt(0).toUpperCase()}${value.slice(1)}` +} + +function listedEfforts(row: Record): AgentSessionOptionChoice[] { + return row.supportsEffort === true && Array.isArray(row.supportedEffortLevels) + ? row.supportedEffortLevels.flatMap((value) => { + const effort = text(value) + return effort ? [{ value: effort, label: effortLabel(effort) }] : [] + }) + : [] +} + +function listedModels(value: unknown): ListedModel[] { + const response = record(value) + const rows = Array.isArray(response?.models) + ? response.models.map(record).filter((row): row is Record => row !== null) + : [] + const defaultRow = rows.find((row) => text(row.value) === 'default') + const defaultResolvedModel = text(defaultRow?.resolvedModel) + const seen = new Set() + return rows.flatMap((row) => { + const id = text(row.value) + if (!id || id === 'default' || seen.has(id)) { + return [] + } + seen.add(id) + const resolvedModel = text(row.resolvedModel) + const description = text(row.description) + return [ + { + id, + label: text(row.displayName) ?? id, + ...(description ? { description } : {}), + isDefault: resolvedModel !== null && resolvedModel === defaultResolvedModel, + efforts: listedEfforts(row), + resolvedModel + } + ] + }) +} + +function seedEfforts(model: CatalogModel): AgentSessionOptionChoice[] { + const effort = model.options.find((option) => option.id === 'effort') + return effort?.kind.type === 'select' ? effort.kind.choices : [] +} + +function seedModels(): ListedModel[] { + return CLAUDE_SESSION_OPTION_CATALOG.models.map((model) => ({ + id: model.id, + label: model.label, + ...(model.description ? { description: model.description } : {}), + isDefault: model.isDefault === true, + efforts: seedEfforts(model), + resolvedModel: null + })) +} + +function currentModelId(models: ListedModel[], reportedModel: string | undefined): string { + const matched = reportedModel + ? models.find((model) => model.id === reportedModel || model.resolvedModel === reportedModel) + : undefined + return ( + matched?.id ?? reportedModel ?? models.find((model) => model.isDefault)?.id ?? models[0]!.id + ) +} + +/** + * The model the session is running. A report the CLI made after the last write + * outranks the write: it names the model the session ran. An older one does not + * — a model set between turns has no report yet, and deferring to the previous + * turn's would flip the pill back. + * + * Sole resolver of that question: every surface that acts on "the current model" + * — the pill, the effort guard, the rejection it names — reads it here, so two + * of them cannot answer it differently and offer an effort a third then refuses. + */ +export function readClaudeCurrentModel(session: ClaudeSession): { + id: string | undefined + confirmed: boolean +} { + const confirmed = + session.reportedModelMutation === session.optionMutationSequence && + session.reportedOptions.model !== undefined + return { + id: confirmed + ? session.reportedOptions.model + : (session.options.get('model') ?? session.reportedOptions.model), + confirmed + } +} + +/** + * The effort levels the session's current model advertises, with the catalog id + * that matched so a refusal names the model the pill shows. Levels are null when + * nothing identified the model: `apply_flag_settings` accepts and stores any + * level for a model with no effort control, so the catalog is the only evidence + * of a refusal — and an absent or unlisted one is not evidence, or a live CLI + * that predates `list_models` would have every effort refused under it. + */ +export async function readClaudeModelEffortLevels( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise<{ modelId: string | undefined; levels: ReadonlySet | null }> { + const modelId = readClaudeCurrentModel(session).id + if (!modelId) { + return { modelId, levels: null } + } + const catalog = await session.connection.supportedModels({ timeoutMs }).catch(() => null) + const matched = catalog + ? listedModels({ models: catalog }).find( + (model) => model.id === modelId || model.resolvedModel === modelId + ) + : undefined + return { + modelId: matched?.id ?? modelId, + levels: matched ? new Set(matched.efforts.map((choice) => choice.value)) : null + } +} + +export async function readClaudeStructuredSessionOptions( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise { + const catalog = await session.connection.supportedModels({ timeoutMs }).catch(() => null) + const discovered = listedModels(catalog ? { models: catalog } : null) + const models = discovered.length > 0 ? discovered : seedModels() + const current = readClaudeCurrentModel(session) + const model = currentModelId(models, current.id) + if (!models.some((entry) => entry.id === model)) { + models.push({ id: model, label: model, isDefault: false, efforts: [], resolvedModel: null }) + } + const effort = session.options.get('effort') ?? session.reportedOptions.effort + const confirmed = [ + ...(current.confirmed ? ['model'] : []), + ...(effort && session.confirmedOptions.has('effort') ? ['effort'] : []) + ] + return { + models: models.map((entry) => ({ + id: entry.id, + label: entry.label, + ...(entry.description ? { description: entry.description } : {}), + isDefault: entry.isDefault, + efforts: entry.efforts + })), + current: { + model, + ...(effort ? { effort } : {}), + ...(confirmed.length > 0 ? { confirmed } : {}) + } + } +} diff --git a/src/main/claude/claude-structured-session-publication.ts b/src/main/claude/claude-structured-session-publication.ts new file mode 100644 index 00000000000..29d1113c814 --- /dev/null +++ b/src/main/claude/claude-structured-session-publication.ts @@ -0,0 +1,68 @@ +import type { AgentSessionAcquisition } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeInitObservation } from './claude-structured-init-proof' +import { claudeProviderHandleLink } from './claude-structured-owner-identity' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeSession } from './claude-structured-session-state' + +export function createClaudeSessionPublication(input: { + connection: ClaudeSession['connection'] + init: ClaudeInitObservation + claudeConfigDir: string + leafUuid: string | null + fence: number + acquisitionGeneration: string + resumed: boolean + prompts: ClaudePromptRegistry + translator: ClaudeJournalTranslator | null + events: ClaudeSession['events'] + process: AgentSessionAcquisition['process'] + linkId?: string + observedAt: number + options?: ReadonlyMap + capabilities: readonly string[] + /** Read from `get_settings`; `system/init` never reports an effort. */ + effort: string | null +}): { acquisition: AgentSessionAcquisition; session: ClaudeSession } { + const model = input.init.model + const effort = input.effort + return { + acquisition: { + process: input.process, + link: claudeProviderHandleLink({ + sessionId: input.init.providerSessionId, + leafUuid: input.leafUuid, + resumed: input.resumed, + fence: input.fence, + ...(input.linkId ? { linkId: input.linkId } : {}), + observedAt: input.observedAt + }), + acquisitionGeneration: input.acquisitionGeneration + }, + session: { + connection: input.connection, + providerSessionId: input.init.providerSessionId, + claudeConfigDir: input.claudeConfigDir, + leafUuid: input.leafUuid, + fence: input.fence, + acquisitionGeneration: input.acquisitionGeneration, + prompts: input.prompts, + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(input.options), + capabilities: input.capabilities, + reportedOptions: { + ...(model ? { model } : {}), + ...(effort ? { effort } : {}) + }, + reportedModelMutation: 0, + confirmedOptions: new Set(effort ? ['effort'] : []), + restoreSkippedOptions: new Set(), + translator: input.translator, + events: input.events + } + } +} diff --git a/src/main/claude/claude-structured-session-recovery.test.ts b/src/main/claude/claude-structured-session-recovery.test.ts new file mode 100644 index 00000000000..5bfe56cf156 --- /dev/null +++ b/src/main/claude/claude-structured-session-recovery.test.ts @@ -0,0 +1,619 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { ClaudeTranscriptPreviousCursorMissingError } from './claude-transcript-branch-proof' +import { + adapterFor, + fakeClaude, + identityFor, + invokeCanUseTool, + PROVIDER_SESSION_ID, + tick +} from './claude-structured-session-test-support' + +describe('ClaudeStructuredSessionAdapter transcript-derived recovery', () => { + it('shares concurrent close finalization and emits lifecycle once', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistence = Promise.withResolvers() + const persistHandle = vi.fn(() => persistence.promise) + const adapter = adapterFor(claude, {}, events, [], undefined, undefined, persistHandle) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + const first = adapter.closeSession('session-1') + const second = adapter.closeSession('session-1') + await tick() + expect(persistHandle).toHaveBeenCalledOnce() + expect(claude.connections[0].closeCount).toBe(1) + + persistence.resolve() + await expect(Promise.all([first, second])).resolves.toEqual([true, true]) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) + + it('still emits ended and disposes state when handle delivery throws', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const callbackError = new Error('handle delivery failed') + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => { + events.push(event) + if (event.type === 'handle') { + throw callbackError + } + }, + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + persistHandle: vi.fn(async () => undefined) + }) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + await expect(adapter.closeSession('session-1')).rejects.toBe(callbackError) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) + + it('retains a closed session until its durable cursor persistence succeeds', async () => { + const claude = fakeClaude() + const persistenceError = new Error('store unavailable') + const persistHandle = vi + .fn>() + .mockRejectedValueOnce(persistenceError) + .mockResolvedValueOnce(undefined) + const adapter = adapterFor(claude, {}, [], [], undefined, undefined, persistHandle) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + await expect(adapter.closeSession('session-1')).rejects.toBe(persistenceError) + expect(persistHandle).toHaveBeenCalledTimes(1) + await expect(adapter.closeSession('session-1')).resolves.toBe(true) + expect(persistHandle).toHaveBeenCalledTimes(2) + }) + + it('persists only the last transcript-entry uuid before graceful close', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'assistant-leaf' + }) + claude.connections[0].handlers.onMessage?.({ + type: 'result', + session_id: PROVIDER_SESSION_ID, + uuid: 'result-frame-uuid' + }) + claude.connections[0].handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION_ID, + uuid: 'stream-event-frame-uuid' + }) + + await adapter.closeSession('session-1') + + expect(persistedHandles).toEqual([ + { + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'assistant-leaf', + fence: 7 + } + ]) + expect(events.at(-2)).toEqual({ + type: 'handle', + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'assistant-leaf', + fence: 7 + }) + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('prefers a validated durable transcript leaf at graceful close', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-tail') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'durable-tail' }) + }) + + it('passes the pinned Claude account home to transcript validation', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-tail') + const adapter = adapterFor( + claude, + { claudeConfigDir: '/accounts/selected' }, + [], + persistedHandles, + undefined, + readTranscriptLeaf + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/selected' + }) + }) + + it('re-proves from the transcript root when the observed cursor is missing', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValueOnce(new ClaudeTranscriptPreviousCursorMissingError()) + .mockResolvedValueOnce('reproved-main-leaf') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(1, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(2, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: null, + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'reproved-main-leaf' }) + }) + + it('keeps the observed leaf when transcript validation proves a sibling branch', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValue(new Error('latest marker is on a sibling branch')) + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledTimes(1) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'observed-tail' }) + }) + + it('persists the last transcript leaf before an unexpected first-hand exit', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'crash-leaf' + }) + + claude.connections[0].handlers.onExit?.( + new Error('claude stream-json exited (code 1): crashed unexpectedly') + ) + await tick() + + expect(persistedHandles).toContainEqual({ + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'crash-leaf', + fence: 7 + }) + expect(events.at(-1)).toMatchObject({ + type: 'ended', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: expect.any(String) + }) + }) + + it('derives the crash cursor from the validated transcript tail', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const adapter = adapterFor( + claude, + {}, + [], + persistedHandles, + undefined, + vi.fn().mockResolvedValue('durable-crash-leaf') + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'stale-observed-tail' + }) + claude.connections[0].handlers.onExit?.( + new Error('claude stream-json exited (signal SIGKILL): crashed') + ) + await tick() + + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'durable-crash-leaf' }) + }) + + it('re-proves a first-hand crash cursor from the transcript root after stale validation', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValueOnce(new ClaudeTranscriptPreviousCursorMissingError()) + .mockResolvedValueOnce('reproved-crash-leaf') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'stale-observed-tail' + }) + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(1, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'stale-observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(2, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: null, + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'reproved-crash-leaf' }) + }) + + it('keeps the observed crash leaf when transcript validation proves a sibling branch', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValue(new Error('latest marker is on a sibling branch')) + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-crash-tail' + }) + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(readTranscriptLeaf).toHaveBeenCalledTimes(1) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'observed-crash-tail' }) + }) + + it('publishes lifecycle recovery even when crash-cursor persistence fails', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor( + claude, + {}, + events, + [], + undefined, + undefined, + vi.fn().mockRejectedValue(new Error('store unavailable')) + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(events.at(-1)).toMatchObject({ type: 'ended', cause: 'unexpected-exit' }) + }) + + it('runs the child close proof before publishing unexpected-exit recovery', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const close = vi.spyOn(claude.connections[0], 'close').mockResolvedValue(true) + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(close).toHaveBeenCalledOnce() + expect(events.at(-1)).toMatchObject({ type: 'ended', cause: 'unexpected-exit' }) + }) + + it('does not publish recovery while an unexpected-exit close proof is false', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + expect(persistedHandles).toEqual([]) + }) + + it('retains pending prompts while an unexpected-exit proof is unproven', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-1', 'tool-1') + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValue(false) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(answered.settled()).toBe(false) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + }) + + it('publishes unexpected recovery exactly once after a retained proof retries successfully', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + await tick() + + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + }) + + it('launches the first replacement from the settled retained transcript cursor', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-retained-leaf') + let durableLeafUuid: string | null = null + const resolveLaunch = vi.fn(async ({ identity }) => { + if ( + identity.providerHandle.kind !== 'claude' || + identity.providerHandle.sessionId !== PROVIDER_SESSION_ID || + identity.providerHandle.leafUuid !== durableLeafUuid + ) { + throw new Error('claude durable resume identity changed before spawn') + } + if (durableLeafUuid === null) { + return { + pathToClaudeCodeExecutable: 'claude', + options: { sessionId: PROVIDER_SESSION_ID }, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + } + } + return { + pathToClaudeCodeExecutable: 'claude', + options: { resume: PROVIDER_SESSION_ID, resumeSessionAt: durableLeafUuid }, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: durableLeafUuid, + resumed: true + } + }) + const persistHandle = vi.fn>( + async (handle) => { + durableLeafUuid = handle.leafUuid + persistedHandles.push(handle) + } + ) + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch, + openConnection: claude.openConnection, + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + readTranscriptLeaf, + persistHandle + }) + const firstAcquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const first = claude.connections[0] + const oldPrompt = invokeCanUseTool(first, 'Bash', 'permission-retained', 'tool-retained') + const oldSession = ( + adapter as unknown as { + sessions: Map< + string, + { + translator: { dispose: () => void } | null + prompts: { + find: (itemId: string) => { prompt: { settle: (value: unknown) => void } } | null + } + } + > + } + ).sessions.get('session-1') + expect(oldSession?.translator).not.toBeNull() + const disposeTranslator = vi.spyOn(oldSession!.translator!, 'dispose') + const pendingPrompt = oldSession?.prompts.find('permission-retained') + expect(pendingPrompt).not.toBeNull() + const settlePrompt = vi.spyOn(pendingPrompt!.prompt, 'settle') + first.handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-retained-leaf' + }) + first.close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof first)['close'] + first.handlers.onExit?.(new Error('crashed before replacement')) + await tick() + + expect(oldPrompt.settled()).toBe(false) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + + const replacement = await adapter.acquire({ + identity: { + ...identityFor(), + providerHandle: { + kind: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'observed-retained-leaf' + } + }, + fence: 8, + spawnToken: 'spawn-10', + events: journalSink + }) + + expect(disposeTranslator).toHaveBeenCalledOnce() + expect(settlePrompt).toHaveBeenCalledOnce() + expect(settlePrompt).toHaveBeenCalledWith(null) + expect(persistHandle).toHaveBeenCalledOnce() + expect(persistedHandles).toEqual([ + { + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf', + fence: 7 + } + ]) + expect(readTranscriptLeaf).toHaveBeenCalledOnce() + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-retained-leaf', + claudeConfigDir: '/accounts/claude' + }) + expect(resolveLaunch).toHaveBeenNthCalledWith(2, { + identity: { + ...identityFor(), + providerHandle: { + kind: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf' + } + } + }) + expect(oldPrompt.settled()).toBe(true) + expect(events.filter((event) => event.type === 'ended')).toEqual([ + { + type: 'ended', + sessionId: 'session-1', + reason: 'crashed before replacement', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: firstAcquisition.acquisitionGeneration + } + ]) + expect(replacement.link).toMatchObject({ + handle: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf' + }, + origin: 'resumed', + mintedAtFence: 8 + }) + expect(claude.connections[1]?.launch.options).toMatchObject({ + resume: PROVIDER_SESSION_ID, + resumeSessionAt: 'durable-retained-leaf' + }) + expect(claude.connections).toHaveLength(2) + }) +}) diff --git a/src/main/claude/claude-structured-session-state.ts b/src/main/claude/claude-structured-session-state.ts new file mode 100644 index 00000000000..346ff686f76 --- /dev/null +++ b/src/main/claude/claude-structured-session-state.ts @@ -0,0 +1,276 @@ +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import type { + ClaudeStreamJsonConnection, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution' +import type { ClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudePendingPrompt, ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { cancelProcessAcquisition } from '../../shared/child-process/cancel-process-acquisition' +import { randomUUID } from 'node:crypto' + +export type ClaudeAuthDiagnostic = { + apiKeySourceConfigured: boolean + baseUrlConfigured: boolean + authTokenConfigured: boolean + apiKeyConfigured: boolean + settingSources: readonly string[] +} + +export type ClaudeStructuredSessionEvent = + | { + type: 'message' + sessionId: string + message: Record + /** Present only when this replay acknowledged Orca's in-flight dispatch. */ + startsTurn?: true + } + | { type: 'provider-frame'; sessionId: string; kind: string; payload: unknown } + | { type: 'prompt'; sessionId: string; prompt: ClaudePendingPrompt } + | { type: 'prompt-cancelled'; sessionId: string; promptKey: string } + | { type: 'options'; sessionId: string; models: unknown[] } + | { + type: 'handle' + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + } + | { type: 'auth-diagnostic'; sessionId: string; diagnostic: ClaudeAuthDiagnostic } + | { + type: 'ended' + sessionId: string + reason: string + /** Present for first-hand child exits so the host can fence recovery. */ + cause?: 'unexpected-exit' | 'requested-close' + fence?: number + acquisitionGeneration?: string + settlementRetryRequired?: boolean + } + +export type ClaudeStructuredSessionAdapterDeps = { + resolveLaunch: (input: { + identity: AgentSessionJournalIdentity + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + openConnection?: typeof openClaudeStreamJsonConnection + readProcessStartTime?: (pid: number) => Promise + mintLinkId?: () => string + mintAcquisitionGeneration?: () => string + now?: () => number + requestTimeoutMs?: number + initTimeoutMs?: number + dispatchAckTimeoutMs?: number + persistHandle?: (input: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + /** Read the durable transcript branch after a child has flushed its final rows. */ + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + /** Account-scoped Claude config root that owns this provider session. */ + claudeConfigDir: string + }) => Promise +} + +export type ClaudeDispatchWaiter = { + resolve: (uuid: string | null) => void + timer: ReturnType + acceptsResult: boolean + /** Client uuid echoed by Claude so a replay is tied to its own dispatch. */ + sentUuid: string + /** Sequence used to fence a late identity from a newer dispatch. */ + dispatchSequence: number + /** Set when the provider replay settled this waiter before send returned. */ + settledUuid?: string + /** The waiter timed out or its write failed, but its replay may still arrive. */ + retired?: boolean + /** Bounded digest/summary for compatibility CLIs that mint UUIDs. */ + replayContentKey: string +} + +export type ClaudeSession = { + connection: ClaudeStreamJsonConnection + providerSessionId: string + /** Durable transcript files live under this account's `projects` directory. */ + claudeConfigDir: string + leafUuid: string | null + fence: number + acquisitionGeneration: string + prompts: ClaudePromptRegistry + dispatchWaiters: ClaudeDispatchWaiter[] + /** Bounded identities for dispatches whose ack was unknown when they returned. */ + retiredDispatchWaiters: ClaudeDispatchWaiter[] + /** Once a retired waiter is evicted, legacy content-only replay matching is unsafe. */ + replayContentFallbackBlocked: boolean + options: Map + reportedOptions: { model?: string; effort?: string } + /** `optionMutationSequence` when `reportedOptions.model` was last observed, so a + * write still awaiting its first turn outranks the report it will replace. */ + reportedModelMutation: number + /** Options whose recorded value the provider reported, not merely accepted. */ + confirmedOptions: Set + restoreSkippedOptions: Set + /** CLI-advertised protocol capabilities from init; gates interrupt-receipt handling. */ + capabilities: readonly string[] + /** Provider uuid of the most recently admitted turn, if one is active. */ + activeTurnId?: string + /** Monotonic fence advanced when a dispatch starts, including unresolved dispatches. */ + dispatchSequence: number + /** Dispatch sequence that admitted activeTurnId. */ + activeTurnSequence?: number + /** Fences overlapping option writes so a late completion cannot restore stale state. */ + optionMutationSequence: number + /** Shared durable-close write; a failed write clears this for a retry. */ + closePersistence?: Promise + /** Shared full close/finalization operation; a failed operation clears this for a retry. */ + closeFinalization?: Promise + /** Set only after the durable close write succeeds, before lifecycle emission. */ + closeFinalized?: boolean + /** Set once `ended` has been emitted, so a persistence retry cannot repeat it. */ + closeEnded?: boolean + translator: ClaudeJournalTranslator | null + events: StructuredAgentSessionEventSink | undefined +} + +export function mintClaudeAcquisitionGeneration(deps: ClaudeStructuredSessionAdapterDeps): string { + return deps.mintAcquisitionGeneration?.() ?? randomUUID() +} + +/** + * The first-hand exit that removed a published session. Kept until the session + * is acquired again so acquisition cleanup that arrives after the exit finds + * what the ladder observed, not an absence it would otherwise report as proven. + */ +export type ClaudeSessionExit = { + connection: ClaudeStreamJsonConnection + /** Full session identity retained until its child tree is proven gone. */ + session: ClaudeSession + error: Error + /** The exit path's first proof attempt; retries must observe this result. */ + closePromise?: Promise + /** Shared lifecycle settlement for concurrent proof retries. */ + settlementPromise?: Promise +} + +export type ClaudeAcquisitionAttempt = { + connection: ClaudeStreamJsonConnection | null + prompts: ClaudePromptRegistry + buffered: (() => void)[] + published: boolean + cancelled: boolean + exitProven: boolean + finished: Promise + finish: () => void +} + +export function createClaudeAcquisitionAttempt( + prompts: ClaudePromptRegistry +): ClaudeAcquisitionAttempt { + let finish = (): void => {} + const finished = new Promise((resolve) => { + finish = resolve + }) + return { + connection: null, + prompts, + buffered: [], + published: false, + cancelled: false, + exitProven: false, + finished, + finish + } +} + +export class ClaudeAcquisitionRegistry { + private readonly attempts = new Map() + private closing = false + + get size(): number { + return this.attempts.size + } + + start( + sessionId: string, + prompts: ClaudePromptRegistry + ): { + previous: ClaudeAcquisitionAttempt | undefined + attempt: ClaudeAcquisitionAttempt + } { + if (this.closing) { + throw new Error('claude structured session adapter is closing') + } + const previous = this.attempts.get(sessionId) + const attempt = createClaudeAcquisitionAttempt(prompts) + this.attempts.set(sessionId, attempt) + return { previous, attempt } + } + + assertCurrent(sessionId: string, attempt: ClaudeAcquisitionAttempt): void { + if (this.closing || attempt.cancelled || this.attempts.get(sessionId) !== attempt) { + throw new Error(`claude session ${sessionId} was superseded while being acquired`) + } + } + + get(sessionId: string): ClaudeAcquisitionAttempt | undefined { + return this.attempts.get(sessionId) + } + + deleteIfCurrent(sessionId: string, attempt: ClaudeAcquisitionAttempt): void { + if (this.attempts.get(sessionId) === attempt) { + this.attempts.delete(sessionId) + } + } + + restoreIfCurrent( + sessionId: string, + replacement: ClaudeAcquisitionAttempt, + previous: ClaudeAcquisitionAttempt + ): void { + if (this.attempts.get(sessionId) === replacement) { + this.attempts.set(sessionId, previous) + } + } + + sessionIds(): IterableIterator { + return this.attempts.keys() + } + + close(): void { + this.closing = true + } +} + +export async function cancelClaudeAcquisitionAttempt( + attempt: ClaudeAcquisitionAttempt | undefined +): Promise { + if (!attempt) { + return true + } + return cancelProcessAcquisition({ + cancel: () => { + attempt.cancelled = true + }, + connection: () => attempt.connection, + exitProven: () => attempt.exitProven, + finished: attempt.finished + }) +} + +/** What an acquisition hands back to the adapter that owns the session map: + * event delivery ordered against publication, and the two exit settlements. */ +export type ClaudeAcquireCallbacks = { + deliver: (attempt: ClaudeAcquisitionAttempt, sessionId: string, event: () => void) => void + emit: ( + session: ClaudeSession | null, + events: StructuredAgentSessionEventSink | undefined, + event: ClaudeStructuredSessionEvent + ) => void + handleExit: (sessionId: string, attempt: ClaudeAcquisitionAttempt, error: Error) => void + settleExit: (sessionId: string, exit: ClaudeSessionExit) => Promise +} diff --git a/src/main/claude/claude-structured-session-test-support.ts b/src/main/claude/claude-structured-session-test-support.ts new file mode 100644 index 00000000000..6b0768b5134 --- /dev/null +++ b/src/main/claude/claude-structured-session-test-support.ts @@ -0,0 +1,260 @@ +import type { + AgentJournalMessageItem, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import type { + ClaudeStreamJsonConnection, + ClaudeStreamJsonConnectionHandlers, + ClaudeStreamJsonLaunch, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredLaunch, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' + +export const PROVIDER_SESSION_ID = '819cf9f8-e43c-4ad7-b50f-54aa158a726a' + +export const USER_MESSAGE: AgentJournalMessageItem = { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'ship it' }] +} + +export function identityFor(sessionId = 'session-1'): AgentSessionJournalIdentity { + return { + sessionId, + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null } + } +} + +type Route = (params: Record | undefined) => unknown + +export type FakeConnection = Omit & { + closed: boolean + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + launch: ClaudeStreamJsonLaunch + handlers: ClaudeStreamJsonConnectionHandlers + calls: { subtype: string; params?: Record }[] + sent: Record[] + closeCount: number +} + +export function fakeClaude( + options: { + initSessionId?: string + initUuid?: string + initModel?: string + initProof?: 'init' | 'session-start' | 'none' + initAccount?: unknown + exitBeforeInit?: string + settings?: unknown + replayUuid?: string | null + replayUuids?: (string | null)[] + capabilities?: string[] + unprovenCloseVerdict?: ClaudeStreamJsonConnection['exitVerdict'] + routes?: Record + } = {} +): { + connections: FakeConnection[] + openConnection: typeof openClaudeStreamJsonConnection + routes: Record +} { + const connections: FakeConnection[] = [] + const routes = options.routes ?? {} + let replayIndex = 0 + const routed = (subtype: string, params?: Record): unknown => { + const route = routes[subtype] + return route ? route(params) : undefined + } + const openConnection = (async (launch, handlers = {}) => { + const connection: FakeConnection = { + launch, + handlers, + calls: [], + sent: [], + closeCount: 0, + pid: 4321, + closed: false, + initializationResult: async () => { + connection.calls.push({ subtype: 'initialize' }) + if (options.exitBeforeInit) { + handlers.onExit?.(new Error(options.exitBeforeInit)) + return { models: [] } + } + if (options.initProof === 'session-start') { + handlers.onMessage?.({ + type: 'system', + subtype: 'hook_started', + hook_name: 'SessionStart:startup', + session_id: options.initSessionId ?? PROVIDER_SESSION_ID, + uuid: options.initUuid ?? 'init-uuid' + }) + } else if (options.initProof !== 'none') { + // Keys mirror the real system/init frame, which carries `model` but no + // effort of any kind: the current effort only comes back from + // get_settings. Never add a field the CLI does not send. + handlers.onMessage?.({ + type: 'system', + subtype: 'init', + session_id: options.initSessionId ?? PROVIDER_SESSION_ID, + uuid: options.initUuid ?? 'init-uuid', + model: options.initModel ?? 'claude-sonnet-5', + apiKeySource: 'none', + ...(options.capabilities ? { capabilities: options.capabilities } : {}) + }) + } + return { + models: [{ value: 'claude-sonnet', displayName: 'Sonnet' }], + ...(options.initAccount === undefined ? {} : { account: options.initAccount }) + } + }, + getSettings: async () => { + connection.calls.push({ subtype: 'get_settings' }) + // Shape measured from Claude Code 2.1.258: {applied, effective, sources}, + // and the only place the session's current effort is reported. + return ( + options.settings ?? { + applied: { model: 'claude-sonnet-5', effort: 'high', advisor: null, ultracode: false }, + effective: { model: 'claude-sonnet-5', effortLevel: 'high', env: {} }, + sources: {} + } + ) + }, + supportedModels: async () => { + connection.calls.push({ subtype: 'list_models' }) + return (routed('list_models') as unknown[] | undefined) ?? [] + }, + setModel: async (model) => { + connection.calls.push({ subtype: 'set_model', params: { model } }) + routed('set_model', { model }) + }, + setPermissionMode: async (mode) => { + connection.calls.push({ subtype: 'set_permission_mode', params: { mode } }) + routed('set_permission_mode', { mode }) + }, + applyFlagSettings: async (settings) => { + connection.calls.push({ subtype: 'apply_flag_settings', params: { settings } }) + routed('apply_flag_settings', { settings }) + }, + interrupt: async (interruptOptions) => { + connection.calls.push({ + subtype: 'interrupt', + params: interruptOptions?.cancelQueued ? { cancelQueued: true } : {} + }) + return routed('interrupt', interruptOptions) as + | Awaited> + | undefined + }, + cancelAsyncMessage: async (uuid) => { + connection.calls.push({ subtype: 'cancel_async_message', params: { uuid } }) + routed('cancel_async_message', { uuid }) + }, + send: async (message) => { + connection.sent.push(message) + if (message.type === 'user' && options.replayUuid !== null) { + const configuredReplayUuid = options.replayUuids + ? options.replayUuids[replayIndex++] + : options.replayUuid + const replayUuid = + configuredReplayUuid === undefined ? `user-uuid-${replayIndex}` : configuredReplayUuid + if (replayUuid !== null) { + handlers.onMessage?.({ + ...message, + uuid: replayUuid + }) + } + } + }, + exitVerdict: options.unprovenCloseVerdict ?? { root: 'live', tree: 'unverifiable' }, + close: async () => { + connection.closeCount += 1 + connection.closed = true + return options.unprovenCloseVerdict === undefined + } + } + connections.push(connection) + return connection + }) as typeof openClaudeStreamJsonConnection + return { connections, openConnection, routes } +} + +export function adapterFor( + claude: ReturnType, + launch: Partial = {}, + events: ClaudeStructuredSessionEvent[] = [], + persistedHandles: unknown[] = [], + initTimeoutMs?: number, + readTranscriptLeaf?: ClaudeStructuredSessionAdapterDeps['readTranscriptLeaf'], + persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'] +): ClaudeStructuredSessionAdapter { + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false, + ...launch + }), + onEvent: (event) => events.push(event), + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + ...(initTimeoutMs === undefined ? {} : { initTimeoutMs }), + dispatchAckTimeoutMs: 10, + persistHandle: + persistHandle ?? + (async (handle) => { + persistedHandles.push(handle) + }), + ...(readTranscriptLeaf ? { readTranscriptLeaf } : {}) + }) +} + +export async function acquired( + claude: ReturnType, + launch: Partial = {}, + events: ClaudeStructuredSessionEvent[] = [] +): Promise { + const adapter = adapterFor(claude, launch, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + return adapter +} + +export function tick(): Promise { + return new Promise((resolve) => setImmediate(resolve)) +} + +export function invokeCanUseTool( + connection: FakeConnection, + toolName: string, + requestId: string, + toolUseID: string, + extra: { + input?: Record + suggestions?: unknown[] + signal?: AbortSignal + } = {} +): { promise: Promise; settled: () => boolean } { + const options = { + requestId, + toolUseID, + signal: extra.signal ?? new AbortController().signal, + ...(extra.suggestions ? { suggestions: extra.suggestions } : {}) + } as unknown as Parameters>[2] + let done = false + const promise = Promise.resolve( + connection.handlers.canUseTool?.(toolName, extra.input ?? {}, options) + ).finally(() => { + done = true + }) + return { promise, settled: () => done } +} diff --git a/src/main/claude/claude-transcript-branch-proof.ts b/src/main/claude/claude-transcript-branch-proof.ts index d7065caa275..605f619eb92 100644 --- a/src/main/claude/claude-transcript-branch-proof.ts +++ b/src/main/claude/claude-transcript-branch-proof.ts @@ -5,6 +5,10 @@ const MAX_CLAUDE_TRANSCRIPT_ANCESTRY = 10_000 type TranscriptNode = { parentUuid: string | null sessionId: string | null + /** First line where this UUID was observed in the append-only transcript. */ + lineIndex: number + /** UUIDs from result/init/stream frames and sidechains are never leaves. */ + disallowedLeaf: boolean } export type ClaudeTranscriptBranchProof = { @@ -27,6 +31,54 @@ export class ClaudeTranscriptTailIncompleteError extends Error { } } +/** The sampled cursor is no longer present, so a root proof may still recover safely. */ +export class ClaudeTranscriptPreviousCursorMissingError extends Error { + constructor() { + super( + 'Claude transcript branch proof failed: previous cursor is missing from the session graph' + ) + this.name = 'ClaudeTranscriptPreviousCursorMissingError' + } +} + +function proveMainLineAncestry( + nodes: Map, + startUuid: string, + providerSessionId: string +): void { + const visited = new Set() + let cursor: string | null = startUuid + for (let depth = 0; cursor !== null && depth < MAX_CLAUDE_TRANSCRIPT_ANCESTRY; depth += 1) { + if (visited.has(cursor)) { + throw transcriptError('cycle in parentUuid ancestry') + } + visited.add(cursor) + const node = nodes.get(cursor) + if (!node || node.sessionId !== providerSessionId) { + throw transcriptError(`missing ancestor ${cursor}`) + } + if (node.disallowedLeaf) { + throw transcriptError(`ancestor ${cursor} is not on the main transcript`) + } + cursor = node.parentUuid + } + if (cursor !== null) { + throw transcriptError('ancestry exceeds the bounded proof limit') + } +} + +function proveAppendOrder(nodes: Map): void { + for (const node of nodes.values()) { + if (!node.parentUuid) { + continue + } + const parent = nodes.get(node.parentUuid) + if (parent && parent.lineIndex >= node.lineIndex) { + throw transcriptError('parent row follows descendant') + } + } +} + export function proveClaudeTranscriptBranchFromJsonl(input: { contents: string providerSessionId: string @@ -34,6 +86,7 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { }): ClaudeTranscriptBranchProof { const nodes = new Map() let leafUuid: string | null = null + let leafMarkerLineIndex = -1 const lines = input.contents.split('\n') for (const [index, line] of lines.entries()) { if (!line.trim()) { @@ -59,6 +112,7 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { throw transcriptError('invalid last-prompt marker') } leafUuid = markerLeaf + leafMarkerLineIndex = index } const uuid = nonEmptyString(row.uuid) if (!uuid) { @@ -70,27 +124,60 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { } const sessionId = nonEmptyString(row.sessionId) const existing = nodes.get(uuid) - if (existing && (existing.parentUuid !== parentUuid || existing.sessionId !== sessionId)) { + const disallowedLeaf = + row.isSidechain === true || + row.parent_tool_use_id != null || + row.type === 'result' || + row.type === 'stream_event' || + (row.type === 'system' && row.subtype === 'init') + if ( + existing && + (existing.parentUuid !== parentUuid || + existing.sessionId !== sessionId || + existing.disallowedLeaf !== disallowedLeaf) + ) { throw transcriptError(`record ${uuid} has conflicting ancestry`) } - nodes.set(uuid, { parentUuid, sessionId }) + nodes.set(uuid, { + parentUuid, + sessionId, + lineIndex: existing?.lineIndex ?? index, + disallowedLeaf + }) } if (!leafUuid) { throw transcriptError('missing last-prompt marker') } const leaf = nodes.get(leafUuid) - if (!leaf || leaf.sessionId !== input.providerSessionId) { + if (!leaf || leaf.sessionId !== input.providerSessionId || leaf.disallowedLeaf) { throw transcriptError('marker leaf is missing from the session graph') } + if (leaf.lineIndex > leafMarkerLineIndex) { + throw transcriptError('marker precedes its leaf record') + } const previousLeafUuid = input.previousLeafUuid if (!previousLeafUuid) { + proveMainLineAncestry(nodes, leafUuid, input.providerSessionId) + // A branch proof is based on an append-only snapshot. A child that appears + // before its claimed parent is not a post-snapshot descendant observation; + // accepting that graph would turn reordered/torn rows into durable ancestry. + proveAppendOrder(nodes) return { leafUuid, relation: 'initial' } } const previous = nodes.get(previousLeafUuid) - if (!previous || previous.sessionId !== input.providerSessionId) { - throw transcriptError('previous cursor is missing from the session graph') + if (!previous) { + throw new ClaudeTranscriptPreviousCursorMissingError() } + if (previous.sessionId !== input.providerSessionId || previous.disallowedLeaf) { + throw transcriptError('previous cursor is not on the main transcript') + } + // The latest marker can be equal to, or descend from, a sampled cursor. In + // either case prove the sampled cursor's own ancestry before accepting it; + // otherwise a cursor that descended through a parent-tool-use sidechain + // could be persisted and resumed as if it were on the main transcript. + proveMainLineAncestry(nodes, previousLeafUuid, input.providerSessionId) if (leafUuid === previousLeafUuid) { + proveAppendOrder(nodes) return { leafUuid, relation: 'same' } } const visited = new Set() @@ -104,8 +191,12 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { if (!node || node.sessionId !== input.providerSessionId) { throw transcriptError(`missing ancestor ${cursor}`) } + if (node.disallowedLeaf) { + throw transcriptError(`ancestor ${cursor} is not on the main transcript`) + } cursor = node.parentUuid if (cursor === previousLeafUuid) { + proveAppendOrder(nodes) return { leafUuid, relation: 'descendant' } } } @@ -126,3 +217,37 @@ export async function proveClaudeTranscriptBranch(input: { previousLeafUuid: input.previousLeafUuid }) } + +/** Re-run a durable branch proof from the transcript root when a sampled cursor is stale. */ +export async function readClaudeTranscriptLeafWithReproof(input: { + readTranscriptLeaf: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise + claudeConfigDir: string + providerSessionId: string + previousLeafUuid: string | null +}): Promise { + try { + return await input.readTranscriptLeaf({ + providerSessionId: input.providerSessionId, + previousLeafUuid: input.previousLeafUuid, + claudeConfigDir: input.claudeConfigDir + }) + } catch (error) { + // A missing cursor can be stale after compaction and is safe to re-prove from the root. A torn + // tail is still being written; dropping the cursor would make a later sibling look admissible. + if ( + input.previousLeafUuid === null || + !(error instanceof ClaudeTranscriptPreviousCursorMissingError) + ) { + throw error + } + return input.readTranscriptLeaf({ + providerSessionId: input.providerSessionId, + previousLeafUuid: null, + claudeConfigDir: input.claudeConfigDir + }) + } +} diff --git a/src/main/claude/claude-tui-exit.test.ts b/src/main/claude/claude-tui-exit.test.ts new file mode 100644 index 00000000000..6e1140b0f4d --- /dev/null +++ b/src/main/claude/claude-tui-exit.test.ts @@ -0,0 +1,159 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + completeClaudeTuiExit, + readClaudeTranscriptEntryUuid, + readClaudeTranscriptLeafUuid +} from './claude-tui-exit' + +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('Claude TUI exit', () => { + it('does not sample UUIDs from subagent stdout frames with a parent tool use', () => { + expect( + readClaudeTranscriptEntryUuid({ + type: 'assistant', + uuid: 'subagent-assistant', + parent_tool_use_id: 'parent-tool' + }) + ).toBeNull() + expect( + readClaudeTranscriptEntryUuid({ + type: 'assistant', + uuid: 'main-assistant', + parent_tool_use_id: null + }) + ).toBe('main-assistant') + }) + + it('reads the authoritative last-prompt leaf from a transcript tail', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-exit-')) + roots.push(root) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'user', uuid: 'user-one' }, + { type: 'assistant', uuid: 'assistant-one' }, + { type: 'last-prompt', leafUuid: 'chain-head' }, + { type: 'file-history-snapshot', snapshot: {} } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('chain-head') + }) + + it('falls back to the last persisted message when last-prompt metadata is absent', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-exit-')) + roots.push(root) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'user', uuid: 'user-one' }, + { type: 'assistant', uuid: 'assistant-one' }, + { type: 'system', subtype: 'init', uuid: 'init-frame' }, + { type: 'result', uuid: 'result-frame' }, + { type: 'stream_event', uuid: 'stream-event-frame' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('assistant-one') + }) + + it('ignores sidechain messages when selecting a fallback transcript leaf', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-sidechain-leaf-')) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'assistant', uuid: 'main-assistant' }, + { type: 'assistant', uuid: 'subagent-assistant', isSidechain: true }, + { type: 'result', uuid: 'result-frame' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('main-assistant') + }) + + it('persists the resumed chain head only after the exact Claude child exits', async () => { + let resolveExit!: (exit: { + pid: number + exitCode: number | null + signal: string | null + }) => void + const exitPromise = new Promise<{ + pid: number + exitCode: number | null + signal: string | null + }>((resolve) => { + resolveExit = resolve + }) + const persistHandle = vi.fn(async () => undefined) + const completion = completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: () => exitPromise, + sessionId: 'provider-session', + transcriptPath: '/accounts/claude/session.jsonl', + fence: 7, + persistHandle, + readLeafUuid: async () => 'tui-leaf', + linkId: 'tui-resumed-link', + now: () => 12 + }) + + expect(persistHandle).not.toHaveBeenCalled() + resolveExit({ pid: 4210, exitCode: 0, signal: null }) + + await expect(completion).resolves.toMatchObject({ + link: { + linkId: 'tui-resumed-link', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'tui-leaf' }, + origin: 'resumed', + mintedAtFence: 7, + observedAt: 12 + } + }) + expect(persistHandle).toHaveBeenCalledTimes(1) + }) + + it('refuses another process exit and a missing transcript leaf', async () => { + const persistHandle = vi.fn(async () => undefined) + await expect( + completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: async () => ({ pid: 4211, exitCode: 0, signal: null }), + sessionId: 'provider-session', + transcriptPath: '/session.jsonl', + fence: 2, + persistHandle, + readLeafUuid: async () => 'leaf' + }) + ).rejects.toThrow(/did not belong to the Claude child/) + await expect( + completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: async () => ({ pid: 4210, exitCode: 1, signal: null }), + sessionId: 'provider-session', + transcriptPath: '/session.jsonl', + fence: 2, + persistHandle, + readLeafUuid: async () => null + }) + ).rejects.toThrow(/resumable transcript leaf/) + expect(persistHandle).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/claude/claude-tui-exit.ts b/src/main/claude/claude-tui-exit.ts new file mode 100644 index 00000000000..3772e9c5e52 --- /dev/null +++ b/src/main/claude/claude-tui-exit.ts @@ -0,0 +1,119 @@ +import { open } from 'node:fs/promises' +import type { AgentSessionProviderHandleLink } from '../../shared/agent-session-provider-handle' +import { claudeProviderHandleLink } from './claude-structured-owner-identity' + +const TRANSCRIPT_TAIL_CHUNK_BYTES = 64 * 1024 +const TRANSCRIPT_TAIL_READ_LIMIT_BYTES = 4 * 1024 * 1024 + +type TranscriptLeafCandidate = { leafUuid: string; authoritative: boolean } + +function validLeafUuid(value: unknown): string | null { + if (typeof value !== 'string' || value.length === 0 || value.length > 512) { + return null + } + const hasControlCharacter = [...value].some((character) => { + const code = character.codePointAt(0) ?? 0 + return code <= 0x1f || code === 0x7f + }) + return value === value.trim() && !hasControlCharacter ? value : null +} + +export function readClaudeTranscriptEntryUuid(value: Record): string | null { + return value.isSidechain === true || + value.parent_tool_use_id != null || + (value.type !== 'user' && value.type !== 'assistant') + ? null + : validLeafUuid(value.uuid) +} + +function readLeafCandidate(line: string): TranscriptLeafCandidate | null { + try { + const value = JSON.parse(line) as Record + const lastPromptLeaf = value.type === 'last-prompt' ? validLeafUuid(value.leafUuid) : null + if (lastPromptLeaf) { + return { leafUuid: lastPromptLeaf, authoritative: true } + } + const messageLeaf = readClaudeTranscriptEntryUuid(value) + return messageLeaf ? { leafUuid: messageLeaf, authoritative: false } : null + } catch { + return null + } +} + +export async function readClaudeTranscriptLeafUuid(transcriptPath: string): Promise { + const file = await open(transcriptPath, 'r') + try { + const { size } = await file.stat() + let position = size + let suffix = '' + let fallback: string | null = null + let scanned = 0 + while (position > 0 && scanned < TRANSCRIPT_TAIL_READ_LIMIT_BYTES) { + const length = Math.min(TRANSCRIPT_TAIL_CHUNK_BYTES, position) + position -= length + scanned += length + const buffer = Buffer.alloc(length) + await file.read(buffer, 0, length, position) + const lines = `${buffer.toString('utf8')}${suffix}`.split(/\r?\n/) + suffix = position > 0 ? (lines.shift() ?? '') : '' + for (let index = lines.length - 1; index >= 0; index -= 1) { + const line = lines[index]?.trim() + if (!line) { + continue + } + const candidate = readLeafCandidate(line) + if (!candidate) { + continue + } + if (candidate.authoritative) { + return candidate.leafUuid + } + fallback ??= candidate.leafUuid + } + } + return fallback + } finally { + await file.close() + } +} + +export type ClaudeTuiChildExit = { + pid: number + exitCode: number | null + signal: string | null +} + +export async function completeClaudeTuiExit(input: { + childPid: number + waitForChildExit: () => Promise + sessionId: string + transcriptPath: string + fence: number + persistHandle: (link: AgentSessionProviderHandleLink) => Promise + readLeafUuid?: (transcriptPath: string) => Promise + linkId?: string + now?: () => number +}): Promise<{ + exit: ClaudeTuiChildExit + transcriptPath: string + link: AgentSessionProviderHandleLink +}> { + const exit = await input.waitForChildExit() + if (exit.pid !== input.childPid) { + throw new Error('The observed process exit did not belong to the Claude child.') + } + const leafUuid = await (input.readLeafUuid ?? readClaudeTranscriptLeafUuid)(input.transcriptPath) + if (!leafUuid) { + throw new Error('The exited Claude TUI did not persist a resumable transcript leaf.') + } + const link = claudeProviderHandleLink({ + sessionId: input.sessionId, + leafUuid, + resumed: true, + fence: input.fence, + ...(input.linkId ? { linkId: input.linkId } : {}), + observedAt: input.now?.() ?? Date.now() + }) + await input.persistHandle(link) + return { exit, transcriptPath: input.transcriptPath, link } +} diff --git a/src/main/claude/claude-tui-resume-launch.test.ts b/src/main/claude/claude-tui-resume-launch.test.ts new file mode 100644 index 00000000000..f907d1fcb4e --- /dev/null +++ b/src/main/claude/claude-tui-resume-launch.test.ts @@ -0,0 +1,227 @@ +import { chmodSync, mkdtempSync, mkdirSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { delimiter, join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { CLAUDE_AUTH_ENV_CONFLICT_MESSAGE } from '../claude-accounts/environment' +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' +import { createClaudeTuiResumeLaunchBuilder } from './claude-tui-resume-launch' + +function record(overrides: Partial = {}): AgentSessionRecord { + return { + sessionId: 'orca-session-1', + provider: 'claude', + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-folder', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/accounts/claude-one' }, + providerHandleChain: [ + { + linkId: 'created', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'leaf-one' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + ...overrides + } as AgentSessionRecord +} + +function makeExecutable(path: string): void { + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, '') + if (process.platform !== 'win32') { + chmodSync(path, 0o755) + } +} + +describe('Claude TUI resume launch', () => { + it('pins the workspace, account home, setting sources, and launch identity', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async (workspaceId) => `/workspaces/${workspaceId}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ + SELECTED_ACCOUNT: 'one', + ANTHROPIC_AUTH_TOKEN: 'selected-account-token' + }), + inheritedEnv: { + ANTHROPIC_API_KEY: 'inherited-gateway-key', + ANTHROPIC_BASE_URL: 'https://inherited-gateway.invalid', + CLAUDE_CODE_SESSION_ID: 'parent-session', + SAFE_PARENT: 'kept' + } + }) + + const launch = await build({ record: record(), spawnToken: 'spawn-one' }) + + expect(launch).toMatchObject({ + command: '/usr/local/bin/claude', + args: [ + '--setting-sources', + CLAUDE_DEFAULT_SETTING_SOURCES.join(','), + '--resume', + 'provider-session' + ], + cwd: '/workspaces/workspace-folder', + providerSessionId: 'provider-session', + resumeLeafUuid: 'leaf-one' + }) + expect(launch.env).toMatchObject({ + SAFE_PARENT: 'kept', + SELECTED_ACCOUNT: 'one', + CLAUDE_CONFIG_DIR: '/accounts/claude-one', + ORCA_AGENT_LAUNCH_TOKEN: 'spawn-one', + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-one', + ANTHROPIC_AUTH_TOKEN: 'selected-account-token' + }) + // System auth (the only state an explicit ANTHROPIC_AUTH_TOKEN overlay is legal in): + // the user's own inherited key is their sign-in and survives. The managed-account + // half — where it is stripped — is covered by 'structured-to-TUI handoff auth'. + expect(launch.env.ANTHROPIC_API_KEY).toBe('inherited-gateway-key') + // Endpoint selection is not credential material; the existing adapter pinning preserves it. + expect(launch.env.ANTHROPIC_BASE_URL).toBe('https://inherited-gateway.invalid') + expect(launch.env.CLAUDE_CODE_SESSION_ID).toBeUndefined() + }) + + it('pairs the resumed Claude CLI with its sibling Node runtime', async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-claude-resume-')) + const binDir = join(root, 'bin') + const claudeCommand = join(binDir, process.platform === 'win32' ? 'claude.cmd' : 'claude') + const nodeCommand = join(binDir, process.platform === 'win32' ? 'node.cmd' : 'node') + makeExecutable(claudeCommand) + makeExecutable(nodeCommand) + + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => claudeCommand, + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + resolveEnv: () => ({ PATH: '/usr/bin' }), + inheritedEnv: {} + }) + + const launch = await build({ record: record(), spawnToken: 'spawn' }) + + expect((launch.env.PATH ?? launch.env.Path)?.split(delimiter)[0]).toBe(binDir) + }) + + it('uses the durable session environment instead of current account settings', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ ANTHROPIC_AUTH_TOKEN: 'pinned-token' }), + inheritedEnv: {} + }) + + const launch = await build({ record: record(), spawnToken: 'spawn' }) + + expect(launch.env.ANTHROPIC_AUTH_TOKEN).toBe('pinned-token') + }) + + it('resolves the durable chain head instead of an earlier Claude leaf', async () => { + const nextRecord = record({ + providerHandleChain: [ + ...record().providerHandleChain, + { + linkId: 'resumed', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'leaf-two' }, + origin: 'resumed', + mintedAtFence: 2, + observedAt: 2 + } + ] + }) + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + await expect(build({ record: nextRecord, spawnToken: 'spawn-two' })).resolves.toMatchObject({ + providerSessionId: 'provider-session', + resumeLeafUuid: 'leaf-two' + }) + }) + + it('preserves durable Claude launch arguments before resume defaults', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + const launch = await build({ + record: record({ launchArgs: ['--model', 'claude-sonnet-4-5'] }), + spawnToken: 'spawn' + }) + + expect(launch.args.slice(0, 3)).toEqual(['--model', 'claude-sonnet-4-5', '--setting-sources']) + }) + + it('rejects missing Claude handles and unpinned account homes', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + await expect( + build({ record: record({ providerHandleChain: [] }), spawnToken: 'spawn' }) + ).rejects.toThrow('claude_tui_resume_handle_required') + await expect( + build({ + record: record({ accountHome: { variable: 'CODEX_HOME', path: '/wrong' } }), + spawnToken: 'spawn' + }) + ).rejects.toThrow(/CLAUDE_CONFIG_DIR/) + }) +}) + +// buildClaudeChildProcessEnv strips its inherited half unconditionally, so this module +// would have signed a system-auth user out of the session the structured path had just +// honoured. It is not wired up yet; the required policy is what stops the next caller +// from inheriting that. +describe('structured-to-TUI handoff auth', () => { + it('carries a system-auth user their own inherited credential', async () => { + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + inheritedEnv: { ANTHROPIC_API_KEY: 'sk-ant-SHELL', PATH: '/usr/bin' } + })({ record: record(), spawnToken: 'token-1' }) + + expect(launch.env.ANTHROPIC_API_KEY).toBe('sk-ant-SHELL') + }) + + it('still strips it once a managed account owns the credential', async () => { + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: { ANTHROPIC_API_KEY: 'sk-ant-SHELL', PATH: '/usr/bin' } + })({ record: record(), spawnToken: 'token-1' }) + + expect(launch.env.ANTHROPIC_API_KEY).toBeUndefined() + }) + + it('refuses a configured override of a pinned managed account, as the terminal path does', async () => { + await expect( + createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + resolveEnv: () => ({ ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' }), + inheritedEnv: {} + })({ record: record(), spawnToken: 'token-1' }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) +}) diff --git a/src/main/claude/claude-tui-resume-launch.ts b/src/main/claude/claude-tui-resume-launch.ts new file mode 100644 index 00000000000..8c09335fd9d --- /dev/null +++ b/src/main/claude/claude-tui-resume-launch.ts @@ -0,0 +1,101 @@ +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { agentSessionProviderHandleChainHead } from '../../shared/agent-session-provider-handle' +import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { resolveClaudeCommand } from '../codex-cli/command' +import { getSpawnArgsForWindows } from '../win32-utils' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + claudeAuthEnvCarriedForward, + hasClaudeAuthEnvConflict +} from '../claude-accounts/environment' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' +import { claudeConfigDirEnvPatch } from './claude-config-dir-pin' +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' + +export const CLAUDE_TUI_RESUME_BASE_ARGS = [ + '--setting-sources', + CLAUDE_DEFAULT_SETTING_SOURCES.join(',') +] as const + +export type ClaudeTuiResumeLaunch = { + command: string + args: string[] + cwd: string + env: Record + providerSessionId: string + resumeLeafUuid: string | null +} + +export type ClaudeTuiResumeLaunchBuilderDeps = { + resolveWorkspacePath: (workspaceId: string) => Promise + resolveCommand?: () => string + resolveEnv?: () => Record + inheritedEnv?: NodeJS.ProcessEnv + /** + * Required so whoever wires this module up has to answer the question rather than + * inherit the wrong default: buildClaudeChildProcessEnv strips its inherited half + * unconditionally, which would sign out a system-auth user whose own ANTHROPIC_* + * is their only credential. Build it with claudeStructuredAuthPolicyForSettings. + */ + resolveAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy +} + +export function createClaudeTuiResumeLaunchBuilder( + deps: ClaudeTuiResumeLaunchBuilderDeps +): (input: { record: AgentSessionRecord; spawnToken: string }) => Promise { + return async ({ record, spawnToken }) => { + if (record.provider !== 'claude') { + throw new Error(`session ${record.sessionId} is a ${record.provider} session`) + } + if (record.accountHome.variable !== 'CLAUDE_CONFIG_DIR') { + throw new Error(`claude sessions pin CLAUDE_CONFIG_DIR, not ${record.accountHome.variable}`) + } + const head = agentSessionProviderHandleChainHead(record.providerHandleChain) + if (head?.handle.provider !== 'claude') { + throw new Error('claude_tui_resume_handle_required') + } + + const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const { spawnCmd, spawnArgs } = getSpawnArgsForWindows(command, [ + ...(record.launchArgs ?? []), + ...CLAUDE_TUI_RESUME_BASE_ARGS, + '--resume', + head.handle.sessionId + ]) + const auth = await deps.resolveAuthPolicy() + const configuredEnv = deps.resolveEnv?.() ?? {} + if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(configuredEnv)) { + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + } + // The inherited half is always stripped downstream, so a system-auth user's own + // credential only reaches the resumed TUI if it is carried in the configured half. + const carriedAuth = auth.stripAuthEnv + ? {} + : claudeAuthEnvCarriedForward(deps.inheritedEnv ?? process.env) + // Compared against what the child would otherwise inherit, so the record's account + // home still wins over a diverging overlay without a needless pin. + const inheritedEnv = { ...(deps.inheritedEnv ?? process.env), ...configuredEnv } + const env = buildClaudeChildProcessEnv( + { + ...carriedAuth, + ...configuredEnv, + ...claudeConfigDirEnvPatch(record.accountHome.path, { env: inheritedEnv }), + ORCA_AGENT_LAUNCH_TOKEN: spawnToken, + [CLAUDE_SPAWN_TOKEN_ENV]: spawnToken + }, + { inheritedEnv: deps.inheritedEnv } + ) + const pairedEnv = withCliRuntimeOnPath(command, env, { platform: process.platform }) + + return { + command: spawnCmd, + args: spawnArgs, + cwd: await deps.resolveWorkspacePath(record.location.workspaceId), + env: pairedEnv, + providerSessionId: head.handle.sessionId, + resumeLeafUuid: head.handle.leafUuid + } + } +} diff --git a/src/main/claude/claude-tui-resume-proof.test.ts b/src/main/claude/claude-tui-resume-proof.test.ts new file mode 100644 index 00000000000..6d2e99d4b00 --- /dev/null +++ b/src/main/claude/claude-tui-resume-proof.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from 'vitest' +import { proveClaudeTuiResume, readClaudeTuiSessionStartEvidence } from './claude-tui-resume-proof' + +const SESSION = '91deba8d-a398-4b69-a05d-35041536fe8e' +const TRANSCRIPT = '/accounts/claude/projects/workspace/transcript.jsonl' + +function envelope(overrides: Record = {}): Record { + return { + launchToken: 'spawn-one', + payload: JSON.stringify({ + hook_event_name: 'SessionStart', + source: 'resume', + session_id: SESSION, + transcript_path: TRANSCRIPT, + ...overrides + }) + } +} + +describe('Claude TUI resume proof', () => { + it('reads SessionStart identity from the hook envelope', () => { + expect(readClaudeTuiSessionStartEvidence(envelope())).toEqual({ + hookEventName: 'SessionStart', + source: 'resume', + sessionId: SESSION, + transcriptPath: TRANSCRIPT, + launchToken: 'spawn-one' + }) + }) + + it('proves the exact launched session and transcript without terminal output', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-one', + waitForSessionStart: async () => envelope() + }) + ).resolves.toMatchObject({ sessionId: SESSION, transcriptPath: TRANSCRIPT }) + }) + + it.each([ + ['source', { source: 'startup' }, /resume SessionStart/], + ['session', { session_id: 'other-session' }, /different Claude session/], + ['transcript', { transcript_path: '/other/transcript.jsonl' }, /different Claude transcript/] + ])('rejects a mismatched %s', async (_name, overrides, expected) => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-one', + waitForSessionStart: async () => envelope(overrides) + }) + ).rejects.toThrow(expected) + }) + + it('rejects a SessionStart from another launched process', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-two', + waitForSessionStart: async () => envelope() + }) + ).rejects.toThrow(/different launched process/) + }) + + it('compares Windows paths using host path semantics', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: 'C:\\Users\\Dev\\session.jsonl', + expectedLaunchToken: 'spawn-one', + platform: 'win32', + waitForSessionStart: async () => + envelope({ transcript_path: 'c:\\users\\dev\\session.jsonl' }) + }) + ).resolves.toMatchObject({ sessionId: SESSION }) + }) +}) diff --git a/src/main/claude/claude-tui-resume-proof.ts b/src/main/claude/claude-tui-resume-proof.ts new file mode 100644 index 00000000000..f352423116e --- /dev/null +++ b/src/main/claude/claude-tui-resume-proof.ts @@ -0,0 +1,111 @@ +import { posix, win32 } from 'node:path' + +export type ClaudeTuiSessionStartEvidence = { + hookEventName: 'SessionStart' + source: 'resume' + sessionId: string + transcriptPath: string + launchToken: string +} + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +function nonEmptyString(value: unknown): string | null { + return typeof value === 'string' && value.length > 0 ? value : null +} + +function hookPayload(envelope: Record): Record | null { + if (typeof envelope.payload === 'string') { + try { + return record(JSON.parse(envelope.payload)) + } catch { + return null + } + } + return record(envelope.payload) ?? envelope +} + +export function readClaudeTuiSessionStartEvidence( + value: unknown +): ClaudeTuiSessionStartEvidence | null { + const envelope = record(value) + if (!envelope) { + return null + } + const payload = hookPayload(envelope) + if (!payload) { + return null + } + const hookEventName = nonEmptyString(payload.hook_event_name ?? payload.hookEventName) + const source = nonEmptyString(payload.source) + const sessionId = nonEmptyString(payload.session_id ?? payload.sessionId) + const transcriptPath = nonEmptyString(payload.transcript_path ?? payload.transcriptPath) + const launchToken = nonEmptyString(envelope.launchToken ?? payload.launchToken) + return hookEventName === 'SessionStart' && + source === 'resume' && + sessionId && + transcriptPath && + launchToken + ? { hookEventName, source, sessionId, transcriptPath, launchToken } + : null +} + +function comparablePath(value: string, platform: NodeJS.Platform): string | null { + if (value.includes('\0')) { + return null + } + const path = platform === 'win32' ? win32 : posix + if (!path.isAbsolute(value)) { + return null + } + const normalized = path.normalize(value) + return platform === 'win32' ? normalized.toLowerCase() : normalized +} + +export async function proveClaudeTuiResume(input: { + expectedSessionId: string + expectedTranscriptPath: string + expectedLaunchToken: string + waitForSessionStart: () => Promise + timeoutMs?: number + platform?: NodeJS.Platform +}): Promise { + const timeoutMs = input.timeoutMs ?? 15_000 + let timer: ReturnType | undefined + try { + const evidence = readClaudeTuiSessionStartEvidence( + await Promise.race([ + input.waitForSessionStart(), + new Promise((_resolve, reject) => { + timer = setTimeout( + () => reject(new Error('The agent terminal did not prove the expected Claude resume.')), + timeoutMs + ) + timer.unref?.() + }) + ]) + ) + if (!evidence) { + throw new Error('The agent terminal did not emit a Claude resume SessionStart proof.') + } + if (evidence.launchToken !== input.expectedLaunchToken) { + throw new Error('The Claude resume proof came from a different launched process.') + } + if (evidence.sessionId !== input.expectedSessionId) { + throw new Error('The agent terminal resumed a different Claude session.') + } + const platform = input.platform ?? process.platform + const expectedPath = comparablePath(input.expectedTranscriptPath, platform) + const observedPath = comparablePath(evidence.transcriptPath, platform) + if (!expectedPath || !observedPath || observedPath !== expectedPath) { + throw new Error('The agent terminal resumed a different Claude transcript.') + } + return evidence + } finally { + clearTimeout(timer) + } +} diff --git a/src/main/claude/claude-tui-resume-real-binary.integration.test.ts b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts new file mode 100644 index 00000000000..9ba3daf2285 --- /dev/null +++ b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts @@ -0,0 +1,279 @@ +import { spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { homedir, tmpdir } from 'node:os' +import { join } from 'node:path' +import * as pty from 'node-pty' +import { afterEach, describe, expect, it } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { resolveClaudeCommand } from '../codex-cli/command' +import { readStructuredTuiProcessIdentity } from '../runtime/structured-tui-process-identity' +import { getSpawnArgsForWindows } from '../win32-utils' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { createClaudeTuiResumeLaunchBuilder } from './claude-tui-resume-launch' +import { proveClaudeTuiResume } from './claude-tui-resume-proof' + +const command = resolveClaudeCommand() +const claudeAvailable = + spawnSync(command, ['--version'], { stdio: 'ignore', timeout: 5_000 }).status === 0 +const authStatusLaunch = getSpawnArgsForWindows(command, ['auth', 'status', '--json']) +const claudeAuthenticated = (() => { + if (!claudeAvailable) { + return false + } + const result = spawnSync(authStatusLaunch.spawnCmd, authStatusLaunch.spawnArgs, { + encoding: 'utf8', + windowsHide: true, + timeout: 5_000 + }) + return result.status === 0 && /"loggedIn"\s*:\s*true/.test(result.stdout) +})() +const roots: string[] = [] +const transcripts: string[] = [] + +function shellQuote(value: string): string { + return process.platform === 'win32' + ? `"${value.replace(/"/g, '""')}"` + : `'${value.replace(/'/g, `'"'"'`)}'` +} + +async function installCaptureHook( + root: string +): Promise<{ eventsPath: string; settingsPath: string }> { + const scriptPath = join(root, 'capture-session-start.cjs') + const eventsPath = join(root, 'session-start.jsonl') + const settingsPath = join(root, 'settings.json') + await writeFile( + scriptPath, + [ + "const { appendFileSync } = require('node:fs')", + "let input = ''", + "process.stdin.setEncoding('utf8')", + "process.stdin.on('data', (chunk) => { input += chunk })", + "process.stdin.on('end', () => {", + ' const payload = JSON.parse(input)', + ' payload.launchToken = process.env.ORCA_AGENT_LAUNCH_TOKEN', + ' appendFileSync(process.argv[2], `${JSON.stringify(payload)}\\n`)', + '})', + '' + ].join('\n') + ) + await writeFile( + settingsPath, + JSON.stringify({ + theme: 'dark', + hooks: { + SessionStart: [ + { + hooks: [ + { + type: 'command', + command: [process.execPath, scriptPath, eventsPath].map(shellQuote).join(' ') + } + ] + } + ] + } + }) + ) + return { eventsPath, settingsPath } +} + +async function waitForHook( + eventsPath: string, + source: 'startup' | 'resume' +): Promise> { + const deadline = Date.now() + 15_000 + while (Date.now() < deadline) { + const contents = await readFile(eventsPath, 'utf8').catch(() => '') + for (const line of contents.split(/\r?\n/)) { + if (!line.trim()) { + continue + } + const event = JSON.parse(line) as Record + if (event.hook_event_name === 'SessionStart' && event.source === source) { + return event + } + } + await new Promise((resolve) => setTimeout(resolve, 50)) + } + throw new Error(`Claude did not emit a ${source} SessionStart hook`) +} + +type RunningTui = { proc: pty.IPty; exited: Promise } + +function spawnResumeTui(args: string[], env: Record): RunningTui { + const direct = process.platform === 'win32' + const proc = pty.spawn( + direct ? command : process.env.SHELL || '/bin/zsh', + direct ? args : ['-l'], + { + name: 'xterm-256color', + cols: 100, + rows: 30, + cwd: process.cwd(), + env: { ...env, TERM: 'xterm-256color' } + } + ) + if (!direct) { + setTimeout(() => { + proc.write(`${[command, ...args].map(shellQuote).join(' ')}\r`) + }, 100).unref() + } + return { proc, exited: new Promise((resolve) => proc.onExit(() => resolve())) } +} + +function structuredIdentity(providerSessionId: string): AgentSessionJournalIdentity { + return { + sessionId: 'orca-real-claude-resume', + workspaceId: 'workspace-real', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: providerSessionId, leafUuid: null } + } +} + +async function waitForStructuredResult(events: ClaudeStructuredSessionEvent[]): Promise { + const deadline = Date.now() + 30_000 + while (Date.now() < deadline) { + if (events.some((event) => event.type === 'message' && event.message.type === 'result')) { + return + } + await new Promise((resolve) => setTimeout(resolve, 50)) + } + throw new Error('Claude structured session did not finish its product-path turn') +} + +async function stopTui(tui: RunningTui): Promise { + try { + tui.proc.kill('SIGKILL') + } catch { + return + } + await Promise.race([ + tui.exited, + new Promise((_resolve, reject) => + setTimeout(() => reject(new Error('Claude TUI did not exit after cleanup')), 5_000) + ) + ]) +} + +afterEach(async () => { + await Promise.all(transcripts.splice(0).map((path) => rm(path, { force: true }))) + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe.skipIf(!claudeAuthenticated)('real Claude TUI resume proof', () => { + it('resumes a product-created structured session and proves its exact child', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-resume-')) + roots.push(root) + const { eventsPath, settingsPath } = await installCaptureHook(root) + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: command, + options: { + ...CLAUDE_STRUCTURED_BASE_OPTIONS, + extraArgs: { ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs, settings: settingsPath }, + sessionId: providerSessionId + }, + cwd: process.cwd(), + claudeConfigDir, + providerSessionId, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1 + }) + let resumed: RunningTui | null = null + try { + const acquisition = await adapter.acquire({ + identity: structuredIdentity(providerSessionId), + fence: 1, + spawnToken: 'real-create' + }) + await expect( + adapter.dispatch({ + sessionId: 'orca-real-claude-resume', + clientMessageId: 'real-product-turn', + fence: 1, + body: { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'Reply only with ORCA_RESUME_READY.' }] + } + }) + ).resolves.toMatchObject({ state: 'accepted' }) + await waitForStructuredResult(events) + const started = await waitForHook(eventsPath, 'startup') + const transcriptPath = String(started.transcript_path) + transcripts.push(transcriptPath) + expect(started.session_id).toBe(providerSessionId) + await adapter.closeAll() + + const record = { + sessionId: 'orca-real-claude-resume', + provider: 'claude', + location: { workspaceId: 'workspace-real' }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: claudeConfigDir }, + providerHandleChain: [ + { + linkId: 'created-real', + handle: acquisition.link.handle, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ] + } as AgentSessionRecord + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => process.cwd(), + resolveCommand: () => command, + // The real binary authenticates from the developer's own environment here, + // which is the system-auth case: stripping it would sign the resume out. + resolveAuthPolicy: () => ({ stripAuthEnv: false }) + })({ record, spawnToken: 'real-resume' }) + resumed = spawnResumeTui([...launch.args, '--settings', settingsPath], launch.env) + let resumedOutput = '' + resumed.proc.onData((data) => { + resumedOutput = `${resumedOutput}${data}`.slice(-4_000) + }) + + const [processIdentity, proof] = await Promise.all([ + readStructuredTuiProcessIdentity({ + hostId: 'local', + rootPid: resumed.proc.pid, + spawnToken: 'real-resume', + agent: 'claude' + }), + proveClaudeTuiResume({ + expectedSessionId: providerSessionId, + expectedTranscriptPath: transcriptPath, + expectedLaunchToken: 'real-resume', + waitForSessionStart: () => waitForHook(eventsPath, 'resume') + }).catch((error) => { + throw new Error(`${String(error)}\nClaude output: ${resumedOutput}`) + }) + ]) + expect(processIdentity).toMatchObject({ + hostId: 'local', + spawnToken: 'real-resume', + pid: expect.any(Number) + }) + expect(proof).toMatchObject({ sessionId: providerSessionId, transcriptPath }) + } finally { + await adapter.closeAll() + if (resumed) { + await stopTui(resumed) + } + } + }, 30_000) +}) diff --git a/src/main/codex/codex-app-server-session.ts b/src/main/codex/codex-app-server-session.ts index 2e176e13ae2..35537f59d0d 100644 --- a/src/main/codex/codex-app-server-session.ts +++ b/src/main/codex/codex-app-server-session.ts @@ -2,6 +2,7 @@ import { spawn, type ChildProcess, type ChildProcessWithoutNullStreams } from 'n import { waitForProcessExitUntil } from './codex-process-exit-deadline' import { stderrIndicatesMissingAppServer } from './codex-app-server-capability-signal' import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { admitProcessTreeKill } from '../../shared/child-process/process-tree-kill-gate' // Why: `codex app-server` is Orca's sanctioned RPC surface into Codex-owned // state (hook trust hashes, the sqlite thread index). This module owns the @@ -74,6 +75,18 @@ export function killCodexAppServerProcessTree( const platform = options.platform ?? process.platform const spawnImpl = options.spawnImpl ?? spawn if (platform === 'win32' && child.pid) { + if ( + !admitProcessTreeKill({ + pid: child.pid, + site: 'codex-app-server-session-deadline', + scope: 'win-taskkill-tree' + }) + ) { + // Refusal blocks the tree walk, not the termination: the root kill is + // handle-addressed, so it cannot reach the recycled pid we refused. + child.kill('SIGKILL') + return + } try { // Why: npm-installed Codex runs behind cmd.exe; killing only that wrapper // leaves the app-server child alive after a timeout or failed shutdown. diff --git a/src/main/codex/codex-prompt-registry-bounds.ts b/src/main/codex/codex-prompt-registry-bounds.ts index 84b3cda6151..e5f79fe3a31 100644 --- a/src/main/codex/codex-prompt-registry-bounds.ts +++ b/src/main/codex/codex-prompt-registry-bounds.ts @@ -20,10 +20,7 @@ export function codexJournalPromptIdPart(value: string): string { } const suffix = `#${digestPayload(value).slice(0, 32)}` const bounded = boundPayload(value, { - inlineHeadBytes: CODEX_JOURNAL_PROMPT_ID_COMPONENT_MAX_BYTES - suffix.length, - maxSessionBytes: Number.MAX_SAFE_INTEGER, - maxAppendsPerWindow: Number.MAX_SAFE_INTEGER, - appendWindowMs: Number.MAX_SAFE_INTEGER + inlineHeadBytes: CODEX_JOURNAL_PROMPT_ID_COMPONENT_MAX_BYTES - suffix.length }) return `${bounded.head}${suffix}` } diff --git a/src/main/codex/codex-structured-item-stream-bounds.ts b/src/main/codex/codex-structured-item-stream-bounds.ts index e84d8dd2efa..572c954b010 100644 --- a/src/main/codex/codex-structured-item-stream-bounds.ts +++ b/src/main/codex/codex-structured-item-stream-bounds.ts @@ -17,14 +17,8 @@ export function codexStructuredItemKey(threadId: string, itemId: string): string return `${key.slice(0, 960)}:${(hash >>> 0).toString(16)}` } -export function pendingPatchBytes(pending: { - body: unknown - blobs: readonly { payload: string }[] -}): number { - return ( - Buffer.byteLength(JSON.stringify(pending.body), 'utf8') + - pending.blobs.reduce((total, blob) => total + Buffer.byteLength(blob.payload, 'utf8'), 0) - ) +export function pendingPatchBytes(pending: { body: unknown }): number { + return Buffer.byteLength(JSON.stringify(pending.body), 'utf8') } export function boundStreamItem(item: Record): Record { diff --git a/src/main/codex/codex-structured-item-stream-contracts.ts b/src/main/codex/codex-structured-item-stream-contracts.ts index cf8aff5795d..f252d048210 100644 --- a/src/main/codex/codex-structured-item-stream-contracts.ts +++ b/src/main/codex/codex-structured-item-stream-contracts.ts @@ -24,7 +24,6 @@ export type CodexItemStreamState = { export type CodexPendingItemPatch = { identity: AgentJournalItemIdentity body: NonNullable['body']> - blobs: ReturnType['blobs'] } export type CodexStructuredItemStreamAdmission = diff --git a/src/main/codex/codex-structured-item-streams.ts b/src/main/codex/codex-structured-item-streams.ts index dda5e9688bd..a765f8339da 100644 --- a/src/main/codex/codex-structured-item-streams.ts +++ b/src/main/codex/codex-structured-item-streams.ts @@ -103,8 +103,8 @@ export function createCodexStructuredItemStreams( } const options = { coalescingKey: `checkpoint:${agentJournalItemKey(state.identity)}` } const admission = deps.sink.tryAppendItem - ? deps.sink.tryAppendItem(state.identity, translated.body, translated.blobs, options) - : (deps.sink.appendItem(state.identity, translated.body, translated.blobs, options), + ? deps.sink.tryAppendItem(state.identity, translated.body, options) + : (deps.sink.appendItem(state.identity, translated.body, options), { accepted: true as const }) if (!admission.accepted) { return false @@ -167,9 +167,8 @@ export function createCodexStructuredItemStreams( } for (const [key, pending] of pendingPatches) { const admission = deps.sink.tryAppendItem - ? deps.sink.tryAppendItem(pending.identity, pending.body, pending.blobs) - : (deps.sink.appendItem(pending.identity, pending.body, pending.blobs), - { accepted: true as const }) + ? deps.sink.tryAppendItem(pending.identity, pending.body) + : (deps.sink.appendItem(pending.identity, pending.body), { accepted: true as const }) if (!admission.accepted) { flushed = false continue @@ -193,9 +192,8 @@ export function createCodexStructuredItemStreams( return { accepted: true } } const admission = deps.sink.tryAppendItem - ? deps.sink.tryAppendItem(pending.identity, pending.body, pending.blobs) - : (deps.sink.appendItem(pending.identity, pending.body, pending.blobs), - { accepted: true as const }) + ? deps.sink.tryAppendItem(pending.identity, pending.body) + : (deps.sink.appendItem(pending.identity, pending.body), { accepted: true as const }) if (!admission.accepted) { return admission } @@ -237,8 +235,7 @@ export function createCodexStructuredItemStreams( if (translated.body) { const nextPending: CodexPendingItemPatch = { identity: state.identity, - body: translated.body, - blobs: translated.blobs + body: translated.body } const previous = pendingPatches.get(key) const previousBytes = previous ? pendingPatchBytes(previous) : 0 diff --git a/src/main/codex/codex-structured-item-translation.test.ts b/src/main/codex/codex-structured-item-translation.test.ts index 63c3d59b0b9..f0f843e4ed6 100644 --- a/src/main/codex/codex-structured-item-translation.test.ts +++ b/src/main/codex/codex-structured-item-translation.test.ts @@ -220,12 +220,6 @@ describe('codex item bodies', () => { throw new Error('expected bounded command output') } expect(body.output.head.length).toBeLessThan(20_000) - expect(translated.blobs).toEqual([ - { - digest: body.output.digest, - payload: output - } - ]) }) it('continues to accept camel-case command completion output', () => { diff --git a/src/main/codex/codex-structured-item-translation.ts b/src/main/codex/codex-structured-item-translation.ts index f640ad5fb3d..19052dca365 100644 --- a/src/main/codex/codex-structured-item-translation.ts +++ b/src/main/codex/codex-structured-item-translation.ts @@ -172,7 +172,6 @@ function commandState(item: CodexThreadItem): 'running' | 'completed' | 'failed' export type CodexJournalItem = { body: AgentJournalItemBody | null - blobs: { digest: string; payload: string }[] handled: boolean } @@ -190,10 +189,6 @@ function commandItem(item: CodexThreadItem): CodexJournalItem { state: commandState(item), ...(bounded === null ? {} : { output: bounded.bounded }) }, - blobs: - output !== null && bounded?.bounded.truncated - ? [{ digest: bounded.bounded.digest, payload: output }] - : [], handled: true } } @@ -215,7 +210,6 @@ function fileChangeItem(item: CodexThreadItem): CodexJournalItem { input: boundToolInput({ changes: item.changes ?? null }, DEFAULT_JOURNAL_PAYLOAD_LIMITS), state: commandState(item) }, - blobs: [], handled: true } } @@ -227,7 +221,6 @@ function fileChangeItem(item: CodexThreadItem): CodexJournalItem { path: changes.length === 1 ? changes[0]!.path : `${changes.length} files`, patch: bounded }, - blobs: bounded.truncated ? [{ digest: bounded.digest, payload: patch }] : [], handled: true } } @@ -246,7 +239,6 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem { blocks.length === 0 ? null : { kind: 'message', role: item.type === 'userMessage' ? 'user' : 'assistant', blocks }, - blobs: [], handled: true } } @@ -266,14 +258,11 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem { text === null ? null : { kind: 'status', text: boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text }, - blobs: [], handled: true } } const unhandled = unhandledProviderFrameJournalItem('codex', `item:${item.type}`, item) - return unhandled - ? { body: unhandled.body, blobs: unhandled.blobs, handled: false } - : { body: null, blobs: [], handled: true } + return unhandled ? { body: unhandled.body, handled: false } : { body: null, handled: true } } export function codexItemBody(item: CodexThreadItem): AgentJournalItemBody | null { @@ -292,7 +281,7 @@ export function codexStreamingMessageBody(text: string): AgentJournalItemBody { /** Snapshot body for any item-level stream, keyed onto its parent item. */ export function codexStreamingJournalItem(item: CodexThreadItem, text: string): CodexJournalItem { if (item.type === 'agentMessage') { - return { body: codexStreamingMessageBody(text), blobs: [], handled: true } + return { body: codexStreamingMessageBody(text), handled: true } } if (item.type === 'commandExecution') { return commandItem({ ...item, aggregatedOutput: text }) @@ -304,10 +293,9 @@ export function codexStreamingJournalItem(item: CodexThreadItem, text: string): const bounded = boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).bounded return { body: { kind: 'diff', path: path ?? 'pending patch', patch: bounded }, - blobs: bounded.truncated ? [{ digest: bounded.digest, payload: text }] : [], handled: true } } const bounded = boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS) - return { body: { kind: 'status', text: bounded.text }, blobs: [], handled: true } + return { body: { kind: 'status', text: bounded.text }, handled: true } } diff --git a/src/main/codex/codex-structured-journal-generic-frames.ts b/src/main/codex/codex-structured-journal-generic-frames.ts index f6b9ff375d3..6c211a8f5a4 100644 --- a/src/main/codex/codex-structured-journal-generic-frames.ts +++ b/src/main/codex/codex-structured-journal-generic-frames.ts @@ -91,13 +91,11 @@ export class CodexJournalGenericFrames { const admission = this.deps.sink.tryAppendItem ? this.deps.sink.tryAppendItem( { provider: 'orca', clientMessageId: `provider-frame:codex:${this.fallbackSequence}` }, - translated.body, - translated.blobs + translated.body ) : (this.deps.sink.appendItem( { provider: 'orca', clientMessageId: `provider-frame:codex:${this.fallbackSequence}` }, - translated.body, - translated.blobs + translated.body ), CODEX_JOURNAL_ADMITTED) if (!admission.accepted) { @@ -136,13 +134,11 @@ export class CodexJournalGenericFrames { kind: 'status', text }, - [], { coalescingKey: `provider-frame-suppressed:codex:${bucket}` } ) : (this.deps.sink.appendItem( { provider: 'orca', clientMessageId: `provider-frame-suppressed:codex:${bucket}` }, { kind: 'status', text }, - [], { coalescingKey: `provider-frame-suppressed:codex:${bucket}` } ), CODEX_JOURNAL_ADMITTED) diff --git a/src/main/codex/codex-structured-journal-items.ts b/src/main/codex/codex-structured-journal-items.ts index 2b5d8c04bba..0e4e3f5a900 100644 --- a/src/main/codex/codex-structured-journal-items.ts +++ b/src/main/codex/codex-structured-journal-items.ts @@ -2,7 +2,7 @@ import type { AgentJournalItemBody, AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' -import { requiresTerminalSettlement } from '../native-chat/agent-session-journal/journal-lifecycle-capacity' +import { requiresTerminalSettlement } from '../native-chat/agent-session-journal/journal-terminal-settlement' import { codexItemIdentity, codexJournalItem, @@ -126,19 +126,13 @@ export class CodexJournalItems { return CODEX_JOURNAL_ADMITTED } if (method === 'item/completed') { - const admission = appendCodexLifecycleItem( - this.deps.sink, - identity, - translated.body, - translated.blobs - ) + const admission = appendCodexLifecycleItem(this.deps.sink, identity, translated.body) return admission.accepted ? publishCodexLifecycle(this.deps.sink) : admission } const options = requiresTerminalSettlement(translated.body) ? { lifecycle: true } : {} const admission = this.deps.sink.tryAppendItem - ? this.deps.sink.tryAppendItem(identity, translated.body, translated.blobs, options) - : (this.deps.sink.appendItem(identity, translated.body, translated.blobs), - CODEX_JOURNAL_ADMITTED) + ? this.deps.sink.tryAppendItem(identity, translated.body, options) + : (this.deps.sink.appendItem(identity, translated.body), CODEX_JOURNAL_ADMITTED) if (!admission.accepted) { return admission } diff --git a/src/main/codex/codex-structured-journal-settlement.ts b/src/main/codex/codex-structured-journal-settlement.ts index f4378d1a16f..2b8eb627d4f 100644 --- a/src/main/codex/codex-structured-journal-settlement.ts +++ b/src/main/codex/codex-structured-journal-settlement.ts @@ -243,14 +243,12 @@ function appendLifecycleMutations( for (const mutation of chunk) { if (mutation.kind === 'item') { if (sink.tryAppendItem) { - admission = sink.tryAppendItem(mutation.identity, mutation.body, [], { - lifecycle: true - }) + admission = sink.tryAppendItem(mutation.identity, mutation.body, { lifecycle: true }) if (!admission.accepted) { return admission } } else { - sink.appendItem(mutation.identity, mutation.body, [], { lifecycle: true }) + sink.appendItem(mutation.identity, mutation.body, { lifecycle: true }) } } else { if (sink.tryAppendTombstone) { diff --git a/src/main/codex/codex-structured-journal-sink.ts b/src/main/codex/codex-structured-journal-sink.ts index b135c89ec0a..5c4ecec9658 100644 --- a/src/main/codex/codex-structured-journal-sink.ts +++ b/src/main/codex/codex-structured-journal-sink.ts @@ -4,7 +4,6 @@ import type { } from '../../shared/agent-session-journal-types' import type { StructuredAgentSessionEventSink, - StructuredAgentSessionJournalBlob, StructuredAgentSessionSinkAdmission } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' import type { CodexPendingJournalPrompt } from './codex-structured-journal-settlement' @@ -20,13 +19,12 @@ function criticalAdmission( export function appendCodexLifecycleItem( sink: StructuredAgentSessionEventSink, identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly StructuredAgentSessionJournalBlob[] = [] + body: AgentJournalItemBody ): CodexJournalTranslationAdmission { if (sink.tryAppendItem) { - return criticalAdmission(sink.tryAppendItem(identity, body, blobs, { lifecycle: true })) + return criticalAdmission(sink.tryAppendItem(identity, body, { lifecycle: true })) } - sink.appendItem(identity, body, blobs, { lifecycle: true }) + sink.appendItem(identity, body, { lifecycle: true }) return CODEX_JOURNAL_ADMITTED } diff --git a/src/main/codex/codex-structured-journal-translation-streams.test.ts b/src/main/codex/codex-structured-journal-translation-streams.test.ts index c06d468a1ba..a9bc79b71c4 100644 --- a/src/main/codex/codex-structured-journal-translation-streams.test.ts +++ b/src/main/codex/codex-structured-journal-translation-streams.test.ts @@ -89,11 +89,11 @@ describe('codex journal translation', () => { const { translator, tap } = translatorWith() let rejectTerminal = true const appendItem = tap.sink.appendItem - tap.sink.tryAppendItem = (identity, body, blobs, options) => { + tap.sink.tryAppendItem = (identity, body, options) => { if (rejectTerminal && body.kind === 'tool-call' && body.state === 'failed') { return { accepted: false as const, reason: 'backpressure' as const } } - appendItem(identity, body, blobs, options) + appendItem(identity, body, options) return { accepted: true as const } } for (let index = 0; index <= 256; index += 1) { diff --git a/src/main/codex/codex-structured-journal-translation-turns.ts b/src/main/codex/codex-structured-journal-translation-turns.ts index 3f295d2add1..06bb28f85f9 100644 --- a/src/main/codex/codex-structured-journal-translation-turns.ts +++ b/src/main/codex/codex-structured-journal-translation-turns.ts @@ -40,7 +40,6 @@ export function publishCodexTurnLifecycle(input: { text: 'Codex is working…', turnLifecycle: { turnId: input.turnId, state: input.state } }, - [], { lifecycle: true } ) : (input.sink.appendItem( @@ -50,7 +49,6 @@ export function publishCodexTurnLifecycle(input: { text: 'Codex is working…', turnLifecycle: { turnId: input.turnId, state: input.state } }, - [], { lifecycle: true } ), ADMITTED) diff --git a/src/main/codex/codex-structured-session-close.test.ts b/src/main/codex/codex-structured-session-close.test.ts index 4238c75de9e..b04e7bc2540 100644 --- a/src/main/codex/codex-structured-session-close.test.ts +++ b/src/main/codex/codex-structured-session-close.test.ts @@ -11,6 +11,8 @@ import { } from './codex-structured-session-adapter' import { handleCodexSessionExit } from './codex-structured-session-close' import type { CodexSession } from './codex-structured-session-state' +import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router' const THREAD = 'thread-1' @@ -60,6 +62,16 @@ function adapterFixture() { return { adapter, connections, events } } +function claudeAdapterStub(): StructuredAgentSessionAdapter { + return { + acquire: vi.fn(async () => ({ process: { pid: 1 } }) as never), + dispatch: vi.fn(), + cancelTurn: vi.fn(), + answerPrompt: vi.fn(), + setOption: vi.fn() + } +} + describe('Codex structured session close lifecycle', () => { it('forwards a one-shot exit when lifecycle admission is rejected', () => { const connection: CodexAppServerConnection = { @@ -164,4 +176,27 @@ describe('Codex structured session close lifecycle', () => { { cause: 'unexpected-exit', reason: 'sink failed', fence: 7 } ]) }) + + it('routes Codex sink-failure recovery through force-close and preserves unexpected-exit settlement', async () => { + const { adapter, connections, events } = adapterFixture() + const router = new StructuredAgentSessionAdapterRouter( + { claude: claudeAdapterStub(), codex: adapter }, + async () => {} + ) + await router.acquire({ identity: identity('session-1'), fence: 7, spawnToken: 'spawn-1' }) + const current = connections[0] + if (!current) { + throw new Error('missing connection') + } + current.connection.close = async () => { + current.handlers.onExit?.(new Error('journal sink failed')) + return true + } + + const forceCloseSession = router.forceCloseSession + await expect(forceCloseSession('session-1')).resolves.toBe(true) + expect(events.filter((event) => event.type === 'ended')).toMatchObject([ + { cause: 'unexpected-exit', reason: 'journal sink failed', fence: 7 } + ]) + }) }) diff --git a/src/main/codex/codex-structured-turn-processes.ts b/src/main/codex/codex-structured-turn-processes.ts index 6bb4b960a1b..163cbb098fb 100644 --- a/src/main/codex/codex-structured-turn-processes.ts +++ b/src/main/codex/codex-structured-turn-processes.ts @@ -57,6 +57,8 @@ async function terminateWindowsAddedProcesses( const added = current.filter((row) => baseline.get(row.pid) !== windowsIdentity(row)) const addedPids = new Set(added.map((row) => row.pid)) const roots = added.filter((row) => !addedPids.has(row.ppid)) + // Added roots come from a table walk, not a spawn, so a refused tree walk has + // no handle to fall back to: the row stays in `remaining` and this reports false. await Promise.all( roots.map((row) => terminateWindowsProcessTree(row.pid, { site: 'codex-turn-added-roots' })) ) diff --git a/src/main/codex/codex-turn-ordinals.ts b/src/main/codex/codex-turn-ordinals.ts index 7087c28e4ac..e2b4132d2b2 100644 --- a/src/main/codex/codex-turn-ordinals.ts +++ b/src/main/codex/codex-turn-ordinals.ts @@ -37,10 +37,7 @@ export class CodexTurnOrdinals { const suffix = `#${digestPayload(value).slice(0, 24)}` return `${ boundPayload(encoded, { - inlineHeadBytes: 256 - Buffer.byteLength(suffix, 'utf8'), - maxSessionBytes: Number.MAX_SAFE_INTEGER, - maxAppendsPerWindow: Number.MAX_SAFE_INTEGER, - appendWindowMs: Number.MAX_SAFE_INTEGER + inlineHeadBytes: 256 - Buffer.byteLength(suffix, 'utf8') }).head }${suffix}` } diff --git a/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts b/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts index c57c1796a35..e958dac42f8 100644 --- a/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts +++ b/src/main/crash-reporting/self-initiated-tree-kill-log.test.ts @@ -17,17 +17,17 @@ import { recordProcessGoneCrash, type ProcessGoneCrashEvent } from './process-go import { resetProcessGoneSiblingCorrelationForTest } from './process-gone-sibling-correlation' import { findSelfInitiatedTreeKills, - installProcessTreeKillBreadcrumbObserver, recordRefusedOwnChromiumTreeKill, recordSelfInitiatedTreeKill, resetSelfInitiatedTreeKillLogForTest, selfInitiatedTreeKillDetails } from './self-initiated-tree-kill-log' import { - notifyProcessTreeKill, - setProcessTreeKillObserver -} from '../../shared/child-process/process-tree-kill-observer' + admitProcessTreeKill, + setProcessTreeKillGate +} from '../../shared/child-process/process-tree-kill-gate' import { terminateWindowsProcessTree } from '../windows-process-tree-kill' +import { installMainProcessTreeKillGate } from '../own-chromium-tree-kill-guard' import { _resetTracerForTests, setActiveSink } from '../observability/tracer' /** The field shape: renderer, `reason=killed exitCode=1`, win32 (#G2). */ @@ -252,12 +252,90 @@ describe('self-initiated tree kill breadcrumb', () => { expect(String(details.selfInitiatedKills)).toContain('more)') }) - it('records a kill issued through the shared runProcess choke point', () => { - installProcessTreeKillBreadcrumbObserver() + it('keeps the pid-addressed kill when a window-close burst overruns the ring', () => { + // Review probe: one taskkill, then 32 routine Job Object teardowns. Under + // plain FIFO the discriminating entry is evicted and the persisted detail + // becomes byte-identical to the external-kill arm. + const goneAt = 5_000_000 + recordSelfInitiatedTreeKill({ + pid: 4242, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree', + at: goneAt - 4_000 + }) + for (let index = 0; index < 32; index += 1) { + recordSelfInitiatedTreeKill({ + pid: 6000 + index, + site: 'windows-pty-job-teardown', + scope: 'win-pty-job', + at: goneAt - 100 + }) + } - notifyProcessTreeKill({ pid: 3131, site: 'run-process-tree', scope: 'posix-process-group' }) + const details = selfInitiatedTreeKillDetails(goneAt) + + expect(details.selfInitiatedTreeKillCount).toBe(1) + expect(details.selfInitiatedGroupKillCount).toBe(31) + expect(String(details.selfInitiatedKills)).toMatch( + /^win-taskkill-tree\/pty-descendant-sweep\/pid4242 -4000ms/ + ) + }) + + it('keeps the newest teardown when a session has saturated the ring with pid kills', () => { + // Review probe, the mirror of the case above: 32 session-old taskkills (six + // routine families feed them) then the Job Object teardown 50ms before the + // death. A scope-preference eviction with no floor splices the entry it just + // pushed, and `{}` is byte-identical to the external-kill arm. + const goneAt = 5_000_000 + for (let index = 0; index < 32; index += 1) { + recordSelfInitiatedTreeKill({ + pid: 6000 + index, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree', + at: goneAt - 600_000 + index * 1_000 + }) + } + recordSelfInitiatedTreeKill({ + pid: 7777, + site: 'windows-pty-job-teardown', + scope: 'win-pty-job', + at: goneAt - 50 + }) + + const details = selfInitiatedTreeKillDetails(goneAt) + + expect(details.selfInitiatedGroupKillCount).toBe(1) + expect(String(details.selfInitiatedKills)).toContain( + 'win-pty-job/windows-pty-job-teardown/pid7777 -50ms' + ) + }) + + it('evicts the oldest pid kill, not the newest, once every candidate is pid-addressed', () => { + const goneAt = 5_000_000 + for (let index = 0; index < 33; index += 1) { + recordSelfInitiatedTreeKill({ + pid: 6000 + index, + site: 'git-command-tree-kill', + scope: 'win-taskkill-tree', + at: goneAt - 1_000 + }) + } + + const pids = findSelfInitiatedTreeKills(goneAt).map((kill) => kill.pid) + + expect(pids).toHaveLength(32) + expect(pids).toContain(6032) + expect(pids).not.toContain(6000) + }) + + it('records a kill issued through the shared runProcess choke point', () => { + installMainProcessTreeKillGate() + + expect( + admitProcessTreeKill({ pid: 3131, site: 'run-process-tree', scope: 'posix-process-group' }) + ).toBe(true) expect(findSelfInitiatedTreeKills(Date.now()).map((kill) => kill.pid)).toEqual([3131]) - setProcessTreeKillObserver(null) + setProcessTreeKillGate(null) }) }) diff --git a/src/main/crash-reporting/self-initiated-tree-kill-log.ts b/src/main/crash-reporting/self-initiated-tree-kill-log.ts index e819795074f..809d421ac61 100644 --- a/src/main/crash-reporting/self-initiated-tree-kill-log.ts +++ b/src/main/crash-reporting/self-initiated-tree-kill-log.ts @@ -1,8 +1,5 @@ import type { CrashReportDetailValue } from '../../shared/crash-reporting' -import { - setProcessTreeKillObserver, - type ProcessTreeKillScope -} from '../../shared/child-process/process-tree-kill-observer' +import type { ProcessTreeKillScope } from '../../shared/child-process/process-tree-kill-gate' import { recordCoalescedDurableCrashBreadcrumb } from './durable-crash-breadcrumb' /** @@ -19,24 +16,38 @@ import { recordCoalescedDurableCrashBreadcrumb } from './durable-crash-breadcrum * The ring is per-process and its only reader is `process-gone-recorder`, which * exists in Electron main. So a count reported on a `render-process-gone` covers * kills issued *from Electron main*, and nothing else: - * - Main only: the three `taskkill /T /F` families that gate on - * `admitSelfInitiatedTreeKill` (`terminateWindowsProcessTree` and the codex / - * claude account-login teardowns) and the codex app-server POSIX group + * - Main only: the families that import the gate directly — + * `terminateWindowsProcessTree`, the codex and claude account-login + * teardowns, the git command-runner abort, the notebook-cell and + * automation-precheck timeouts — plus the codex app-server POSIX group * teardowns. - * - Main *and* other hosts: `signalProcessTree` (the `runProcess` choke point, - * reached from the CLI, relay and daemon too — a fourth pid-addressed - * `taskkill` family, gated on the child not being reaped rather than on the - * Chromium set it cannot read), the POSIX PTY process-group sweep and the - * Windows PTY Job Object (relay `pty-handler`, daemon - * `subprocess-handle`). When those run outside main they record into that - * process's own ring, which nothing reads — no observer is installed there, - * and the tracer sink is a no-op. - * - Never instrumented: the direct `process.kill(-pid)` calls in the browser - * routes, notebooks, automation prechecks and ephemeral-VM recipes. + * - Main *and* other hosts, through the `process-tree-kill-gate` seam main + * installs the same guard into: `signalProcessTree` (the `runProcess` choke + * point, reached from the CLI, relay and daemon too), the codex app-server + * deadline kill (compiled into the CLI as well) and the ephemeral-VM recipe + * kill. Also host-spanning but recording directly: the POSIX PTY + * process-group sweep and the Windows PTY Job Object (relay `pty-handler`, + * daemon `subprocess-handle`). When any of these run outside main they record + * into that process's own ring, which nothing reads — no gate is installed + * there, and the tracer sink is a no-op. + * - Never instrumented, and none of them a pid-addressed kill issued from main: + * the POSIX `process.kill(-pid, …)` group arms of the notebook, precheck, + * browser-route and ephemeral-VM kills, plus the macOS keyboard-input-source + * probe's group kill in `ipc/app.ts`; the relay's own + * `subprocess-tree-termination` taskkill and the CLI's login-interruption + * taskkill (neither runs in main); and the browser-route Electron probes, + * which are reached only from `*.electron.test.ts`. + * + * `main-process-tree-kill-gate.test.ts` is the ratchet that keeps that list + * closed: it counts `/pid` call sites against gate admissions per file, so a new + * pid-addressed kill fails it whether it lands in a new file or inside a family + * that already asks the gate. It does not see a `/pid` argument built from a + * variable. * * A daemon or relay kill missing from the count is a diagnostics gap, not a * missed suspect: those hosts cannot reach a Chromium pid in the first place - * (see `orca-chromium-process-pids.ts`). Absence is evidence, not proof. + * (see `orca-chromium-process-pids.ts`), and a group or Job-Object kill can + * only contain what Orca put in it. Absence is evidence, not proof. */ /** Which mechanism issued the kill; each has a different blast radius. */ @@ -81,6 +92,25 @@ function isPidAddressedTreeKill(scope: SelfInitiatedTreeKillScope): boolean { return scope === 'win-taskkill-tree' } +/** + * Drop one entry, newest-first-preserving. + * + * Two rules, in order. The entry just recorded is never a candidate: it is the + * one closest to any death that follows, and evicting it leaves a detail + * byte-identical to the external-kill arm. Among the rest, routine group/job + * teardown goes before a pid-addressed kill — a window-close burst is 30+ group + * kills and plain FIFO would drop the one entry that can explain the death — + * falling back to plain FIFO once every candidate is pid-addressed, which is + * what an ordinary session saturates the ring with. + */ +function evictOneSelfInitiatedTreeKill(): void { + const lastCandidate = selfInitiatedKills.length - 1 + const oldestGroupKill = selfInitiatedKills.findIndex( + (kill, index) => index < lastCandidate && !isPidAddressedTreeKill(kill.scope) + ) + selfInitiatedKills.splice(Math.max(oldestGroupKill, 0), 1) +} + export function recordSelfInitiatedTreeKill({ pid, site, @@ -96,13 +126,15 @@ export function recordSelfInitiatedTreeKill({ return } selfInitiatedKills.push({ pid, site, scope, at }) - if (selfInitiatedKills.length > MAX_TRACKED_SELF_KILLS) { - selfInitiatedKills = selfInitiatedKills.slice(-MAX_TRACKED_SELF_KILLS) + while (selfInitiatedKills.length > MAX_TRACKED_SELF_KILLS) { + evictOneSelfInitiatedTreeKill() } // Durable so it survives into the diagnostic bundle even when the kill takes // the reporting renderer with it; coalesced because the crash detail above is // the primary record and a teardown burst must not cost 30 ring slots plus a - // forced disk flush each. The newest pid still rides the emitted crumb. + // forced disk flush each. The retained ring crumb carries the newest pid, but + // the span trail emits only the first of a coalesced burst — read + // `selfInitiatedKills` for the rest. recordCoalescedDurableCrashBreadcrumb({ name: 'self_tree_kill', data: { pid, site, scope }, @@ -133,11 +165,6 @@ export function recordRefusedOwnChromiumTreeKill(target: { }) } -/** Routes the `runProcess` choke point's kills here; shared code cannot import us. */ -export function installProcessTreeKillBreadcrumbObserver(): void { - setProcessTreeKillObserver((kill) => recordSelfInitiatedTreeKill(kill)) -} - export function findSelfInitiatedTreeKills(at: number): SelfInitiatedTreeKill[] { return selfInitiatedKills.filter((kill) => { const offsetMs = kill.at - at diff --git a/src/main/daemon/agent-startup-prompt-latency.node-pty.test.ts b/src/main/daemon/agent-startup-prompt-latency.node-pty.test.ts new file mode 100644 index 00000000000..cfa01fed8e5 --- /dev/null +++ b/src/main/daemon/agent-startup-prompt-latency.node-pty.test.ts @@ -0,0 +1,123 @@ +import { appendFileSync, existsSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { createPtySubprocess } from './pty-subprocess' +import { Session } from './session' + +const SHELLS = process.platform === 'win32' ? [] : ['/bin/bash', '/bin/zsh'].filter(existsSync) +const COMMAND = "printf 'AGENT_%s\\n' STARTED" + +async function launch( + shell: string, + slow: boolean, + legacy = false +): Promise<{ output: string; ms: number }> { + const root = mkdtempSync(join(tmpdir(), 'orca-startup-latency-')) + const bash = shell.endsWith('bash') + const pause = slow ? 'sleep 0.6\n' : '' + const prompt = slow ? "PS1='$(sleep 0.3)prompt> '\n" : "PS1='prompt> '\n" + writeFileSync( + join(root, bash ? '.bash_profile' : '.zshrc'), + `${pause}${bash ? '' : 'setopt PROMPT_SUBST\n'}${prompt}` + ) + vi.stubEnv('HOME', root) + vi.stubEnv('ZDOTDIR', root) + vi.stubEnv('ORCA_ORIG_ZDOTDIR', root) + let session: Session | undefined + let timer: ReturnType | undefined + let legacyTimer: ReturnType | undefined + const readinessEvents: string[] = [] + const started = performance.now() + try { + const subprocess = await createPtySubprocess({ + sessionId: 'startup-latency', + cols: 120, + rows: 30, + cwd: root, + shellOverride: shell, + command: COMMAND, + env: { HOME: root, SHELL: shell, TERM: 'xterm-256color' } + }) + session = new Session({ + sessionId: 'startup-latency', + cols: 120, + rows: 30, + subprocess, + shellReadySupported: !legacy, + reportReadinessEvent: (event) => readinessEvents.push(event) + }) + const active = session + return await new Promise((resolve, reject) => { + let output = '' + timer = setTimeout( + () => reject(new Error(`Startup timed out: ${JSON.stringify(output)}`)), + 5000 + ) + active.attachClient({ + onExit: () => {}, + onData: (data) => { + output += data + if (output.includes('AGENT_STARTED')) { + resolve({ output, ms: performance.now() - started }) + } + } + }) + if (legacy) { + legacyTimer = setTimeout(() => active.write(`${COMMAND}\n`), 300) + } else { + active.write(`${COMMAND}\n`) + } + }) + } finally { + clearTimeout(timer) + clearTimeout(legacyTimer) + if (session) { + await session.forceKillAndWaitForExit(3000) + session.dispose() + } + vi.unstubAllEnvs() + rmSync(root, { recursive: true, force: true }) + expect(readinessEvents).toEqual([]) + } +} + +describe('agent startup at the rendered shell prompt', () => { + afterEach(() => vi.unstubAllEnvs()) + it.each(SHELLS)( + '%s displays the command once after slow startup and prompt expansion', + async (shell) => { + const before = await launch(shell, true, true) + expect(before.output.split(COMMAND)).toHaveLength(3) + const result = await launch(shell, true) + expect(result.output).not.toContain('orca-shell-ready') + expect(result.output.split(COMMAND)).toHaveLength(2) + expect(result.output.indexOf('prompt> ')).toBeLessThan(result.output.indexOf(COMMAND)) + } + ) + + it.skipIf(!process.env.ORCA_STARTUP_BENCH || SHELLS.length === 0)( + 'compares legacy input timing with prompt delivery', + async () => { + for (const shell of SHELLS) { + for (const slow of [false, true]) { + const legacy: number[] = [] + const current: number[] = [] + for (let i = 0; i < 5; i++) { + legacy.push((await launch(shell, slow, true)).ms) + const result = await launch(shell, slow) + expect(result.output).not.toContain('orca-shell-ready') + expect(result.output.split(COMMAND)).toHaveLength(2) + current.push(result.ms) + } + const result = JSON.stringify({ shell, slow, legacy, current }) + if (process.env.ORCA_STARTUP_BENCH_OUTPUT) { + appendFileSync(process.env.ORCA_STARTUP_BENCH_OUTPUT, `${result}\n`) + } + console.log(result) + } + } + }, + 60_000 + ) +}) diff --git a/src/main/daemon/daemon-bash-shell-ready-rcfile.ts b/src/main/daemon/daemon-bash-shell-ready-rcfile.ts index 142c0c131d9..f7834dba9d7 100644 --- a/src/main/daemon/daemon-bash-shell-ready-rcfile.ts +++ b/src/main/daemon/daemon-bash-shell-ready-rcfile.ts @@ -2,7 +2,6 @@ import { getPosixOmpShellWrapper } from '../pty/omp-shell-wrapper' import { getPosixCodexShellLaunchPreflight } from '../pty/codex-shell-launch-preflight' import { BASH_PROMPT_COMMAND_COMPOSITION_BLOCK } from '../bash-prompt-command-composition' import { BASH_FEATURE_CHANNEL_BLOCK, SHELL_STARTUP_IDENTITY_MARKER_BLOCK } from '../shell-templates' -import { SHELL_READY_MARKER } from './daemon-shell-ready-marker' export function getDaemonBashShellReadyRcfileContent(): string { return `# Orca daemon bash shell-ready wrapper @@ -56,10 +55,6 @@ __orca_osc133_precmd() { unset __orca_in_command fi printf "\\033]133;A\\007" - # Why: emit the shell-ready marker here (not a trailing PROMPT_COMMAND entry) - # so a framework that must be last in PROMPT_COMMAND — bash-preexec — is not - # displaced by one of Orca's own hooks. - [[ -n "$__orca_ready_marker" ]] && printf "${SHELL_READY_MARKER}" return "$exit_code" } __orca_osc133_preexec() { @@ -117,6 +112,11 @@ __orca_osc133_epilogue() { unset __orca_in_prompt_command __orca_adopt_outer_debug_trap trap '__orca_osc133_preexec' DEBUG + # Readline renders PS1 after entering raw mode; prompt hooks still run in cooked mode. + if [[ -n "$__orca_ready_marker" ]]; then + PS1="\${PS1-}"'\\[\\e]777;orca-shell-ready\\a\\]' + __orca_ready_marker="" + fi } ${BASH_PROMPT_COMMAND_COMPOSITION_BLOCK} __orca_prepend_prompt_command "__orca_osc133_precmd" diff --git a/src/main/daemon/daemon-pty-adapter.test.ts b/src/main/daemon/daemon-pty-adapter.test.ts index fc2f9365f23..d0c0198af65 100644 --- a/src/main/daemon/daemon-pty-adapter.test.ts +++ b/src/main/daemon/daemon-pty-adapter.test.ts @@ -344,35 +344,32 @@ describe('DaemonPtyAdapter (IPtyProvider)', () => { } }) - itOnPosix('keeps plain Codex startup on the short daemon shell-ready timeout', async () => { - await adapter.spawn({ - cols: 80, - rows: 24, - command: 'codex', - env: { SHELL: '/bin/zsh' } - }) - + itOnPosix('preserves the existing fast-start timing for fish', async () => { + await adapter.spawn({ cols: 80, rows: 24, command: 'codex', env: { SHELL: '/usr/bin/fish' } }) await waitFor(() => vi.mocked(lastSubprocess.write).mock.calls.length > 0) - expect(lastSubprocess.write).toHaveBeenCalledWith('codex\n') + expect(lastSubprocess.write).toHaveBeenCalledExactlyOnceWith('codex\n') + expect(lastSpawnOpts).not.toEqual( + expect.objectContaining({ startupCommandDelivery: 'shell-ready' }) + ) }) - itOnPosix('waits for shell-ready for delivery-hinted Codex startup', async () => { - await adapter.spawn({ - cols: 80, - rows: 24, - command: "codex 'linked issue context'", - startupCommandDelivery: 'shell-ready', - env: { SHELL: '/bin/zsh' } - }) + itOnPosix.each([ + { command: 'codex' }, + { command: 'codex', startupCommandDelivery: 'fast' as const }, + { command: "codex 'linked issue context'", startupCommandDelivery: 'shell-ready' as const } + ])('waits past 300ms and submits once after readiness: %j', async (startup) => { + await adapter.spawn({ cols: 80, rows: 24, ...startup, env: { SHELL: '/bin/zsh' } }) await new Promise((resolve) => setTimeout(resolve, 350)) expect(lastSubprocess.write).not.toHaveBeenCalled() - + expect(lastSpawnOpts).toEqual( + expect.objectContaining({ startupCommandDelivery: 'shell-ready' }) + ) lastSubprocess._simulateData('\x1b]777;orca-shell-ready\x07') lastSubprocess._simulateData('\r\nuser@host $ ') await waitFor(() => vi.mocked(lastSubprocess.write).mock.calls.length > 0) - expect(lastSubprocess.write).toHaveBeenCalledWith("codex 'linked issue context'\n") + expect(lastSubprocess.write).toHaveBeenCalledExactlyOnceWith(`${startup.command}\n`) }) }) diff --git a/src/main/daemon/daemon-pty-session-spawn.ts b/src/main/daemon/daemon-pty-session-spawn.ts index 235b286b0bc..00f19777878 100644 --- a/src/main/daemon/daemon-pty-session-spawn.ts +++ b/src/main/daemon/daemon-pty-session-spawn.ts @@ -1,5 +1,6 @@ import { recognizeAgentProcessFromCommandLine } from '../../shared/agent-process-recognition' import { shouldUseShellReadyStartupDelivery } from '../../shared/codex-startup-delivery' +import { CODEX_SHELL_READY_TIMEOUT_MS } from './session-shell-ready-barrier' import type { HistoryRecoveryContext, PendingDaemonSpawnOperation @@ -11,8 +12,11 @@ import { DaemonPtySpawnResult } from './daemon-pty-spawn-result' import type { DaemonPtySpawnContext } from './daemon-pty-spawn-request' import type { ColdRestoreInfo } from './history-reader' import { mintPtySessionId } from './pty-session-id' -import { CODEX_SHELL_READY_TIMEOUT_MS } from './session-shell-ready-barrier' -import { supportsPtyStartupBarrier } from './shell-ready' +import { + supportsPtyStartupBarrier, + shellReadyMarkerComesFromLineEditor, + resolvePtyShellPath +} from './shell-ready' import { getRecoveredHistorySeedSegments } from './terminal-history-seed-segments' import { AGENT_SESSION_CLAIM_DAEMON_PROTOCOL_VERSION, type CreateOrAttachResult } from './types' import { normalizeWslColdRestoreCwd } from './wsl-cold-restore-cwd' @@ -213,21 +217,23 @@ export abstract class DaemonPtySessionSpawn extends DaemonPtySpawnResult { let effectiveRows = restoreInfo?.rows ?? opts.rows const shellReadySupported = opts.command ? supportsPtyStartupBarrier(opts.env ?? {}) : false - const isCodexStartupCommand = - recognizeAgentProcessFromCommandLine(opts.command)?.agent === 'codex' - const shouldWaitForShellReady = - isCodexStartupCommand && - shouldUseShellReadyStartupDelivery({ + const immediateMarker = + process.platform !== 'win32' && + shellReadyMarkerComesFromLineEditor(opts.shellOverride || resolvePtyShellPath(opts.env ?? {})) + const shellReadyTimeoutMs = + shellReadySupported && + !immediateMarker && + recognizeAgentProcessFromCommandLine(opts.command)?.agent === 'codex' && + !shouldUseShellReadyStartupDelivery({ command: opts.command, startupCommandDelivery: opts.startupCommandDelivery }) - const shellReadyTimeoutMs = - shellReadySupported && isCodexStartupCommand && !shouldWaitForShellReady ? CODEX_SHELL_READY_TIMEOUT_MS : undefined - const context: DaemonPtySpawnContext = { - opts, + // Older daemons also need the existing hint to enable their ready marker. + opts: + opts.command && immediateMarker ? { ...opts, startupCommandDelivery: 'shell-ready' } : opts, operation, historyRecovery, requestedSessionId, diff --git a/src/main/daemon/post-ready-flush-gate.test.ts b/src/main/daemon/post-ready-flush-gate.test.ts index f8e58ba7c7d..06ff64db99f 100644 --- a/src/main/daemon/post-ready-flush-gate.test.ts +++ b/src/main/daemon/post-ready-flush-gate.test.ts @@ -20,6 +20,14 @@ describe('PostReadyFlushGate', () => { vi.useRealTimers() }) + it('flushes synchronously when the marker comes from the line editor', () => { + gate = new PostReadyFlushGate(onFlush, true) + gate.arm() + expect(onFlush).toHaveBeenCalledTimes(1) + expect(gate.isPending).toBe(false) + expect(vi.getTimerCount()).toBe(0) + }) + it('does not flush immediately when armed', () => { gate.arm() expect(onFlush).not.toHaveBeenCalled() diff --git a/src/main/daemon/post-ready-flush-gate.ts b/src/main/daemon/post-ready-flush-gate.ts index 199bb601024..f7f7a2f1869 100644 --- a/src/main/daemon/post-ready-flush-gate.ts +++ b/src/main/daemon/post-ready-flush-gate.ts @@ -1,23 +1,5 @@ -/** - * Defers a flush callback until after the shell has drawn its prompt and - * switched the PTY into raw mode. - * - * Why: the OSC 777 shell-ready marker fires from zsh's precmd_functions / - * bash's PROMPT_COMMAND — before the shell draws its prompt and before - * zle/readline flips the PTY into raw mode. Flushing queued input then lets - * the kernel (ECHO still on) echo the command once, and the line editor - * redraws it under the prompt — producing a visible duplicate (e.g. "claude" - * appears twice on agent launch). - * - * Strategy: after arm() is called, wait for prompt bytes plus a short delay - * for the tcsetattr() that enables raw mode. If the marker-completing scan - * already saw post-marker bytes, use that same short path immediately. - * A conservative wall-clock fallback covers ambiguous marker-only cases. - * - * Mirrors the gate in local-pty-shell-ready.ts::writeStartupCommandWhenShellReady, - * which solves the same race on the non-daemon path. - */ - +// Bash's prompt and zsh's line-init marker are ready for input immediately. +// Other shells retain the existing settling delay. export const POST_READY_FLUSH_DELAY_MS = 30 export const POST_READY_FLUSH_FALLBACK_MS = 200 @@ -26,7 +8,10 @@ export class PostReadyFlushGate { private postDataTimer: ReturnType | null = null private fallbackTimer: ReturnType | null = null - constructor(private readonly onFlush: () => void) {} + constructor( + private readonly onFlush: () => void, + private readonly markerIsLineEditorReady = false + ) {} /** True between arm() and the actual flush firing. Callers should treat * input as still-queued during this window to preserve ordering. */ @@ -38,6 +23,10 @@ export class PostReadyFlushGate { * wall-clock fallback unless the marker scan already observed post-marker * bytes, in which case the short post-data settle path is enough. */ arm(postMarkerBytesObserved = false): void { + if (this.markerIsLineEditorReady) { + this.onFlush() + return + } this.awaitingPromptDraw = true if (postMarkerBytesObserved) { this.notifyData() diff --git a/src/main/daemon/pty-subprocess-managed-agent-env.test.ts b/src/main/daemon/pty-subprocess-managed-agent-env.test.ts index 5267778088c..d5dbc33246c 100644 --- a/src/main/daemon/pty-subprocess-managed-agent-env.test.ts +++ b/src/main/daemon/pty-subprocess-managed-agent-env.test.ts @@ -261,7 +261,7 @@ describe('createPtySubprocess', () => { expect(lastCall[2].env.ORCA_SHELL_FEATURES).not.toContain('ready') }) - it('keeps plain Codex startup commands on the no-marker wrapper', async () => { + it('enables readiness and shell identity for plain Codex startup', async () => { const proc = mockPtyProcess() spawnMock.mockReturnValue(proc) const platform = Object.getOwnPropertyDescriptor(process, 'platform') @@ -285,7 +285,8 @@ describe('createPtySubprocess', () => { const lastCall = spawnMock.mock.calls.at(-1)! expect(lastCall[1]).toEqual(['-l']) expect(lastCall[2].env.ZDOTDIR).toMatch(ZSH_SHELL_READY_DIR) - expect(lastCall[2].env.ORCA_SHELL_FEATURES).not.toContain('ready') + expect(lastCall[2].env.ORCA_SHELL_FEATURES).toContain('ready') + expect(lastCall[2].env.ORCA_SHELL_FEATURES).toContain('identity') }) it('uses shell-ready wrapper for delivery-hinted Codex startup commands', async () => { diff --git a/src/main/daemon/pty-subprocess/shell-launch-plan.ts b/src/main/daemon/pty-subprocess/shell-launch-plan.ts index ed8cd5e6a22..12e049d9f72 100644 --- a/src/main/daemon/pty-subprocess/shell-launch-plan.ts +++ b/src/main/daemon/pty-subprocess/shell-launch-plan.ts @@ -1,3 +1,4 @@ +import { shouldUseShellReadyStartupDelivery } from '../../../shared/codex-startup-delivery' import { win32 as pathWin32 } from 'node:path' import { isWindowsGitBashShellPath, resolveWindowsGitBashShellPath } from '../../git-bash' import { isPwshAvailable } from '../../pwsh' @@ -32,10 +33,13 @@ import { recognizeAgentProcessFromCommandLine, type RecognizedAgentProcess } from '../../../shared/agent-process-recognition' -import { shouldUseShellReadyStartupDelivery } from '../../../shared/codex-startup-delivery' import { ORCA_HERMES_STARTUP_QUERY_ENV } from '../../../shared/hermes-startup-query' import { WINDOWS_GIT_BASH_SHELL } from '../../../shared/windows-terminal-shell' -import { getShellLaunchConfig, resolvePtyShellPath } from '../shell-ready' +import { + getShellLaunchConfig, + resolvePtyShellPath, + shellReadyMarkerComesFromLineEditor +} from '../shell-ready' import { resolveWslSessionContext } from '../wsl-session-context' import { finalizeDaemonPtyEnvironment, rescrubDaemonPtyEnvironment } from './spawn-environment' import type { PtySubprocessOptions } from '../pty-subprocess' @@ -60,7 +64,6 @@ export function createPtyShellLaunchPlan( let startupCommandDeliveredInShellArgs = false let windowsFallbackAttempts: WindowsShellSpawnAttempt[] = [] const startupAgentRecognition = recognizeAgentProcessFromCommandLine(opts.command) - const isCodexStartupCommand = startupAgentRecognition?.agent === 'codex' const requestedCwd = opts.cwd || resolveSafePtyDefaultCwd() if (opts.command && startupAgentRecognition) { assertSafeAgentStartupCwd(requestedCwd, opts.command) @@ -192,9 +195,10 @@ export function createPtyShellLaunchPlan( } const waitsForShellReady = Boolean(opts.command) && - (!isCodexStartupCommand || + (startupAgentRecognition?.agent !== 'codex' || + shellReadyMarkerComesFromLineEditor(shellPath) || shouldUseShellReadyStartupDelivery({ - command: opts.command as string, + command: opts.command, startupCommandDelivery: opts.startupCommandDelivery })) delete env.ORCA_SHELL_FEATURES diff --git a/src/main/daemon/session-shell-ready-barrier.ts b/src/main/daemon/session-shell-ready-barrier.ts index e7562e23224..538fd57a49a 100644 --- a/src/main/daemon/session-shell-ready-barrier.ts +++ b/src/main/daemon/session-shell-ready-barrier.ts @@ -1,3 +1,4 @@ +import { shellReadyMarkerComesFromLineEditor } from './shell-ready' import { installDeviceAttributesResponder, STARTUP_DA1_RESPONSE @@ -19,7 +20,6 @@ import { basename } from 'node:path' import type { ShellReadyState } from './types' const SHELL_READY_TIMEOUT_MS = 15_000 -// Why: Codex skips marker-gated command delivery; this only bounds older daemon/local paths that still report shell-ready for Codex. export const CODEX_SHELL_READY_TIMEOUT_MS = 300 export type SessionShellReadyBarrierDeps = { @@ -69,7 +69,10 @@ export class SessionShellReadyBarrier { this._state = 'unsupported' } - this.postReadyFlushGate = new PostReadyFlushGate(() => this.flushPreReadyQueue()) + this.postReadyFlushGate = new PostReadyFlushGate( + () => this.flushPreReadyQueue(), + shellReadyMarkerComesFromLineEditor(deps.subprocess.shellPath ?? '') + ) } get state(): ShellReadyState { diff --git a/src/main/daemon/shell-ready.ts b/src/main/daemon/shell-ready.ts index 9dc60b7c980..d51f6efb80a 100644 --- a/src/main/daemon/shell-ready.ts +++ b/src/main/daemon/shell-ready.ts @@ -101,6 +101,11 @@ export function resolvePtyShellPath(env: Record): string { return env.SHELL || process.env.SHELL || '/bin/zsh' } +export function shellReadyMarkerComesFromLineEditor(shellPath: string): boolean { + const shellName = pathWin32.basename(basename(shellPath)).toLowerCase() + return shellName === 'bash' || shellName === 'zsh' +} + export function shellPathSupportsPtyStartupBarrier(shellPath: string): boolean { const shellName = pathWin32.basename(basename(shellPath)).toLowerCase() // Why fish: markerless, its startup command is written before fish's reader owns diff --git a/src/main/git/command-runner/git-exec-file.ts b/src/main/git/command-runner/git-exec-file.ts index dbd861d4897..acfeb9f8a3e 100644 --- a/src/main/git/command-runner/git-exec-file.ts +++ b/src/main/git/command-runner/git-exec-file.ts @@ -10,6 +10,7 @@ import { prepareWslLinkedWorktreeGitRouting } from '../wsl-linked-worktree-git-routing' import { resolveCommand, type ResolvedCommand } from './wsl-command-resolution' +import { annotateWslHostFailure } from './wsl-host-failure' import type { GitAdmissionTier, GitExecOptions } from './git-exec-options' import { execFileCapture, execFileCaptureToTermination } from './exec-file-capture' import { @@ -96,7 +97,7 @@ async function gitExecFileAsyncUnlocked( ? {} : { createTimeoutError: () => new GitCommandTimeoutError(timeoutMs) }) } - return options.terminationBarrier + const captured = options.terminationBarrier ? execFileCaptureToTermination( command.binary, command.args, @@ -104,6 +105,10 @@ async function gitExecFileAsyncUnlocked( command.termination ) : execFileCapture(command.binary, command.args, captureOptions) + // Why: a dead WSL distro fails with an empty stderr, so the span would carry no cause at all. + return captured.catch((error: unknown) => { + throw annotateWslHostFailure(error, command) + }) } const runCapturedCommand = async (): Promise<{ stdout: string; stderr: string }> => { let result: { stdout: string | Buffer; stderr: string | Buffer } diff --git a/src/main/git/command-runner/git-process-env.ts b/src/main/git/command-runner/git-process-env.ts index 6154364e5ab..9064f5ed900 100644 --- a/src/main/git/command-runner/git-process-env.ts +++ b/src/main/git/command-runner/git-process-env.ts @@ -63,6 +63,11 @@ export function nonInteractiveGitEnv( platform: NodeJS.Platform = process.platform ): NodeJS.ProcessEnv { const next = promptGuardGitEnv(env, platform) + if (platform === 'win32') { + // Why: without it wsl.exe writes its OWN failures ("no distribution with the supplied name") as + // UTF-16LE (#9010), which is how a dead distro reached telemetry as an error with no text. + next.WSL_UTF8 = '1' + } if (!next.GIT_SSH_COMMAND) { next.GIT_SSH_COMMAND = 'ssh -o BatchMode=yes' if (platform === 'win32') { diff --git a/src/main/git/command-runner/spawned-command-tree-kill.ts b/src/main/git/command-runner/spawned-command-tree-kill.ts index c2fefcba1bc..324e04f1db3 100644 --- a/src/main/git/command-runner/spawned-command-tree-kill.ts +++ b/src/main/git/command-runner/spawned-command-tree-kill.ts @@ -1,4 +1,5 @@ import { spawn, type ChildProcess } from 'node:child_process' +import { admitSelfInitiatedTreeKill } from '../../own-chromium-tree-kill-guard' const WINDOWS_TREE_KILL_WAIT_MS = 2_000 @@ -8,6 +9,14 @@ export function killSpawnedCommandTree(child: ChildProcess): Promise { child.kill() return Promise.resolve() } + if ( + !admitSelfInitiatedTreeKill({ pid, site: 'git-command-tree-kill', scope: 'win-taskkill-tree' }) + ) { + // Refusal blocks the pid-addressed tree walk, never the termination: the + // handle-addressed root kill cannot reach a recycled pid. + child.kill() + return Promise.resolve() + } return new Promise((resolve) => { let killer: ChildProcess try { diff --git a/src/main/git/command-runner/wsl-host-failure.test.ts b/src/main/git/command-runner/wsl-host-failure.test.ts new file mode 100644 index 00000000000..4b495817f6f --- /dev/null +++ b/src/main/git/command-runner/wsl-host-failure.test.ts @@ -0,0 +1,143 @@ +import { EventEmitter } from 'node:events' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { execFileMock } = vi.hoisted(() => ({ execFileMock: vi.fn() })) + +vi.mock('node:child_process', () => ({ + execFile: execFileMock, + execFileSync: vi.fn(), + spawn: vi.fn() +})) +vi.mock('../../observability/instrumentation', () => ({ + withGitSpan: (_attributes: unknown, run: (span: unknown) => unknown) => + run({ setAttribute: () => {} }) +})) +vi.mock('../../diagnostics/main-thread-churn-probe', () => ({ recordSubprocessSpawn: vi.fn() })) + +import { gitExecFileAsync } from '../runner' +import { _resetGitAdmissionForTests } from './git-subprocess-admission' +import { nonInteractiveGitEnv } from './git-process-env' +import { annotateWslHostFailure, readWslHostFailureDiagnostic } from './wsl-host-failure' +import type { ResolvedCommand } from './wsl-command-resolution' + +const WSL_COMMAND: ResolvedCommand = { + binary: 'wsl.exe', + args: ['-d', 'kali-linux', '--exec', 'sh', '-lc', 'git worktree list'], + cwd: 'C:\\Users\\paulius', + wsl: { distro: 'kali-linux', linuxPath: '/home/paulius/bugbounty' }, + wslMode: 'login-shell' +} + +const WSL_DIAGNOSTIC = + 'There is no distribution with the supplied name.\r\nError code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\r\n' + +/** wsl.exe without WSL_UTF8 writes its own diagnostic as UTF-16LE, which reaches Node as NUL-riddled text. */ +function asUtf16Mojibake(text: string): string { + return [...text].map((character) => `${character}\u0000`).join('') +} + +function hostFailure(stdout: string): Error { + return Object.assign(new Error('Command failed: wsl.exe -d kali-linux --exec sh -lc ...'), { + code: 4294967295, + stdout, + stderr: '' + }) +} + +async function withPlatform(platform: NodeJS.Platform, run: () => Promise): Promise { + const original = process.platform + Object.defineProperty(process, 'platform', { configurable: true, value: platform }) + try { + return await run() + } finally { + Object.defineProperty(process, 'platform', { configurable: true, value: original }) + } +} + +describe('wsl.exe host failure classification', () => { + it('reads the diagnostic wsl.exe left on stdout, including UTF-16 output', () => { + expect(readWslHostFailureDiagnostic(hostFailure(WSL_DIAGNOSTIC), WSL_COMMAND)).toContain( + 'Wsl/Service/WSL_E_DISTRO_NOT_FOUND' + ) + expect( + readWslHostFailureDiagnostic(hostFailure(asUtf16Mojibake(WSL_DIAGNOSTIC)), WSL_COMMAND) + ).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + }) + + it('leaves a guest failure and a non-WSL command alone', () => { + const guestFailure = Object.assign(new Error('Command failed'), { + code: 1, + stdout: '', + stderr: 'fatal: not a git repository\n' + }) + expect(readWslHostFailureDiagnostic(guestFailure, WSL_COMMAND)).toBeNull() + // Same exit code, but wsl.exe was never involved. + expect( + readWslHostFailureDiagnostic(hostFailure(WSL_DIAGNOSTIC), { + binary: 'git', + args: ['status'], + cwd: '/repo', + wsl: null, + wslMode: null + }) + ).toBeNull() + }) + + it('moves the diagnostic into the message the span records', () => { + const error = annotateWslHostFailure(hostFailure(WSL_DIAGNOSTIC), WSL_COMMAND) as Error & { + wslHostFailure?: boolean + wslDistro?: string + code?: number + } + expect(error.message).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + expect(error.message).toContain('kali-linux') + expect(error.wslHostFailure).toBe(true) + expect(error.wslDistro).toBe('kali-linux') + // The original failure detail must survive for callers that classify on it. + expect(error.code).toBe(4294967295) + expect(error.message).toContain('Command failed: wsl.exe') + }) +}) + +describe('WSL-routed git subprocess', () => { + beforeEach(() => { + execFileMock.mockReset() + }) + + afterEach(() => { + _resetGitAdmissionForTests() + }) + + it('sets WSL_UTF8 so wsl.exe explains itself in UTF-8', () => { + expect(nonInteractiveGitEnv({}, 'win32').WSL_UTF8).toBe('1') + expect(nonInteractiveGitEnv({}, 'darwin').WSL_UTF8).toBeUndefined() + }) + + it('reports a dead distro instead of an empty git error', async () => { + execFileMock.mockImplementation((_command, _args, _options, callback) => { + const child = new EventEmitter() as EventEmitter & { pid: number; kill: () => void } + child.pid = 4321 + child.kill = () => {} + queueMicrotask(() => + callback?.( + hostFailure(asUtf16Mojibake(WSL_DIAGNOSTIC)), + asUtf16Mojibake(WSL_DIAGNOSTIC), + '' + ) + ) + return child + }) + + const failure = await withPlatform('win32', () => + gitExecFileAsync(['worktree', 'list', '--porcelain', '-z'], { + cwd: '\\\\wsl.localhost\\kali-linux\\home\\paulius\\bugbounty' + }).then( + () => null, + (error: unknown) => error as Error + ) + ) + + expect(failure?.message).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + expect(execFileMock.mock.calls.at(-1)?.[2]?.env?.WSL_UTF8).toBe('1') + }) +}) diff --git a/src/main/git/command-runner/wsl-host-failure.ts b/src/main/git/command-runner/wsl-host-failure.ts new file mode 100644 index 00000000000..74f98509d0b --- /dev/null +++ b/src/main/git/command-runner/wsl-host-failure.ts @@ -0,0 +1,55 @@ +import type { ResolvedCommand } from './wsl-command-resolution' + +/** wsl.exe's own launch-failure exit, distinct from any status the guest process can return. */ +export const WSL_HOST_FAILURE_EXIT_CODE = 0xffffffff + +function outputText(value: unknown): string { + if (typeof value === 'string') { + return value + } + return Buffer.isBuffer(value) ? value.toString('utf8') : '' +} + +/** + * The message wsl.exe prints when it — not the guest — failed: a distro that was renamed or + * removed, or a VM that would not start. + * + * Why this needs decoding at all: wsl.exe exits 0xFFFFFFFF, leaves stderr EMPTY, and writes + * `Error code: Wsl/Service/WSL_E_*` to stdout, so every caller that reports stderr reports nothing. + * NULs are stripped because a wsl.exe that ignores WSL_UTF8 writes that line as UTF-16LE (#9010). + */ +export function readWslHostFailureDiagnostic( + error: unknown, + command: ResolvedCommand +): string | null { + if (!command.wsl || !error || typeof error !== 'object') { + return null + } + const { code, status, stdout, stderr } = error as { + code?: unknown + status?: unknown + stdout?: unknown + stderr?: unknown + } + const exitCode = typeof code === 'number' ? code : typeof status === 'number' ? status : null + // A guest failure always explains itself on stderr; an empty one plus this exit is the host. + if (exitCode !== WSL_HOST_FAILURE_EXIT_CODE || outputText(stderr).trim().length > 0) { + return null + } + const diagnostic = outputText(stdout).replaceAll('\u0000', '').trim() + return diagnostic.length > 0 ? diagnostic : 'wsl.exe reported no diagnostic.' +} + +/** + * Move a wsl.exe host failure into the error's message, which is what `git.exec` spans record ahead + * of the stack. Left untouched when the failure came from git itself. + */ +export function annotateWslHostFailure(error: unknown, command: ResolvedCommand): unknown { + const diagnostic = readWslHostFailureDiagnostic(error, command) + if (diagnostic === null || !(error instanceof Error) || !command.wsl) { + return error + } + const distro = command.wsl.distro + error.message = `wsl.exe host failure (distro "${distro}"): ${diagnostic}\n${error.message}` + return Object.assign(error, { wslHostFailure: true, wslDistro: distro }) +} diff --git a/src/main/git/source-control/submodule-status.ts b/src/main/git/source-control/submodule-status.ts index f3c808c9f2f..fb4ca8449e3 100644 --- a/src/main/git/source-control/submodule-status.ts +++ b/src/main/git/source-control/submodule-status.ts @@ -26,17 +26,22 @@ export async function getSubmoduleStatus( const submoduleWorktreePath = resolveSubmoduleWorktreePath(worktreePath, submodulePath) const limit = resolveGitStatusLimit(options.limit) // Why: staged expansion only represents HEAD→index; scanning the submodule worktree is wasted work. - const workingResult = options.staged - ? ({ entries: [], conflictOperation: 'unknown' } satisfies GitStatusResult) - : await getStatus(submoduleWorktreePath, options) - // Why: a moved gitlink (clean worktree) has no status rows; surface the parent-commit→checkout range as inner rows. - const fromOid = options.staged - ? await readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options) - : (await readGitlinkOidFromIndex(worktreePath, submodulePath, options)) || - (await readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options)) - const toOid = options.staged - ? await readGitlinkOidFromIndex(worktreePath, submodulePath, options) - : await readWorkingSubmoduleHead(submoduleWorktreePath, options) + // These three reads are independent, so they run concurrently — on SSH/WSL each one is a real round trip. + const [workingResult, fromOid, toOid] = await Promise.all([ + options.staged + ? Promise.resolve({ entries: [], conflictOperation: 'unknown' }) + : getStatus(submoduleWorktreePath, options), + // Why: a moved gitlink (clean worktree) has no status rows; surface the parent-commit→checkout range as inner rows. + options.staged + ? readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options) + : readGitlinkOidFromIndex(worktreePath, submodulePath, options).then( + (indexOid) => + indexOid || readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options) + ), + options.staged + ? readGitlinkOidFromIndex(worktreePath, submodulePath, options) + : readWorkingSubmoduleHead(submoduleWorktreePath, options) + ]) if (fromOid && toOid && fromOid !== toOid) { const rangeEntries = await computeSubmoduleRangeEntries( submoduleWorktreePath, diff --git a/src/main/git/status-submodule.test.ts b/src/main/git/status-submodule.test.ts index b67a4eca9a0..6528d373a78 100644 --- a/src/main/git/status-submodule.test.ts +++ b/src/main/git/status-submodule.test.ts @@ -395,4 +395,35 @@ describe('getSubmoduleStatus', () => { expect(result.didHitLimit).toBe(true) expect(result.statusLength).toBe(2) }) + + it('overlaps the inner status, index and HEAD reads instead of serializing them', async () => { + const OLD_OID = 'a'.repeat(40) + const NEW_OID = 'b'.repeat(40) + readFileMock.mockResolvedValue('gitdir: /repo/flutter_mine/.git\n') + existsSyncMock.mockReturnValue(false) + const events: string[] = [] + gitExecFileAsyncMock.mockReset() + gitExecFileAsyncMock.mockImplementation(async (args: string[]) => { + const name = args.includes('status') ? 'status' : args[0] + events.push(`start:${name}`) + if (name === 'status') { + // Hold the inner status open so a serialized caller could not have started the oid reads. + await new Promise((resolve) => setTimeout(resolve, 5)) + } + events.push(`end:${name}`) + if (name === 'ls-files') { + return { stdout: `160000 ${OLD_OID} 0\tflutter_mine\n` } + } + if (name === 'rev-parse') { + return { stdout: `${NEW_OID}\n` } + } + return { stdout: '' } + }) + + await getSubmoduleStatus('/repo', 'flutter_mine') + + // Serialized reads only start ls-files/rev-parse once the inner status has resolved. + expect(events.indexOf('start:ls-files')).toBeLessThan(events.indexOf('end:status')) + expect(events.indexOf('start:rev-parse')).toBeLessThan(events.indexOf('end:status')) + }) }) diff --git a/src/main/git/worktree-listing.ts b/src/main/git/worktree-listing.ts index a60819181d4..f027bac4bc1 100644 --- a/src/main/git/worktree-listing.ts +++ b/src/main/git/worktree-listing.ts @@ -35,17 +35,7 @@ export async function listWorktreeGraph( ? worktrees : worktrees.filter((worktree) => !isWorktreeCreatePreparation(worktree)) } catch (err) { - if (getErrorCode(err) === 'ENOENT') { - try { - await stat(repoPath) - } catch (statErr) { - if (getErrorCode(statErr) === 'ENOENT') { - console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) - return [] - } - } - } - if (isNotGitRepositoryError(err)) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { return [] } console.warn(`[git/worktree] listWorktreeGraph failed for ${repoPath}:`, err) @@ -64,17 +54,7 @@ export async function listWorktreesUnshared( : worktrees.filter((worktree) => !isWorktreeCreatePreparation(worktree)) return annotateSparseCheckoutStatus(repoPath, visibleWorktrees, options) } catch (err) { - if (getErrorCode(err) === 'ENOENT') { - try { - await stat(repoPath) - } catch (statErr) { - if (getErrorCode(statErr) === 'ENOENT') { - console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) - return [] - } - } - } - if (isNotGitRepositoryError(err)) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { return [] } // Why: don't swallow git-compat/repo-state failures — else they resurface as opaque "created but not found in listing" errors. @@ -83,6 +63,24 @@ export async function listWorktreesUnshared( } } +/** + * The two failures where an empty listing is the repo's true answer, not a broken scan: the repo + * path is gone, or it is not a Git repo. Every other failure means the scan could not read Git. + */ +async function isTrueEmptyWorktreeListing(repoPath: string, err: unknown): Promise { + if (getErrorCode(err) === 'ENOENT') { + try { + await stat(repoPath) + } catch (statErr) { + if (getErrorCode(statErr) === 'ENOENT') { + console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) + return true + } + } + } + return isNotGitRepositoryError(err) +} + export async function listWorktreesStrict( repoPath: string, options: GitWorktreeExecOptions = {} @@ -97,6 +95,28 @@ export async function listWorktreesStrict( return annotateSparseCheckoutStatus(repoPath, visibleWorktrees, options) } +/** + * Strict except for the two true empties above. + * + * Why: a Git or host failure (dead WSL distro, hung mount) softened to `[]` reaches the detected + * listing as an *authoritative* empty scan, which then permanently prunes the repo's worktrees and + * the agent tabs attached to them. Rejecting keeps that listing non-authoritative, while a deleted + * repo still reports empty so real removals prune. + */ +export async function listWorktreesStrictAllowingTrueEmpty( + repoPath: string, + options: GitWorktreeExecOptions = {} +): Promise { + try { + return await listWorktreesStrict(repoPath, options) + } catch (err) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { + return [] + } + throw err + } +} + export async function annotateSparseCheckoutStatus( repoPath: string, worktrees: GitWorktreeInfo[], diff --git a/src/main/git/worktree-scan-cache.ts b/src/main/git/worktree-scan-cache.ts index 59a7691fe7f..a35f0a4f275 100644 --- a/src/main/git/worktree-scan-cache.ts +++ b/src/main/git/worktree-scan-cache.ts @@ -2,7 +2,8 @@ import type { GitWorktreeInfo } from '../../shared/worktree/types' import { annotateSparseCheckoutStatus, listWorktreeGraph as listWorktreeGraphUnshared, - listWorktreesStrict as listWorktreesStrictUnshared + listWorktreesStrict as listWorktreesStrictUnshared, + listWorktreesStrictAllowingTrueEmpty as listWorktreesStrictAllowingTrueEmptyUnshared } from './worktree-listing' import type { GitWorktreeExecOptions } from './worktree-operation-options' import { WORKTREE_LIST_TIMEOUT_MS } from './worktree-operation-options' @@ -10,7 +11,7 @@ import { WORKTREE_LIST_TIMEOUT_MS } from './worktree-operation-options' // Why: share concurrent `git worktree list` scans, which are expensive on Windows. const inFlightWorktreeScans = new Map>() -type WorktreeScanKind = 'graph' | 'lenient' | 'strict' +type WorktreeScanKind = 'graph' | 'lenient' | 'strict' | 'strict-true-empty' // Why: mutation generations prevent listings from joining stale scans. const worktreeScanGenerations = new Map() @@ -139,3 +140,20 @@ export function listWorktreesSharedStrict( ): Promise { return shareWorktreeScan(repoPath, options, 'strict', listWorktreesStrictUnshared) } + +/** + * The detected scan's discipline: reject a Git/host failure so it cannot publish as an + * authoritative empty listing, but still answer `[]` for a repo that is gone or not a repo. + * Its own kind because neither a strict nor a lenient joiner may inherit that middle contract. + */ +export function listWorktreesSharedStrictAllowingTrueEmpty( + repoPath: string, + options: GitWorktreeExecOptions = {} +): Promise { + return shareWorktreeScan( + repoPath, + options, + 'strict-true-empty', + listWorktreesStrictAllowingTrueEmptyUnshared + ) +} diff --git a/src/main/git/worktree.ts b/src/main/git/worktree.ts index 18e92e2873a..024485b6981 100644 --- a/src/main/git/worktree.ts +++ b/src/main/git/worktree.ts @@ -32,7 +32,8 @@ export { _resetWorktreeScanCacheForTests, listWorktreeGraph, listWorktrees, - listWorktreesSharedStrict + listWorktreesSharedStrict, + listWorktreesSharedStrictAllowingTrueEmpty } from './worktree-scan-cache' export { bumpWorktreeScanGeneration as notifyPreparedWorktreeMutation } from './worktree-scan-cache' export { addSparseWorktree } from './worktree-sparse-add' diff --git a/src/main/ipc/notebook.ts b/src/main/ipc/notebook.ts index 9255ab7383d..d90ab3616de 100644 --- a/src/main/ipc/notebook.ts +++ b/src/main/ipc/notebook.ts @@ -4,6 +4,7 @@ import { dirname } from 'node:path' import { ipcMain } from 'electron' import type { Store } from '../persistence' import { resolveAuthorizedPath } from './filesystem-auth' +import { admitSelfInitiatedTreeKill } from '../own-chromium-tree-kill-guard' export type NotebookRunResult = { stdout: string @@ -53,7 +54,8 @@ function appendBounded(capture: BoundedCapture, chunk: Buffer): void { capture.truncated = true } -function terminateNotebookProcessTree( +/** Exported for the refusal-fallback test; the timeout path is otherwise unreachable. */ +export function terminateNotebookProcessTree( child: ChildProcessWithoutNullStreams ): ReturnType | null { if (!child.pid) { @@ -62,6 +64,18 @@ function terminateNotebookProcessTree( } if (process.platform === 'win32') { + if ( + !admitSelfInitiatedTreeKill({ + pid: child.pid, + site: 'notebook-cell-timeout', + scope: 'win-taskkill-tree' + }) + ) { + // Refusal blocks the tree walk, not the termination: killing the root by + // handle cannot reach a recycled pid, and a timed-out cell must still stop. + child.kill() + return null + } try { // Why: a timed-out cell can spawn descendants. taskkill /T is the // Windows equivalent of terminating the whole process group. diff --git a/src/main/ipc/orca-profile-auth-status-broadcast.ts b/src/main/ipc/orca-profile-auth-status-broadcast.ts new file mode 100644 index 00000000000..b5ac8483943 --- /dev/null +++ b/src/main/ipc/orca-profile-auth-status-broadcast.ts @@ -0,0 +1,15 @@ +import { BrowserWindow } from 'electron' +import { ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL } from '../../shared/orca-profiles' + +export function broadcastOrcaProfileAuthStatusChanged(): void { + for (const window of BrowserWindow.getAllWindows()) { + if (window.isDestroyed()) { + continue + } + try { + window.webContents.send(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL) + } catch { + // A renderer can disappear between isDestroyed() and send(). + } + } +} diff --git a/src/main/ipc/orca-profiles.ts b/src/main/ipc/orca-profiles.ts index bf1bf666270..480c6f350f9 100644 --- a/src/main/ipc/orca-profiles.ts +++ b/src/main/ipc/orca-profiles.ts @@ -45,6 +45,8 @@ import { signOutCurrentOrcaProfile } from '../orca-profiles/profile-cloud-service' import { registerOrcaProfileOrgMemberHandlers } from './orca-profile-org-members-handlers' +import { onOrcaCloudSessionInvalidated } from '../orca-profiles/profile-cloud-session-invalidation' +import { broadcastOrcaProfileAuthStatusChanged } from './orca-profile-auth-status-broadcast' type RegisterOrcaProfileHandlersOptions = { onBeforeRelaunch?: () => void | Promise @@ -178,6 +180,12 @@ export function registerOrcaProfileHandlers( getCurrentOrcaProfileAuthStatus(getProfileUserDataPath()) ) + // Why: a background refresh can revoke the session with no renderer request in + // flight, so push the change instead of waiting for the next pane to ask. + // Why not options.onAuthMutation: that hook drives the relay coordinator, which + // is the caller that just failed the refresh — re-entering it here would be a loop. + onOrcaCloudSessionInvalidated(broadcastOrcaProfileAuthStatusChanged) + ipcMain.handle( 'orcaProfiles:createLocal', (_event, args?: CreateLocalOrcaProfileArgs): CreateLocalOrcaProfileResult => { diff --git a/src/main/ipc/pty/ipc/spawn-env.ts b/src/main/ipc/pty/ipc/spawn-env.ts index af5da3858bd..94f1acf363e 100644 --- a/src/main/ipc/pty/ipc/spawn-env.ts +++ b/src/main/ipc/pty/ipc/spawn-env.ts @@ -7,7 +7,11 @@ import { isRemoteAgentHooksEnabled } from '../../../../shared/agent-hook-relay' import { isOpaqueRemintedPaneKey } from '../../../../shared/pane-key-alias' import { isValidTerminalTabId } from '../../../../shared/terminal-tab-id' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' -import { hasClaudeAuthEnvConflict } from '../../../claude-accounts/environment' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + hasClaudeAuthEnvConflict +} from '../../../claude-accounts/environment' import { LocalPtyProvider } from '../../../providers/local-pty-provider' import { resolvePathEnvKey } from '../../../pty/windows-environment-path' import { routesFreshSpawnsToLocalProvider } from '../host-env/fresh-spawn-routing' @@ -20,12 +24,10 @@ import { assemblePtyIpcSpawnCodexEnv } from './spawn-env-codex' export async function assemblePtyIpcSpawnEnv(ctx: PtyIpcSpawnState): Promise { const args = ctx.args if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } if (ctx.claudeAuth?.stripAuthEnv && hasClaudeAuthEnvConflict(args.env)) { - throw new Error( - 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' - ) + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) } // Why: the daemon-backed provider skips LocalPtyProvider's buildSpawnEnv, so assemble the same host-local env here for parity. // Safety: skip entirely for SSH — every injection is a loopback secret or a local path that leaks or misleads on the remote host. diff --git a/src/main/ipc/pty/ipc/spawn-preflight.ts b/src/main/ipc/pty/ipc/spawn-preflight.ts index f40b46e5dd5..f885d6e2f6f 100644 --- a/src/main/ipc/pty/ipc/spawn-preflight.ts +++ b/src/main/ipc/pty/ipc/spawn-preflight.ts @@ -4,6 +4,7 @@ import { } from '../../../../shared/local-windows-terminal-runtime' import { isWslUncPath, toWindowsWslPath } from '../../../../shared/wsl-paths' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' +import { CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE } from '../../../claude-accounts/environment' import { mintPtySessionId } from '../../../daemon/pty-session-id' import { resolveWslSessionContext } from '../../../daemon/wsl-session-context' import { LocalPtyProvider } from '../../../providers/local-pty-provider' @@ -193,7 +194,7 @@ export async function preparePtyIpcSpawnPreflight(ctx: PtyIpcSpawnState): Promis ctx.isClaudeLaunch = !ctx.preAdoptedStablePane && !args.connectionId && isClaudeLaunchCommand(args.command) if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } ctx.terminalRuntimeOptions = process.platform === 'win32' && !args.connectionId diff --git a/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts b/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts new file mode 100644 index 00000000000..86e16121f5c --- /dev/null +++ b/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts @@ -0,0 +1,134 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { TERMINAL_INPUT_CHUNK_MAX_BYTES } from '../../../../shared/terminal-input' +import { agentSessionPtyWriteGate } from '../../../runtime/agent-session-pty-write-gate' +import { ptyOwnership } from '../provider/ownership-state' +import { createPtyWriteInput } from './write-input' + +const PTY_ID = 'pty-chunk-yield' + +const { provider } = vi.hoisted(() => ({ provider: { write: vi.fn() } })) + +vi.mock('../provider/registry', () => ({ + tryGetProviderForPty: (id: string) => (id === PTY_ID ? provider : undefined) +})) + +const realSetImmediate = globalThis.setImmediate +const realReadmit = agentSessionPtyWriteGate.readmit.bind(agentSessionPtyWriteGate) +const THREE_CHUNK_INPUT = 'x'.repeat(TERMINAL_INPUT_CHUNK_MAX_BYTES * 2 + 8) + +const mainWindow = { + isDestroyed: () => false, + webContents: { isDestroyed: () => false, send: vi.fn() } +} + +/** Resolves after `turns` real check-phase passes; never touches the (faked) timer queue. */ +function afterImmediateTurns(turns: number): Promise<'stalled'> { + return new Promise((resolve) => { + const step = (remaining: number): void => { + if (remaining === 0) { + resolve('stalled') + return + } + realSetImmediate(() => step(remaining - 1)) + } + step(turns) + }) +} + +function createWriteInput(): ReturnType['writePtyInput'] { + return createPtyWriteInput({ + mainWindow: mainWindow as never, + clearHiddenRendererResizeOutput: vi.fn() + }).writePtyInput +} + +beforeEach(() => { + ptyOwnership.set(PTY_ID, null) + provider.write.mockReset() + mainWindow.webContents.send.mockReset() + // Why: only setTimeout is faked. A setTimeout(0) yield would stall the write forever here, + // while a setImmediate yield still runs in Node's check phase — the race below is deterministic. + vi.useFakeTimers({ toFake: ['setTimeout'] }) +}) + +afterEach(() => { + ptyOwnership.delete(PTY_ID) + vi.restoreAllMocks() + vi.useRealTimers() +}) + +describe('chunked pty write yield', () => { + it('yields between chunks via setImmediate, not a timer, and readmits before each later chunk', async () => { + const events: string[] = [] + provider.write.mockImplementation((_id: string, data: string) => { + events.push(`write:${data.length}`) + }) + vi.spyOn(agentSessionPtyWriteGate, 'readmit').mockImplementation((...args) => { + events.push('readmit') + return realReadmit(...args) + }) + vi.spyOn(globalThis, 'setImmediate').mockImplementation(((callback: () => void) => { + events.push('yield') + return realSetImmediate(callback) + }) as typeof setImmediate) + + const outcome = await Promise.race([ + createWriteInput()({ id: PTY_ID, data: THREE_CHUNK_INPUT }), + afterImmediateTurns(50) + ]) + + expect(outcome).toBe(true) + expect(vi.getTimerCount()).toBe(0) + expect(events).toEqual([ + `write:${TERMINAL_INPUT_CHUNK_MAX_BYTES}`, + 'yield', + 'readmit', + `write:${TERMINAL_INPUT_CHUNK_MAX_BYTES}`, + 'yield', + 'readmit', + 'write:8' + ]) + expect(mainWindow.webContents.send).not.toHaveBeenCalled() + }) + + it('does not yield for input that fits in a single chunk', async () => { + const immediate = vi.spyOn(globalThis, 'setImmediate') + + const outcome = createWriteInput()({ + id: PTY_ID, + data: 'x'.repeat(TERMINAL_INPUT_CHUNK_MAX_BYTES) + }) + + expect(outcome).toBe(true) + expect(provider.write).toHaveBeenCalledTimes(1) + expect(immediate).not.toHaveBeenCalled() + }) + + it('stops after a yield once readmission refuses', async () => { + const readmit = vi.spyOn(agentSessionPtyWriteGate, 'readmit').mockReturnValue({ + admitted: false, + refusal: { + code: 'agent_session_checkpoint_stale', + sessionId: 'session-1', + ownerRuntimeKind: null, + handoffStage: null, + ownerPid: null, + runtimeFence: null + } + }) + + const outcome = await Promise.race([ + createWriteInput()({ id: PTY_ID, data: THREE_CHUNK_INPUT }), + afterImmediateTurns(50) + ]) + + expect(outcome).toBe(false) + expect(provider.write).toHaveBeenCalledTimes(1) + expect(readmit).toHaveBeenCalledTimes(1) + expect(mainWindow.webContents.send).toHaveBeenCalledWith( + 'pty:writeUnavailable', + expect.objectContaining({ id: PTY_ID }) + ) + }) +}) diff --git a/src/main/ipc/pty/ipc/write-input.ts b/src/main/ipc/pty/ipc/write-input.ts index f27604dc9f8..6713e49000b 100644 --- a/src/main/ipc/pty/ipc/write-input.ts +++ b/src/main/ipc/pty/ipc/write-input.ts @@ -152,7 +152,9 @@ export function createPtyWriteInput(deps: { first = false provider.write(id, chunk.value) if (!nextChunk.done) { - await new Promise((resolve) => setTimeout(resolve, 0)) + // setImmediate, not setTimeout(0): the yield exists to let abort/data callbacks run + // between chunks, and a clamped timer tick per 16 KiB is pure latency. + await new Promise((resolve) => setImmediate(resolve)) } chunk = nextChunk nextChunk = chunks.next() diff --git a/src/main/ipc/pty/runtime/spawn-preflight.ts b/src/main/ipc/pty/runtime/spawn-preflight.ts index aed89b44df8..43fd2778119 100644 --- a/src/main/ipc/pty/runtime/spawn-preflight.ts +++ b/src/main/ipc/pty/runtime/spawn-preflight.ts @@ -23,7 +23,11 @@ import { import { stripRemotePaneEnvWhenHooksDisabled } from '../provider/liveness' import { isTuiAgent } from '../../../../shared/tui-agent-config' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' -import { hasClaudeAuthEnvConflict } from '../../../claude-accounts/environment' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + hasClaudeAuthEnvConflict +} from '../../../claude-accounts/environment' import { isSafePtySessionId, mintPtySessionId, @@ -65,7 +69,7 @@ export async function prepareRuntimePtySpawn( ctx.isClaudeLaunch = !ctx.preAdoptedStablePane && !args.connectionId && isClaudeLaunchCommand(args.command) if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } // Why: runtime-created terminals carry no renderer-computed projectRuntime; resolve from worktreeId to honor the project's Windows runtime. ctx.terminalRuntimeOptions = @@ -134,12 +138,10 @@ export async function prepareRuntimePtySpawn( ? await ctx.deps.prepareClaudeAuth(ctx.codexSelectionTarget) : null if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } if (ctx.claudeAuth?.stripAuthEnv && hasClaudeAuthEnvConflict(args.env)) { - throw new Error( - 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' - ) + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) } ctx.shouldPersistHostSessionBinding = args.persistHostSessionBinding === true diff --git a/src/main/ipc/runtime.test.ts b/src/main/ipc/runtime.test.ts index dc7f8a01cd7..07010087363 100644 --- a/src/main/ipc/runtime.test.ts +++ b/src/main/ipc/runtime.test.ts @@ -136,6 +136,40 @@ describe('registerRuntimeHandlers', () => { }) }) + it('projects Claude structured tabs to the same-version desktop client', async () => { + const claudeTab = { + type: 'agent-session', + id: 'agent-session:claude-1', + title: 'Claude Chat', + sessionId: 'claude-1', + agent: 'claude', + isActive: true + } + const runtime = { + getRuntimeId: vi.fn().mockReturnValue('runtime-1'), + restoreStructuredAgentSessionTabs: vi.fn(async () => undefined), + listMobileSessionTabs: vi.fn(async () => ({ + worktree: 'workspace-1', + publicationEpoch: 'epoch-1', + snapshotVersion: 1, + activeGroupId: 'group-1', + activeTabId: claudeTab.id, + activeTabType: 'agent-session', + tabGroups: [{ id: 'group-1', activeTabId: claudeTab.id, tabOrder: [claudeTab.id] }], + tabs: [claudeTab] + })) + } + + registerRuntimeHandlers(runtime as never) + const callRegistration = handleMock.mock.calls.find(([channel]) => channel === 'runtime:call') + const result = await callRegistration![1](runtimeCallEvent(), { + method: 'session.tabs.list', + params: { worktree: 'id:workspace-1' } + }) + + expect(result).toMatchObject({ ok: true, result: { tabs: [claudeTab] } }) + }) + it('registers project group runtime RPC methods for local desktop callers', async () => { const runtime = { syncWindowGraph: vi.fn(), diff --git a/src/main/ipc/runtime.ts b/src/main/ipc/runtime.ts index 901d14bfce6..3901d8b1ffa 100644 --- a/src/main/ipc/runtime.ts +++ b/src/main/ipc/runtime.ts @@ -10,7 +10,10 @@ import type { import type { RuntimeRpcResponse } from '../../shared/runtime-rpc-envelope' import type { ClientHostedBrowserRowsEvent } from '../../shared/client-hosted-browser-rows' import { TERMINAL_FIT_RESTORE_DEADLINE_MS } from '../../shared/terminal-fit-restore-deadline' -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' +import { + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../shared/protocol-version' import { RpcDispatcher } from '../runtime/rpc/dispatcher' import { ALL_RPC_METHODS } from '../runtime/rpc/methods' import { DesktopRuntimeSenderLifecycle } from './desktop-runtime-sender-lifecycle' @@ -76,7 +79,10 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientId: 'desktop-renderer', clientKind: 'runtime', connectionId: desktopSenders.connectionIdFor(event.sender), - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] } )) as RuntimeRpcResponse } @@ -121,7 +127,10 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientId: 'desktop-renderer', clientKind: 'runtime', connectionId, - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] } ) .finally(stop) diff --git a/src/main/ipc/worktrees-test-module-mocks.ts b/src/main/ipc/worktrees-test-module-mocks.ts index f31f6ae1dd6..a1925d1ce72 100644 --- a/src/main/ipc/worktrees-test-module-mocks.ts +++ b/src/main/ipc/worktrees-test-module-mocks.ts @@ -115,6 +115,7 @@ export const gitWorktreeModuleMock = () => ({ listWorktrees: listWorktreesMock, listWorktreesStrict: listWorktreesMock, listWorktreesSharedStrict: listWorktreesMock, + listWorktreesSharedStrictAllowingTrueEmpty: listWorktreesMock, describeCreatedWorktree: describeCreatedWorktreeMock, parseWorktreeList: parseWorktreeListMock, assertWorktreeCleanForRemoval: assertWorktreeCleanForRemovalMock, diff --git a/src/main/ipc/worktrees-windows.test.ts b/src/main/ipc/worktrees-windows.test.ts index 4d4115264a3..7a54200d9f8 100644 --- a/src/main/ipc/worktrees-windows.test.ts +++ b/src/main/ipc/worktrees-windows.test.ts @@ -74,6 +74,7 @@ vi.mock('../git/worktree', () => ({ listWorktrees: listWorktreesMock, listWorktreesStrict: listWorktreesMock, listWorktreesSharedStrict: listWorktreesMock, + listWorktreesSharedStrictAllowingTrueEmpty: listWorktreesMock, describeCreatedWorktree: vi.fn().mockResolvedValue(undefined), assertWorktreeCleanForRemoval: assertWorktreeCleanForRemovalMock, addWorktree: addWorktreeMock, diff --git a/src/main/ipc/worktrees/listing/detected-provider-listing.ts b/src/main/ipc/worktrees/listing/detected-provider-listing.ts index 388c825530f..ec5e7606e45 100644 --- a/src/main/ipc/worktrees/listing/detected-provider-listing.ts +++ b/src/main/ipc/worktrees/listing/detected-provider-listing.ts @@ -25,7 +25,11 @@ import { type DetectedWorktreeMetadataPrune, type DetectedWorktreeSideEffectToken } from './detected-worktree-scan-cache' -import { loggedWorktreeListFailures, warnOnce } from './worktree-listing-diagnostics' +import { + describeWorktreeScanFailure, + loggedWorktreeListFailures, + warnOnce +} from './worktree-listing-diagnostics' import { readAllWorktreeMetaForRepo } from '../../../persistence/host-qualified-worktree-meta' export async function listDetectedWorktreesForCapturedRepo( @@ -157,16 +161,25 @@ export async function listDetectedWorktreesForCapturedRepo( `[worktrees] failed to list detected worktrees for repo "${repo.displayName}" (${repo.id}) at ${repo.path}`, err ) + // Why: retention alone leaves inert rows with no explanation; the cause rides with the listing. + const unavailableReason = describeWorktreeScanFailure(err) if (repo.connectionId) { const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex()) return { repoId: repo.id, authoritative: false, source: 'metadata-fallback', - worktrees: buildDisconnectedDetectedWorktrees(store, repo, worktrees) + worktrees: buildDisconnectedDetectedWorktrees(store, repo, worktrees), + unavailableReason } } - return { repoId: repo.id, authoritative: false, source: 'metadata-fallback', worktrees: [] } + return { + repoId: repo.id, + authoritative: false, + source: 'metadata-fallback', + worktrees: [], + unavailableReason + } } } diff --git a/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts b/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts new file mode 100644 index 00000000000..4e436b8dcf1 --- /dev/null +++ b/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts @@ -0,0 +1,166 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { DetectedWorktreeListResult } from '../../../../shared/worktree/types' + +const gitExecFileAsyncMock = vi.hoisted(() => vi.fn()) + +vi.mock('electron', () => ({ + ipcMain: { handle: vi.fn(), removeHandler: vi.fn() }, + app: { getPath: () => '/tmp/orca-test' } +})) +vi.mock('../../../git/runner', async (importOriginal) => ({ + ...(await importOriginal>()), + gitExecFileAsync: gitExecFileAsyncMock +})) + +const { listDetectedWorktreesForCapturedRepo } = await import('./detected-provider-listing') +const { __resetDetectedWorktreeScanCacheForTests } = await import('./detected-worktree-scan-cache') +const { _resetWorktreeScanCacheForTests } = await import('../../../git/worktree-scan-cache') +const { isRegisteredWorktreePath, invalidateAuthorizedRootsCache } = + await import('../../registered-worktree-roots-cache') + +const REPO_PATH = '/workspace/repo' +const repo = { + id: 'repo-1', + path: REPO_PATH, + displayName: 'repo', + badgeColor: '#000', + addedAt: 0 +} as Repo + +const removeWorktreeLineage = vi.fn() + +function createStore() { + return { + getRepo: () => repo, + getRepos: () => [repo], + getProjects: () => [], + getSettings: () => ({}), + getAllWorktreeMeta: () => ({}), + getProjectHostSetups: () => [], + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getAllWorkspaceLineage: () => ({}), + removeWorktreeLineage, + captureNativeLocalWorktreeMetadataScanExpectation: () => undefined + } as never +} + +/** The field failure: wsl.exe exits 0xFFFFFFFF, says nothing on stderr, and git never ran. */ +function wslHostFailure(): Error { + return Object.assign(new Error('Command failed: wsl.exe -d kali-linux --exec sh -lc ...'), { + code: 4294967295, + stdout: 'Error code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\r\n', + stderr: '' + }) +} + +async function listDetected(): Promise { + const result = await listDetectedWorktreesForCapturedRepo(createStore(), repo, () => true) + return result as DetectedWorktreeListResult +} + +describe('detected worktree listing authority', () => { + beforeEach(() => { + gitExecFileAsyncMock.mockReset() + removeWorktreeLineage.mockReset() + __resetDetectedWorktreeScanCacheForTests() + _resetWorktreeScanCacheForTests() + invalidateAuthorizedRootsCache() + }) + + it('reports a failed scan as non-authoritative and prunes nothing', async () => { + gitExecFileAsyncMock.mockRejectedValue(wslHostFailure()) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.source).toBe('metadata-fallback') + expect(result.worktrees).toEqual([]) + // Why: the retained rows must carry the cause, or the user sees inert worktrees with no explanation. + expect(result.unavailableReason).toContain('Command failed: wsl.exe') + // The destructive halves of a fresh scan must not run against a listing that failed. + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(false) + expect(removeWorktreeLineage).not.toHaveBeenCalled() + }) + + it('surfaces the annotated wsl.exe diagnostic as the unavailable reason', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign( + new Error( + 'wsl.exe host failure (distro "kali-linux"): There is no distribution with the supplied name.\r\nError code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\nCommand failed: wsl.exe -d kali-linux --exec sh -lc ...' + ), + { code: 4294967295, stdout: '', stderr: '' } + ) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.unavailableReason).toBe( + 'wsl.exe host failure (distro "kali-linux"): There is no distribution with the supplied name. Error code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND' + ) + }) + + // Why (measured on a real Windows host): under WSL the spawn cwd is the interop directory, so a + // deleted guest repo fails as `bash: cd` exit 1 — not ENOENT — and must stay retained, not pruned. + it('retains a WSL repo whose guest directory is gone, and says why', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('bash: line 1: cd: /home/neil/repo: No such file or directory'), { + code: 1, + stdout: '', + stderr: 'bash: line 1: cd: /home/neil/repo: No such file or directory\n' + }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.unavailableReason).toContain('No such file or directory') + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(false) + expect(removeWorktreeLineage).not.toHaveBeenCalled() + }) + + it('keeps an empty listing authoritative when the path is not a Git repo', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('Command failed: git worktree list'), { + code: 128, + stderr: 'fatal: not a git repository (or any of the parent directories): .git\n' + }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.source).toBe('git') + expect(result.worktrees).toEqual([]) + expect(result.unavailableReason).toBeUndefined() + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(true) + }) + + it('keeps an empty listing authoritative when the repo path is gone', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('spawn git ENOENT'), { code: 'ENOENT', stderr: '' }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.source).toBe('git') + expect(result.worktrees).toEqual([]) + }) + + it('stays authoritative for a healthy scan', async () => { + gitExecFileAsyncMock.mockResolvedValue({ + stdout: `worktree ${REPO_PATH}\u0000HEAD abc\u0000branch refs/heads/main\u0000\u0000`, + stderr: '' + }) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.worktrees.map((worktree) => worktree.path)).toEqual([REPO_PATH]) + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(true) + }) +}) diff --git a/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts b/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts index b694bfbf11d..09b0156cb54 100644 --- a/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts +++ b/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts @@ -3,7 +3,7 @@ import type { Store } from '../../../persistence/loading-store/store' import type { Repo } from '../../../../shared/repo-types' import { getLocalProjectWorktreeGitOptions } from '../../../project-runtime-git-options' import { isFolderRepo } from '../../../../shared/repo-kind' -import { listRepoWorktrees } from '../../../repo-worktrees' +import { listRepoWorktreesForDetectedScan } from '../../../repo-worktrees' import { getRegisteredWorktreeRootsRevision, registerWorktreeRootsForRepo @@ -111,7 +111,7 @@ export async function listDetectedGitWorktrees( const localWorktreeGitOptions = getLocalProjectWorktreeGitOptions(store, repo) if (repo.connectionId || isFolderRepo(repo)) { return { - gitWorktrees: await listRepoWorktrees(repo, localWorktreeGitOptions), + gitWorktrees: await listRepoWorktreesForDetectedScan(repo, localWorktreeGitOptions), fresh: true } } @@ -144,7 +144,7 @@ export async function listDetectedGitWorktrees( : undefined const scan: DetectedWorktreeScan = { invalidated: false, - promise: listRepoWorktrees(repo, localWorktreeGitOptions), + promise: listRepoWorktreesForDetectedScan(repo, localWorktreeGitOptions), sideEffectToken: { generation, authorizedRootsRevision }, hygieneDue, ...(metadataPruneExpectation diff --git a/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts b/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts index cb8c48833ef..655cca4fa77 100644 --- a/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts +++ b/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts @@ -10,7 +10,9 @@ const { listRepoWorktreesMock, pruneLineageMock, pruneMetadataMock, registerWork registerWorktreeRootsMock: vi.fn() })) -vi.mock('../../../repo-worktrees', () => ({ listRepoWorktrees: listRepoWorktreesMock })) +vi.mock('../../../repo-worktrees', () => ({ + listRepoWorktreesForDetectedScan: listRepoWorktreesMock +})) vi.mock('../../../project-runtime-git-options', () => ({ getLocalProjectWorktreeGitOptions: () => ({}) })) diff --git a/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts b/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts index 128bb46dd28..56bbe5d4ad2 100644 --- a/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts +++ b/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts @@ -14,3 +14,23 @@ export function warnOnce(keySet: Set, key: string, message: string, erro console.warn(message) } } + +const SCAN_FAILURE_REASON_MAX_CHARS = 240 + +/** + * The cause a retained-but-unscannable repo shows the user. The first two lines carry the + * classifier's summary plus its `Wsl/Service/WSL_E_*` code; everything after is the raw command. + */ +export function describeWorktreeScanFailure(error: unknown): string { + const message = error instanceof Error ? error.message : String(error) + const summary = message + .split(/\r?\n/) + .map((line) => line.trim()) + .filter((line) => line.length > 0) + .slice(0, 2) + .join(' ') + const reason = summary.length > 0 ? summary : 'Worktree scan failed with no diagnostic.' + return reason.length > SCAN_FAILURE_REASON_MAX_CHARS + ? `${reason.slice(0, SCAN_FAILURE_REASON_MAX_CHARS - 1)}…` + : reason +} diff --git a/src/main/main-process-tree-kill-gate.test.ts b/src/main/main-process-tree-kill-gate.test.ts new file mode 100644 index 00000000000..40b1421c32f --- /dev/null +++ b/src/main/main-process-tree-kill-gate.test.ts @@ -0,0 +1,184 @@ +import { readFileSync, readdirSync, statSync } from 'node:fs' +import { join, relative, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * The ratchet behind the guard's claim to be a choke point. + * + * `admitSelfInitiatedTreeKill` is only "one decision" for as long as every + * pid-addressed `taskkill /pid /t /f` in Electron main asks it. Each such + * kill can land on a recycled pid that is now one of Orca's own Chromium + * processes (#10680), and an ungated one is also invisible to + * `selfInitiatedTreeKillCount`, which makes a zero read as exculpatory when it + * is not. A new family fails here rather than in the field. + * + * Exactly what is enforced, so no comment elsewhere claims more: per file, the + * number of gate admissions must be at least the number of `/pid` call sites. + * Counting sites rather than files is the point — a file-granular scan would let + * a second, ungated taskkill land inside a family that already mentions the gate, + * which is the shape the six highest-risk files now have. What it still cannot + * see: a site that pairs an ungated kill with a second admission of an already + * gated one in the same file, and a kill whose `/pid` argument is itself built + * from a variable. + */ +const REPOSITORY_ROOT = resolve(__dirname, '..', '..') +const MAIN_DIRECTORY = 'src/main/' +const SCANNED_EXTENSIONS = ['.ts', '.tsx'] +const IGNORED_DIRECTORIES = new Set([ + 'node_modules', + 'dist', + 'out', + 'build', + '.git', + '__fixtures__' +]) + +/** + * One match per call site. Keyed on the `/pid` argument rather than the program + * name because `/pid ` is what makes the kill pid-addressed — it walks + * whatever tree owns that pid *now* — and because the literal survives a + * `taskkill` spawned through a constant or a variable, which a quoted-program + * pattern misses entirely. + */ +const PID_ADDRESSED_KILL_SITE = /['"]\/pid['"]/gi + +/** + * A call, not an import or a comment: `admitSelfInitiatedTreeKill` in main, and + * `admitProcessTreeKill` for the `src/shared` seam main installs the same gate + * into, which shared code cannot import directly. + */ +const GATE_ADMISSION = /\badmit(?:SelfInitiatedTreeKill|ProcessTreeKill)\s*\(/g + +function countMatches(source: string, pattern: RegExp): number { + return source.match(pattern)?.length ?? 0 +} + +/** Sites left over once each admission in the file has claimed one. */ +function ungatedKillSiteCount(source: string): number { + return Math.max( + countMatches(source, PID_ADDRESSED_KILL_SITE) - countMatches(source, GATE_ADMISSION), + 0 + ) +} + +/** + * Only ever shrinks. Each entry states why the gate cannot reach it — never + * "not got to yet", which is what a new ungated family would also look like. + */ +const UNGATED_TASKKILL_ALLOWLIST = new Map([ + [ + 'src/main/browser/browser-route-egress-electron-launch.ts', + 'Electron probe reached only from *.electron.test.ts; kills the probe Electron it spawned' + ], + [ + 'src/main/browser/browser-route-persisted-worker-electron-process.ts', + 'Electron probe reached only from *.electron.test.ts; kills the probe Electron it spawned' + ], + [ + 'src/cli/handlers/interactive-login-interruption.ts', + 'CLI host: no Chromium pid on the machine to reach, and no reader for the ring' + ], + [ + 'src/relay/subprocess-tree-termination.ts', + 'Relay host: same, and the relay cannot import the main-process gate' + ] +]) + +function isTestFile(path: string): boolean { + return /\.(?:test|spec)\.tsx?$/.test(path) || /(?:test-harness|test-fixture|fixture)/.test(path) +} + +function scanSourceFiles(directory: string, found: string[] = []): string[] { + for (const entry of readdirSync(directory)) { + if (IGNORED_DIRECTORIES.has(entry)) { + continue + } + const path = join(directory, entry) + if (statSync(path).isDirectory()) { + scanSourceFiles(path, found) + continue + } + if (SCANNED_EXTENSIONS.some((extension) => entry.endsWith(extension)) && !isTestFile(path)) { + found.push(path) + } + } + return found +} + +// Only the Node-side hosts: a renderer or preload cannot spawn a process at all. +const SCANNED_HOSTS = ['src/main', 'src/shared', 'src/cli', 'src/relay'] + +/** Scanned once at import: 10k files is seconds, and every case below reuses it. */ +const PID_ADDRESSED_KILL_FILES = SCANNED_HOSTS.flatMap((host) => + scanSourceFiles(join(REPOSITORY_ROOT, host)) + .map((path) => ({ + path: relative(REPOSITORY_ROOT, path).split('\\').join('/'), + source: readFileSync(path, 'utf8') + })) + .filter((file) => countMatches(file.source, PID_ADDRESSED_KILL_SITE) > 0) +) + +function pidAddressedKillFiles(): { path: string; source: string }[] { + return PID_ADDRESSED_KILL_FILES +} + +describe('main-process tree-kill gate', () => { + it('finds the taskkill families it is meant to police', () => { + // Falsifiable: a scanner that matched nothing would pass every case below. + expect(pidAddressedKillFiles().map((file) => file.path)).toContain( + 'src/main/windows-process-tree-kill.ts' + ) + }) + + it('routes every pid-addressed taskkill in Electron main through the gate', () => { + const ungated = pidAddressedKillFiles() + .filter((file) => file.path.startsWith(MAIN_DIRECTORY)) + .filter((file) => ungatedKillSiteCount(file.source) > 0) + .map((file) => file.path) + .filter((path) => !UNGATED_TASKKILL_ALLOWLIST.has(path)) + + expect(ungated).toEqual([]) + }) + + it('leaves no pid-addressed taskkill outside main unaccounted for', () => { + const unaccounted = pidAddressedKillFiles() + .filter((file) => !file.path.startsWith(MAIN_DIRECTORY)) + .filter((file) => ungatedKillSiteCount(file.source) > 0) + .map((file) => file.path) + .filter((path) => !UNGATED_TASKKILL_ALLOWLIST.has(path)) + + expect(unaccounted).toEqual([]) + }) + + it('counts call sites, not files: a second ungated kill in a gated file is caught', () => { + // The failure a file-granular scan let through: one gate mention exempting + // every taskkill in the file. + const gated = ` + import { admitSelfInitiatedTreeKill } from './own-chromium-tree-kill-guard' + if (admitSelfInitiatedTreeKill({ pid, site: 's', scope: 'win-taskkill-tree' })) { + spawn('taskkill', ['/pid', String(pid), '/t', '/f']) + } + ` + + expect(ungatedKillSiteCount(gated)).toBe(0) + expect( + ungatedKillSiteCount(`${gated}\nspawn('taskkill', ['/pid', String(other), '/t', '/f'])`) + ).toBe(1) + }) + + it('sees a kill whose program name comes from a constant', () => { + // A quoted-program pattern misses this shape; the `/pid` argument does not. + expect( + ungatedKillSiteCount(` + const KILLER = 'taskkill' + spawn(KILLER, ['/pid', String(pid), '/t', '/f']) + `) + ).toBe(1) + }) + + it('keeps the allowlist honest: every entry still spawns a taskkill', () => { + const spawning = new Set(pidAddressedKillFiles().map((file) => file.path)) + + expect([...UNGATED_TASKKILL_ALLOWLIST.keys()].filter((path) => !spawning.has(path))).toEqual([]) + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-blob-store.ts b/src/main/native-chat/agent-session-journal/journal-blob-store.ts deleted file mode 100644 index 9b02877cec7..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-blob-store.ts +++ /dev/null @@ -1,110 +0,0 @@ -// Content-addressed store for the remainder of a bounded payload. -// -// Blobs are named by their sha256, so writing the same output twice costs one -// file and re-import is idempotent. They live beside the journal (host-side -// per-workspace state, never inside the user's working tree) and share the -// epoch's retention: compaction prunes every blob no retained row references. - -import { mkdir, readFile, readdir, rm, stat } from 'node:fs/promises' -import { join } from 'node:path' -import { durableWriteTempPath, writeFileDurable } from '../../durable-file-write' - -export const JOURNAL_BLOB_DIR = 'blobs' -const DIGEST_PATTERN = /^[0-9a-f]{64}$/ - -/** A digest arrives back from a row on disk, so it is untrusted by the time it - * reaches the filesystem: anything but a bare sha256 could escape the store. */ -function blobPath(journalDir: string, digest: string): string | null { - return DIGEST_PATTERN.test(digest) ? join(journalDir, JOURNAL_BLOB_DIR, digest) : null -} - -/** Persist `payload` under its digest. Returns the digest so the caller can - * stamp it on the row it is about to append. */ -export async function putJournalBlob( - journalDir: string, - digest: string, - payload: string -): Promise { - const target = blobPath(journalDir, digest) - if (!target) { - throw new Error('refusing to write a journal blob under a name that is not a sha256 digest') - } - // Content addressing makes a rewrite pointless: identical digest, identical bytes. - if (await pathExists(target)) { - return digest - } - await mkdir(join(journalDir, JOURNAL_BLOB_DIR), { recursive: true }) - await writeFileDurable(durableWriteTempPath(target), target, payload) - return digest -} - -export async function readJournalBlob(journalDir: string, digest: string): Promise { - const source = blobPath(journalDir, digest) - if (!source) { - return null - } - try { - return await readFile(source, 'utf-8') - } catch { - return null - } -} - -/** Remove a blob written speculatively for a row that was rejected. */ -export async function removeJournalBlob(journalDir: string, digest: string): Promise { - const target = blobPath(journalDir, digest) - if (target) { - await rm(target, { force: true }) - } -} - -/** Drop every blob outside `retained`. Called from compaction, under the - * current lease fence, after the snapshot is durable — so a crash mid-prune - * leaves extra blobs rather than dangling references. */ -export async function pruneJournalBlobs( - journalDir: string, - retained: ReadonlySet -): Promise { - let removed = 0 - let names: string[] - try { - names = await readdir(join(journalDir, JOURNAL_BLOB_DIR)) - } catch { - return 0 - } - for (const name of names) { - if (retained.has(name)) { - continue - } - await rm(join(journalDir, JOURNAL_BLOB_DIR, name), { force: true }).catch(() => {}) - removed += 1 - } - return removed -} - -async function pathExists(path: string): Promise { - try { - await stat(path) - return true - } catch { - return false - } -} - -export async function journalBlobFileSize( - journalDir: string, - digest: string -): Promise { - const target = blobPath(journalDir, digest) - if (!target) { - return null - } - try { - return (await stat(target)).size - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return null - } - throw error - } -} diff --git a/src/main/native-chat/agent-session-journal/journal-close-retry.ts b/src/main/native-chat/agent-session-journal/journal-close-retry.ts new file mode 100644 index 00000000000..e73be0a59b8 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-close-retry.ts @@ -0,0 +1,68 @@ +// Where a journal goes when its close REJECTS. +// +// `AgentSessionJournal.close()` is deliberately retryable: a rejection does not +// release the handle, and a second call is a real second attempt. Every caller +// that did `close().catch(() => undefined)` and then threw or overwrote its map +// entry defeated that contract — the store became unreachable with its SQLite +// connection still open. On POSIX that is a silent leak; on Windows the open +// handle blocks renaming or removing the journal directory outright. +// +// So a rejected close hands the journal here instead of dropping it, and host +// teardown retries everything this holds. Retention is bounded by construction: +// an entry leaves the set the moment its close fulfils, and a journal can be +// retained only once because the set is keyed by identity. + +/** Everything the registry needs; `AgentSessionJournal` satisfies it. */ +export type RetryableJournalClose = { + close: () => Promise + readonly directory: string +} + +export class JournalCloseRetryRegistry { + private readonly retained = new Set() + + /** Directories still held by a journal whose close has not fulfilled. */ + get pendingDirectories(): string[] { + return [...this.retained].map((journal) => journal.directory) + } + + /** + * Close it. Returns true when the handle is actually released; on a rejection + * the journal is RETAINED for `retryAll` and the rejection is returned rather + * than thrown, because every caller of this is already unwinding a different + * failure it must not lose. + */ + async closeOrRetain( + journal: RetryableJournalClose + ): Promise<{ closed: boolean; error?: unknown }> { + try { + await journal.close() + this.retained.delete(journal) + return { closed: true } + } catch (error) { + this.retained.add(journal) + return { closed: false, error } + } + } + + /** Retry every retained close. Ones that fulfil are dropped; ones that reject + * stay retained and their rejections are returned for the caller to report. */ + async retryAll(): Promise { + const entries = [...this.retained] + const failures: unknown[] = [] + for (const journal of entries) { + const result = await this.closeOrRetain(journal) + if (!result.closed) { + failures.push(result.error) + } + } + return failures + } +} + +/** + * Process-wide, because ownership of these handles is process-wide: the attach + * path, the recovery wrapper and runtime teardown are separate call trees that + * must all be able to reach the same orphan. + */ +export const agentSessionJournalCloseRetries = new JournalCloseRetryRegistry() diff --git a/src/main/native-chat/agent-session-journal/journal-compaction.ts b/src/main/native-chat/agent-session-journal/journal-compaction.ts deleted file mode 100644 index b9600a4f4e9..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-compaction.ts +++ /dev/null @@ -1,206 +0,0 @@ -// Retention and compaction. -// -// The snapshot carries the retained tail with it, so publishing both is ONE -// atomic write and there is no window where the folded state exists without the -// rows a reconnecting client still needs. Truncating the log afterwards is -// idempotent: a crash before it leaves the log a superset of the tail. -// -// The retained tail must cover the longest reconnect window Orca supports, or a -// client that was merely asleep gets a full snapshot reload instead of a resume. - -import { blobDigestsInBody, renderJournalState, type JournalReducerState } from './journal-reducer' -import { pruneJournalBlobs } from './journal-blob-store' -import { - rewriteJournalLog, - writeJournalSnapshotFile, - type JournalSnapshotFile -} from './journal-log-file' -import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' -import type { JournalRow } from './journal-row-schema' -import { AgentSessionJournalError } from './journal-write-guards' -import { assertJournalPhysicalCapacity, journalDirectoryBytes } from './journal-physical-quota' - -export type JournalCompactionPolicy = { - /** Always keep at least this many rows, however old they are. */ - minTailRows: number - /** Keep every row observed within this window. */ - retainTailMs: number - /** - * `window` honours `retainTailMs` outright. `budget-pressure` lets it yield: - * the alternative is refusing the user's writes until the window ages out, - * and the tail is only a resume optimization — compaction folds every shed - * row into the snapshot before truncating the log, so a client that loses - * its resume point reloads instead of losing conversation. Defaults to - * `window`. - */ - retention?: 'window' | 'budget-pressure' -} - -/** Two hours of tail comfortably covers a phone that slept through a commute, - * which is the longest reconnect Orca resumes rather than reloads. */ -export const DEFAULT_JOURNAL_COMPACTION_POLICY: JournalCompactionPolicy = { - minTailRows: 512, - retainTailMs: 2 * 60 * 60 * 1000 -} - -export type JournalCompactionResult = { - tailRows: JournalRow[] - compactedThrough: number - oldestSequence: number -} - -export async function compactJournal(input: { - journalDir: string - /** Parent quota root when compacting an in-directory staging journal. */ - physicalQuotaRoot?: string - state: JournalReducerState - tailRows: readonly JournalRow[] - policy?: JournalCompactionPolicy - now: number - maxSessionBytes: number - sessionId?: string -}): Promise { - const policy = input.policy ?? DEFAULT_JOURNAL_COMPACTION_POLICY - const retained = retainTail(input.tailRows, policy, input.now) - const rendered = renderJournalState(input.state) - const compactedThrough = input.state.lastSequence - - const snapshot: JournalSnapshotFile = { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: input.state.epoch, - compactedThrough, - highestFence: input.state.highestFence, - items: rendered.items, - submissions: rendered.submissions, - receipts: [...input.state.receipts.values()].map((receipt) => ({ - clientMessageId: receipt.clientMessageId, - providerItemId: receipt.providerItemId, - epoch: receipt.cursor.epoch, - sequence: receipt.cursor.sequence, - acceptedAt: receipt.acceptedAt - })), - aliases: [...input.state.aliases.entries()].map(([providerItemId, itemId]) => ({ - providerItemId, - itemId - })), - tombstones: [...input.state.tombstones.entries()].map(([itemId, revision]) => ({ - itemId, - revision - })), - appliedSettlementIds: [...input.state.appliedSettlementIds], - tail: retained - } - - const snapshotBytes = Buffer.byteLength(JSON.stringify(snapshot), 'utf8') - if (snapshotBytes > input.maxSessionBytes) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal snapshot reached its ${input.maxSessionBytes}-byte bound` - ) - } - - const sessionId = input.sessionId ?? input.state.sessionId - const quotaRoot = input.physicalQuotaRoot ?? input.journalDir - const retainedLogBytes = retained.reduce( - (total, row) => total + Buffer.byteLength(JSON.stringify(row), 'utf8') + 1, - 0 - ) - // Durable writes keep the old final alongside the new temp until rename. - // Reserve the complete compaction peak up front so a later copy cannot leave - // a half-published snapshot/log pair when the quota is tight. - await assertJournalPhysicalCapacity({ - journalDir: quotaRoot, - sessionId, - maxBytes: input.maxSessionBytes, - peakAdditionalBytes: snapshotBytes + retainedLogBytes - }) - await writeJournalSnapshotFile(input.journalDir, snapshot) - await rewriteJournalLog(input.journalDir, retained) - // Blobs are pruned last: a crash before this leaks bytes, whereas pruning - // first would strand a snapshot pointing at a payload that no longer exists. - // Recompute from exactly what the durable snapshot and retained log carry; - // this preserves reused/pre-existing blobs while allowing stale payloads to - // be pruned safely after both files are published. - const retainedDigests = new Set() - for (const item of snapshot.items) { - blobDigestsInBody(item.body, retainedDigests) - } - for (const row of retained) { - if (row.kind === 'item') { - blobDigestsInBody(row.body, retainedDigests) - } else if (row.kind === 'lifecycle-batch') { - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - blobDigestsInBody(mutation.body, retainedDigests) - } - } - } - } - await pruneJournalBlobs(input.journalDir, retainedDigests) - - if ((await journalDirectoryBytes(quotaRoot)) > input.maxSessionBytes) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${sessionId} exceeds its physical bound after compaction` - ) - } - - return { - tailRows: retained, - compactedThrough, - oldestSequence: retained[0]?.seq ?? compactedThrough + 1 - } -} - -function retainTail( - rows: readonly JournalRow[], - policy: JournalCompactionPolicy, - now: number -): JournalRow[] { - if (rows.length <= policy.minTailRows) { - return [...rows] - } - const floor = now - policy.retainTailMs - const byAge = rows.findIndex((row) => row.ts >= floor) - const byCount = rows.length - policy.minTailRows - const start = byAge === -1 ? byCount : Math.min(byAge, byCount) - if (policy.retention !== 'budget-pressure') { - return rows.slice(start) - } - // Halve rather than empty: the newer half keeps live clients resuming, and - // shedding at least one row guarantees the append that triggered this makes - // progress instead of latching the session read-only. - return rows.slice(Math.max(start, Math.ceil(rows.length / 2))) -} - -/** Only when the retention window would actually drop rows: inside it, - * compaction rewrites an identical log, and doing that per append is a full - * state serialization on the hot path. */ -export function journalTailIsReadyToCompact( - tailRows: readonly JournalRow[], - policy: JournalCompactionPolicy, - now: number -): boolean { - if (tailRows.length <= policy.minTailRows * 2) { - return false - } - return (tailRows[0]?.ts ?? now) < now - policy.retainTailMs -} - -/** The policy an append falls back to when the size bound would otherwise - * refuse it: both floors that normally protect the tail step aside. */ -export function budgetPressurePolicy(policy: JournalCompactionPolicy): JournalCompactionPolicy { - return { ...policy, minTailRows: 0, retention: 'budget-pressure' } -} - -/** Budget pressure may need to shed rows before the ordinary batching threshold. - * Pass a `budget-pressure` policy, or a tail wholly inside the retention - * window answers false and the size bound refuses every append until it ages - * out — two hours of a session the user cannot write to. */ -export function journalTailCanShedRows( - tailRows: readonly JournalRow[], - policy: JournalCompactionPolicy, - now: number -): boolean { - return retainTail(tailRows, policy, now).length < tailRows.length -} diff --git a/src/main/native-chat/agent-session-journal/journal-corruption-quarantine.ts b/src/main/native-chat/agent-session-journal/journal-corruption-quarantine.ts deleted file mode 100644 index e12d5b67a0a..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-corruption-quarantine.ts +++ /dev/null @@ -1,102 +0,0 @@ -// Corruption never deletes history. A journal that cannot be read end to end -// keeps its intact prefix live and moves the unreadable remainder aside, so the -// bytes stay on disk for inspection instead of being rebuilt into an empty epoch. - -import { readFile } from 'node:fs/promises' -import { join } from 'node:path' -import { - JOURNAL_SNAPSHOT_FILE, - quarantineJournalRemainder, - readJournalLog, - rewriteJournalLog -} from './journal-log-file' -import type { JournalRow } from './journal-row-schema' -import { assertJournalPhysicalCapacity } from './journal-physical-quota' - -/** Keep the readable prefix and set the unreadable suffix aside. */ -export async function quarantineCorruptSuffix( - journalDir: string, - retainedRows: readonly JournalRow[], - remainder: string | undefined, - quota?: { sessionId: string; maxBytes: number } -): Promise { - if (remainder) { - if (quota) { - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: Buffer.byteLength(remainder, 'utf8') - }) - } - await quarantineJournalRemainder(journalDir, remainder) - } - if (quota) { - const retainedBytes = retainedRows.reduce( - (total, row) => total + Buffer.byteLength(JSON.stringify(row), 'utf8') + 1, - 0 - ) - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: retainedBytes - }) - } - await rewriteJournalLog(journalDir, retainedRows) -} - -/** Copy everything aside before a read-only journal is rebuilt under a newer - * schema: those rows are unreadable to THIS build, not worthless. The - * snapshot is preserved as raw bytes — a future-version snapshot does not - * parse under this build's schema, and its bytes must survive verbatim. */ -export async function quarantineUnreadableSchema( - journalDir: string, - quota?: { sessionId: string; maxBytes: number } -): Promise { - const snapshot = await readSnapshotBytes(journalDir) - const log = await readJournalLog(journalDir) - const preserved = [ - snapshot ?? '', - log.rows.map((row) => JSON.stringify(row)).join('\n'), - log.remainder ?? '' - ] - .filter(Boolean) - .join('\n') - if (preserved) { - if (quota) { - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: Buffer.byteLength(preserved, 'utf8') - }) - } - await quarantineJournalRemainder(journalDir, preserved) - } -} - -async function readSnapshotBytes(journalDir: string): Promise { - try { - return await readFile(join(journalDir, JOURNAL_SNAPSHOT_FILE), 'utf-8') - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return null - } - throw error - } -} - -/** The disclosure row for lines that failed to parse. Skipped lines are lost - * rows; counting them silently is the drop this exists to prevent. */ -export function malformedRowsDisclosure(count: number): { - identity: { provider: 'orca'; clientMessageId: string } - body: { kind: 'status'; text: string } -} { - const plural = count === 1 ? '' : 's' - return { - // One stable identity, so a reopen upserts the same row instead of adding one. - identity: { provider: 'orca', clientMessageId: 'journal-malformed-lines' }, - body: { - kind: 'status', - text: `${count} journal line${plural} could not be read and ${count === 1 ? 'was' : 'were'} skipped` - } - } -} diff --git a/src/main/native-chat/agent-session-journal/journal-corruption-repair.test.ts b/src/main/native-chat/agent-session-journal/journal-corruption-repair.test.ts new file mode 100644 index 00000000000..978f076d6e9 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-corruption-repair.test.ts @@ -0,0 +1,265 @@ +// A repair drops what it cannot replay, and says so. +// +// Two things make a suffix unreplayable: a row this build cannot parse, and a +// sequence gap that makes every later row unanchored. The rejected suffix is +// DELETED; the load reports `corrupt`, and recovery rebuilds the epoch from +// provider history. Every case here asserts the same two halves: the live epoch +// holds only the replayable prefix, AND the epoch stays anchored so nothing +// replays a repaired journal as a clean timeline. + +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import type Database from '../../sqlite/sync-database' +import { openJournalDatabase } from './journal-database' +import { journalDatabaseFile } from './journal-paths' +import { parseJournalRow, type JournalRow } from './journal-row-schema' +import { loadJournal } from './journal-open' +import type { openAgentSessionJournal } from './journal-store-factory' +import { createTrackedJournalOpener } from './journal-store-test-open' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} + +let root: string +let clock = 1_000 +const journals = createTrackedJournalOpener() + +function tick(): number { + clock += 1 + return clock +} + +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } +} + +function body(value: string): AgentJournalItemBody { + return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } +} + +function open(overrides: Partial[0]> = {}) { + return journals.open({ + identity: IDENTITY, + journalDir: root, + now: tick, + mintEpoch: () => `epoch-${clock}`, + ...overrides + }) +} + +async function withJournalDatabase(run: (db: Database.Database) => void): Promise { + const opened = openJournalDatabase(journalDatabaseFile(root)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} + +/** The row replay anchors on, parsed exactly as replay parses it. */ +function firstLiveRow(): Promise { + let row: JournalRow | null = null + return withJournalDatabase((db) => { + const stored = db.prepare('SELECT row_json FROM journal_rows ORDER BY seq LIMIT 1').get() as + | { row_json: string } + | undefined + const parsed = stored ? parseJournalRow(stored.row_json) : null + row = parsed?.ok ? parsed.row : null + }).then(() => row) +} + +function liveSequences(): Promise { + let sequences: number[] = [] + return withJournalDatabase((db) => { + sequences = ( + db.prepare('SELECT seq FROM journal_rows ORDER BY seq').all() as { seq: number }[] + ).map((row) => row.seq) + }).then(() => sequences) +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-repair-')) + clock = 1_000 +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('a malformed row', () => { + it('keeps the readable prefix live and drops the rest of the epoch', async () => { + const journal = await open() + await journal.appendItem(item(0), body('readable'), { fence: 1 }) + await journal.appendItem(item(1), body('unreadable'), { fence: 1 }) + await journal.appendItem(item(2), body('after the fault'), { fence: 1 }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('UPDATE journal_rows SET row_json = ? WHERE seq = ?').run('{"not":"a row"}', 3) + }) + + const reopened = await open() + expect(reopened.repair.malformedRows).toBe(1) + // 1..2 is the surviving prefix; 3 is the disclosure the repair appends. + expect(await liveSequences()).toEqual([1, 2, 3]) + }) + + it('discloses the line it could not read', async () => { + const journal = await open() + await journal.appendItem(item(0), body('readable'), { fence: 1 }) + await journal.appendItem(item(1), body('later'), { fence: 1 }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('UPDATE journal_rows SET row_json = ? WHERE seq = ?').run('}{', 2) + }) + + const reopened = await open() + const disclosure = reopened + .snapshot() + .items.map((entry) => entry.body) + .find((entry) => entry.kind === 'status') + expect(disclosure).toMatchObject({ kind: 'status' }) + expect(disclosure && 'text' in disclosure ? disclosure.text : '').toContain( + '1 journal line could not be read' + ) + }) +}) + +describe('a sequence gap', () => { + it('drops every row after the hole and reports the epoch corrupt', async () => { + const journal = await open() + for (let ordinal = 0; ordinal < 5; ordinal += 1) { + await journal.appendItem(item(ordinal), body(`m${ordinal}`), { fence: 1 }) + } + await journal.close() + // Sequence 1 is the epoch row, so the items occupy 2..6. Removing 4 leaves + // 5 and 6 valid but unanchored. + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(4) + }) + + const reopened = await open() + expect(await liveSequences()).toEqual([1, 2, 3]) + expect(reopened.repair).toEqual({ malformedRows: 0 }) + expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([body('m0'), body('m1')]) + }) + + // The prefix survives, so there is no emptied epoch to re-anchor and — a gap + // costing no malformed row — no disclosure either. Without a durable marker + // the next probe reads a contiguous anchored prefix and calls it clean, and + // the rows the repair deleted are never asked for again. + it('still reports corrupt on the next probe, with the deleted suffix unrebuilt', async () => { + const journal = await open() + for (let ordinal = 0; ordinal < 5; ordinal += 1) { + await journal.appendItem(item(ordinal), body(`m${ordinal}`), { fence: 1 }) + } + await journal.close() + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(4) + }) + + const repaired = await open() + await repaired.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true }) + + // Same policy the emptied-epoch repair takes: a session that writes into the + // epoch owns it, and a later import must not replace rows the user has seen. + const writable = await open() + await writable.appendItem(item(9), body('typed after the repair'), { fence: 1 }) + await writable.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: false }) + }) + + // The disclosure is the repair talking about itself, not the session writing: + // counting it as content would retire the marker the instant it was raised. + it('is not settled by the repair disclosure it appends for a malformed row', async () => { + const journal = await open() + for (let ordinal = 0; ordinal < 3; ordinal += 1) { + await journal.appendItem(item(ordinal), body(`m${ordinal}`), { fence: 1 }) + } + await journal.close() + await withJournalDatabase((db) => { + db.prepare('UPDATE journal_rows SET row_json = ? WHERE seq = ?').run('}{', 3) + }) + + const repaired = await open() + expect(repaired.repair.malformedRows).toBe(1) + await repaired.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true }) + }) +}) + +describe('a missing epoch row', () => { + // Sequence 1 is the anchor for the whole epoch. Validating from the first row + // that HAPPENS to remain declares the leftovers contiguous, and replay then + // renders a repaired journal as a clean timeline. + it('rejects the whole surviving range rather than declaring it contiguous', async () => { + const journal = await open() + await journal.appendItem(item(0), body('anchor'), { fence: 1 }) + await journal.appendSubmission({ + clientMessageId: 'client-message-1', + payloadFingerprint: 'fingerprint-1', + body: { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'the user typed this' }] + }, + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'client-message-1', + state: 'accepted', + providerIdentity: item(1), + fence: 1 + }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(1) + }) + + const reopened = await open() + expect(reopened.repair).toEqual({ malformedRows: 0 }) + expect(reopened.snapshot().items).toEqual([]) + + // The epoch cannot be left row-less. An ordinary append would then take + // sequence 1, and replay would call that non-epoch row a clean timeline. + expect(await liveSequences()).toEqual([1]) + const anchor = await firstLiveRow() + expect(anchor).toMatchObject({ kind: 'epoch', reason: 'unreconcilable_prefix' }) + }) + + // The repair epoch is a placeholder for history it could not rebuild. Left + // clean it would end automatic recovery: the provider transcript is never + // consulted again and the dropped rows never come back. + it('keeps asking for provider history until the epoch has content of its own', async () => { + const journal = await open() + await journal.appendItem(item(0), body('anchor'), { fence: 1 }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(1) + }) + + const repaired = await open() + await repaired.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true }) + + // A session that writes into the epoch owns it: its own rows are not a + // repair placeholder, and a later import must not replace them. + const writable = await open() + await writable.appendItem(item(1), body('typed after the repair'), { fence: 1 }) + await writable.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: false }) + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts b/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts index c58f1b57bc8..a7f54a14a4c 100644 --- a/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts @@ -21,7 +21,7 @@ import { type ProviderHistoryItem, type ProviderHistoryWindow } from './journal-submission-reconciler' -import { openAgentSessionJournal } from './journal-store-factory' +import { createTrackedJournalOpener } from './journal-store-test-open' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -52,8 +52,10 @@ function userMessage(text: string): AgentJournalMessageItem { return { kind: 'message', role: 'user', blocks: [{ type: 'text', text }] } } +const journals = createTrackedJournalOpener() + async function open() { - return openAgentSessionJournal({ + return journals.open({ identity: IDENTITY, journalDir: root, now: tick, @@ -89,6 +91,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -195,14 +198,8 @@ describe('crash between provider accept and journal commit', () => { ).toEqual([]) }) - it('keeps the receipt after the row that minted it was compacted away', async () => { - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - now: tick, - mintEpoch: () => `epoch-${clock}`, - compaction: { minTailRows: 1, retainTailMs: 0 } - }) + it('keeps the receipt across a reopen', async () => { + const journal = await open() await journal.appendSubmission({ clientMessageId: 'cm_1', payloadFingerprint: digestPayload('kept'), @@ -215,7 +212,7 @@ describe('crash between provider accept and journal commit', () => { providerIdentity: ACCEPTED_IDENTITY, fence: 1 }) - await journal.compact() + await journal.close() const reopened = await open() expect(reopened.receiptFor('cm_1')?.providerItemId).toBe(agentJournalItemKey(ACCEPTED_IDENTITY)) diff --git a/src/main/native-chat/agent-session-journal/journal-cursor.ts b/src/main/native-chat/agent-session-journal/journal-cursor.ts index bdec8adfd19..bd43623ae49 100644 --- a/src/main/native-chat/agent-session-journal/journal-cursor.ts +++ b/src/main/native-chat/agent-session-journal/journal-cursor.ts @@ -77,7 +77,7 @@ export function findSequenceGap( export function readJournalSince( source: { state: { epoch: string; lastSequence: number; oldestSequence: number } - tailRows: readonly JournalRow[] + rowsAfter: (afterSequence: number) => JournalRow[] readOnly: boolean }, cursor: AgentJournalCursor, @@ -92,7 +92,7 @@ export function readJournalSince( } return { ok: true, - rows: source.tailRows.filter((row) => row.seq > resume.afterSequence), + rows: source.rowsAfter(resume.afterSequence), cursor: currentCursor() } } diff --git a/src/main/native-chat/agent-session-journal/journal-database-schema.ts b/src/main/native-chat/agent-session-journal/journal-database-schema.ts new file mode 100644 index 00000000000..37675fbcd8e --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-database-schema.ts @@ -0,0 +1,37 @@ +// Table shape for one session's journal database. +// +// `journal_rows` is the append-only log; `journal_sessions` is the derived +// projection, upserted in the SAME transaction as every row insert so the live +// epoch and the rows that belong to it can never disagree. `journal_repairs` +// carries at most one row per session: the standing demand for a rebuild a +// partial repair leaves behind (see journal-repair-marker.ts). + +/** DB shape version, carried in `PRAGMA user_version`. Independent of the row + * body version (`JournalRow.v`): a newer build can change either alone. + * v2 added `journal_repairs`; a build without it would replay a partially + * repaired journal as clean, so it must latch read-only rather than write. */ +export const JOURNAL_DB_SCHEMA_VERSION = 2 + +export function createJournalTablesSql(): string { + return ` +CREATE TABLE IF NOT EXISTS journal_rows ( + session_id TEXT NOT NULL, + epoch TEXT NOT NULL, + seq INTEGER NOT NULL, + ts INTEGER NOT NULL, + row_json TEXT NOT NULL, + PRIMARY KEY (session_id, epoch, seq) +); +CREATE TABLE IF NOT EXISTS journal_sessions ( + session_id TEXT PRIMARY KEY, + epoch TEXT NOT NULL, + updated_at INTEGER NOT NULL +); +CREATE TABLE IF NOT EXISTS journal_repairs ( + session_id TEXT PRIMARY KEY, + epoch TEXT NOT NULL, + content_from INTEGER NOT NULL, + repaired_at INTEGER NOT NULL +); +` +} diff --git a/src/main/native-chat/agent-session-journal/journal-database.test.ts b/src/main/native-chat/agent-session-journal/journal-database.test.ts new file mode 100644 index 00000000000..358c42c8d55 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-database.test.ts @@ -0,0 +1,201 @@ +import { mkdtemp, rm, stat } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import Database from '../../sqlite/sync-database' +import { + JOURNAL_BUSY_TIMEOUT_MS, + journalPragmaNumber, + openJournalDatabase +} from './journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' +import { journalDatabaseFile } from './journal-paths' +import { + deleteAllJournalRows, + deleteJournalRowSuffix, + insertJournalRow, + readJournalEpochRows, + readJournalRowsAfter, + readJournalSessionEpoch, + upsertJournalSessionRow +} from './journal-row-table' +import type { JournalRow } from './journal-row-schema' +import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' + +let root: string +let dbPath: string + +function epochRow(seq: number, epoch = 'epoch-1'): JournalRow { + return { + kind: 'epoch', + reason: 'session_created', + providerHandle: { kind: 'codex', threadId: 'thread-1' }, + v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + epoch, + seq, + fence: 0, + ts: 1_700_000_000_000 + seq + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-db-')) + dbPath = journalDatabaseFile(root) +}) + +afterEach(async () => { + vi.restoreAllMocks() + await rm(root, { recursive: true, force: true }) +}) + +describe('journal database open', () => { + it('creates both tables and reads back every load-bearing pragma', () => { + const opened = openJournalDatabase(dbPath) + try { + const tables = opened.db + .prepare("SELECT name FROM sqlite_master WHERE type = 'table' ORDER BY name") + .all() + .map((entry) => (entry as { name: string }).name) + expect(tables).toContain('journal_rows') + expect(tables).toContain('journal_sessions') + expect(opened.db.pragma('journal_mode', { simple: true })).toBe('wal') + expect(journalPragmaNumber(opened.db, 'synchronous')).toBe(2) + expect(journalPragmaNumber(opened.db, 'busy_timeout')).toBe(JOURNAL_BUSY_TIMEOUT_MS) + expect(journalPragmaNumber(opened.db, 'foreign_keys')).toBe(1) + expect(journalPragmaNumber(opened.db, 'user_version')).toBe(JOURNAL_DB_SCHEMA_VERSION) + expect(opened.readOnly).toBe(false) + } finally { + opened.db.close() + } + }) + + it('latches read-only on a future user_version without touching the file', async () => { + const seeded = openJournalDatabase(dbPath) + upsertJournalSessionRow(seeded.db, 'session-1', 'epoch-1', 1) + insertJournalRow(seeded.db, 'session-1', epochRow(1)) + seeded.db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 5}`) + seeded.db.close() + const before = await stat(dbPath) + + const latched = openJournalDatabase(dbPath) + try { + expect(latched.readOnly).toBe(true) + expect(journalPragmaNumber(latched.db, 'user_version')).toBe(JOURNAL_DB_SCHEMA_VERSION + 5) + expect(readJournalEpochRows(latched.db, 'session-1', 'epoch-1')).toHaveLength(1) + expect(() => latched.db.exec("INSERT INTO journal_sessions VALUES ('x', 'y', 1)")).toThrow() + } finally { + latched.db.close() + } + expect((await stat(dbPath)).size).toBe(before.size) + expect(journalPragmaNumber(openJournalDatabase(dbPath).db, 'user_version')).toBe( + JOURNAL_DB_SCHEMA_VERSION + 5 + ) + }) + + // Site 1: the raw connection is owned by the open call until it returns. + it('closes the raw connection when schema setup throws', async () => { + const failing = join(root, 'nested', 'journal.db') + expect(() => openJournalDatabase(failing)).toThrow() + await expect(stat(`${failing}-wal`)).rejects.toThrow() + await expect(rm(root, { recursive: true, force: true })).resolves.toBeUndefined() + root = await mkdtemp(join(tmpdir(), 'orca-journal-db-')) + }) +}) + +describe('journal row statements', () => { + it('serves replay, resume, discard and suffix truncation from the primary key', () => { + const opened = openJournalDatabase(dbPath) + try { + const { db } = opened + db.exec('BEGIN IMMEDIATE') + for (let seq = 1; seq <= 5; seq += 1) { + insertJournalRow(db, 'session-1', epochRow(seq)) + } + insertJournalRow(db, 'session-1', epochRow(1, 'epoch-old')) + upsertJournalSessionRow(db, 'session-1', 'epoch-1', 42) + db.exec('COMMIT') + + expect(readJournalSessionEpoch(db, 'session-1')).toBe('epoch-1') + expect(readJournalSessionEpoch(db, 'absent')).toBeNull() + expect(readJournalEpochRows(db, 'session-1', 'epoch-1').map((row) => row.seq)).toEqual([ + 1, 2, 3, 4, 5 + ]) + expect(readJournalRowsAfter(db, 'session-1', 'epoch-1', 3).map((row) => row.seq)).toEqual([ + 4, 5 + ]) + + // The rejected suffix leaves `journal_rows`, scoped to its own epoch. + expect(deleteJournalRowSuffix(db, 'session-1', 'epoch-1', 4)).toBe(2) + expect(readJournalEpochRows(db, 'session-1', 'epoch-1').map((row) => row.seq)).toEqual([ + 1, 2, 3 + ]) + expect(readJournalEpochRows(db, 'session-1', 'epoch-old')).toHaveLength(1) + + deleteAllJournalRows(db) + expect(readJournalEpochRows(db, 'session-1', 'epoch-1')).toHaveLength(0) + expect(readJournalEpochRows(db, 'session-1', 'epoch-old')).toHaveLength(0) + expect(readJournalSessionEpoch(db, 'session-1')).toBe('epoch-1') + } finally { + opened.db.close() + } + }) + + it('refuses a duplicate sequence inside one epoch', () => { + const opened = openJournalDatabase(dbPath) + try { + insertJournalRow(opened.db, 'session-1', epochRow(1)) + expect(() => insertJournalRow(opened.db, 'session-1', epochRow(1))).toThrow() + insertJournalRow(opened.db, 'session-1', epochRow(1, 'epoch-2')) + } finally { + opened.db.close() + } + }) + + it('upserts the session projection in place', () => { + const opened = openJournalDatabase(dbPath) + try { + upsertJournalSessionRow(opened.db, 'session-1', 'epoch-1', 1) + upsertJournalSessionRow(opened.db, 'session-1', 'epoch-2', 2) + expect(readJournalSessionEpoch(opened.db, 'session-1')).toBe('epoch-2') + expect( + opened.db.prepare('SELECT count(*) AS total FROM journal_sessions').get() + ).toMatchObject({ total: 1 }) + } finally { + opened.db.close() + } + }) +}) + +describe('schema creation', () => { + // Creating the tables outside the migration transaction left a v2-shaped + // database still reporting version 0, which an older build does not latch + // read-only: it stamps its own version on and writes through v1 SQL. + it('publishes no table until the version bump commits with it', () => { + const original = Database.prototype.pragma + const pragma = vi.spyOn(Database.prototype, 'pragma').mockImplementation(function ( + this: Database.Database, + sql: string, + options?: { simple?: boolean } + ) { + if (sql.startsWith('user_version =')) { + throw new Error('crash before the version is published') + } + return original.call(this, sql, options) + }) + + expect(() => openJournalDatabase(dbPath)).toThrow('crash before the version is published') + pragma.mockRestore() + + const inspected = new Database(dbPath) + try { + expect(inspected.pragma('user_version', { simple: true })).toBe(0) + expect( + inspected + .prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'journal_rows'") + .get() + ).toBeUndefined() + } finally { + inspected.close() + } + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-database.ts b/src/main/native-chat/agent-session-journal/journal-database.ts new file mode 100644 index 00000000000..6f1cd60733b --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-database.ts @@ -0,0 +1,80 @@ +// Opening one session's journal database. +// +// `PRAGMA user_version` is read FIRST, on a connection that has set no +// persistent pragma and run no DDL: a future-schema database must be left +// byte-identical, and `journal_mode = WAL` writes the file header. + +import Database from '../../sqlite/sync-database' +import { hardenSqliteDatabaseFiles } from '../../sqlite/harden-database-files' +import { createJournalTablesSql, JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' + +export const JOURNAL_BUSY_TIMEOUT_MS = 5000 + +export type OpenJournalDatabase = { + db: Database.Database + /** A newer `user_version` was met: this build reads and never writes. */ + readOnly: boolean +} + +export function journalPragmaNumber(db: Database.Database, name: string): number { + return Number(db.pragma(name, { simple: true }) ?? 0) +} + +export function openJournalDatabase(dbPath: string): OpenJournalDatabase { + const probe = new Database(dbPath) + let stored: number + try { + stored = journalPragmaNumber(probe, 'user_version') + } catch (error) { + probe.close() + throw error + } + if (stored > JOURNAL_DB_SCHEMA_VERSION) { + probe.close() + return { db: new Database(dbPath, { readonly: true, fileMustExist: true }), readOnly: true } + } + let transferred = false + try { + configureJournalPragmas(probe) + createJournalSchema(probe, stored) + hardenSqliteDatabaseFiles(dbPath) + const opened = { db: probe, readOnly: false } + transferred = true + return opened + } finally { + if (!transferred) { + probe.close() + } + } +} + +function configureJournalPragmas(db: Database.Database): void { + db.pragma('journal_mode = WAL') + db.pragma(`busy_timeout = ${JOURNAL_BUSY_TIMEOUT_MS}`) + db.pragma('foreign_keys = ON') + // Why FULL rather than the house NORMAL: the write-ahead submission row must + // survive a power loss before the adapter dispatches anything, and NORMAL in + // WAL mode does not fsync at commit. + db.pragma('synchronous = FULL') +} + +/** + * Table creation and the `user_version` bump are ONE transaction. Creating the + * tables first left a shaped database still reporting version 0, which an older + * build does not latch read-only: it stamped its own version on and wrote + * through SQL for a schema it did not have. + */ +function createJournalSchema(db: Database.Database, stored: number): void { + if (stored >= JOURNAL_DB_SCHEMA_VERSION) { + return + } + db.exec('BEGIN IMMEDIATE') + try { + db.exec(createJournalTablesSql()) + db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION}`) + db.exec('COMMIT') + } catch (error) { + db.exec('ROLLBACK') + throw error + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts b/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts index 087ac9e6a26..f428675fb2c 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts @@ -2,28 +2,21 @@ import type { AgentJournalCursor, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import type { JournalCompactionPolicy } from './journal-compaction' -import { quarantineUnreadableSchema } from './journal-corruption-quarantine' +import type Database from '../../sqlite/sync-database' import { replaceJournalEpoch, type JournalReplacementItem } from './journal-epoch-replacement' import { publishNewEpoch } from './journal-epoch-rollover' import type { JournalLoad } from './journal-open' import type { AgentJournalEpochReason } from './journal-row-schema' -import { - assertJournalFence, - assertJournalWritable, - type JournalAppendBudget -} from './journal-write-guards' +import { assertJournalFence, assertJournalWritable } from './journal-write-guards' export class JournalEpochController { constructor( private readonly deps: { identity: AgentSessionJournalIdentity - journalDir: string - budget: JournalAppendBudget - compaction: JournalCompactionPolicy now: () => number mintEpoch: () => string serialize: (run: () => Promise) => Promise + database: () => { db: Database.Database } readOnly: () => boolean setReadOnly: (readOnly: boolean) => void highestFence: () => number @@ -32,33 +25,33 @@ export class JournalEpochController { } ) {} - async start(reason: AgentJournalEpochReason, fence: number): Promise { - this.deps.adopt( - await publishNewEpoch({ - journalDir: this.deps.journalDir, - sessionId: this.deps.identity.sessionId, - providerHandle: this.deps.identity.providerHandle, - epoch: this.deps.mintEpoch(), - reason, - fence, - now: this.deps.now(), - maxSessionBytes: this.deps.budget.maxSessionBytes - }) - ) + start(reason: AgentJournalEpochReason, fence: number): void { + publishNewEpoch({ + db: this.deps.database().db, + sessionId: this.deps.identity.sessionId, + providerHandle: this.deps.identity.providerHandle, + epoch: this.deps.mintEpoch(), + reason, + fence, + now: this.deps.now(), + onPublished: this.deps.adopt + }) } - async roll(reason: AgentJournalEpochReason, fence: number): Promise { - if (reason !== 'schema_unreadable') { + /** + * Every reason takes the same writable guard. A latched store refuses a roll + * like any other write, and `schema_unreadable` has no production caller. + * + * Serialized like every other write, so the discard cannot land between an + * admitted append's sequence assignment and its commit. + */ + roll(reason: AgentJournalEpochReason, fence: number): Promise { + return this.deps.serialize(async () => { assertJournalWritable(this.deps.readOnly(), this.deps.identity.sessionId) - } else if (this.deps.readOnly()) { - await quarantineUnreadableSchema(this.deps.journalDir, { - sessionId: this.deps.identity.sessionId, - maxBytes: this.deps.budget.maxSessionBytes - }) - } - await this.start(reason, fence) - this.deps.setReadOnly(false) - return this.deps.cursor() + this.start(reason, fence) + this.deps.setReadOnly(false) + return this.deps.cursor() + }) } replace( @@ -69,17 +62,15 @@ export class JournalEpochController { return this.deps.serialize(async () => { assertJournalWritable(this.deps.readOnly(), this.deps.identity.sessionId) assertJournalFence(fence, this.deps.highestFence()) - await replaceJournalEpoch({ - journalDir: this.deps.journalDir, + replaceJournalEpoch({ + db: this.deps.database().db, identity: this.deps.identity, reason, fence, items, - budget: this.deps.budget.fork(), - compaction: this.deps.compaction, now: this.deps.now, mintEpoch: this.deps.mintEpoch, - onSnapshotPublished: this.deps.adopt + onPublished: this.deps.adopt }) return this.deps.cursor() }) diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts index d4036755fda..96273366d5f 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts @@ -1,18 +1,19 @@ -import { mkdtemp, readdir, rm } from 'node:fs/promises' +// Republishing an epoch is ONE transaction. + +import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' import type { - AgentJournalItemBody, + AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { DEFAULT_JOURNAL_COMPACTION_POLICY } from './journal-compaction' +import { openJournalDatabase, type OpenJournalDatabase } from './journal-database' import { replaceJournalEpoch } from './journal-epoch-replacement' -import { putJournalBlob, readJournalBlob } from './journal-blob-store' -import { boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalDirectoryBytes } from './journal-physical-quota' -import { JournalAppendBudget } from './journal-write-guards' -import { openAgentSessionJournal } from './journal-store-factory' +import type { JournalLoad } from './journal-open' +import { journalDatabaseFile } from './journal-paths' +import { readJournalEpochRows, readJournalSessionEpoch } from './journal-row-table' +import { createTrackedJournalOpener } from './journal-store-test-open' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -24,150 +25,79 @@ const IDENTITY: AgentSessionJournalIdentity = { let root: string let clock = 1_000 - -beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-replace-')) - clock = 1_000 -}) - -afterEach(async () => { - await rm(root, { recursive: true, force: true }) -}) +let database: OpenJournalDatabase +const journals = createTrackedJournalOpener() function now(): number { clock += 1 return clock } -function toolBody(output: ReturnType): AgentJournalItemBody { - return { - kind: 'tool-call', - name: 'shell', - input: {}, - state: 'completed', - output - } +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } } -describe('journal epoch replacement', () => { - it('publishes one observable replacement and prunes stale root blobs afterward', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 8 } - const stalePayload = 'stale'.repeat(1_000) - const retainedPayload = 'retained'.repeat(1_000) - const stale = boundPayload(stalePayload, limits) - const retained = boundPayload(retainedPayload, limits) - const published: unknown[] = [] - await putJournalBlob(root, stale.digest, stalePayload) +function replace(input: { + items: Parameters[0]['items'] + onPublished?: (loaded: JournalLoad) => void +}): void { + replaceJournalEpoch({ + db: database.db, + identity: IDENTITY, + reason: 'legacy_import', + fence: 1, + items: input.items, + now, + mintEpoch: () => `epoch-${clock}`, + onPublished: input.onPublished ?? (() => undefined) + }) +} - await replaceJournalEpoch({ - journalDir: root, - identity: IDENTITY, - reason: 'handle_forked', - fence: 2, - items: [ - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 }, - body: toolBody(retained), - blobs: [{ digest: retained.digest, payload: retainedPayload }] - } - ], - budget: new JournalAppendBudget(IDENTITY.sessionId, { - ...limits, - maxSessionBytes: 512 * 1024 - }), - compaction: DEFAULT_JOURNAL_COMPACTION_POLICY, - now, - mintEpoch: () => 'epoch-new', - onSnapshotPublished: (loaded) => published.push(loaded) +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-replace-')) + clock = 1_000 + database = openJournalDatabase(journalDatabaseFile(root)) +}) + +afterEach(async () => { + try { + database.db.close() + } catch { + // Already closed by the case. + } + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('journal epoch replacement', () => { + it('publishes one observable replacement', () => { + const published: JournalLoad[] = [] + + replace({ + items: [{ identity: item(1), body: { kind: 'status', text: 'republished' } }], + onPublished: (loaded) => published.push(loaded) }) expect(published).toHaveLength(1) - expect(await readJournalBlob(root, stale.digest)).toBeNull() - expect(await readJournalBlob(root, retained.digest)).toBe(retainedPayload) - expect((published[0] as { sizeBytes: number }).sizeBytes).toBe( - await journalDirectoryBytes(root) - ) + const epoch = readJournalSessionEpoch(database.db, IDENTITY.sessionId) + expect(epoch).toBe(published[0]?.state.epoch) + expect(readJournalEpochRows(database.db, IDENTITY.sessionId, epoch ?? '')).toHaveLength(2) }) - it('keeps root blobs and reports no publication when replacement never becomes authoritative', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 8, maxSessionBytes: 6_000 } - const stalePayload = 'stale'.repeat(500) - const stale = boundPayload(stalePayload, limits) - const published: unknown[] = [] - await putJournalBlob(root, stale.digest, stalePayload) + it('discards every superseded row in the same transaction', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.appendItem(item(1), { kind: 'status', text: 'old' }, { fence: 1 }) + await journal.appendItem(item(2), { kind: 'status', text: 'older' }, { fence: 1 }) + const before = journal.epoch - await expect( - replaceJournalEpoch({ - journalDir: root, - identity: IDENTITY, - reason: 'handle_forked', - fence: 2, - items: [ - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 }, - body: { - kind: 'message', - role: 'assistant', - blocks: [{ type: 'text', text: 'x'.repeat(10_000) }] - } - } - ], - budget: new JournalAppendBudget(IDENTITY.sessionId, limits), - compaction: DEFAULT_JOURNAL_COMPACTION_POLICY, - now, - mintEpoch: () => 'epoch-new', - onSnapshotPublished: (loaded) => published.push(loaded) - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) + await journal.replaceEpochItems('legacy_import', 1, [ + { identity: item(9), body: { kind: 'status', text: 'republished' } } + ]) - expect(published).toHaveLength(0) - expect(await readJournalBlob(root, stale.digest)).toBe(stalePayload) - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - }) - - it('charges replacement blobs cumulatively and rolls back staging on quota refusal', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 8, maxSessionBytes: 7_000 } - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits, - autoCompact: false, - now, - mintEpoch: () => `epoch-${clock}` - }) - const existingPayload = 'existing'.repeat(250) - const existing = boundPayload(existingPayload, limits) - await journal.appendItemWithBlobs( - { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 }, - toolBody(existing), - [{ digest: existing.digest, payload: existingPayload }], - { fence: 1 } - ) - - const replacementPayload = 'replacement'.repeat(200) - const replacement = boundPayload(replacementPayload, limits) - const secondPayload = 'second'.repeat(200) - const second = boundPayload(secondPayload, limits) - await expect( - journal.replaceEpochItems('handle_forked', 2, [ - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 1 }, - body: toolBody(replacement), - blobs: [{ digest: replacement.digest, payload: replacementPayload }] - }, - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 2 }, - body: toolBody(second), - blobs: [{ digest: second.digest, payload: secondPayload }] - } - ]) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toMatch(/^epoch-/) - expect(await readJournalBlob(root, existing.digest)).toBe(existingPayload) - expect(await readJournalBlob(root, replacement.digest)).toBeNull() - expect(await readJournalBlob(root, second.digest)).toBeNull() - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) + expect(journal.epoch).not.toBe(before) + expect(readJournalEpochRows(database.db, IDENTITY.sessionId, before)).toHaveLength(0) + expect(journal.snapshot().items.map((entry) => entry.body)).toEqual([ + { kind: 'status', text: 'republished' } + ]) }) }) diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts index 342c60c4f1d..9512d5e9fa1 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts @@ -1,298 +1,85 @@ -import { mkdir, mkdtemp, rm, stat } from 'node:fs/promises' -import { join } from 'node:path' -import { copyFileDurable } from '../../durable-file-write' +// Republishing a live item set into a fresh epoch. +// +// One transaction: discard every row, insert the epoch row plus the replacement +// items, move the session projection, and retire any repair marker — this +// republished history is exactly what the marker was holding out for. + import type { AgentJournalItemBody, AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { compactJournal, type JournalCompactionPolicy } from './journal-compaction' -import { JOURNAL_LOG_FILE, JOURNAL_SNAPSHOT_FILE, appendJournalRows } from './journal-log-file' -import { - applyJournalRow, - blobDigestsInBody, - createJournalReducerState, - referencedBlobDigests, - type JournalReducerState -} from './journal-reducer' -import { buildJournalItemRow, journalRowBase } from './journal-row-builders' -import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -import { journalRowByteLength } from './journal-row-schema' -import { assertJournalFence, type JournalAppendBudget } from './journal-write-guards' +import type Database from '../../sqlite/sync-database' import type { JournalLoad } from './journal-open' -import { assertJournalPhysicalCapacity, journalDirectoryBytes } from './journal-physical-quota' +import { clearJournalRepairMarker } from './journal-repair-marker' +import { applyJournalRow, createJournalReducerState } from './journal-reducer' +import { buildJournalItemRow, journalRowBase } from './journal-row-builders' import { - JOURNAL_BLOB_DIR, - journalBlobFileSize, - putJournalBlob, - pruneJournalBlobs, - removeJournalBlob -} from './journal-blob-store' + deleteAllJournalRows, + insertJournalRow, + upsertJournalSessionRow +} from './journal-row-table' +import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' +import { assertJournalFence } from './journal-write-guards' export type JournalReplacementItem = { identity: AgentJournalItemIdentity body: AgentJournalItemBody - blobs?: readonly { digest: string; payload: string }[] observedAt?: number } -export async function replaceJournalEpoch(input: { - journalDir: string +export function replaceJournalEpoch(input: { + db: Database.Database identity: AgentSessionJournalIdentity reason: AgentJournalEpochReason fence: number items: readonly JournalReplacementItem[] - budget: JournalAppendBudget - compaction: JournalCompactionPolicy now: () => number mintEpoch: () => string - onSnapshotPublished: (loaded: JournalLoad) => void -}): Promise { - const stagingDir = await mkdtemp(join(input.journalDir, '.epoch-replacement-')) - const stagedBlobDigests = new Set() - const publishedBlobDigests: string[] = [] - let snapshotPublished = false - let adoptionReported = false - let publishedLoad: JournalLoad | null = null + /** Called the instant the transaction commits, before any fallible follow-up. */ + onPublished: (loaded: JournalLoad) => void +}): void { + const epoch = input.mintEpoch() + const state = createJournalReducerState(input.identity.sessionId, epoch) + const epochRow: JournalRow = { + kind: 'epoch', + reason: input.reason, + providerHandle: input.identity.providerHandle, + ...journalRowBase(epoch, 1, input.fence, input.now()) + } + const rows: JournalRow[] = [epochRow] + applyJournalRow(state, epochRow) + for (const item of input.items) { + const row = buildJournalItemRow({ + state, + identity: item.identity, + body: item.body, + seq: state.lastSequence + 1, + fence: input.fence, + ts: item.observedAt ?? input.now() + }) + assertJournalFence(row.fence, state.highestFence) + applyJournalRow(state, row) + rows.push(row) + } + + input.db.exec('BEGIN IMMEDIATE') try { - const epoch = input.mintEpoch() - const state = createJournalReducerState(input.identity.sessionId, epoch) - const epochRow: JournalRow = { - kind: 'epoch', - reason: input.reason, - providerHandle: input.identity.providerHandle, - ...journalRowBase(epoch, 1, input.fence, input.now()) - } - const rows: JournalRow[] = [epochRow] - applyJournalRow(state, epochRow) - let sizeBytes = journalRowByteLength(epochRow) - await assertStagingCapacity(input, sizeBytes) - await appendJournalRows(stagingDir, [epochRow]) - - for (const item of input.items) { - sizeBytes += await stageReplacementBlobs({ - journalDir: input.journalDir, - stagingDir, - identity: input.identity, - budget: input.budget, - stagedBlobDigests, - blobs: item.blobs ?? [] - }) - const appendTime = input.now() - const row = buildJournalItemRow({ - state, - identity: item.identity, - body: item.body, - seq: state.lastSequence + 1, - fence: input.fence, - ts: item.observedAt ?? appendTime - }) - assertJournalFence(row.fence, state.highestFence) - input.budget.assert(row, appendTime, sizeBytes) - await assertStagingCapacity(input, journalRowByteLength(row)) - await appendJournalRows(stagingDir, [row]) - applyJournalRow(state, row) - rows.push(row) - sizeBytes += journalRowByteLength(row) - } - - const compacted = await compactJournal({ - journalDir: stagingDir, - physicalQuotaRoot: input.journalDir, - state, - tailRows: rows, - policy: input.compaction, - now: input.now(), - maxSessionBytes: input.budget.maxSessionBytes - }) - // All destination publishes use durable temp files while the staging - // source and existing finals remain present. Reserve the whole publication - // peak before touching the live epoch so a later file cannot fail halfway - // through replacement. - const stagedSnapshotBytes = (await stat(join(stagingDir, JOURNAL_SNAPSHOT_FILE))).size - const stagedLogBytes = (await stat(join(stagingDir, JOURNAL_LOG_FILE))).size - let stagedPublishBytes = stagedSnapshotBytes + stagedLogBytes - for (const digest of stagedBlobDigests) { - stagedPublishBytes += (await stat(join(stagingDir, JOURNAL_BLOB_DIR, digest))).size - } - await assertJournalPhysicalCapacity({ - journalDir: input.journalDir, - sessionId: input.identity.sessionId, - maxBytes: input.budget.maxSessionBytes, - peakAdditionalBytes: stagedPublishBytes - }) - for (const digest of stagedBlobDigests) { - if ( - await publishPreparedBlob( - stagingDir, - input.journalDir, - digest, - input.identity.sessionId, - input.budget.maxSessionBytes - ) - ) { - publishedBlobDigests.push(digest) - } - } - await publishPreparedFile( - stagingDir, - input.journalDir, - JOURNAL_SNAPSHOT_FILE, - input.identity.sessionId, - input.budget.maxSessionBytes - ) - snapshotPublished = true - state.oldestSequence = compacted.oldestSequence - publishedLoad = { - state, - tailRows: compacted.tailRows, - compactedThrough: compacted.compactedThrough, - readOnly: false, - corrupt: false, - malformedRows: 0, - sizeBytes: 0 - } - await publishPreparedFile( - stagingDir, - input.journalDir, - JOURNAL_LOG_FILE, - input.identity.sessionId, - input.budget.maxSessionBytes - ) - await pruneJournalBlobs( - input.journalDir, - replacementRetainedBlobDigests(state, compacted.tailRows) - ) - } finally { - if (!snapshotPublished) { - for (const digest of publishedBlobDigests) { - await removeJournalBlob(input.journalDir, digest) - } - } - await rm(stagingDir, { recursive: true, force: true }) - if (snapshotPublished && publishedLoad && !adoptionReported) { - adoptionReported = true - input.onSnapshotPublished({ - ...publishedLoad, - sizeBytes: await journalDirectoryBytes(input.journalDir) - }) + deleteAllJournalRows(input.db) + clearJournalRepairMarker(input.db, input.identity.sessionId) + for (const row of rows) { + insertJournalRow(input.db, input.identity.sessionId, row) } + upsertJournalSessionRow(input.db, input.identity.sessionId, epoch, epochRow.ts) + input.db.exec('COMMIT') + } catch (error) { + input.db.exec('ROLLBACK') + throw error } -} -function replacementRetainedBlobDigests( - state: JournalReducerState, - tailRows: readonly JournalRow[] -): Set { - const retained = referencedBlobDigests(state) - for (const row of tailRows) { - if (row.kind === 'item') { - blobDigestsInBody(row.body, retained) - } else if (row.kind === 'lifecycle-batch') { - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - blobDigestsInBody(mutation.body, retained) - } - } - } - } - return retained -} - -async function stageReplacementBlobs(input: { - journalDir: string - stagingDir: string - identity: AgentSessionJournalIdentity - budget: JournalAppendBudget - stagedBlobDigests: Set - blobs: readonly { digest: string; payload: string }[] -}): Promise { - const toStage: { digest: string; payload: string; bytes: number }[] = [] - const unique = new Map(input.blobs.map((blob) => [blob.digest, blob])) - for (const blob of unique.values()) { - if (input.stagedBlobDigests.has(blob.digest)) { - continue - } - if ((await journalBlobFileSize(input.journalDir, blob.digest)) !== null) { - continue - } - const bytes = Buffer.byteLength(blob.payload, 'utf8') - toStage.push({ ...blob, bytes }) - } - // Reserve all new payloads together. The staging directory lives under the - // journal root, so the capacity check includes existing session bytes and - // every other .epoch-replacement-* directory already present. - const stagedBytes = toStage.reduce((total, blob) => total + blob.bytes, 0) - await assertStagingCapacity(input, stagedBytes) - for (const blob of toStage) { - await putJournalBlob(input.stagingDir, blob.digest, blob.payload) - input.stagedBlobDigests.add(blob.digest) - } - return stagedBytes -} - -function assertStagingCapacity( - input: { - journalDir: string - identity: AgentSessionJournalIdentity - budget: JournalAppendBudget - }, - additionalBytes: number -): Promise { - return assertJournalPhysicalCapacity({ - journalDir: input.journalDir, - sessionId: input.identity.sessionId, - maxBytes: input.budget.maxSessionBytes, - peakAdditionalBytes: additionalBytes - }) -} - -async function publishPreparedFile( - stagingDir: string, - journalDir: string, - fileName: string, - sessionId: string, - maxBytes: number -): Promise { - await assertJournalPhysicalCapacity({ - journalDir, - sessionId, - maxBytes, - peakAdditionalBytes: (await stat(join(stagingDir, fileName))).size - }) - const copied = await copyFileDurable(join(stagingDir, fileName), join(journalDir, fileName)) - if (!copied) { - throw new Error(`prepared journal file disappeared before publish: ${fileName}`) - } -} - -async function publishPreparedBlob( - stagingDir: string, - journalDir: string, - digest: string, - sessionId: string, - maxBytes: number -): Promise { - if ((await journalBlobFileSize(journalDir, digest)) !== null) { - return false - } - const size = await journalBlobFileSize(stagingDir, digest) - if (size === null) { - throw new Error(`prepared journal blob disappeared before publish: ${digest}`) - } - await assertJournalPhysicalCapacity({ - journalDir, - sessionId, - maxBytes, - peakAdditionalBytes: size - }) - await mkdir(join(journalDir, JOURNAL_BLOB_DIR), { recursive: true }) - const copied = await copyFileDurable( - join(stagingDir, JOURNAL_BLOB_DIR, digest), - join(journalDir, JOURNAL_BLOB_DIR, digest) - ) - if (!copied) { - throw new Error(`prepared journal blob disappeared before publish: ${digest}`) - } - return true + // COMMIT landed: on disk the superseded rows are gone and this epoch is the + // live one. The caller adopts that immediately, or a later failure leaves the + // live store writing into an epoch whose rows were just deleted. + state.oldestSequence = 1 + input.onPublished({ state, readOnly: false, corrupt: false, malformedRows: 0 }) } diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts b/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts index 40e1bc3a542..e95ff821058 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts @@ -1,29 +1,34 @@ // Opening a new epoch. // -// The snapshot is what names the live epoch, so it is published BEFORE the log -// is reset. A crash mid-rollover therefore leaves stale-epoch rows behind the -// new snapshot, which `loadJournal` drops — the reverse order would leave a -// journal whose log no longer matches any epoch anyone can name. +// One transaction: discard every row of the superseded epoch, insert the new +// epoch row at sequence 1, move the session projection onto it, and retire any +// repair marker the superseded epoch was carrying. Superseded rows are DELETED +// rather than retained — nothing would ever shed them. import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' import type { AgentSessionProviderHandle } from '../../../shared/agent-session-journal-types' -import { compactJournal } from './journal-compaction' -import { applyJournalRow, createJournalReducerState } from './journal-reducer' -import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -import { journalRowByteLength } from './journal-row-schema' +import type Database from '../../sqlite/sync-database' import type { JournalLoad } from './journal-open' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' +import { clearJournalRepairMarker } from './journal-repair-marker' +import { applyJournalRow, createJournalReducerState } from './journal-reducer' +import { + deleteAllJournalRows, + insertJournalRow, + upsertJournalSessionRow +} from './journal-row-table' +import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -export async function publishNewEpoch(input: { - journalDir: string +export function publishNewEpoch(input: { + db: Database.Database sessionId: string providerHandle: AgentSessionProviderHandle epoch: string reason: AgentJournalEpochReason fence: number now: number - maxSessionBytes?: number -}): Promise { + /** Called the instant the transaction commits, before any fallible follow-up. */ + onPublished: (loaded: JournalLoad) => void +}): void { const row: JournalRow = { kind: 'epoch', reason: input.reason, @@ -34,25 +39,24 @@ export async function publishNewEpoch(input: { fence: input.fence, ts: input.now } + + input.db.exec('BEGIN IMMEDIATE') + try { + deleteAllJournalRows(input.db) + clearJournalRepairMarker(input.db, input.sessionId) + insertJournalRow(input.db, input.sessionId, row) + upsertJournalSessionRow(input.db, input.sessionId, input.epoch, input.now) + input.db.exec('COMMIT') + } catch (error) { + input.db.exec('ROLLBACK') + throw error + } + + // COMMIT landed: on disk the superseded prefix is gone and this epoch is the + // live one. The caller adopts that immediately, or a later failure leaves the + // store writing into an epoch that no longer exists. const state = createJournalReducerState(input.sessionId, input.epoch) - await compactJournal({ - journalDir: input.journalDir, - state, - tailRows: [row], - policy: { minTailRows: 1, retainTailMs: Number.POSITIVE_INFINITY }, - now: input.now, - maxSessionBytes: input.maxSessionBytes ?? DEFAULT_JOURNAL_PAYLOAD_LIMITS.maxSessionBytes, - sessionId: input.sessionId - }) applyJournalRow(state, row) state.oldestSequence = 1 - return { - state, - tailRows: [row], - compactedThrough: 0, - readOnly: false, - corrupt: false, - malformedRows: 0, - sizeBytes: journalRowByteLength(row) - } + input.onPublished({ state, readOnly: false, corrupt: false, malformedRows: 0 }) } diff --git a/src/main/native-chat/agent-session-journal/journal-handle-ownership.test.ts b/src/main/native-chat/agent-session-journal/journal-handle-ownership.test.ts new file mode 100644 index 00000000000..592f7471c5c --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-handle-ownership.test.ts @@ -0,0 +1,161 @@ +// Every path that can open a SQLite connection releases it. +// +// Asserting that the happy path closes cleanly proves nothing: these sites are +// reached only when something has already gone wrong. On POSIX a leak is +// SILENT — the unlink succeeds — so each case asserts BOTH that the sidecars +// are gone and that the directory renames and removes, which is the half that +// actually fails on Windows. + +import { access, mkdtemp, rename, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import Database from '../../sqlite/sync-database' +import { openJournalDatabase } from './journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' +import { loadJournal } from './journal-open' +import { journalDatabaseFile } from './journal-paths' +import { createTrackedJournalOpener } from './journal-store-test-open' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} + +let base: string +let root: string +const journals = createTrackedJournalOpener() + +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } +} + +function runningTool(): AgentJournalItemBody { + return { kind: 'tool-call', name: 'command', input: {}, state: 'running' } +} + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +/** The platform-independent proof: an open handle blocks both of these on + * Windows, where every leak in this file actually shows up. */ +async function expectNothingHoldsTheDirectory(): Promise { + const dbPath = journalDatabaseFile(root) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + const moved = `${root}-recovered-vtest` + await rename(root, moved) + await rm(moved, { recursive: true }) + root = join(base, `journal-${Math.random().toString(36).slice(2)}`) +} + +beforeEach(async () => { + base = await mkdtemp(join(tmpdir(), 'orca-journal-handles-')) + root = join(base, 'journal') +}) + +afterEach(async () => { + await journals.closeAll() + await rm(base, { recursive: true, force: true }) +}) + +describe('the standalone probe owns its own connection', () => { + it('leaves no handle behind after fifty repeated loads', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.appendItem(item(1), { kind: 'message', role: 'user', blocks: [] }, { fence: 1 }) + await journal.close() + + for (let attempt = 0; attempt < 50; attempt += 1) { + const loaded = await loadJournal(root, IDENTITY.sessionId) + expect(loaded?.readOnly).toBe(false) + } + await expectNothingHoldsTheDirectory() + }) + + it('leaves no handle behind after fifty loads of a latched future schema', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.close() + const seeded = openJournalDatabase(journalDatabaseFile(root)) + seeded.db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 3}`) + seeded.db.close() + + for (let attempt = 0; attempt < 50; attempt += 1) { + // The latched open closes the probe connection and returns the read-only + // reopen, so probing repeatedly leaves nothing on a file we must not touch. + expect((await loadJournal(root, IDENTITY.sessionId))?.readOnly).toBe(true) + } + // A read-only connection cannot remove the sidecars it materialized, so only + // the rename/remove half is expected to hold here. + const moved = `${root}-recovered-vtest` + await rename(root, moved) + await rm(moved, { recursive: true }) + root = join(base, 'journal-after-latched') + }) + + it('returns null for a session with no journal, without creating one', async () => { + expect(await loadJournal(root, IDENTITY.sessionId)).toBeNull() + expect(await exists(journalDatabaseFile(root))).toBe(false) + }) +}) + +describe('failure paths inside the open call', () => { + // Site 1: the raw connection is owned by `openJournalDatabase` until it returns. + it('closes the raw connection when the version read cannot run', async () => { + await journals.open({ identity: IDENTITY, journalDir: root }).then((journal) => journal.close()) + await writeFile(journalDatabaseFile(root), 'this is not a database', 'utf8') + + expect(() => openJournalDatabase(journalDatabaseFile(root))).toThrow() + await expectNothingHoldsTheDirectory() + }) + + it('closes the raw connection when the migration cannot start', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.close() + const blocker = openJournalDatabase(journalDatabaseFile(root)) + // Roll the stored version back so the migration runs, then hold the write + // lock it needs: the throw lands after the connection already exists. + blocker.db.pragma('user_version = 0') + blocker.db.exec('BEGIN IMMEDIATE') + blocker.db.exec("INSERT INTO journal_sessions VALUES ('other', 'e', 1)") + try { + expect(() => openJournalDatabase(journalDatabaseFile(root))).toThrow() + } finally { + blocker.db.exec('ROLLBACK') + blocker.db.close() + } + await expectNothingHoldsTheDirectory() + }, 60_000) + + // Sites 2 and 3: `open()` closes its own connection on any throw after the + // connection exists, which is what lets the factory need no `finally`. + it('leaves nothing open when a post-connection step of open() throws', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.appendItem(item(1), runningTool(), { fence: 1 }) + await journal.close() + + // Replay runs after the connection is open, so a read it cannot serve + // throws with the handle already held. + const exec = vi.spyOn(Database.prototype, 'prepare').mockImplementation(() => { + throw new Error('replay cannot read this journal') + }) + try { + await expect(journals.open({ identity: IDENTITY, journalDir: root })).rejects.toThrow( + 'replay cannot read this journal' + ) + } finally { + exec.mockRestore() + } + await expectNothingHoldsTheDirectory() + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-item-appender.ts b/src/main/native-chat/agent-session-journal/journal-item-appender.ts index 3ed2b58b230..1bbc3d4fab3 100644 --- a/src/main/native-chat/agent-session-journal/journal-item-appender.ts +++ b/src/main/native-chat/agent-session-journal/journal-item-appender.ts @@ -5,23 +5,16 @@ import type { } from '../../../shared/agent-session-journal-types' import { journalItemRowBuilder } from './journal-row-builders' import type { JournalReducerState } from './journal-reducer' -import type { AgentSessionJournal } from './journal-store' import type { JournalAppendResult } from './journal-store-contracts' import type { JournalRow } from './journal-row-schema' -import { appendToolOutputFallback } from './journal-tool-output-fallback' type ItemAppendOptions = { fence: number; observedAt?: number; recovered?: true } -type JournalBlob = { digest: string; payload: string } export class JournalItemAppender { constructor( private readonly deps: { - journal: () => AgentSessionJournal state: () => JournalReducerState - enqueue: ( - build: (seq: number, ts: number) => JournalRow, - blobs?: readonly JournalBlob[] - ) => Promise + enqueue: (build: (seq: number, ts: number) => JournalRow) => Promise } ) {} @@ -33,37 +26,10 @@ export class JournalItemAppender { const itemId = agentJournalItemKey(identity) return this.deps .enqueue(journalItemRowBuilder(this.deps.state, identity, body, options)) - .then((row) => itemAppendResult(row, itemId)) - } - - appendWithBlobs( - identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly JournalBlob[], - options: ItemAppendOptions - ): Promise { - const itemId = agentJournalItemKey(identity) - return this.deps - .enqueue(journalItemRowBuilder(this.deps.state, identity, body, options), blobs) - .then((row) => itemAppendResult(row, itemId)) - .catch((error: unknown) => - appendToolOutputFallback({ - journal: this.deps.journal(), - error, - identity, - body, - blobs, - itemId, - fence: options.fence - }) - ) - } -} - -function itemAppendResult(row: JournalRow, itemId: string): JournalAppendResult { - return { - cursor: { epoch: row.epoch, sequence: row.seq }, - itemId, - revision: (row as Extract).revision + .then((row) => ({ + cursor: { epoch: row.epoch, sequence: row.seq }, + itemId, + revision: (row as Extract).revision + })) } } diff --git a/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts b/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts index a5a3e2be852..ed86d81861e 100644 --- a/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts @@ -2,19 +2,21 @@ // results by identity read off the same raw lines. Fixtures are shaped like the // files the providers actually write. -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' +import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' -import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { JOURNAL_BLOB_DIR, readJournalBlob } from './journal-blob-store' +import type { + AgentSessionJournalIdentity, + AgentSessionProviderHandle +} from '../../../shared/agent-session-journal-types' import { createLegacyIdentityTracker } from './journal-legacy-identity' import { appendLegacyTranscriptMessages, importLegacyTranscriptIntoJournal } from './journal-legacy-import' -import { boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' +import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' import { openAgentSessionJournal } from './journal-store-factory' import type { AgentSessionJournal } from './journal-store' @@ -29,21 +31,29 @@ function tick(): number { return clock } -function identity(agent: 'claude' | 'codex', sessionId: string): AgentSessionJournalIdentity { +type ImportAgent = 'claude' | 'codex' | 'grok' | 'omp' + +function providerHandle(agent: ImportAgent, sessionId: string): AgentSessionProviderHandle { + if (agent === 'claude') { + return { kind: 'claude', sessionId, leafUuid: null } + } + return agent === 'codex' + ? { kind: 'codex', threadId: sessionId } + : { kind: 'opaque', agent, value: sessionId } +} + +function identity(agent: ImportAgent, sessionId: string): AgentSessionJournalIdentity { return { sessionId, workspaceId: 'ws-1', hostId: 'host-1', agent, - providerHandle: - agent === 'claude' - ? { kind: 'claude', sessionId, leafUuid: null } - : { kind: 'codex', threadId: sessionId } + providerHandle: providerHandle(agent, sessionId) } } async function open( - agent: 'claude' | 'codex', + agent: ImportAgent, sessionId: string, overrides: Partial[0]> = {} ): Promise { @@ -387,7 +397,7 @@ describe('codex import', () => { }) describe('payload bounds on import', () => { - it('marks a clipped tool result and parks the remainder in the blob store', async () => { + it('marks a clipped tool result and discards the remainder', async () => { const output = 'y'.repeat(64 * 1024) const filePath = await writeFixture('claude-big.jsonl', [ { @@ -421,167 +431,13 @@ describe('payload bounds on import', () => { expect(body.output.truncated).toBe(true) expect(body.output.byteLength).toBe(64 * 1024) expect(body.output.head).toHaveLength(1_024) - expect(await readJournalBlob(root, body.output.digest)).toBe(output) - }) - - it('deduplicates staged blobs while importing a replacement epoch', async () => { - const journalDir = join(root, 'dedupe-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 512, - maxSessionBytes: 512 * 1024 - } - const output = 'd'.repeat(32 * 1024) - const bounded = boundPayload(output, limits) - const toolResultLine = (uuid: string) => ({ - parentUuid: null, - isSidechain: false, - type: 'user', - message: { - role: 'user', - content: [{ type: 'tool_result', tool_use_id: `toolu_${uuid}`, content: output }] - }, - uuid, - timestamp: '2026-08-05T10:00:09.000Z', - sessionId: CLAUDE_SESSION - }) - const filePath = await writeFixture('claude-duplicate-blobs.jsonl', [ - toolResultLine('aa11bb22-cc33-4d44-8e55-6f7788990011'), - toolResultLine('bb22cc33-dd44-4e55-8f66-778899001122') - ]) - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) - - await importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 1, - options: { filePath, limits } - }) - - expect(await readJournalBlob(journalDir, bounded.digest)).toBe(output) - expect(await readdir(join(journalDir, JOURNAL_BLOB_DIR))).toEqual([bounded.digest]) - expect( - journal - .snapshot() - .items.map((item) => (item.body.kind === 'tool-call' ? item.body.output?.digest : null)) - ).toEqual([bounded.digest, bounded.digest]) - }) - - it('prunes root-level blobs made stale by a later legacy import', async () => { - const journalDir = join(root, 'prune-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 512, - maxSessionBytes: 512 * 1024 - } - const output = 's'.repeat(32 * 1024) - const bounded = boundPayload(output, limits) - const first = await writeFixture('claude-stale-blob.jsonl', [ - { - parentUuid: null, - isSidechain: false, - type: 'user', - message: { - role: 'user', - content: [{ type: 'tool_result', tool_use_id: 'toolu_stale', content: output }] - }, - uuid: 'aa11bb22-cc33-4d44-8e55-6f7788990011', - timestamp: '2026-08-05T10:00:09.000Z', - sessionId: CLAUDE_SESSION - } - ]) - const second = await writeFixture('claude-without-blob.jsonl', [ - { - parentUuid: null, - isSidechain: false, - type: 'assistant', - message: { role: 'assistant', content: [{ type: 'text', text: 'replacement' }] }, - uuid: 'cc33dd44-ee55-4666-8777-889900112233', - timestamp: '2026-08-05T10:00:10.000Z', - sessionId: CLAUDE_SESSION - } - ]) - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) - - await importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 1, - options: { filePath: first, limits } - }) - expect(await readJournalBlob(journalDir, bounded.digest)).toBe(output) - - await importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 2, - options: { filePath: second, limits } - }) - - expect(await readJournalBlob(journalDir, bounded.digest)).toBeNull() - expect(journal.snapshot().items[0]?.body).toMatchObject({ - kind: 'message', - blocks: [{ type: 'text', text: 'replacement' }] - }) - }) - - it('uses managed catch-up appends when a tool-result blob exceeds quota', async () => { - const journalDir = join(root, 'catchup-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 128, - maxSessionBytes: 8_000 - } - const journal = await open('codex', CODEX_SESSION, { - journalDir, - limits, - autoCompact: false - }) - const output = 'z'.repeat(12_000) - const bounded = boundPayload(output, limits) - - await expect( - appendLegacyTranscriptMessages({ - journal, - agent: 'codex', - sessionId: CODEX_SESSION, - fence: 1, - messages: [ - { - id: 'catchup-tool-output', - role: 'tool', - blocks: [{ type: 'tool-result', output }], - timestamp: 1_800_000_000_000, - source: 'transcript' - } - ] - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(await readJournalBlob(journalDir, bounded.digest)).toBeNull() - expect(journal.snapshot().items).toEqual([]) }) }) describe('import failures', () => { it('rejects a legacy source above the fixed 16 MiB import cap before decoding', async () => { const journalDir = join(root, 'oversized-source-journal') - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 256 * 1024 * 1024 }, - autoCompact: false - }) + const journal = await open('claude', CLAUDE_SESSION, { journalDir }) const filePath = join(root, 'oversized-source.jsonl') await writeFile(filePath, 'x'.repeat(16 * 1024 * 1024 + 1), 'utf8') const epoch = journal.epoch @@ -602,115 +458,10 @@ describe('import failures', () => { expect(journal.snapshot().items).toEqual([]) }) - it('keeps the live epoch intact when a staged rebuild runs out of budget', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - const journal = await open('codex', CODEX_SESSION, { limits }) - await appendLegacyTranscriptMessages({ - journal, - agent: 'codex', - sessionId: CODEX_SESSION, - fence: 1, - messages: [ - { - id: 'durable-prefix', - role: 'assistant', - blocks: [{ type: 'text', text: 'keep me' }], - timestamp: 1_800_000_000_000, - source: 'transcript' - } - ] - }) - const filePath = await writeFixture('oversized-rollout.jsonl', [ - CODEX_LINES[0], - CODEX_LINES[1], - CODEX_LINES[2], - { - type: 'event_msg', - timestamp: '2026-08-05T10:00:03.000Z', - payload: { type: 'agent_message', message: 'x'.repeat(2_000) } - } - ]) - const epoch = journal.epoch - const snapshotPath = join(root, 'snapshot.json') - const logPath = join(root, 'log.jsonl') - const before = { - snapshot: await readFile(snapshotPath, 'utf-8'), - log: await readFile(logPath, 'utf-8') - } - - await expect( - importLegacyTranscriptIntoJournal({ - journal, - agent: 'codex', - sessionId: CODEX_SESSION, - fence: 1, - options: { filePath, limits } - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - expect(journal.epoch).toBe(epoch) - expect(await readFile(snapshotPath, 'utf-8')).toBe(before.snapshot) - expect(await readFile(logPath, 'utf-8')).toBe(before.log) - expect(journal.snapshot().items[0]?.body).toMatchObject({ - kind: 'message', - blocks: [{ type: 'text', text: 'keep me' }] - }) - }) - - it('cleans staged replacement blobs when legacy import exceeds physical quota', async () => { - const journalDir = join(root, 'replacement-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 128, - maxSessionBytes: 8_000 - } - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) - const output = 'q'.repeat(12_000) - const bounded = boundPayload(output, limits) - const filePath = await writeFixture('oversized-tool-result.jsonl', [ - { - parentUuid: null, - isSidechain: false, - type: 'user', - message: { - role: 'user', - content: [{ type: 'tool_result', tool_use_id: 'toolu_oversized', content: output }] - }, - uuid: 'ba11ad00-1111-4222-8333-444455556666', - timestamp: '2026-08-05T10:00:09.000Z', - sessionId: CLAUDE_SESSION - } - ]) - const epoch = journal.epoch - - await expect( - importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 1, - options: { filePath, limits } - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toBe(epoch) - expect(await readJournalBlob(journalDir, bounded.digest)).toBeNull() - expect((await readdir(journalDir)).some((name) => name.startsWith('.epoch-replacement-'))).toBe( - false - ) - }) - it('bounds oversized legacy tool-call input before journal publication', async () => { const journalDir = join(root, 'bounded-tool-input-journal') const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 64 } - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) + const journal = await open('claude', CLAUDE_SESSION, { journalDir }) const filePath = await writeFixture('oversized-tool-input.jsonl', [ { parentUuid: null, @@ -768,6 +519,39 @@ describe('import failures', () => { expect(journal.epoch).toBe(before) }) + // A transcript with no decodable messages recovers nothing. Publishing an + // empty replacement would roll the epoch and drop whatever the journal held — + // including a repair's own anchor and disclosure. + it('leaves the epoch untouched when the transcript decodes to no messages', async () => { + const journal = await open('codex', CODEX_SESSION) + await journal.appendItem( + { provider: 'codex', threadId: CODEX_SESSION, turnId: 'turn-1', ordinal: 1 }, + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'kept' }] }, + { fence: 1 } + ) + const before = journal.epoch + const metadataOnly = await writeFixture('metadata-only.jsonl', [ + { + type: 'session_meta', + timestamp: '2026-08-05T10:00:00.000Z', + payload: { id: CODEX_SESSION, session_id: CODEX_SESSION, cwd: '/Users/dev/project' } + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'codex', + sessionId: CODEX_SESSION, + fence: 1, + options: { filePath: metadataOnly } + }) + + expect(result).toMatchObject({ ok: true, imported: 0, replaced: false }) + expect(journal.epoch).toBe(before) + expect(journal.snapshot().items).toHaveLength(1) + await journal.close() + }) + it('rejects an agent with no transcript decoder', async () => { const journal = await open('claude', CLAUDE_SESSION) const result = await importLegacyTranscriptIntoJournal({ @@ -780,3 +564,132 @@ describe('import failures', () => { expect(result).toMatchObject({ ok: false }) }) }) + +// A tool call is only the SOLE block of its message when the provider wrote it +// that way. Claude interleaves it with narration, Grok hangs `tool_calls` off a +// row that also has text, and omp's execution cells always pair the invocation +// with its output — so the multi-block path carries untrusted tool input too. +describe('multi-block legacy messages', () => { + const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 64 } + const oversized = 'x'.repeat(10_000) + + /** The tool-call block of the first imported multi-block message. */ + function importedToolCallBlock(journal: AgentSessionJournal): unknown { + for (const entry of journal.snapshot().items) { + if (entry.body.kind !== 'message') { + continue + } + const block = entry.body.blocks.find((candidate) => candidate.type === 'tool-call') + if (block) { + return block.input + } + } + return null + } + + it('bounds a Claude tool call that shares its message with narration', async () => { + const journal = await open('claude', CLAUDE_SESSION, { + journalDir: join(root, 'claude-mixed-journal') + }) + const filePath = await writeFixture('claude-mixed.jsonl', [ + { + parentUuid: null, + isSidechain: false, + type: 'assistant', + message: { + role: 'assistant', + content: [ + { type: 'text', text: 'Editing the file.' }, + { + type: 'tool_use', + id: 'toolu_mixed', + name: 'Edit', + input: { file_path: 'a.ts', patch: oversized } + } + ] + }, + uuid: 'dd22be00-1111-4222-8333-444455556666', + timestamp: '2026-08-05T10:00:09.000Z', + sessionId: CLAUDE_SESSION + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'claude', + sessionId: CLAUDE_SESSION, + fence: 1, + options: { filePath, limits } + }) + + expect(result.ok).toBe(true) + expect(importedToolCallBlock(journal)).toMatchObject({ + truncated: true, + byteLength: expect.any(Number), + digest: expect.stringMatching(/^[0-9a-f]{64}$/), + head: expect.any(String) + }) + expect(JSON.stringify(journal.snapshot().items)).not.toContain('x'.repeat(1_000)) + await journal.close() + }) + + it('bounds a Grok tool call that shares its row with assistant text', async () => { + const journal = await open('grok', CODEX_SESSION, { + journalDir: join(root, 'grok-mixed-journal') + }) + const filePath = await writeFixture('grok-mixed.jsonl', [ + { + type: 'assistant', + id: 'asst-mixed', + timestamp: '2026-08-05T10:00:09.000Z', + content: [{ type: 'text', text: 'Searching.' }], + tool_calls: [{ id: 'c1', name: 'grep', arguments: JSON.stringify({ pattern: oversized }) }] + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'grok', + sessionId: CODEX_SESSION, + fence: 1, + options: { filePath, limits } + }) + + expect(result.ok).toBe(true) + expect(importedToolCallBlock(journal)).toMatchObject({ truncated: true }) + expect(JSON.stringify(journal.snapshot().items)).not.toContain('x'.repeat(1_000)) + await journal.close() + }) + + it('bounds an omp execution cell, whose invocation always ships with its output', async () => { + const journal = await open('omp', CODEX_SESSION, { + journalDir: join(root, 'omp-mixed-journal') + }) + const filePath = await writeFixture('omp-mixed.jsonl', [ + { + type: 'message', + id: 'omp-mixed-1', + timestamp: '2026-08-05T10:00:09.000Z', + message: { + role: 'bashExecution', + command: `echo ${oversized}`, + output: 'done', + exitCode: 0 + } + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'omp', + sessionId: CODEX_SESSION, + fence: 1, + options: { filePath, limits } + }) + + expect(result.ok).toBe(true) + expect(importedToolCallBlock(journal)).toMatchObject({ truncated: true }) + expect(JSON.stringify(journal.snapshot().items)).not.toContain('x'.repeat(1_000)) + await journal.close() + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-legacy-import.ts b/src/main/native-chat/agent-session-journal/journal-legacy-import.ts index 126196af38e..6a2ff099dbc 100644 --- a/src/main/native-chat/agent-session-journal/journal-legacy-import.ts +++ b/src/main/native-chat/agent-session-journal/journal-legacy-import.ts @@ -49,7 +49,14 @@ export type LegacyImportOptions = ResolveSessionFileOptions & { const MAX_LEGACY_IMPORT_SOURCE_BYTES = 16 * 1024 * 1024 export type LegacyImportResult = - | { ok: true; epoch: string; cursor: AgentJournalCursor; imported: number } + | { + ok: true + epoch: string + cursor: AgentJournalCursor + imported: number + /** False when the transcript held no messages and the epoch was left as it stood. */ + replaced: boolean + } | { ok: false; error: string } export async function appendLegacyTranscriptMessages(input: { @@ -61,16 +68,14 @@ export async function appendLegacyTranscriptMessages(input: { }): Promise { let appended = 0 for (const message of input.messages) { - const mapped = legacyItemBody(message, DEFAULT_JOURNAL_PAYLOAD_LIMITS) - await input.journal.appendItemWithBlobs( + await input.journal.appendItem( { provider: 'legacy', agent: input.agent, sessionId: input.sessionId, recordId: message.id }, - mapped.body, - mapped.blobs, + legacyItemBody(message, DEFAULT_JOURNAL_PAYLOAD_LIMITS), { fence: input.fence, observedAt: message.timestamp ?? undefined } ) appended += 1 @@ -134,16 +139,28 @@ export async function importLegacyTranscriptIntoJournal(input: { if (!identity) { continue } - const mapped = legacyItemBody(message, limits) replacement.push({ identity, - body: mapped.body, - blobs: mapped.blobs, + body: legacyItemBody(message, limits), observedAt: message.timestamp ?? undefined }) } + // A transcript that decodes to nothing reconstructs nothing, and an empty + // replacement is not a harmless no-op: it would delete the repair's anchor and + // its disclosure, leaving nothing to ask for the history again. The epoch + // stands so a later read can still rebuild it. + if (replacement.length === 0) { + const current = input.journal.cursor() + return { ok: true, epoch: current.epoch, cursor: current, imported: 0, replaced: false } + } const cursor = await input.journal.replaceEpochItems('legacy_import', input.fence, replacement) - return { ok: true, epoch: cursor.epoch, cursor, imported: decoded.messages.length } + return { + ok: true, + epoch: cursor.epoch, + cursor, + imported: decoded.messages.length, + replaced: true + } } const TRANSCRIPT_DECODERS = { @@ -199,11 +216,6 @@ async function decodeWithIdentities(input: { return { messages, identities } } -type MappedLegacyItem = { - body: AgentJournalItemBody - blobs: { digest: string; payload: string }[] -} - /** * A message whose only content is a tool invocation becomes a tool-call item so * the reducer renders it as one. Everything else stays a message item with its @@ -212,49 +224,40 @@ type MappedLegacyItem = { function legacyItemBody( message: NativeChatMessage, limits: JournalPayloadLimits -): MappedLegacyItem { +): AgentJournalItemBody { const only = message.blocks.length === 1 ? message.blocks[0] : undefined if (only?.type === 'tool-call') { + // Legacy transcripts are untrusted and can contain arbitrarily large tool + // arguments. Keep them on the same bounded path as live events before the + // replacement epoch is published. return { - // Legacy transcripts are untrusted and can contain arbitrarily large - // tool arguments. Keep them on the same bounded path as live events - // before the replacement epoch is staged or published. - body: { - kind: 'tool-call', - name: only.name, - input: boundToolInput(only.input, limits), - state: 'completed' - }, - blobs: [] + kind: 'tool-call', + name: only.name, + input: boundToolInput(only.input, limits), + state: 'completed' } } if (only?.type === 'tool-result') { - const output = boundPayload(only.output, limits) return { - body: { - kind: 'tool-call', - name: 'tool-result', - input: null, - state: only.isError ? 'failed' : 'completed', - output - }, - blobs: output.truncated ? [{ digest: output.digest, payload: only.output }] : [] + kind: 'tool-call', + name: 'tool-result', + input: null, + state: only.isError ? 'failed' : 'completed', + output: boundPayload(only.output, limits) } } return { - body: { - kind: 'message', - role: message.role, - blocks: message.blocks.map((block) => boundBlock(block, limits)) - }, - blobs: [] + kind: 'message', + role: message.role, + blocks: message.blocks.map((block) => boundBlock(block, limits)) } } -/** Inline block text keeps only a bounded head plus an explicit marker. No blob - * is written: the marker carries the digest and byte length, and the source - * transcript remains the full copy — a blob here would be unreferenced by the - * render model and pruned at the next compaction. */ +/** Every block that can carry untrusted bulk is bounded here, tool calls + * included: a provider decoder is free to put one alongside narration, and the + * sole-block path above never sees those. The remainder is discarded rather + * than stored elsewhere — the marker keeps its digest and byte length, and the + * source transcript remains the full copy. */ function boundBlock(block: NativeChatBlock, limits: JournalPayloadLimits): NativeChatBlock { if (block.type === 'text') { return { ...block, text: boundInlineText(block.text, limits).text } @@ -262,5 +265,8 @@ function boundBlock(block: NativeChatBlock, limits: JournalPayloadLimits): Nativ if (block.type === 'tool-result') { return { ...block, output: boundInlineText(block.output, limits).text } } + if (block.type === 'tool-call') { + return { ...block, input: boundToolInput(block.input, limits) } + } return block } diff --git a/src/main/native-chat/agent-session-journal/journal-lifecycle-admission.ts b/src/main/native-chat/agent-session-journal/journal-lifecycle-admission.ts deleted file mode 100644 index 063e807038e..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-lifecycle-admission.ts +++ /dev/null @@ -1,160 +0,0 @@ -import type { - AgentJournalItemBody, - AgentJournalSnapshot -} from '../../../shared/agent-session-journal-types' -import { - dispatchReservationId, - JournalLifecycleCapacity, - lifecycleReservationIdForItem, - requiresTerminalSettlement, - terminalReservationBytes, - type JournalLifecycleReservation -} from './journal-lifecycle-capacity' -import type { JournalRow } from './journal-row-schema' -import { journalRowByteLength } from './journal-row-schema' -import { AgentSessionJournalError } from './journal-write-guards' - -export type JournalLifecycleRowAdmission = { - releaseAfter: string[] - protectedBytes: number - lifecycleCovered: boolean - proposedCapacity: JournalLifecycleCapacity -} - -export class JournalLifecycleAdmission { - private readonly capacity = new JournalLifecycleCapacity() - - constructor( - private readonly sessionId: string, - private readonly maxBytes: number, - private readonly canonicalItemId: (itemId: string) => string, - private readonly maxAppendSlots = Number.MAX_SAFE_INTEGER - ) {} - - get state(): { reservedBytes: number; reservedAppendSlots: number } { - return { - reservedBytes: this.capacity.reservedBytes, - reservedAppendSlots: this.capacity.reservedAppendSlots - } - } - - rebuild(snapshot: AgentJournalSnapshot, currentPhysicalBytes: number): void { - if ( - !this.capacity.rebuild(snapshot, this.maxBytes, currentPhysicalBytes, this.maxAppendSlots) - ) { - throw this.capacityError('cannot rebuild lifecycle capacity') - } - } - - reserve(token: JournalLifecycleReservation, currentPhysicalBytes: number): boolean { - return this.capacity.reserve(token, currentPhysicalBytes, this.maxBytes, this.maxAppendSlots) - } - - transfer(fromId: string, toId: string): boolean { - return this.capacity.transfer(fromId, toId) - } - - release(id: string): void { - this.capacity.release(id) - } - - prepare(row: JournalRow, currentPhysicalBytes: number): JournalLifecycleRowAdmission { - const proposedCapacity = this.capacity.clone() - this.ensureActionable(row, currentPhysicalBytes, proposedCapacity) - const releaseAfter = this.reservationsSettledBy(row, proposedCapacity) - const releasedBytes = releaseAfter.reduce( - (total, id) => total + (proposedCapacity.token(id)?.bytes ?? 0), - 0 - ) - return { - releaseAfter, - protectedBytes: proposedCapacity.reservedBytes - releasedBytes, - lifecycleCovered: proposedCapacity.covers(releaseAfter, journalRowByteLength(row), 1), - proposedCapacity - } - } - - commit(admission: JournalLifecycleRowAdmission): void { - this.capacity.replaceFrom(admission.proposedCapacity) - for (const id of admission.releaseAfter) { - this.capacity.release(id) - } - } - - private ensureActionable( - row: JournalRow, - currentPhysicalBytes: number, - capacity: JournalLifecycleCapacity - ): void { - if (row.kind === 'item') { - this.ensureActionableItem(row.itemId, row.body, currentPhysicalBytes, capacity) - return - } - if (row.kind !== 'lifecycle-batch') { - return - } - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - this.ensureActionableItem(mutation.itemId, mutation.body, currentPhysicalBytes, capacity) - } - } - } - - private ensureActionableItem( - itemId: string, - body: AgentJournalItemBody, - currentPhysicalBytes: number, - capacity: JournalLifecycleCapacity - ): void { - if (!requiresTerminalSettlement(body)) { - return - } - const id = lifecycleReservationIdForItem(this.canonicalItemId(itemId)) - if (body.kind === 'status' && body.turnLifecycle?.state === 'running' && !capacity.has(id)) { - capacity.claimFirst('tentative-turn:', id) - } - if ( - !capacity.reserve( - { id, bytes: terminalReservationBytes(body), appendSlots: 1 }, - currentPhysicalBytes, - this.maxBytes, - this.maxAppendSlots - ) - ) { - throw this.capacityError('cannot reserve terminal capacity') - } - } - - private reservationsSettledBy(row: JournalRow, capacity: JournalLifecycleCapacity): string[] { - if (row.kind === 'dispatch') { - const id = dispatchReservationId(row.clientMessageId) - return capacity.has(id) ? [id] : [] - } - const itemIds = - row.kind === 'item' - ? requiresTerminalSettlement(row.body) - ? [] - : [row.itemId] - : row.kind === 'tombstone' - ? [row.itemId] - : row.kind === 'lifecycle-batch' - ? row.mutations.flatMap((mutation) => - mutation.kind === 'item' && requiresTerminalSettlement(mutation.body) - ? [] - : [mutation.itemId] - ) - : [] - return [ - ...new Set( - itemIds.map((itemId) => lifecycleReservationIdForItem(this.canonicalItemId(itemId))) - ) - ].filter((id) => capacity.has(id)) - } - - private capacityError(detail: string): AgentSessionJournalError { - return new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} ${detail}` - ) - } -} diff --git a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.test.ts b/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.test.ts deleted file mode 100644 index 331bc63c9dc..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.test.ts +++ /dev/null @@ -1,30 +0,0 @@ -import { describe, expect, it } from 'vitest' -import type { AgentJournalSnapshot } from '../../../shared/agent-session-journal-types' -import { JournalLifecycleCapacity } from './journal-lifecycle-capacity' - -describe('JournalLifecycleCapacity', () => { - it('enforces append-slot limits for both rebuilt submission reservations', () => { - const snapshot: AgentJournalSnapshot = { - sessionId: 'session-1', - cursor: { epoch: 'epoch-1', sequence: 1 }, - items: [], - submissions: [ - { - clientMessageId: 'message-1', - fence: 0, - payloadFingerprint: 'fingerprint', - dispatchState: 'pending', - providerItemId: null, - reason: null, - submittedAt: 1, - resolvedAt: null - } - ] - } - - const capacity = new JournalLifecycleCapacity() - - expect(capacity.rebuild(snapshot, Number.MAX_SAFE_INTEGER, 0, 1)).toBe(false) - expect(capacity.reservedAppendSlots).toBe(1) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.ts b/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.ts deleted file mode 100644 index 0c99070b1e4..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.ts +++ /dev/null @@ -1,193 +0,0 @@ -import type { - AgentJournalItemBody, - AgentJournalSnapshot -} from '../../../shared/agent-session-journal-types' - -export type JournalLifecycleReservation = { - id: string - bytes: number - appendSlots: number -} - -export const JOURNAL_TURN_TERMINAL_RESERVATION_BYTES = 128 * 1024 -export const JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES = 64 * 1024 -export const JOURNAL_DISPATCH_RESERVATION_BYTES = 32 * 1024 - -export class JournalLifecycleCapacity { - private readonly reservations = new Map() - - get reservedBytes(): number { - return [...this.reservations.values()].reduce((total, token) => total + token.bytes, 0) - } - - get reservedAppendSlots(): number { - return [...this.reservations.values()].reduce((total, token) => total + token.appendSlots, 0) - } - - has(id: string): boolean { - return this.reservations.has(id) - } - - token(id: string): JournalLifecycleReservation | null { - return this.reservations.get(id) ?? null - } - - clone(): JournalLifecycleCapacity { - const copy = new JournalLifecycleCapacity() - for (const token of this.reservations.values()) { - copy.reservations.set(token.id, { ...token }) - } - return copy - } - - replaceFrom(source: JournalLifecycleCapacity): void { - this.reservations.clear() - for (const token of source.reservations.values()) { - this.reservations.set(token.id, { ...token }) - } - } - - reserve( - token: JournalLifecycleReservation, - currentPhysicalBytes: number, - maxBytes: number, - maxAppendSlots = Number.MAX_SAFE_INTEGER - ): boolean { - if (this.reservations.has(token.id)) { - return true - } - if (currentPhysicalBytes + this.reservedBytes + token.bytes > maxBytes) { - return false - } - if (this.reservedAppendSlots + token.appendSlots > maxAppendSlots) { - return false - } - this.reservations.set(token.id, token) - return true - } - - transfer(fromId: string, toId: string): boolean { - const existing = this.reservations.get(fromId) - if (!existing) { - return false - } - this.reservations.delete(fromId) - this.reservations.set(toId, { ...existing, id: toId }) - return true - } - - claimFirst(prefix: string, toId: string): boolean { - const fromId = [...this.reservations.keys()].find((id) => id.startsWith(prefix)) - return fromId ? this.transfer(fromId, toId) : false - } - - release(id: string): void { - this.reservations.delete(id) - } - - covers(ids: readonly string[], bytes: number, appendSlots: number): boolean { - const tokens = ids.flatMap((id) => { - const token = this.reservations.get(id) - return token ? [token] : [] - }) - return ( - tokens.length > 0 && - tokens.reduce((total, token) => total + token.bytes, 0) >= bytes && - tokens.reduce((total, token) => total + token.appendSlots, 0) >= appendSlots - ) - } - - rebuild( - snapshot: AgentJournalSnapshot, - maxBytes: number, - currentPhysicalBytes: number, - maxAppendSlots = Number.MAX_SAFE_INTEGER - ): boolean { - this.reservations.clear() - for (const item of snapshot.items) { - if (!requiresTerminalSettlement(item.body)) { - continue - } - if ( - !this.reserve( - { - id: lifecycleReservationIdForItem(item.itemId), - bytes: terminalReservationBytes(item.body), - appendSlots: 1 - }, - currentPhysicalBytes, - maxBytes, - maxAppendSlots - ) - ) { - return false - } - } - for (const submission of snapshot.submissions) { - if (submission.dispatchState !== 'pending' && submission.dispatchState !== 'unknown') { - continue - } - // A write-ahead submission owns both its dispatch attempt and the - // terminal turn settlement. Rebuild both reservations after restart; - // restoring only the tentative turn token would let a new send consume - // the dispatch headroom still owed to this unresolved submission. - if ( - !this.reserve( - { - id: dispatchReservationId(submission.clientMessageId), - bytes: JOURNAL_DISPATCH_RESERVATION_BYTES, - appendSlots: 1 - }, - currentPhysicalBytes, - maxBytes, - maxAppendSlots - ) - ) { - return false - } - if ( - !this.reserve( - { - id: tentativeTurnReservationId(submission.clientMessageId), - bytes: JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - appendSlots: 1 - }, - currentPhysicalBytes, - maxBytes, - maxAppendSlots - ) - ) { - return false - } - } - return true - } -} - -export function lifecycleReservationIdForItem(itemId: string): string { - return `item:${itemId}` -} - -export function dispatchReservationId(clientMessageId: string): string { - return `dispatch:${clientMessageId}` -} - -export function tentativeTurnReservationId(clientMessageId: string): string { - return `tentative-turn:${clientMessageId}` -} - -export function requiresTerminalSettlement(body: AgentJournalItemBody): boolean { - if (body.kind === 'tool-call') { - return body.state === 'running' - } - if (body.kind === 'approval' || body.kind === 'question') { - return body.resolution.state === 'pending' - } - return body.kind === 'status' && body.turnLifecycle?.state === 'running' -} - -export function terminalReservationBytes(body: AgentJournalItemBody): number { - return body.kind === 'status' && body.turnLifecycle?.state === 'running' - ? JOURNAL_TURN_TERMINAL_RESERVATION_BYTES - : JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES -} diff --git a/src/main/native-chat/agent-session-journal/journal-log-file.test.ts b/src/main/native-chat/agent-session-journal/journal-log-file.test.ts deleted file mode 100644 index 98c0b5029d3..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-log-file.test.ts +++ /dev/null @@ -1,374 +0,0 @@ -import { mkdtemp, readdir, rm, writeFile } from 'node:fs/promises' -import type * as FsPromises from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { appendJournalRows, JOURNAL_SNAPSHOT_FILE, readJournalSnapshot } from './journal-log-file' -import type { JournalSnapshotFile } from './journal-log-file' -import type { JournalRow } from './journal-row-schema' -import { openAgentSessionJournal } from './journal-store-factory' -import { - projectStructuredAgentSessionStatus, - projectStructuredItemsToNativeChat -} from '../../../shared/structured-agent-session-projection' - -type FakeDirectoryHandle = { sync: ReturnType; close: ReturnType } - -let openDirectoryHook: ((path: unknown, flags: unknown) => FakeDirectoryHandle | undefined) | null = - null - -vi.mock('node:fs/promises', async (importOriginal) => { - const actual = await importOriginal() - return { - ...actual, - open: (async (...args: Parameters) => { - const fake = openDirectoryHook?.(args[0], args[1]) - return fake ?? actual.open(...args) - }) as typeof actual.open - } -}) - -let root: string - -beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-log-file-')) - openDirectoryHook = null -}) - -afterEach(async () => { - openDirectoryHook = null - await rm(root, { recursive: true, force: true }) -}) - -function validSnapshot(): JournalSnapshotFile { - return { - v: 1, - epoch: 'epoch-A', - compactedThrough: 2, - highestFence: 1, - items: [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'hi' }] }, - sequence: 2, - observedAt: 1_000 - } - ], - submissions: [], - receipts: [], - aliases: [], - tombstones: [{ itemId: 'codex:thread-1:turn-1:2', revision: 3 }], - tail: [] - } -} - -async function writeSnapshot(snapshot: unknown): Promise { - await writeFile(join(root, JOURNAL_SNAPSHOT_FILE), JSON.stringify(snapshot), 'utf-8') -} - -describe('readJournalSnapshot validation', () => { - it('accepts a well-formed snapshot, with and without the tombstones collection', async () => { - await writeSnapshot(validSnapshot()) - expect((await readJournalSnapshot(root)).status).toBe('valid') - - const { tombstones: _tombstones, ...withoutTombstones } = validSnapshot() - await writeSnapshot(withoutTombstones) - expect((await readJournalSnapshot(root)).status).toBe('valid') - }) - - it('accepts every canonical item kind and a fully-formed submission', async () => { - const snapshot = validSnapshot() - const payload = { head: 'x', byteLength: 4, digest: 'd'.repeat(64), truncated: true } - snapshot.items = [ - { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] }, - { - kind: 'tool-call', - name: 'Read', - input: { path: 'a' }, - state: 'completed', - output: payload - }, - { kind: 'diff', path: 'a.ts', patch: payload }, - { - kind: 'approval', - title: 'Run?', - detail: null, - options: [{ id: 'a', label: 'Yes' }], - resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } - }, - { - kind: 'question', - question: 'Deploy?', - options: [{ id: 'a', label: 'Yes' }], - freeTextQuestionId: 'q-free', - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 5 } - }, - { - kind: 'status', - text: 'working', - turnLifecycle: { turnId: 'turn-1', state: 'running' }, - providerFrame: { provider: 'codex', kind: 'raw', payload } - } - ].map((body, index) => ({ - itemId: `codex:thread-1:turn-1:${index + 1}`, - revision: 1, - body: body as JournalSnapshotFile['items'][number]['body'], - sequence: index + 1, - observedAt: 1_000, - ...(index === 0 ? { recovered: true as const } : {}) - })) - snapshot.compactedThrough = snapshot.items.length - snapshot.submissions = [ - { - clientMessageId: 'm-1', - fence: 1, - payloadFingerprint: 'a'.repeat(64), - dispatchState: 'accepted', - providerItemId: 'codex:thread-1:turn-1:1', - reason: null, - submittedAt: 1_000, - resolvedAt: 1_001 - } - ] - await writeSnapshot(snapshot) - expect((await readJournalSnapshot(root)).status).toBe('valid') - }) - - it('classifies a JSON-valid non-array tombstones collection as invalid instead of valid', async () => { - await writeSnapshot({ ...validSnapshot(), tombstones: {} }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects tombstone entries that would poison seeding', async () => { - for (const tombstones of [ - [{ itemId: 42, revision: 1 }], - [{ itemId: 'codex:thread-1:turn-1:1', revision: 'one' }], - [{ itemId: 'codex:thread-1:turn-1:1', revision: Number.NaN }], - ['codex:thread-1:turn-1:1'] - ]) { - await writeSnapshot({ ...validSnapshot(), tombstones }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - } - }) - - it('rejects JSON-valid nested item corruption instead of admitting it', async () => { - // A resolved question with `options: null` used to pass shallow admission and - // then throw `TypeError` in the shared projection's `options.map`. - const poisonedQuestion = validSnapshot() - poisonedQuestion.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { - kind: 'question', - question: 'Deploy?', - options: null, - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 1 } - }, - sequence: 2, - observedAt: 1_000 - } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(poisonedQuestion) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - - // Pending-prompt surfaces read `resolution.state` before anything else. - const nullResolution = validSnapshot() - nullResolution.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'question', question: 'Deploy?', options: [], resolution: null }, - sequence: 2, - observedAt: 1_000 - } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(nullResolution) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects a JSON-valid nested corruption in the retained tail', async () => { - const poisonedTail = validSnapshot() - poisonedTail.tail = [ - { - v: 1, - epoch: 'epoch-A', - seq: 3, - fence: 1, - ts: 1_000, - kind: 'item', - itemId: 'codex:thread-1:turn-1:3', - revision: 1, - body: { - kind: 'question', - question: 'Deploy?', - options: null, - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 1 } - } - } - ] as unknown as JournalSnapshotFile['tail'] - await writeSnapshot(poisonedTail) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects a submission that only carries a client message id', async () => { - const shallowSubmission = validSnapshot() - shallowSubmission.submissions = [ - { clientMessageId: 'm-1' } - ] as unknown as JournalSnapshotFile['submissions'] - await writeSnapshot(shallowSubmission) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects items and counters that only look shallowly plausible', async () => { - const missingSequence = validSnapshot() - missingSequence.items = [ - { itemId: 'codex:thread-1:turn-1:1', revision: 1, body: { kind: 'status', text: 'x' } } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(missingSequence) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - - await writeSnapshot({ ...validSnapshot(), compactedThrough: Number.NaN }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) -}) - -describe('future-version snapshot classification', () => { - it('classifies a future version before shape validation so unknown bodies stay unreadable', async () => { - // The version can only advance because bodies changed, so a future snapshot - // legitimately carries kinds this build cannot parse. That is the - // schema-unreadable contract, not corruption. - const future = validSnapshot() as unknown as Record - future.v = 99 - future.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'future-render-kind', payload: { anything: true } }, - sequence: 2, - observedAt: 1_000 - } - ] - await writeSnapshot(future) - expect((await readJournalSnapshot(root)).status).toBe('unreadable') - }) - - it('classifies a future version as unreadable even when its shapes still parse today', async () => { - await writeSnapshot({ ...validSnapshot(), v: 99 }) - expect((await readJournalSnapshot(root)).status).toBe('unreadable') - }) - - it('treats a non-integer or sub-1 version as invalid, matching row admission', async () => { - for (const v of [0, 1.5]) { - await writeSnapshot({ ...validSnapshot(), v }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - } - }) -}) - -describe('journal startup isolation from a malformed snapshot', () => { - it('quarantines a JSON-valid malformed snapshot instead of throwing through open', async () => { - await writeSnapshot({ ...validSnapshot(), tombstones: {} }) - - const journal = await openAgentSessionJournal({ - identity: { - sessionId: 'session-1', - workspaceId: 'ws-1', - hostId: 'host-1', - agent: 'codex', - providerHandle: { kind: 'codex', threadId: 'thread-1' } - }, - journalDir: root - }) - - // Degraded exactly like other corrupt snapshots: quarantined on disk, never - // silently deleted, and the session does not adopt state it cannot trust. - const entries = await readdir(root) - expect(entries.some((entry) => entry.startsWith('quarantine-snapshot-'))).toBe(true) - expect(entries.includes(JOURNAL_SNAPSHOT_FILE)).toBe(false) - expect(journal.snapshot().items).toEqual([]) - }) -}) - -describe('reopen after a persisted JSON-valid poisoned question', () => { - it('quarantines the snapshot so reopen-to-render cannot throw in projection', async () => { - const poisoned = validSnapshot() - poisoned.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { - kind: 'question', - question: 'Deploy?', - options: null, - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 1 } - }, - sequence: 2, - observedAt: 1_000 - } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(poisoned) - - const journal = await openAgentSessionJournal({ - identity: { - sessionId: 'session-1', - workspaceId: 'ws-1', - hostId: 'host-1', - agent: 'codex', - providerHandle: { kind: 'codex', threadId: 'thread-1' } - }, - journalDir: root - }) - - // The poisoned item must land in quarantine, not in the reopened state: - // pre-fix it was admitted and the render path below threw - // `TypeError: Cannot read properties of null (reading 'map')`. - const entries = await readdir(root) - expect(entries.some((entry) => entry.startsWith('quarantine-snapshot-'))).toBe(true) - const items = journal.snapshot().items - expect(() => projectStructuredItemsToNativeChat(items)).not.toThrow() - expect(() => projectStructuredAgentSessionStatus(items)).not.toThrow() - expect(items).toEqual([]) - }) -}) - -describe('appendJournalRows directory fsync', () => { - const ROW: JournalRow = { - kind: 'epoch', - reason: 'session_created', - providerHandle: { kind: 'codex', threadId: 'thread-1' }, - v: 1, - epoch: 'epoch-A', - seq: 1, - fence: 0, - ts: 1_000 - } - - function hookDirectoryOpen(sync: ReturnType): FakeDirectoryHandle { - const fake: FakeDirectoryHandle = { sync, close: vi.fn(async () => undefined) } - openDirectoryHook = (path, flags) => (path === root && flags === 'r' ? fake : undefined) - return fake - } - - it('closes the directory handle when directory fsync fails', async () => { - const fake = hookDirectoryOpen( - vi.fn(async () => { - throw new Error('EINVAL: sync') - }) - ) - - // Tolerating unsupported directory fsync must not turn into a leak. - await expect(appendJournalRows(root, [ROW])).resolves.toBeUndefined() - expect(fake.sync).toHaveBeenCalledTimes(1) - expect(fake.close).toHaveBeenCalledTimes(1) - }) - - it('closes the directory handle when directory fsync succeeds', async () => { - const fake = hookDirectoryOpen(vi.fn(async () => undefined)) - - await expect(appendJournalRows(root, [ROW])).resolves.toBeUndefined() - expect(fake.sync).toHaveBeenCalledTimes(1) - expect(fake.close).toHaveBeenCalledTimes(1) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-log-file.ts b/src/main/native-chat/agent-session-journal/journal-log-file.ts deleted file mode 100644 index 44cbc26d0d5..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-log-file.ts +++ /dev/null @@ -1,365 +0,0 @@ -// On-disk layout for one session's journal. -// -// /log.jsonl append-only rows, fsynced before the caller is told the write landed -// /snapshot.json folded state at a compaction boundary PLUS the retained tail -// /blobs/ bounded-payload remainders -// -// The snapshot carries its own tail so compaction is one atomic write. A crash -// between publishing the snapshot and truncating the log leaves the log a -// superset of the tail, and recovery unions the two by sequence — never a hole. - -import { appendFile, mkdir, open, readFile, stat, type FileHandle } from 'node:fs/promises' -import { randomUUID } from 'node:crypto' -import { join } from 'node:path' -import { durableWriteTempPath, renameDurable, writeFileDurable } from '../../durable-file-write' -import { - AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - type AgentJournalRenderItem, - type AgentJournalSubmission -} from '../../../shared/agent-session-journal-types' -import { - isAdmissibleAgentJournalRenderItem, - isAdmissibleAgentJournalSubmission -} from '../../../shared/agent-session-journal-schemas' -import { parseJournalRow, serializeJournalRow, type JournalRow } from './journal-row-schema' -import { assertJournalPhysicalCapacity } from './journal-physical-quota' - -export const JOURNAL_LOG_FILE = 'log.jsonl' -export const JOURNAL_SNAPSHOT_FILE = 'snapshot.json' - -export type JournalSnapshotFile = { - v: number - epoch: string - /** Highest sequence folded into `items`; the tail starts after it. */ - compactedThrough: number - /** Fence monotonicity survives compaction and restart. */ - highestFence: number - items: AgentJournalRenderItem[] - submissions: AgentJournalSubmission[] - /** Receipts outlive the rows that minted them: a client reconnecting after - * compaction must still get the same answer instead of re-sending. */ - receipts: { - clientMessageId: string - providerItemId: string - epoch: string - sequence: number - acceptedAt: number - }[] - /** Provider item id → submission slot, preserved so a post-compaction echo - * still reconciles into the bubble it belongs to. */ - aliases: { providerItemId: string; itemId: string }[] - tombstones: { itemId: string; revision: number }[] - /** Bounded by compaction retention; used to deduplicate a replayed settlement. */ - appliedSettlementIds?: string[] - tail: JournalRow[] -} - -export type JournalReadResult = { - rows: JournalRow[] - /** True when a line used a schema version this build cannot read. Reading - * STOPS there — the row must not be skipped — and the host degrades to - * read-only: no writes, no compaction, no deletion. */ - unreadable: boolean - /** Lines that failed to parse for reasons other than schema version. */ - malformed: number - /** Raw suffix beginning at the first malformed line, if any. */ - remainder?: string - /** Distinguishes an absent/empty log from bytes that could not name an epoch. */ - hasBytes: boolean -} - -export type JournalSnapshotReadResult = - | { status: 'missing' } - | { status: 'valid'; snapshot: JournalSnapshotFile } - | { status: 'invalid' } - /** A future schema version: unreadable by this build, not corrupt. The file - * stays authoritative in place and the caller degrades to read-only. */ - | { status: 'unreadable' } - -const NEWLINE_BYTE = 0x0a - -export async function ensureJournalDir(journalDir: string): Promise { - await mkdir(journalDir, { recursive: true }) -} - -export async function readJournalSnapshot(journalDir: string): Promise { - try { - const raw = await readFile(join(journalDir, JOURNAL_SNAPSHOT_FILE), 'utf-8') - const parsed: unknown = JSON.parse(raw) - const version = snapshotSchemaVersion(parsed) - if (version === null) { - return { status: 'invalid' } - } - // Version is classified BEFORE shape validation, matching row admission: a - // version only advances because bodies changed, so a valid newer snapshot - // carries kinds this build cannot parse — unreadable, never corruption. - if (version > AGENT_SESSION_JOURNAL_SCHEMA_VERSION) { - return { status: 'unreadable' } - } - return isJournalSnapshotFile(parsed) - ? { status: 'valid', snapshot: parsed } - : { status: 'invalid' } - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return { status: 'missing' } - } - if (error instanceof SyntaxError) { - return { status: 'invalid' } - } - throw error - } -} - -export async function quarantineInvalidJournalSnapshot( - journalDir: string, - quota?: { sessionId: string; maxBytes: number } -): Promise { - // Rename is normally same-filesystem and size-neutral, but admission must - // happen before retaining evidence so a full journal never creates an - // unbounded quarantine artifact (or relies on a copy fallback). - if (quota) { - const source = join(journalDir, JOURNAL_SNAPSHOT_FILE) - // Account for the complete source bytes: rename is usually neutral, but a - // cross-device/filesystem fallback may briefly retain both inodes. - const sourceBytes = await stat(source) - .then((info) => info.size) - .catch((error) => { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return 0 - } - throw error - }) - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: sourceBytes - }) - } - const source = join(journalDir, JOURNAL_SNAPSHOT_FILE) - const target = join(journalDir, `quarantine-snapshot-${Date.now()}-${randomUUID()}.json`) - await renameDurable(source, target) - return target -} - -export async function writeJournalSnapshotFile( - journalDir: string, - snapshot: JournalSnapshotFile -): Promise { - const target = join(journalDir, JOURNAL_SNAPSHOT_FILE) - await writeFileDurable(durableWriteTempPath(target), target, JSON.stringify(snapshot)) -} - -export async function readJournalLog(journalDir: string): Promise { - let raw: string - try { - raw = await readFile(join(journalDir, JOURNAL_LOG_FILE), 'utf-8') - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return { rows: [], unreadable: false, malformed: 0, hasBytes: false } - } - throw error - } - const rows: JournalRow[] = [] - let unreadable = false - let malformed = 0 - const lines = raw.split('\n') - let offset = 0 - for (const line of lines) { - if (!line.trim()) { - offset += line.length + 1 - continue - } - const parsed = parseJournalRow(line) - if (parsed.ok) { - rows.push(parsed.row) - offset += line.length + 1 - continue - } - if (parsed.unreadable) { - unreadable = true - return { rows, unreadable, malformed, remainder: raw.slice(offset), hasBytes: raw.length > 0 } - } - malformed += 1 - return { rows, unreadable, malformed, remainder: raw.slice(offset), hasBytes: raw.length > 0 } - } - return { rows, unreadable, malformed, hasBytes: raw.length > 0 } -} - -/** Row admission requires an integer version of at least 1; a snapshot whose - * version cannot even be read is malformed, not a schema statement. */ -function snapshotSchemaVersion(value: unknown): number | null { - const snapshot = recordOf(value) - const version = snapshot?.v - return typeof version === 'number' && Number.isInteger(version) && version >= 1 ? version : null -} - -function isJournalSnapshotFile(value: unknown): value is JournalSnapshotFile { - if (!value || typeof value !== 'object' || Array.isArray(value)) { - return false - } - const snapshot = value as Record - return ( - typeof snapshot.v === 'number' && - typeof snapshot.epoch === 'string' && - snapshot.epoch.length > 0 && - Number.isInteger(snapshot.compactedThrough) && - (snapshot.compactedThrough as number) >= 0 && - Number.isInteger(snapshot.highestFence) && - // Deep discriminated admission: a JSON-valid item with a corrupt nested - // shape (e.g. a question whose options are null) must land this snapshot - // in quarantine rather than throw later in projection or prompt render. - arrayOf(snapshot.items, isAdmissibleAgentJournalRenderItem) && - arrayOf(snapshot.submissions, isAdmissibleAgentJournalSubmission) && - arrayOf(snapshot.receipts, isReceipt) && - arrayOf(snapshot.aliases, isAlias) && - // Older snapshots predate tombstones; absence is fine, a non-array is not — - // seeding iterates this collection, so a JSON-valid wrong shape must land - // in quarantine rather than throw through startup restoration. - (snapshot.tombstones === undefined || arrayOf(snapshot.tombstones, isTombstone)) && - (snapshot.appliedSettlementIds === undefined || - arrayOf(snapshot.appliedSettlementIds, (entry) => typeof entry === 'string')) && - arrayOf(snapshot.tail, (row) => parseJournalRow(JSON.stringify(row)).ok) - ) -} - -function arrayOf(value: unknown, predicate: (entry: unknown) => boolean): value is unknown[] { - return Array.isArray(value) && value.every(predicate) -} - -function recordOf(value: unknown): Record | null { - return value && typeof value === 'object' && !Array.isArray(value) - ? (value as Record) - : null -} - -function isTombstone(value: unknown): boolean { - const tombstone = recordOf(value) - return Boolean( - tombstone && typeof tombstone.itemId === 'string' && Number.isInteger(tombstone.revision) - ) -} - -function isReceipt(value: unknown): boolean { - const receipt = recordOf(value) - return Boolean( - receipt && - typeof receipt.clientMessageId === 'string' && - typeof receipt.providerItemId === 'string' && - typeof receipt.epoch === 'string' && - typeof receipt.sequence === 'number' && - typeof receipt.acceptedAt === 'number' - ) -} - -function isAlias(value: unknown): boolean { - const alias = recordOf(value) - return Boolean( - alias && typeof alias.providerItemId === 'string' && typeof alias.itemId === 'string' - ) -} - -/** - * Append rows and fsync before returning. The caller treats a resolved promise - * as "this row survives a power loss" — the write-ahead submission row depends - * on exactly that, so this must never be relaxed to a buffered write. - */ -export async function appendJournalRows( - journalDir: string, - rows: readonly JournalRow[] -): Promise { - if (rows.length === 0) { - return - } - const path = join(journalDir, JOURNAL_LOG_FILE) - // A process death can leave a final JSON fragment without its newline. Never - // concatenate a new durable row onto that fragment: truncate the torn tail - // first, then fsync the repair before acknowledging this append. - try { - await repairJournalLogTail(path) - } catch (error) { - if ((error as NodeJS.ErrnoException).code !== 'ENOENT') { - throw error - } - // The append below creates a missing log. - } - const payload = `${rows.map(serializeJournalRow).join('\n')}\n` - await appendFile(path, payload, 'utf-8') - const handle = await open(path, 'r+') - try { - await handle.sync() - } finally { - await handle.close() - } - let directory: FileHandle | undefined - try { - directory = await open(journalDir, 'r') - await directory.sync() - } catch { - // Directory fsync is unavailable on some platforms (notably Windows). - } finally { - // The tolerance above must not leak the descriptor when open succeeded - // but sync failed — one leaked handle per append adds up fast. - await directory?.close().catch(() => undefined) - } -} - -/** Repair only a torn final row. The normal append path reads one byte; scanning - * backward is reserved for the crash-recovery case and never rereads the log. */ -async function repairJournalLogTail(path: string): Promise { - const handle = await open(path, 'r+') - try { - const { size } = await handle.stat() - if (size === 0) { - return - } - const lastByte = Buffer.alloc(1) - await handle.read(lastByte, 0, 1, size - 1) - if (lastByte[0] === NEWLINE_BYTE) { - return - } - - const scanChunkBytes = 64 * 1024 - let scanEnd = size - let boundary = -1 - while (scanEnd > 0 && boundary === -1) { - const scanStart = Math.max(0, scanEnd - scanChunkBytes) - const chunk = Buffer.alloc(scanEnd - scanStart) - await handle.read(chunk, 0, chunk.length, scanStart) - const newline = chunk.lastIndexOf(NEWLINE_BYTE) - if (newline !== -1) { - boundary = scanStart + newline - } - scanEnd = scanStart - } - - const lineStart = boundary + 1 - const finalLine = Buffer.alloc(size - lineStart) - await handle.read(finalLine, 0, finalLine.length, lineStart) - // A whole row that merely lost its newline is kept; a real fragment goes. - const complete = parseJournalRow(finalLine.toString('utf-8')).ok - await (complete ? handle.write('\n', size) : handle.truncate(lineStart)) - await handle.sync() - } finally { - await handle.close() - } -} - -export async function quarantineJournalRemainder( - journalDir: string, - remainder: string -): Promise { - const path = join(journalDir, `quarantine-${Date.now()}-${randomUUID()}.jsonl`) - await writeFileDurable(durableWriteTempPath(path), path, remainder) - return path -} - -/** Replace the log with exactly the retained tail. Runs only after the snapshot - * carrying that tail is durable, so a crash here loses nothing. */ -export async function rewriteJournalLog( - journalDir: string, - rows: readonly JournalRow[] -): Promise { - const target = join(journalDir, JOURNAL_LOG_FILE) - const payload = rows.length ? `${rows.map(serializeJournalRow).join('\n')}\n` : '' - await writeFileDurable(durableWriteTempPath(target), target, payload) -} diff --git a/src/main/native-chat/agent-session-journal/journal-open.ts b/src/main/native-chat/agent-session-journal/journal-open.ts index 94fb3137f26..350d2e9cfb9 100644 --- a/src/main/native-chat/agent-session-journal/journal-open.ts +++ b/src/main/native-chat/agent-session-journal/journal-open.ts @@ -1,194 +1,205 @@ -// Loading a journal from disk: snapshot + log → folded state. +// Loading a journal: the session projection names the live epoch, and that +// epoch's rows are folded through the reducer in sequence order. // -// The snapshot is authoritative for the current epoch. Log rows belonging to a -// superseded epoch are dropped rather than merged — a crash between publishing -// a rollover snapshot and rewriting the log is the ordinary way that happens. -// A gap in the surviving sequence is corruption, and the caller rolls the epoch +// There is no snapshot to anchor to and no superseded-epoch rows to drop — a +// roll deletes them in the same transaction that publishes the new epoch. A gap +// in the surviving sequence is corruption, and the caller rolls the epoch // rather than rendering a partial timeline. -import type { AgentJournalSubmission } from '../../../shared/agent-session-journal-types' +import { existsSync } from 'node:fs' +import type Database from '../../sqlite/sync-database' import { findSequenceGap } from './journal-cursor' -import { - quarantineInvalidJournalSnapshot, - readJournalLog, - readJournalSnapshot, - type JournalSnapshotFile -} from './journal-log-file' +import { openJournalDatabase } from './journal-database' +import { journalDatabaseFile } from './journal-paths' import { applyJournalRow, createJournalReducerState, - rememberAppliedSettlementId, type JournalReducerState } from './journal-reducer' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' +import { + readJournalEpochRows, + readJournalRowsAfter, + readJournalSessionEpoch +} from './journal-row-table' +import { JOURNAL_REPAIR_DISCLOSURE_ITEM_ID } from './journal-repair-disclosure' +import { pendingJournalRepairSequence } from './journal-repair-marker' +import { parseJournalRow, type JournalRow } from './journal-row-schema' + +/** Every epoch row is sequence 1, and no compaction moves that floor. */ +const FIRST_JOURNAL_SEQUENCE = 1 export type JournalLoad = { state: JournalReducerState - /** Rows still individually replayable, oldest first. */ - tailRows: JournalRow[] - /** Highest sequence folded into the snapshot; the tail starts after it. */ - compactedThrough: number - /** A future schema version was met: no writes, no compaction, no deletion. */ + /** A future schema version was met: no writes, no deletion. */ readOnly: boolean /** Set when the surviving prefix is unusable and the caller must roll the epoch. */ corrupt: boolean - /** Log lines skipped because they failed to parse (schema-version rows are + /** Rows skipped because their body failed to parse (future-version rows are * `readOnly`, never counted here). The store discloses these in the timeline. */ malformedRows: number - sizeBytes: number - /** Raw unreadable suffix retained for quarantine instead of deletion. */ - quarantineRemainder?: string + /** Directory-internal: the first sequence of an unusable suffix. The store + * deletes from here before it accepts a write; a probe leaves it alone. */ + truncateFrom?: number } -/** Returns null when no journal exists yet for this session. */ -export async function loadJournal( - journalDir: string, - sessionId: string, - quota?: { maxBytes: number } -): Promise { - const snapshotRead = await readJournalSnapshot(journalDir) - if (snapshotRead.status === 'unreadable') { - // Written by a newer schema: not corrupt, so never quarantined. The file - // stays authoritative in place, and this build must not write, compact, - // delete, or render a partial timeline from rows it cannot anchor to the - // snapshot it cannot read. +/** + * Replay on a connection this function does NOT own. Returns null when the + * session has no journal yet. + */ +export function replayJournal( + db: Database.Database, + readOnly: boolean, + sessionId: string +): JournalLoad | null { + if (readOnly) { return emptyReadOnlyLoad(sessionId) } - if (snapshotRead.status === 'invalid') { - await quarantineInvalidJournalSnapshot( - journalDir, - quota ? { sessionId, maxBytes: quota.maxBytes } : undefined - ) - } - const snapshot = snapshotRead.status === 'valid' ? snapshotRead.snapshot : null - const log = await readJournalLog(journalDir) - const epoch = resolveEpoch(snapshot, log.rows) + const epoch = readJournalSessionEpoch(db, sessionId) if (!epoch) { - return snapshotRead.status === 'invalid' || log.hasBytes ? emptyReadOnlyLoad(sessionId) : null + return null + } + const state = createJournalReducerState(sessionId, epoch) + const stored = readJournalEpochRows(db, sessionId, epoch) + // A partial repair keeps its prefix, so the surviving rows look contiguous and + // anchored however much of the timeline it deleted. Its marker is what still + // says otherwise, naming the sequence past which the epoch would be its own + // history again. + const repairedFrom = pendingJournalRepairSequence(db, sessionId, epoch) + const rows: JournalRow[] = [] + let malformedRows = 0 + let latched = false + let truncateFrom: number | undefined + for (const entry of stored) { + const parsed = parseJournalRow(entry.rowJson) + if (parsed.ok) { + rows.push(parsed.row) + continue + } + // Reading STOPS at the first row this build cannot represent. A future + // version latches read-only; anything else is one skipped row, disclosed. + truncateFrom = entry.seq + if (parsed.unreadable) { + latched = true + } else { + malformedRows = 1 + } + break } - const compactedThrough = snapshot?.epoch === epoch ? snapshot.compactedThrough : 0 - const state = seedState(sessionId, epoch, snapshot?.epoch === epoch ? snapshot : null) - const liveRows = log.rows.filter((row) => row.epoch === epoch) - let tailRows = unionBySequence(snapshot?.epoch === epoch ? snapshot.tail : [], liveRows, epoch) - - const oldest = tailRows[0]?.seq ?? compactedThrough + 1 + // Anchored at 1, never at the first row that HAPPENS to remain: nothing trims + // a prefix, so a missing epoch row is a hole like any other and everything + // behind it is unanchored. Validating from `rows[0].seq` would call the + // leftovers contiguous and leave them out of the repair that runs before + // provider history replaces the epoch. const gap = findSequenceGap( - tailRows.map((row) => row.seq), - oldest + rows.map((row) => row.seq), + FIRST_JOURNAL_SEQUENCE ) - // A hole below the snapshot boundary is unrecoverable too: the snapshot only - // covers `compactedThrough`, so a tail that starts above it lost rows. - let corrupt = Boolean(gap) || oldest > compactedThrough + 1 || log.malformed > 0 - let quarantineRemainder = log.remainder if (gap) { - const firstBad = tailRows.findIndex((row, index) => { - const expected = (tailRows[0]?.seq ?? compactedThrough + 1) + index - return row.seq !== expected - }) + const firstBad = rows.findIndex((row, index) => row.seq !== FIRST_JOURNAL_SEQUENCE + index) if (firstBad !== -1) { - const suffix = tailRows.slice(firstBad) - tailRows = tailRows.slice(0, firstBad) - quarantineRemainder ??= `${suffix.map((row) => JSON.stringify(row)).join('\n')}\n` + truncateFrom = rows[firstBad]?.seq ?? truncateFrom + rows.length = firstBad } } - - for (const row of tailRows) { - if (row.seq > compactedThrough) { - applyJournalRow(state, row) - } + // Contiguity from 1 is not the whole invariant: sequence 1 has to BE the epoch + // row. An ordinary row there is an epoch nothing anchors, and replaying it as + // clean is how a repaired journal silently adopts a timeline whose real + // history was never rebuilt. + if (rows.length > 0 && rows[0]?.kind !== 'epoch') { + truncateFrom = rows[0]?.seq ?? truncateFrom + rows.length = 0 } - state.oldestSequence = oldest - state.lastSequence = Math.max(state.lastSequence, compactedThrough) + for (const row of rows) { + applyJournalRow(state, row) + } + state.oldestSequence = FIRST_JOURNAL_SEQUENCE + // A latched journal reduces to nothing by design; only a writable one can be + // held to the anchor. + const unanchored = !latched && rows[0]?.kind !== 'epoch' return { state, - tailRows, - compactedThrough, - // A future-version snapshot never reaches here: it is classified - // unreadable above, so `valid` implies a version this build can write. - readOnly: log.unreadable, - corrupt, - malformedRows: log.malformed, - sizeBytes: tailRows.reduce((total, row) => total + journalRowByteLength(row), 0), - quarantineRemainder + readOnly: latched, + corrupt: + Boolean(gap) || + malformedRows > 0 || + unanchored || + (repairedFrom !== null && awaitsRebuild(rows, repairedFrom)) || + awaitsProviderHistory(rows), + malformedRows, + ...(truncateFrom !== undefined && !latched ? { truncateFrom } : {}) + } +} + +/** + * The epoch a total repair published, still holding nothing but its own anchor + * and disclosure. The rows it dropped were never reconstructed, so provider + * history has to be retried rather than this being called a clean timeline. + */ +function awaitsProviderHistory(rows: readonly JournalRow[]): boolean { + const anchor = rows[0] + if (anchor?.kind !== 'epoch' || anchor.reason !== 'unreconcilable_prefix') { + return false + } + // The anchor sits at sequence 1, so content of the epoch's own starts at 2. + return awaitsRebuild(rows, FIRST_JOURNAL_SEQUENCE + 1) +} + +/** + * True while everything at or above `contentFrom` is the repair's own + * bookkeeping: the deleted history was never rebuilt, so the provider has to be + * asked again. The moment the session writes content of its own past that + * sequence the epoch IS its own history, and the retry stops rather than a + * later import replacing rows the user has since seen. + */ +function awaitsRebuild(rows: readonly JournalRow[], contentFrom: number): boolean { + return rows.every( + (row) => + row.seq < contentFrom || + (row.kind === 'item' && row.itemId === JOURNAL_REPAIR_DISCLOSURE_ITEM_ID) + ) +} + +/** Rows after a cursor, in sequence order. Stops at the first row this build + * cannot parse, exactly as replay does. */ +export function readJournalRowsAfterCursor( + db: Database.Database, + sessionId: string, + epoch: string, + afterSequence: number +): JournalRow[] { + const rows: JournalRow[] = [] + for (const stored of readJournalRowsAfter(db, sessionId, epoch, afterSequence)) { + const parsed = parseJournalRow(stored.rowJson) + if (!parsed.ok) { + break + } + rows.push(parsed.row) + } + return rows +} + +/** Standalone probe. Opens its own connection and closes it before returning, + * so a caller holding only the returned value holds no handle. */ +export function loadJournal(journalDir: string, sessionId: string): JournalLoad | null { + const dbPath = journalDatabaseFile(journalDir) + if (!existsSync(dbPath)) { + return null + } + const opened = openJournalDatabase(dbPath) + try { + return replayJournal(opened.db, opened.readOnly, sessionId) + } finally { + opened.db.close() } } function emptyReadOnlyLoad(sessionId: string): JournalLoad { - const state = createJournalReducerState(sessionId, '') return { - state, - tailRows: [], - compactedThrough: 0, + state: createJournalReducerState(sessionId, ''), readOnly: true, corrupt: false, - malformedRows: 0, - sizeBytes: 0 + malformedRows: 0 } } - -/** The snapshot names the live epoch; without one, the newest valid row does. */ -function resolveEpoch(snapshot: JournalSnapshotFile | null, rows: JournalRow[]): string | null { - if (snapshot?.epoch) { - return snapshot.epoch - } - return rows.at(-1)?.epoch ?? null -} - -function seedState( - sessionId: string, - epoch: string, - snapshot: JournalSnapshotFile | null -): JournalReducerState { - const state = createJournalReducerState(sessionId, epoch) - if (!snapshot) { - return state - } - for (const item of snapshot.items) { - state.items.set(item.itemId, item) - } - for (const submission of snapshot.submissions) { - state.submissions.set(submission.clientMessageId, { ...submission } as AgentJournalSubmission) - } - for (const receipt of snapshot.receipts) { - state.receipts.set(receipt.clientMessageId, { - clientMessageId: receipt.clientMessageId, - providerItemId: receipt.providerItemId, - cursor: { epoch: receipt.epoch, sequence: receipt.sequence }, - acceptedAt: receipt.acceptedAt - }) - } - for (const alias of snapshot.aliases) { - state.aliases.set(alias.providerItemId, alias.itemId) - } - for (const tombstone of snapshot.tombstones ?? []) { - state.tombstones.set(tombstone.itemId, tombstone.revision) - } - for (const settlementId of snapshot.appliedSettlementIds ?? []) { - rememberAppliedSettlementId(state, settlementId) - } - state.highestFence = snapshot.highestFence ?? 0 - state.lastSequence = snapshot.compactedThrough - state.oldestSequence = snapshot.compactedThrough + 1 - return state -} - -/** Merge the snapshot's retained tail with the live log, preferring the log's - * copy of any sequence both hold, and dropping rows from a superseded epoch. */ -function unionBySequence( - retained: readonly JournalRow[], - live: readonly JournalRow[], - epoch: string -): JournalRow[] { - const bySequence = new Map() - for (const row of retained) { - if (row.epoch === epoch) { - bySequence.set(row.seq, row) - } - } - for (const row of live) { - bySequence.set(row.seq, row) - } - return [...bySequence.values()].sort((a, b) => a.seq - b.seq) -} diff --git a/src/main/native-chat/agent-session-journal/journal-paths.ts b/src/main/native-chat/agent-session-journal/journal-paths.ts index 87616b25b96..c21a7675f00 100644 --- a/src/main/native-chat/agent-session-journal/journal-paths.ts +++ b/src/main/native-chat/agent-session-journal/journal-paths.ts @@ -40,3 +40,10 @@ export function journalDirectoryFor( export function defaultJournalRoot(): Promise { return Promise.resolve(getAppEnvironment().getPath('userData')) } + +export const JOURNAL_DATABASE_FILE = 'journal.db' + +/** The session's SQLite database, inside the directory `journalDirectoryFor` names. */ +export function journalDatabaseFile(journalDir: string): string { + return join(journalDir, JOURNAL_DATABASE_FILE) +} diff --git a/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts b/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts index b47d1a1511c..ea09ecf6df3 100644 --- a/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts +++ b/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts @@ -1,11 +1,9 @@ // Payload bounds for tool output and diffs. // -// A looping agent must not be able to fill the host disk, and a 40 MB tool -// result must not be inlined into a row that every reconnecting client -// replays. A bounded payload keeps a head plus the original byte length and -// digest; the remainder lives in the content-addressed blob store under the -// same retention as its epoch. Crossing a bound is always marked — never a -// silent drop. +// A 40 MB tool result must not be inlined into a row that every reconnecting +// client replays. A bounded payload keeps a head plus the original byte length +// and digest; the remainder is discarded. Crossing a bound is always marked — +// never a silent drop. import { createHash } from 'node:crypto' import type { AgentJournalBoundedPayload } from '../../../shared/agent-session-journal-types' @@ -13,18 +11,10 @@ import type { AgentJournalBoundedPayload } from '../../../shared/agent-session-j export type JournalPayloadLimits = { /** Bytes of the payload kept inline on the row. */ inlineHeadBytes: number - /** Total bytes of journal rows one session may hold before appends are refused. */ - maxSessionBytes: number - /** Appends allowed inside `appendWindowMs`, bounding a runaway agent's rate. */ - maxAppendsPerWindow: number - appendWindowMs: number } export const DEFAULT_JOURNAL_PAYLOAD_LIMITS: JournalPayloadLimits = { - inlineHeadBytes: 16 * 1024, - maxSessionBytes: 256 * 1024 * 1024, - maxAppendsPerWindow: 5000, - appendWindowMs: 60_000 + inlineHeadBytes: 16 * 1024 } /** Marker appended to a clipped inline string so the UI never presents a @@ -38,10 +28,8 @@ export function digestPayload(payload: string): string { return createHash('sha256').update(payload, 'utf8').digest('hex') } -/** - * Clip `payload` to the inline head. `truncated` means the remainder must be - * written to the blob store under `digest` before the row is appended. - */ +/** Clip `payload` to the inline head. `truncated` means the remainder was + * discarded; `digest` and `byteLength` describe the original. */ export function boundPayload( payload: string, limits: JournalPayloadLimits @@ -60,7 +48,7 @@ export function boundPayload( } /** Bound a plain string that must stay a string (a tool-result block's output), - * keeping the explicit marker inline. Returns the blob payload to persist. */ + * keeping the explicit marker inline. */ export function boundInlineText( payload: string, limits: JournalPayloadLimits @@ -75,7 +63,7 @@ export function boundInlineText( } } -/** Keep arbitrary tool input JSON bounded before lifecycle admission. */ +/** Keep arbitrary tool input JSON bounded before it reaches a row. */ export function boundToolInput(input: unknown, limits: JournalPayloadLimits): unknown { let encoded: string try { diff --git a/src/main/native-chat/agent-session-journal/journal-physical-quota.test.ts b/src/main/native-chat/agent-session-journal/journal-physical-quota.test.ts deleted file mode 100644 index b3ee4699fc0..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-physical-quota.test.ts +++ /dev/null @@ -1,125 +0,0 @@ -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' -import type { - AgentJournalItemBody, - AgentJournalItemIdentity, - AgentSessionJournalIdentity -} from '../../../shared/agent-session-journal-types' -import { JOURNAL_SNAPSHOT_FILE } from './journal-log-file' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalDirectoryBytes } from './journal-physical-quota' -import { openAgentSessionJournal } from './journal-store-factory' - -const IDENTITY: AgentSessionJournalIdentity = { - sessionId: 'session-1', - workspaceId: 'ws-1', - hostId: 'host-1', - agent: 'codex', - providerHandle: { kind: 'codex', threadId: 'thread-1' } -} - -let root: string - -function item(ordinal: number): AgentJournalItemIdentity { - return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } -} - -function body(value: string): AgentJournalItemBody { - return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } -} - -beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-quota-')) -}) - -afterEach(async () => { - await rm(root, { recursive: true, force: true }) -}) - -describe('journal physical quota peaks', () => { - it('refuses an epoch replacement whose staging peak exceeds the quota', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 8_000 } - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits, - autoCompact: false - }) - await journal.appendItem(item(1), body('old'.repeat(500)), { fence: 1 }) - const epoch = journal.epoch - - await expect( - journal.replaceEpochItems('handle_forked', 2, [ - { identity: item(2), body: body('replacement'.repeat(250)) } - ]) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toBe(epoch) - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) - }) - - it('refuses schema quarantine when its peak copy would exceed the quota', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 7_000 } - await openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - snapshot.items = [{ body: { kind: 'future', payload: 'x'.repeat(4_000) } }] - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - const reopened = await openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - - await expect(reopened.rollEpoch('schema_unreadable', 2)).rejects.toMatchObject({ - code: 'journal_bound_exceeded' - }) - - expect(reopened.isReadOnly).toBe(true) - expect((await readdir(root)).some((name) => name.startsWith('quarantine-'))).toBe(false) - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) - }) - - it('does not rename an invalid snapshot when the directory is already full', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - await openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - await writeFile(snapshotPath, '{"invalid":', 'utf8') - const current = await journalDirectoryBytes(root) - await writeFile( - join(root, 'quota-filler'), - 'x'.repeat(Math.max(0, limits.maxSessionBytes - current)), - 'utf8' - ) - - await expect( - openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - expect(await readFile(snapshotPath, 'utf8')).toBe('{"invalid":') - expect((await readdir(root)).some((name) => name.startsWith('quarantine-snapshot-'))).toBe( - false - ) - }) - - it('counts pre-existing durable-write temps while staging an epoch replacement', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 8_000 } - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits, - autoCompact: false - }) - await journal.appendItem(item(1), body('old'), { fence: 1 }) - // Simulate a temp left by a crash. Replacement must refuse before writing - // its epoch row or creating a staging blob beside this file. - await writeFile(join(root, 'snapshot.json.crashed-write.tmp'), 'x'.repeat(7_500), 'utf8') - const epoch = journal.epoch - - await expect( - journal.replaceEpochItems('handle_forked', 2, [{ identity: item(2), body: body('new') }]) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toBe(epoch) - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-physical-quota.ts b/src/main/native-chat/agent-session-journal/journal-physical-quota.ts deleted file mode 100644 index 478451b615c..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-physical-quota.ts +++ /dev/null @@ -1,41 +0,0 @@ -import { lstat, readdir } from 'node:fs/promises' -import type { Dirent } from 'node:fs' -import { join } from 'node:path' -import { AgentSessionJournalError } from './journal-write-guards' - -/** Counts every physical file owned by one session, including blobs, durable - * write temps, and retained quarantine evidence. Symlinks are charged as files - * but never followed outside the journal directory. */ -export async function journalDirectoryBytes(directory: string): Promise { - let entries: Dirent[] - try { - entries = await readdir(directory, { withFileTypes: true, encoding: 'utf8' }) - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return 0 - } - throw error - } - let total = 0 - for (const entry of entries) { - const path = join(directory, entry.name) - total += entry.isDirectory() ? await journalDirectoryBytes(path) : (await lstat(path)).size - } - return total -} - -export async function assertJournalPhysicalCapacity(input: { - journalDir: string - sessionId: string - maxBytes: number - peakAdditionalBytes?: number -}): Promise { - const current = await journalDirectoryBytes(input.journalDir) - if (current + (input.peakAdditionalBytes ?? 0) > input.maxBytes) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${input.sessionId} reached its ${input.maxBytes}-byte physical bound` - ) - } - return current -} diff --git a/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts b/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts index fb8243b6926..58d12dfcf89 100644 --- a/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts +++ b/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts @@ -12,10 +12,7 @@ import { export const MAX_JOURNAL_PROMPT_OPTIONS = 64 -const JOURNAL_PROMPT_OPTION_LIMITS = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 1024 -} +const JOURNAL_PROMPT_OPTION_LIMITS = { inlineHeadBytes: 1024 } const JOURNAL_PROMPT_ID_MAX_BYTES = 1024 export function cancelledJournalPromptBody( @@ -78,11 +75,6 @@ function boundPromptIdentifier(value: string): string { if (Buffer.byteLength(value, 'utf8') <= JOURNAL_PROMPT_ID_MAX_BYTES) { return value } - const bounded = boundPayload(value, { - inlineHeadBytes: JOURNAL_PROMPT_ID_MAX_BYTES - 33, - maxSessionBytes: Number.MAX_SAFE_INTEGER, - maxAppendsPerWindow: Number.MAX_SAFE_INTEGER, - appendWindowMs: Number.MAX_SAFE_INTEGER - }) + const bounded = boundPayload(value, { inlineHeadBytes: JOURNAL_PROMPT_ID_MAX_BYTES - 33 }) return `${bounded.head}#${bounded.digest.slice(0, 32)}` } diff --git a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts index cbdc35a4698..676a37f63a7 100644 --- a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts @@ -11,7 +11,6 @@ import { applyJournalRow, createJournalReducerState, MAX_JOURNAL_APPLIED_SETTLEMENT_IDS, - referencedBlobDigests, renderJournalState, type JournalReducerState } from './journal-reducer' @@ -379,58 +378,6 @@ describe('lifecycle settlement deduplication', () => { }) }) -describe('blob retention', () => { - it('reports the digests live rows still reference', () => { - const state = fold([ - { - kind: 'item', - itemId: 'tool', - revision: 1, - body: { - kind: 'tool-call', - name: 'bash', - input: {}, - state: 'completed', - output: { head: 'x', byteLength: 999, digest: 'digest-a', truncated: true } - }, - ...base(1) - }, - { - kind: 'item', - itemId: 'inline', - revision: 1, - body: { - kind: 'tool-call', - name: 'bash', - input: {}, - state: 'completed', - output: { head: 'y', byteLength: 1, digest: 'digest-b', truncated: false } - }, - ...base(2) - } - ]) - expect([...referencedBlobDigests(state)]).toEqual(['digest-a']) - }) - - it('stops referencing a digest once its item is tombstoned', () => { - const state = fold([ - { - kind: 'item', - itemId: 'tool', - revision: 1, - body: { - kind: 'diff', - path: 'a.ts', - patch: { head: 'x', byteLength: 999, digest: 'digest-a', truncated: true } - }, - ...base(1) - }, - { kind: 'tombstone', itemId: 'tool', revision: 2, ...base(2) } - ]) - expect(referencedBlobDigests(state).size).toBe(0) - }) -}) - describe('malformed persisted item keys', () => { it('degrades a malformed-percent item id to an opaque key instead of throwing', () => { // A user-message body drives identity resolution through the key parser; diff --git a/src/main/native-chat/agent-session-journal/journal-reducer.ts b/src/main/native-chat/agent-session-journal/journal-reducer.ts index c660f80611f..3dc4385d797 100644 --- a/src/main/native-chat/agent-session-journal/journal-reducer.ts +++ b/src/main/native-chat/agent-session-journal/journal-reducer.ts @@ -9,7 +9,6 @@ import type { AgentJournalAcceptanceReceipt, - AgentJournalItemBody, AgentJournalRenderItem, AgentJournalSnapshot, AgentJournalSubmission @@ -272,26 +271,3 @@ export function renderJournalState(state: JournalReducerState): AgentJournalSnap submissions: [...state.submissions.values()].sort((a, b) => a.submittedAt - b.submittedAt) } } - -/** Blob digests one body points at. A retained row can outlive its render item - * (a tombstone drops the item), so compaction reads rows through this too. */ -export function blobDigestsInBody(body: AgentJournalItemBody, into: Set): void { - if (body.kind === 'tool-call' && body.output?.truncated) { - into.add(body.output.digest) - } - if (body.kind === 'diff' && body.patch.truncated) { - into.add(body.patch.digest) - } - if (body.kind === 'status' && body.providerFrame?.payload.truncated) { - into.add(body.providerFrame.payload.digest) - } -} - -/** Digests referenced by live rows, so compaction knows which blobs to keep. */ -export function referencedBlobDigests(state: JournalReducerState): Set { - const digests = new Set() - for (const item of state.items.values()) { - blobDigestsInBody(item.body, digests) - } - return digests -} diff --git a/src/main/native-chat/agent-session-journal/journal-repair-disclosure.ts b/src/main/native-chat/agent-session-journal/journal-repair-disclosure.ts new file mode 100644 index 00000000000..fee5edf142b --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-repair-disclosure.ts @@ -0,0 +1,32 @@ +// What a repair tells the user it did. +// +// The identity is a constant because replay reads it back: an epoch holding +// nothing but its anchor and this row is a repair that has not been +// reconstructed yet, not a timeline. + +import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import type { AgentJournalItemIdentity } from '../../../shared/agent-session-journal-types' + +/** One stable identity, so a reopen upserts the same row instead of adding one. */ +export const JOURNAL_REPAIR_DISCLOSURE_IDENTITY: AgentJournalItemIdentity = { + provider: 'orca', + clientMessageId: 'journal-malformed-lines' +} + +export const JOURNAL_REPAIR_DISCLOSURE_ITEM_ID = agentJournalItemKey( + JOURNAL_REPAIR_DISCLOSURE_IDENTITY +) + +export type JournalRepairDisclosure = { + identity: AgentJournalItemIdentity + body: { kind: 'status'; text: string } +} + +/** Disclosed when a repair skipped a row it could not read. */ +export function journalRepairDisclosure(input: { malformedRows: number }): JournalRepairDisclosure { + const lines = `${input.malformedRows} journal line${input.malformedRows === 1 ? '' : 's'}` + return { + identity: JOURNAL_REPAIR_DISCLOSURE_IDENTITY, + body: { kind: 'status', text: `${lines} could not be read` } + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-repair-marker.ts b/src/main/native-chat/agent-session-journal/journal-repair-marker.ts new file mode 100644 index 00000000000..f9ef09e688d --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-repair-marker.ts @@ -0,0 +1,69 @@ +// The standing demand for a rebuild that a repair leaves behind. +// +// A repair that empties the epoch republishes an `unreconcilable_prefix` anchor, +// and replay reads that back as history still owed. A repair that KEEPS a prefix +// has no such anchor to publish and — for a plain sequence gap — no malformed +// row to disclose either, so nothing on disk would record that the deleted +// suffix was never reconstructed. This marker is that record, written in the +// SAME transaction as the deletion: a crash between the two would otherwise +// leave the rows gone with nothing left asking for them back. +// +// It records the first sequence at which the epoch would hold content of its +// own again, because it retires under exactly the rule the emptied-epoch anchor +// takes: a fresh epoch carries the rebuild, and a session that writes past that +// sequence owns the epoch and stops the retry. + +import type Database from '../../sqlite/sync-database' +import { deleteJournalRowSuffix } from './journal-row-table' + +const SELECT_REPAIR = 'SELECT epoch, content_from FROM journal_repairs WHERE session_id = ?' +const UPSERT_REPAIR = `INSERT INTO journal_repairs (session_id, epoch, content_from, repaired_at) +VALUES (?, ?, ?, ?) +ON CONFLICT(session_id) DO UPDATE SET + epoch = excluded.epoch, content_from = excluded.content_from, repaired_at = excluded.repaired_at` +const DELETE_REPAIR = 'DELETE FROM journal_repairs WHERE session_id = ?' + +/** + * The sequence a pending repair on THIS epoch left free, or null when none is + * pending. Epoch-scoped: a marker raised on an epoch that has since been + * superseded says nothing about the live one. + */ +export function pendingJournalRepairSequence( + db: Database.Database, + sessionId: string, + epoch: string +): number | null { + const row = db.prepare(SELECT_REPAIR).get(sessionId) as + | { epoch?: string; content_from?: number } + | undefined + return row?.epoch === epoch ? (row.content_from ?? null) : null +} + +/** Retires the marker. Called from inside the epoch transactions, whose new + * epoch is the rebuilt history the marker was holding out for. */ +export function clearJournalRepairMarker(db: Database.Database, sessionId: string): void { + db.prepare(DELETE_REPAIR).run(sessionId) +} + +/** Drop the rejected suffix and record that it is owed, atomically. */ +export function deleteJournalRepairedSuffix(input: { + db: Database.Database + sessionId: string + epoch: string + /** First sequence of the rejected suffix. */ + fromSeq: number + /** First sequence left free once the suffix is gone. */ + contentFrom: number + now: number +}): number { + input.db.exec('BEGIN IMMEDIATE') + try { + const deleted = deleteJournalRowSuffix(input.db, input.sessionId, input.epoch, input.fromSeq) + input.db.prepare(UPSERT_REPAIR).run(input.sessionId, input.epoch, input.contentFrom, input.now) + input.db.exec('COMMIT') + return deleted + } catch (error) { + input.db.exec('ROLLBACK') + throw error + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-row-table.ts b/src/main/native-chat/agent-session-journal/journal-row-table.ts new file mode 100644 index 00000000000..306b3b300f2 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-row-table.ts @@ -0,0 +1,92 @@ +// Every statement the journal issues against `journal_rows` / `journal_sessions`. +// +// Each one is a prefix or range scan on the `(session_id, epoch, seq)` primary +// key; there is no secondary index, and no `max(seq)` tip query — replay folds +// the epoch to obtain `lastSequence`, so nothing needs the tip from SQL. +// Columns are always named: `SELECT *` is uncacheable and can drop a column. + +import type Database from '../../sqlite/sync-database' +import { serializeJournalRow, type JournalRow } from './journal-row-schema' + +export type JournalStoredRow = { epoch: string; seq: number; ts: number; rowJson: string } + +const SELECT_SESSION = 'SELECT epoch FROM journal_sessions WHERE session_id = ?' +const UPSERT_SESSION = `INSERT INTO journal_sessions (session_id, epoch, updated_at) +VALUES (?, ?, ?) +ON CONFLICT(session_id) DO UPDATE SET epoch = excluded.epoch, updated_at = excluded.updated_at` +const INSERT_ROW = + 'INSERT INTO journal_rows (session_id, epoch, seq, ts, row_json) VALUES (?, ?, ?, ?, ?)' +const SELECT_EPOCH_ROWS = `SELECT epoch, seq, ts, row_json FROM journal_rows +WHERE session_id = ? AND epoch = ? ORDER BY seq ASC` +const SELECT_ROWS_AFTER = `SELECT epoch, seq, ts, row_json FROM journal_rows +WHERE session_id = ? AND epoch = ? AND seq > ? ORDER BY seq ASC` +const DELETE_SUFFIX = 'DELETE FROM journal_rows WHERE session_id = ? AND epoch = ? AND seq >= ?' + +export function readJournalSessionEpoch(db: Database.Database, sessionId: string): string | null { + const row = db.prepare(SELECT_SESSION).get(sessionId) as { epoch?: string } | undefined + return row?.epoch ?? null +} + +export function upsertJournalSessionRow( + db: Database.Database, + sessionId: string, + epoch: string, + updatedAt: number +): void { + db.prepare(UPSERT_SESSION).run(sessionId, epoch, updatedAt) +} + +export function insertJournalRow( + db: Database.Database, + sessionId: string, + row: JournalRow +): number { + const rowJson = serializeJournalRow(row) + db.prepare(INSERT_ROW).run(sessionId, row.epoch, row.seq, row.ts, rowJson) + return Buffer.byteLength(rowJson, 'utf8') +} + +export function readJournalEpochRows( + db: Database.Database, + sessionId: string, + epoch: string +): JournalStoredRow[] { + return toStoredRows(db.prepare(SELECT_EPOCH_ROWS).all(sessionId, epoch)) +} + +export function readJournalRowsAfter( + db: Database.Database, + sessionId: string, + epoch: string, + afterSeq: number +): JournalStoredRow[] { + return toStoredRows(db.prepare(SELECT_ROWS_AFTER).all(sessionId, epoch, afterSeq)) +} + +/** + * Unqualified on purpose. One database per session means every row here belongs + * to this session, and the unqualified form takes SQLite's truncate + * optimization: measured at 0.26% of the database in WAL bytes where the + * `WHERE session_id = ?` form rewrote every emptied leaf at up to 99%. + */ +export function deleteAllJournalRows(db: Database.Database): void { + db.exec('DELETE FROM journal_rows') +} + +/** Drop the rejected suffix a repair found, from `fromSeq` to the tip. */ +export function deleteJournalRowSuffix( + db: Database.Database, + sessionId: string, + epoch: string, + fromSeq: number +): number { + const deleted = db.prepare(DELETE_SUFFIX).run(sessionId, epoch, fromSeq) + return Number(deleted.changes ?? 0) +} + +function toStoredRows(rows: readonly unknown[]): JournalStoredRow[] { + return rows.map((entry) => { + const record = entry as { epoch: string; seq: number; ts: number; row_json: string } + return { epoch: record.epoch, seq: record.seq, ts: record.ts, rowJson: record.row_json } + }) +} diff --git a/src/main/native-chat/agent-session-journal/journal-row-writer-read-only-latch.test.ts b/src/main/native-chat/agent-session-journal/journal-row-writer-read-only-latch.test.ts deleted file mode 100644 index 2b318bc5320..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-row-writer-read-only-latch.test.ts +++ /dev/null @@ -1,362 +0,0 @@ -import { mkdtemp, rm, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' -import type { AgentJournalItemBody } from '../../../shared/agent-session-journal-types' -import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' -import { readJournalBlob } from './journal-blob-store' -import { appendJournalRows } from './journal-log-file' -import { JournalLifecycleAdmission } from './journal-lifecycle-admission' -import { JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES } from './journal-lifecycle-capacity' -import { loadJournal } from './journal-open' -import { boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' -import { JournalRowWriter } from './journal-row-writer' -import { JournalAppendBudget } from './journal-write-guards' - -const SESSION_ID = 'session-1' - -function row(seq: number, ts: number): JournalRow { - return { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: 'epoch-1', - seq, - fence: 0, - ts, - kind: 'item', - itemId: 'item-1', - revision: 1, - body: { kind: 'status', text: 'ambiguous append' } - } -} - -function rowWithBlob(seq: number, ts: number, output: ReturnType): JournalRow { - return { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: 'epoch-1', - seq, - fence: 0, - ts, - kind: 'item', - itemId: 'item-with-blob', - revision: 1, - body: { - kind: 'tool-call', - name: 'shell', - input: {}, - state: 'completed', - output - } - } -} - -function runningToolRow(seq: number, ts: number, itemId = 'running-tool'): JournalRow { - return { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: 'epoch-1', - seq, - fence: 0, - ts, - kind: 'item', - itemId, - revision: 1, - body: runningToolBody() - } -} - -function runningToolBody(): AgentJournalItemBody { - return { kind: 'tool-call', name: 'shell', input: {}, state: 'running' } -} - -describe('journal row writer read-only latch', () => { - let root: string - let readOnly = false - - beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-row-writer-')) - readOnly = false - }) - - afterEach(async () => { - await rm(root, { recursive: true, force: true }) - }) - - it('enforces the lifecycle append rate and allows a retry after the window', () => { - const appendWindowMs = 100 - const budget = new JournalAppendBudget(SESSION_ID, { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxAppendsPerWindow: 1, - appendWindowMs - }) - - budget.assertLifecycle(row(1, 1), 0) - expect(() => budget.assertLifecycle(row(2, 1), 0)).toThrow( - expect.objectContaining({ code: 'journal_rate_exceeded' }) - ) - expect(() => budget.assertLifecycle(row(2, appendWindowMs + 1), 0)).not.toThrow() - }) - - it('refuses lifecycle reservations once aggregate append capacity is saturated', () => { - const admission = new JournalLifecycleAdmission(SESSION_ID, 1_000_000, (itemId) => itemId, 2) - expect(admission.reserve({ id: 'first', bytes: 1, appendSlots: 1 }, 0)).toBe(true) - expect(admission.reserve({ id: 'second', bytes: 1, appendSlots: 1 }, 0)).toBe(true) - expect(admission.reserve({ id: 'third', bytes: 1, appendSlots: 1 }, 0)).toBe(false) - }) - - function writerHarness( - overrides: { - limits?: typeof DEFAULT_JOURNAL_PAYLOAD_LIMITS - physicalBytes?: number - appendRows?: (journalDir: string, rows: readonly JournalRow[]) => Promise - commit?: (row: JournalRow, physicalBytes: number) => void - } = {} - ) { - const limits = overrides.limits ?? DEFAULT_JOURNAL_PAYLOAD_LIMITS - const lifecycleAdmission = new JournalLifecycleAdmission( - SESSION_ID, - limits.maxSessionBytes, - (itemId) => itemId - ) - let physicalBytes = overrides.physicalBytes ?? 0 - let nextSequence = 1 - const committedRows: JournalRow[] = [] - const writer = new JournalRowWriter({ - journalDir: root, - sessionId: SESSION_ID, - budget: new JournalAppendBudget(SESSION_ID, limits), - lifecycleAdmission, - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 }, - now: () => 1, - serialize: (run) => run(), - readOnly: () => readOnly, - setReadOnly: (value) => { - readOnly = value - }, - physicalBytes: () => physicalBytes, - highestFence: () => 0, - nextSequence: () => nextSequence, - tailRows: () => committedRows, - referencedBlobDigests: () => new Set(), - compact: async () => undefined, - commit: (row, nextPhysicalBytes) => { - overrides.commit?.(row, nextPhysicalBytes) - committedRows.push(row) - physicalBytes = nextPhysicalBytes - nextSequence = row.seq + 1 - }, - ...(overrides.appendRows ? { appendRows: overrides.appendRows } : {}) - }) - return { writer, lifecycleAdmission, committedRows } - } - - it('latches read-only when a post-append failure makes durability ambiguous', async () => { - let committed = false - const writer = new JournalRowWriter({ - journalDir: root, - sessionId: 'session-1', - budget: new JournalAppendBudget('session-1', DEFAULT_JOURNAL_PAYLOAD_LIMITS), - lifecycleAdmission: new JournalLifecycleAdmission( - 'session-1', - DEFAULT_JOURNAL_PAYLOAD_LIMITS.maxSessionBytes, - (itemId) => itemId - ), - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 }, - now: () => 1, - serialize: (run) => run(), - readOnly: () => readOnly, - setReadOnly: (value) => { - readOnly = value - }, - physicalBytes: () => 0, - highestFence: () => 0, - nextSequence: () => 1, - tailRows: () => [], - referencedBlobDigests: () => new Set(), - compact: async () => undefined, - commit: () => { - committed = true - }, - appendRows: async (journalDir, rows) => { - await appendJournalRows(journalDir, rows) - throw new Error('fsync failed after append') - } - }) - - await expect(writer.enqueue(row)).rejects.toThrow('fsync failed after append') - - expect(readOnly).toBe(true) - expect(committed).toBe(false) - await expect(writer.enqueue(row)).rejects.toMatchObject({ code: 'journal_read_only' }) - }) - - it('keeps blobs for a durable row when a post-append crash is reported', async () => { - const payload = 'durable blob payload'.repeat(2_000) - const bounded = boundPayload(payload, { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 32 - }) - const writer = new JournalRowWriter({ - journalDir: root, - sessionId: 'session-1', - budget: new JournalAppendBudget('session-1', DEFAULT_JOURNAL_PAYLOAD_LIMITS), - lifecycleAdmission: new JournalLifecycleAdmission( - 'session-1', - DEFAULT_JOURNAL_PAYLOAD_LIMITS.maxSessionBytes, - (itemId) => itemId - ), - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 }, - now: () => 1, - serialize: (run) => run(), - readOnly: () => readOnly, - setReadOnly: (value) => { - readOnly = value - }, - physicalBytes: () => 0, - highestFence: () => 0, - nextSequence: () => 1, - tailRows: () => [], - referencedBlobDigests: () => new Set(), - compact: async () => undefined, - commit: () => undefined, - appendRows: async (journalDir, rows) => { - await appendJournalRows(journalDir, rows) - throw new Error('crash after row append') - } - }) - - await expect( - writer.enqueue( - (seq, ts) => rowWithBlob(seq, ts, bounded), - [{ digest: bounded.digest, payload }] - ) - ).rejects.toThrow('crash after row append') - - expect(readOnly).toBe(true) - await expect(writer.enqueue(row)).rejects.toMatchObject({ code: 'journal_read_only' }) - expect(await readJournalBlob(root, bounded.digest)).toBe(payload) - const reopened = await loadJournal(root, 'session-1') - const item = reopened?.state.items.get('item-with-blob') - expect(item?.body).toMatchObject({ - kind: 'tool-call', - output: { digest: bounded.digest, truncated: true } - }) - }) - - it('does not leak a lifecycle reservation after budget refusal', async () => { - const probe = runningToolRow(1, 1) - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxSessionBytes: JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES + journalRowByteLength(probe) - 1 - } - const { writer, lifecycleAdmission, committedRows } = writerHarness({ limits }) - - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).rejects.toMatchObject({ - code: 'journal_bound_exceeded' - }) - - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - await expect(writer.enqueue(row)).resolves.toMatchObject({ kind: 'item', itemId: 'item-1' }) - expect( - committedRows.map((entry) => (entry.kind === 'item' ? entry.itemId : 'non-item')) - ).toEqual(['item-1']) - }) - - it('preflights existing durable-write temps before creating a blob or row', async () => { - const tempBytes = 512 - const tempPath = join(root, 'log.jsonl.existing-write.tmp') - await writeFile(tempPath, 't'.repeat(tempBytes), 'utf8') - const probe = row(1, 1) - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxSessionBytes: tempBytes + journalRowByteLength(probe) - 1 - } - const { writer, committedRows } = writerHarness({ limits }) - - await expect(writer.enqueue((seq, ts) => row(seq, ts))).rejects.toMatchObject({ - code: 'journal_bound_exceeded' - }) - expect(committedRows).toHaveLength(0) - expect(await readJournalBlob(root, 'a'.repeat(64))).toBeNull() - }) - - it('does not leak a lifecycle reservation after blob lookup failure', async () => { - const { writer, lifecycleAdmission } = writerHarness() - const digest = 'a'.repeat(64) - await writeFile(join(root, 'blobs'), 'not a directory', 'utf8') - - await expect( - writer.enqueue((seq, ts) => runningToolRow(seq, ts), [{ digest, payload: 'payload' }]) - ).rejects.toThrow() - - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - await rm(join(root, 'blobs'), { force: true }) - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).resolves.toMatchObject({ - kind: 'item', - itemId: 'running-tool' - }) - expect(lifecycleAdmission.state).toEqual({ - reservedBytes: JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES, - reservedAppendSlots: 1 - }) - }) - - it('rolls back ordinary append-rate reservation after blob preflight failure', async () => { - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxAppendsPerWindow: 1, - appendWindowMs: 100 - } - const { writer, committedRows } = writerHarness({ limits }) - const payload = 'retryable blob payload'.repeat(100) - const bounded = boundPayload(payload, limits) - await writeFile(join(root, 'blobs'), 'not a directory', 'utf8') - - await expect( - writer.enqueue( - (seq, ts) => rowWithBlob(seq, ts, bounded), - [{ digest: bounded.digest, payload }] - ) - ).rejects.toThrow() - - await rm(join(root, 'blobs'), { force: true }) - await expect( - writer.enqueue( - (seq, ts) => rowWithBlob(seq, ts, bounded), - [{ digest: bounded.digest, payload }] - ) - ).resolves.toMatchObject({ kind: 'item', itemId: 'item-with-blob' }) - expect(committedRows).toHaveLength(1) - }) - - it('does not leak a lifecycle reservation after durable append failure', async () => { - const { writer, lifecycleAdmission } = writerHarness({ - appendRows: async () => { - throw new Error('append failed before a durable row existed') - } - }) - - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).rejects.toThrow( - 'append failed before a durable row existed' - ) - - expect(readOnly).toBe(true) - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('does not leak a lifecycle reservation after reducer commit failure', async () => { - const { writer, lifecycleAdmission } = writerHarness({ - commit: () => { - throw new Error('commit failed after durable append') - } - }) - - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).rejects.toThrow( - 'commit failed after durable append' - ) - - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-row-writer.test.ts b/src/main/native-chat/agent-session-journal/journal-row-writer.test.ts new file mode 100644 index 00000000000..dc08739a8bc --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-row-writer.test.ts @@ -0,0 +1,94 @@ +// The write path's transaction. +// +// A transaction either commits or does not, so the old "a post-append failure +// makes durability ambiguous" latch has nothing left to latch on: the case that +// used to assert the latch asserts the rollback instead. + +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' +import { openJournalDatabase, type OpenJournalDatabase } from './journal-database' +import { journalDatabaseFile } from './journal-paths' +import { + insertJournalRow, + readJournalEpochRows, + upsertJournalSessionRow +} from './journal-row-table' +import type { JournalRow } from './journal-row-schema' +import { JournalRowWriter } from './journal-row-writer' + +const SESSION_ID = 'session-1' +const EPOCH = 'epoch-1' + +function row(seq: number, ts: number): JournalRow { + return { + v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + epoch: EPOCH, + seq, + fence: 0, + ts, + kind: 'item', + itemId: 'item-1', + revision: 1, + body: { kind: 'status', text: 'plain append' } + } +} + +describe('journal row writer', () => { + let root: string + let database: OpenJournalDatabase + let readOnly = false + + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-row-writer-')) + database = openJournalDatabase(journalDatabaseFile(root)) + upsertJournalSessionRow(database.db, SESSION_ID, EPOCH, 1) + readOnly = false + }) + + afterEach(async () => { + try { + database.db.close() + } catch { + // Already closed by the case. + } + await rm(root, { recursive: true, force: true }) + }) + + function writerHarness() { + const committedRows: JournalRow[] = [] + let sequence = 1 + const writer = new JournalRowWriter({ + sessionId: SESSION_ID, + now: () => 1, + serialize: (run) => run(), + database: () => database, + readOnly: () => readOnly, + highestFence: () => 0, + nextSequence: () => sequence, + commit: (committed) => { + committedRows.push(committed) + sequence = committed.seq + 1 + } + }) + return { writer, committedRows } + } + + it('rolls the transaction back and sets no latch when the insert fails', async () => { + const { writer, committedRows } = writerHarness() + // A row already occupies sequence 1, so the insert violates the primary key. + insertJournalRow(database.db, SESSION_ID, row(1, 1)) + + await expect(writer.enqueue(row)).rejects.toThrow() + + expect(readOnly).toBe(false) + expect(committedRows).toHaveLength(0) + expect(readJournalEpochRows(database.db, SESSION_ID, EPOCH)).toHaveLength(1) + // Still writable: there is no ambiguity for a latch to protect against. + await expect(writer.enqueue((seq, ts) => row(seq + 1, ts))).resolves.toMatchObject({ + kind: 'item' + }) + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-row-writer.ts b/src/main/native-chat/agent-session-journal/journal-row-writer.ts index 980275d4c72..85ff7da7a3f 100644 --- a/src/main/native-chat/agent-session-journal/journal-row-writer.ts +++ b/src/main/native-chat/agent-session-journal/journal-row-writer.ts @@ -1,188 +1,42 @@ -import { - budgetPressurePolicy, - journalTailCanShedRows, - journalTailIsReadyToCompact, - type JournalCompactionPolicy -} from './journal-compaction' -import { journalBlobFileSize, putJournalBlob, removeJournalBlob } from './journal-blob-store' -import { appendJournalRows } from './journal-log-file' -import { blobDigestsInBody } from './journal-reducer' -import { journalDirectoryBytes } from './journal-physical-quota' -import type { JournalLifecycleAdmission } from './journal-lifecycle-admission' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' -import { - AgentSessionJournalError, - assertJournalFence, - assertJournalWritable, - type JournalAppendBudget -} from './journal-write-guards' - -type JournalBlob = { digest: string; payload: string } +import type Database from '../../sqlite/sync-database' +import { insertJournalRow, upsertJournalSessionRow } from './journal-row-table' +import type { JournalRow } from './journal-row-schema' +import { assertJournalFence, assertJournalWritable } from './journal-write-guards' export type JournalRowWriterDeps = { - journalDir: string sessionId: string - budget: JournalAppendBudget - lifecycleAdmission: JournalLifecycleAdmission - autoCompact: boolean - compaction: JournalCompactionPolicy now: () => number serialize: (run: () => Promise) => Promise + database: () => { db: Database.Database } readOnly: () => boolean - setReadOnly: (readOnly: boolean) => void - physicalBytes: () => number highestFence: () => number nextSequence: () => number - tailRows: () => readonly JournalRow[] - referencedBlobDigests: () => ReadonlySet - compact: (now: number, policy: JournalCompactionPolicy) => Promise - commit: (row: JournalRow, physicalBytes: number) => void - appendRows?: (journalDir: string, rows: readonly JournalRow[]) => Promise + commit: (row: JournalRow) => void } export class JournalRowWriter { constructor(private readonly deps: JournalRowWriterDeps) {} - enqueue( - build: (seq: number, ts: number) => JournalRow, - blobs: readonly JournalBlob[] = [] - ): Promise { + enqueue(build: (seq: number, ts: number) => JournalRow): Promise { return this.deps.serialize(async () => { assertJournalWritable(this.deps.readOnly(), this.deps.sessionId) - const ts = this.deps.now() - const row = build(this.deps.nextSequence(), ts) + const row = build(this.deps.nextSequence(), this.deps.now()) assertJournalFence(row.fence, this.deps.highestFence()) - // The in-memory counter is an optimization, not the quota source of - // truth: a prior crash may have left a durable-write temp beside the - // finals, and a concurrent/retried opener may have materialized files - // after the last commit callback. Recount before any speculative write - // so the peak check includes those bytes. - let physicalBytes = Math.max( - this.deps.physicalBytes(), - await journalDirectoryBytes(this.deps.journalDir) - ) - const admission = this.deps.lifecycleAdmission.prepare(row, physicalBytes) - const newBlobs = await uniqueNewBlobs(this.deps.journalDir, blobs) - const blobBytes = newBlobs.reduce( - (total, blob) => total + Buffer.byteLength(blob.payload, 'utf8'), - 0 - ) - const budgetCompaction = budgetPressurePolicy(this.deps.compaction) - let effectiveSize = physicalBytes + blobBytes + admission.protectedBytes - if ( - this.deps.autoCompact && - this.deps.budget.wouldExceedSize(row, effectiveSize) && - journalTailCanShedRows(this.deps.tailRows(), budgetCompaction, ts) - ) { - await this.deps.compact(ts, budgetCompaction) - physicalBytes = this.deps.physicalBytes() - effectiveSize = physicalBytes + blobBytes + admission.protectedBytes - } - const lifecycleRateCheckpoint = admission.lifecycleCovered - ? this.deps.budget.checkpoint() - : null - const appendRateCheckpoint = this.deps.budget.checkpoint() - let committed = false - let appendMayHaveLanded = false + const { db } = this.deps.database() + db.exec('BEGIN IMMEDIATE') try { - if (admission.lifecycleCovered) { - this.deps.budget.assertReservedLifecycle(row, effectiveSize) - } else { - this.deps.budget.assert(row, ts, effectiveSize) - } - const appendedBytes = blobBytes + journalRowByteLength(row) - if ( - physicalBytes + appendedBytes > - this.deps.budget.maxSessionBytes - admission.protectedBytes - ) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.deps.sessionId} reached its ${this.deps.budget.maxSessionBytes}-byte physical bound` - ) - } - await this.commitFiles(row, newBlobs, () => { - appendMayHaveLanded = true - }) - physicalBytes += appendedBytes - this.deps.commit(row, physicalBytes) - this.deps.lifecycleAdmission.commit(admission) - committed = true + insertJournalRow(db, this.deps.sessionId, row) + upsertJournalSessionRow(db, this.deps.sessionId, row.epoch, row.ts) + db.exec('COMMIT') } catch (error) { - if (!committed && lifecycleRateCheckpoint) { - this.deps.budget.restore(lifecycleRateCheckpoint) - } - if (!committed && !appendMayHaveLanded) { - this.deps.budget.restore(appendRateCheckpoint) - } + db.exec('ROLLBACK') throw error } - if ( - this.deps.autoCompact && - journalTailIsReadyToCompact(this.deps.tailRows(), this.deps.compaction, ts) - ) { - await this.deps.compact(ts, this.deps.compaction) - } + // COMMIT landed, so the row is durable: adopt it before anything that can + // fail. Rejecting here instead would leave the next append reusing a + // sequence the table already holds. + this.deps.commit(row) return row }) } - - private async commitFiles( - row: JournalRow, - blobs: readonly JournalBlob[], - markAppendLanded: () => void - ): Promise { - const persisted: string[] = [] - let appendMayHaveLanded = false - try { - for (const blob of blobs) { - await putJournalBlob(this.deps.journalDir, blob.digest, blob.payload) - persisted.push(blob.digest) - } - appendMayHaveLanded = true - markAppendLanded() - await (this.deps.appendRows ?? appendJournalRows)(this.deps.journalDir, [row]) - } catch (error) { - if (appendMayHaveLanded) { - this.deps.setReadOnly(true) - throw error - } - const retained = this.referencedBlobDigestsIncludingTail() - for (const digest of persisted) { - if (!retained.has(digest)) { - await removeJournalBlob(this.deps.journalDir, digest) - } - } - throw error - } - } - - private referencedBlobDigestsIncludingTail(): Set { - const retained = new Set(this.deps.referencedBlobDigests()) - for (const row of this.deps.tailRows()) { - if (row.kind === 'item') { - blobDigestsInBody(row.body, retained) - } else if (row.kind === 'lifecycle-batch') { - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - blobDigestsInBody(mutation.body, retained) - } - } - } - } - return retained - } -} - -async function uniqueNewBlobs( - journalDir: string, - blobs: readonly JournalBlob[] -): Promise { - const unique = new Map(blobs.map((blob) => [blob.digest, blob])) - const result: JournalBlob[] = [] - for (const blob of unique.values()) { - if ((await journalBlobFileSize(journalDir, blob.digest)) === null) { - result.push(blob) - } - } - return result } diff --git a/src/main/native-chat/agent-session-journal/journal-store-close.test.ts b/src/main/native-chat/agent-session-journal/journal-store-close.test.ts new file mode 100644 index 00000000000..960c65a638a --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-close.test.ts @@ -0,0 +1,259 @@ +// `close()`: enqueue-time admission, and the fulfilled/rejected split. +// +// The two failures this file exists to prevent: an append enqueued in the same +// turn as `close()` being rejected AFTER the queue accepted it, and a rejected +// close leaving a connection live but permanently unreachable through the API. + +import { access, mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import { journalDatabaseFile } from './journal-paths' +import type { AgentSessionJournal } from './journal-store' +import { createTrackedJournalOpener } from './journal-store-test-open' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} + +let root: string +const journals = createTrackedJournalOpener() + +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } +} + +function body(value: string): AgentJournalItemBody { + return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } +} + +function openJournal(): Promise { + return journals.open({ identity: IDENTITY, journalDir: root }) +} + +/** Replaces the store's own release step, which is the only step that can + * reject the attempt in production. */ +function injectReleaseFailure(journal: AgentSessionJournal): { + calls: () => number + stopFailing: () => void +} { + const internals = journal as unknown as { database: { db: { close: () => void } } } + const release = internals.database.db.close.bind(internals.database.db) + let calls = 0 + let failing = true + internals.database.db.close = () => { + calls += 1 + if (failing) { + throw new Error('injected release failure') + } + release() + } + return { calls: () => calls, stopFailing: () => (failing = false) } +} + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-close-')) +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('closed-state admission happens at enqueue', () => { + it('completes a write enqueued in the same turn as the close', async () => { + const journal = await openJournal() + const append = journal.appendItem(item(1), body('before'), { fence: 1 }) + const closed = journal.close() + + await expect(append).resolves.toBeDefined() + await expect(closed).resolves.toBeUndefined() + const reopened = await openJournal() + expect(reopened.snapshot().items).toHaveLength(1) + }) + + it('refuses a write offered while the close is still in flight, without queueing it', async () => { + const journal = await openJournal() + const closing = journal.close() + const refused = journal.appendItem(item(1), body('during'), { fence: 1 }) + + // The rejection is available before the close step has run: it never joined + // the queue, so nothing is ever chained behind a close. + await expect(refused).rejects.toMatchObject({ code: 'journal_closed' }) + await expect(closing).resolves.toBeUndefined() + }) + + it('refuses every write entry point after the close has settled', async () => { + const journal = await openJournal() + await journal.close() + const settle = (attempt: Promise): Promise => + attempt.then( + () => new Error('resolved instead of refusing'), + (error: unknown) => error + ) + const refusals = [ + settle(journal.appendItem(item(1), body('after'), { fence: 1 })), + settle(journal.appendTombstone(item(3), { fence: 1 })), + settle( + journal.appendSubmission({ + clientMessageId: 'cm_1', + payloadFingerprint: 'f', + body: { kind: 'message', role: 'user', blocks: [] }, + fence: 1 + }) + ), + settle(journal.resolveDispatch({ clientMessageId: 'cm_1', state: 'rejected', fence: 1 })), + settle( + journal.appendLifecycleBatch({ + settlementId: 'settle', + fence: 1, + mutations: [{ kind: 'item', identity: item(4), body: body('x') }] + }) + ), + settle(journal.rollEpoch('handle_forked', 1)), + settle(journal.replaceEpochItems('handle_forked', 1, [])) + ] + for (const refusal of refusals) { + expect(await refusal).toMatchObject({ code: 'journal_closed' }) + } + }) + + // The property part 1 rests on: every entry point reaches the gate in the + // caller's own turn, so a refusal never advances the queue. + it('rejects without waiting for the queue to advance', async () => { + const journal = await openJournal() + const inFlight = journal.appendItem(item(1), body('admitted'), { fence: 1 }) + const closing = journal.close() + + // Settles while the admitted append is still running: it reached the gate in + // the caller's own turn and never joined the queue behind it. + await expect(journal.appendItem(item(2), body('later'), { fence: 1 })).rejects.toMatchObject({ + code: 'journal_closed' + }) + await expect(inFlight).resolves.toBeDefined() + await expect(closing).resolves.toBeUndefined() + }) + + it('resolves a second close as a no-op without running the routine again', async () => { + const journal = await openJournal() + const injected = injectReleaseFailure(journal) + injected.stopFailing() + await journal.close() + expect(injected.calls()).toBe(1) + + await expect(journal.close()).resolves.toBeUndefined() + expect(injected.calls()).toBe(1) + }) +}) + +describe('a rejected close is a real retry', () => { + it('retries the release, releases the handle, and then goes terminal', async () => { + const journal = await openJournal() + await journal.appendItem(item(1), body('durable'), { fence: 1 }) + const injected = injectReleaseFailure(journal) + + await expect(journal.close()).rejects.toThrow('injected release failure') + expect(injected.calls()).toBe(1) + + // Write-closed anyway: retry exists to release the OS handle, never to + // resurrect the store. + await expect(journal.appendItem(item(2), body('after'), { fence: 1 })).rejects.toMatchObject({ + code: 'journal_closed' + }) + + injected.stopFailing() + await expect(journal.close()).resolves.toBeUndefined() + // The retry RE-ENTERED the release. A completion flag on it would skip the + // one step that had not succeeded, and the handle would never be released. + expect(injected.calls()).toBe(2) + const dbPath = journalDatabaseFile(root) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + + // Fulfilment is terminal: a third call must not issue a second db.close(), + // which `node:sqlite` answers with ERR_INVALID_STATE. + await expect(journal.close()).resolves.toBeUndefined() + expect(injected.calls()).toBe(2) + }) + + it('hands concurrent callers the same outcome', async () => { + const journal = await openJournal() + const injected = injectReleaseFailure(journal) + const first = journal.close() + const second = journal.close() + await expect(first).rejects.toThrow('injected release failure') + await expect(second).rejects.toThrow('injected release failure') + expect(injected.calls()).toBe(1) + }) +}) + +// The two rejected readings of the contract, as models, because a green run on +// the real store proves nothing about what the alternatives would have done. +describe('negative controls', () => { + class UnconditionalNoOpClose { + calls = 0 + private called = false + constructor(private readonly release: () => void) {} + async close(): Promise { + if (this.called) { + return + } + this.called = true + this.calls += 1 + this.release() + } + } + + class AlwaysReentrantClose { + calls = 0 + async close(release: () => void): Promise { + this.calls += 1 + release() + } + } + + it('an unconditional second-call no-op leaves a failed close unreleasable', async () => { + let failing = true + let released = false + const model = new UnconditionalNoOpClose(() => { + if (failing) { + throw new Error('injected release failure') + } + released = true + }) + await expect(model.close()).rejects.toThrow('injected release failure') + failing = false + await expect(model.close()).resolves.toBeUndefined() + // Fulfilled without the routine running: the handle is still held. + expect(model.calls).toBe(1) + expect(released).toBe(false) + }) + + it('an always-reentrant close issues the second db.close() that throws', async () => { + let open = true + const release = (): void => { + if (!open) { + throw new Error('ERR_INVALID_STATE: database is not open') + } + open = false + } + const model = new AlwaysReentrantClose() + await model.close(release) + await expect(model.close(release)).rejects.toThrow('database is not open') + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-store-close.ts b/src/main/native-chat/agent-session-journal/journal-store-close.ts new file mode 100644 index 00000000000..f1b56d6a0db --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-close.ts @@ -0,0 +1,106 @@ +// Releasing the one SQLite connection a store holds for its lifetime. +// +// The contract, in five parts: +// +// 1. Closed-state admission happens at ENQUEUE, once, and is permanent. The +// flag lives on the store's write gate; it is never cleared, not by a +// rejection and not by a retry. A store that failed to close is still a +// store nobody may write to. +// 2. The close step shares the write queue and BYPASSES that gate — a close +// that consulted the flag it had just set would refuse itself. Same queue +// orders the release behind admitted work; separate gate lets it run. +// 3. One in-flight attempt, shared: concurrent callers get the same outcome. +// 4. Fulfilled is TERMINAL; rejected is not. A later call after fulfilment is a +// genuine no-op — `DatabaseSync.close()` throws `ERR_INVALID_STATE` on a +// second call, so re-entering after success is the bug, not the fix. +// 5. The release is deliberately unguarded. There is no way to ask whether a +// `db.close()` that threw released the handle first, and guarding the step +// would skip it on retry — guaranteeing a permanent leak in exactly the case +// where it did not release. + +import { AgentSessionJournalError } from './journal-write-guards' +import type Database from '../../sqlite/sync-database' + +/** + * The write queue and its closed gate. Admission is checked at ENQUEUE and is + * permanent; the close step reaches the same queue through `serializePastGate`, + * because a close that consulted the flag it had just set would refuse itself. + */ +export class JournalWriteQueue { + private writes: Promise = Promise.resolve() + private closed = false + + constructor(private readonly sessionId: string) {} + + markClosed(): void { + this.closed = true + } + + serialize(run: () => Promise): Promise { + if (this.closed) { + return Promise.reject( + new AgentSessionJournalError( + 'journal_closed', + `agent-session journal for ${this.sessionId} is closed` + ) + ) + } + return this.serializePastGate(run) + } + + serializePastGate(run: () => Promise): Promise { + const started = this.writes.then(run) + this.writes = started.catch(() => undefined) + return started + } +} + +export class JournalConnectionCloser { + private released = false + private inFlight: Promise | null = null + + constructor( + private readonly deps: { + connection: () => Database.Database | null + /** Chains onto the store's write queue past the closed gate. */ + enqueue: (run: () => Promise) => Promise + } + ) {} + + get isReleased(): boolean { + return this.released + } + + close(): Promise { + if (this.released) { + return Promise.resolve() + } + if (this.inFlight) { + return this.inFlight + } + const attempt = this.deps + .enqueue(() => this.release()) + .then( + () => { + this.released = true + this.inFlight = null + }, + (error: unknown) => { + this.inFlight = null + throw error + } + ) + this.inFlight = attempt + return attempt + } + + private async release(): Promise { + const db = this.deps.connection() + if (!db) { + return + } + // SQLite checkpoints and removes the WAL itself when the last connection to + // the database closes. + db.close() + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store-collaborators.ts b/src/main/native-chat/agent-session-journal/journal-store-collaborators.ts new file mode 100644 index 00000000000..a4ff455b31f --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-collaborators.ts @@ -0,0 +1,88 @@ +// Wiring for the store's collaborators. +// +// Split out of the store itself so the class stays a description of the public +// surface rather than sixty lines of constructor plumbing. + +import type { + AgentJournalCursor, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import type { OpenJournalDatabase } from './journal-database' +import { JournalEpochController } from './journal-epoch-controller' +import { JournalItemAppender } from './journal-item-appender' +import { JournalLifecycleBatchAppender } from './journal-lifecycle-batch-appender' +import type { JournalLoad } from './journal-open' +import type { JournalReducerState } from './journal-reducer' +import { JournalRowWriter } from './journal-row-writer' +import { restoreJournalStore } from './journal-store-restore' +import type { JournalRow } from './journal-row-schema' +import type { AgentSessionJournal } from './journal-store' + +export type JournalStoreHost = { + identity: AgentSessionJournalIdentity + journalDir: string + now: () => number + mintEpoch: () => string + serialize: (run: () => Promise) => Promise + database: () => OpenJournalDatabase + state: () => JournalReducerState + readOnly: () => boolean + setReadOnly: (readOnly: boolean) => void + cursor: () => AgentJournalCursor + adopt: (loaded: JournalLoad) => void + commit: (row: JournalRow) => void + /** A caller-supplied load, which suppresses replay entirely when present. */ + loaded: () => JournalLoad | null | undefined + malformedRows: () => number + setMalformedRows: (count: number) => void + journal: () => AgentSessionJournal + enqueue: (build: (seq: number, ts: number) => JournalRow) => Promise +} + +export type JournalStoreCollaborators = { + rowWriter: JournalRowWriter + epochController: JournalEpochController + itemAppender: JournalItemAppender + lifecycleBatchAppender: JournalLifecycleBatchAppender + /** Restores the store's state from disk. Owned here because it needs the same + * collaborators the constructor just built. */ + restore: () => Promise +} + +export function createJournalStoreCollaborators(host: JournalStoreHost): JournalStoreCollaborators { + const epochController = new JournalEpochController({ + identity: host.identity, + now: host.now, + mintEpoch: host.mintEpoch, + serialize: host.serialize, + database: host.database, + readOnly: host.readOnly, + setReadOnly: host.setReadOnly, + highestFence: () => host.state().highestFence, + cursor: host.cursor, + adopt: host.adopt + }) + return { + epochController, + restore: () => restoreJournalStore(host, { epochController }), + rowWriter: new JournalRowWriter({ + sessionId: host.identity.sessionId, + now: host.now, + serialize: host.serialize, + database: host.database, + readOnly: host.readOnly, + highestFence: () => host.state().highestFence, + nextSequence: () => host.state().lastSequence + 1, + commit: host.commit + }), + itemAppender: new JournalItemAppender({ + state: host.state, + enqueue: host.enqueue + }), + lifecycleBatchAppender: new JournalLifecycleBatchAppender({ + state: host.state, + cursor: host.cursor, + enqueue: host.enqueue + }) + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store-contracts.ts b/src/main/native-chat/agent-session-journal/journal-store-contracts.ts index 45a723a0f23..22e3a4c7cca 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-contracts.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-contracts.ts @@ -6,19 +6,13 @@ import type { AgentJournalResetReason, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import type { JournalCompactionPolicy } from './journal-compaction' import type { JournalLoad } from './journal-open' -import type { JournalPayloadLimits } from './journal-payload-bounds' import type { JournalLifecycleMutationInput } from './journal-row-builders' import type { JournalRow } from './journal-row-schema' export type AgentSessionJournalOptions = { identity: AgentSessionJournalIdentity journalDir: string - limits?: JournalPayloadLimits - compaction?: JournalCompactionPolicy - /** Compact as the tail grows. Defaults on: without it the log never sheds. */ - autoCompact?: boolean now?: () => number mintEpoch?: () => string /** A caller that already loaded the journal can avoid reading the same files again. */ @@ -45,7 +39,6 @@ export type JournalAppendResult = { } export type JournalItemAppendOptions = { fence: number; observedAt?: number; recovered?: true } -export type JournalBlobInput = { digest: string; payload: string } export type JournalTombstoneInput = { fence: number } export type JournalLifecycleBatchInput = { diff --git a/src/main/native-chat/agent-session-journal/journal-store-open.ts b/src/main/native-chat/agent-session-journal/journal-store-open.ts index 67d05c0b0c3..e002048a094 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-open.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-open.ts @@ -1,16 +1,14 @@ -import type { AgentJournalSnapshot } from '../../../shared/agent-session-journal-types' -import { malformedRowsDisclosure, quarantineCorruptSuffix } from './journal-corruption-quarantine' -import { ensureJournalDir } from './journal-log-file' -import { loadJournal, type JournalLoad } from './journal-open' -import { assertJournalPhysicalCapacity, journalDirectoryBytes } from './journal-physical-quota' -import type { JournalRow } from './journal-row-schema' +import { mkdir } from 'node:fs/promises' +import type { JournalLoad } from './journal-open' +import { journalRepairDisclosure, type JournalRepairDisclosure } from './journal-repair-disclosure' + +export async function ensureJournalDir(journalDir: string): Promise { + await mkdir(journalDir, { recursive: true }) +} export function journalStoreLoadedFields(loaded: JournalLoad) { return { state: loaded.state, - tailRows: loaded.tailRows, - compactedThrough: loaded.compactedThrough, - sizeBytes: loaded.sizeBytes, readOnly: loaded.readOnly, malformedRows: loaded.malformedRows } @@ -18,60 +16,47 @@ export function journalStoreLoadedFields(loaded: JournalLoad) { export async function openJournalStoreState(input: { journalDir: string - sessionId: string - maxBytes: number loaded: JournalLoad | null | undefined - start: () => Promise + replay: () => JournalLoad | null + /** Drops the rejected suffix and records the rebuild it owes, in ONE + * transaction. Corruption is not preserved; replay keeps reporting `corrupt` + * until provider history republishes the epoch or the session writes past + * `contentFrom`, the first sequence the repair left free. */ + deleteSuffix: (fromSeq: number, contentFrom: number) => number + start: () => void adopt: (loaded: JournalLoad) => void - tailRows: () => readonly JournalRow[] - snapshot: () => AgentJournalSnapshot - rebuildLifecycle: (snapshot: AgentJournalSnapshot, physicalBytes: number) => void + /** Republishes an anchor row for an epoch a repair emptied. */ + publishRepairEpoch: () => void appendDisclosure: ( - identity: ReturnType['identity'], - body: ReturnType['body'], + identity: JournalRepairDisclosure['identity'], + body: JournalRepairDisclosure['body'], fence: number ) => Promise highestFence: () => number malformedRows: () => number + setMalformedRows: (count: number) => void readOnly: () => boolean - setPhysicalBytes: (bytes: number) => void }): Promise { - await ensureJournalDir(input.journalDir) - input.setPhysicalBytes( - await assertJournalPhysicalCapacity({ - journalDir: input.journalDir, - sessionId: input.sessionId, - maxBytes: input.maxBytes - }) - ) - const loaded = - input.loaded !== undefined - ? input.loaded - : await loadJournal(input.journalDir, input.sessionId, { maxBytes: input.maxBytes }) + const loaded = input.loaded !== undefined ? input.loaded : input.replay() if (!loaded) { - await input.start() - input.setPhysicalBytes(await journalDirectoryBytes(input.journalDir)) + input.start() return } input.adopt(loaded) - if (loaded.corrupt && !loaded.readOnly) { - await quarantineCorruptSuffix(input.journalDir, input.tailRows(), loaded.quarantineRemainder, { - sessionId: input.sessionId, - maxBytes: input.maxBytes - }) + if (loaded.truncateFrom !== undefined && !loaded.readOnly) { + input.deleteSuffix(loaded.truncateFrom, loaded.state.lastSequence + 1) } - let physicalBytes = await journalDirectoryBytes(input.journalDir) - input.setPhysicalBytes(physicalBytes) - // A future-schema/read-only journal is inspection-only. Its reduced state is - // intentionally empty, and rebuilding reservations from it would mutate the - // in-memory quota model (and could influence later admission decisions). - if (!loaded.readOnly) { - input.rebuildLifecycle(input.snapshot(), physicalBytes) + // A repair that took every live row leaves the epoch with no anchor. Publish + // one before anything can append into it: an ordinary row at sequence 1 would + // replay as a clean timeline and hide that the history was never rebuilt. + if (!loaded.readOnly && loaded.state.lastSequence === 0) { + input.publishRepairEpoch() + // The replacement epoch adopts a clean load; what this open's repair did is + // still the answer `repair` and the disclosure below owe the caller. + input.setMalformedRows(loaded.malformedRows) } if (input.malformedRows() > 0 && !input.readOnly()) { - const disclosure = malformedRowsDisclosure(input.malformedRows()) + const disclosure = journalRepairDisclosure({ malformedRows: input.malformedRows() }) await input.appendDisclosure(disclosure.identity, disclosure.body, input.highestFence()) } - physicalBytes = await journalDirectoryBytes(input.journalDir) - input.setPhysicalBytes(physicalBytes) } diff --git a/src/main/native-chat/agent-session-journal/journal-store-restore.ts b/src/main/native-chat/agent-session-journal/journal-store-restore.ts new file mode 100644 index 00000000000..53683a797d1 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-restore.ts @@ -0,0 +1,49 @@ +// Bringing a store's in-memory state up from disk. +// +// Split out of the store for the same reason its collaborators were: this is the +// ORDERING between replay, suffix repair and disclosure, and none of it belongs +// to the store's public surface. Every step here reads or writes through the +// same host the collaborators use, so the store keeps the state and this owns +// the sequence. + +import type { JournalEpochController } from './journal-epoch-controller' +import { replayJournal } from './journal-open' +import type { JournalStoreHost } from './journal-store-collaborators' +import { openJournalStoreState } from './journal-store-open' +import { deleteJournalRepairedSuffix } from './journal-repair-marker' + +export function restoreJournalStore( + host: JournalStoreHost, + collaborators: { epochController: JournalEpochController } +): Promise { + return openJournalStoreState({ + journalDir: host.journalDir, + loaded: host.loaded(), + replay: () => { + const opened = host.database() + return replayJournal(opened.db, opened.readOnly, host.identity.sessionId) + }, + deleteSuffix: (fromSeq, contentFrom) => + deleteJournalRepairedSuffix({ + db: host.database().db, + sessionId: host.identity.sessionId, + epoch: host.state().epoch, + fromSeq, + contentFrom, + now: host.now() + }), + start: () => collaborators.epochController.start('session_created', 0), + // `unreconcilable_prefix` is the durable statement that this epoch exists + // because a repair emptied one: replay reads it back and keeps asking for + // provider history until the timeline is rebuilt or the session writes. + publishRepairEpoch: () => + collaborators.epochController.start('unreconcilable_prefix', host.state().highestFence), + adopt: host.adopt, + appendDisclosure: (identity, body, fence) => + host.journal().appendItem(identity, body, { fence }), + highestFence: () => host.state().highestFence, + malformedRows: host.malformedRows, + setMalformedRows: host.setMalformedRows, + readOnly: host.readOnly + }) +} diff --git a/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts b/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts index d78adeb30f5..df34ef73e73 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts @@ -1,4 +1,11 @@ -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' +// Two independent version axes, both fail closed. +// +// `PRAGMA user_version` the DB SHAPE, known before the first read +// the row's `v` field the row BODY shape, met during replay +// +// A newer build can change either alone, so both are needed. + +import { mkdtemp, rm, stat } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -7,8 +14,12 @@ import type { AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { JOURNAL_LOG_FILE, JOURNAL_SNAPSHOT_FILE } from './journal-log-file' -import { openAgentSessionJournal } from './journal-store-factory' +import type Database from '../../sqlite/sync-database' +import { openJournalDatabase } from './journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' +import { journalDatabaseFile } from './journal-paths' +import type { AgentSessionJournal } from './journal-store' +import { createTrackedJournalOpener } from './journal-store-test-open' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -20,6 +31,7 @@ const IDENTITY: AgentSessionJournalIdentity = { let root: string let clock = 1_000 +const journals = createTrackedJournalOpener() function tick(): number { clock += 1 @@ -34,87 +46,139 @@ function body(value: string): AgentJournalItemBody { return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } } -async function open(overrides: Partial[0]> = {}) { - return openAgentSessionJournal({ +function open(): Promise { + return journals.open({ identity: IDENTITY, journalDir: root, now: tick, - mintEpoch: () => `epoch-${clock}`, - ...overrides + mintEpoch: () => `epoch-${clock}` + }) +} + +async function withDatabase(run: (db: Database.Database) => void): Promise { + const opened = openJournalDatabase(journalDatabaseFile(root)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} + +/** Appends a raw `row_json` the way a newer build or a bad write would leave it. */ +async function appendRawRow(epoch: string, seq: number, rowJson: string): Promise { + await withDatabase((db) => { + db.prepare( + 'INSERT INTO journal_rows (session_id, epoch, seq, ts, row_json) VALUES (?, ?, ?, ?, ?)' + ).run(IDENTITY.sessionId, epoch, seq, 1, rowJson) }) } beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-')) + root = await mkdtemp(join(tmpdir(), 'orca-journal-schema-')) clock = 1_000 }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) -describe('schema', () => { - it('quarantines an invalid compacted snapshot without replacing its tail', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - await journal.compact() - const epoch = journal.epoch - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const logPath = join(root, JOURNAL_LOG_FILE) - const invalidSnapshot = '{"folded history":' - await writeFile(snapshotPath, invalidSnapshot, 'utf-8') - const retainedTail = await readFile(logPath, 'utf-8') - expect(retainedTail).not.toContain('"kind":"epoch"') - - const reopened = await open() - expect(reopened.epoch).toBe(epoch) - expect(await readFile(logPath, 'utf-8')).toBe(retainedTail) - const quarantined = (await readdir(root)).find((name) => - name.startsWith('quarantine-snapshot-') - ) - expect(quarantined).toBeDefined() - expect(await readFile(join(root, quarantined!), 'utf-8')).toBe(invalidSnapshot) - }) - - it('degrades to read-only on a row from a newer build, without skipping or deleting it', async () => { +describe('axis 1: the database shape', () => { + it('latches read-only on a newer user_version and writes nothing', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - const future = JSON.stringify({ - v: 99, - kind: 'item', - epoch: journal.epoch, - seq: 99, - fence: 1, - ts: 1, - itemId: 'future', - revision: 1, - body: { kind: 'status', text: 'from a newer host' } - }) - const before = await readFile(logPath, 'utf-8') - await writeFile(logPath, `${before}${future}\n`, 'utf-8') + await journal.close() + await withDatabase((db) => db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`)) + const before = await stat(journalDatabaseFile(root)) const reopened = await open() expect(reopened.isReadOnly).toBe(true) await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ code: 'journal_read_only' }) - await expect(reopened.compact()).rejects.toMatchObject({ code: 'journal_read_only' }) - expect(reopened.readSince({ epoch: reopened.epoch, sequence: 0 })).toEqual({ - ok: false, - reset: 'schema_unreadable' + // The file this build must not touch is byte-identical afterwards. + await reopened.close() + expect((await stat(journalDatabaseFile(root))).size).toBe(before.size) + await withDatabase((db) => { + expect(db.pragma('user_version', { simple: true })).toBe(JOURNAL_DB_SCHEMA_VERSION + 1) }) - // The unreadable row is still on disk, and nothing was compacted past it. - expect(await readFile(logPath, 'utf-8')).toContain('"v":99') }) - it('skips a malformed line without giving up the journal, and discloses the skip', async () => { + it('refuses the schema escape hatch on a latched store', async () => { + const journal = await open() + await journal.close() + await withDatabase((db) => db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`)) + + const reopened = await open() + // With byte-copy quarantine gone there is nothing for `schema_unreadable` to + // do differently, so it takes the same writable guard as every other reason. + await expect(reopened.rollEpoch('schema_unreadable', 2)).rejects.toMatchObject({ + code: 'journal_read_only' + }) + expect(reopened.isReadOnly).toBe(true) + }) + + it('migrates an older user_version forward on reopen', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}{not json\n`, 'utf-8') + await journal.close() + await withDatabase((db) => db.pragma('user_version = 0')) + + const reopened = await open() + expect(reopened.isReadOnly).toBe(false) + expect(reopened.snapshot().items).toHaveLength(1) + await reopened.close() + await withDatabase((db) => { + expect(db.pragma('user_version', { simple: true })).toBe(JOURNAL_DB_SCHEMA_VERSION) + }) + }) +}) + +describe('axis 2: the row body shape', () => { + it('degrades to read-only on a row from a newer build, without skipping it', async () => { + const journal = await open() + await journal.appendItem(item(0), body('a'), { fence: 1 }) + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() + await appendRawRow( + epoch, + nextSeq, + JSON.stringify({ + v: 99, + kind: 'item', + epoch, + seq: nextSeq, + fence: 1, + ts: 1, + itemId: 'future', + revision: 1, + body: { kind: 'status', text: 'from a newer build' } + }) + ) + + const reopened = await open() + expect(reopened.isReadOnly).toBe(true) + await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ + code: 'journal_read_only' + }) + await reopened.close() + // Never skipped, never deleted: the row this build cannot read is still there. + await withDatabase((db) => { + const stored = db.prepare('SELECT row_json FROM journal_rows WHERE seq = ?').get(nextSeq) as { + row_json: string + } + expect(stored.row_json).toContain('"v":99') + }) + }) + + it('skips a malformed row without giving up the journal, and discloses the skip', async () => { + const journal = await open() + await journal.appendItem(item(0), body('a'), { fence: 1 }) + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() + await appendRawRow(epoch, nextSeq, '{not json') const reopened = await open() expect(reopened.isReadOnly).toBe(false) @@ -132,10 +196,12 @@ describe('schema', () => { it('keeps one disclosure row across reopens instead of stacking duplicates', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}{not json\n`, 'utf-8') + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() + await appendRawRow(epoch, nextSeq, '{not json') - await open() + await open().then((first) => first.close()) const reopened = await open() expect( reopened @@ -146,168 +212,32 @@ describe('schema', () => { ).toHaveLength(1) }) - it('repairs a torn tail before acknowledging the next append', async () => { + it('reopens a journal holding an admitted malformed-percent item id without throwing', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - const intact = await readFile(logPath, 'utf-8') - await writeFile(logPath, intact.slice(0, -1), 'utf-8') - - await journal.appendItem(item(1), body('b'), { fence: 1 }) - const reopened = await open() - expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([body('a'), body('b')]) - }) - - // Transcripts are full of emoji and CJK, so the repair's file offsets must be - // bytes: string indices would truncate mid-character and corrupt the prefix. - it('repairs a torn tail whose rows contain multi-byte characters', async () => { - const journal = await open() - await journal.appendItem(item(0), body('안녕하세요 🌊 café'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - const intact = await readFile(logPath) - // Kill mid-row: keep the complete first row plus a fragment of the second. - const torn = Buffer.concat([intact, Buffer.from('{"seq":2,"kind":"it', 'utf-8')]) - await writeFile(logPath, torn) - - await journal.appendItem(item(1), body('b'), { fence: 1 }) - const reopened = await open() - expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([ - body('안녕하세요 🌊 café'), - body('b') - ]) - }) - - it('degrades to read-only when the snapshot comes from a newer schema', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - - const reopened = await open() - expect(reopened.isReadOnly).toBe(true) - await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ - code: 'journal_read_only' - }) - }) - - it('preserves a future-version snapshot with an unknown body kind in place instead of quarantining it', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - // The version advances because bodies changed: a valid newer snapshot - // carries kinds this build cannot parse and must stay unreadable in place. - snapshot.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'future-render-kind', payload: { anything: true } }, - sequence: 1, - observedAt: 1_000 - } - ] - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - - const reopened = await open() - const entries = await readdir(root) - expect(entries.some((name) => name.startsWith('quarantine-'))).toBe(false) - expect(entries.includes(JOURNAL_SNAPSHOT_FILE)).toBe(true) - expect(reopened.isReadOnly).toBe(true) - expect(reopened.snapshot().items).toHaveLength(0) - await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ - code: 'journal_read_only' - }) - }) - - it('keeps the future-version snapshot bytes when the schema escape hatch rolls the epoch', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - snapshot.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'future-render-kind', payload: { anything: true } }, - sequence: 1, - observedAt: 1_000 - } - ] - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - const reopened = await open() - // Still live in place before the explicit escape hatch runs. - expect((await readdir(root)).some((name) => name.startsWith('quarantine-'))).toBe(false) - - await reopened.rollEpoch('schema_unreadable', 2) - expect(reopened.isReadOnly).toBe(false) - const quarantine = (await readdir(root)).find((name) => name.startsWith('quarantine-')) - expect(quarantine).toBeDefined() - expect(await readFile(join(root, quarantine!), 'utf-8')).toContain('future-render-kind') - }) - - it('reopens a log holding an admitted malformed-percent item id without throwing', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() // `parseJournalRow` admits any string itemId, so replay must degrade a // malformed percent key to an opaque id instead of throwing URIError. - const malformedKeyRow = JSON.stringify({ - v: 1, - epoch: journal.epoch, - seq: journal.cursor().sequence + 1, - fence: 1, - ts: 1, - kind: 'item', - itemId: '%', - revision: 1, - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] } - }) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}${malformedKeyRow}\n`, 'utf-8') + await appendRawRow( + epoch, + nextSeq, + JSON.stringify({ + v: 1, + epoch, + seq: nextSeq, + fence: 1, + ts: 1, + kind: 'item', + itemId: '%', + revision: 1, + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] } + }) + ) const reopened = await open() expect(reopened.isReadOnly).toBe(false) expect(reopened.snapshot().items.some((entry) => entry.itemId === '%')).toBe(true) }) - - it('allows the explicit schema-unreadable epoch escape hatch while preserving the old files', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - const reopened = await open() - - await reopened.rollEpoch('schema_unreadable', 2) - expect(reopened.isReadOnly).toBe(false) - expect(reopened.snapshot().items).toHaveLength(0) - expect((await readdir(root)).some((name) => name.startsWith('quarantine-'))).toBe(true) - }) - - it('keeps the unreadable log suffix in the schema escape quarantine', async () => { - const journal = await open() - const logPath = join(root, JOURNAL_LOG_FILE) - const future = JSON.stringify({ - v: 99, - kind: 'item', - epoch: journal.epoch, - seq: 2, - fence: 1, - ts: 1, - itemId: 'future', - revision: 1, - body: { kind: 'status', text: 'preserve these bytes' } - }) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}${future}\n`, 'utf-8') - const reopened = await open() - - await reopened.rollEpoch('schema_unreadable', 2) - const quarantine = (await readdir(root)).find((name) => name.startsWith('quarantine-')) - expect(quarantine).toBeDefined() - expect(await readFile(join(root, quarantine!), 'utf-8')).toContain('preserve these bytes') - }) }) diff --git a/src/main/native-chat/agent-session-journal/journal-store-test-open.ts b/src/main/native-chat/agent-session-journal/journal-store-test-open.ts new file mode 100644 index 00000000000..fc8bce5af39 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-test-open.ts @@ -0,0 +1,34 @@ +// Shared tracked-open helper for journal tests. +// +// Tracking every opened INSTANCE rather than a variable is the point: `close()` +// is idempotent, a module-level `journal` binding can be reassigned to a second +// store mid-suite, and `allSettled` means one failing close cannot skip the rest +// or the directory removal behind it. + +import type { AgentSessionJournal } from './journal-store' +import type { AgentSessionJournalOptions } from './journal-store-contracts' +import { openAgentSessionJournal } from './journal-store-factory' + +export type TrackedJournalOpener = { + open: (options: AgentSessionJournalOptions) => Promise + track: (journal: T) => T + closeAll: () => Promise +} + +export function createTrackedJournalOpener(): TrackedJournalOpener { + const opened: AgentSessionJournal[] = [] + return { + open: async (options) => { + const journal = await openAgentSessionJournal(options) + opened.push(journal) + return journal + }, + track: (journal) => { + opened.push(journal) + return journal + }, + closeAll: async () => { + await Promise.allSettled(opened.splice(0).map((journal) => journal.close())) + } + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store.test.ts b/src/main/native-chat/agent-session-journal/journal-store.test.ts index 9ff37774d27..97eac02fe75 100644 --- a/src/main/native-chat/agent-session-journal/journal-store.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-store.test.ts @@ -1,4 +1,4 @@ -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' +import { mkdtemp, readdir, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -11,24 +11,17 @@ import { boundJournalKeyComponent, MAX_JOURNAL_KEY_COMPONENT_CHARS } from '../../../shared/agent-session-journal-item-key' -import { readJournalBlob } from './journal-blob-store' -import { JOURNAL_LOG_FILE, JOURNAL_SNAPSHOT_FILE } from './journal-log-file' import { loadJournal } from './journal-open' import { boundInlineText, boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalDirectoryFor, journalPathSegment } from './journal-paths' -import { journalDirectoryBytes } from './journal-physical-quota' -import type { JournalLifecycleMutationInput } from './journal-row-builders' +import { journalDatabaseFile, journalDirectoryFor, journalPathSegment } from './journal-paths' import { AgentSessionJournalError, type AgentSessionJournal } from './journal-store' -import { openAgentSessionJournal } from './journal-store-factory' -import { - JOURNAL_DISPATCH_RESERVATION_BYTES, - JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES, - JOURNAL_TURN_TERMINAL_RESERVATION_BYTES -} from './journal-lifecycle-capacity' +import type { openAgentSessionJournal } from './journal-store-factory' +import { createTrackedJournalOpener } from './journal-store-test-open' +import type Database from '../../sqlite/sync-database' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -54,8 +47,10 @@ function body(value: string): AgentJournalItemBody { return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } } +const journals = createTrackedJournalOpener() + async function open(overrides: Partial[0]> = {}) { - return openAgentSessionJournal({ + return journals.open({ identity: IDENTITY, journalDir: root, now: tick, @@ -70,6 +65,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -151,14 +147,12 @@ describe('fences', () => { }) describe('replay', () => { - it('adopts a caller-provided load without reading the journal files again', async () => { + it('adopts a caller-provided load without replaying the rows again', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) const loaded = await loadJournal(root, IDENTITY.sessionId) expect(loaded).not.toBeNull() - - await rm(join(root, JOURNAL_LOG_FILE), { force: true }) - await rm(join(root, JOURNAL_SNAPSHOT_FILE), { force: true }) + await journal.close() const reopened = await open({ loaded }) expect(reopened.snapshot()).toEqual(journal.snapshot()) @@ -200,208 +194,27 @@ describe('replay', () => { expect(reopened.snapshot().items).toHaveLength(0) }) - it('preserves the intact prefix and quarantines a corrupt suffix', async () => { + it('keeps the intact prefix and drops the rejected suffix', async () => { const journal = await open() for (let index = 0; index < 4; index += 1) { await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) } const before = journal.epoch - const logPath = join(root, JOURNAL_LOG_FILE) - const lines = (await readFile(logPath, 'utf-8')).split('\n').filter(Boolean) - await writeFile(logPath, `${[...lines.slice(0, 2), ...lines.slice(3)].join('\n')}\n`, 'utf-8') + await journal.close() + await withJournalDatabase(root, (db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(3) + }) const reopened = await open() expect(reopened.epoch).toBe(before) expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([body('m0')]) - const files = await readdir(root) - expect(files.some((name) => name.startsWith('quarantine-'))).toBe(true) - }) -}) - -describe('automatic compaction', () => { - // Production passes no policy and never called compact(), so the log only - // ever grew — until the size bound refused every append for good. - it('compacts on append once the retention window has rows to shed', async () => { - const policy = { minTailRows: 2, retainTailMs: 0 } - const journal = await open({ compaction: policy }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - expect(journal.compactionBoundary).toBeGreaterThan(0) - // The log sheds instead of growing with every append (7 = epoch row + 6). - const log = await readFile(join(root, JOURNAL_LOG_FILE), 'utf-8') - expect(log.trim().split('\n').length).toBeLessThan(7) - // Nothing is lost: the folded prefix is served from the snapshot. - expect(journal.snapshot().items).toHaveLength(6) - }) - - it('does not rewrite the log while every row is inside the retention window', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 60_000 } }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - expect(journal.compactionBoundary).toBe(0) - }) - - it('can be turned off explicitly', async () => { - const journal = await open({ - autoCompact: false, - compaction: { minTailRows: 2, retainTailMs: 0 } + // Sequences 4 and 5 are VALID rows that the gap at 3 made unreplayable. + // Nothing preserves them; recovery rebuilds the epoch from provider history. + await withJournalDatabase(root, (db) => { + const rows = db.prepare('SELECT seq FROM journal_rows ORDER BY seq').all() + expect(rows.map((row) => (row as { seq: number }).seq)).toEqual([1, 2]) }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - expect(journal.compactionBoundary).toBe(0) - }) - - it('refuses an append when a tail shorter than the row floor cannot make room', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 900 }, - // The tail never reaches the floor, so honouring it would shed nothing. - compaction: { minTailRows: 512, retainTailMs: 10_000 } - }) - let rejected = 0 - for (let index = 0; index < 20; index += 1) { - try { - await journal.appendItem(item(index), body('x'.repeat(96)), { fence: 1 }) - } catch (error) { - expect(error).toMatchObject({ code: 'journal_bound_exceeded' }) - rejected += 1 - } - } - expect(rejected).toBeGreaterThan(0) - }) - - it('refuses once the retained snapshot itself reaches the session bound', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 10_000 }, - compaction: { minTailRows: 10, retainTailMs: 2 * 60 * 60 * 1000 } - }) - let rejected = 0 - for (let index = 0; index < 30; index += 1) { - try { - await journal.appendItem(item(index), body('x'.repeat(128)), { fence: 1 }) - } catch (error) { - expect(error).toMatchObject({ code: 'journal_bound_exceeded' }) - rejected += 1 - } - } - - expect(rejected).toBeGreaterThan(0) - expect(journal.snapshot().items.length).toBeLessThan(30) - }) - - it('keeps the newest rows resumable while shedding under budget pressure', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 10_000 }, - compaction: { minTailRows: 10, retainTailMs: 2 * 60 * 60 * 1000 } - }) - for (let index = 0; index < 30; index += 1) { - await journal.appendItem(item(index), body('x'.repeat(64)), { fence: 1 }) - } - - // The window yields oldest-first, never wholesale: the latest append is - // still in the log, so a client resuming from it does not reload. - const log = (await readFile(join(root, JOURNAL_LOG_FILE), 'utf-8')).trim().split('\n') - expect(log.length).toBeGreaterThan(0) - expect(log.at(-1)).toContain('"seq"') - }) -}) - -describe('compaction and retention', () => { - it('preserves the highest fence across compaction and reopen', async () => { - const journal = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await journal.appendItem(item(0), body('a'), { fence: 7 }) - await journal.compact() - const reopened = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await expect(reopened.appendItem(item(1), body('stale'), { fence: 6 })).rejects.toMatchObject({ - code: 'journal_stale_fence' - }) - }) - - it('preserves tombstones across compaction and reopen', async () => { - const journal = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await journal.appendItem(item(0), body('a'), { fence: 1 }) - await journal.appendTombstone(item(0), { fence: 1 }) - await journal.compact() - const reopened = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await reopened.appendItem(item(0), body('stale'), { fence: 1 }) - expect(reopened.snapshot().items).toHaveLength(0) - }) - - it('folds the prefix into the snapshot and keeps serving the retained tail', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - const rendered = journal.snapshot() - const tip = journal.cursor() - await journal.compact() - - expect(journal.snapshot()).toEqual(rendered) - expect(journal.readSince({ epoch: tip.epoch, sequence: 1 })).toEqual({ - ok: false, - reset: 'cursor_compacted' - }) - const nearTip = journal.readSince({ epoch: tip.epoch, sequence: tip.sequence - 1 }) - expect(nearTip.ok && nearTip.rows).toHaveLength(1) - - const reopened = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - expect(reopened.snapshot()).toEqual(rendered) - expect(reopened.compactionBoundary).toBe(tip.sequence) - }) - - it('publishes the snapshot and its tail as one write, so a crash before the log rewrite loses nothing', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - for (let index = 0; index < 5; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - const rendered = journal.snapshot() - const logBefore = await readFile(join(root, JOURNAL_LOG_FILE), 'utf-8') - await journal.compact() - const persistedSnapshot = JSON.parse( - await readFile(join(root, JOURNAL_SNAPSHOT_FILE), 'utf-8') - ) as { tail: unknown[] } - expect(persistedSnapshot.tail).toHaveLength(2) - // Simulate the crash: the snapshot landed, the truncation did not. - await writeFile(join(root, JOURNAL_LOG_FILE), logBefore, 'utf-8') - - const reopened = await open() - expect(reopened.snapshot()).toEqual(rendered) - expect(reopened.snapshot().items).toHaveLength(5) - }) - - it('prunes blobs no live row references and keeps the ones that survive', async () => { - const journal = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - const kept = boundPayload('k'.repeat(64), { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 8 - }) - const dropped = boundPayload('d'.repeat(64), { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 8 - }) - const { putJournalBlob } = await import('./journal-blob-store') - await putJournalBlob(root, kept.digest, 'k'.repeat(64)) - await putJournalBlob(root, dropped.digest, 'd'.repeat(64)) - await journal.appendItem( - item(0), - { kind: 'tool-call', name: 'bash', input: {}, state: 'completed', output: kept }, - { fence: 1 } - ) - await journal.compact() - - expect(await readJournalBlob(root, kept.digest)).toBe('k'.repeat(64)) - expect(await readJournalBlob(root, dropped.digest)).toBeNull() - }) - - it('refuses a blob name that is not a bare digest, on either slash', async () => { - const { putJournalBlob } = await import('./journal-blob-store') - // A corrupt or crafted row must not steer a read or a write out of the store. - for (const name of ['../../escape', '..\\..\\escape', 'nested/name', 'NOTHEX']) { - expect(await readJournalBlob(root, name)).toBeNull() - await expect(putJournalBlob(root, name, 'payload')).rejects.toThrow('sha256 digest') - } + expect(reopened.repair).toEqual({ malformedRows: 0 }) }) }) @@ -429,244 +242,9 @@ describe('bounds', () => { expect(bounded.head).toBe('small') expect(boundInlineText('small', DEFAULT_JOURNAL_PAYLOAD_LIMITS).text).toBe('small') }) - - it('refuses a single row larger than the per-session size bound', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - }) - // Shedding the whole tail still cannot make room, so the bound holds. - await expect( - journal.appendItem(item(0), body('x'.repeat(4_096)), { fence: 1 }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - }) - - it('refuses an append past the per-session size bound when compaction is off', async () => { - const journal = await open({ - autoCompact: false, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - }) - await expect( - (async () => { - for (let index = 0; index < 50; index += 1) { - await journal.appendItem(item(index), body('x'.repeat(64)), { fence: 1 }) - } - })() - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - }) - - it('refuses an append past the per-window rate bound', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxAppendsPerWindow: 3, appendWindowMs: 60_000 } - }) - await expect( - (async () => { - for (let index = 0; index < 10; index += 1) { - await journal.appendItem(item(index), body('x'), { fence: 1 }) - } - })() - ).rejects.toMatchObject({ code: 'journal_rate_exceeded' }) - }) - - it('charges unique blobs and abandoned staging files to one physical quota', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 8_000 } - const journal = await open({ limits, autoCompact: false }) - const payload = 'z'.repeat(1_200) - const bounded = boundPayload(payload, { ...limits, inlineHeadBytes: 8 }) - const toolBody: AgentJournalItemBody = { - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed', - output: bounded - } - - await journal.appendItemWithBlobs(item(1), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - const afterFirst = await journalDirectoryBytes(root) - await journal.appendItemWithBlobs(item(2), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - const afterDuplicate = await journalDirectoryBytes(root) - - expect(afterDuplicate - afterFirst).toBeLessThan(payload.length) - expect(await readdir(join(root, 'blobs'))).toEqual([bounded.digest]) - expect(afterDuplicate).toBeLessThanOrEqual(limits.maxSessionBytes) - - await writeFile(join(root, 'log.jsonl.abandoned.tmp'), 's'.repeat(2_000), 'utf8') - const physical = await journalDirectoryBytes(root) - await expect( - open({ limits: { ...limits, maxSessionBytes: physical - 1 } }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - }) - - it('uses a running tool reservation when its authoritative blob cannot fit', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 220 * 1024 } - const journal = await open({ limits, autoCompact: false }) - await journal.appendItem( - item(1), - { kind: 'tool-call', name: 'command', input: {}, state: 'running' }, - { fence: 1 } - ) - for (let ordinal = 10; ordinal < 100; ordinal += 1) { - try { - await journal.appendItem(item(ordinal), body('f'.repeat(4_000)), { fence: 1 }) - } catch (error) { - expect(error).toMatchObject({ code: 'journal_bound_exceeded' }) - break - } - } - const payload = 'o'.repeat(100 * 1024) - const bounded = boundPayload(payload, { ...limits, inlineHeadBytes: 16 * 1024 }) - - await journal.appendItemWithBlobs( - item(1), - { - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed', - output: bounded - }, - [{ digest: bounded.digest, payload }], - { fence: 1 } - ) - - const tool = journal.snapshot().items.find((entry) => entry.itemId.includes('turn-1:1')) - expect(tool?.body).toEqual({ - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed' - }) - expect( - journal - .snapshot() - .items.some( - (entry) => - entry.body.kind === 'status' && entry.body.text.includes('could not be retained') - ) - ).toBe(true) - expect(await readJournalBlob(root, bounded.digest)).toBeNull() - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('keeps cached physical bytes aligned after blob dedupe and compaction', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 45_000 } - const journal = await open({ - limits, - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 } - }) - const payload = 'p'.repeat(20_000) - const bounded = boundPayload(payload, { ...limits, inlineHeadBytes: 8 }) - const toolBody: AgentJournalItemBody = { - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed', - output: bounded - } - - await journal.appendItemWithBlobs(item(1), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - await journal.appendItemWithBlobs(item(2), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - await journal.compact(tick() + 10, { minTailRows: 0, retainTailMs: 0 }) - const compactedBytes = await journalDirectoryBytes(root) - expect(compactedBytes).toBeLessThan(limits.maxSessionBytes) - - await journal.appendItem(item(3), body('after compaction'), { fence: 1 }) - - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) - expect(await readdir(join(root, 'blobs'))).toEqual([bounded.digest]) - }) }) describe('lifecycle batches', () => { - it('uses a reserved append slot after ordinary rate pressure', async () => { - const journal = await open({ - autoCompact: false, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxAppendsPerWindow: 1, appendWindowMs: 60_000 } - }) - const identity: AgentJournalItemIdentity = { - provider: 'orca', - clientMessageId: 'reserved-prompt' - } - const pending: AgentJournalItemBody = { - kind: 'approval', - title: 'Run a command?', - detail: null, - options: [{ id: 'accept', label: 'Allow' }], - resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } - } - - // The pending row spends the only ordinary slot while reserving its - // terminal append slot for recovery. - await journal.appendLifecycleBatch({ - settlementId: 'reserved-start', - fence: 1, - mutations: [{ kind: 'item', identity, body: pending }] - }) - await expect( - journal.appendItem( - identity, - { - ...pending, - resolution: { - state: 'resolved', - selectedOptionId: 'accept', - resolvedBy: 'test', - resolvedAt: 1 - } - }, - { fence: 1 } - ) - ).resolves.toBeDefined() - }) - - it('rate-limits an unreserved lifecycle batch', async () => { - const journal = await open({ - autoCompact: false, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxAppendsPerWindow: 1, appendWindowMs: 60_000 } - }) - const mutation = (id: string): JournalLifecycleMutationInput => ({ - kind: 'item', - identity: { provider: 'orca', clientMessageId: id }, - body: { kind: 'status', text: 'provider diagnostic' } - }) - await journal.appendLifecycleBatch({ - settlementId: 'unreserved-1', - fence: 1, - mutations: [mutation('one')] - }) - await expect( - journal.appendLifecycleBatch({ - settlementId: 'unreserved-2', - fence: 1, - mutations: [mutation('two')] - }) - ).rejects.toMatchObject({ code: 'journal_rate_exceeded' }) - }) - - it('rebuilds dispatch and turn reservations for pending submissions after reopen', async () => { - const journal = await open({ autoCompact: false }) - await journal.appendSubmission({ - clientMessageId: 'pending-send', - payloadFingerprint: 'fingerprint', - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, - fence: 1 - }) - - const reopened = await open({ autoCompact: false }) - expect(reopened.lifecycleCapacityState()).toEqual({ - reservedBytes: JOURNAL_DISPATCH_RESERVATION_BYTES + JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - reservedAppendSlots: 2 - }) - }) - it('deduplicates concurrent submissions before appending a second row', async () => { const journal = await open() const input = { @@ -684,7 +262,7 @@ describe('lifecycle batches', () => { }) it('applies every mutation at one sequence and deduplicates a replay across reopen', async () => { - const journal = await open({ autoCompact: false }) + const journal = await open() const turn: AgentJournalItemIdentity = { provider: 'legacy', agent: 'codex', @@ -715,9 +293,9 @@ describe('lifecycle batches', () => { .snapshot() .items.some((entry) => entry.body.kind === 'status' && entry.body.text === 'working') ).toBe(false) - await journal.compact(tick() + 10, { minTailRows: 0, retainTailMs: 0 }) + await journal.close() - const reopened = await open({ autoCompact: false }) + const reopened = await open() const beforeReplay = reopened.cursor() const replay = await reopened.appendLifecycleBatch({ settlementId: 'exit:turn-1', @@ -737,53 +315,6 @@ describe('lifecycle batches', () => { ) ).toBe(false) }) - - it('reserves and releases terminal prompts created inside lifecycle batches', async () => { - const journal = await open() - const identity: AgentJournalItemIdentity = { provider: 'orca', clientMessageId: 'prompt-1' } - const pending: AgentJournalItemBody = { - kind: 'approval', - title: 'Run a command?', - detail: null, - options: [{ id: 'accept', label: 'Allow' }], - resolution: { - state: 'pending', - selectedOptionId: null, - resolvedBy: null, - resolvedAt: null - } - } - - await journal.appendLifecycleBatch({ - settlementId: 'prompt-start', - fence: 1, - mutations: [{ kind: 'item', identity, body: pending }] - }) - - expect(journal.lifecycleCapacityState()).toEqual({ - reservedBytes: JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES, - reservedAppendSlots: 1 - }) - - await journal.appendItem( - identity, - { - ...pending, - resolution: { - state: 'resolved', - selectedOptionId: 'accept', - resolvedBy: 'test', - resolvedAt: tick() - } - }, - { fence: 1 } - ) - - expect(journal.lifecycleCapacityState()).toEqual({ - reservedBytes: 0, - reservedAppendSlots: 0 - }) - }) }) describe('journal location', () => { @@ -808,12 +339,32 @@ describe('journal location', () => { }) describe('on-disk layout', () => { - it('writes the log and snapshot beside each other', async () => { + it('keeps the session database and its projection in one directory', async () => { const journal: AgentSessionJournal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - await expect(readFile(join(root, JOURNAL_LOG_FILE), 'utf-8')).resolves.toContain( - '"kind":"item"' - ) - await expect(readFile(join(root, JOURNAL_SNAPSHOT_FILE), 'utf-8')).resolves.toContain('"epoch"') + expect(await readdir(root)).toContain('journal.db') + await journal.close() + await withJournalDatabase(root, (db) => { + const row = db.prepare('SELECT row_json FROM journal_rows WHERE seq = 2').get() + expect((row as { row_json: string }).row_json).toContain('"kind":"item"') + expect(db.prepare('SELECT epoch FROM journal_sessions').get()).toMatchObject({ + epoch: journal.epoch + }) + }) }) }) + +/** Opens the session database directly, so a case can stage a fault or read + * back what a commit actually stored. */ +async function withJournalDatabase( + journalDir: string, + run: (db: Database.Database) => void +): Promise { + const { openJournalDatabase } = await import('./journal-database') + const opened = openJournalDatabase(journalDatabaseFile(journalDir)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store.ts b/src/main/native-chat/agent-session-journal/journal-store.ts index 33f9be0c2e4..ab2715d0d86 100644 --- a/src/main/native-chat/agent-session-journal/journal-store.ts +++ b/src/main/native-chat/agent-session-journal/journal-store.ts @@ -11,20 +11,16 @@ import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' -import { - compactJournal, - DEFAULT_JOURNAL_COMPACTION_POLICY, - type JournalCompactionPolicy -} from './journal-compaction' +import { agentSessionJournalCloseRetries } from './journal-close-retry' +import { openJournalDatabase, type OpenJournalDatabase } from './journal-database' import type { JournalReplacementItem } from './journal-epoch-replacement' import { readJournalSince } from './journal-cursor' -import type { JournalLoad } from './journal-open' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' +import { readJournalRowsAfterCursor, type JournalLoad } from './journal-open' +import { journalDatabaseFile } from './journal-paths' import { markJournalPendingSubmissionsUnknown } from './journal-pending-submission-recovery' import { applyJournalRow, createJournalReducerState, - referencedBlobDigests, renderJournalState, resolveJournalItemId, type JournalReducerState @@ -37,7 +33,6 @@ import { import type { AgentSessionJournalOptions, JournalAppendResult, - JournalBlobInput, JournalItemAppendOptions, JournalLifecycleBatchInput, JournalReadSince, @@ -46,112 +41,79 @@ import type { ResolveDispatchInput } from './journal-store-contracts' import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -import { assertJournalWritable, JournalAppendBudget } from './journal-write-guards' -import { journalDirectoryBytes } from './journal-physical-quota' -import type { JournalLifecycleReservation } from './journal-lifecycle-capacity' -import { JournalLifecycleAdmission } from './journal-lifecycle-admission' -import { JournalRowWriter } from './journal-row-writer' -import { JournalEpochController } from './journal-epoch-controller' -import { journalStoreLoadedFields, openJournalStoreState } from './journal-store-open' -import { JournalItemAppender } from './journal-item-appender' -import { JournalLifecycleBatchAppender } from './journal-lifecycle-batch-appender' +import { AgentSessionJournalError } from './journal-write-guards' +import type { JournalRowWriter } from './journal-row-writer' +import type { JournalEpochController } from './journal-epoch-controller' +import { JournalConnectionCloser, JournalWriteQueue } from './journal-store-close' +import { createJournalStoreCollaborators } from './journal-store-collaborators' +import { ensureJournalDir, journalStoreLoadedFields } from './journal-store-open' +import type { JournalItemAppender } from './journal-item-appender' +import type { JournalLifecycleBatchAppender } from './journal-lifecycle-batch-appender' export { AgentSessionJournalError } from './journal-write-guards' export class AgentSessionJournal { private readonly identity: AgentSessionJournalIdentity private readonly journalDir: string - private readonly budget: JournalAppendBudget - private readonly compaction: JournalCompactionPolicy - private readonly autoCompact: boolean + private readonly dbPath: string private readonly now: () => number private readonly mintEpoch: () => string private readonly loaded: JournalLoad | null | undefined private state: JournalReducerState - private tailRows: JournalRow[] = [] - private compactedThrough = 0 - private sizeBytes = 0 private readOnly = false private malformedRows = 0 - private readonly lifecycleAdmission: JournalLifecycleAdmission + private database: OpenJournalDatabase | null = null + private readonly queue: JournalWriteQueue + private readonly closer: JournalConnectionCloser private readonly rowWriter: JournalRowWriter private readonly epochController: JournalEpochController private readonly itemAppender: JournalItemAppender private readonly lifecycleBatchAppender: JournalLifecycleBatchAppender - /** Serializes sequence assignment with the durable write behind it. */ - private writes: Promise = Promise.resolve() + private readonly restore: () => Promise constructor(options: AgentSessionJournalOptions) { this.identity = options.identity this.journalDir = options.journalDir - this.budget = new JournalAppendBudget( - options.identity.sessionId, - options.limits ?? DEFAULT_JOURNAL_PAYLOAD_LIMITS - ) - this.autoCompact = options.autoCompact ?? true - this.compaction = options.compaction ?? DEFAULT_JOURNAL_COMPACTION_POLICY + this.dbPath = journalDatabaseFile(options.journalDir) this.now = options.now ?? (() => Date.now()) this.mintEpoch = options.mintEpoch ?? randomUUID this.loaded = options.loaded this.state = createJournalReducerState(options.identity.sessionId, '') - this.lifecycleAdmission = new JournalLifecycleAdmission( - options.identity.sessionId, - this.budget.maxSessionBytes, - (itemId) => resolveJournalItemId(this.state, itemId), - this.budget.maxAppendsPerWindow - ) - this.rowWriter = new JournalRowWriter({ - journalDir: this.journalDir, - sessionId: options.identity.sessionId, - budget: this.budget, - lifecycleAdmission: this.lifecycleAdmission, - autoCompact: this.autoCompact, - compaction: this.compaction, - now: this.now, - serialize: (run) => this.serializeWrite(run), - readOnly: () => this.readOnly, - setReadOnly: (readOnly) => { - this.readOnly = readOnly - }, - physicalBytes: () => this.sizeBytes, - highestFence: () => this.state.highestFence, - nextSequence: () => this.state.lastSequence + 1, - tailRows: () => this.tailRows, - referencedBlobDigests: () => referencedBlobDigests(this.state), - compact: (now, policy) => this.compact(now, policy), - commit: (row, physicalBytes) => { - applyJournalRow(this.state, row) - this.tailRows.push(row) - this.sizeBytes = physicalBytes - } + // Serializes sequence assignment with the durable write behind it. + this.queue = new JournalWriteQueue(options.identity.sessionId) + this.closer = new JournalConnectionCloser({ + connection: () => this.database?.db ?? null, + enqueue: (run) => this.queue.serializePastGate(run) }) - this.epochController = new JournalEpochController({ + const collaborators = createJournalStoreCollaborators({ identity: this.identity, journalDir: this.journalDir, - budget: this.budget, - compaction: this.compaction, now: this.now, mintEpoch: this.mintEpoch, - serialize: (run) => this.serializeWrite(run), + serialize: (run) => this.queue.serialize(run), + database: () => this.requireDatabase(), + state: () => this.state, readOnly: () => this.readOnly, setReadOnly: (readOnly) => { this.readOnly = readOnly }, - highestFence: () => this.state.highestFence, cursor: this.cursor, - adopt: (loaded) => this.adoptLoadedJournal(loaded) - }) - this.itemAppender = new JournalItemAppender({ + adopt: (loaded) => this.adoptLoadedJournal(loaded), + commit: (row) => applyJournalRow(this.state, row), + loaded: () => this.loaded, + malformedRows: () => this.malformedRows, + setMalformedRows: (count) => { + this.malformedRows = count + }, journal: () => this, - state: () => this.state, - enqueue: (build, blobs) => this.enqueue(build, blobs) - }) - this.lifecycleBatchAppender = new JournalLifecycleBatchAppender({ - state: () => this.state, - cursor: this.cursor, enqueue: (build) => this.enqueue(build) }) + this.rowWriter = collaborators.rowWriter + this.epochController = collaborators.epochController + this.itemAppender = collaborators.itemAppender + this.lifecycleBatchAppender = collaborators.lifecycleBatchAppender + this.restore = collaborators.restore } get isReadOnly(): boolean { @@ -166,31 +128,31 @@ export class AgentSessionJournal { return this.journalDir } - /** Highest sequence folded into the snapshot; rows at or below it are no - * longer individually replayable. */ - get compactionBoundary(): number { - return this.compactedThrough + /** What the last open's repair did. */ + get repair(): { malformedRows: number } { + return { malformedRows: this.malformedRows } } async open(): Promise { - await openJournalStoreState({ - journalDir: this.journalDir, - sessionId: this.identity.sessionId, - maxBytes: this.budget.maxSessionBytes, - loaded: this.loaded, - start: () => this.epochController.start('session_created', 0), - adopt: (loaded) => this.adoptLoadedJournal(loaded), - tailRows: () => this.tailRows, - snapshot: this.snapshot, - rebuildLifecycle: (snapshot, bytes) => this.lifecycleAdmission.rebuild(snapshot, bytes), - appendDisclosure: (identity, body, fence) => this.appendItem(identity, body, { fence }), - highestFence: () => this.state.highestFence, - malformedRows: () => this.malformedRows, - readOnly: () => this.readOnly, - setPhysicalBytes: (bytes) => { - this.sizeBytes = bytes - } - }) + await ensureJournalDir(this.journalDir) + this.database = openJournalDatabase(this.dbPath) + try { + await this.restore() + } catch (error) { + // Nothing else holds a reference to this connection, so a throw here is + // the leak site unless the store releases it itself — and a close that + // REJECTS has not released it, so the store is retained for a later retry + // instead of being dropped with its handle open. + await agentSessionJournalCloseRetries.closeOrRetain(this) + throw error + } + } + + /** Releases the session's SQLite handle. Idempotent on success, a real retry + * after a failure, and permanently closed to writes either way (§ close). */ + close(): Promise { + this.queue.markClosed() + return this.closer.close() } cursor = (): AgentJournalCursor => ({ @@ -212,27 +174,19 @@ export class AgentSessionJournal { canonicalItemId = (itemId: string): string => resolveJournalItemId(this.state, itemId) - reserveLifecycleCapacity(token: JournalLifecycleReservation): Promise { - return this.serializeCapacityMutation(async () => { - this.sizeBytes = await journalDirectoryBytes(this.journalDir) - return this.lifecycleAdmission.reserve(token, this.sizeBytes) - }) - } - - transferLifecycleCapacity(fromId: string, toId: string): Promise { - return this.serializeCapacityMutation(() => this.lifecycleAdmission.transfer(fromId, toId)) - } - - releaseLifecycleCapacity(id: string): Promise { - return this.serializeCapacityMutation(() => this.lifecycleAdmission.release(id)) - } - - lifecycleCapacityState = (): { reservedBytes: number; reservedAppendSlots: number } => - this.lifecycleAdmission.state - readSince(cursor: AgentJournalCursor): JournalReadSince { return readJournalSince( - { state: this.state, tailRows: this.tailRows, readOnly: this.readOnly }, + { + state: this.state, + rowsAfter: (afterSequence) => + readJournalRowsAfterCursor( + this.requireDatabase().db, + this.identity.sessionId, + this.state.epoch, + afterSequence + ), + readOnly: this.readOnly + }, cursor, () => this.cursor() ) @@ -248,16 +202,6 @@ export class AgentSessionJournal { return this.itemAppender.append(identity, body, options) } - /** Blob-before-row admission on the same serialized path as sequence assignment. */ - appendItemWithBlobs( - identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly JournalBlobInput[], - options: JournalItemAppendOptions = { fence: 0 } - ): Promise { - return this.itemAppender.appendWithBlobs(identity, body, blobs, options) - } - appendTombstone( identity: AgentJournalItemIdentity, options: JournalTombstoneInput @@ -303,26 +247,6 @@ export class AgentSessionJournal { return markJournalPendingSubmissionsUnknown(this, fence) } - async compact( - now = this.now(), - policy: JournalCompactionPolicy = this.compaction - ): Promise { - assertJournalWritable(this.readOnly, this.identity.sessionId) - const result = await compactJournal({ - journalDir: this.journalDir, - state: this.state, - tailRows: this.tailRows, - policy, - now, - maxSessionBytes: this.budget.maxSessionBytes, - sessionId: this.identity.sessionId - }) - this.tailRows = result.tailRows - this.compactedThrough = result.compactedThrough - this.state.oldestSequence = result.oldestSequence - this.sizeBytes = await journalDirectoryBytes(this.journalDir) - } - /** The escape hatch for corruption, an unreconcilable prefix, a forked handle, * and an unreadable schema. It invalidates every cursor; clients reload. */ async rollEpoch(reason: AgentJournalEpochReason, fence: number): Promise { @@ -341,24 +265,22 @@ export class AgentSessionJournal { Object.assign(this, journalStoreLoadedFields(loaded)) } + private requireDatabase(): OpenJournalDatabase { + if (!this.database) { + throw new AgentSessionJournalError( + 'journal_closed', + `agent-session journal for ${this.identity.sessionId} is not open` + ) + } + return this.database + } + /** * Assign the next sequence, make the row durable, and fold it through the * SAME reducer replay uses — all inside one serialized step, so concurrent * callers cannot interleave and mint the same sequence. */ - private enqueue( - build: (seq: number, ts: number) => JournalRow, - blobs: readonly JournalBlobInput[] = [] - ): Promise { - return this.rowWriter.enqueue(build, blobs) - } - - private serializeCapacityMutation = (runMutation: () => Promise | T): Promise => - this.serializeWrite(async () => runMutation()) - - private serializeWrite(runWrite: () => Promise): Promise { - const run = this.writes.then(runWrite) - this.writes = run.catch(() => undefined) - return run + private enqueue(build: (seq: number, ts: number) => JournalRow): Promise { + return this.rowWriter.enqueue(build) } } diff --git a/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts new file mode 100644 index 00000000000..ade04667174 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts @@ -0,0 +1,13 @@ +import type { AgentJournalItemBody } from '../../../shared/agent-session-journal-types' + +/** True while an item is still awaiting the row that settles it, so a sink can + * treat that row as lifecycle-critical rather than sheddable under pressure. */ +export function requiresTerminalSettlement(body: AgentJournalItemBody): boolean { + if (body.kind === 'tool-call') { + return body.state === 'running' + } + if (body.kind === 'approval' || body.kind === 'question') { + return body.resolution.state === 'pending' + } + return body.kind === 'status' && body.turnLifecycle?.state === 'running' +} diff --git a/src/main/native-chat/agent-session-journal/journal-tool-output-fallback.ts b/src/main/native-chat/agent-session-journal/journal-tool-output-fallback.ts deleted file mode 100644 index 5b3209ea594..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-tool-output-fallback.ts +++ /dev/null @@ -1,58 +0,0 @@ -import type { - AgentJournalItemBody, - AgentJournalItemIdentity -} from '../../../shared/agent-session-journal-types' -import type { AgentSessionJournal } from './journal-store' -import type { JournalAppendResult } from './journal-store-contracts' -import { AgentSessionJournalError } from './journal-write-guards' -import { boundToolInput, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' - -export async function appendToolOutputFallback(input: { - journal: AgentSessionJournal - error: unknown - identity: AgentJournalItemIdentity - body: AgentJournalItemBody - blobs: readonly { digest: string; payload: string }[] - itemId: string - fence: number -}): Promise { - if ( - !(input.error instanceof AgentSessionJournalError) || - input.error.code !== 'journal_bound_exceeded' || - input.body.kind !== 'tool-call' || - input.body.state === 'running' || - input.blobs.length === 0 - ) { - throw input.error - } - const digest = input.blobs[0]?.digest ?? 'unknown' - const cursor = await input.journal.appendLifecycleBatch({ - settlementId: `tool-output-unavailable:${input.itemId}:${digest}`, - fence: input.fence, - mutations: [ - { - kind: 'item', - identity: input.identity, - body: { - kind: 'tool-call', - name: input.body.name, - input: boundToolInput(input.body.input, DEFAULT_JOURNAL_PAYLOAD_LIMITS), - state: input.body.state - } - }, - { - kind: 'item', - identity: { provider: 'orca', clientMessageId: `output-unavailable:${input.itemId}` }, - body: { - kind: 'status', - text: 'The tool completed, but its output could not be retained within the session storage limit.' - } - } - ] - }) - const item = input.journal.snapshot().items.find((entry) => entry.itemId === input.itemId) - if (!item) { - throw new Error('journal_tool_output_fallback_lost') - } - return { cursor, itemId: input.itemId, revision: item.revision } -} diff --git a/src/main/native-chat/agent-session-journal/journal-write-guards.ts b/src/main/native-chat/agent-session-journal/journal-write-guards.ts index 9f76de55745..e1869ae0c73 100644 --- a/src/main/native-chat/agent-session-journal/journal-write-guards.ts +++ b/src/main/native-chat/agent-session-journal/journal-write-guards.ts @@ -1,18 +1,11 @@ // Guards an append clears before it becomes durable. // -// All four refuse loudly rather than degrade: a silent drop here is a message +// Both refuse loudly rather than degrade: a silent drop here is a message // missing from the transcript with nothing to explain it. -import type { JournalPayloadLimits } from './journal-payload-bounds' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' - export class AgentSessionJournalError extends Error { constructor( - readonly code: - | 'journal_read_only' - | 'journal_stale_fence' - | 'journal_bound_exceeded' - | 'journal_rate_exceeded', + readonly code: 'journal_read_only' | 'journal_stale_fence' | 'journal_closed', message: string ) { super(message) @@ -41,94 +34,3 @@ export function assertJournalFence(fence: number, highestFence: number): void { ) } } - -/** Total size and append rate for one session, bounding a runaway agent. */ -export class JournalAppendBudget { - private windowStart = 0 - private appendsInWindow = 0 - - constructor( - private readonly sessionId: string, - private readonly limits: JournalPayloadLimits - ) {} - - fork(): JournalAppendBudget { - return new JournalAppendBudget(this.sessionId, this.limits) - } - - get maxSessionBytes(): number { - return this.limits.maxSessionBytes - } - - get maxAppendsPerWindow(): number { - return this.limits.maxAppendsPerWindow - } - - /** Capture rate state so a speculative append can be rolled back safely. */ - checkpoint(): { windowStart: number; appendsInWindow: number } { - return { windowStart: this.windowStart, appendsInWindow: this.appendsInWindow } - } - - restore(checkpoint: { windowStart: number; appendsInWindow: number }): void { - this.windowStart = checkpoint.windowStart - this.appendsInWindow = checkpoint.appendsInWindow - } - - wouldExceedSize(row: JournalRow, sizeBytes: number): boolean { - return sizeBytes + journalRowByteLength(row) > this.limits.maxSessionBytes - } - - assert(row: JournalRow, ts: number, sizeBytes: number): void { - if (this.wouldExceedSize(row, sizeBytes)) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} reached its ${this.limits.maxSessionBytes}-byte bound` - ) - } - this.assertRate(ts) - } - - /** Lifecycle capacity cannot bypass the session-wide append rate. */ - assertLifecycle(row: JournalRow, sizeBytes: number): void { - if (this.wouldExceedSize(row, sizeBytes)) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} reached its ${this.limits.maxSessionBytes}-byte bound` - ) - } - this.assertRate(row.ts) - } - - /** - * Consume a lifecycle row covered by a pre-reserved append slot. Reserved - * rows still observe the physical quota, but do not spend ordinary window - * rate headroom that may be needed by unrelated traffic. - */ - assertReservedLifecycle(row: JournalRow, sizeBytes: number): void { - if (this.wouldExceedSize(row, sizeBytes)) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} reached its ${this.limits.maxSessionBytes}-byte bound` - ) - } - } - - private assertRate(ts: number): void { - let windowStart = this.windowStart - let appendsInWindow = this.appendsInWindow - if (ts - windowStart >= this.limits.appendWindowMs) { - windowStart = ts - appendsInWindow = 0 - } - appendsInWindow += 1 - if (appendsInWindow > this.limits.maxAppendsPerWindow) { - // A refusal must not consume a slot, so a later retry can succeed. - throw new AgentSessionJournalError( - 'journal_rate_exceeded', - `agent-session journal for ${this.sessionId} exceeded ${this.limits.maxAppendsPerWindow} appends per ${this.limits.appendWindowMs}ms` - ) - } - this.windowStart = windowStart - this.appendsInWindow = appendsInWindow - } -} diff --git a/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts b/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts index 3930abb6fe8..cda878f5a01 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts @@ -1,4 +1,4 @@ -import { appendFile, mkdtemp, rm } from 'node:fs/promises' +import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -19,15 +19,16 @@ import { serializeRemoteRuntimePayload } from '../../../shared/remote-runtime-memory-limits' import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' -import { JOURNAL_LOG_FILE } from '../agent-session-journal/journal-log-file' -import { - serializeJournalRow, - type JournalItemRow, - type JournalRow, - type JournalTombstoneRow +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { insertJournalRow } from '../agent-session-journal/journal-row-table' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import type { + JournalItemRow, + JournalRow, + JournalTombstoneRow } from '../agent-session-journal/journal-row-schema' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' import { projectJournalBatch } from './agent-session-journal-batch' import { readAgentSessionHistory, resolveHistoryLimit } from './agent-session-history-page' @@ -39,6 +40,7 @@ const IDENTITY: AgentSessionJournalIdentity = { providerHandle: { kind: 'codex', threadId: 'thread-1' } } +const journals = createTrackedJournalOpener() let root: string let clock = 1_000 let epochs = 0 @@ -67,7 +69,7 @@ beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-wire-history-')) clock = 1_000 epochs = 0 - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: IDENTITY, journalDir: root, now: tick, @@ -79,6 +81,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -477,12 +480,20 @@ async function reopenWithRawRows(rows: readonly RawSeedRow[]): Promise { diff --git a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts index 2efcc08cde9..214c889b5c9 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts @@ -1,7 +1,9 @@ // Recovery drives the real journal loader against real on-disk damage: a hole -// punched in the log, and a row stamped with a schema this host cannot read. +// punched in the row sequence, and a row stamped with a schema this host cannot +// read — on both version axes, because only one of them is detectable before a +// read. -import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -9,7 +11,13 @@ import type { AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from '../agent-session-journal/journal-database-schema' +import { loadJournal } from '../agent-session-journal/journal-open' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { readJournalEpochRows } from '../agent-session-journal/journal-row-table' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import type Database from '../../sqlite/sync-database' import { openAgentSessionJournalWithRecovery, providerHistoryId, @@ -53,14 +61,15 @@ const CODEX_LINES = [ let root: string let journalDir: string let historyFilePath: string +const journals = createTrackedJournalOpener() function item(ordinal: number): AgentJournalItemIdentity { return { provider: 'codex', threadId: CODEX_SESSION, turnId: 'turn-1', ordinal } } -/** Fills a journal with `count` items and hands back the raw log lines. */ -async function seedJournal(count: number): Promise { - const journal = await openAgentSessionJournal({ identity: IDENTITY, journalDir }) +/** Fills a journal with `count` items and hands back its epoch. */ +async function seedJournal(count: number): Promise { + const journal = await journals.open({ identity: IDENTITY, journalDir }) for (let ordinal = 1; ordinal <= count; ordinal += 1) { await journal.appendItem( item(ordinal), @@ -68,8 +77,48 @@ async function seedJournal(count: number): Promise { { fence: 1 } ) } - const raw = await readFile(join(journalDir, 'log.jsonl'), 'utf-8') - return raw.split('\n').filter((line) => line.trim().length > 0) + const epoch = journal.epoch + await journal.close() + return epoch +} + +/** A journal whose epoch row is gone: every surviving row is unanchored, so a + * repair has to set aside the whole range. */ +async function seedRepairableSession(): Promise { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + await journal.appendSubmission({ + clientMessageId: 'client-message-1', + payloadFingerprint: 'fingerprint-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'add a retry' }] }, + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'client-message-1', + state: 'accepted', + providerIdentity: item(1), + fence: 1 + }) + await journal.close() + await deleteRow(1) +} + +async function withJournalDatabase( + directory: string, + run: (db: Database.Database) => void +): Promise { + const opened = openJournalDatabase(journalDatabaseFile(directory)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} + +/** The same logical hole `findSequenceGap` detects at replay. */ +async function deleteRow(seq: number): Promise { + await withJournalDatabase(journalDir, (db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(seq) + }) } beforeEach(async () => { @@ -84,6 +133,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -99,20 +149,20 @@ describe('providerHistoryId', () => { describe('openAgentSessionJournalWithRecovery', () => { it('opens a healthy journal untouched', async () => { await seedJournal(2) - const opened = await openAgentSessionJournalWithRecovery({ - identity: IDENTITY, - journalDir, - fence: 1, - historyFilePath - }) - expect(opened.recovery).toBeNull() - expect(opened.journal.snapshot().items).toHaveLength(2) + const opened = journals.track( + await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }).then((result) => result.journal) + ) + expect(opened.snapshot().items).toHaveLength(2) }) it('rebuilds a holed journal in place on a fresh epoch', async () => { - const lines = await seedJournal(3) - const holed = lines.filter((_line, index) => index !== 1) - await writeFile(join(journalDir, 'log.jsonl'), `${holed.join('\n')}\n`, 'utf-8') + await seedJournal(3) + await deleteRow(3) const opened = await openAgentSessionJournalWithRecovery({ identity: IDENTITY, @@ -120,6 +170,7 @@ describe('openAgentSessionJournalWithRecovery', () => { fence: 1, historyFilePath }) + journals.track(opened.journal) expect(opened.recovery).toMatchObject({ trigger: 'journal_corrupt', reset: 'epoch_changed' }) expect(opened.recovery?.imported).toBeGreaterThan(0) expect(opened.journal.isReadOnly).toBe(false) @@ -130,12 +181,46 @@ describe('openAgentSessionJournalWithRecovery', () => { expect(texts.some((text) => text.includes('add a retry'))).toBe(true) }) - it('reconstructs a future-schema journal into a schema-scoped sibling, never in place', async () => { - const lines = await seedJournal(1) - await writeFile( - join(journalDir, 'log.jsonl'), - `${lines.join('\n')}\n${JSON.stringify({ v: 99, seq: 2, epoch: 'e', kind: 'item' })}\n`, - 'utf-8' + it('reconstructs a future row-body version into a sibling, never in place', async () => { + const epoch = await seedJournal(1) + await withJournalDatabase(journalDir, (db) => { + db.prepare( + 'INSERT INTO journal_rows (session_id, epoch, seq, ts, row_json) VALUES (?, ?, ?, ?, ?)' + ).run( + CODEX_SESSION, + epoch, + 3, + 1, + JSON.stringify({ v: 99, seq: 3, epoch, kind: 'item', fence: 1, ts: 1 }) + ) + }) + + const opened = journals.track( + await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }).then((result) => result.journal) + ) + + // The unreadable journal is left exactly as found; a newer host still owns it. + await withJournalDatabase(journalDir, (db) => { + const rows = readJournalEpochRows(db, CODEX_SESSION, epoch) + expect(rows.some((entry) => entry.rowJson.includes('"v":99'))).toBe(true) + expect(rows).toHaveLength(3) + }) + await opened.close() + await withJournalDatabase(recoveryJournalDir(journalDir), (db) => { + const sibling = db.prepare('SELECT row_json FROM journal_rows').all() + expect(JSON.stringify(sibling)).toContain('add a retry') + }) + }) + + it('reconstructs a future database version into a sibling, never in place', async () => { + await seedJournal(1) + await withJournalDatabase(journalDir, (db) => + db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`) ) const opened = await openAgentSessionJournalWithRecovery({ @@ -144,23 +229,71 @@ describe('openAgentSessionJournalWithRecovery', () => { fence: 1, historyFilePath }) + journals.track(opened.journal) expect(opened.recovery).toMatchObject({ trigger: 'schema_unreadable', reset: 'schema_unreadable' }) expect(opened.recovery?.imported).toBeGreaterThan(0) + // No schema change, no row written, no row deleted. + await withJournalDatabase(journalDir, (db) => { + expect(db.pragma('user_version', { simple: true })).toBe(JOURNAL_DB_SCHEMA_VERSION + 1) + expect(db.prepare('SELECT count(*) AS total FROM journal_rows').get()).toMatchObject({ + total: 2 + }) + }) + }) - // The unreadable journal is left exactly as found; a newer host still owns it. - const untouched = await readFile(join(journalDir, 'log.jsonl'), 'utf-8') - expect(untouched).toContain('"v":99') - const sibling = await readFile(join(recoveryJournalDir(journalDir), 'log.jsonl'), 'utf-8') - expect(sibling).toContain('add a retry') + // The rehydrate deletes every live row to publish its replacement epoch, so + // everything replay rejected is gone for good by the time the import runs. + // Orca minted the submission, receipt and lifecycle identities; no provider + // transcript can hand them back. + it('rebuilds from provider history when the epoch row itself is gone', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + await journal.appendSubmission({ + clientMessageId: 'client-message-1', + payloadFingerprint: 'fingerprint-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'add a retry' }] }, + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'client-message-1', + state: 'accepted', + providerIdentity: item(1), + fence: 1 + }) + await journal.appendLifecycleBatch({ + settlementId: 'settlement-1', + fence: 1, + mutations: [ + { + kind: 'item', + identity: { provider: 'orca', clientMessageId: 'approval-1' }, + body: { kind: 'status', text: 'approved' } + } + ] + }) + await journal.close() + // Sequence 1 is the epoch row: everything behind it is valid but unanchored. + await deleteRow(1) + + const opened = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(opened.journal) + expect(opened.recovery).toMatchObject({ trigger: 'journal_corrupt' }) + expect(opened.recovery?.imported).toBeGreaterThan(0) + expect(JSON.stringify(opened.journal.snapshot().items.map((entry) => entry.body))).toContain( + 'add a retry' + ) }) it('still opens the session when provider history cannot be read', async () => { - const lines = await seedJournal(3) - const holed = lines.filter((_line, index) => index !== 2) - await writeFile(join(journalDir, 'log.jsonl'), `${holed.join('\n')}\n`, 'utf-8') + await seedJournal(3) + await deleteRow(3) const opened = await openAgentSessionJournalWithRecovery({ identity: IDENTITY, @@ -168,9 +301,170 @@ describe('openAgentSessionJournalWithRecovery', () => { fence: 1, historyFilePath: join(root, 'missing.jsonl') }) + journals.track(opened.journal) expect(opened.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) expect(opened.recovery?.error).toBeTruthy() // A missing provider transcript must not clear the intact journal prefix. - expect(opened.journal.snapshot().items).toHaveLength(1) + expect(opened.journal.snapshot().items.map((entry) => entry.body.kind)).toEqual(['message']) + }) + + // A repair that KEEPS a prefix has no emptied epoch to anchor, so nothing + // about the surviving rows records that the deleted suffix was never rebuilt. + // Unmarked, the next probe reads a contiguous anchored prefix, calls it clean, + // and the dropped stretch of timeline is gone for good. + it('keeps a partially repaired journal corrupt until provider history replaces it', async () => { + await seedJournal(3) + await deleteRow(3) + const empty = join(root, 'empty.jsonl') + await writeFile(empty, '', 'utf-8') + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: empty + }) + journals.track(first.journal) + expect(first.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) + expect(first.recovery?.error).toBeTruthy() + // Only the unanchored suffix went; the prefix the repair kept is still live. + expect(first.journal.snapshot().items.map((entry) => entry.body.kind)).toEqual(['message']) + await first.journal.close() + + // The deletion is durable, so the demand for a rebuild has to be too. + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: true }) + + // A readable transcript rebuilds the epoch, and THAT is what retires it. + const retried = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(retried.journal) + expect(retried.recovery?.imported).toBeGreaterThan(0) + await retried.journal.close() + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: false }) + }) + + // The reproduced path. Deleting sequence 1 leaves every surviving row + // unanchored, so the repair drops ALL of them — and provider history is not + // there to publish a replacement. A journal in that state used to reopen as + // clean: an append took sequence 1 as an ordinary row, replay accepted it, + // and recovery never asked the provider for the timeline again. + it('does not normalize an epoch a repair emptied while provider history was unavailable', async () => { + await seedRepairableSession() + const missing = join(root, 'missing.jsonl') + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: missing + }) + journals.track(first.journal) + expect(first.recovery?.error).toBeTruthy() + expect(first.recovery?.imported).toBe(0) + await first.journal.close() + + // Reopen: the epoch still holds nothing but the repair, so recovery runs again. + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: true }) + const reopened = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: missing + }) + journals.track(reopened.journal) + expect(reopened.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) + // Nothing but the anchor: the repair rebuilt no history of its own. + expect(reopened.journal.snapshot().items).toEqual([]) + + // The append lands ABOVE the epoch anchor, never on top of it. + await reopened.journal.appendItem( + item(2), + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'typed later' }] }, + { fence: 1 } + ) + const epoch = reopened.journal.epoch + await reopened.journal.close() + await withJournalDatabase(journalDir, (db) => { + const rows = readJournalEpochRows(db, CODEX_SESSION, epoch) + expect(JSON.parse(rows[0]?.rowJson ?? '{}')).toMatchObject({ kind: 'epoch', seq: 1 }) + }) + }) + + // An empty transcript is a plausible transient provider state, and it used to + // end recovery for good: the import published an empty replacement epoch that + // deleted the repair's anchor, the next probe called that clean, and the user's + // timeline was never rebuilt. + it('does not retire the repair marker when provider history exists but holds no messages', async () => { + await seedRepairableSession() + const empty = join(root, 'empty.jsonl') + await writeFile(empty, '', 'utf-8') + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: empty + }) + journals.track(first.journal) + expect(first.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) + expect(first.recovery?.error).toBeTruthy() + // The anchor the repair published is still the epoch; the empty import did + // not replace it with a clean one. + expect(first.journal.snapshot().items).toEqual([]) + const epoch = first.journal.epoch + await first.journal.close() + await withJournalDatabase(journalDir, (db) => { + const rows = readJournalEpochRows(db, CODEX_SESSION, epoch) + expect(JSON.parse(rows[0]?.rowJson ?? '{}')).toMatchObject({ + kind: 'epoch', + seq: 1, + reason: 'unreconcilable_prefix' + }) + }) + + // The session still reports corrupt, so the next attach retries. + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: true }) + + // And a transcript that DOES have content still rebuilds the timeline. + const retried = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(retried.journal) + expect(retried.recovery?.imported).toBeGreaterThan(0) + expect(JSON.stringify(retried.journal.snapshot().items.map((entry) => entry.body))).toContain( + 'add a retry' + ) + }) + + it('rebuilds the emptied epoch once provider history is readable again', async () => { + await seedRepairableSession() + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: join(root, 'missing.jsonl') + }) + journals.track(first.journal) + await first.journal.close() + + const retried = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(retried.journal) + expect(retried.recovery?.imported).toBeGreaterThan(0) + expect(JSON.stringify(retried.journal.snapshot().items.map((entry) => entry.body))).toContain( + 'add a retry' + ) }) }) diff --git a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts index 05f58a25afe..6e771a31809 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts @@ -14,6 +14,7 @@ import { type AgentSessionJournalIdentity, type AgentSessionProviderHandle } from '../../../shared/agent-session-journal-types' +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' import { importLegacyTranscriptIntoJournal } from '../agent-session-journal/journal-legacy-import' import { loadJournal } from '../agent-session-journal/journal-open' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' @@ -25,7 +26,8 @@ export type AgentSessionJournalRecovery = { reset: AgentJournalResetReason epoch: string imported: number - /** Set when provider history could not be read; the intact journal prefix remains live. */ + /** Set when provider history could not be read, or held nothing to restore; the + * intact journal prefix remains live. */ error?: string } @@ -55,16 +57,13 @@ export async function openAgentSessionJournalWithRecovery(input: { /** Resolve directly to a transcript instead of discovering it by session id. */ historyFilePath?: string | null }): Promise { - const probe = await loadJournal(input.journalDir, input.identity.sessionId) + const probe = loadJournal(input.journalDir, input.identity.sessionId) if (probe?.readOnly) { const journal = await openAgentSessionJournal({ identity: input.identity, journalDir: recoveryJournalDir(input.journalDir) }) - return { - journal, - recovery: await rehydrate({ ...input, journal, trigger: 'schema_unreadable' }) - } + return { journal, recovery: await rehydrateOrClose(input, journal, 'schema_unreadable') } } const journal = await openAgentSessionJournal({ identity: input.identity, @@ -73,9 +72,30 @@ export async function openAgentSessionJournalWithRecovery(input: { if (!probe?.corrupt) { return { journal, recovery: null } } - // `open()` quarantines the unusable suffix; a successful import rolls once - // more so the rebuilt timeline is the only content of its epoch. - return { journal, recovery: await rehydrate({ ...input, journal, trigger: 'journal_corrupt' }) } + // `open()` drops the unusable suffix; a successful import rolls once more so + // the rebuilt timeline is the only content of its epoch. + return { journal, recovery: await rehydrateOrClose(input, journal, 'journal_corrupt') } +} + +/** `importLegacyTranscriptIntoJournal` can THROW rather than report `ok: false` + * — a journal write failure, for instance — and nothing else holds a reference + * to the journal this function just opened. A close that rejects is retryable, + * so the journal is retained rather than dropped with its handle still open. */ +async function rehydrateOrClose( + input: { + identity: AgentSessionJournalIdentity + fence: number + historyFilePath?: string | null + }, + journal: AgentSessionJournal, + trigger: AgentSessionJournalRecovery['trigger'] +): Promise { + try { + return await rehydrate({ ...input, journal, trigger }) + } catch (error) { + await agentSessionJournalCloseRetries.closeOrRetain(journal) + throw error + } } async function rehydrate(input: { @@ -94,13 +114,16 @@ async function rehydrate(input: { fence: input.fence, ...(input.historyFilePath ? { options: { filePath: input.historyFilePath } } : {}) }) - if (!result.ok) { + // A transcript that held nothing is the same outcome as one that could not be + // read: nothing was restored, so the repair's marker has to stand and be + // retried on a later attach rather than being retired as a completed recovery. + if (!result.ok || !result.replaced) { return { trigger: input.trigger, reset, epoch: input.journal.epoch, imported: 0, - error: result.error + error: result.ok ? 'Provider history held no messages to restore' : result.error } } return { diff --git a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts index 4f3ef118af5..4f21285e417 100644 --- a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts +++ b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts @@ -1,4 +1,4 @@ -// SDKMessage discriminators from Claude Agent SDK 0.3.231 / Claude Code 2.1.231. +// SDKMessage discriminators from Claude Agent SDK 0.3.251 / Claude Code 2.1.258. export const CLAUDE_STREAM_JSON_FRAME_KINDS = [ 'message:assistant', 'message:user', @@ -42,7 +42,15 @@ export const CLAUDE_STREAM_JSON_FRAME_KINDS = [ 'message:prompt_suggestion', 'message:system:mirror_error', 'message:system:informational', - 'message:conversation_reset' + 'message:conversation_reset', + // Queue bookkeeping the CLI emits per client-supplied command uuid. Absent + // from the SDK's SDKMessage union, which is why it reached users as raw JSON. + 'message:command_lifecycle', + 'message:result:success', + 'message:result:error_during_execution', + 'message:result:error_max_turns', + 'message:result:error_max_budget_usd', + 'message:result:error_max_structured_output_retries' ] as const export type ClaudeStreamJsonFrameKind = (typeof CLAUDE_STREAM_JSON_FRAME_KINDS)[number] diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts index ad9ca66c52a..22bd645d8a6 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts @@ -65,6 +65,29 @@ describe('provider frame classification catalog', () => { ).toBe('error-surface') }) + it('keeps command queue bookkeeping off the transcript without hiding a failed one', () => { + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'started' + }) + ).toBe('status-chrome') + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'cancelled' + }) + ).toBe('status-chrome') + // Payload inspection outranks the catalogue, so suppressing the kind cannot + // swallow a state the provider reports as a failure. + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'failed' + }) + ).toBe('error-surface') + }) + it('keeps unknown future frames on the substantive bounded fallback path', () => { expect(classifyProviderFrame('codex', 'notification:future/event', {})).toBe( 'timeline-substantive' diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts index 8d11df995a6..474b1385a4f 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts @@ -131,7 +131,18 @@ export const PROVIDER_FRAME_CLASSIFICATIONS = { 'message:prompt_suggestion': 'status-chrome', 'message:system:mirror_error': 'error-surface', 'message:system:informational': 'timeline-substantive', - 'message:conversation_reset': 'status-chrome' + 'message:conversation_reset': 'status-chrome', + // A `started`/`completed`/`cancelled` state for one queued command uuid and + // nothing else; the CLI keeps it out of its own transcript too. A state that + // reads as a failure still surfaces, via the payload check in classify. + 'message:command_lifecycle': 'status-chrome', + // The turn-complete signal: lifecycle, never a transcript row. Error subtypes + // included — the turn's assistant frames already carry any user-facing text. + 'message:result:success': 'status-chrome', + 'message:result:error_during_execution': 'status-chrome', + 'message:result:error_max_turns': 'status-chrome', + 'message:result:error_max_budget_usd': 'status-chrome', + 'message:result:error_max_structured_output_retries': 'status-chrome' } } as const satisfies ProviderFrameClassificationTable diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts new file mode 100644 index 00000000000..c6566083eac --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts @@ -0,0 +1,114 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionAdapterRouter } from './structured-agent-session-adapter-router' + +function adapterOf( + releaseAcquisition: StructuredAgentSessionAdapter['releaseAcquisition'] +): StructuredAgentSessionAdapter { + return { + acquire: vi.fn(async () => ({ process: { pid: 1 } }) as never), + releaseAcquisition, + dispatch: vi.fn(), + cancelTurn: vi.fn(), + answerPrompt: vi.fn(), + setOption: vi.fn() + } as unknown as StructuredAgentSessionAdapter +} + +describe('StructuredAgentSessionAdapterRouter.releaseAcquisition', () => { + it('drops the owner even when its release reports a typed failure', async () => { + const failure = new Error('root exited') + const claude = adapterOf(vi.fn().mockRejectedValueOnce(failure).mockResolvedValue(false)) + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + + await expect(router.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBe(failure) + // With no owner left, a later release asks every adapter instead of the stale one. + await expect(router.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(false) + expect(claude.releaseAcquisition).toHaveBeenCalledTimes(2) + expect(codex.releaseAcquisition).toHaveBeenCalledTimes(1) + }) +}) + +describe('StructuredAgentSessionAdapterRouter.closeSession', () => { + it('retains the owner after an unproven close so a later retry reaches the same adapter', async () => { + const claude = adapterOf(vi.fn(async () => true)) + const closeSession = vi.fn().mockResolvedValueOnce(false).mockResolvedValueOnce(true) + const dispatch = vi.fn().mockResolvedValue({ state: 'unknown', reason: 'test' }) + claude.closeSession = closeSession + claude.dispatch = dispatch + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + + await expect(router.closeSession('session-1')).resolves.toBe(false) + await expect( + router.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: {} as never, + fence: 1 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + await expect(router.closeSession('session-1')).resolves.toBe(true) + expect(closeSession).toHaveBeenCalledTimes(2) + expect(dispatch).toHaveBeenCalledTimes(1) + }) +}) + +describe('StructuredAgentSessionAdapterRouter optional lifecycle methods', () => { + it.each([ + ['forceCloseSession', 'forceCloseSession'], + ['disposeSession', 'disposeSession'] + ] as const)( + '%s forwards to the owner and retains it until proven stopped', + async (_label, method) => { + const claude = adapterOf(vi.fn(async () => true)) + const stop = vi.fn().mockResolvedValueOnce(false).mockResolvedValueOnce(true) + claude[method] = stop + const dispatch = vi.fn().mockResolvedValue({ state: 'unknown', reason: 'test' }) + claude.dispatch = dispatch + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + const stopSession = router[method] + + await expect(stopSession('session-1')).resolves.toBe(false) + await expect( + router.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: {} as never, + fence: 1 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + await expect(stopSession('session-1')).resolves.toBe(true) + expect(stop).toHaveBeenCalledTimes(2) + expect(dispatch).toHaveBeenCalledOnce() + } + ) + + it.each(['forceCloseSession', 'disposeSession'] as const)( + 'falls back to closeSession when an owner lacks %s', + async (method) => { + const closeSession = vi.fn().mockResolvedValue(true) + const claude = adapterOf(vi.fn(async () => true)) + claude.closeSession = closeSession + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + await router.acquire({ + identity: { sessionId: 'session-1', agent: 'claude' } as never, + fence: 1, + spawnToken: 'spawn-1' + }) + const stopSession = router[method] + + await expect(stopSession('session-1')).resolves.toBe(true) + expect(closeSession).toHaveBeenCalledWith('session-1') + } + ) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts new file mode 100644 index 00000000000..6ac0c8e0fbf --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts @@ -0,0 +1,124 @@ +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import type { AgentSessionExecutionLocation } from '../../../shared/agent-session-record' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' + +type RoutedAgent = 'claude' | 'codex' + +export class StructuredAgentSessionAdapterRouter implements StructuredAgentSessionAdapter { + private readonly owners = new Map() + + constructor( + private readonly adapters: Record, + private readonly closeAdapters: () => Promise + ) {} + + supportsCreate = (location: AgentSessionExecutionLocation, agent: string): boolean => { + const adapter = this.adapterForAgent(agent) + return adapter ? (adapter.supportsLocation?.(location) ?? false) : false + } + + supportsLocation = (location: AgentSessionExecutionLocation): boolean => + Object.values(this.adapters).some((adapter) => adapter.supportsLocation?.(location) ?? false) + + async acquire(input: Parameters[0]) { + const adapter = this.requireAgent(input.identity) + const acquired = await adapter.acquire(input) + this.owners.set(input.identity.sessionId, adapter) + return acquired + } + + async releaseAcquisition(input: { sessionId: string }): Promise { + const adapter = this.owners.get(input.sessionId) + if (adapter) { + try { + return (await adapter.releaseAcquisition?.(input)) === true + } finally { + this.owners.delete(input.sessionId) + } + } + let released = false + for (const candidate of Object.values(this.adapters)) { + released = (await candidate.releaseAcquisition?.(input)) === true || released + } + return released + } + + dispatch: StructuredAgentSessionAdapter['dispatch'] = (input) => + this.owner(input.sessionId).dispatch(input) + + cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => + this.owner(input.sessionId).cancelTurn(input) + + answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => + this.owner(input.sessionId).answerPrompt(input) + + setOption: StructuredAgentSessionAdapter['setOption'] = (input) => + this.owner(input.sessionId).setOption(input) + + readOptions = (input: { sessionId: string; fence: number }) => { + const reader = this.owner(input.sessionId).readOptions + if (!reader) { + throw new Error(`structured session ${input.sessionId} does not report options`) + } + return reader(input) + } + + readOptionRestoreFailures = (sessionId: string): readonly string[] => + this.owner(sessionId).readOptionRestoreFailures?.(sessionId) ?? [] + + historyFilePath = (input: { identity: AgentSessionJournalIdentity }) => + this.requireAgent(input.identity).historyFilePath?.(input) ?? Promise.resolve(null) + + closeSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.closeSession) + + forceCloseSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.forceCloseSession ?? adapter.closeSession) + + disposeSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.disposeSession ?? adapter.closeSession) + + private async stopSession( + sessionId: string, + selectStop: ( + adapter: StructuredAgentSessionAdapter + ) => NonNullable | undefined + ): Promise { + const adapter = this.owners.get(sessionId) + if (!adapter) { + return false + } + const stop = selectStop(adapter) + const stopped = await stop?.call(adapter, sessionId) + if (stopped === true) { + this.owners.delete(sessionId) + return true + } + return false + } + + async closeAll(): Promise { + this.owners.clear() + await this.closeAdapters() + } + + private owner(sessionId: string): StructuredAgentSessionAdapter { + const adapter = this.owners.get(sessionId) + if (!adapter) { + throw new Error(`no live structured adapter owns ${sessionId}`) + } + return adapter + } + + private requireAgent(identity: AgentSessionJournalIdentity): StructuredAgentSessionAdapter { + const adapter = this.adapterForAgent(identity.agent) + if (!adapter) { + throw new Error(`structured sessions do not support ${identity.agent}`) + } + return adapter + } + + private adapterForAgent(agent: string): StructuredAgentSessionAdapter | null { + return agent === 'claude' || agent === 'codex' ? this.adapters[agent] : null + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts index 5f67240c1c6..77cce9153f5 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, rethrowAfterAgentSessionAcquisitionCleanup } from './structured-agent-session-adapter' @@ -28,6 +29,27 @@ describe('failed agent-session acquisition cleanup', () => { ).rejects.toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) }) + it('keeps a first-hand root exit that cleanup observed, with the provider diagnostic', async () => { + const cause = new Error('proof failed') + const exit = new AgentSessionAcquisitionRootExitObservedError( + new Error('claude stream-json exited (code 1): crashed') + ) + const error = await rethrowAfterAgentSessionAcquisitionCleanup( + { + releaseAcquisition: vi.fn(async () => { + throw exit + }) + }, + 'session-1', + cause + ).catch((thrown: unknown) => thrown) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): crashed') + expect((error as Error).cause).toMatchObject({ errors: [cause, exit] }) + }) + it('reports unproven exit when cleanup throws', async () => { const error = await rethrowAfterAgentSessionAcquisitionCleanup( { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts index cccc8ce6f13..01c16a60e55 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts @@ -32,6 +32,20 @@ export class AgentSessionAcquisitionRefusal extends Error { } } +/** + * The provider's own root process was observed to exit, but its descendant tree + * could not be verified. The lease keys on the root's pid and start time, so its + * observed death releases the reservation; nothing is claimed about descendants. + * Never thrown when a descendant was observed still alive — that stays unproven. + */ +export class AgentSessionAcquisitionRootExitObservedError extends Error { + constructor(cause: unknown) { + // The provider's own diagnostic is the only thing the user can act on. + super(cause instanceof Error ? cause.message : String(cause), { cause }) + this.name = 'AgentSessionAcquisitionRootExitObservedError' + } +} + export class AgentSessionAcquisitionExitUnprovenError extends Error { constructor(cause: unknown) { super('agent_session_acquisition_exit_unproven', { cause }) @@ -49,7 +63,7 @@ export type AgentSessionAcquisition = { acquisitionGeneration?: string } -/** Acquisition validation failed before the adapter attempted to spawn. */ +/** Acquisition failed with first-hand proof that no provider process existed. */ export class AgentSessionPreSpawnError extends Error { constructor(cause: unknown) { super(cause instanceof Error ? cause.message : String(cause), { cause }) @@ -105,7 +119,9 @@ export type StructuredAgentSessionAdapter = { * at — the store rejects a link minted at any other fence. */ acquire(input: StructuredAgentSessionAcquireInput): Promise /** Reaps an acquired provider when the host cannot commit or prove its lease. - * Returns true only after provider child exit is proven. */ + * Returns true only after provider child exit is proven. Throws + * `AgentSessionAcquisitionRootExitObservedError` when the provider root's own + * exit was observed first-hand but its descendants could not be verified. */ releaseAcquisition?(input: { sessionId: string }): Promise dispatch(input: { sessionId: string @@ -133,6 +149,8 @@ export type StructuredAgentSessionAdapter = { input: StructuredAgentSessionSetOptionInput ): Promise>> readOptions?(input: { sessionId: string; fence: number }): Promise + /** Option keys skipped after a provider rejected their persisted restore value. */ + readOptionRestoreFailures?(sessionId: string): readonly string[] /** Transcript path for journal recovery. Omit to let the existing session-file * resolver discover it from the provider session id. */ historyFilePath?(input: { identity: AgentSessionJournalIdentity }): Promise @@ -154,9 +172,15 @@ export async function rethrowAfterAgentSessionAcquisitionCleanup( try { released = (await adapter.releaseAcquisition?.({ sessionId })) === true } catch (cleanupError) { - throw new AgentSessionAcquisitionExitUnprovenError( - new AggregateError([cause, cleanupError], 'agent session acquisition cleanup failed') - ) + // A root exit the cleanup observed first-hand keeps its classification and its + // provider diagnostic; the failure that triggered cleanup rides along as cause. + throw cleanupError instanceof AgentSessionAcquisitionRootExitObservedError + ? new AgentSessionAcquisitionRootExitObservedError( + new AggregateError([cause, cleanupError], cleanupError.message) + ) + : new AgentSessionAcquisitionExitUnprovenError( + new AggregateError([cause, cleanupError], 'agent session acquisition cleanup failed') + ) } if (released) { throw cause diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts index 05c3d8c9e5e..7113be8d54b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts @@ -5,7 +5,6 @@ import type { AgentSessionWireRefusal } from '../../../shared/agent-session-wire' import type { AgentJournalResetReason } from '../../../shared/agent-session-journal-types' -import type { AgentSessionAttachParams } from './structured-agent-session-attach' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { StructuredAgentSessionHostDeps, @@ -30,8 +29,6 @@ export type StructuredAgentSessionAttachContext = { } tasks: StructuredAgentSessionTaskQueue reconcileLeases: (sessionId: string) => Promise - /** Retries a durable provider-exit journal settlement before a new owner is reserved. */ - retryPendingSettlement?: (sessionId: string, params: AgentSessionAttachParams) => Promise serialize: (sessionId: string, task: () => Promise) => Promise now: () => number } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts index 5be2d8ed09e..b08a56ea4d9 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts @@ -26,6 +26,7 @@ import type { AgentSessionRecordStore } from '../../runtime/agent-session-record import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, AgentSessionAcquisitionRefusal, AgentSessionPreSpawnError, isAgentSessionPreSpawnError, @@ -58,8 +59,9 @@ export type AttachFlowInput = { onAcquiring?: () => Promise | void /** Settles writes already captured by the superseded journal before opening another. */ beforeJournalOpen?: () => Promise | void - /** Removes any partial host publication after journal attachment fails. */ - onAttachFailed?: () => void + /** Removes any partial host publication after journal attachment fails, and + * closes the journal handle of the map entry it drops. Awaited: see eviction. */ + onAttachFailed?: () => Promise } export async function performAttach( @@ -118,7 +120,9 @@ export async function performAttach( ? 'processless' : error instanceof AgentSessionAcquisitionExitUnprovenError ? 'unproven' - : 'exit-proven' + : error instanceof AgentSessionAcquisitionRootExitObservedError + ? 'root-exit-observed' + : 'exit-proven' const outcome = error instanceof AgentSessionAcquisitionExitUnprovenError ? { @@ -208,15 +212,21 @@ async function settlePostAcquisitionAttachFailure( cause: unknown ): Promise { let cleanupError: unknown = cause - let exitProof: 'exit-proven' | 'unproven' = 'unproven' + let exitProof: 'exit-proven' | 'root-exit-observed' | 'unproven' = 'unproven' try { await rethrowAfterAgentSessionAcquisitionCleanup(input.adapter, record.sessionId, cause) } catch (error) { cleanupError = error exitProof = - error instanceof AgentSessionAcquisitionExitUnprovenError ? 'unproven' : 'exit-proven' + error instanceof AgentSessionAcquisitionExitUnprovenError + ? 'unproven' + : error instanceof AgentSessionAcquisitionRootExitObservedError + ? 'root-exit-observed' + : 'exit-proven' } - input.onAttachFailed?.() + // Why: the close is awaited so the map entry is gone only once its handle is + // released, but a failed close must not also cost the store settlement below. + await Promise.resolve(input.onAttachFailed?.()).catch(() => undefined) try { await input.store.settleFailedPostAcquisitionAttachment({ sessionId: record.sessionId, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts index bd79bb1bb45..a22bbdcbb3e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts @@ -17,7 +17,11 @@ import { pinnedAgentSessionLaunchEnv } from './structured-agent-session-launch-env' import { refuseAgentSessionMutation } from './structured-agent-session-mutation-admission' +import { retryPendingStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' +import type { DeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' export function attachStructuredAgentSession( context: StructuredAgentSessionAttachContext, @@ -38,14 +42,20 @@ export function attachStructuredAgentSession( return refuseAgentSessionMutation(unreconciled) } await context.runtimeState.resolveRecovery(sessionId) - if (context.retryPendingSettlement) { - const settled = await context.retryPendingSettlement(sessionId, params) - if (!settled) { - return refuseAgentSessionMutation({ - code: 'agent_session_ownership_unknown', - message: 'The provider-exit terminal journal settlement is still pending; retry attach.' - }) - } + // Retries a durable provider-exit journal settlement before a new owner is reserved. Answers + // settled when the record has none pending, so every attach can ask unconditionally. + const settled = await retryPendingStructuredAgentSessionSettlement({ + deps: context.deps, + sessions: context.sessions, + sessionId, + params, + now: () => context.now() + }) + if (!settled) { + return refuseAgentSessionMutation({ + code: 'agent_session_ownership_unknown', + message: 'The provider-exit terminal journal settlement is still pending; retry attach.' + }) } const eventSink = context.runtimeState.eventSinkFor(sessionId) const attached = await performAttach({ @@ -71,7 +81,10 @@ export function attachStructuredAgentSession( callerKey, params, now: () => context.now(), - onAttachFailed: () => { + // Site 9: this closes the PRIOR map entry it drops, never the provisional + // journal — it has no reference to that one. `onAttached` owns that. + onAttachFailed: async () => { + await context.sessions.get(sessionId)?.journal.close() context.sessions.delete(sessionId) eventSink.close() context.runtimeState.discardEventSink(sessionId) @@ -80,14 +93,27 @@ export function attachStructuredAgentSession( const fence = context.deps.store.getRecord(sessionId)?.lease.runtimeFence ?? 0 const previous = context.sessions.get(sessionId) const previousFence = previous?.fence - eventSink.bind({ - journal: attached.journal, - fence, - publish: () => context.subscribers.publish(sessionId, attached.journal) - }) - const barrier = await eventSink.drained() - if (!barrier.ok) { - throw barrier.error + // Site 8: the provisional journal has no owner until the map takes it, + // and the barrier below throws by design. + try { + await bindAndDrain(eventSink, attached.journal, fence, () => + context.subscribers.publish(sessionId, attached.journal) + ) + } catch (error) { + await agentSessionJournalCloseRetries.closeOrRetain(attached.journal) + throw error + } + // Site 10: a `set` over a live entry would orphan its handle — and a + // close that REJECTED did not release it. The replacement is therefore + // ABORTED rather than completed over a handle nothing can reach again: + // `previous` stays indexed, so teardown still owns it and can retry. + if (previous && previous.journal !== attached.journal) { + try { + await previous.journal.close() + } catch (error) { + await agentSessionJournalCloseRetries.closeOrRetain(attached.journal) + throw error + } } context.sessions.set(sessionId, { journal: attached.journal, @@ -115,3 +141,18 @@ export function attachStructuredAgentSession( }) return context.tasks.trackAttach(attaching) } + +/** Binds the sink to the journal and waits for the barrier the host publishes + * behind. It throws by design when a sink barrier fails. */ +async function bindAndDrain( + eventSink: DeferredStructuredAgentSessionEventSink, + journal: AgentSessionJournal, + fence: number, + publish: () => void +): Promise { + eventSink.bind({ journal, fence, publish }) + const barrier = await eventSink.drained() + if (!barrier.ok) { + throw barrier.error + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts index cfdbf786e14..ce58e31b4ee 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts @@ -5,7 +5,6 @@ // the record store's compare-and-swap, which also owns the idempotency row, so // a retried attach replays instead of reserving a second owner. -import type { AgentType } from '../../../shared/agent-status-types' import type { AgentSessionJournalIdentity, AgentSessionProviderHandle @@ -32,6 +31,7 @@ import { } from '../../../shared/agent-session-mutation-envelope' import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { agentSessionProviderHandleChainHead } from '../../../shared/agent-session-provider-handle' +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' import { journalDirectoryFor } from '../agent-session-journal/journal-paths' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import { @@ -51,7 +51,7 @@ export type AgentSessionAttachParams = { envelope: AgentSessionMutationEnvelope location: AgentSessionExecutionLocation provider: AgentSessionHandleProvider - agent: AgentType + agent: AgentSessionHandleProvider accountHome: AgentSessionAccountHome runtimeKind: AgentSessionOwnerRuntimeKind /** Omitted only for create-by-intent; the adapter proves the durable handle. */ @@ -162,9 +162,18 @@ export async function attachJournal(input: { fence, historyFilePath }) - return { - ...opened, - unconfirmedClientMessageIds: await opened.journal.markPendingSubmissionsUnknown(fence) + try { + // That await is a WRITE. A failure in it leaves the journal with no caller + // holding a reference to close it. + return { + ...opened, + unconfirmedClientMessageIds: await opened.journal.markPendingSubmissionsUnknown(fence) + } + } catch (error) { + // A rejected close leaves the handle open, so the journal is retained for a + // later retry rather than dropped along with the only reference to it. + await agentSessionJournalCloseRetries.closeOrRetain(opened.journal) + throw error } } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts new file mode 100644 index 00000000000..87bc33bc4b9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts @@ -0,0 +1,182 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffRequest, + AgentSessionMutationEnvelope +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +const CALLER = { callerKey: 'client-claude' } +const CLAUDE_SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' +const DEFAULT_MODEL = 'sonnet' +const PICKED_MODEL = 'opus' + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let activeModel: string +let transcriptPath: string + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function handoff(direction: AgentSessionHandoffDirection): AgentSessionHandoffRequest { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { envelope: envelope('agentSession.requestHandoff', fields), ...fields } +} + +function owner(fence: number, spawnToken: string): StructuredTuiOwner { + return { + terminal: { + handle: 'term-claude', + tabId: 'tab-claude', + paneKey: 'pane-claude', + ptyId: 'pty-claude' + }, + process: { hostId: 'local', pid: 5200, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `claude-tui-${fence}`, + handle: { provider: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'tui-leaf' }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + transcriptPath + } +} + +function transport(): StructuredAgentSessionHandoffTransport { + return { + hostLabel: 'Test host', + launchTui: async ({ fence, spawnToken }) => owner(fence, spawnToken), + reproveTuiOwner: async ({ owner: current }) => current, + recoverTuiOwner: async (record) => + owner( + record.lease.runtimeFence, + record.lease.ownerProcess?.spawnToken ?? record.lease.reservedSpawnToken ?? 'recovered' + ), + stopRecoveredOwner: async () => undefined, + waitForTuiExit: async (current) => ({ transcriptPath: current.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle' + } +} + +function adapter(): StructuredAgentSessionAdapter { + acquire = vi.fn(async ({ fence, spawnToken, options }) => { + activeModel = options?.model ?? DEFAULT_MODEL + return { + process: { hostId: 'local', pid: 4200, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `claude-native-${fence}`, + handle: { provider: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'native-leaf' }, + origin: acquire.mock.calls.length === 1 ? 'created' : 'resumed', + mintedAtFence: fence, + observedAt: NOW + } + } + }) + return { + acquire, + dispatch: vi.fn(), + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt: vi.fn(async () => undefined), + setOption: vi.fn(async ({ value }) => { + activeModel = value + return { model: value } + }), + readOptions: vi.fn(async () => ({ current: { model: activeModel }, models: [] })), + closeSession: vi.fn(async () => { + activeModel = DEFAULT_MODEL + return true + }) + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-claude-handoff-options-')) + resetHostTestOperationIds() + activeModel = DEFAULT_MODEL + transcriptPath = join(root, 'claude.jsonl') + await writeFile(transcriptPath, '', 'utf8') + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-claude', + handoffTransport: transport(), + now: () => NOW + }) + expect( + await host.attach( + CALLER, + hostTestAttachParams(null, { + provider: 'claude', + agent: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: join(root, 'claude-home') }, + providerHandle: { kind: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'native-leaf' } + }) + ) + ).toMatchObject({ ok: true }) +}) + +afterEach(async () => { + await new Promise((resolve) => setTimeout(resolve, 100)) + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true, maxRetries: 3, retryDelay: 50 }) +}) + +describe('Claude structured session handoff options', () => { + it('keeps a directly selected model through chat to TUI to chat', async () => { + const fields = { key: 'model', value: PICKED_MODEL } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', fields), + ...fields + }) + ).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } }) + + expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }) + ) + expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }) + ) + + expect(acquire.mock.calls[1]?.[0].options).toEqual({ model: PICKED_MODEL }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + expect(activeModel).toBe(PICKED_MODEL) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts new file mode 100644 index 00000000000..a4666b4f045 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts @@ -0,0 +1,257 @@ +// A close that REJECTED did not release the handle. +// +// `AgentSessionJournal.close()` is retryable by design: the release step is +// unguarded precisely so a second call is a second attempt. Callers that did +// `close().catch(() => undefined)` and then threw or overwrote their map entry +// turned that retryable failure into a permanent orphan — on POSIX a silent +// leak, on Windows a handle that blocks renaming or removing the directory. +// +// These drive the REAL callers: the attach orchestration's `onAttached`, and +// host teardown, which is what runtime stop calls. Only the lease/record +// machinery around them is stubbed. + +import { access, mkdtemp, rename, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import { + agentSessionJournalCloseRetries, + JournalCloseRetryRegistry +} from '../agent-session-journal/journal-close-retry' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { attachStructuredAgentSession } from './structured-agent-session-attach-orchestration' +import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' +import { tearDownStructuredAgentSessionHost } from './structured-agent-session-host-teardown' +import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' + +const attachFlow = vi.hoisted(() => ({ + journal: null as AgentSessionJournal | null +})) + +// The lease reservation, the record store and the provider child are not what +// these cases are about; `onAttached` is, and it is the real one. +vi.mock('./structured-agent-session-attach-flow', () => ({ + performAttach: async (input: { + onAttached: ( + attached: { journal: AgentSessionJournal; recovery: null }, + generation: string | null + ) => Promise + }) => { + await input.onAttached({ journal: attachFlow.journal!, recovery: null }, null) + return { ok: true, value: {} } + } +})) + +const SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: SESSION, + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: SESSION } +} + +let root: string +const journals = createTrackedJournalOpener() + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +async function expectNothingHoldsTheDirectory(directory: string): Promise { + const dbPath = journalDatabaseFile(directory) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + // The half that actually fails on Windows when a handle is still open. + const moved = `${directory}-moved` + await rename(directory, moved) + await rm(moved, { recursive: true }) +} + +function hostSession(journal: AgentSessionJournal): StructuredAgentSessionHostSession { + return { + journal, + params: {} as StructuredAgentSessionHostSession['params'], + fence: 1, + hasProviderChild: false, + acquisitionGeneration: null + } +} + +/** A journal whose close rejects until `failures` is exhausted, wrapping a real + * store so the handle it holds is a real one. */ +function flakyClose(journal: AgentSessionJournal, failures: number): AgentSessionJournal { + let remaining = failures + return new Proxy(journal, { + get(target, property, receiver) { + if (property !== 'close') { + return Reflect.get(target, property, receiver) + } + return async () => { + if (remaining > 0) { + remaining -= 1 + throw new Error('close rejected') + } + await target.close() + } + } + }) +} + +function attachContext( + sessions: Map +): StructuredAgentSessionAttachContext { + const eventSink = { + sink: {}, + drained: async () => ({ ok: true }) as const, + unbind: () => undefined, + bind: () => undefined, + close: () => undefined + } + return { + deps: { store: { getRecord: () => null }, claimKeyId: 'key-1', journalRoot: root }, + runtimeState: { + resolveRecovery: async () => undefined, + eventSinkFor: () => eventSink, + probeOwner: async () => ({ outcome: 'pid-absent' }), + discardEventSink: () => undefined + }, + sessions, + subscribers: { + reset: () => undefined, + snapshot: () => undefined, + publish: () => undefined + }, + tasks: { trackAttach: (task: Promise) => task }, + reconcileLeases: async () => null, + serialize: (_sessionId: string, task: () => Promise) => task(), + now: () => 1 + } as unknown as StructuredAgentSessionAttachContext +} + +const attachParams = { + envelope: { sessionId: SESSION, clientOperationId: 'op-1' } +} as unknown as Parameters[2] + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-close-retry-')) + // The registry is process-wide; drain it so one case cannot see another's. + await agentSessionJournalCloseRetries.retryAll() +}) + +afterEach(async () => { + await agentSessionJournalCloseRetries.retryAll() + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('the registry', () => { + it('retains a journal whose close rejected and releases it on the retry', async () => { + const directory = join(root, 'retained') + const registry = new JournalCloseRetryRegistry() + const journal = flakyClose( + await journals.open({ identity: IDENTITY, journalDir: directory }), + 1 + ) + + const first = await registry.closeOrRetain(journal) + expect(first.closed).toBe(false) + expect(registry.pendingDirectories).toEqual([directory]) + + expect(await registry.retryAll()).toEqual([]) + expect(registry.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(directory) + }) +}) + +describe('the attach orchestration', () => { + it('ABORTS the map replacement when the previous journal will not close', async () => { + const previousDir = join(root, 'previous') + const provisionalDir = join(root, 'provisional') + const previous = flakyClose( + await journals.open({ identity: IDENTITY, journalDir: previousDir }), + 1 + ) + const provisional = await journals.open({ + identity: IDENTITY, + journalDir: provisionalDir + }) + attachFlow.journal = provisional + const sessions = new Map([[SESSION, hostSession(previous)]]) + + await expect( + attachStructuredAgentSession(attachContext(sessions), 'caller-1', attachParams) + ).rejects.toThrow('close rejected') + + // The live entry is UNTOUCHED: overwriting it would have left its handle + // open with nothing able to reach it again. + expect(sessions.get(SESSION)?.journal).toBe(previous) + // And the provisional journal is owned by the registry, not orphaned. + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(provisionalDir) + }) + + it('retains the provisional journal when its own close rejects on the barrier path', async () => { + const provisionalDir = join(root, 'provisional-barrier') + const provisional = flakyClose( + await journals.open({ identity: IDENTITY, journalDir: provisionalDir }), + 1 + ) + attachFlow.journal = provisional + const sessions = new Map() + const context = attachContext(sessions) + const failing = { + sink: {}, + drained: async () => ({ ok: false, error: new Error('sink barrier failed') }) as const, + unbind: () => undefined, + bind: () => undefined, + close: () => undefined + } + context.runtimeState.eventSinkFor = (() => + failing) as unknown as typeof context.runtimeState.eventSinkFor + + await expect(attachStructuredAgentSession(context, 'caller-1', attachParams)).rejects.toThrow( + 'sink barrier failed' + ) + + expect(sessions.size).toBe(0) + // Retained rather than dropped, so teardown can still release the handle. + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([provisionalDir]) + }) +}) + +describe('teardown, which is what runtime stop calls', () => { + it('retries the journals earlier failure paths could not close', async () => { + const orphanDir = join(root, 'orphan') + const orphan = flakyClose(await journals.open({ identity: IDENTITY, journalDir: orphanDir }), 1) + expect((await agentSessionJournalCloseRetries.closeOrRetain(orphan)).closed).toBe(false) + + // The first teardown reports the still-failing close instead of hiding it. + await tearDownStructuredAgentSessionHost({ phases: [], sessions: new Map() }) + + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(orphanDir) + }) + + it('surfaces a retained close that still rejects, and keeps it for the next stop', async () => { + const orphanDir = join(root, 'stubborn') + const orphan = flakyClose(await journals.open({ identity: IDENTITY, journalDir: orphanDir }), 2) + await agentSessionJournalCloseRetries.closeOrRetain(orphan) + + await expect( + tearDownStructuredAgentSessionHost({ phases: [], sessions: new Map() }) + ).rejects.toMatchObject({ errors: [expect.objectContaining({ message: 'close rejected' })] }) + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([orphanDir]) + + // A later stop is a real retry, not a no-op. + await tearDownStructuredAgentSessionHost({ phases: [], sessions: new Map() }) + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(orphanDir) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts index 4aed742f51d..e16f6a63c9d 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts @@ -2,16 +2,10 @@ import type { AgentJournalItemBody, AgentJournalItemIdentity } from '../../../shared/agent-session-journal-types' -import type { StructuredAgentSessionJournalBlob } from './structured-agent-session-event-sink' export function estimateStructuredAgentSessionItemBytes( identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly StructuredAgentSessionJournalBlob[] + body: AgentJournalItemBody ): number { - return ( - Buffer.byteLength(JSON.stringify({ identity, body }), 'utf8') + - blobs.reduce((total, blob) => total + Buffer.byteLength(blob.payload, 'utf8'), 0) + - 512 - ) + return Buffer.byteLength(JSON.stringify({ identity, body }), 'utf8') + 512 } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts index 6befa3b0b62..c97161ce3dd 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts @@ -287,13 +287,13 @@ describe('deferred structured agent-session event sink', () => { releaseSecond?.() }) - it('replaces a queued same-item checkpoint before any blob is created', async () => { + it('replaces a queued same-item checkpoint before it runs', async () => { const log: Recorded[] = [] const deferred = createDeferredStructuredAgentSessionEventSink() const options = { coalescingKey: 'checkpoint:item-1' } - deferred.sink.appendItem(identity(0), BODY, [], options) - deferred.sink.appendItem(identity(1), BODY, [], options) + deferred.sink.appendItem(identity(0), BODY, options) + deferred.sink.appendItem(identity(1), BODY, options) expect(deferred.state().queuedOperations).toBe(1) deferred.bind(target(6, log)) @@ -306,9 +306,9 @@ describe('deferred structured agent-session event sink', () => { const deferred = createDeferredStructuredAgentSessionEventSink() const options = { coalescingKey: 'checkpoint:item-1' } - deferred.sink.appendItem(identity(0), BODY, [], options) + deferred.sink.appendItem(identity(0), BODY, options) deferred.sink.appendItem(identity(1), BODY) - deferred.sink.appendItem(identity(2), BODY, [], options) + deferred.sink.appendItem(identity(2), BODY, options) deferred.bind(target(6, log)) await deferred.drained() diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts index e0d91f93b71..7e0192f179c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts @@ -8,8 +8,6 @@ import type { JournalLifecycleMutationInput } from '../agent-session-journal/jou import { estimateStructuredAgentSessionItemBytes } from './structured-agent-session-event-sink-estimate' import { StructuredAgentSessionSinkQueue } from './structured-agent-session-event-sink-queue' -export type StructuredAgentSessionJournalBlob = { digest: string; payload: string } - export type StructuredAgentSessionSinkAdmission = | { accepted: true } | { accepted: false; reason: 'backpressure' | 'failed' | 'closed' } @@ -24,7 +22,7 @@ export type StructuredAgentSessionSinkState = { export type StructuredAgentSessionSinkBarrier = { ok: true } | { ok: false; error: unknown } export type StructuredAgentSessionAppendOptions = { - /** Pending checkpoints with this key replace one another before blob writes. */ + /** Pending checkpoints with this key replace one another before they run. */ coalescingKey?: string /** Marks a critical lifecycle operation for lifecycle barriers and diagnostics. */ lifecycle?: boolean @@ -34,7 +32,6 @@ export type StructuredAgentSessionEventSink = { appendItem( identity: AgentJournalItemIdentity, body: AgentJournalItemBody, - blobs?: readonly StructuredAgentSessionJournalBlob[], options?: StructuredAgentSessionAppendOptions ): void appendTombstone( @@ -49,7 +46,6 @@ export type StructuredAgentSessionEventSink = { tryAppendItem?( identity: AgentJournalItemIdentity, body: AgentJournalItemBody, - blobs?: readonly StructuredAgentSessionJournalBlob[], options?: StructuredAgentSessionAppendOptions ): StructuredAgentSessionSinkAdmission appendLifecycleBatch?( @@ -159,32 +155,22 @@ export function createDeferredStructuredAgentSessionEventSink( return { sink: { - appendItem: (identity, body, blobs = [], options = {}) => { + appendItem: (identity, body, options = {}) => { queue.submit( { - bytes: estimateStructuredAgentSessionItemBytes(identity, body, blobs), + bytes: estimateStructuredAgentSessionItemBytes(identity, body), coalescingKey: options.coalescingKey, - run: (bound) => - blobs.length > 0 && typeof bound.journal.appendItemWithBlobs === 'function' - ? bound.journal.appendItemWithBlobs(identity, body, blobs, { - fence: bound.fence - }) - : bound.journal.appendItem(identity, body, { fence: bound.fence }) + run: (bound) => bound.journal.appendItem(identity, body, { fence: bound.fence }) }, options ) }, - tryAppendItem: (identity, body, blobs = [], options = {}) => + tryAppendItem: (identity, body, options = {}) => queue.submit( { - bytes: estimateStructuredAgentSessionItemBytes(identity, body, blobs), + bytes: estimateStructuredAgentSessionItemBytes(identity, body), coalescingKey: options.coalescingKey, - run: (bound) => - blobs.length > 0 && typeof bound.journal.appendItemWithBlobs === 'function' - ? bound.journal.appendItemWithBlobs(identity, body, blobs, { - fence: bound.fence - }) - : bound.journal.appendItem(identity, body, { fence: bound.fence }) + run: (bound) => bound.journal.appendItem(identity, body, { fence: bound.fence }) }, options ), diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts index d1430ac237c..0d2c693c75f 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts @@ -26,7 +26,9 @@ function context(): StructuredAgentSessionEvictionContext & { order: string[] } return true }) } as unknown as StructuredAgentSessionEvictionContext['adapter'], - forget: vi.fn(() => order.push('forget')), + forget: vi.fn(async () => { + order.push('forget') + }), discardSink: vi.fn(() => order.push('discardSink')), releaseLease: vi.fn(async () => { order.push('releaseLease') @@ -122,7 +124,7 @@ describe('rows the provider emits while closing', () => { return true } } as never, - forget: () => {}, + forget: async () => {}, discardSink: () => state.discardEventSink(sessionId), releaseLease: async () => {} }) @@ -171,7 +173,7 @@ describe('eviction against the real sink cache', () => { sessionId, eventSink: state.eventSinkFor(sessionId), adapter: { closeSession: async () => true } as never, - forget: () => {}, + forget: async () => {}, discardSink: () => state.discardEventSink(sessionId), releaseLease: async () => {} }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts index 5d1bcaf180c..c2591bba567 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts @@ -25,7 +25,10 @@ export type StructuredAgentSessionEvictionContext = { hasProviderChild?: boolean eventSink: DeferredStructuredAgentSessionEventSink adapter: StructuredAgentSessionAdapter - forget: () => void + /** Closes the session's journal handle and drops the map entry. Async and + * awaited: `close()` is ordered behind queued writes, and a delete that + * returns while the close is still queued leaves nothing to retry. */ + forget: () => Promise /** Drops the cached sink so a later attach mints a fresh one. */ discardSink: () => void /** Hands the lease back now that this host's child is proven gone. No-ops when the record is diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts new file mode 100644 index 00000000000..6082ab074f5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts @@ -0,0 +1,166 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionMutationEnvelope } from '../../../shared/agent-session-wire' +import { encodeAgentSessionQuestionAnswers } from '../../../shared/agent-session-question-answer' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { journalDirectoryFor } from '../agent-session-journal/journal-paths' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import type { + AgentSessionDispatchOutcome, + StructuredAgentSessionAdapter +} from './structured-agent-session-adapter' +import type { AgentSessionAttachParams } from './structured-agent-session-attach' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' + +const CALLER = { callerKey: 'client-1' } + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +const attachParams = (): AgentSessionAttachParams => hostTestAttachParams(null) + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let answerPrompt: Mock +let ordinal = 0 + +function adapter(): StructuredAgentSessionAdapter { + const dispatch = vi.fn(async (): Promise => { + ordinal += 1 + return { + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal } + } + }) + return { + acquire, + releaseAcquisition: vi.fn(async () => true), + dispatch, + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt, + setOption: vi.fn(async () => undefined) + } +} + +async function seedGroupedQuestion(): Promise<{ itemId: string; revision: number }> { + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: journalDirectoryFor(root, { workspaceId: 'workspace-1', sessionId: SESSION }) + }) + const appended = await journal.appendItem( + { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 100 }, + { + kind: 'question', + question: '2 grouped questions from Claude', + options: [], + questions: [ + { + id: 'q1', + question: 'Targets', + multiSelect: true, + options: [ + { id: 'target-web', label: 'Web' }, + { id: 'target-mobile', label: 'Mobile' } + ] + }, + { + id: 'q2', + question: 'Host', + multiSelect: false, + options: [], + freeTextQuestionId: 'q2' + } + ], + resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } + }, + { fence: 1 } + ) + return { itemId: appended.itemId, revision: appended.revision } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-wire-grouped-')) + resetHostTestOperationIds() + ordinal = 0 + acquire = vi.fn(async ({ fence }) => ({ + process: { + hostId: 'local', + pid: 4242, + processStartTimeMs: 1_700_000_000_000, + spawnToken: store.getRecord(SESSION)?.lease.reservedSpawnToken ?? 'spawn-a' + }, + link: { + linkId: `link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: store.getRecord(SESSION)?.providerHandleChain.length ? 'resumed' : 'created', + mintedAtFence: fence, + observedAt: NOW + } + })) + answerPrompt = vi.fn(async () => undefined) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-a', + now: () => NOW + }) +}) + +afterEach(async () => { + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +describe('grouped question admission', () => { + it('admits renderer question-group payloads with child ids and multi-select answers', async () => { + const prompt = await seedGroupedQuestion() + const attached = await host.attach(CALLER, attachParams()) + expect(attached.ok).toBe(true) + const optionId = encodeAgentSessionQuestionAnswers([ + { questionId: 'q1', optionIds: ['target-web', 'target-mobile'] }, + { questionId: 'q2', optionIds: [], other: 'SSH host' } + ]) + const fields = { itemId: prompt.itemId, expectedRevision: prompt.revision, optionId } + const result = await host.respondToPrompt(CALLER, { + envelope: envelope('agentSession.respondTo:question', fields), + kind: 'question', + ...fields + }) + expect(result).toMatchObject({ ok: true, value: { resolution: { state: 'resolved' } } }) + expect(answerPrompt).toHaveBeenCalledWith( + expect.objectContaining({ itemId: prompt.itemId, optionId }) + ) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts new file mode 100644 index 00000000000..b881d55e771 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts @@ -0,0 +1,138 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionOperationOutcome } from '../../../shared/agent-session-operation-ledger' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffResult, + AgentSessionHandoffStatus, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import { AGENT_SESSION_WIRE_REFUSAL_CODES } from '../../../shared/agent-session-wire' +import { + agentSessionFingerprintConflict, + computeAgentSessionPayloadFingerprint +} from '../../../shared/agent-session-mutation-envelope' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export type StructuredHandoffAdmission = + | { decision: 'continue'; record: AgentSessionRecord; fingerprint: string } + | { decision: 'replay'; outcome: AgentSessionOperationOutcome } + | { decision: 'refused'; refusal: AgentSessionWireRefusal } + +export async function admitStructuredHandoffRequest(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + record: AgentSessionRecord + status?: AgentSessionHandoffStatus +}): Promise { + const action = input.params.action ?? 'start' + const requestFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: input.record.sessionId, + fields: { direction: input.params.direction, mode: input.params.mode, action } + }) + const conflict = agentSessionFingerprintConflict(input.params.envelope, requestFingerprint) + if (conflict) { + return { decision: 'refused', refusal: conflict } + } + const fingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff.operation', + sessionId: input.record.sessionId, + fields: { direction: input.params.direction } + }) + const operation = await input.operationGuard.check({ + callerKey: input.callerKey, + sessionId: input.record.sessionId, + operationId: input.params.envelope.clientOperationId, + fingerprint, + action, + ...(input.status ? { status: input.status } : {}), + now: input.deps.now() + }) + if (operation.decision === 'replay') { + return { decision: 'replay', outcome: operation.outcome } + } + if (operation.decision === 'refused') { + return { + decision: 'refused', + refusal: { + code: operation.code as 'agent_session_operation_conflict', + message: 'This handoff operation could not be admitted.' + } + } + } + if (input.params.envelope.expectedRuntimeFence !== input.record.lease.runtimeFence) { + await input.deps.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_checkpoint_stale' } + }) + input.operationGuard.finish(input.record.sessionId, input.params.envelope.clientOperationId) + return { + decision: 'refused', + refusal: { + code: 'agent_session_checkpoint_stale', + message: 'The session owner changed before the handoff request arrived.', + currentFence: input.record.lease.runtimeFence + } + } + } + return { decision: 'continue', record: input.record, fingerprint } +} + +export function replayedStructuredHandoffRefusal( + outcome: AgentSessionOperationOutcome +): AgentSessionWireRefusal | null { + if ( + outcome.status !== 'failed' || + !AGENT_SESSION_WIRE_REFUSAL_CODES.includes( + outcome.code as (typeof AGENT_SESSION_WIRE_REFUSAL_CODES)[number] + ) + ) { + return null + } + return { + code: outcome.code as (typeof AGENT_SESSION_WIRE_REFUSAL_CODES)[number], + message: 'This handoff request was previously refused.' + } +} + +export async function refuseAdmittedStructuredHandoff(input: { + deps: StructuredAgentSessionHandoffDeps + callerKey: string + params: AgentSessionHandoffRequest + refusal: AgentSessionWireRefusal +}): Promise> { + await input.deps.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.params.envelope.clientOperationId, + outcome: { status: 'failed', code: input.refusal.code } + }) + return { ok: false, refusal: input.refusal } +} + +export function structuredHandoffRetryIsAdmissible( + status: AgentSessionHandoffStatus, + params: AgentSessionHandoffRequest +): boolean { + return ( + status.phase === 'failed' && + status.direction === params.direction && + status.operationId === params.envelope.clientOperationId && + status.error?.recoverableOwner !== 'none' + ) +} + +export function structuredHandoffRetryResumesStoppedOwner( + record: AgentSessionRecord, + params: AgentSessionHandoffRequest +): boolean { + return ( + record.lease.claimStatus === 'released' && + record.lease.handoffStage === 'old-owner-stopped' && + record.lease.handoffOperationId === params.envelope.clientOperationId + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts new file mode 100644 index 00000000000..7f2bec98962 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts @@ -0,0 +1,98 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { StructuredAgentSessionHandoffFlowRunner } from './structured-agent-session-handoff-flow-runner' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-flow-runner-outcome-write-failure' +const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' +const OPERATION = `${NOW}-00000000000000000000000000000002` +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('structured handoff flow runner outcome-write failure', () => { + it('still reports the flow failure when the failed-outcome ledger write throws', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-handoff-flow-runner-')) + roots.push(root) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + // Materialize the store file so its later disappearance reads as corruption, + // making every subsequent ledger write reject. + await store.admitOperation({ + callerKey: 'seed', + operationId: `${NOW}-00000000000000000000000000000009`, + fingerprint: 'seed', + now: NOW + }) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: join(root, 'journal') + }) + await rm(join(root, 'store'), { recursive: true, force: true }) + const failures: unknown[] = [] + const fields = { + direction: 'to-native' as const, + mode: 'now' as const, + action: 'retry' as const + } + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: SESSION, + fields + }) + }, + ...fields + } + const runner = new StructuredAgentSessionHandoffFlowRunner({ + deps: { + store, + claimKeyId: 'key-1', + session: () => ({ journal, fence: 1 }), + suspendNative: async () => ({ state: 'stopped' as const }), + acquireNative: async () => { + throw new Error('unused') + }, + importTuiHistory: async () => {}, + publish: () => {}, + schedule: async () => { + throw new Error('scheduling failed') + }, + now: () => NOW + }, + operationGuard: new StructuredAgentSessionHandoffOperationGuard(store), + flowContext: (): StructuredAgentSessionHandoffFlowContext => { + throw new Error('unreachable: scheduling rejects before the flow needs context') + }, + fail: (_params, error) => { + failures.push(error) + } + }) + runner.begin({ callerKey: 'client-1', params, turnId: null, fingerprint: 'fp' }) + await runner.drain() + expect(failures).toHaveLength(1) + expect((failures[0] as Error).message).toBe('scheduling failed') + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts new file mode 100644 index 00000000000..7278502ce1f --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts @@ -0,0 +1,110 @@ +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { stopStructuredNativeTurn } from './structured-agent-session-handoff-flow-context' +import { handoffStructuredSessionToTui } from './structured-agent-session-handoff-forward' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { assertScheduledStructuredHandoffIsAdmissible } from './structured-agent-session-handoff-revalidation' +import { handoffStructuredSessionToNative } from './structured-agent-session-handoff-reverse' +import { structuredTuiStatus } from './structured-agent-session-handoff-status' +import type { + StructuredAgentSessionHandoffDeps, + StructuredAgentSessionHandoffFlowContext +} from './structured-agent-session-handoff-types' + +export class StructuredAgentSessionHandoffFlowRunner { + private readonly active = new Set>() + + constructor( + private readonly input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + flowContext: () => StructuredAgentSessionHandoffFlowContext + fail: (params: AgentSessionHandoffRequest, error: unknown) => void + } + ) {} + + async drain(): Promise { + await Promise.allSettled(this.active) + } + + track(task: Promise): void { + this.active.add(task) + void task.finally(() => this.active.delete(task)) + } + + begin(input: { + callerKey: string + params: AgentSessionHandoffRequest + turnId: string | null + fingerprint: string + tuiAlreadyExited?: boolean + }): void { + const { callerKey, params, turnId, fingerprint, tuiAlreadyExited = false } = input + const sessionId = params.envelope.sessionId + const journalSequence = this.input.deps.session(sessionId).journal.cursor().sequence + this.input.operationGuard.start(sessionId, { + callerKey, + operationId: params.envelope.clientOperationId, + fingerprint + }) + const flow = this.run(params, turnId, tuiAlreadyExited, journalSequence) + .then(() => { + this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId) + return this.input.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId } + }) + }) + .catch(async (error) => { + try { + await this.input.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_handoff_failed' } + }) + } catch { + // Best-effort: a store write failure must not suppress the client's failure + // notification or leak the flow as an unhandled rejection. + } + this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId) + this.input.fail(params, error) + }) + .finally(() => this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId)) + this.track(flow) + } + + private run( + params: AgentSessionHandoffRequest, + turnId: string | null, + tuiAlreadyExited: boolean, + journalSequence: number + ): Promise { + const sessionId = params.envelope.sessionId + return this.input.deps.schedule(sessionId, async () => { + const context = this.input.flowContext() + assertScheduledStructuredHandoffIsAdmissible({ + record: context.requireRecord(sessionId), + journal: this.input.deps.session(sessionId).journal, + params, + turnId, + journalSequence, + tuiAlreadyExited, + tuiStatus: structuredTuiStatus(context.owner(sessionId), this.input.deps.transport) + }) + if (turnId && params.mode === 'stop-turn') { + const stopped = await stopStructuredNativeTurn(this.input.deps, sessionId, turnId) + if (!stopped) { + throw new Error('The current turn did not acknowledge cancellation.') + } + } + await (params.direction === 'to-tui' + ? handoffStructuredSessionToTui(context, params, params.action === 'retry') + : handoffStructuredSessionToNative( + context, + params, + params.action === 'retry', + tuiAlreadyExited + )) + }) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts new file mode 100644 index 00000000000..7c77806ad91 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts @@ -0,0 +1,221 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + agentSessionLeaseFixture, + agentSessionRecordFixture +} from '../../../shared/agent-session-record.test-fixture' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { + queuedStructuredHandoffCanBegin, + StructuredAgentSessionHandoffQueue +} from './structured-agent-session-handoff-queue' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { assertScheduledStructuredHandoffIsAdmissible } from './structured-agent-session-handoff-revalidation' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-alpha-1' +const OPERATION_A = `${NOW}-00000000000000000000000000000001` +const OPERATION_B = `${NOW}-00000000000000000000000000000002` + +let root: string | null = null + +afterEach(async () => { + if (root) { + await rm(root, { recursive: true, force: true }) + root = null + } +}) + +async function createGuard() { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-operation-guard-')) + const store = await AgentSessionRecordStore.open({ directory: root, hostId: 'local' }) + return { guard: new StructuredAgentSessionHandoffOperationGuard(store), store } +} + +function status(phase: 'switching' | 'queued' | 'idle'): AgentSessionHandoffStatus { + return { + owner: phase === 'idle' ? 'native' : 'none', + direction: phase === 'idle' ? null : 'to-tui', + phase, + stage: phase === 'switching' ? 'preparing' : null, + operationId: phase === 'idle' ? null : OPERATION_A + } +} + +describe('structured handoff operation ownership', () => { + it('reserves one winner across concurrent admissions', async () => { + const { guard } = await createGuard() + const check = (operationId: string) => + guard.check({ + callerKey: operationId, + sessionId: SESSION, + operationId, + fingerprint: operationId, + action: 'start', + now: NOW + }) + + const decisions = await Promise.all([check(OPERATION_A), check(OPERATION_B)]) + + expect(decisions.map(({ decision }) => decision).sort()).toEqual(['new', 'refused']) + }) + + it.each(['switching', 'queued'] as const)( + 'durably refuses a distinct operation while the %s operation owns the session', + async (phase) => { + const { guard } = await createGuard() + guard.start(SESSION, { + callerKey: 'client-a', + operationId: OPERATION_A, + fingerprint: 'fingerprint-a' + }) + + expect( + await guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'start', + status: status(phase), + now: NOW + }) + ).toEqual({ decision: 'refused', code: 'agent_session_operation_conflict' }) + + guard.finish(SESSION, OPERATION_A) + expect( + await guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'start', + status: status('idle'), + now: NOW + }) + ).toMatchObject({ + decision: 'replay', + outcome: { status: 'failed', code: 'agent_session_operation_conflict' } + }) + } + ) + + it('admits only cancellation beside a queued operation', async () => { + const { guard } = await createGuard() + guard.start(SESSION, { + callerKey: 'client-a', + operationId: OPERATION_A, + fingerprint: 'fingerprint-a' + }) + + await expect( + guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'cancel-queued', + status: status('queued'), + now: NOW + }) + ).resolves.toEqual({ decision: 'new' }) + }) +}) + +describe('queued handoff fence revalidation', () => { + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION_A, + expectedRuntimeFence: 7, + payloadFingerprint: 'fingerprint' + }, + direction: 'to-tui', + mode: 'after-turn', + action: 'start' + } + const queued = status('queued') + + it('accepts the same live owner and fence', () => { + const record = agentSessionRecordFixture( + agentSessionLeaseFixture({ runtimeKind: 'native', ownerProcess: null }) + ) + expect(queuedStructuredHandoffCanBegin(record, queued, params)).toBe(true) + }) + + it.each([ + agentSessionLeaseFixture({ runtimeKind: 'native', runtimeFence: 8, ownerProcess: null }), + agentSessionLeaseFixture({ runtimeKind: 'tui' }), + agentSessionLeaseFixture({ + runtimeKind: 'native', + ownerProcess: null, + handoffStage: 'preparing' + }) + ])('refuses a changed durable owner or fence', (lease) => { + expect(queuedStructuredHandoffCanBegin(agentSessionRecordFixture(lease), queued, params)).toBe( + false + ) + }) + + it('cannot cancel after the idle waiter claims the queued operation', async () => { + const queue = new StructuredAgentSessionHandoffQueue() + const ready = vi.fn() + queue.enqueue(SESSION, () => true, ready) + await vi.waitFor(() => expect(ready).toHaveBeenCalledOnce()) + expect(queue.cancel(SESSION)).toBe(false) + }) +}) + +describe('scheduled handoff revalidation', () => { + it('refuses a native turn accepted ahead of the scheduled handoff', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-revalidation-')) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: SESSION, leafUuid: null } + }, + journalDir: join(root, 'journal') + }) + const journalSequence = journal.cursor().sequence + await journal.appendItem( + { provider: 'orca', clientMessageId: 'turn-running' }, + { kind: 'status', text: 'running', turnLifecycle: { turnId: 'turn-1', state: 'running' } }, + { fence: 7 } + ) + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION_A, + expectedRuntimeFence: 7, + payloadFingerprint: 'fingerprint' + }, + direction: 'to-tui', + mode: 'now', + action: 'start' + } + + expect(() => + assertScheduledStructuredHandoffIsAdmissible({ + record: agentSessionRecordFixture( + agentSessionLeaseFixture({ runtimeKind: 'native', ownerProcess: null }) + ), + journal, + params, + turnId: null, + journalSequence, + tuiAlreadyExited: false, + tuiStatus: 'busy' + }) + ).toThrow('session changed') + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts new file mode 100644 index 00000000000..da8d2eaad3d --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts @@ -0,0 +1,125 @@ +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { + AgentSessionOperationOutcome, + AgentSessionOperationRefusalCode +} from '../../../shared/agent-session-operation-ledger' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' + +type ActiveOperation = { callerKey: string; operationId: string; fingerprint: string } + +export type HandoffOperationDecision = + | { decision: 'new' } + | { decision: 'replay'; outcome: AgentSessionOperationOutcome } + | { decision: 'retry' } + | { decision: 'refused'; code: AgentSessionOperationRefusalCode } + +export class StructuredAgentSessionHandoffOperationGuard { + private readonly activeBySession = new Map() + + constructor(private readonly store: AgentSessionRecordStore) {} + + async check(input: { + callerKey: string + sessionId: string + operationId: string + fingerprint: string + action: 'start' | 'cancel-queued' | 'retry' | 'recover' + status?: AgentSessionHandoffStatus + now: number + }): Promise { + const ledger = await this.store.admitOperation({ + callerKey: input.callerKey, + operationId: input.operationId, + fingerprint: input.fingerprint, + now: input.now + }) + if (ledger.decision === 'refused') { + return { decision: 'refused', code: ledger.code } + } + const active = this.activeBySession.get(input.sessionId) + const queuedCancellation = + input.action === 'cancel-queued' && + input.status?.phase === 'queued' && + input.status.operationId === active?.operationId + const activeConflict = Boolean( + active && + ((active.operationId === input.operationId && + (active.fingerprint !== input.fingerprint || active.callerKey !== input.callerKey)) || + (active.operationId !== input.operationId && !queuedCancellation)) + ) + const queuedConflict = Boolean( + !active && + input.status?.phase === 'queued' && + input.status.operationId !== input.operationId && + input.action !== 'cancel-queued' + ) + if (activeConflict || queuedConflict) { + if (ledger.decision === 'admit') { + await this.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.operationId, + outcome: { status: 'failed', code: 'agent_session_operation_conflict' } + }) + } + return { decision: 'refused', code: 'agent_session_operation_conflict' } + } + if (ledger.decision === 'admit') { + this.reserve(input) + return { decision: 'new' } + } + if (input.action === 'retry' && ledger.row.outcome.status === 'failed') { + await this.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.operationId, + outcome: { status: 'pending' } + }) + this.reserve(input) + return { decision: 'retry' } + } + if ( + ledger.row.outcome.status === 'pending' && + !active && + input.status?.operationId !== input.operationId + ) { + this.reserve(input) + return { decision: 'new' } + } + return { decision: 'replay', outcome: ledger.row.outcome } + } + + start(sessionId: string, operation: ActiveOperation): void { + this.activeBySession.set(sessionId, operation) + } + + private reserve(input: { + action: 'start' | 'cancel-queued' | 'retry' | 'recover' + callerKey: string + sessionId: string + operationId: string + fingerprint: string + }): void { + if (input.action !== 'cancel-queued') { + this.start(input.sessionId, input) + } + } + + finish(sessionId: string, operationId: string): void { + if (this.activeBySession.get(sessionId)?.operationId === operationId) { + this.activeBySession.delete(sessionId) + } + } + + async settle( + sessionId: string, + operationId: string, + outcome: AgentSessionOperationOutcome + ): Promise { + const active = this.activeBySession.get(sessionId) + await this.store.recordOperationOutcome({ + ...(active?.operationId === operationId ? { callerKey: active.callerKey } : {}), + operationId, + outcome + }) + this.finish(sessionId, operationId) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts new file mode 100644 index 00000000000..e5ee4f7ca9b --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts @@ -0,0 +1,286 @@ +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffRequest, + AgentSessionMutationEnvelope +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { AgentSessionOptionRejectedError } from './structured-agent-session-option-error' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestMessage, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +const CALLER = { callerKey: 'client-1' } +const DEFAULT_MODEL = 'gpt-default' +const PICKED_MODEL = 'gpt-picked' +const PICKED_EFFORT = 'medium' + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let activeModel: string +let activeEffort: string | null +let transcriptPath: string +let optionFailure: Error | null +const dispatchedModels: string[] = [] +const launchedOptions: (Readonly> | undefined)[] = [] +const closedTuiOwners: StructuredTuiOwner[] = [] + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function handoff(direction: AgentSessionHandoffDirection): AgentSessionHandoffRequest { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { envelope: envelope('agentSession.requestHandoff', fields), ...fields } +} + +function tuiOwner(fence: number, spawnToken: string): StructuredTuiOwner { + return { + terminal: { handle: 'term-tui', tabId: 'tab-tui', paneKey: 'pane-tui', ptyId: 'pty-tui' }, + process: { + hostId: 'local', + pid: 5200, + processStartTimeMs: NOW, + spawnToken + }, + link: { + linkId: `tui-link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + transcriptPath + } +} + +function handoffTransport(): StructuredAgentSessionHandoffTransport { + return { + hostLabel: 'Test host', + launchTui: async ({ record, fence, spawnToken }) => { + launchedOptions.push(record.options) + return tuiOwner(fence, spawnToken) + }, + reproveTuiOwner: async ({ owner }) => owner, + recoverTuiOwner: async (record) => + tuiOwner( + record.lease.runtimeFence, + record.lease.ownerProcess?.spawnToken ?? record.lease.reservedSpawnToken ?? 'recovered' + ), + stopRecoveredOwner: async () => undefined, + closeTuiOwner: async (owner) => { + closedTuiOwners.push(owner) + return { transcriptPath: owner.transcriptPath } + }, + waitForTuiExit: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle' + } +} + +function adapter(): StructuredAgentSessionAdapter { + acquire = vi.fn(async ({ fence, spawnToken, options }) => { + activeModel = options?.model ?? DEFAULT_MODEL + activeEffort = options?.effort ?? null + return { + process: { + hostId: 'local', + pid: 4200 + acquire.mock.calls.length, + processStartTimeMs: NOW, + spawnToken + }, + link: { + linkId: `native-link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: acquire.mock.calls.length === 1 ? 'created' : 'resumed', + mintedAtFence: fence, + observedAt: NOW + } + } + }) + return { + acquire, + dispatch: vi.fn(async () => { + dispatchedModels.push(activeModel) + return { + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 1 } + } + }), + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt: vi.fn(async () => undefined), + setOption: vi.fn(async ({ key, value }) => { + if (optionFailure) { + const error = optionFailure + optionFailure = null + throw error + } + if (key === 'model') { + activeModel = value + } else if (key === 'effort') { + activeEffort = value + } + return { + model: activeModel, + ...(activeEffort ? { effort: activeEffort } : {}) + } + }), + readOptions: vi.fn(async () => ({ + current: { model: activeModel, ...(activeEffort ? { effort: activeEffort } : {}) }, + models: [] + })), + closeSession: vi.fn(async () => { + activeModel = DEFAULT_MODEL + return true + }) + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-options-')) + resetHostTestOperationIds() + activeModel = DEFAULT_MODEL + activeEffort = null + optionFailure = null + dispatchedModels.length = 0 + launchedOptions.length = 0 + closedTuiOwners.length = 0 + const accountHome = join(root, 'codex-home') + const sessionsDir = join(accountHome, 'sessions', '2026', '08', '12') + transcriptPath = join(sessionsDir, `rollout-2026-08-12T10-00-00-${THREAD}.jsonl`) + await mkdir(sessionsDir, { recursive: true }) + await writeFile( + transcriptPath, + `${JSON.stringify({ + type: 'session_meta', + timestamp: '2026-08-12T10:00:00.000Z', + payload: { id: THREAD, session_id: THREAD } + })}\n`, + 'utf8' + ) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-native', + handoffTransport: handoffTransport(), + now: () => NOW + }) + const attached = await host.attach( + CALLER, + hostTestAttachParams(null, { accountHome: { variable: 'CODEX_HOME', path: accountHome } }) + ) + expect(attached).toMatchObject({ ok: true }) +}) + +afterEach(async () => { + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +describe('structured session handoff options', () => { + it('settles a pre-mutation rejection so a fresh retry can succeed', async () => { + optionFailure = new AgentSessionOptionRejectedError('model list unavailable') + const fields = { key: 'model', value: PICKED_MODEL } + const rejected = { + envelope: envelope('agentSession.setOption', fields), + ...fields + } + + expect(await host.setOption(CALLER, rejected)).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_operation_invalid', message: 'model list unavailable' } + }) + expect(await host.setOption(CALLER, rejected)).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_operation_invalid' } + }) + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', fields), + ...fields + }) + ).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + }) + + it('keeps a picked model through a native to TUI to native round trip', async () => { + const optionFields = { key: 'model', value: PICKED_MODEL } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', optionFields), + ...optionFields + }) + ).toMatchObject({ ok: true }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + + const effortFields = { key: 'effort', value: PICKED_EFFORT } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', effortFields), + ...effortFields + }) + ).toMatchObject({ ok: true }) + expect(store.getRecord(SESSION)?.options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + + expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }) + ) + expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }) + ) + + expect(launchedOptions).toEqual([{ model: PICKED_MODEL, effort: PICKED_EFFORT }]) + expect(closedTuiOwners).toHaveLength(1) + expect(acquire.mock.calls[1]?.[0].options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + expect(store.getRecord(SESSION)?.options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + const body = hostTestMessage('use the selected model') + expect( + await host.send(CALLER, { + envelope: envelope('agentSession.send', { body }), + body + }) + ).toMatchObject({ ok: true }) + expect(dispatchedModels).toEqual([PICKED_MODEL]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts new file mode 100644 index 00000000000..a5afd9891a1 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts @@ -0,0 +1,23 @@ +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' + +export async function readNativeHandoffSessionOptions(input: { + adapter: Pick + sessionId: string + fence: number + priorOptions?: Readonly> +}): Promise> | undefined> { + const { adapter, sessionId, fence, priorOptions } = input + const reported = await adapter.readOptions?.({ + sessionId, + fence + }) + if (!reported) { + return undefined + } + const { model: _model, effort: _effort, ...restored } = priorOptions ?? {} + return { + ...restored, + model: reported.current.model, + ...(reported.current.effort ? { effort: reported.current.effort } : {}) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts new file mode 100644 index 00000000000..f8d6db2bace --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts @@ -0,0 +1,46 @@ +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { StructuredAgentSessionHandoffQueue } from './structured-agent-session-handoff-queue' +import type { + StructuredAgentSessionHandoffDeps, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +export function queueStructuredHandoffAfterTurn(input: { + callerKey: string + params: AgentSessionHandoffRequest + deps: StructuredAgentSessionHandoffDeps + queue: StructuredAgentSessionHandoffQueue + owner: (sessionId: string) => StructuredTuiOwner | undefined + setStatus: ( + sessionId: string, + status: Parameters[1] + ) => void + begin: (callerKey: string, params: AgentSessionHandoffRequest, tuiAlreadyExited?: boolean) => void +}): void { + const { callerKey, params, deps, queue, owner, setStatus, begin } = input + const sessionId = params.envelope.sessionId + let tuiReadiness: 'idle' | 'exited' | null = null + setStatus(sessionId, { + owner: params.direction === 'to-tui' ? 'native' : 'tui', + direction: params.direction, + phase: 'queued', + stage: null, + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + const tuiOwner = owner(sessionId) + queue.enqueue( + sessionId, + async (signal) => { + if (params.direction === 'to-tui') { + return !activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items) + } + tuiReadiness = tuiOwner + ? ((await deps.transport?.waitForTuiIdleOrExit(tuiOwner, signal)) ?? null) + : null + return tuiReadiness !== null + }, + () => begin(callerKey, { ...params, mode: 'now' }, tuiReadiness === 'exited') + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts new file mode 100644 index 00000000000..9ea3d7ff08f --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts @@ -0,0 +1,133 @@ +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { + StructuredAgentSessionHandoffDeps, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +export class StructuredAgentSessionHandoffQueue { + private readonly controllers = new Map() + + cancel(sessionId: string): boolean { + const controller = this.controllers.get(sessionId) + controller?.abort() + this.controllers.delete(sessionId) + return controller !== undefined + } + + enqueue( + sessionId: string, + isIdle: (signal: AbortSignal) => boolean | Promise, + onReady: () => void + ): void { + this.cancel(sessionId) + const controller = new AbortController() + this.controllers.set(sessionId, controller) + void this.waitUntilIdle(sessionId, controller, isIdle).then((ready) => { + if (ready) { + onReady() + } + }) + } + + private async waitUntilIdle( + sessionId: string, + controller: AbortController, + isIdle: (signal: AbortSignal) => boolean | Promise + ): Promise { + while (this.controllers.get(sessionId) === controller && !controller.signal.aborted) { + try { + if (await isIdle(controller.signal)) { + this.controllers.delete(sessionId) + return true + } + } catch { + if (controller.signal.aborted) { + return false + } + } + await new Promise((resolve) => setTimeout(resolve, 150)) + } + return false + } +} + +export function queuedStructuredHandoffCanBegin( + record: AgentSessionRecord, + status: AgentSessionHandoffStatus, + params: AgentSessionHandoffRequest +): boolean { + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + return ( + record.sessionId === params.envelope.sessionId && + status.phase === 'queued' && + status.direction === params.direction && + status.operationId === params.envelope.clientOperationId && + record.lease.runtimeFence === params.envelope.expectedRuntimeFence && + record.lease.runtimeKind === expectedOwner && + record.lease.claimStatus === 'live' && + record.lease.handoffStage === null && + !record.lease.unreconciled + ) +} + +export function enqueueStructuredHandoffAfterTurn(input: { + deps: StructuredAgentSessionHandoffDeps + queue: StructuredAgentSessionHandoffQueue + params: AgentSessionHandoffRequest + tuiOwner: StructuredTuiOwner | undefined + status: () => AgentSessionHandoffStatus + requireRecord: () => AgentSessionRecord + setStatus: (status: AgentSessionHandoffStatus) => void + begin: (params: AgentSessionHandoffRequest, tuiAlreadyExited: boolean) => void + refuse: (record: AgentSessionRecord) => void +}): void { + const { deps, params, queue, tuiOwner } = input + const sessionId = params.envelope.sessionId + let tuiReadiness: 'idle' | 'exited' | null = null + let observedTuiQueue = false + input.setStatus({ + owner: params.direction === 'to-tui' ? 'native' : 'tui', + direction: params.direction, + phase: 'queued', + stage: null, + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + queue.enqueue( + sessionId, + async (signal) => { + if (params.direction === 'to-tui') { + return !activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items) + } + if (!observedTuiQueue) { + observedTuiQueue = true + return false + } + tuiReadiness = tuiOwner + ? ((await deps.transport?.waitForTuiIdleOrExit(tuiOwner, signal)) ?? null) + : null + if (tuiReadiness === 'exited') { + return true + } + if (!activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items)) { + tuiReadiness = 'idle' + return true + } + return false + }, + () => { + const record = input.requireRecord() + const status = input.status() + if (!queuedStructuredHandoffCanBegin(record, status, params)) { + input.refuse(record) + return + } + input.begin(params, tuiReadiness === 'exited') + } + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts new file mode 100644 index 00000000000..0aab4f0335b --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts @@ -0,0 +1,30 @@ +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { + beginStructuredManualRecovery, + structuredManualRecoveryIsAdmissible +} from './structured-agent-session-manual-recovery' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' + +export async function requestStructuredManualRecovery(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + fingerprint: string + record: AgentSessionRecord + status: AgentSessionHandoffStatus + requireRecord: (sessionId: string) => AgentSessionRecord + restore: (sessionId: string) => Promise + setStatus: (sessionId: string, status: AgentSessionHandoffStatus) => void +}): Promise { + if (!structuredManualRecoveryIsAdmissible(input.record, input.status)) { + return false + } + beginStructuredManualRecovery(input) + return true +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts new file mode 100644 index 00000000000..fe480d6359c --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts @@ -0,0 +1,32 @@ +import type { + AgentSessionHandoffResult, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export function structuredHandoffRefusal( + code: AgentSessionWireRefusal['code'], + message: string +): AgentSessionWireRefusal { + return { code, message } +} + +export function structuredHandoffSuccess( + deps: StructuredAgentSessionHandoffDeps, + sessionId: string, + replayed: boolean, + status: AgentSessionHandoffResult['status'] +): AgentSessionMutationResult { + const record = deps.store.getRecord(sessionId) + if (!record) { + throw new Error('agent_session_identity_required') + } + return { + ok: true, + replayed, + fence: record.lease.runtimeFence, + cursor: deps.session(sessionId).journal.cursor(), + value: { status } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts new file mode 100644 index 00000000000..66005959378 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts @@ -0,0 +1,52 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { structuredHandoffRetryResumesStoppedOwner } from './structured-agent-session-handoff-admission' +import { structuredSessionHasPendingPrompt } from './structured-agent-session-handoff-status' + +export function assertScheduledStructuredHandoffIsAdmissible(input: { + record: AgentSessionRecord + journal: AgentSessionJournal + params: AgentSessionHandoffRequest + turnId: string | null + journalSequence: number + tuiAlreadyExited: boolean + tuiStatus: 'idle' | 'busy' +}): void { + const { params, record } = input + if (params.action === 'retry' && structuredHandoffRetryResumesStoppedOwner(record, params)) { + return + } + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + if ( + record.lease.runtimeFence !== params.envelope.expectedRuntimeFence || + record.lease.runtimeKind !== expectedOwner || + record.lease.claimStatus !== 'live' || + record.lease.handoffStage !== null || + record.lease.unreconciled + ) { + throw new Error('agent_session_checkpoint_stale') + } + if (structuredSessionHasPendingPrompt(input.journal)) { + throw new Error('Resolve the pending question or approval before switching.') + } + if (params.mode !== 'stop-turn' && input.journal.cursor().sequence !== input.journalSequence) { + throw new Error('The session changed before the handoff started.') + } + const activeTurn = activeStructuredAgentSessionTurnId(input.journal.snapshot().items) + if (params.direction === 'to-tui') { + const expectedTurn = params.mode === 'stop-turn' ? input.turnId : null + if (activeTurn !== expectedTurn) { + throw new Error('The native turn changed before the handoff started.') + } + return + } + if ( + !input.tuiAlreadyExited && + input.tuiStatus !== 'idle' && + (params.mode !== 'after-turn' || activeTurn !== null) + ) { + throw new Error('The agent terminal became busy before the handoff started.') + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts new file mode 100644 index 00000000000..91c6163bd17 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts @@ -0,0 +1,100 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import { handoffStructuredSessionToNative } from './structured-agent-session-handoff-reverse' +import type { StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' + +const OPERATION_ID = 'operation-1' +const SESSION_ID = 'session-1' + +vi.mock('../../runtime/agent-session-handoff-record-transitions', () => ({ + abandonStoredAgentSessionHandoffAttempt: vi.fn(async () => undefined), + reserveStoredAgentSessionHandoffOwner: vi.fn(async () => record()), + rollbackStoredAgentSessionHandoffPreparation: vi.fn(async () => undefined), + stopStoredAgentSessionOwnerForHandoff: vi.fn(async () => record()) +})) + +function record(): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + lease: { + runtimeFence: 3, + handoffStage: 'old-owner-stopped', + handoffOperationId: OPERATION_ID + } + } as unknown as AgentSessionRecord +} + +function contextWith( + revealNativeSession: () => Promise, + statuses: AgentSessionHandoffStatus[] +): StructuredAgentSessionHandoffFlowContext { + return { + deps: { + store: {} as never, + claimKeyId: 'key-1', + now: () => 1_800_000_000_000, + importTuiHistory: vi.fn(async () => undefined), + acquireNative: vi.fn(async () => record()), + transport: { revealNativeSession } + } as never, + owner: () => undefined, + retainOwner: vi.fn(), + releaseOwner: vi.fn(), + setStatus: (_sessionId, status) => statuses.push(status), + enterPreparing: vi.fn(async () => undefined), + publishStage: vi.fn(), + requireRecord: () => record() + } +} + +// Why this ordering matters: releaseOwner has already run by the time the reveal fires, +// so a reveal that rejects before the status flip leaves the session released but never +// marked native — a stuck chat with no owner on either side. +describe('handoffStructuredSessionToNative', () => { + it('marks the session native before revealing it', async () => { + const statuses: AgentSessionHandoffStatus[] = [] + const order: string[] = [] + const context = contextWith(async () => { + order.push('reveal') + }, statuses) + const setStatus = context.setStatus + context.setStatus = (sessionId, status) => { + order.push('status') + setStatus(sessionId, status) + } + + await handoffStructuredSessionToNative( + context, + { envelope: { sessionId: SESSION_ID, clientOperationId: OPERATION_ID } } as never, + true + ) + + expect(order).toEqual(['status', 'reveal']) + expect(statuses.at(-1)).toMatchObject({ owner: 'native', direction: null, phase: 'idle' }) + }) + + it('still leaves the session marked native when the reveal rejects', async () => { + const statuses: AgentSessionHandoffStatus[] = [] + const context = contextWith(async () => { + throw new Error('publish failed') + }, statuses) + + await expect( + handoffStructuredSessionToNative( + context, + { envelope: { sessionId: SESSION_ID, clientOperationId: OPERATION_ID } } as never, + true + ) + ).rejects.toThrow('publish failed') + + expect(statuses.at(-1)).toMatchObject({ owner: 'native' }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts index f59f7735245..ebfca81525c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts @@ -130,12 +130,8 @@ export async function handoffStructuredSessionToNative( throw error } context.releaseOwner(sessionId) - await deps.transport?.revealNativeSession?.({ - workspaceId: record.location.workspaceId, - sessionId, - agent: record.provider, - ...(owner?.adoptedTerminal ? { adoptedTerminal: true } : {}) - }) + // Why status lands before the reveal: the native owner is already proven here, and a + // reveal that rejects must not leave the session released but never marked native. context.setStatus(sessionId, { owner: 'native', direction: null, @@ -143,4 +139,10 @@ export async function handoffStructuredSessionToNative( stage: record.lease.handoffStage, operationId: record.lease.handoffOperationId }) + await deps.transport?.revealNativeSession?.({ + workspaceId: record.location.workspaceId, + sessionId, + agent: record.provider, + ...(owner?.adoptedTerminal ? { adoptedTerminal: true } : {}) + }) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts new file mode 100644 index 00000000000..e5dd478f719 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts @@ -0,0 +1,80 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +type TestCoordinatorInput = { + store: AgentSessionRecordStore + journal: AgentSessionJournal + sessionId: string + provider: 'claude' | 'codex' + claudeSessionId: string + codexThreadId: string + now: number + launchTui: StructuredAgentSessionHandoffTransport['launchTui'] + reproveTuiOwner: StructuredAgentSessionHandoffTransport['reproveTuiOwner'] + stopRecoveredOwner: StructuredAgentSessionHandoffTransport['stopRecoveredOwner'] + closeTuiOwner: NonNullable + waitForTuiExit: StructuredAgentSessionHandoffTransport['waitForTuiExit'] + waitForTuiIdleOrExit: StructuredAgentSessionHandoffTransport['waitForTuiIdleOrExit'] + stopFailedTuiLaunch: NonNullable + recoverTuiOwner: (record: AgentSessionRecord) => Promise + tuiStatus: () => 'idle' | 'busy' + acquireNative: (input: { + sessionId: string + fence: number + spawnToken: string + }) => Promise + acquireNativeStop: (turnId: string) => Promise + takeImportFailure: () => Error | null + statuses: AgentSessionHandoffStatus[] +} + +export function createStructuredAgentSessionHandoffTestCoordinator( + input: TestCoordinatorInput +): StructuredAgentSessionHandoffCoordinator { + return new StructuredAgentSessionHandoffCoordinator({ + store: input.store, + claimKeyId: 'key-1', + transport: { + hostLabel: 'Test host', + launchTui: input.launchTui, + reproveTuiOwner: input.reproveTuiOwner, + recoverTuiOwner: input.recoverTuiOwner, + stopRecoveredOwner: input.stopRecoveredOwner, + closeTuiOwner: input.closeTuiOwner, + waitForTuiExit: input.waitForTuiExit, + waitForTuiIdleOrExit: input.waitForTuiIdleOrExit, + tuiStatus: input.tuiStatus, + stopFailedTuiLaunch: input.stopFailedTuiLaunch + }, + session: () => ({ + journal: input.journal, + fence: input.store.getRecord(input.sessionId)?.lease.runtimeFence ?? 1 + }), + suspendNative: async () => ({ state: 'stopped' as const }), + acquireNative: input.acquireNative, + acquireNativeStop: (_sessionId, turnId) => input.acquireNativeStop(turnId), + importTuiHistory: async ({ fence }) => { + const importFailure = input.takeImportFailure() + if (importFailure) { + throw importFailure + } + await input.journal.appendItem( + input.provider === 'claude' + ? { provider: 'claude', sessionId: input.claudeSessionId, uuid: 'tui-turn' } + : { provider: 'codex', threadId: input.codexThreadId, turnId: 'tui-turn', ordinal: 0 }, + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'from tui' }] }, + { fence, recovered: true } + ) + }, + publish: (_sessionId, status) => input.statuses.push(status), + schedule: async (_sessionId, task) => task(), + now: () => input.now + }) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts new file mode 100644 index 00000000000..ca33e486999 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts @@ -0,0 +1,40 @@ +export type StructuredHandoffProviderCase = { + provider: 'claude' | 'codex' + accountHome: { variable: 'CLAUDE_CONFIG_DIR' | 'CODEX_HOME'; pathName: string } +} + +export const STRUCTURED_HANDOFF_PROVIDER_CASES: StructuredHandoffProviderCase[] = [ + { provider: 'codex', accountHome: { variable: 'CODEX_HOME', pathName: 'codex-home' } }, + { + provider: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', pathName: 'claude-home' } + } +] + +export function structuredHandoffTestProcess(now: number, spawnToken: string, pid: number) { + return { hostId: 'local', pid, processStartTimeMs: now - 1_000, spawnToken } +} + +export function structuredHandoffTestLink(input: { + provider: 'claude' | 'codex' + fence: number + id: string + now: number + claudeSessionId: string + codexThreadId: string +}) { + return { + linkId: input.id, + handle: + input.provider === 'claude' + ? ({ + provider: 'claude' as const, + sessionId: input.claudeSessionId, + leafUuid: input.id.startsWith('native-link') ? 'tui-exit-leaf' : 'current-leaf' + } as const) + : ({ provider: 'codex' as const, threadId: input.codexThreadId } as const), + origin: 'resumed' as const, + mintedAtFence: input.fence, + observedAt: input.now + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts new file mode 100644 index 00000000000..550ebded0da --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts @@ -0,0 +1,53 @@ +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffAction, + AgentSessionHandoffMode, + AgentSessionHandoffRequest +} from '../../../shared/agent-session-wire' + +export type StructuredHandoffTestRequestOptions = { + action?: AgentSessionHandoffAction + operationId?: string +} + +export class StructuredHandoffTestRequests { + private operations = 0 + + constructor( + private readonly now: number, + private readonly sessionId: string, + private readonly readFence: () => number + ) {} + + reset(): void { + this.operations = 0 + } + + operationId(): string { + this.operations += 1 + return `${this.now}-${this.operations.toString(16).padStart(32, '0')}` + } + + request( + direction: AgentSessionHandoffDirection, + mode: AgentSessionHandoffMode, + options: StructuredHandoffTestRequestOptions = {} + ): AgentSessionHandoffRequest { + const action = options.action ?? 'start' + const fields = { direction, mode, action } + return { + envelope: { + sessionId: this.sessionId, + clientOperationId: options.operationId ?? this.operationId(), + expectedRuntimeFence: this.readFence(), + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: this.sessionId, + fields + }) + }, + ...fields + } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts index e125d3e49ae..f0f410b66aa 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts @@ -12,7 +12,8 @@ import { setStoredAgentSessionHandoffStage, stopStoredAgentSessionOwnerForHandoff } from '../../runtime/agent-session-handoff-record-transitions' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import type { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' import { createStructuredHandoffFlowContext } from './structured-agent-session-handoff-flow-context' import { handoffStructuredSessionToTui } from './structured-agent-session-handoff-forward' @@ -21,6 +22,8 @@ import type { StructuredTuiOwner } from './structured-agent-session-handoff-types' +const journals = createTrackedJournalOpener() + const NOW = 1_800_000_000_000 const SESSION = 'session-handoff' const PLAIN_RESIDUE = 'session-plain-residue' @@ -211,7 +214,7 @@ beforeEach(async () => { stopRecoveredOwner = vi.fn(async () => undefined) store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) await establishNativeOwner() - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -232,6 +235,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -426,7 +430,7 @@ describe('structured session ownership recovery on restore', () => { : { outcome: 'pid-absent' }, now: NOW + 1_000 }) - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts index 0e213921609..57333d1c90c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts @@ -1,63 +1,286 @@ import type { AgentSessionRecord } from '../../../shared/agent-session-record' -import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffResult, + AgentSessionHandoffStatus, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import { + admitStructuredHandoffRequest, + refuseAdmittedStructuredHandoff, + replayedStructuredHandoffRefusal, + structuredHandoffRetryIsAdmissible +} from './structured-agent-session-handoff-admission' import { createStructuredHandoffFlowContext, requireStructuredHandoffRecord } from './structured-agent-session-handoff-flow-context' -import { restoreStructuredAgentSessionHandoff } from './structured-agent-session-handoff-restart' +import { StructuredAgentSessionHandoffFlowRunner } from './structured-agent-session-handoff-flow-runner' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { StructuredAgentSessionHandoffQueue } from './structured-agent-session-handoff-queue' +import { queueStructuredHandoffAfterTurn } from './structured-agent-session-handoff-queue-start' import { closeRetainedTuiOwner } from './structured-agent-session-handoff-owner-close' +import { requestStructuredManualRecovery } from './structured-agent-session-handoff-recover' +import { restoreStructuredAgentSessionHandoff } from './structured-agent-session-handoff-restart' +import { + structuredHandoffRefusal as refusal, + structuredHandoffSuccess +} from './structured-agent-session-handoff-result' +import { + failedStructuredHandoffStatus, + idleStructuredHandoffStatus, + structuredSessionHasPendingPrompt, + structuredTuiStatus +} from './structured-agent-session-handoff-status' import type { StructuredAgentSessionHandoffDeps, StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' import { StructuredAgentSessionHandoffState } from './structured-agent-session-handoff-state' - export class StructuredAgentSessionHandoffCoordinator { private readonly state: StructuredAgentSessionHandoffState - + private readonly queue = new StructuredAgentSessionHandoffQueue() + private readonly operationGuard: StructuredAgentSessionHandoffOperationGuard + private readonly flowRunner: StructuredAgentSessionHandoffFlowRunner constructor(private readonly deps: StructuredAgentSessionHandoffDeps) { - // oxfmt-ignore - this.state = new StructuredAgentSessionHandoffState({ requireRecord: (sessionId) => this.requireRecord(sessionId), publish: deps.publish, hostLabel: deps.transport?.hostLabel }) - } - - status = (sessionId: string) => this.state.status(sessionId) - - closeRetainedTuiOwner = (sessionId: string): Promise => - closeRetainedTuiOwner({ - sessionId, - deps: this.deps, - owner: this.state.owner, - requireRecord: this.requireRecord, - releaseOwner: this.state.releaseOwner + this.state = new StructuredAgentSessionHandoffState({ + requireRecord: (sessionId) => this.requireRecord(sessionId), + publish: deps.publish, + hostLabel: deps.transport?.hostLabel }) - + this.operationGuard = new StructuredAgentSessionHandoffOperationGuard(deps.store) + this.flowRunner = new StructuredAgentSessionHandoffFlowRunner({ + deps, + operationGuard: this.operationGuard, + flowContext: () => this.flowContext(), + fail: (params, error) => this.fail(params, error) + }) + } + status = (sessionId: string): AgentSessionHandoffStatus => this.state.status(sessionId) + drain = (): Promise => this.flowRunner.drain() + closeRetainedTuiOwner = (sessionId: string): Promise => + this.closeRetainedOwner(sessionId) setStatus = (sessionId: string, status: AgentSessionHandoffStatus): void => this.state.setStatus(sessionId, status) - + async request( + callerKey: string, + params: AgentSessionHandoffRequest + ): Promise> { + const record = this.requireRecord(params.envelope.sessionId) + const currentStatus = this.state.cachedStatus(record.sessionId) + const admission = await admitStructuredHandoffRequest({ + deps: this.deps, + operationGuard: this.operationGuard, + callerKey, + params, + record, + ...(currentStatus ? { status: currentStatus } : {}) + }) + if (admission.decision === 'replay') { + const replayedRefusal = replayedStructuredHandoffRefusal(admission.outcome) + if (replayedRefusal) { + return { ok: false, refusal: replayedRefusal } + } + return this.success(record.sessionId, true) + } + if (admission.decision === 'refused') { + return { ok: false, refusal: admission.refusal } + } + const { fingerprint } = admission + const action = params.action ?? 'start' + if (action === 'cancel-queued') { + if (currentStatus?.phase !== 'queued' || currentStatus?.direction !== params.direction) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'No matching queued handoff exists.' + ) + } + this.queue.cancel(record.sessionId) + this.setStatus(record.sessionId, idleStructuredHandoffStatus(record)) + await this.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId: record.sessionId } + }) + return this.success(record.sessionId, false) + } + if (!this.deps.transport) { + return this.refuseAdmitted( + callerKey, + params, + 'structured_agent_session_unsupported', + 'Agent TUI handoff is unavailable on this host.' + ) + } + if (action === 'recover') { + const status = this.status(record.sessionId) + const started = await requestStructuredManualRecovery({ + deps: this.deps, + operationGuard: this.operationGuard, + callerKey, + params, + fingerprint, + record, + status, + requireRecord: this.requireRecord, + restore: this.restore, + setStatus: this.setStatus + }) + if (!started) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'This handoff is no longer eligible for proof recovery.' + ) + } + return this.success(record.sessionId, false) + } + if (action === 'retry') { + if (!structuredHandoffRetryIsAdmissible(this.status(record.sessionId), params)) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'This handoff is no longer retryable.' + ) + } + this.begin(callerKey, params, null, fingerprint) + return this.success(record.sessionId, false) + } + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + if (record.lease.runtimeKind !== expectedOwner || record.lease.claimStatus !== 'live') { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + `The ${expectedOwner} runtime does not own this session.` + ) + } + if (structuredSessionHasPendingPrompt(this.deps.session(record.sessionId).journal)) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + 'Resolve the pending question or approval before switching.' + ) + } + const turnId = activeStructuredAgentSessionTurnId( + this.deps.session(record.sessionId).journal.snapshot().items + ) + const tuiOwner = this.state.owner(record.sessionId) + const busy = + expectedOwner === 'native' + ? turnId !== null + : structuredTuiStatus(tuiOwner, this.deps.transport) !== 'idle' + if (busy && params.mode === 'now') { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + 'The current turn must finish before switching.' + ) + } + if (busy && params.mode === 'after-turn') { + queueStructuredHandoffAfterTurn({ + callerKey, + params, + deps: this.deps, + queue: this.queue, + owner: (sessionId) => this.state.owner(sessionId), + setStatus: this.setStatus, + begin: (key, next, tuiAlreadyExited) => + this.begin(key, next, null, fingerprint, tuiAlreadyExited) + }) + return this.success(record.sessionId, false) + } + if (busy && expectedOwner === 'tui' && params.mode === 'stop-turn') { + return this.refuseAdmitted( + callerKey, + params, + 'structured_agent_session_unsupported', + 'Exit the agent terminal after this turn to continue in chat.' + ) + } + this.begin(callerKey, params, turnId, fingerprint) + return this.success(record.sessionId, false) + } async restore(sessionId: string): Promise { await restoreStructuredAgentSessionHandoff( { deps: this.deps, requireRecord: (id) => this.requireRecord(id), flowContext: () => this.flowContext(), - retainOwner: this.state.retainOwner, - setStatus: this.state.setStatus + retainOwner: (id, owner) => this.state.retainOwner(id, owner), + setStatus: (id, status) => this.state.setStatus(id, status) }, sessionId ) } - + private refuseAdmitted( + callerKey: string, + params: AgentSessionHandoffRequest, + code: AgentSessionWireRefusal['code'], + message: string + ): Promise> { + return refuseAdmittedStructuredHandoff({ + deps: this.deps, + callerKey, + params, + refusal: refusal(code, message) + }) + } + private success( + sessionId: string, + replayed: boolean + ): AgentSessionMutationResult { + return structuredHandoffSuccess(this.deps, sessionId, replayed, this.status(sessionId)) + } + private begin( + callerKey: string, + params: AgentSessionHandoffRequest, + turnId: string | null, + fingerprint: string, + tuiAlreadyExited = false + ): void { + this.flowRunner.begin({ + callerKey, + params, + turnId, + fingerprint, + tuiAlreadyExited + }) + } private flowContext(): StructuredAgentSessionHandoffFlowContext { return createStructuredHandoffFlowContext({ deps: this.deps, - owner: this.state.owner, - retainOwner: this.state.retainOwner, - releaseOwner: this.state.releaseOwner, - setStatus: this.state.setStatus, + owner: (sessionId) => this.state.owner(sessionId), + retainOwner: (sessionId, owner) => this.state.retainOwner(sessionId, owner), + releaseOwner: (sessionId) => this.state.releaseOwner(sessionId), + setStatus: (sessionId, status) => this.state.setStatus(sessionId, status), requireRecord: (sessionId) => this.requireRecord(sessionId) }) } - + private fail(params: AgentSessionHandoffRequest, error: unknown): void { + const record = this.requireRecord(params.envelope.sessionId) + this.setStatus( + record.sessionId, + failedStructuredHandoffStatus(record, params, error, this.deps.transport?.hostLabel) + ) + } + private closeRetainedOwner(sessionId: string): Promise { + return closeRetainedTuiOwner({ + sessionId, + deps: this.deps, + owner: this.state.owner, + requireRecord: this.requireRecord, + releaseOwner: this.state.releaseOwner + }) + } private requireRecord = (sessionId: string): AgentSessionRecord => requireStructuredHandoffRecord(this.deps, sessionId) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts index 1828807f887..4c807d879b6 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts @@ -8,13 +8,15 @@ import type { } from '../../../shared/agent-session-record' import { LOCAL_EXECUTION_HOST_ID } from '../../../shared/execution-host' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { createDeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' import { acquireNativeHandoffOwner, structuredTuiTranscriptImportOptions } from './structured-agent-session-host-handoff' +const journals = createTrackedJournalOpener() + function importRecord(provider: 'claude' | 'codex', accountHome: string): AgentSessionRecord { return { provider, @@ -54,6 +56,7 @@ describe('native handoff acquisition', () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -82,7 +85,7 @@ describe('native handoff acquisition', () => { }, now }) - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId, workspaceId: location.workspaceId, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts index 5f3bbc5c731..2afd94ba128 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts @@ -48,7 +48,10 @@ export async function evictHeldStructuredAgentSession( hasProviderChild: hasProviderChild(context, sessionId), eventSink: context.runtimeState.eventSinkFor(sessionId), adapter: context.deps.adapter, - forget: () => context.sessions.delete(sessionId), + forget: async () => { + await context.sessions.get(sessionId)?.journal.close() + context.sessions.delete(sessionId) + }, discardSink: () => context.runtimeState.discardEventSink(sessionId), releaseLease: () => releaseStoredStructuredAgentSessionOwner({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-teardown.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-teardown.ts new file mode 100644 index 00000000000..2b230db1357 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-teardown.ts @@ -0,0 +1,53 @@ +// Host teardown, made failure-complete. +// +// A trailing "close every journal" statement is skipped on exactly the path +// that leaks: `flushAllEventSinks` throws BY DESIGN when a sink barrier fails, +// and the attach drain can reject too. Every connection would then be left open +// with the global runtime reference already cleared — the one state from which +// nothing can ever close them. + +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' +import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' + +export type StructuredAgentSessionTeardownPhase = { + name: string + run: () => Promise | void +} + +export async function tearDownStructuredAgentSessionHost(input: { + phases: readonly StructuredAgentSessionTeardownPhase[] + sessions: Map +}): Promise { + const failures: unknown[] = [] + for (const phase of input.phases) { + try { + await phase.run() + } catch (error) { + failures.push(error) + } + } + + const entries = [...input.sessions.entries()] + // `allSettled`, so one rejected close cannot skip the others. + const closed = await Promise.allSettled(entries.map(([, session]) => session.journal.close())) + closed.forEach((result, index) => { + const sessionId = entries[index]?.[0] + if (result.status === 'fulfilled') { + // Only a FULFILLED close drops the entry. One that rejected stays indexed, + // which is what makes a later close a real retry rather than a no-op. + if (sessionId !== undefined) { + input.sessions.delete(sessionId) + } + return + } + failures.push(result.reason) + }) + + // Journals an earlier failure path could not close are retried HERE, which is + // the only place that owns them once their caller has unwound. + failures.push(...(await agentSessionJournalCloseRetries.retryAll())) + + if (failures.length > 0) { + throw new AggregateError(failures, 'agent session host teardown failed') + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts index 8de64e3f419..5354670fa0b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts @@ -13,7 +13,7 @@ import type { import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { journalDirectoryFor } from '../agent-session-journal/journal-paths' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import type { AgentSessionDispatchOutcome, StructuredAgentSessionAdapter @@ -30,6 +30,8 @@ import { resetHostTestOperationIds } from './structured-agent-session-host-test-data' +const journals = createTrackedJournalOpener() + const CALLER = { callerKey: 'client-1' } function envelope( @@ -97,7 +99,7 @@ async function attach(): Promise { async function seedApproval(optionId = 'allow'): Promise<{ itemId: string; revision: number }> { const identity = { provider: 'codex' as const, threadId: THREAD, turnId: 'turn-1', ordinal: 99 } const journalDir = journalDirectoryFor(root, { workspaceId: 'workspace-1', sessionId: SESSION }) - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -157,6 +159,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await host.flushAllStreamedEvents() await rm(root, { recursive: true, force: true }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index 3c5256cece0..c38f64c3318 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -6,6 +6,8 @@ import type { AgentSessionAttachResult, AgentSessionHistoryRequest, AgentSessionHistoryResult, + AgentSessionHandoffRequest, + AgentSessionHandoffResult, AgentSessionHandoffStatus, AgentSessionMutationResult, AgentSessionOptionsResult, @@ -49,13 +51,13 @@ import { type StructuredAgentSessionMutationContext } from './structured-agent-session-host-mutations' import { StructuredAgentSessionReadableRestorer } from './structured-agent-session-readable-restorer' +import { tearDownStructuredAgentSessionHost } from './structured-agent-session-host-teardown' import type { StructuredAgentSessionCaller, StructuredAgentSessionHostDeps, StructuredAgentSessionHostSession } from './structured-agent-session-host-types' import { readStructuredAgentSessionHistoryResult } from './structured-agent-session-history-result' -import { retryPendingStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' import { StructuredAgentSessionEventRecovery } from './structured-agent-session-event-recovery' export type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' export class StructuredAgentSessionHost { @@ -108,6 +110,8 @@ export class StructuredAgentSessionHost { resolveRecovery: (sessionId) => this.runtimeState.resolveRecovery(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), hasSession: (sessionId) => this.sessions.has(sessionId), + // Site 10: cannot overwrite a live entry — the restorer returns early on + // `hasSession` inside the same serialized step as this `set`. onReadable: (sessionId, restored) => this.sessions.set(sessionId, restored), restoreHandoff: (sessionId) => this.handoffs.restore(sessionId) }) @@ -178,14 +182,6 @@ export class StructuredAgentSessionHost { subscribers: this.subscribers, tasks: this.tasks, reconcileLeases: (sessionId) => this.reconcileLeases(sessionId), - retryPendingSettlement: (sessionId, params) => - retryPendingStructuredAgentSessionSettlement({ - deps: this.deps, - sessions: this.sessions, - sessionId, - params, - now: () => this.now() - }), serialize: (sessionId, task) => this.serialize(sessionId, task), now: () => this.now() } @@ -249,11 +245,16 @@ export class StructuredAgentSessionHost { this.runtimeState.flushEventSink(sessionId) async flushAllStreamedEvents(): Promise { - this.holds.dispose() - this.runtimeState.stopLeaseRenewal() - this.handoffs.stopTuiHistoryCatchup() - await this.tasks.drainAttaches() - await this.runtimeState.flushAllEventSinks() + await tearDownStructuredAgentSessionHost({ + phases: [ + { name: 'dispose-holds', run: () => this.holds.dispose() }, + { name: 'stop-lease-renewal', run: () => this.runtimeState.stopLeaseRenewal() }, + { name: 'stop-tui-catchup', run: () => this.handoffs.stopTuiHistoryCatchup() }, + { name: 'drain-attaches', run: () => this.tasks.drainAttaches() }, + { name: 'flush-event-sinks', run: () => this.runtimeState.flushAllEventSinks() } + ], + sessions: this.sessions + }) } private mutationContext(): StructuredAgentSessionMutationContext { @@ -291,6 +292,12 @@ export class StructuredAgentSessionHost { ): ReturnType => setStructuredAgentSessionOption(this.mutationContext(), caller, params) + requestHandoff = ( + caller: StructuredAgentSessionCaller, + params: AgentSessionHandoffRequest + ): Promise> => + this.handoffs.request(caller.callerKey, params) + readOptions = (sessionId: string): Promise => readStructuredAgentSessionOptions(this.mutationContext(), sessionId) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-journal-handles.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-journal-handles.test.ts new file mode 100644 index 00000000000..b1cfd81b2a5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-journal-handles.test.ts @@ -0,0 +1,250 @@ +// Journal handle ownership across the wire layer. +// +// Every one of these sites is reached only when something has already gone +// wrong, so a happy-path assertion proves nothing about them. On POSIX a leak +// is silent; the rename/remove pair below is the half that actually fails on +// Windows. + +import { access, mkdtemp, rename, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import type * as JournalLegacyImport from '../agent-session-journal/journal-legacy-import' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { openAgentSessionJournalWithRecovery } from './agent-session-journal-recovery' +import { + evictStructuredAgentSession, + STRUCTURED_AGENT_SESSION_EVICTION_STEPS, + type StructuredAgentSessionEvictionContext +} from './structured-agent-session-eviction' +import { tearDownStructuredAgentSessionHost } from './structured-agent-session-host-teardown' +import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' + +const legacyImport = vi.hoisted(() => ({ throws: false })) + +vi.mock('../agent-session-journal/journal-legacy-import', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + importLegacyTranscriptIntoJournal: async ( + input: Parameters[0] + ) => { + if (legacyImport.throws) { + throw new Error('legacy import threw instead of reporting a failure') + } + return actual.importLegacyTranscriptIntoJournal(input) + } + } +}) + +const SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: SESSION, + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: SESSION } +} + +let root: string +let journalDir: string +const journals = createTrackedJournalOpener() + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +async function expectNothingHoldsTheDirectory(directory: string): Promise { + const dbPath = journalDatabaseFile(directory) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + const moved = `${directory}-moved` + await rename(directory, moved) + await rm(moved, { recursive: true }) +} + +function hostSession(journal: AgentSessionJournal): StructuredAgentSessionHostSession { + return { + journal, + params: {} as StructuredAgentSessionHostSession['params'], + fence: 1, + hasProviderChild: false, + acquisitionGeneration: null + } +} + +function evictionContext( + overrides: Partial +): StructuredAgentSessionEvictionContext { + return { + sessionId: SESSION, + hasProviderChild: false, + eventSink: { + drained: async () => ({ ok: true }) as const, + unbind: () => undefined, + close: () => undefined + } as unknown as StructuredAgentSessionEvictionContext['eventSink'], + adapter: {} as StructuredAgentSessionEvictionContext['adapter'], + forget: async () => undefined, + discardSink: () => undefined, + releaseLease: async () => undefined, + ...overrides + } +} + +beforeEach(async () => { + legacyImport.throws = false + root = await mkdtemp(join(tmpdir(), 'orca-wire-handles-')) + journalDir = join(root, 'journal') +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('site 6: recovery rehydration', () => { + it('closes the journal it opened when the legacy import throws', async () => { + const seeded = await journals.open({ identity: IDENTITY, journalDir }) + for (let ordinal = 1; ordinal <= 3; ordinal += 1) { + await seeded.appendItem( + { provider: 'codex', threadId: SESSION, turnId: 'turn-1', ordinal }, + { kind: 'status', text: `seed-${ordinal}` }, + { fence: 1 } + ) + } + await seeded.close() + // Punch a hole in the middle so recovery takes the `journal_corrupt` branch. + const { openJournalDatabase } = await import('../agent-session-journal/journal-database') + const opened = openJournalDatabase(journalDatabaseFile(journalDir)) + opened.db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(3) + opened.db.close() + legacyImport.throws = true + + await expect( + openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: join(root, 'missing.jsonl') + }) + ).rejects.toThrow('legacy import threw') + await expectNothingHoldsTheDirectory(journalDir) + }) +}) + +describe('sites 9 and 10: the delete and overwrite callbacks', () => { + it('awaits the journal close before dropping the map entry', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + const sessions = new Map([[SESSION, hostSession(journal)]]) + const order: string[] = [] + + await evictStructuredAgentSession( + evictionContext({ + forget: async () => { + order.push('close-started') + await sessions.get(SESSION)?.journal.close() + order.push('closed') + sessions.delete(SESSION) + order.push('forgotten') + } + }), + STRUCTURED_AGENT_SESSION_EVICTION_STEPS + ) + + expect(order).toEqual(['close-started', 'closed', 'forgotten']) + expect(sessions.size).toBe(0) + await expectNothingHoldsTheDirectory(journalDir) + }) + + it('aborts the eviction with the session still indexed when the close rejects', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + const sessions = new Map([[SESSION, hostSession(journal)]]) + + await expect( + evictStructuredAgentSession( + evictionContext({ + forget: async () => { + await Promise.reject(new Error('close rejected')) + } + }), + STRUCTURED_AGENT_SESSION_EVICTION_STEPS + ) + ).rejects.toMatchObject({ step: 'forget-session' }) + // Still indexed, so the next close is a real retry. + expect(sessions.has(SESSION)).toBe(true) + }) +}) + +describe('site 11: host teardown is failure-complete', () => { + async function twoSessions(): Promise> { + const first = await journals.open({ identity: IDENTITY, journalDir }) + const second = await journals.open({ + identity: { ...IDENTITY, sessionId: `${SESSION}-b` }, + journalDir: join(root, 'journal-b') + }) + return new Map([ + [SESSION, hostSession(first)], + [`${SESSION}-b`, hostSession(second)] + ]) + } + + it('closes every journal and clears the map on the happy path', async () => { + const sessions = await twoSessions() + await tearDownStructuredAgentSessionHost({ phases: [], sessions }) + + expect(sessions.size).toBe(0) + await expectNothingHoldsTheDirectory(journalDir) + await expectNothingHoldsTheDirectory(join(root, 'journal-b')) + }) + + // Against a trailing-statement design this case fails: `flushAllEventSinks` + // throws by design, so the close would be skipped on exactly the leaking path. + it('still closes every journal when a teardown phase throws', async () => { + const sessions = await twoSessions() + const barrierError = new Error('sink barrier failed') + + await expect( + tearDownStructuredAgentSessionHost({ + phases: [ + { + name: 'flush-event-sinks', + run: () => { + throw barrierError + } + } + ], + sessions + }) + ).rejects.toMatchObject({ errors: [barrierError] }) + + expect(sessions.size).toBe(0) + await expectNothingHoldsTheDirectory(journalDir) + await expectNothingHoldsTheDirectory(join(root, 'journal-b')) + }) + + it('keeps the entry whose close rejected, and surfaces the rejection', async () => { + const sessions = await twoSessions() + const failing = sessions.get(SESSION) + const closeError = new Error('close rejected') + if (failing) { + failing.journal = { + close: () => Promise.reject(closeError) + } as unknown as AgentSessionJournal + } + + await expect( + tearDownStructuredAgentSessionHost({ phases: [], sessions }) + ).rejects.toMatchObject({ errors: [closeError] }) + + // Only the failure stays indexed — `status === 'fulfilled'`, not "settled". + expect([...sessions.keys()]).toEqual([SESSION]) + await expectNothingHoldsTheDirectory(join(root, 'journal-b')) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts new file mode 100644 index 00000000000..91be1a15aa5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts @@ -0,0 +1,103 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { setStoredAgentSessionHandoffStage } from '../../runtime/agent-session-handoff-record-transitions' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { idleStructuredHandoffStatus } from './structured-agent-session-handoff-status' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export function structuredManualRecoveryIsAdmissible( + record: AgentSessionRecord, + status: AgentSessionHandoffStatus | undefined +): boolean { + return ( + record.lease.handoffStage === 'manual-recovery' && + record.lease.runtimeKind === 'tui' && + record.lease.ownerProcess !== null && + status?.error?.canRetryProof === true + ) +} + +export function beginStructuredManualRecovery(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + fingerprint: string + requireRecord: (sessionId: string) => AgentSessionRecord + restore: (sessionId: string) => Promise + setStatus: (sessionId: string, status: AgentSessionHandoffStatus) => void +}): Promise { + const { + callerKey, + deps, + fingerprint, + operationGuard, + params, + requireRecord, + restore, + setStatus + } = input + const sessionId = params.envelope.sessionId + operationGuard.start(sessionId, { + callerKey, + operationId: params.envelope.clientOperationId, + fingerprint + }) + setStatus(sessionId, { + owner: 'none', + direction: params.direction, + phase: 'switching', + stage: 'recovering', + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + return deps + .schedule(sessionId, async () => { + let record = requireRecord(sessionId) + if (record.lease.claimStatus === 'reserved' && record.lease.handoffOperationId !== null) { + record = await setStoredAgentSessionHandoffStage(deps.store, { + sessionId, + fence: record.lease.runtimeFence, + stage: 'new-owner-proving', + handoffOperationId: record.lease.handoffOperationId, + now: deps.now() + }) + } + await restore(record.sessionId) + if (requireRecord(sessionId).lease.handoffStage === 'manual-recovery') { + throw new Error('The TUI owner proof is still unavailable.') + } + }) + .then(() => { + operationGuard.finish(sessionId, params.envelope.clientOperationId) + return deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId } + }) + }) + .catch(async (error) => { + await deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_handoff_failed' } + }) + operationGuard.finish(sessionId, params.envelope.clientOperationId) + const status = idleStructuredHandoffStatus(requireRecord(sessionId)) + setStatus(sessionId, { + ...status, + ...(status.error + ? { + error: { + ...status.error, + details: error instanceof Error ? error.message : String(error) + } + } + : {}) + }) + }) + .finally(() => operationGuard.finish(sessionId, params.envelope.clientOperationId)) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts index b9ba03ff327..6e71f822170 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts @@ -1,7 +1,7 @@ import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' export async function readNativeSessionOptions(input: { - adapter: Pick + adapter: Pick sessionId: string fence: number priorOptions?: Readonly> @@ -11,7 +11,13 @@ export async function readNativeSessionOptions(input: { if (!reported) { return undefined } - const { model: _model, effort: _effort, ...restored } = priorOptions ?? {} + const skipped = new Set(input.adapter.readOptionRestoreFailures?.(sessionId) ?? []) + const restored = priorOptions ? { ...priorOptions } : {} + delete restored.model + delete restored.effort + for (const key of skipped) { + delete restored[key] + } return { ...restored, model: reported.current.model, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts new file mode 100644 index 00000000000..3caba894cb9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts @@ -0,0 +1,178 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { recoverStoredDeadTuiOwnerForHandoff } from '../../runtime/agent-session-handoff-record-transitions' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' +import type { StructuredAgentSessionHandoffTransport } from './structured-agent-session-handoff-types' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-proven-dead-retry' +const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' +const CREATE_OPERATION = `${NOW}-00000000000000000000000000000000` +const OPERATION = `${NOW}-00000000000000000000000000000001` +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('structured session proven-dead TUI retry', () => { + it('acquires native ownership without trying to close the dead TUI again', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-handoff-dead-retry-')) + roots.push(root) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + const reserved = await store.reserveOwner({ + sessionId: SESSION, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + provider: 'codex', + accountHome: { variable: 'CODEX_HOME', path: join(root, 'codex-home') }, + runtimeKind: 'tui', + expectedFence: null, + spawnToken: 'tui-spawn', + claimKeyId: 'key-1', + handoffOperationId: null, + probe: { outcome: 'reservation-unused' }, + operation: { callerKey: 'test', operationId: CREATE_OPERATION, fingerprint: 'create' }, + now: NOW + }) + const tuiFence = reserved.record.lease.runtimeFence + await store.commitProcessIdentity({ + sessionId: SESSION, + fence: tuiFence, + process: { + hostId: 'local', + pid: 4200, + processStartTimeMs: NOW - 1_000, + spawnToken: 'tui-spawn' + }, + now: NOW + }) + await store.proveOwner({ + sessionId: SESSION, + fence: tuiFence, + link: { + linkId: 'tui-link', + handle: { provider: 'codex', threadId: THREAD }, + origin: 'created', + mintedAtFence: tuiFence, + observedAt: NOW + }, + now: NOW + }) + await recoverStoredDeadTuiOwnerForHandoff(store, { + sessionId: SESSION, + expectedFence: tuiFence, + operationId: OPERATION, + probe: { outcome: 'pid-absent' }, + now: NOW + }) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: join(root, 'journal') + }) + const closeTuiOwner = + vi.fn>() + const coordinator = new StructuredAgentSessionHandoffCoordinator({ + store, + claimKeyId: 'key-1', + transport: { + hostLabel: 'Test host', + launchTui: vi.fn(), + reproveTuiOwner: vi.fn(), + recoverTuiOwner: vi.fn(), + stopRecoveredOwner: vi.fn(), + closeTuiOwner, + waitForTuiExit: vi.fn(), + waitForTuiIdleOrExit: vi.fn(), + tuiStatus: () => 'busy' + }, + session: () => ({ journal, fence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1 }), + suspendNative: vi.fn(), + acquireNative: async ({ fence, spawnToken }) => { + await store.commitProcessIdentity({ + sessionId: SESSION, + fence, + process: { + hostId: 'local', + pid: 4300, + processStartTimeMs: NOW, + spawnToken + }, + now: NOW + }) + return store.proveOwner({ + sessionId: SESSION, + fence, + link: { + linkId: 'native-link', + handle: { provider: 'codex', threadId: THREAD }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + now: NOW + }) + }, + acquireNativeStop: vi.fn(async () => true), + importTuiHistory: vi.fn(), + publish: vi.fn(), + schedule: async (_sessionId, task) => task(), + now: () => NOW + }) + const fields = { + direction: 'to-native' as const, + mode: 'now' as const, + action: 'retry' as const + } + const request: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: SESSION, + fields + }) + }, + ...fields + } + + expect(coordinator.status(SESSION)).toMatchObject({ phase: 'failed', owner: 'tui' }) + expect( + await ( + coordinator as { + request: (callerKey: string, params: AgentSessionHandoffRequest) => Promise + } + ).request('client-1', request) + ).toMatchObject({ ok: true }) + await vi.waitFor(() => expect(coordinator.status(SESSION).owner).toBe('native')) + // Settle the flow's trailing outcome write before afterEach removes the store root. + await coordinator.drain() + expect(closeTuiOwner).not.toHaveBeenCalled() + expect(store.getRecord(SESSION)?.lease).toMatchObject({ + runtimeKind: 'native', + claimStatus: 'live', + handoffStage: null + }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts index 5f2589de4f2..5bed8f2920e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts @@ -39,7 +39,7 @@ export async function restoreStructuredAgentSessionRead( workspaceId: record.location.workspaceId, sessionId }) - const loaded = await loadJournal(journalDir, sessionId) + const loaded = loadJournal(journalDir, sessionId) if (!loaded || loaded.corrupt) { return null } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts index a5927dc0c14..5cd888cc5cc 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts @@ -8,7 +8,7 @@ import { spawnProcess } from '../../../shared/child-process/run-process' import { CODEX_SPAWN_TOKEN_ENV } from '../../codex/codex-structured-owner-identity' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { readProcessStartTimeMs } from '../../runtime/agent-session-process-identity-probe' -import { createStructuredAgentSessionOwnerProbe } from '../../runtime/structured-agent-session-runtime' +import { createStructuredAgentSessionOwnerProbe } from '../../runtime/structured-agent-session-owner-probe' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { StructuredAgentSessionHost } from './structured-agent-session-host' import type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts index b06ec51018f..581743633c0 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts @@ -4,17 +4,19 @@ import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { AgentJournalMessageItem } from '../../../shared/agent-session-journal-types' import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { performSend, type AgentSessionTurnContext } from './structured-agent-session-turns' +const journals = createTrackedJournalOpener() + let root: string let journal: AgentSessionJournal beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-send-idempotency-')) - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: 'session-1', workspaceId: 'workspace-1', @@ -27,6 +29,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts index bd4dd1958ef..6b627726f98 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts @@ -1,4 +1,4 @@ -import { appendFile, mkdtemp, rm } from 'node:fs/promises' +import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -11,26 +11,30 @@ import { REMOTE_RUNTIME_MAX_OUTBOUND_JSON_BYTES, serializeRemoteRuntimePayload } from '../../../shared/remote-runtime-memory-limits' -import { JOURNAL_LOG_FILE } from '../agent-session-journal/journal-log-file' -import { serializeJournalRow, type JournalRow } from '../agent-session-journal/journal-row-schema' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { insertJournalRow } from '../agent-session-journal/journal-row-table' +import type { JournalRow } from '../agent-session-journal/journal-row-schema' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { AgentSessionSubscribers } from './structured-agent-session-subscribers' const SESSION = 'subscriber-session' let root: string +const journals = createTrackedJournalOpener() beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-agent-subscribers-')) }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) describe('AgentSessionSubscribers', () => { it('publishes the current fence when a resumed cursor is already caught up', async () => { - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -67,7 +71,7 @@ describe('AgentSessionSubscribers', () => { }) it('publishes handoff-only changes without serializing a transcript snapshot', async () => { - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -112,7 +116,7 @@ describe('AgentSessionSubscribers', () => { it('catches a subscriber up past a pre-existing unsendable removal with a bounded reset', async () => { const journalDir = join(root, 'oversized-removal-journal') - const seeded = await openAgentSessionJournal({ + const seeded = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -150,12 +154,20 @@ describe('AgentSessionSubscribers', () => { ts: 2_001 } ] - await appendFile( - join(journalDir, JOURNAL_LOG_FILE), - `${rows.map(serializeJournalRow).join('\n')}\n`, - 'utf-8' - ) - const journal = await openAgentSessionJournal({ + // Staged straight into the session database, exactly as a previous writer + // would have committed them. + await seeded.close() + const opened = openJournalDatabase(journalDatabaseFile(journalDir)) + try { + opened.db.exec('BEGIN IMMEDIATE') + for (const row of rows) { + insertJournalRow(opened.db, SESSION, row) + } + opened.db.exec('COMMIT') + } finally { + opened.db.close() + } + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts index 9d16e19a7f7..26ad85b5cfb 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts @@ -1,6 +1,11 @@ import { parseAgentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import { + decodeAgentSessionQuestionAnswers, + isValidAgentSessionQuestionAnswers +} from '../../../shared/agent-session-question-answer' import type { AgentJournalItemBody, + AgentJournalQuestion, AgentJournalResolution } from '../../../shared/agent-session-journal-types' import type { AgentSessionPromptResult } from '../../../shared/agent-session-wire' @@ -14,6 +19,7 @@ function invalid(message: string): TurnOutcome { function promptBodyOf(body: AgentJournalItemBody): { options: readonly { id: string }[] freeTextQuestionId?: string + questions?: AgentJournalQuestion[] resolution: AgentJournalResolution } | null { return body.kind === 'approval' || body.kind === 'question' ? body : null @@ -64,7 +70,19 @@ export async function performPrompt( prompt.freeTextQuestionId !== undefined && freeText?.questionId === prompt.freeTextQuestionId && freeText.answer.trim().length > 0 - if (!acceptsFreeText && !prompt.options.some((option) => option.id === input.optionId)) { + const grouped = + item.body.kind === 'question' && prompt.questions + ? decodeAgentSessionQuestionAnswers(input.optionId) + : null + const acceptsGrouped = + grouped !== null && + prompt.questions !== undefined && + isValidAgentSessionQuestionAnswers(prompt.questions, grouped) + if ( + !acceptsFreeText && + !acceptsGrouped && + !prompt.options.some((option) => option.id === input.optionId) + ) { return invalid(`Option ${input.optionId} is not offered by item ${input.itemId}.`) } const identity = parseAgentJournalItemKey(input.itemId) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts index 47edf3470e6..5222f9557f9 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts @@ -3,14 +3,9 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' -import { - performCancel, - performSend, - type AgentSessionTurnContext -} from './structured-agent-session-turns' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from '../agent-session-journal/journal-payload-bounds' +import { performCancel, type AgentSessionTurnContext } from './structured-agent-session-turns' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -21,8 +16,10 @@ const IDENTITY: AgentSessionJournalIdentity = { } let root: string | null = null +const journals = createTrackedJournalOpener() afterEach(async () => { + await journals.closeAll() if (root) { await rm(root, { recursive: true, force: true }) root = null @@ -32,7 +29,7 @@ afterEach(async () => { describe('performCancel', () => { it('acknowledges only the request and leaves the running lifecycle row intact', async () => { root = await mkdtemp(join(tmpdir(), 'orca-turn-cancel-')) - const journal = await openAgentSessionJournal({ identity: IDENTITY, journalDir: root }) + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) const lifecycleIdentity = { provider: 'legacy' as const, agent: 'codex' as const, @@ -77,190 +74,3 @@ describe('performCancel', () => { ]) }) }) - -describe('performSend lifecycle capacity', () => { - it('refuses before provider contact when dispatch plus terminal capacity cannot fit', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 100 * 1024 } - }) - const dispatch = vi.fn() - const ctx = turnContext(journal, { dispatch } as unknown as StructuredAgentSessionAdapter) - - const result = await performSend(ctx, { - clientMessageId: 'message-1', - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - - expect(result).toMatchObject({ ok: false }) - expect(dispatch).not.toHaveBeenCalled() - expect(journal.submissions()).toEqual([]) - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('binds a synchronous turn start to tentative capacity and releases only on terminality', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 400 * 1024 } - }) - const turnIdentity = { - provider: 'legacy' as const, - agent: 'codex' as const, - sessionId: 'session-1', - recordId: 'turn-lifecycle:turn-1' - } - const dispatch = vi.fn(async () => { - await journal.appendItem( - turnIdentity, - { - kind: 'status', - text: 'Agent is working…', - turnLifecycle: { turnId: 'turn-1', state: 'running' } - }, - { fence: 1 } - ) - return { - state: 'accepted' as const, - providerIdentity: { - provider: 'codex' as const, - threadId: 'thread-1', - turnId: 'turn-1', - ordinal: 0 - } - } - }) - const ctx = turnContext(journal, { dispatch } as unknown as StructuredAgentSessionAdapter) - - const result = await performSend(ctx, { - clientMessageId: 'message-1', - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - - expect(result).toMatchObject({ ok: true }) - expect(journal.lifecycleCapacityState()).toEqual({ - reservedBytes: 128 * 1024, - reservedAppendSlots: 1 - }) - await journal.appendLifecycleBatch({ - settlementId: 'turn-completed:turn-1', - fence: 1, - mutations: [{ kind: 'tombstone', identity: turnIdentity }] - }) - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('keeps response-before-start capacity on the Codex turn lifecycle identity', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 220 * 1024 } - }) - const turnIdentity = { - provider: 'legacy' as const, - agent: 'codex' as const, - sessionId: 'session-1', - recordId: 'turn-lifecycle:turn-1' - } - const ctx = turnContext(journal, { - dispatch: vi.fn(async () => ({ - state: 'accepted' as const, - providerIdentity: { - provider: 'codex' as const, - threadId: 'thread-1', - turnId: 'turn-1', - ordinal: 0 - } - })) - } as unknown as StructuredAgentSessionAdapter) - - await expect( - performSend(ctx, { - clientMessageId: 'message-1', - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - ).resolves.toMatchObject({ ok: true }) - await expect( - journal.appendItem( - turnIdentity, - { - kind: 'status', - text: 'Agent is working…', - turnLifecycle: { turnId: 'turn-1', state: 'running' } - }, - { fence: 1 } - ) - ).resolves.toBeDefined() - expect( - journal - .snapshot() - .items.some( - (item) => - item.body.kind === 'status' && - item.body.turnLifecycle?.turnId === 'turn-1' && - item.body.turnLifecycle.state === 'running' - ) - ).toBe(true) - }) - - it('transfers non-Codex reservations so repeated sends can settle without leaking capacity', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 220 * 1024 } - }) - const dispatch = vi.fn(async ({ clientMessageId }: { clientMessageId: string }) => ({ - state: 'accepted' as const, - providerIdentity: { - provider: 'claude' as const, - sessionId: 'claude-session', - uuid: `turn-${clientMessageId}` - } - })) - const ctx = turnContext(journal, { dispatch } as unknown as StructuredAgentSessionAdapter) - - for (let index = 0; index < 6; index += 1) { - const clientMessageId = `message-${index}` - const result = await performSend(ctx, { - clientMessageId, - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - expect(result).toMatchObject({ ok: true }) - await journal.appendItem( - { - provider: 'claude', - sessionId: 'claude-session', - uuid: `turn-${clientMessageId}` - }, - { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'done' }] }, - { fence: 1 } - ) - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - } - }) -}) - -function turnContext( - journal: Awaited>, - adapter: StructuredAgentSessionAdapter -): AgentSessionTurnContext { - return { - sessionId: 'session-1', - journal, - fence: 1, - adapter, - persistOptions: async () => undefined, - resolvedBy: 'client-1', - publish: vi.fn(), - now: () => 1 - } -} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts index 0c8f43efe1b..f1717027b8b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts @@ -7,7 +7,6 @@ // turn the provider already accepted. import type { AgentJournalMessageItem } from '../../../shared/agent-session-journal-types' -import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' import type { AgentSessionCancelResult, AgentSessionSendResult, @@ -18,14 +17,6 @@ import type { AgentSessionDispatchOutcome, StructuredAgentSessionAdapter } from './structured-agent-session-adapter' -import { - dispatchReservationId, - JOURNAL_DISPATCH_RESERVATION_BYTES, - JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - lifecycleReservationIdForItem, - tentativeTurnReservationId -} from '../agent-session-journal/journal-lifecycle-capacity' - export { performSetOption } from './structured-agent-session-turns-options' export { performPrompt } from './structured-agent-session-turns-prompt' @@ -104,42 +95,8 @@ export async function performSend( } } if (!(input.retryUnknown && existing?.dispatchState === 'unknown')) { - const dispatchReservation = dispatchReservationId(input.clientMessageId) - const tentativeReservation = tentativeTurnReservationId(input.clientMessageId) - const dispatchReserved = await ctx.journal.reserveLifecycleCapacity({ - id: dispatchReservation, - bytes: JOURNAL_DISPATCH_RESERVATION_BYTES, - appendSlots: 1 - }) - const turnReserved = - dispatchReserved && - (await ctx.journal.reserveLifecycleCapacity({ - id: tentativeReservation, - bytes: JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - appendSlots: 1 - })) - if (!dispatchReserved || !turnReserved) { - await ctx.journal.releaseLifecycleCapacity(dispatchReservation) - await ctx.journal.releaseLifecycleCapacity(tentativeReservation) - return invalid('The session does not have enough durable capacity to start another turn.') - } - try { - await ctx.journal.appendSubmission({ ...input, fence: ctx.fence }) - } catch (error) { - await ctx.journal.releaseLifecycleCapacity(dispatchReservation) - await ctx.journal.releaseLifecycleCapacity(tentativeReservation) - throw error - } + await ctx.journal.appendSubmission({ ...input, fence: ctx.fence }) ctx.publish() - } else { - const retryReserved = await ctx.journal.reserveLifecycleCapacity({ - id: dispatchReservationId(input.clientMessageId), - bytes: JOURNAL_DISPATCH_RESERVATION_BYTES, - appendSlots: 1 - }) - if (!retryReserved) { - return invalid('The session does not have enough durable capacity to retry this turn.') - } } const outcome = await dispatchSafely(ctx, input.clientMessageId, input.body) @@ -161,7 +118,7 @@ export async function performSend( ) } catch (error) { // A failed resolution must not strand a pending row; an unknown result is - // explicitly replayable and keeps tentative capacity for that retry. + // explicitly replayable. try { await ctx.journal.resolveDispatch({ clientMessageId: input.clientMessageId, @@ -171,34 +128,11 @@ export async function performSend( recovered: true }) } catch { - await ctx.journal.releaseLifecycleCapacity(dispatchReservationId(input.clientMessageId)) + // Nothing further to record; the pending row is settled on the next attach. } ctx.publish() throw error } - if (outcome.state === 'accepted') { - // Codex publishes its running lifecycle row under the legacy turn identity, - // while the dispatch response identifies the user's message item. Bind the - // tentative turn reservation to the lifecycle identity so a response that - // wins the race with turn/started cannot strand that row at the quota edge. - const reservationTarget = - outcome.providerIdentity.provider === 'codex' - ? { - provider: 'legacy' as const, - agent: 'codex' as const, - sessionId: ctx.sessionId, - recordId: `turn-lifecycle:${outcome.providerIdentity.turnId}` - } - : outcome.providerIdentity - await ctx.journal.transferLifecycleCapacity( - tentativeTurnReservationId(input.clientMessageId), - lifecycleReservationIdForItem( - ctx.journal.canonicalItemId(agentJournalItemKey(reservationTarget)) - ) - ) - } else if (outcome.state === 'rejected') { - await ctx.journal.releaseLifecycleCapacity(tentativeTurnReservationId(input.clientMessageId)) - } ctx.publish() const submission = ctx.journal diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts index f1b127c3fd4..fcd5625a48a 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts @@ -9,8 +9,13 @@ import type { import type { AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' import { REMOTE_RUNTIME_MAX_OUTBOUND_JSON_BYTES } from '../../../shared/remote-runtime-memory-limits' import { mobileE2EETextPayloadAdmissionBytes } from '../../runtime/rpc/mobile-e2ee-outbound-admission' +import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { insertJournalRow } from '../agent-session-journal/journal-row-table' +import type { JournalRow } from '../agent-session-journal/journal-row-schema' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { readAgentSessionHistory } from './agent-session-history-page' import { AgentSessionSubscribers } from './structured-agent-session-subscribers' @@ -19,10 +24,11 @@ const LARGE_TEXT = 'x'.repeat(250 * 1024) let root: string let journal: AgentSessionJournal +const journals = createTrackedJournalOpener() beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-wire-admission-')) - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -30,8 +36,7 @@ beforeEach(async () => { agent: 'codex', providerHandle: { kind: 'codex', threadId: 'thread-1' } }, - journalDir: root, - autoCompact: false + journalDir: root }) for (let ordinal = 1; ordinal <= 20; ordinal += 1) { await journal.appendItem(item(ordinal), body(`${ordinal}:${LARGE_TEXT}`), { fence: 1 }) @@ -39,6 +44,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -91,23 +97,16 @@ describe('structured agent-session outbound admission', () => { expect(epochHistory).toMatchObject({ ok: false, reset: 'epoch_changed' }) expectAdmitted(epochHistory) - await journal.compact(Date.now() + 1, { minTailRows: 0, retainTailMs: 0 }) - const compactedReset: AgentSessionSubscribeEvent[] = [] - subscribers.open({ - id: 'compacted', - sessionId: SESSION, - journal, - fence: 2, - cursor: { epoch: journal.epoch, sequence: 0 }, - emit: (event) => compactedReset.push(event) - }) - expect(compactedReset[0]).toMatchObject({ type: 'reset', reset: 'cursor_compacted' }) - expectAdmitted(compactedReset[0]) - - const history = readAgentSessionHistory(journal, { + // The store can no longer produce a `cursor_compacted` reset — with no row + // shedding inside an epoch, `oldestSequence` is always 1. The reset reason + // stays in the wire vocabulary through the over-budget page path, which is + // where this file's subject — is such a frame admitted outbound? — now lives. + const cursorBefore = journal.cursor() + const overBudget = await reopenWithOversizedRemoval(cursorBefore.sequence) + const history = readAgentSessionHistory(overBudget, { sessionId: SESSION, direction: 'after', - cursor: { epoch: journal.epoch, sequence: 0 } + cursor: cursorBefore }) expect(history).toMatchObject({ ok: false, reset: 'cursor_compacted' }) expectAdmitted(history) @@ -156,3 +155,42 @@ function item(ordinal: number): AgentJournalItemIdentity { function body(text: string): AgentJournalItemBody { return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text }] } } + +/** Stages a pre-bounding oversized removal id — the one remaining producer of a + * `cursor_compacted` reset — straight into the session database. */ +async function reopenWithOversizedRemoval(afterSequence: number): Promise { + const hugeItemId = `codex:thread-1:${'h'.repeat(5 * 1024 * 1024)}:1` + const base = { v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, epoch: journal.epoch, fence: 1, ts: 1 } + const rows: JournalRow[] = [ + { + ...base, + kind: 'item', + itemId: hugeItemId, + revision: 1, + seq: afterSequence + 1, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'big' }] } + }, + { ...base, kind: 'tombstone', itemId: hugeItemId, revision: 2, seq: afterSequence + 2 } + ] + await journal.close() + const opened = openJournalDatabase(journalDatabaseFile(root)) + try { + opened.db.exec('BEGIN IMMEDIATE') + for (const row of rows) { + insertJournalRow(opened.db, SESSION, row) + } + opened.db.exec('COMMIT') + } finally { + opened.db.close() + } + return journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + journalDir: root + }) +} diff --git a/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts b/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts index 956e8a0eaa8..6e4cf64b5bf 100644 --- a/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts @@ -3,9 +3,11 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { StructuredTuiTranscriptCatchup } from './structured-tui-transcript-catchup' +const journals = createTrackedJournalOpener() + const NOW = 1_800_000_000_000 const SESSION = 'session-catchup' const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' @@ -27,6 +29,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -83,7 +86,7 @@ async function createCatchupFixture() { }, now: NOW }) - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', diff --git a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts index d287838a13c..e389b30aba2 100644 --- a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts +++ b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts @@ -8,12 +8,7 @@ describe('unhandled provider frame journal fallback', () => { 'future-provider', 'notification:new/event', { body: 'abcdefghij' }, - { - inlineHeadBytes: 8, - maxSessionBytes: 1024, - maxAppendsPerWindow: 10, - appendWindowMs: 1000 - } + { inlineHeadBytes: 8 } ) expect(item).not.toBeNull() @@ -32,12 +27,6 @@ describe('unhandled provider frame journal fallback', () => { expect( Buffer.byteLength(item.body.providerFrame?.payload.head ?? '', 'utf8') ).toBeLessThanOrEqual(8) - expect(item.blobs).toEqual([ - { - digest: item.body.providerFrame?.payload.digest, - payload: '{"body":"abcdefghij"}' - } - ]) }) it('turns an unserializable message-shaped payload into an explicit visible value', () => { @@ -198,12 +187,7 @@ describe('unhandled provider frame journal fallback', () => { 'codex', 'notification:warning', { message }, - { - inlineHeadBytes: 8, - maxSessionBytes: 1024, - maxAppendsPerWindow: 10, - appendWindowMs: 1000 - } + { inlineHeadBytes: 8 } ) expect(row?.body.text).toContain('abcdefgh') diff --git a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts index b4651cfc952..60f34707ac9 100644 --- a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts +++ b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts @@ -9,7 +9,6 @@ import { classifyProviderFrame } from './provider-frame-disposition' export type UnhandledProviderFrameJournalItem = { body: AgentJournalStatusItem - blobs: { digest: string; payload: string }[] /** Why the frame surfaced. Error frames are exempt from generic-row caps. */ classification: 'timeline-substantive' | 'error-surface' } @@ -55,7 +54,8 @@ function directReadableMessage(payload: unknown): string | null { return null } -function readableMessage(payload: unknown): string | null { +/** The provider's own sentence for a frame, when it carries one. */ +export function readableProviderFrameText(payload: unknown): string | null { const direct = directReadableMessage(payload) if (direct || typeof payload !== 'object' || payload === null || Array.isArray(payload)) { return direct @@ -90,7 +90,7 @@ export function unhandledProviderFrameJournalItem( // Why: the opcode alone ("codex · notification:warning") tells the user nothing // and reads as protocol noise. Lead with the provider's own sentence when it has // one; the raw frame stays behind the row's disclosure either way. - const message = readableMessage(payload) + const message = readableProviderFrameText(payload) const display = message ? boundInlineText(message, limits) : null return { body: { @@ -98,7 +98,6 @@ export function unhandledProviderFrameJournalItem( text: display?.text ?? `${provider} · ${kind}`, providerFrame: { provider, kind, payload: bounded } }, - blobs: bounded.truncated ? [{ digest: bounded.digest, payload: serialized }] : [], classification: classification === 'error-surface' ? 'error-surface' : 'timeline-substantive' } } diff --git a/src/main/native-chat/claude-structured-managed-account-support.test.ts b/src/main/native-chat/claude-structured-managed-account-support.test.ts new file mode 100644 index 00000000000..f647579d03e --- /dev/null +++ b/src/main/native-chat/claude-structured-managed-account-support.test.ts @@ -0,0 +1,154 @@ +import { describe, expect, it } from 'vitest' +import { getSelectedClaudeAccountIdForTarget } from '../claude-accounts/runtime-selection' +import { + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from './claude-structured-managed-account-support' + +function account(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +function settings( + overrides: Partial +): ClaudeManagedAccountGateSettings { + return { claudeManagedAccounts: [], activeClaudeManagedAccountId: null, ...overrides } +} + +describe('structuredClaudeMatchesActiveManagedAccount', () => { + it('allows an unmanaged install, where nothing claims an identity', () => { + expect(structuredClaudeMatchesActiveManagedAccount(settings({}))).toBe(true) + }) + + it('allows a selected host account, which the runtime syncs into the ambient config', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } + }) + ) + ).toBe(true) + }) + + it('refuses a WSL-only managed account, which never reaches the ambient config', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } + }) + ) + ).toBe(false) + }) + + it('refuses when a host selection names an account that is WSL-bound or missing', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountIdsByRuntime: { host: 'wsl-1', wsl: {} } + }) + ) + ).toBe(false) + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountIdsByRuntime: { host: 'gone', wsl: {} } + }) + ) + ).toBe(false) + }) + + /** Absent and empty are the same answer: this user has no managed Claude accounts, so nothing + * claims an identity and the ambient path is legitimate. Only settings that cannot be READ are + * unknown. Treating a missing key as unknown strands profiles that simply never wrote it — the + * auth policy's own predicate takes `(accounts ?? [])` for exactly this reason. */ + it('treats an absent account list the same as an empty one', () => { + expect( + structuredClaudeMatchesActiveManagedAccount(settings({ claudeManagedAccounts: [] })) + ).toBe(true) + expect( + structuredClaudeMatchesActiveManagedAccount({ + activeClaudeManagedAccountId: null + } as unknown as ClaudeManagedAccountGateSettings) + ).toBe(true) + }) + + it('fails closed when the settings cannot be read at all', () => { + expect(structuredClaudeMatchesActiveManagedAccount(null)).toBe(false) + expect(structuredClaudeMatchesActiveManagedAccount(undefined)).toBe(false) + }) + + /** The four states this gate exists to tell apart, pinned together so a change to one is visible + * against the others. */ + it.each([ + ['no managed accounts', [], null, true], + ['accounts present, none active, no WSL account', [account('host-1', 'host')], null, true], + ['host account selected', [account('host-1', 'host')], 'host-1', true], + ['WSL-only, normalized to no host selection', [account('wsl-1', 'wsl')], null, false] + ] as const)('resolves %s', (_name, claudeManagedAccounts, activeId, expected) => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [...claudeManagedAccounts], + activeClaudeManagedAccountIdsByRuntime: { host: activeId, wsl: {} } + }) + ) + ).toBe(expected) + }) + + /** THE discriminator, and the whole of this rule. With nothing selected for the host runtime the + * settings alone cannot distinguish honest deselection from the WSL-only steady state, because + * `pruneInvalidClaudeRuntimeSelection` empties the host slot in the second case and persists it. + * So the presence of ANY WSL-bound account decides. Simplifying this to "none active -> + * supported" re-opens the auth-identity misrepresentation this gate exists to prevent. */ + it('splits none-active on whether a WSL-bound account exists at all', () => { + const noneActive = (accounts: ReturnType[]) => + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: accounts, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } + }) + ) + + expect(noneActive([account('host-1', 'host')])).toBe(true) + expect(noneActive([account('host-1', 'host'), account('host-2', 'host')])).toBe(true) + expect(noneActive([account('wsl-1', 'wsl')])).toBe(false) + // Mixed list still refuses: the WSL account is present and nothing is selected. + expect(noneActive([account('host-1', 'host'), account('wsl-1', 'wsl')])).toBe(false) + }) + + /** The gate and the auth policy must resolve the SAME account. A legacy settings blob carries the + * selection only in the flat `activeClaudeManagedAccountId`, which is where the accessor's + * fall-through lives — reading the runtime map directly silently disagrees with the policy. */ + it('resolves the same account as the auth policy on a legacy flat selection', () => { + const legacy = settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1' + }) + + expect(getSelectedClaudeAccountIdForTarget(legacy, { runtime: 'host' })).toBe('host-1') + expect(structuredClaudeMatchesActiveManagedAccount(legacy)).toBe(true) + }) + + it('agrees with the auth policy that a legacy flat WSL selection is refused', () => { + const legacy = settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountId: 'wsl-1' + }) + + expect(getSelectedClaudeAccountIdForTarget(legacy, { runtime: 'host' })).toBe('wsl-1') + expect(structuredClaudeMatchesActiveManagedAccount(legacy)).toBe(false) + }) +}) diff --git a/src/main/native-chat/claude-structured-managed-account-support.ts b/src/main/native-chat/claude-structured-managed-account-support.ts new file mode 100644 index 00000000000..dccf6216bda --- /dev/null +++ b/src/main/native-chat/claude-structured-managed-account-support.ts @@ -0,0 +1,61 @@ +import type { GlobalSettings } from '../../shared/global-settings-types' +import { getSelectedClaudeAccountIdForTarget } from '../claude-accounts/runtime-selection' + +export type ClaudeManagedAccountGateSettings = Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' +> + +/** + * A structured Claude session launches against the ambient Claude config, which the account service + * keeps in sync with the selected HOST account. A WSL-bound managed account lives inside the distro + * and is never synced there, so such a session would authenticate as whatever the ambient identity + * happens to be while the UI names the WSL account — the user is told one identity and given + * another. Refuse the structured path there and let the terminal-backed one, which resolves the + * account per runtime, handle that account shape. + * + * Reads the selection through the same accessor the auth policy uses. Resolving it any other way + * lets the two disagree, and a session admitted by this gate would then run under a policy computed + * from a different account than the one approved here. + * + * Unknown answers refuse, and only genuinely unknown ones: settings that cannot be read at all, or + * an active selection this cannot resolve. An install with no managed accounts — the list empty or + * never written — claims no identity and is fine. + */ +export function structuredClaudeMatchesActiveManagedAccount( + settings: ClaudeManagedAccountGateSettings | null | undefined +): boolean { + if (!settings) { + return false + } + // Absent is the same answer as empty — this user has no managed Claude accounts, so nothing + // claims an identity and ambient auth is the truth. Only settings that cannot be READ are + // unknown, and those refuse above. The auth policy reads the list the same way. + const accounts = settings.claudeManagedAccounts ?? [] + if (accounts.length === 0) { + return true + } + const activeHostId = getSelectedClaudeAccountIdForTarget(settings, { runtime: 'host' }) + if (!activeHostId) { + // Nothing selected for the host runtime is two different states that the settings cannot tell + // apart after the fact: honest deselection, where ambient auth is the truth and the UI names no + // identity, and the WSL-only case, where the prune emptied the host slot and persisted null + // while the UI still names the WSL account. The presence of any WSL-bound account decides. + return !accounts.some((candidate) => candidate.managedAuthRuntime === 'wsl') + } + const active = accounts.find((candidate) => candidate.id === activeHostId) + return active ? active.managedAuthRuntime !== 'wsl' : false +} + +/** Reads the gate's settings, answering null when they cannot be read so callers refuse. */ +export function readClaudeManagedAccountGateSettings( + getSettings: () => ClaudeManagedAccountGateSettings +): ClaudeManagedAccountGateSettings | null { + try { + return getSettings() + } catch { + return null + } +} diff --git a/src/main/native-chat/session-file-resolver-claude-roots.test.ts b/src/main/native-chat/session-file-resolver-claude-roots.test.ts new file mode 100644 index 00000000000..87ebd570280 --- /dev/null +++ b/src/main/native-chat/session-file-resolver-claude-roots.test.ts @@ -0,0 +1,97 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const scanned = vi.hoisted(() => ({ dirs: [] as string[], hits: {} as Record })) +vi.mock('../ai-vault/session-scanner-discovery', () => ({ + walkSessionFiles: async (dir: string) => { + scanned.dirs.push(dir) + const hit = scanned.hits[dir] + return hit ? [hit] : [] + } +})) + +import { homedir } from 'node:os' +import { join } from 'node:path' +import { resolveSessionFilePath } from './session-file-resolver' + +const DEFAULT_ROOT = join(homedir(), '.claude', 'projects') +const CONFIG_DIR = '/opt/claude-home' +const CONFIG_ROOT = join(CONFIG_DIR, 'projects') + +let previousConfigDir: string | undefined + +beforeEach(() => { + previousConfigDir = process.env.CLAUDE_CONFIG_DIR + scanned.dirs = [] + scanned.hits = {} +}) + +afterEach(() => { + if (previousConfigDir === undefined) { + delete process.env.CLAUDE_CONFIG_DIR + } else { + process.env.CLAUDE_CONFIG_DIR = previousConfigDir + } +}) + +/** + * Honouring CLAUDE_CONFIG_DIR fixed new sessions but would otherwise hide every + * transcript written before the user adopted the variable. The Codex resolver in this + * same file already searches managed-then-default and de-dupes; Claude does the same. + */ +describe('claude transcript roots', () => { + it('searches the config-dir root first, then the default home', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([CONFIG_ROOT, DEFAULT_ROOT]) + }) + + it('still finds history written before CLAUDE_CONFIG_DIR was adopted', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + const legacy = join(DEFAULT_ROOT, '-repos-old', 'session-1.jsonl') + scanned.hits[DEFAULT_ROOT] = legacy + + await expect(resolveSessionFilePath('claude', 'session-1')).resolves.toBe(legacy) + }) + + it('prefers the config-dir root when both hold the session', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + scanned.hits[CONFIG_ROOT] = join(CONFIG_ROOT, '-repos-new', 'session-1.jsonl') + scanned.hits[DEFAULT_ROOT] = join(DEFAULT_ROOT, '-repos-old', 'session-1.jsonl') + + await expect(resolveSessionFilePath('claude', 'session-1')).resolves.toBe( + scanned.hits[CONFIG_ROOT] + ) + // The default root is never reached, so the common case pays for one scan. + expect(scanned.dirs).toEqual([CONFIG_ROOT]) + }) + + it('scans one root when the variable is unset', async () => { + delete process.env.CLAUDE_CONFIG_DIR + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([DEFAULT_ROOT]) + }) + + it('de-dupes when CLAUDE_CONFIG_DIR names the default home', async () => { + process.env.CLAUDE_CONFIG_DIR = join(homedir(), '.claude') + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([DEFAULT_ROOT]) + }) + + it('honours an explicit root override without adding fallbacks', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + // The account-home callers (structured-claude-runtime-adapter, the host handoff) + // know the exact tree their session pinned; a fallback there could resolve a + // different account's transcript. + await resolveSessionFilePath('claude', 'session-1', { + claudeProjectsDir: '/accounts/pinned/projects' + }) + + expect(scanned.dirs).toEqual(['/accounts/pinned/projects']) + }) +}) diff --git a/src/main/native-chat/session-file-resolver.test.ts b/src/main/native-chat/session-file-resolver.test.ts index 584d8a25a9d..04946f5b464 100644 --- a/src/main/native-chat/session-file-resolver.test.ts +++ b/src/main/native-chat/session-file-resolver.test.ts @@ -1,9 +1,12 @@ import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { dirname, join } from 'node:path' -import { afterEach, describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' -import { ClaudeTranscriptTailIncompleteError } from '../claude/claude-transcript-branch-proof' +import { + ClaudeTranscriptTailIncompleteError, + readClaudeTranscriptLeafWithReproof +} from '../claude/claude-transcript-branch-proof' import { readClaudeTranscriptLeafUuid, resolveSessionFilePath } from './session-file-resolver' let tempRoots: string[] = [] @@ -139,6 +142,263 @@ describe('resolveSessionFilePath', () => { ) }) + it('rejects non-transcript and sidechain UUIDs as the durable leaf', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-leaf-filter-') + const transcript = join(root, 'session.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'sidechain-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + isSidechain: true + }, + { type: 'result', uuid: 'result-frame', parentUuid: 'main-user', sessionId: 'session-1' }, + { + type: 'system', + subtype: 'init', + uuid: 'init-frame', + parentUuid: null, + sessionId: 'session-1' + }, + { type: 'stream_event', uuid: 'stream-frame', parentUuid: null, sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'sidechain-assistant', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'marker leaf is missing from the session graph' + ) + }) + + it('rejects a main leaf whose ancestry crosses a subagent sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-sidechain-ancestry-') + const transcript = join(root, 'session.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'sidechain-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + isSidechain: true + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'sidechain-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'not on the main transcript' + ) + }) + + it('rejects a main leaf whose ancestry crosses a parent-tool-use sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-ancestry-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'not on the main transcript' + ) + }) + + it('rejects a previous cursor descended from a parent-tool-use sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-cursor-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect( + readClaudeTranscriptLeafUuid(transcript, 'session-1', 'main-after-sidechain') + ).rejects.toThrow('not on the main transcript') + }) + + it('rejects a latest marker descended from a parent-tool-use cursor sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-cursor-descendant-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { + type: 'assistant', + uuid: 'latest-after-sidechain', + parentUuid: 'main-after-sidechain', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'latest-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect( + readClaudeTranscriptLeafUuid(transcript, 'session-1', 'main-after-sidechain') + ).rejects.toThrow('not on the main transcript') + }) + + it('rejects a post-snapshot descendant whose parent row was observed later', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-post-snapshot-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { + type: 'assistant', + uuid: 'descendant', + parentUuid: 'previous', + sessionId: 'session-1' + }, + { type: 'assistant', uuid: 'previous', parentUuid: null, sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'descendant', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1', 'previous')).rejects.toThrow( + 'parent row follows descendant' + ) + }) + + it('does not re-prove a divergent sibling after the sampled cursor rejects', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-sibling-reproof-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'root', parentUuid: null, sessionId: 'session-1' }, + { type: 'assistant', uuid: 'old', parentUuid: 'root', sessionId: 'session-1' }, + { type: 'assistant', uuid: 'new', parentUuid: 'root', sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'new', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + const calls: (string | null)[] = [] + const readTranscriptLeaf = async ({ + previousLeafUuid + }: { + previousLeafUuid: string | null + }) => { + calls.push(previousLeafUuid) + return readClaudeTranscriptLeafUuid(transcript, 'session-1', previousLeafUuid) + } + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1', 'old')).rejects.toThrow( + 'sibling branch' + ) + + await expect( + readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf, + claudeConfigDir: '/accounts/claude', + providerSessionId: 'session-1', + previousLeafUuid: 'old' + }) + ).rejects.toThrow('sibling branch') + expect(calls).toEqual(['old']) + }) + + it('does not accept a divergent sibling after a truncated-tail reproof', async () => { + const calls: (string | null)[] = [] + const readTranscriptLeaf = vi.fn( + async ({ previousLeafUuid }: { previousLeafUuid: string | null }) => { + calls.push(previousLeafUuid) + if (calls.length === 1) { + throw new ClaudeTranscriptTailIncompleteError() + } + return 'divergent-sibling' + } + ) + + await expect( + readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf, + claudeConfigDir: '/accounts/claude', + providerSessionId: 'session-1', + previousLeafUuid: 'old' + }) + ).rejects.toBeInstanceOf(ClaudeTranscriptTailIncompleteError) + expect(calls).toEqual(['old']) + }) + it('globs Claude project subdirs for .jsonl', async () => { const root = await makeRoot('orca-native-chat-resolve-claude-') const claudeProjectsDir = join(root, 'claude-projects') @@ -410,3 +670,42 @@ describe('resolveSessionFilePath', () => { expect(resolved).toBe(target) }) }) + +// Mobile native chat resolves with no root override (transcript-read-cache.ts:104), +// while the account home a structured Claude session pins is +// `CLAUDE_CONFIG_DIR || ~/.claude` (runtime-paths.ts:15). When the two disagree the +// CLI writes one place and mobile reads another, and the chat goes dark with no +// wire-level error — so the default root has to honour the same variable. +describe('the default Claude transcript root mobile falls back to', () => { + it('follows CLAUDE_CONFIG_DIR, the same variable the pinned account home follows', async () => { + const configDir = await makeRoot('orca-native-chat-claude-config-dir-') + const slugDir = join(configDir, 'projects', '-repos-workspace-1') + await mkdir(slugDir, { recursive: true }) + const transcript = join(slugDir, 'session-under-config-dir.jsonl') + await writeFile(transcript, '', 'utf8') + const previous = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = configDir + + try { + // No `claudeProjectsDir` override: exactly the call mobile makes. + await expect(resolveSessionFilePath('claude', 'session-under-config-dir')).resolves.toBe( + transcript + ) + } finally { + restoreEnv('CLAUDE_CONFIG_DIR', previous) + } + }) + + it('ignores a blank CLAUDE_CONFIG_DIR rather than resolving against the filesystem root', async () => { + const previous = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = ' ' + + try { + await expect( + resolveSessionFilePath('claude', 'session-that-does-not-exist') + ).resolves.toBeNull() + } finally { + restoreEnv('CLAUDE_CONFIG_DIR', previous) + } + }) +}) diff --git a/src/main/native-chat/session-file-resolver.ts b/src/main/native-chat/session-file-resolver.ts index 0ee73fddc8d..12d2e615742 100644 --- a/src/main/native-chat/session-file-resolver.ts +++ b/src/main/native-chat/session-file-resolver.ts @@ -31,8 +31,20 @@ import { proveClaudeTranscriptBranch } from '../claude/claude-transcript-branch- // the remote main resolves its local home, so we never hardcode an absolute // user path — homedir()/CODEX_HOME resolution stays runtime-relative and is // computed per call (not at module load) so it tracks the live home. -function claudeProjectsDir(): string { - return join(homedir(), '.claude', 'projects') +// Why CLAUDE_CONFIG_DIR and not just homedir(): a structured Claude session pins its +// account home to `CLAUDE_CONFIG_DIR || ~/.claude` (claude-accounts/runtime-paths.ts), +// and the CLI writes its transcript under whatever home it was given. Mobile native chat +// resolves with no root override, so a default that ignored the variable read a different +// tree than the CLI wrote — a silent blackout, not an error. +// Why both roots and not just that one: adopting the variable would otherwise hide every +// transcript written before it was set. Same managed-then-default shape as +// codexSessionsDirs below, de-duped so the usual case still scans once. +function claudeProjectsDirs(): string[] { + const candidates = [ + join(process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude'), 'projects'), + join(homedir(), '.claude', 'projects') + ] + return candidates.filter((dir, index) => candidates.indexOf(dir) === index) } // Why: Orca launches Codex with ORCA_CODEX_HOME pointing at its own managed @@ -173,9 +185,11 @@ async function resolveSessionFileById( } if (transcriptAgent === 'claude') { + // An explicit root is the caller naming the exact account tree its session pinned; + // adding a fallback there could resolve a different account's transcript. return resolveClaudeSessionFile( trimmedId, - options.claudeProjectsDir ?? claudeProjectsDir(), + options.claudeProjectsDir ? [options.claudeProjectsDir] : claudeProjectsDirs(), signal ) } @@ -205,16 +219,22 @@ async function resolveSessionFileById( async function resolveClaudeSessionFile( sessionId: string, - projectsDir: string, + projectsDirs: readonly string[], signal?: AbortSignal ): Promise { const targetName = `${sessionId}.jsonl` - const files = await walkSessionFiles(projectsDir, 'claude', [], { - extensions: new Set(['.jsonl']), - filePredicate: (path) => basename(path) === targetName, - signal - }) - return files[0] ?? null + for (const projectsDir of projectsDirs) { + // No existence pre-check: walkSessionFiles already yields [] for a missing root. + const files = await walkSessionFiles(projectsDir, 'claude', [], { + extensions: new Set(['.jsonl']), + filePredicate: (path) => basename(path) === targetName, + signal + }) + if (files[0]) { + return files[0] + } + } + return null } async function resolveCodexSessionFile( diff --git a/src/main/native-chat/structured-agent-session-create-support.test.ts b/src/main/native-chat/structured-agent-session-create-support.test.ts new file mode 100644 index 00000000000..ab19d369bac --- /dev/null +++ b/src/main/native-chat/structured-agent-session-create-support.test.ts @@ -0,0 +1,83 @@ +import { describe, expect, it } from 'vitest' +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import type { ClaudeManagedAccountGateSettings } from './claude-structured-managed-account-support' +import { resolveStructuredAgentSessionCreateSupport } from './structured-agent-session-create-support' + +const LOCAL: AgentSessionExecutionLocation = { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' +} + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +const WSL_ONLY: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +function support( + overrides: Partial[0]> = {} +) { + return resolveStructuredAgentSessionCreateSupport({ + agent: 'claude', + location: LOCAL, + adapterSupportsCreate: true, + getSettings: () => HOST_SELECTED, + ...overrides + }) +} + +describe('resolveStructuredAgentSessionCreateSupport', () => { + it('supports Claude under a selected host account', () => { + expect(support()).toEqual({ supported: true }) + }) + + it('refuses Claude under a WSL-only managed account', () => { + expect(support({ getSettings: () => WSL_ONLY })).toEqual({ supported: false, reason: 'wsl' }) + }) + + it('fails closed for Claude when the settings throw', () => { + expect( + support({ + getSettings: () => { + throw new Error('no store') + } + }) + ).toEqual({ supported: false, reason: 'wsl' }) + }) + + it('leaves Codex to the adapter answer under the same WSL-only account', () => { + expect(support({ agent: 'codex', getSettings: () => WSL_ONLY })).toEqual({ supported: true }) + }) + + it.each([ + ['remote', { ...LOCAL, executionHostId: 'ssh:host-a' }, 'remote'], + ['wsl workspace', { ...LOCAL, wslDistro: 'Ubuntu' }, 'wsl'], + ['unsupported agent', LOCAL, 'agent'] + ] as const)('keeps the adapter refusal reason for %s', (_name, location, reason) => { + expect(support({ adapterSupportsCreate: false, location })).toEqual({ + supported: false, + reason + }) + }) +}) diff --git a/src/main/native-chat/structured-agent-session-create-support.ts b/src/main/native-chat/structured-agent-session-create-support.ts new file mode 100644 index 00000000000..9b96a1af4be --- /dev/null +++ b/src/main/native-chat/structured-agent-session-create-support.ts @@ -0,0 +1,48 @@ +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { + readClaudeManagedAccountGateSettings, + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from './claude-structured-managed-account-support' + +export type StructuredAgentSessionCreateSupport = { + supported: boolean + reason?: 'agent' | 'remote' | 'wsl' +} + +/** + * The create-support verdict, kept out of the runtime class file because that file is `@ts-nocheck` + * — a call site there is not typechecked, so an auth-identity decision written inline would compile + * however wrong it was. The runtime hands over the two facts it owns and this decides. + */ +export function resolveStructuredAgentSessionCreateSupport(input: { + agent: 'claude' | 'codex' + location: AgentSessionExecutionLocation + adapterSupportsCreate: boolean + getSettings: () => ClaudeManagedAccountGateSettings +}): StructuredAgentSessionCreateSupport { + if (!input.adapterSupportsCreate) { + return { + supported: false, + reason: + input.location.executionHostId !== LOCAL_EXECUTION_HOST_ID + ? 'remote' + : input.location.wslDistro + ? 'wsl' + : 'agent' + } + } + // Claude only: Codex resolves its account on a different path, so its answer is untouched here. + // `wsl` is the closest existing reason — the cause is a WSL-bound account rather than a WSL + // workspace — and no client reads the field, so it stays as-is. + if ( + input.agent === 'claude' && + !structuredClaudeMatchesActiveManagedAccount( + readClaudeManagedAccountGateSettings(input.getSettings) + ) + ) { + return { supported: false, reason: 'wsl' } + } + return { supported: true } +} diff --git a/src/main/orca-chromium-process-pids.ts b/src/main/orca-chromium-process-pids.ts index b2613e42b79..3282f3a4d66 100644 --- a/src/main/orca-chromium-process-pids.ts +++ b/src/main/orca-chromium-process-pids.ts @@ -12,6 +12,12 @@ import { recordCoalescedDurableCrashBreadcrumb } from './crash-reporting/durable * Empty on a Node host and empty on failure: that is "no refusal proven", never * "safe to kill" — callers must keep every other guard they already have. * + * The other direction is real too, and bounded by design: `getAppMetrics()` can + * still list a renderer Electron has not finished reaping, so on Windows a pid + * already recycled onto an unrelated child of ours reads as `own` and its tree + * walk is refused. That is why a refusal only blocks the pid-addressed walk and + * every gated site still kills its own root through the child handle. + * * Why failure stays open rather than refusing everything: a refusal is not free. * `terminateWindowsProcessTree` resolves without killing, and * `killSourceControlAgentProcess` returns that straight to a caller that then diff --git a/src/main/orca-profiles/profile-cloud-client.ts b/src/main/orca-profiles/profile-cloud-client.ts index e7657bbdb87..5893c8d109a 100644 --- a/src/main/orca-profiles/profile-cloud-client.ts +++ b/src/main/orca-profiles/profile-cloud-client.ts @@ -159,19 +159,37 @@ function normalizeSessionResponse(value: unknown): OrcaCloudSessionExchangeRespo const CLOUD_REQUEST_TIMEOUT_MS = 30_000 -async function postJson(url: string, body: unknown, accessToken?: string): Promise { +// Why: refresh tokens rotate, so an aborted refresh is ambiguous — the server +// may have rotated ours before the reply was lost, and the only recovery is a +// replay the server reads as reuse. One long attempt beats a short attempt plus +// a replayed retry. +const CLOUD_REFRESH_TIMEOUT_MS = 60_000 + +type PostJsonOptions = { + accessToken?: string + timeoutMs?: number +} + +// Only a status line proves the server rejected the request without consuming +// what was in it. Everything else — an abort, a dropped socket, a 200 we could +// not parse — leaves a rotating credential possibly already spent. +export function isAmbiguousCloudRequestFailure(error: unknown): boolean { + return !(error instanceof OrcaCloudRequestError) +} + +async function postJson(url: string, body: unknown, options?: PostJsonOptions): Promise { const response = await fetch(url, { method: 'POST', headers: { 'content-type': 'application/json', - ...(accessToken ? { authorization: `Bearer ${accessToken}` } : {}) + ...(options?.accessToken ? { authorization: `Bearer ${options.accessToken}` } : {}) }, body: JSON.stringify(body), // Why: these are fixed first-party token endpoints; following a redirect // would re-send refresh tokens/code verifiers to another origin, and a // stalled server must not hang the renderer's awaited IPC call forever. redirect: 'error', - signal: AbortSignal.timeout(CLOUD_REQUEST_TIMEOUT_MS) + signal: AbortSignal.timeout(options?.timeoutMs ?? CLOUD_REQUEST_TIMEOUT_MS) }) if (!response.ok) { await cancelUnreadResponseBody(response) @@ -204,7 +222,7 @@ export async function refreshOrcaCloudCapabilities( cloud?: unknown organizations?: unknown capabilities: unknown - }>(config.capabilitiesEndpoint, {}, session.accessToken) + }>(config.capabilitiesEndpoint, {}, { accessToken: session.accessToken }) return { cloud: response.cloud === undefined ? undefined : normalizeCloudSummary(response.cloud), organizations: normalizeOrganizations(response.organizations), @@ -217,9 +235,11 @@ export async function refreshOrcaCloudSession( session: OrcaCloudSession ): Promise { return normalizeSessionResponse( - await postJson(config.refreshEndpoint, { - refreshToken: session.refreshToken - }) + await postJson( + config.refreshEndpoint, + { refreshToken: session.refreshToken }, + { timeoutMs: CLOUD_REFRESH_TIMEOUT_MS } + ) ) } @@ -235,7 +255,7 @@ export async function createOrcaCloudProfile( orgId: args.orgId, name: args.name }, - session.accessToken + { accessToken: session.accessToken } ) ) } @@ -249,7 +269,7 @@ export async function selectOrcaCloudOrg( cloud: unknown organizations?: unknown capabilities: unknown - }>(config.orgEndpoint, { orgId }, session.accessToken) + }>(config.orgEndpoint, { orgId }, { accessToken: session.accessToken }) return { cloud: normalizeCloudSummary(response.cloud), organizations: normalizeOrganizations(response.organizations), @@ -261,5 +281,9 @@ export async function revokeOrcaCloudSession( config: OrcaCloudAuthConfig, session: OrcaCloudSession ): Promise { - await postJson(config.logoutEndpoint, { refreshToken: session.refreshToken }, session.accessToken) + await postJson( + config.logoutEndpoint, + { refreshToken: session.refreshToken }, + { accessToken: session.accessToken } + ) } diff --git a/src/main/orca-profiles/profile-cloud-refresh-replay-guard.ts b/src/main/orca-profiles/profile-cloud-refresh-replay-guard.ts new file mode 100644 index 00000000000..d5c22f6b527 --- /dev/null +++ b/src/main/orca-profiles/profile-cloud-refresh-replay-guard.ts @@ -0,0 +1,55 @@ +// Refresh tokens whose server-side fate is unknown: the POST left the client but +// no status came back, so the server may already have rotated the token before +// the reply was lost. Sending it again reads as reuse and revokes the whole +// token family — on 2026-09-04 that turned one slow refresh endpoint into 21,605 +// sign-outs, because every caller's retry loop replayed the same stored token. + +// A replay this soon after the ambiguous attempt is a retry loop, not a person +// asking again; holding it back keeps one lost reply from becoming a storm. +export const AMBIGUOUS_REFRESH_REPLAY_DELAY_MS = 30_000 + +type AmbiguousRefreshAttempt = { + refreshToken: string + attemptedAt: number +} + +const ambiguousRefreshAttempts = new Map() + +export class AmbiguousRefreshReplayBlockedError extends Error { + constructor() { + super('orca_cloud_refresh_replay_blocked') + this.name = 'AmbiguousRefreshReplayBlockedError' + } +} + +export function recordAmbiguousRefreshAttempt( + key: string, + refreshToken: string, + now = Date.now() +): void { + ambiguousRefreshAttempts.set(key, { refreshToken, attemptedAt: now }) +} + +// Call once the token's fate is known: it rotated, or the session it belonged to +// is gone. Leaving the record would mislabel a later, unrelated 401. +export function forgetAmbiguousRefreshAttempt(key: string): void { + ambiguousRefreshAttempts.delete(key) +} + +export function wasRefreshTokenAmbiguouslyAttempted(key: string, refreshToken: string): boolean { + return ambiguousRefreshAttempts.get(key)?.refreshToken === refreshToken +} + +export function blocksAmbiguousRefreshReplay( + key: string, + refreshToken: string, + now = Date.now() +): boolean { + const attempt = ambiguousRefreshAttempts.get(key) + if (!attempt || attempt.refreshToken !== refreshToken) { + return false + } + // Why bounded rather than permanent: the token is only *possibly* spent. A + // permanent block would sign out every desktop whose refresh merely timed out. + return now - attempt.attemptedAt < AMBIGUOUS_REFRESH_REPLAY_DELAY_MS +} diff --git a/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts b/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts index 9db1d830986..53b26a75595 100644 --- a/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts +++ b/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts @@ -53,6 +53,7 @@ vi.mock('./profile-cloud-pkce', () => ({ vi.mock('./profile-cloud-client', () => ({ OrcaCloudRequestError: OrcaCloudRequestErrorMock, + isAmbiguousCloudRequestFailure: (error: unknown) => !(error instanceof OrcaCloudRequestErrorMock), createOrcaCloudProfile: createOrcaCloudProfileMock, exchangeOrcaCloudAuthCode: exchangeOrcaCloudAuthCodeMock, refreshOrcaCloudCapabilities: refreshOrcaCloudCapabilitiesMock, diff --git a/src/main/orca-profiles/profile-cloud-service-refresh.test.ts b/src/main/orca-profiles/profile-cloud-service-refresh.test.ts index 075573eea3a..877c74c3c46 100644 --- a/src/main/orca-profiles/profile-cloud-service-refresh.test.ts +++ b/src/main/orca-profiles/profile-cloud-service-refresh.test.ts @@ -51,6 +51,7 @@ vi.mock('./profile-cloud-pkce', () => ({ vi.mock('./profile-cloud-client', () => ({ OrcaCloudRequestError: OrcaCloudRequestErrorMock, + isAmbiguousCloudRequestFailure: (error: unknown) => !(error instanceof OrcaCloudRequestErrorMock), createOrcaCloudProfile: createOrcaCloudProfileMock, exchangeOrcaCloudAuthCode: exchangeOrcaCloudAuthCodeMock, refreshOrcaCloudCapabilities: refreshOrcaCloudCapabilitiesMock, diff --git a/src/main/orca-profiles/profile-cloud-session-invalidation.ts b/src/main/orca-profiles/profile-cloud-session-invalidation.ts new file mode 100644 index 00000000000..a9415e28ac6 --- /dev/null +++ b/src/main/orca-profiles/profile-cloud-session-invalidation.ts @@ -0,0 +1,30 @@ +type OrcaCloudSessionInvalidationListener = () => void + +const listeners = new Set() + +/** + * Fires when an auth failure (revoked or rotated-away refresh token) clears a + * stored cloud session. Never fires for an explicit user sign-out, which already + * hands the fresh auth status back to its caller. + */ +export function onOrcaCloudSessionInvalidated( + listener: OrcaCloudSessionInvalidationListener +): () => void { + listeners.add(listener) + return () => { + listeners.delete(listener) + } +} + +export function emitOrcaCloudSessionInvalidated(): void { + for (const listener of listeners) { + try { + listener() + } catch (error) { + console.warn( + '[orca-profiles] Cloud session invalidation listener failed:', + error instanceof Error ? error.message : String(error) + ) + } + } +} diff --git a/src/main/orca-profiles/profile-cloud-session-refresh.test.ts b/src/main/orca-profiles/profile-cloud-session-refresh.test.ts index 42b665d4e35..2aa3dc0266e 100644 --- a/src/main/orca-profiles/profile-cloud-session-refresh.test.ts +++ b/src/main/orca-profiles/profile-cloud-session-refresh.test.ts @@ -36,6 +36,9 @@ vi.mock('./profile-cloud-client', async (importOriginal) => { vi.mock('./profile-cloud-index', () => ({ linkOrcaProfileToCloud: linkMock })) import { readFreshOrcaCloudSession } from './profile-cloud-session-refresh' +import { OrcaCloudRequestError } from './profile-cloud-client' +import { onOrcaCloudSessionInvalidated } from './profile-cloud-session-invalidation' +import { forgetAmbiguousRefreshAttempt } from './profile-cloud-refresh-replay-guard' const config = {} as OrcaCloudAuthConfig const active = { @@ -59,6 +62,8 @@ const staleSession = { describe('profile cloud session refresh', () => { beforeEach(() => { vi.clearAllMocks() + vi.restoreAllMocks() + forgetAmbiguousRefreshAttempt('/data\0profile-1') saveIfCurrentMock.mockReturnValue('memory-only') readMock.mockReturnValue({ status: 'found', session: staleSession, persistence: 'memory-only' }) }) @@ -110,4 +115,166 @@ describe('profile cloud session refresh', () => { expect(saveIfCurrentMock).toHaveBeenCalledTimes(1) expect(linkMock).toHaveBeenCalledTimes(1) }) + + it('notifies subscribers when an auth failure clears the stored session', async () => { + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(clearMock).toHaveBeenCalledTimes(1) + expect(invalidated).toHaveBeenCalledTimes(1) + unsubscribe() + }) + + it('stays silent when a concurrent rotation already replaced the failed session', async () => { + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + readMock.mockReturnValueOnce({ + status: 'found', + session: staleSession, + persistence: 'memory-only' + }) + readMock.mockReturnValueOnce({ + status: 'found', + session: staleSession, + persistence: 'memory-only' + }) + readMock.mockReturnValue({ + status: 'found', + session: { ...staleSession, refreshToken: 'rotated-refresh' }, + persistence: 'memory-only' + }) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(clearMock).not.toHaveBeenCalled() + expect(invalidated).not.toHaveBeenCalled() + unsubscribe() + }) +}) + +describe('refresh-token replay after an ambiguous attempt', () => { + const timeout = (): Error => + Object.assign(new Error('The operation timed out.'), { name: 'TimeoutError' }) + const rotatedResponse = { + accessToken: 'new-access', + refreshToken: 'new-refresh', + expiresAt: 4_000_000, + organizations: [], + capabilities: { flags: { 'relay.use': true }, refreshedAt: 2 }, + cloud: { userId: 'user-1', cloudProfileId: 'cloud-profile-1', activeOrgId: 'org-1' } + } + + beforeEach(() => { + vi.clearAllMocks() + vi.restoreAllMocks() + forgetAmbiguousRefreshAttempt('/data\0profile-1') + saveIfCurrentMock.mockReturnValue('memory-only') + readMock.mockReturnValue({ status: 'found', session: staleSession, persistence: 'memory-only' }) + }) + + it('never resends a refresh token whose attempt timed out', async () => { + refreshMock.mockRejectedValue(timeout()) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'The operation timed out.' + ) + expect(refreshMock).toHaveBeenCalledTimes(1) + + // The retry loop above this module re-enters immediately; it must not turn + // one lost reply into a second POST of the same token. + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'orca_cloud_refresh_replay_blocked' + ) + expect(refreshMock).toHaveBeenCalledTimes(1) + expect(clearMock).not.toHaveBeenCalled() + }) + + it('adopts the stored session when a timed-out attempt was rotated elsewhere', async () => { + const rotated = { ...staleSession, refreshToken: 'rotated-refresh', expiresAt: 4_000_000 } + refreshMock.mockRejectedValue(timeout()) + readMock + .mockReturnValueOnce({ status: 'found', session: staleSession, persistence: 'memory-only' }) + .mockReturnValueOnce({ status: 'found', session: staleSession, persistence: 'memory-only' }) + .mockReturnValue({ status: 'found', session: rotated, persistence: 'memory-only' }) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'found', + session: rotated + }) + expect(refreshMock).toHaveBeenCalledTimes(1) + expect(saveIfCurrentMock).not.toHaveBeenCalled() + }) + + it('retries once after a definitive 5xx, which cannot have rotated the token', async () => { + refreshMock + .mockRejectedValueOnce(new OrcaCloudRequestError(503)) + .mockResolvedValueOnce(rotatedResponse) + + const result = await readFreshOrcaCloudSession(config, active, '/data') + + expect(refreshMock).toHaveBeenCalledTimes(2) + expect(refreshMock).toHaveBeenNthCalledWith(2, config, staleSession) + expect(result).toEqual({ + status: 'found', + session: expect.objectContaining({ + accessToken: 'new-access', + refreshToken: 'new-refresh' + }) + }) + }) + + it('gives a definitive 5xx exactly one retry', async () => { + refreshMock.mockRejectedValue(new OrcaCloudRequestError(503)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'orca_cloud_request_failed_503' + ) + expect(refreshMock).toHaveBeenCalledTimes(2) + expect(clearMock).not.toHaveBeenCalled() + }) + + it('marks a 401 that follows an ambiguous attempt as a possible self-replay', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const now = vi.spyOn(Date, 'now').mockReturnValue(1_000_000) + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValueOnce(timeout()) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'The operation timed out.' + ) + + now.mockReturnValue(1_000_000 + 31_000) + refreshMock.mockRejectedValueOnce(new OrcaCloudRequestError(401)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(refreshMock).toHaveBeenCalledTimes(2) + expect(warn.mock.calls.flat().join(' ')).toContain('orca_cloud_refresh_possible_replay') + expect(clearMock).toHaveBeenCalledTimes(1) + expect(invalidated).toHaveBeenCalledTimes(1) + unsubscribe() + }) + + it('does not mark a 401 that follows no ambiguous attempt', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(warn.mock.calls.flat().join(' ')).not.toContain('orca_cloud_refresh_possible_replay') + expect(clearMock).toHaveBeenCalledTimes(1) + }) }) diff --git a/src/main/orca-profiles/profile-cloud-session-refresh.ts b/src/main/orca-profiles/profile-cloud-session-refresh.ts index ced48a16e7b..221b4908bee 100644 --- a/src/main/orca-profiles/profile-cloud-session-refresh.ts +++ b/src/main/orca-profiles/profile-cloud-session-refresh.ts @@ -6,13 +6,26 @@ import { readOrcaCloudSession, saveOrcaCloudSessionIfCurrent } from './profile-cloud-session-store' -import { OrcaCloudRequestError, refreshOrcaCloudSession } from './profile-cloud-client' +import { + isAmbiguousCloudRequestFailure, + OrcaCloudRequestError, + refreshOrcaCloudSession +} from './profile-cloud-client' import { linkOrcaProfileToCloud } from './profile-cloud-index' +import type { OrcaCloudSessionExchangeResponse } from './profile-cloud-session-exchange' +import { + AmbiguousRefreshReplayBlockedError, + blocksAmbiguousRefreshReplay, + forgetAmbiguousRefreshAttempt, + recordAmbiguousRefreshAttempt, + wasRefreshTokenAmbiguouslyAttempted +} from './profile-cloud-refresh-replay-guard' import { captureCloudSessionMutation, cloudSessionIdentity, tombstoneCloudSession } from './profile-cloud-session-mutation' +import { emitOrcaCloudSessionInvalidated } from './profile-cloud-session-invalidation' const CLOUD_SESSION_REFRESH_SKEW_MS = 60_000 @@ -66,6 +79,76 @@ function clearCloudSessionIfUnchanged( ) } clearOrcaCloudSession(profileId, userDataPath) + forgetAmbiguousRefreshAttempt(cloudSessionRefreshKey(profileId, userDataPath)) + // Why: the renderer cached auth status at startup; without this it keeps + // showing "Connected" until the app restarts. + emitOrcaCloudSessionInvalidated() +} + +// Why: support cannot otherwise tell a genuine revocation from a sign-out we +// caused ourselves by resending a refresh token whose first attempt never +// answered. Never log the token itself. +function warnIfPossibleRefreshReplay( + profileId: string, + userDataPath: string, + failed: OrcaCloudSession, + error: unknown +): void { + if (!(error instanceof OrcaCloudRequestError) || error.statusCode !== 401) { + return + } + const key = cloudSessionRefreshKey(profileId, userDataPath) + if (!wasRefreshTokenAmbiguouslyAttempted(key, failed.refreshToken)) { + return + } + console.warn( + '[orca-cloud] orca_cloud_refresh_possible_replay: refresh rejected 401 for a token whose earlier attempt never answered' + ) +} + +type CloudSessionRefreshAttempt = + | { status: 'refreshed'; response: OrcaCloudSessionExchangeResponse } + | { status: 'rotated-elsewhere'; session: OrcaCloudSession } + +function isRetryableCloudRefreshRejection(error: unknown): boolean { + return error instanceof OrcaCloudRequestError && error.statusCode >= 500 +} + +async function attemptCloudSessionRefresh( + key: string, + config: OrcaCloudAuthConfig, + active: ActiveOrcaProfileState, + userDataPath: string, + session: OrcaCloudSession +): Promise { + for (let attempt = 0; ; attempt++) { + try { + const response = await refreshOrcaCloudSession(config, session) + forgetAmbiguousRefreshAttempt(key) + return { status: 'refreshed', response } + } catch (error) { + const ambiguous = isAmbiguousCloudRequestFailure(error) + if (ambiguous) { + recordAmbiguousRefreshAttempt(key, session.refreshToken) + } + // Only a status line proves the server rejected this token without + // rotating it, so a definitive 5xx is the only failure worth retrying. + const retryable = !ambiguous && attempt === 0 && isRetryableCloudRefreshRejection(error) + if (!ambiguous && !retryable) { + throw error + } + // Another caller may have rotated the stored session while this attempt + // was in flight; that result is the one to use, and the token this attempt + // held is no longer ours to send again. + const current = readOrcaCloudSession(active.profile.id, userDataPath) + if (current.status === 'found' && current.session.refreshToken !== session.refreshToken) { + return { status: 'rotated-elsewhere', session: current.session } + } + if (ambiguous) { + throw error + } + } + } } async function refreshStoredCloudSession( @@ -91,9 +174,16 @@ async function refreshStoredCloudSession( if (!active.profile.cloud) { throw new StaleCloudSessionMutationError() } + if (blocksAmbiguousRefreshReplay(key, session.refreshToken)) { + throw new AmbiguousRefreshReplayBlockedError() + } const expectedIdentity = cloudSessionIdentity(active.profile.id, active.profile.cloud) const snapshot = captureCloudSessionMutation(expectedIdentity, userDataPath) - const refreshed = await refreshOrcaCloudSession(config, session) + const attempt = await attemptCloudSessionRefresh(key, config, active, userDataPath, session) + if (attempt.status === 'rotated-elsewhere') { + return attempt.session + } + const refreshed = attempt.response const refreshedIdentity = cloudSessionIdentity(active.profile.id, refreshed.cloud) if ( refreshedIdentity.cloudUserId !== expectedIdentity.cloudUserId || @@ -144,6 +234,7 @@ export async function readFreshOrcaCloudSession( } } catch (error) { if (isOrcaCloudAuthFailure(error)) { + warnIfPossibleRefreshReplay(active.profile.id, userDataPath, session.session, error) clearCloudSessionIfUnchanged(active.profile.id, userDataPath, session.session, active) return { status: 'reconnect-required' } } @@ -164,6 +255,7 @@ export async function forceRefreshOrcaCloudSession( } } catch (error) { if (isOrcaCloudAuthFailure(error)) { + warnIfPossibleRefreshReplay(active.profile.id, userDataPath, session, error) clearCloudSessionIfUnchanged(active.profile.id, userDataPath, session, active) return { status: 'reconnect-required' } } diff --git a/src/main/own-chromium-tree-kill-guard.test.ts b/src/main/own-chromium-tree-kill-guard.test.ts index bd3b1674e18..7b9c30687bc 100644 --- a/src/main/own-chromium-tree-kill-guard.test.ts +++ b/src/main/own-chromium-tree-kill-guard.test.ts @@ -13,7 +13,12 @@ import { import { readOrcaChromiumProcessPids } from './orca-chromium-process-pids' import { classifyWindowsTreeKillTarget } from './windows-pty-root-identity' import { terminateWindowsProcessTree } from './windows-process-tree-kill' -import { admitSelfInitiatedTreeKill } from './own-chromium-tree-kill-guard' +import { + admitSelfInitiatedTreeKill, + installMainProcessTreeKillGate +} from './own-chromium-tree-kill-guard' +import { killCodexAppServerProcessTree } from './codex/codex-app-server-session' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' import { resetSelfInitiatedTreeKillLogForTest } from './crash-reporting/self-initiated-tree-kill-log' import { clearCrashBreadcrumbsForTest, @@ -57,6 +62,7 @@ beforeEach(() => { setActiveSink({ push: () => {}, flush: () => {}, close: () => {} }) clearCrashBreadcrumbsForTest() resetSelfInitiatedTreeKillLogForTest() + installMainProcessTreeKillGate() }) afterEach(() => { @@ -66,6 +72,7 @@ afterEach(() => { vi.restoreAllMocks() _resetTracerForTests() clearCrashBreadcrumbsForTest() + setProcessTreeKillGate(null) }) describe('refusing to tree-kill our own Chromium processes', () => { @@ -143,6 +150,38 @@ describe('refusing to tree-kill our own Chromium processes', () => { ) }) + it('refuses the codex app-server deadline kill against one of our own pids', () => { + const spawnImpl = vi.fn(() => ({ on: vi.fn(), unref: vi.fn() })) + const child = { pid: RENDERER_PID, kill: vi.fn() } + + killCodexAppServerProcessTree(child as never, { + platform: 'win32', + spawnImpl: spawnImpl as never + }) + + // The deadline timer fires on `child.pid` alone; a reaped-then-recycled pid + // is the stale-pid mechanism this gate exists to stop. + expect(spawnImpl).not.toHaveBeenCalled() + expect(getCrashBreadcrumbSnapshot()).toEqual([ + expect.objectContaining({ name: 'self_tree_kill_refused_own_chromium' }) + ]) + }) + + it('still lets the codex app-server deadline kill reach a foreign pid', () => { + const killer = { on: vi.fn(), unref: vi.fn() } + const spawnImpl = vi.fn(() => killer) + + killCodexAppServerProcessTree({ pid: 7777, kill: vi.fn() } as never, { + platform: 'win32', + spawnImpl: spawnImpl as never + }) + + expect(spawnImpl).toHaveBeenCalledWith('taskkill', ['/pid', '7777', '/t', '/f'], { + stdio: 'ignore', + windowsHide: true + }) + }) + /** * Fail-open is the deliberate choice — see `orca-chromium-process-pids.ts` for * why refusing everything is worse — so the crumb is the only thing that keeps diff --git a/src/main/own-chromium-tree-kill-guard.ts b/src/main/own-chromium-tree-kill-guard.ts index 4daedf4faa4..24b6a4b7327 100644 --- a/src/main/own-chromium-tree-kill-guard.ts +++ b/src/main/own-chromium-tree-kill-guard.ts @@ -4,16 +4,23 @@ import { type SelfInitiatedTreeKillScope } from './crash-reporting/self-initiated-tree-kill-log' import { readOrcaChromiumProcessPids } from './orca-chromium-process-pids' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' /** - * Gate every main-process tree-kill through one decision: refuse the pid when - * Electron is currently accounting for it, otherwise put it on the record. + * Gate every main-process tree-kill through one decision: refuse a pid-addressed + * walk when Electron is currently accounting for the pid, otherwise put the + * kill on the record. * * Why a shared gate rather than a check inside `terminateWindowsProcessTree`: - * the codex and claude account-login teardowns run their own `taskkill /T /F` - * with different lifetimes (one sync, one with its own timeout ladder), so a - * guard that only lived in the tree-kill helper would cover one of three - * families. Returns false when the caller must not kill. + * five other families in main run their own `taskkill /T /F` with different + * lifetimes (sync, fire-and-forget, timeout ladder), and three more live in + * `src/shared` and reach this through `process-tree-kill-gate`, so a guard that + * only lived in the tree-kill helper would cover one of nine. + * `main-process-tree-kill-gate.test.ts` holds that set closed by counting `/pid` + * call sites against gate admissions per file, not by file. Returns false + * when the caller must not walk that pid's tree; the caller still kills its own + * root through the child handle (`refused-tree-kill-root-termination.test.ts`), + * so a refusal is never a process leak. * * Electron main only, by construction. `terminateWindowsProcessTree` also runs * in the standalone daemon (the `pty-descendant-sweep` site), where @@ -30,11 +37,26 @@ export function admitSelfInitiatedTreeKill(target: { }): boolean { // Why: no PTY root, codex root or git child is ever one of our own Chromium // processes, so a pid that is means the caller is about to kill a renderer, - // the GPU or the browser itself (#10680). - if (readOrcaChromiumProcessPids().has(target.pid)) { - recordRefusedOwnChromiumTreeKill(target) - return false + // the GPU or the browser itself (#10680). Only the pid-addressed scope can + // land there: a POSIX group holds only what Orca put in it, so that arm is + // recorded and admitted like every other group kill in main, and a stale + // `getAppMetrics()` entry cannot orphan a macOS/Linux tree. + const isOwnChromiumPid = + target.scope === 'win-taskkill-tree' && readOrcaChromiumProcessPids().has(target.pid) + try { + if (isOwnChromiumPid) { + recordRefusedOwnChromiumTreeKill(target) + } else { + recordSelfInitiatedTreeKill(target) + } + } catch { + // Recording must never turn a successful termination into a failed one, and + // never flip the decision: it is taken above, before anything can throw. } - recordSelfInitiatedTreeKill(target) - return true + return !isOwnChromiumPid +} + +/** Hands the gate to the shared choke points, which cannot import main. */ +export function installMainProcessTreeKillGate(): void { + setProcessTreeKillGate((kill) => admitSelfInitiatedTreeKill(kill)) } diff --git a/src/main/ports/local-workspace-platform-port-scanner.ts b/src/main/ports/local-workspace-platform-port-scanner.ts index 7e3b9941617..9a760870540 100644 --- a/src/main/ports/local-workspace-platform-port-scanner.ts +++ b/src/main/ports/local-workspace-platform-port-scanner.ts @@ -3,6 +3,7 @@ import { getProcessOutputFields } from '../../shared/process-output-field-scanne import { readWindowsProcessTable } from '../windows/windows-process-table' import { runPortScanCommand } from './port-scan-command-client' import { + partitionListenersNeedingMetadata, recallListenerMetadata, rememberListenerMetadata, shouldSkipMetadataCommands, @@ -19,25 +20,27 @@ import { export function parseLsofListeningOutput(output: string): RawListeningPort[] { const ports: RawListeningPort[] = [] - let currentPid: number | undefined - let currentProcessName: string | undefined + let pid: number | undefined + let processName: string | undefined + let socketId: string | undefined for (const line of output.split('\n')) { - if (!line) { - continue - } const tag = line[0] const value = line.slice(1) if (tag === 'p') { - const pid = Number.parseInt(value, 10) - currentPid = Number.isFinite(pid) ? pid : undefined - currentProcessName = undefined + const parsedPid = Number.parseInt(value, 10) + pid = Number.isFinite(parsedPid) ? parsedPid : undefined + processName = socketId = undefined } else if (tag === 'c') { - currentProcessName = value + processName = value + } else if (tag === 'f') { + socketId = undefined // each file record restarts; a socket without `d` must not inherit one + } else if (tag === 'd') { + socketId = value || undefined } else if (tag === 'n') { const parsed = parseAddressWithPort(value) if (parsed) { - ports.push({ pid: currentPid, processName: currentProcessName, ...parsed }) + ports.push({ pid, processName, ...(socketId ? { socketId } : {}), ...parsed }) } } } @@ -118,17 +121,23 @@ async function scanDarwinLsofPorts( '-iTCP', '-sTCP:LISTEN', '-F', - 'pcn' + // Why `d`: the socket's kernel identity is free on this command and lets the metadata cache + // tell a recycled pid on the same port apart from the process it remembered. + 'pcnd' ]) const ports = parseLsofListeningOutput(stdout) if (shouldSkipMetadataCommands(spawnMs, options)) { return { ports, metadataAvailable: false } } - const metadata = await loadDarwinProcessMetadata( - new Set(ports.flatMap((p) => (p.pid ? [p.pid] : []))) - ) + // Why: on a quiet machine the same servers keep listening, so the two metadata commands — the + // expensive half of the scan — would re-derive answers the last scan already has. + const { hydrated, pidsNeedingMetadata } = partitionListenersNeedingMetadata(ports, options) + if (pidsNeedingMetadata.size === 0) { + return { ports: hydrated, metadataAvailable: true } + } + const metadata = await loadDarwinProcessMetadata(pidsNeedingMetadata) return { - ports: ports.map((port) => ({ ...metadata.get(port.pid ?? -1), ...port })), + ports: hydrated.map((port) => ({ ...metadata.get(port.pid ?? -1), ...port })), metadataAvailable: true } } diff --git a/src/main/ports/local-workspace-port-scan-state.ts b/src/main/ports/local-workspace-port-scan-state.ts index 582e8b9731e..e4025944c79 100644 --- a/src/main/ports/local-workspace-port-scan-state.ts +++ b/src/main/ports/local-workspace-port-scan-state.ts @@ -7,10 +7,14 @@ import { } from './workspace-port-scan-timeout-backoff' const SLOW_SPAWN_SKIP_METADATA_MS = 2_000 +/** Re-probe a remembered listener every Nth scan (~5 min at 30s) so a cwd change cannot go stale forever. */ +const METADATA_REPROBE_INTERVAL_SCANS = 10 const commandTimeoutBackoff = new WorkspacePortScanTimeoutBackoff() let loggedWorkerUnavailable = false let skippedMetadataOnLastScan = false -let lastListenerMetadata = new Map() +let lastListenerMetadata = new Map() +let metadataScanSequence = 0 +let reusedListenerKeys = new Set() export type WorkspacePortScanOptions = { requireMetadata?: boolean @@ -21,6 +25,8 @@ export type RawListeningPort = { port: number pid?: number processName?: string + /** Kernel socket identity from `lsof -F d`; a new socket on the same pid:port gets a new one. */ + socketId?: string commandLine?: string cwd?: string } @@ -31,6 +37,11 @@ export type ProcessMetadata = { cwd?: string } +type RememberedListenerMetadata = ProcessMetadata & { + socketId?: string + probedAtScan: number +} + export type NormalizedWorkspacePortProbe = { worktree: WorkspacePortProbe normalizedPath: string @@ -58,6 +69,8 @@ export function resetWorkspacePortScanTimeoutBackoffForTests(): void { loggedWorkerUnavailable = false skippedMetadataOnLastScan = false lastListenerMetadata = new Map() + metadataScanSequence = 0 + reusedListenerKeys = new Set() } export function shouldSkipMetadataCommands( @@ -72,17 +85,98 @@ export function shouldSkipMetadataCommands( return skip } +// dedupeRawPorts already collapses rows by connectHost:port:pid, so this key is unique per row. function listenerMetadataKey(port: RawListeningPort): string { return `${port.pid ?? 'unknown'}:${port.host}:${port.port}` } +/** Call after partitionListenersNeedingMetadata: it consumes the reuse set that call recorded. */ export function rememberListenerMetadata(ports: readonly RawListeningPort[]): void { - lastListenerMetadata = new Map( - ports.map((port) => [ - listenerMetadataKey(port), - { processName: port.processName, commandLine: port.commandLine, cwd: port.cwd } - ]) - ) + const previous = lastListenerMetadata + lastListenerMetadata = new Map() + for (const port of ports) { + const key = listenerMetadataKey(port) + // Why: a reused entry keeps its original probe time so the staleness ceiling still expires it. + const probedAtScan = reusedListenerKeys.has(key) + ? (previous.get(key)?.probedAtScan ?? metadataScanSequence) + : metadataScanSequence + lastListenerMetadata.set(key, { + processName: port.processName, + socketId: port.socketId, + commandLine: port.commandLine, + cwd: port.cwd, + probedAtScan + }) + } + reusedListenerKeys = new Set() +} + +/** + * Split listeners into those a previous scan already resolved and the pids still needing a probe. + * + * Records which keys were reused; rememberListenerMetadata reads and clears that on the same scan. + * + * Why: the metadata commands are the expensive half of a macOS scan, and a listener that is still + * the same process on the same address has the same command line it had 30s ago. A remembered + * entry is only trusted when the free `lsof -F c` process name and `-F d` socket identity from + * this scan still match, so a recycled pid re-probes instead of inheriting the dead process's + * metadata; and every entry is re-probed after METADATA_REPROBE_INTERVAL_SCANS so a process that + * chdir'd while listening cannot keep a stale cwd forever. + */ +export function partitionListenersNeedingMetadata( + ports: readonly RawListeningPort[], + options: WorkspacePortScanOptions = {} +): { hydrated: RawListeningPort[]; pidsNeedingMetadata: Set } { + metadataScanSequence += 1 + reusedListenerKeys = new Set() + // Why requireMetadata opts out: that caller is the SIGTERM authorization re-scan, so it must + // attribute the owner from this cycle's probe and never from a remembered cwd. + if (options.requireMetadata) { + return { + hydrated: [...ports], + pidsNeedingMetadata: new Set(ports.flatMap((port) => (port.pid ? [port.pid] : []))) + } + } + const hydrated: RawListeningPort[] = [] + const pidsNeedingMetadata = new Set() + const reusableByPort = new Map() + for (const port of ports) { + const remembered = lastListenerMetadata.get(listenerMetadataKey(port)) + // Why require commandLine: a probe that returned nothing must not be cached as an answer. + if ( + remembered?.commandLine !== undefined && + remembered.processName === port.processName && + remembered.socketId === port.socketId && + metadataScanSequence - remembered.probedAtScan < METADATA_REPROBE_INTERVAL_SCANS && + port.pid !== undefined + ) { + reusableByPort.set(port, remembered) + continue + } + if (port.pid !== undefined) { + pidsNeedingMetadata.add(port.pid) + } + } + // Why the second pass: if any of a pid's sockets needs a probe, none of its sockets may be + // served from cache — otherwise one process reports a fresh cwd on one row and a remembered + // cwd on another, i.e. two different workspace attributions. + for (const port of ports) { + const remembered = + port.pid !== undefined && !pidsNeedingMetadata.has(port.pid) + ? reusableByPort.get(port) + : undefined + if (remembered) { + reusedListenerKeys.add(listenerMetadataKey(port)) + hydrated.push({ + ...port, + commandLine: port.commandLine ?? remembered.commandLine, + cwd: port.cwd ?? remembered.cwd + }) + continue + } + hydrated.push(port) + } + return { hydrated, pidsNeedingMetadata } } export function recallListenerMetadata(port: RawListeningPort): RawListeningPort { diff --git a/src/main/ports/local-workspace-port-scanner.test.ts b/src/main/ports/local-workspace-port-scanner.test.ts index be37ad60eb3..4f6820be0bd 100644 --- a/src/main/ports/local-workspace-port-scanner.test.ts +++ b/src/main/ports/local-workspace-port-scanner.test.ts @@ -50,6 +50,35 @@ describe('local workspace port scanner parsing', () => { ]) }) + it('keeps the socket identity from lsof -F d and tolerates its absence', () => { + const ports = parseLsofListeningOutput( + [ + 'p123', + 'cnode', + 'f18', + 'd0x469ca588d83e7924', + 'n127.0.0.1:5173', + 'f19', + 'n127.0.0.1:5174', + 'p456', + 'cnginx', + 'n*:8080' + ].join('\n') + ) + + expect(ports).toEqual([ + { + pid: 123, + processName: 'node', + socketId: '0x469ca588d83e7924', + host: '127.0.0.1', + port: 5173 + }, + { pid: 123, processName: 'node', host: '127.0.0.1', port: 5174 }, + { pid: 456, processName: 'nginx', host: '*', port: 8080 } + ]) + }) + it('parses multiple lsof listening ports for the same process', () => { const ports = parseLsofListeningOutput( ['p123', 'cnode', 'n127.0.0.1:5173', 'n127.0.0.1:55173'].join('\n') @@ -387,7 +416,19 @@ describe('scanWorkspacePorts with delayed process creation', () => { it('does not let a required-metadata scan reset the background skip parity', async () => { vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') - mockStalledDarwinScan() + // Why a fresh pid each cycle: a listener the previous scan already resolved is served from the + // remembered metadata, so a stable pid would hide whether this scan skipped the probe or not. + let listenerPid = 123 + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + listenerPid += 1 + return { stdout: `p${listenerPid}\ncnode\nn127.0.0.1:5173`, spawnMs: 4_200 } + } + if (command === 'lsof') { + return { stdout: [`p${listenerPid}`, 'n/repo'].join('\n'), spawnMs: 4_200 } + } + return { stdout: `${listenerPid} node /repo/server.js`, spawnMs: 4_200 } + }) await scanWorkspacePorts(worktrees, urlWatcherStub()) await scanWorkspacePorts(worktrees, urlWatcherStub(), { requireMetadata: true }) @@ -397,6 +438,168 @@ describe('scanWorkspacePorts with delayed process creation', () => { expect(runPortScanCommandMock).toHaveBeenCalledTimes(7) }) + it('serves an unchanged listener from remembered metadata instead of re-probing', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: LSOF_LISTEN_OUTPUT, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node /repo/server.js', spawnMs: 5 } + }) + + const first = await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + + const second = await scanWorkspacePorts(worktrees, urlWatcherStub()) + + // Only the listening scan itself runs; the two metadata commands are served from the cache. + expect(runPortScanCommandMock).toHaveBeenCalledTimes(4) + expect(second.ports).toEqual(first.ports) + }) + + it('re-probes every port of a pid when any one of them needs metadata', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let secondSocket = 'd0xbbbb' + let cwd = '/repo' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { + stdout: [ + 'p123', + 'cnode', + 'f10', + 'd0xaaaa', + 'n127.0.0.1:5173', + 'f11', + secondSocket, + 'n127.0.0.1:5174' + ].join('\n'), + spawnMs: 5 + } + } + if (command === 'lsof') { + return { stdout: ['p123', `n${cwd}`].join('\n'), spawnMs: 5 } + } + return { stdout: `123 node ${cwd}/server.js`, spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + + // One socket is replaced and the process has since moved, so this pid must be re-derived for + // BOTH its rows — serving one from cache would report two different workspaces for one process. + secondSocket = 'd0xcccc' + cwd = '/elsewhere' + const scan = await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(6) + expect(new Set(scan.ports.map((entry) => entry.kind)).size).toBe(1) + }) + + it('always probes metadata for a requireMetadata scan, even when the cache is warm', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: LSOF_LISTEN_OUTPUT, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node /repo/server.js', spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + await scanWorkspacePorts(worktrees, urlWatcherStub()) + // Warm cache: the background poll is served without the metadata commands. + expect(runPortScanCommandMock).toHaveBeenCalledTimes(4) + + // Why: this is the SIGTERM authorization re-scan; a remembered cwd must never authorize a kill. + await scanWorkspacePorts(worktrees, urlWatcherStub(), { requireMetadata: true }) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(7) + }) + + it('re-probes when a recycled pid is running a different process', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let processName = 'node' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: `p123\nc${processName}\nn127.0.0.1:5173`, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: `123 ${processName} /repo/server.js`, spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + + // Same pid and address, different process: the remembered metadata must not be reused. + processName = 'python3' + await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(6) + }) + + it('re-probes when the same pid and name listen through a different socket', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let socketId = '0x1' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: `p123\ncnode\nf18\nd${socketId}\nn127.0.0.1:5173`, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node /repo/server.js', spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(4) + + // Same pid, name and address but a new kernel socket: a restarted process, not the cached one. + socketId = '0x2' + await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(7) + }) + + it('re-probes a remembered listener every tenth scan so a changed cwd cannot stay stale', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let cwd = '/repo' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: LSOF_LISTEN_OUTPUT, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', `n${cwd}`].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node server.js', spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + cwd = '/repo/worktrees/feature' + for (let scan = 2; scan <= 10; scan += 1) { + const cached = await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(cached.ports[0]).toMatchObject({ owner: { worktreeId: 'repo::/repo' } }) + } + // 3 for the first scan, then one listening command per cached scan. + expect(runPortScanCommandMock).toHaveBeenCalledTimes(12) + + const reprobed = await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(15) + expect(reprobed.ports[0]).toMatchObject({ + owner: { worktreeId: 'repo::/repo/worktrees/feature' } + }) + }) + // Regression for #11161 review: without carry-forward the panel moves every // workspace port into External on each skipped cycle. it('carries the previous cycle attribution through a skipped scan', async () => { diff --git a/src/main/providers/windows-foreground-process-rows.ts b/src/main/providers/windows-foreground-process-rows.ts index 5f462649e6c..e8320a6d00a 100644 --- a/src/main/providers/windows-foreground-process-rows.ts +++ b/src/main/providers/windows-foreground-process-rows.ts @@ -108,6 +108,26 @@ export async function queryWindowsPaneProcessInventory( } } +/** + * The descendant walk over rows the caller already read. + * + * Why exported: a caller that needs a field this module's projection drops — + * process creation time, for a PID-reuse-safe teardown snapshot — would + * otherwise read the whole table a second time to get it. + * Null when the root is absent, which is a stale or filtered snapshot rather + * than a root with no descendants. + */ +export function windowsDescendantsFromRows( + rows: Row[], + rootPid: number +): (Row & { depth: number })[] | null { + const index = getProcessTableIndex(rows) + if (!index.byPid.has(rootPid)) { + return null + } + return collectDescendantsFromIndex(index, rootPid).sort((a, b) => b.depth - a.depth) +} + /** Test-only: clear the shared snapshot so one case's rows never serve the next. */ export function resetWindowsProcessRowsSnapshotForTests(): void { resetWindowsProcessTableForTests() diff --git a/src/main/pty-descendant-exit-verification.ts b/src/main/pty-descendant-exit-verification.ts index c0ed223704a..4c8471fa955 100644 --- a/src/main/pty-descendant-exit-verification.ts +++ b/src/main/pty-descendant-exit-verification.ts @@ -21,50 +21,143 @@ function waitForDelay(ms: number): Promise { function matchingSnapshotRows( snapshot: DescendantSnapshot, - table: readonly ProcessTableRow[] + table: readonly ProcessTableRow[], + rejectDuplicatePids = false ): ProcessTableRow[] { const expected = new Map(snapshot.descendants.map((row) => [row.pid, row])) - return table.filter((live) => { - const row = expected.get(live.pid) - return row?.startedAt === live.startedAt && row.pgid === live.pgid + const rowsByPid = new Map() + for (const live of table) { + const rows = rowsByPid.get(live.pid) + if (rows) { + rows.push(live) + } else { + rowsByPid.set(live.pid, [live]) + } + } + return [...expected.entries()].flatMap(([pid, row]) => { + const rows = rowsByPid.get(pid) + if (rejectDuplicatePids && rows?.length !== 1) { + // Duplicate PID rows make this non-atomic process-table read ambiguous; + // never signal or count either identity as proof of liveness. + return [] + } + return (rows ?? []).filter((live) => live.startedAt === row.startedAt && live.pgid === row.pgid) }) } +function hasDuplicateSnapshotPids( + snapshot: DescendantSnapshot, + table: readonly ProcessTableRow[] +): boolean { + const expected = new Set(snapshot.descendants.map((row) => row.pid)) + const counts = new Map() + for (const live of table) { + if (expected.has(live.pid)) { + counts.set(live.pid, (counts.get(live.pid) ?? 0) + 1) + } + } + return [...counts.values()].some((count) => count > 1) +} + type VerificationDeps = TerminateDeps & { verifyMs?: number + /** Revalidate identities before signaling; used by Claude's close proof. */ + requireIdentityBeforeSignal?: boolean } +/** + * Orca's verdict vocabulary for a snapshotted tree, with no synonyms: `live` is + * an identity-matched descendant still observed at the deadline; `unverifiable` + * is a table that could not be read, which is never evidence either way. + */ +export type DescendantTreeVerdict = 'exited' | 'live' | 'unverifiable' + /** An unreadable process table is never proof that a stopped descendant exited. */ export async function terminateDescendantSnapshotAndWait( snapshot: DescendantSnapshot, deps: VerificationDeps = {} ): Promise { + return (await terminateDescendantSnapshotWithVerdict(snapshot, deps)) === 'exited' +} + +/** Signals the snapshot, then reports what the last table read observed. */ +export async function terminateDescendantSnapshotWithVerdict( + snapshot: DescendantSnapshot, + deps: VerificationDeps = {} +): Promise { const sendSignal = deps.sendSignal ?? sendDescendantSignal const readTable = deps.readTable ?? readProcessTable const graceMs = deps.graceMs ?? DESCENDANT_KILL_GRACE_MS const verifyMs = deps.verifyMs ?? DESCENDANT_KILL_VERIFY_MS const deadline = Date.now() + verifyMs - for (const row of snapshot.descendants) { - sendSignal(row.pid, 'SIGTERM') - } let forced = false + let signalled = !deps.requireIdentityBeforeSignal + let missingObservations = 0 + if (signalled) { + for (const row of snapshot.descendants) { + sendSignal(row.pid, 'SIGTERM') + } + } while (Date.now() < deadline) { const capture = await readProcessTableBeforeDeadline( readTable, deps.timeoutMs ?? DESCENDANT_SNAPSHOT_TIMEOUT_MS ) - if (!capture) { - return false - } - const live = matchingSnapshotRows(snapshot, capture.rows) - if (live.length === 0) { - return true - } - if (!forced && Date.now() >= deadline - verifyMs + graceMs) { - forced = true - for (const row of live) { - if (hasUnambiguousStartIdentity(row, snapshot.capturedAtMs)) { - sendSignal(row.pid, 'SIGKILL') + // A read that missed its own deadline is not an answer, and surrendering on + // the first slow one spends none of the window this verification was given: + // on a loaded host that reported a tree unverifiable without ever seeing it. + if (capture) { + if (deps.requireIdentityBeforeSignal && hasDuplicateSnapshotPids(snapshot, capture.rows)) { + // A duplicate target pid is an ambiguous non-atomic read. Do not signal + // either row and do not turn that uncertainty into an exited verdict. + await waitForDelay(50) + continue + } + const live = matchingSnapshotRows(snapshot, capture.rows, deps.requireIdentityBeforeSignal) + if (live.length === 0) { + // Before a signal has been sent, an empty identity match means the + // snapshotted descendants already exited or were replaced. Signalling + // those old numeric pids would be unsafe. + if (deps.requireIdentityBeforeSignal) { + // A single process-table read can race a fork or return a partial + // view; require two bounded absences before claiming the tree gone. + missingObservations += 1 + if (missingObservations < 2) { + await waitForDelay(50) + continue + } + } + return 'exited' + } + missingObservations = 0 + if (!signalled) { + // Revalidate every identity immediately before the first signal. A PID + // can be recycled between the original walk and close, so never signal + // from the stale snapshot alone. + for (const row of live) { + sendSignal(row.pid, 'SIGTERM') + } + signalled = true + } + if (!forced && Date.now() >= deadline - verifyMs + graceMs) { + forced = true + for (const row of live) { + // A row a walk re-derived from a live root is ours whatever second it + // was born in, which start time alone can never establish for one born + // in its own capture second. Rows no walk re-derived still answer to + // the second-resolution fence, which is all the evidence they have. + // Scoped to the identity-revalidating callers; the same argument holds + // for the rest, but widening it is a deliberate change of its own. + if ( + (deps.requireIdentityBeforeSignal === true && + snapshot.reDerivedPids?.has(row.pid) === true) || + hasUnambiguousStartIdentity( + row, + snapshot.capturedAtMsByPid?.[String(row.pid)] ?? snapshot.capturedAtMs + ) + ) { + sendSignal(row.pid, 'SIGKILL') + } } } } @@ -74,5 +167,19 @@ export async function terminateDescendantSnapshotAndWait( readTable, deps.timeoutMs ?? DESCENDANT_SNAPSHOT_TIMEOUT_MS ) - return finalCapture !== null && matchingSnapshotRows(snapshot, finalCapture.rows).length === 0 + if (!finalCapture) { + return 'unverifiable' + } + if (deps.requireIdentityBeforeSignal && hasDuplicateSnapshotPids(snapshot, finalCapture.rows)) { + return 'unverifiable' + } + const finalLive = matchingSnapshotRows( + snapshot, + finalCapture.rows, + deps.requireIdentityBeforeSignal + ) + if (finalLive.length > 0) { + return 'live' + } + return deps.requireIdentityBeforeSignal && missingObservations < 2 ? 'unverifiable' : 'exited' } diff --git a/src/main/pty-descendant-termination.test.ts b/src/main/pty-descendant-termination.test.ts index e1255a678d8..0c4bea81306 100644 --- a/src/main/pty-descendant-termination.test.ts +++ b/src/main/pty-descendant-termination.test.ts @@ -15,7 +15,10 @@ import { type ProcessTableCapture, type ProcessTableRow } from './pty-descendant-termination' -import { terminateDescendantSnapshotAndWait } from './pty-descendant-exit-verification' +import { + terminateDescendantSnapshotAndWait, + terminateDescendantSnapshotWithVerdict +} from './pty-descendant-exit-verification' const CAPTURED_AT_MS = Date.parse('Tue Jul 14 12:00:00 2026') @@ -53,7 +56,14 @@ function snapshot( rootPgid: number | null = 10, capturedAtMs = CAPTURED_AT_MS ) { - return { rootPgid, descendants, capturedAtMs } + return { + ...(rootPgid === null ? {} : { root: { pid: 10, startedAt: 'Mon Jul 13 12:54:47 2026' } }), + rootPgid, + descendants, + capturedAtMs, + // Everything a walk returns was re-derived by it. + ...(rootPgid === null ? {} : { reDerivedPids: new Set(descendants.map((row) => row.pid)) }) + } } describe('parseProcessTable', () => { @@ -298,6 +308,32 @@ describe('terminateDescendantSnapshot', () => { expect(sendSignal).not.toHaveBeenCalled() expect(vi.getTimerCount()).toBe(0) }) + + it("uses each row's capture boundary when escalating a merged snapshot", async () => { + const oldBoundary = CAPTURED_AT_MS + 900 + const refreshBoundary = CAPTURED_AT_MS + 2_100 + const retained = row(20, 10, 20, 'Tue Jul 14 12:00:00 2026') + const fresh = row(30, 10, 30, 'Tue Jul 14 12:00:01 2026') + const sendSignal = vi.fn() + terminateDescendantSnapshot( + { + ...snapshot([retained, fresh], 10, refreshBoundary), + capturedAtMsByPid: { '20': oldBoundary, '30': refreshBoundary } + }, + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([retained, fresh])) + } + ) + sendSignal.mockClear() + + await vi.advanceTimersByTimeAsync(DESCENDANT_KILL_GRACE_MS) + + // PID 20 was retained from the earlier capture and is still in its + // capture second; PID 30 was newly observed by the refresh and is old + // enough for a bounded forced cleanup. + expect(sendSignal.mock.calls).toEqual([[30, 'SIGKILL']]) + }) }) describe('terminateDescendantSnapshotAndWait', () => { @@ -333,14 +369,114 @@ describe('terminateDescendantSnapshotAndWait', () => { it('does not claim exit when the verification table is unavailable', async () => { const sendSignal = vi.fn() - const result = await terminateDescendantSnapshotAndWait(snapshot([row(20, 10, 20)]), { + const pending = terminateDescendantSnapshotAndWait(snapshot([row(20, 10, 20)]), { sendSignal, - readTable: vi.fn().mockRejectedValue(new Error('ps exploded')) + readTable: vi.fn().mockRejectedValue(new Error('ps exploded')), + verifyMs: 200 }) + await vi.advanceTimersByTimeAsync(400) - expect(result).toBe(false) + await expect(pending).resolves.toBe(false) expect(sendSignal).toHaveBeenCalledWith(20, 'SIGTERM') }) + + it('keeps polling past a read that missed its deadline rather than surrendering', async () => { + const survivor = row(20, 10, 20) + const readTable = vi + .fn() + // A loaded host can miss one read's deadline with the window still open. + .mockRejectedValueOnce(new Error('ps timed out')) + .mockResolvedValueOnce(tableCapture([survivor])) + .mockResolvedValueOnce(tableCapture([])) + const sendSignal = vi.fn() + + const pending = terminateDescendantSnapshotWithVerdict(snapshot([survivor]), { + sendSignal, + readTable, + graceMs: 0, + verifyMs: 2_000 + }) + await vi.advanceTimersByTimeAsync(500) + + await expect(pending).resolves.toBe('exited') + expect(sendSignal.mock.calls).toEqual([ + [20, 'SIGTERM'], + [20, 'SIGKILL'] + ]) + }) + + it('names a survivor seen at the deadline live, never unverifiable', async () => { + const survivor = row(20, 10, 20) + const pending = terminateDescendantSnapshotWithVerdict(snapshot([survivor]), { + sendSignal: vi.fn(), + readTable: vi.fn().mockResolvedValue(tableCapture([survivor])), + graceMs: 0, + verifyMs: 100 + }) + await vi.advanceTimersByTimeAsync(200) + + await expect(pending).resolves.toBe('live') + }) + + it('names an unreadable verification table unverifiable', async () => { + const pending = terminateDescendantSnapshotWithVerdict(snapshot([row(20, 10, 20)]), { + sendSignal: vi.fn(), + readTable: vi.fn().mockRejectedValue(new Error('ps exploded')), + verifyMs: 200 + }) + await vi.advanceTimersByTimeAsync(400) + + await expect(pending).resolves.toBe('unverifiable') + }) + + it('does not signal a recycled descendant when identity validation is required', async () => { + const sendSignal = vi.fn() + const recycled = row(20, 10, 20, 'Tue Jul 14 13:00:00 2026') + const pending = terminateDescendantSnapshotWithVerdict( + snapshot([row(20, 10, 20, 'Tue Jul 14 12:00:00 2026')]), + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([recycled])), + requireIdentityBeforeSignal: true, + verifyMs: 100 + } + ) + + await vi.advanceTimersByTimeAsync(200) + await expect(pending).resolves.toBe('exited') + expect(sendSignal).not.toHaveBeenCalled() + }) + + it('uses row-scoped boundaries for forced cleanup in the exit verifier', async () => { + const oldBoundary = CAPTURED_AT_MS + 900 + const refreshBoundary = CAPTURED_AT_MS + 2_100 + const retained = row(20, 10, 20, 'Tue Jul 14 12:00:00 2026') + const fresh = row(30, 10, 30, 'Tue Jul 14 12:00:01 2026') + const sendSignal = vi.fn() + const pending = terminateDescendantSnapshotWithVerdict( + { + ...snapshot([retained, fresh], 10, refreshBoundary), + capturedAtMsByPid: { '20': oldBoundary, '30': refreshBoundary }, + // What a merge produces: only the refresh re-derived 30; 20 is retained. + reDerivedPids: new Set([30]) + }, + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([retained, fresh])), + requireIdentityBeforeSignal: true, + graceMs: 0, + verifyMs: 100 + } + ) + await vi.advanceTimersByTimeAsync(200) + + await expect(pending).resolves.toBe('live') + expect(sendSignal.mock.calls).toEqual([ + [20, 'SIGTERM'], + [30, 'SIGTERM'], + [30, 'SIGKILL'] + ]) + }) }) describe('createProcessTableSnapshotReader', () => { diff --git a/src/main/pty-descendant-termination.ts b/src/main/pty-descendant-termination.ts index bf254d03b56..4f56c3e977b 100644 --- a/src/main/pty-descendant-termination.ts +++ b/src/main/pty-descendant-termination.ts @@ -21,12 +21,25 @@ export type ProcessTableRow = { startedAt: string } +export type PosixProcessIdentity = Pick + export type DescendantSnapshot = { + /** Identity of the root observed in the same process-table capture. */ + root?: PosixProcessIdentity rootPgid: number | null descendants: ProcessTableRow[] - /** Wall-clock boundary for deciding whether ps's second-resolution lstart - * can safely distinguish this process from a later PID reuse. */ + /** Wall-clock boundary for an unmerged snapshot (or legacy callers). */ capturedAtMs: number + /** Per-PID identity boundaries for merged captures. */ + capturedAtMsByPid?: Readonly> + /** + * PIDs this walk re-derived from a live root. A ppid walk only reaches what + * the root actually parents, so membership is proof of ownership that owes + * nothing to `lstart`'s one-second resolution: a stranger would have to have + * been forked into our own tree, and then it is not a stranger. Rows a merge + * retained from an earlier walk are absent, and still answer to start time. + */ + reDerivedPids?: ReadonlySet } export type ProcessTableCapture = { @@ -156,9 +169,13 @@ export function collectDescendantRows( ): DescendantSnapshot { const childrenByPpid = new Map() let rootRow: ProcessTableRow | null = null + let duplicateRoot = false for (const row of table) { if (row.pid === rootPid) { - rootRow = row + // A non-atomic process-table read can contain both an old and a recycled + // root row. There is no safe identity to retain in that case. + duplicateRoot = rootRow !== null + rootRow ??= row continue } const siblings = childrenByPpid.get(row.ppid) @@ -172,7 +189,7 @@ export function collectDescendantRows( // An absent root has already exited — its real descendants reparent to pid 1 and // become unreachable by ppid, so any rows still pointing at the vacated PID are a // PID-reuse coincidence. Sweeping them could signal an unrelated process, so bail. - if (!rootRow) { + if (!rootRow || duplicateRoot) { return { rootPgid: null, descendants: [], capturedAtMs } } const descendants: ProcessTableRow[] = [] @@ -191,7 +208,13 @@ export function collectDescendantRows( queue.push(child.pid) } } - return { rootPgid: rootRow.pgid, descendants, capturedAtMs } + return { + root: { pid: rootRow.pid, startedAt: rootRow.startedAt }, + rootPgid: rootRow.pgid, + descendants, + capturedAtMs, + reDerivedPids: new Set(descendants.map((row) => row.pid)) + } } type SnapshotDeps = { @@ -316,7 +339,11 @@ export type TerminateDeps = { } export function hasUnambiguousStartIdentity(row: ProcessTableRow, capturedAtMs: number): boolean { - const startedAtMs = Date.parse(row.startedAt) + return hasUnambiguousStartTime(row.startedAt, capturedAtMs) +} + +export function hasUnambiguousStartTime(startedAt: string, capturedAtMs: number): boolean { + const startedAtMs = Date.parse(startedAt) if (!Number.isFinite(startedAtMs)) { return false } @@ -364,7 +391,10 @@ export function terminateDescendantSnapshot( for (const row of snapshot.descendants) { const live = liveTargets.get(row.pid) if ( - hasUnambiguousStartIdentity(row, snapshot.capturedAtMs) && + hasUnambiguousStartIdentity( + row, + snapshot.capturedAtMsByPid?.[String(row.pid)] ?? snapshot.capturedAtMs + ) && live?.startedAt === row.startedAt && live.pgid === row.pgid ) { diff --git a/src/main/refused-tree-kill-root-termination.test.ts b/src/main/refused-tree-kill-root-termination.test.ts new file mode 100644 index 00000000000..ada4b5942a9 --- /dev/null +++ b/src/main/refused-tree-kill-root-termination.test.ts @@ -0,0 +1,220 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { spawnMock, execFileMock, queryWindowsProcessDescendantsMock } = vi.hoisted(() => ({ + spawnMock: vi.fn(), + execFileMock: vi.fn(), + queryWindowsProcessDescendantsMock: vi.fn() +})) + +vi.mock('node:child_process', async (importOriginal) => ({ + ...(await importOriginal>()), + spawn: spawnMock, + execFile: execFileMock +})) +vi.mock('electron', () => ({ ipcMain: { handle: vi.fn(), on: vi.fn() } })) +vi.mock('./providers/windows-foreground-process-rows', () => ({ + queryWindowsProcessDescendants: queryWindowsProcessDescendantsMock +})) + +import { + getAppEnvironment, + hasAppEnvironment, + setAppEnvironment, + type AppEnvironment +} from '../shared/app-environment' +import { installMainProcessTreeKillGate } from './own-chromium-tree-kill-guard' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' +import { resetSelfInitiatedTreeKillLogForTest } from './crash-reporting/self-initiated-tree-kill-log' +import { + clearCrashBreadcrumbsForTest, + getCrashBreadcrumbSnapshot +} from './crash-reporting/crash-breadcrumb-store' +import { _resetTracerForTests, setActiveSink } from './observability/tracer' +import { terminateNotebookProcessTree } from './ipc/notebook' +import { killLocalPrecheckProcessTree } from './automations/precheck-runner' +import { killRecipeProcess } from '../shared/ephemeral-vm-recipe-process' +import { killSpawnedCommandTree } from './git/command-runner/spawned-command-tree-kill' +import { killCodexAppServerProcessTree } from './codex/codex-app-server-session' +import { signalProcessTree } from '../shared/child-process/process-tree-termination' +import { killSourceControlAgentProcess } from './text-generation/source-control-local-process' +import { terminateCodexTurnProcesses } from './codex/codex-structured-turn-processes' + +/** A pid Electron reports as one of ours: every gate below must refuse it. */ +const RENDERER_PID = 1001 + +function appEnvironment(): AppEnvironment { + return { + getPath: () => process.cwd(), + getAppPath: () => process.cwd(), + getVersion: () => '0.0.0-test', + isPackaged: () => false, + onWillQuit: () => {}, + exit: () => {}, + getAppMetrics: (() => [ + { pid: RENDERER_PID, type: 'Tab' } + ]) as unknown as AppEnvironment['getAppMetrics'] + } +} + +let previousEnvironment: AppEnvironment | null = null +let previousPlatform: PropertyDescriptor | undefined + +function setPlatform(platform: NodeJS.Platform): void { + Object.defineProperty(process, 'platform', { value: platform, configurable: true }) +} + +beforeEach(() => { + previousEnvironment = hasAppEnvironment() ? getAppEnvironment() : null + previousPlatform = Object.getOwnPropertyDescriptor(process, 'platform') + setAppEnvironment(appEnvironment()) + setActiveSink(null) + clearCrashBreadcrumbsForTest() + resetSelfInitiatedTreeKillLogForTest() + installMainProcessTreeKillGate() + spawnMock.mockReset() + execFileMock.mockReset() + queryWindowsProcessDescendantsMock.mockReset() + spawnMock.mockReturnValue({ on: vi.fn(), once: vi.fn(), unref: vi.fn(), kill: vi.fn() }) +}) + +afterEach(() => { + setProcessTreeKillGate(null) + if (previousPlatform) { + Object.defineProperty(process, 'platform', previousPlatform) + } + if (previousEnvironment) { + setAppEnvironment(previousEnvironment) + } + _resetTracerForTests() +}) + +/** + * A refusal must never become a process leak. The gate only blocks the + * pid-addressed tree walk; the root kill is addressed by the child handle, so it + * cannot reach the recycled pid we refused, and skipping it would report a + * timed-out command as stopped while its tree keeps running. + */ +describe('a refused tree-kill still terminates the root it owns', () => { + it('kills the git command root when the tree walk is refused', async () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + await killSpawnedCommandTree(child as never) + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('kills the notebook cell root when the tree walk is refused', () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + expect(terminateNotebookProcessTree(child as never)).toBeNull() + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('kills the automation precheck root when the tree walk is refused', () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + expect(killLocalPrecheckProcessTree(child as never)).toBeNull() + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('kills the ephemeral-VM recipe root when the tree walk is refused', () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + killRecipeProcess(child as never, true) + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the codex app-server root when the deadline tree walk is refused', () => { + const child = { pid: RENDERER_PID, kill: vi.fn() } + + killCodexAppServerProcessTree(child as never, { + platform: 'win32', + spawnImpl: spawnMock as never + }) + + expect(spawnMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the commit-message agent root when the tree walk is refused', async () => { + setPlatform('win32') + const child = { pid: RENDERER_PID, kill: vi.fn() } + + await killSourceControlAgentProcess(child as never) + + expect(execFileMock).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the runProcess root when the Windows arm of the shared choke point is refused', async () => { + setPlatform('win32') + const windowsChild = { pid: RENDERER_PID, kill: vi.fn(), exitCode: null, signalCode: null } + + await expect(signalProcessTree(windowsChild as never, 'SIGKILL')).resolves.toBe(false) + expect(spawnMock).not.toHaveBeenCalled() + expect(windowsChild.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('still signals the POSIX process group: a group only holds what Orca put in it', async () => { + // Same contract as main and as the other three POSIX group arms in main + // (claude-login, codex teardown, PTY sweep): record, never refuse. A stale + // `getAppMetrics()` entry must not orphan a macOS/Linux tree. + setPlatform('linux') + const posixChild = { pid: RENDERER_PID, kill: vi.fn(), exitCode: null, signalCode: null } + const processKill = vi.spyOn(process, 'kill').mockImplementation(() => true) + + await expect(signalProcessTree(posixChild as never, 'SIGKILL')).resolves.toBe(true) + expect(processKill).toHaveBeenCalledWith(-RENDERER_PID, 'SIGKILL') + expect(posixChild.kill).not.toHaveBeenCalled() + expect(getCrashBreadcrumbSnapshot()).toEqual([ + expect.objectContaining({ + name: 'self_tree_kill', + data: expect.objectContaining({ pid: RENDERER_PID, scope: 'posix-process-group' }) + }) + ]) + processKill.mockRestore() + }) +}) + +/** + * The one gated site with nothing to fall back to: the roots it kills are found + * by a process-table walk, not spawned here, so there is no child handle. A + * refusal must then be visible — the refusal crumb is written and the turn is + * reported as not cancelled — rather than resolving as if the tree had gone. + */ +describe('a refused tree-kill with no handle to fall back to', () => { + it('reports the codex turn as not cancelled and records the refused added root', async () => { + const appServerPid = 500 + const addedRoot = { + pid: RENDERER_PID, + ppid: appServerPid, + name: 'node.exe', + command: 'node', + depth: 1 + } + queryWindowsProcessDescendantsMock.mockResolvedValue([addedRoot]) + + await expect( + terminateCodexTurnProcesses(appServerPid, { platform: 'win32', identities: new Map() }) + ).resolves.toBe(false) + + expect(execFileMock).not.toHaveBeenCalled() + expect(getCrashBreadcrumbSnapshot()).toEqual([ + expect.objectContaining({ + name: 'self_tree_kill_refused_own_chromium', + data: expect.objectContaining({ pid: RENDERER_PID, site: 'codex-turn-added-roots' }) + }) + ]) + }) +}) diff --git a/src/main/repo-worktrees.ts b/src/main/repo-worktrees.ts index 228f303bfa2..f5d67523286 100644 --- a/src/main/repo-worktrees.ts +++ b/src/main/repo-worktrees.ts @@ -1,6 +1,11 @@ import type { Repo } from '../shared/repo-types' import type { GitWorktreeInfo } from '../shared/worktree/types' -import { listWorktreeGraph, listWorktrees, listWorktreesStrict } from './git/worktree' +import { + listWorktreeGraph, + listWorktrees, + listWorktreesSharedStrictAllowingTrueEmpty, + listWorktreesStrict +} from './git/worktree' import { isFolderRepo } from '../shared/repo-kind' import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' import { resolveGitRouteForHost } from './providers/execution-host-provider-dispatch' @@ -42,6 +47,29 @@ export function createFolderWorktree(repo: Repo): GitWorktreeInfo { export async function listRepoWorktrees( repo: Repo, options?: LocalRepoWorktreeListOptions +): Promise { + return listRoutedRepoWorktrees(repo, options, listWorktrees) +} + +/** + * The detected scan's listing: a Git or host failure rejects instead of softening to `[]`, so a + * failed scan cannot be published as an authoritative empty listing and prune the repo's worktrees + * (#1158's retention guard only fires when the listing admits it failed). + */ +export async function listRepoWorktreesForDetectedScan( + repo: Repo, + options?: LocalRepoWorktreeListOptions +): Promise { + return listRoutedRepoWorktrees(repo, options, listWorktreesSharedStrictAllowingTrueEmpty) +} + +async function listRoutedRepoWorktrees( + repo: Repo, + options: LocalRepoWorktreeListOptions | undefined, + listLocal: ( + repoPath: string, + options?: LocalRepoWorktreeListOptions + ) => Promise ): Promise { if (isFolderRepo(repo)) { return [createFolderWorktree(repo)] @@ -66,8 +94,8 @@ export async function listRepoWorktrees( return await route.provider.listWorktrees(repo.path) } return hasLocalRepoWorktreeListOptions(options) - ? await listWorktrees(repo.path, options) - : await listWorktrees(repo.path) + ? await listLocal(repo.path, options) + : await listLocal(repo.path) } /** diff --git a/src/main/runtime/agent-session-acquisition-failure-settlement.ts b/src/main/runtime/agent-session-acquisition-failure-settlement.ts index 7ad20397813..c790a149f31 100644 --- a/src/main/runtime/agent-session-acquisition-failure-settlement.ts +++ b/src/main/runtime/agent-session-acquisition-failure-settlement.ts @@ -4,10 +4,28 @@ import { type AgentSessionOperationOutcome } from '../../shared/agent-session-operation-ledger' import { nextAgentSessionFence } from '../../shared/agent-session-next-fence' -import type { AgentSessionRecord } from '../../shared/agent-session-record' +import type { + AgentSessionDeathEvidence, + AgentSessionRecord +} from '../../shared/agent-session-record' import { assertFence, withLease } from './agent-session-lease-transitions' import type { AgentSessionStoreState } from './agent-session-record-store-file' +/** + * How the failed attempt's provider process was accounted for. + * - `exit-proven`: cleanup observed the whole tree gone. + * - `root-exit-observed`: the owner root's exit was observed first-hand, so the + * identity this lease is keyed on is dead, but its descendants could not be + * verified. Releases the lease and says exactly that, claiming nothing more. + * - `processless`: the attempt failed before a process existed. + * - `unproven`: nothing about the process was observed; the reservation latches. + */ +export type AgentSessionAcquisitionExitProof = + | 'exit-proven' + | 'root-exit-observed' + | 'processless' + | 'unproven' + export type AgentSessionFailedAcquisitionSettlement = { sessionId: string fence: number @@ -15,7 +33,7 @@ export type AgentSessionFailedAcquisitionSettlement = { callerKey: string operationId: string outcome: Extract - exitProof: 'exit-proven' | 'processless' | 'unproven' + exitProof: AgentSessionAcquisitionExitProof now: number } @@ -83,11 +101,18 @@ export function settleFailedAgentSessionPostAcquisitionAttachment( claimStatus: 'released', lastRenewedAt: args.now, handoffOperationId: null, - deathEvidence: { - kind: 'exit-observed', - detail: 'post-acquisition cleanup proved no provider child remains', - observedAt: args.now - } + deathEvidence: + args.exitProof === 'root-exit-observed' + ? { + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable', + observedAt: args.now + } + : { + kind: 'exit-observed', + detail: 'post-acquisition cleanup proved no provider child remains', + observedAt: args.now + } }) state.records.set(args.sessionId, next) state.operations = settleAgentSessionOperation(state.operations, args) @@ -127,18 +152,29 @@ function settleFailedLease( claimStatus: 'released', lastRenewedAt: args.now, handoffOperationId: null, - deathEvidence: - args.exitProof === 'processless' - ? { - kind: 'pid-absent', - detail: 'reservation failed before spawn', - observedAt: args.now - } - : { - // Cleanup proved no child of this attempt remains; it may never have spawned. - kind: 'exit-observed', - detail: 'acquisition cleanup proved no provider child remains', - observedAt: args.now - } + deathEvidence: acquisitionDeathEvidence(args.exitProof, args.now) }) } + +/** Records only what was observed: never a tree claim the cleanup did not make. */ +function acquisitionDeathEvidence( + exitProof: AgentSessionAcquisitionExitProof, + observedAt: number +): AgentSessionDeathEvidence { + if (exitProof === 'processless') { + return { kind: 'pid-absent', detail: 'reservation failed before spawn', observedAt } + } + if (exitProof === 'root-exit-observed') { + return { + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable', + observedAt + } + } + // Cleanup proved no child of this attempt remains; it may never have spawned. + return { + kind: 'exit-observed', + detail: 'acquisition cleanup proved no provider child remains', + observedAt + } +} diff --git a/src/main/runtime/agent-session-launch-env-backfill.test.ts b/src/main/runtime/agent-session-launch-env-backfill.test.ts new file mode 100644 index 00000000000..c95705f2aa5 --- /dev/null +++ b/src/main/runtime/agent-session-launch-env-backfill.test.ts @@ -0,0 +1,90 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { AgentSessionRecordStore } from './agent-session-record-store' +import type { AgentSessionReserveRequest } from './agent-session-reservation-admission' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-launch-env' +let directory: string + +function request(overrides: Partial = {}): AgentSessionReserveRequest { + return { + sessionId: SESSION, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + provider: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/dev/.claude' }, + runtimeKind: 'native', + expectedFence: null, + spawnToken: 'spawn-a', + claimKeyId: 'key-1', + handoffOperationId: null, + probe: { outcome: 'reservation-unused' }, + operation: { + callerKey: 'client-1', + operationId: `${NOW}-00000000000000000000000000000001`, + fingerprint: 'fp-1' + }, + now: NOW, + ...overrides + } +} + +beforeEach(async () => { + directory = await mkdtemp(join(tmpdir(), 'orca-agent-session-launch-env-')) +}) + +afterEach(async () => { + await rm(directory, { recursive: true, force: true }) +}) + +describe('legacy agent session launch environment', () => { + it('durably pins the first environment resolved by a current reservation', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + await store.reserveOwner(request()) + await store.reserveOwner( + request({ + expectedFence: 1, + spawnToken: 'spawn-b', + launchEnv: { ANTHROPIC_AUTH_TOKEN: 'pinned-token' }, + operation: { + callerKey: 'client-1', + operationId: `${NOW}-00000000000000000000000000000002`, + fingerprint: 'fp-2' + } + }) + ) + + const reopened = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + expect( + (reopened.getRecord(SESSION) as { launchEnv?: Record } | null)?.launchEnv + ).toBeUndefined() + }) + + it('rejects an environment that could not be reloaded before writing it', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + const launchEnv = Object.fromEntries( + Array.from({ length: 257 }, (_, index) => [`KEY_${index}`, 'value']) + ) + + await expect(store.reserveOwner(request({ launchEnv }))).rejects.toThrow( + 'agent_session_launch_env_invalid' + ) + expect(store.getRecord(SESSION)).toBeNull() + }) + + it('rejects an overlong environment key before writing it', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + + await expect( + store.reserveOwner(request({ launchEnv: { ['K'.repeat(513)]: 'value' } })) + ).rejects.toThrow('agent_session_launch_env_invalid') + expect(store.getRecord(SESSION)).toBeNull() + }) +}) diff --git a/src/main/runtime/agent-session-record-options.test.ts b/src/main/runtime/agent-session-record-options.test.ts index a1dfb9ccdef..5795763d96c 100644 --- a/src/main/runtime/agent-session-record-options.test.ts +++ b/src/main/runtime/agent-session-record-options.test.ts @@ -31,6 +31,20 @@ it('fails option hydration before ownership can be proved', async () => { ).rejects.toThrow('model list unavailable') }) +it('drops provider-rejected persisted options before the next owner proof', async () => { + await expect( + readNativeSessionOptions({ + adapter: { + readOptions: async () => ({ models: [], current: { model: 'provider-model' } }), + readOptionRestoreFailures: () => ['permissionMode'] + }, + sessionId: SESSION, + fence: 2, + priorOptions: { permissionMode: 'retired-mode', other: 'keep' } + }) + ).resolves.toEqual({ model: 'provider-model', other: 'keep' }) +}) + it('persists resumed provider options atomically with owner proof', async () => { const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) const reserved = await store.reserveOwner({ diff --git a/src/main/runtime/agent-session-resume-args.test.ts b/src/main/runtime/agent-session-resume-args.test.ts new file mode 100644 index 00000000000..db4d0b07b8d --- /dev/null +++ b/src/main/runtime/agent-session-resume-args.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { resolveAgentSessionResumeArgs } from './agent-session-resume-args' + +describe('agent session resume arguments', () => { + it('keeps the session creation arguments after mutable defaults change', () => { + expect( + resolveAgentSessionResumeArgs({ + persistedArgs: ['--model', 'claude-created'], + defaultArgs: '--model claude-current', + shell: 'posix' + }) + ).toBe("'--model' 'claude-created'") + }) + + it('keeps an explicit empty snapshot when defaults are toggled off', () => { + expect( + resolveAgentSessionResumeArgs({ + persistedArgs: [], + defaultArgs: '--dangerously-skip-permissions', + shell: 'posix' + }) + ).toBe('') + }) + + it('uses current defaults for legacy records without a snapshot', () => { + expect( + resolveAgentSessionResumeArgs({ + defaultArgs: '--dangerously-skip-permissions', + shell: 'posix' + }) + ).toBe('--dangerously-skip-permissions') + }) +}) diff --git a/src/main/runtime/agent-session-resume-args.ts b/src/main/runtime/agent-session-resume-args.ts new file mode 100644 index 00000000000..dc276726dc6 --- /dev/null +++ b/src/main/runtime/agent-session-resume-args.ts @@ -0,0 +1,17 @@ +import type { AgentSessionLaunchArgs } from '../../shared/agent-session-record' +import { quoteStartupArg, type AgentStartupShell } from '../../shared/tui-agent-startup-shell' + +export function resolveAgentSessionResumeArgs(input: { + requestArgs?: string | null + persistedArgs?: AgentSessionLaunchArgs + defaultArgs?: string | null + shell: AgentStartupShell +}): string | null | undefined { + if (input.requestArgs !== undefined) { + return input.requestArgs + } + if (input.persistedArgs !== undefined) { + return input.persistedArgs.map((arg) => quoteStartupArg(arg, input.shell)).join(' ') + } + return input.defaultArgs +} diff --git a/src/main/runtime/claude-structured-session-integration.test.ts b/src/main/runtime/claude-structured-session-integration.test.ts new file mode 100644 index 00000000000..712543d0428 --- /dev/null +++ b/src/main/runtime/claude-structured-session-integration.test.ts @@ -0,0 +1,747 @@ +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../shared/agent-session-mutation-envelope' +import type { AgentJournalRenderItem } from '../../shared/agent-session-journal-types' +import type { AgentSessionSubscribeEvent } from '../../shared/agent-session-wire' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' +import type { + ClaudeStreamJsonConnection, + ClaudeStreamJsonConnectionHandlers, + ClaudeStreamJsonLaunch, + openClaudeStreamJsonConnection +} from '../claude/claude-stream-json-connection' +import { claudeSessionIdForOrcaSession } from '../claude/claude-structured-launch-resolution' +import { + CLAUDE_SPAWN_TOKEN_ENV, + claudeProviderHandleLink +} from '../claude/claude-structured-owner-identity' +import { attachFingerprintFields } from '../native-chat/agent-session-wire/structured-agent-session-attach' +import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' +import type { OrcaRuntimeService } from './orca-runtime' +import type { RpcRequest, RpcResponse } from './rpc/core' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { RpcDispatcher } from './rpc/dispatcher' +import { STRUCTURED_AGENT_SESSION_METHODS } from './rpc/methods/structured-agent-session' +import { + ensureStructuredAgentSessionHost, + stopStructuredAgentSessionRuntime +} from './structured-agent-session-runtime' + +const SESSION = 'claude-integration-1' +const PROVIDER_SESSION = claudeSessionIdForOrcaSession(SESSION) +const WORKSPACE = 'workspace-claude' +// Why 'runtime': this file exercises the Claude structured integration over agentSession.*, not the +// mobile surface — nothing here asserts anything mobile-specific, and its sibling integration +// suites use 'runtime' too. Mobile additionally requires the experimental structured-chat setting, +// which structured-agent-session.test.ts pins in both its satisfied and refused states. +const CLIENT = { + clientKind: 'runtime' as const, + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] +} + +const { readClaudeTranscriptLeafUuid, resolveSessionFilePath } = vi.hoisted(() => ({ + readClaudeTranscriptLeafUuid: vi.fn(), + resolveSessionFilePath: vi.fn() +})) + +vi.mock('../native-chat/session-file-resolver', () => ({ + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +})) + +type FakeClaudeConnection = Omit & { + closed: boolean + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + launch: ClaudeStreamJsonLaunch + handlers: ClaudeStreamJsonConnectionHandlers + calls: { subtype: string; params?: Record }[] + sent: Record[] +} + +function fakeClaude() { + const connections: FakeClaudeConnection[] = [] + let initializeAccount: unknown + /** A child that dies during start, with the close verdict its ladder observed. */ + let selfExit: { message: string; exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] } | null = + null + const openConnection = (async (launch, handlers = {}) => { + const connection: FakeClaudeConnection = { + launch, + handlers, + calls: [], + sent: [], + pid: 4321 + connections.length, + closed: false, + initializationResult: async () => { + connection.calls.push({ subtype: 'initialize' }) + if (selfExit) { + handlers.onExit?.(new Error(selfExit.message)) + return { models: [] } + } + handlers.onMessage?.({ + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION, + ...(connections.length === 0 ? { uuid: 'init-leaf' } : {}), + model: 'claude-sonnet-5', + apiKeySource: 'none' + }) + return { + models: [{ value: 'sonnet', displayName: 'Sonnet' }], + ...(initializeAccount === undefined ? {} : { account: initializeAccount }) + } + }, + getSettings: async () => { + connection.calls.push({ subtype: 'get_settings' }) + return { env: {} } + }, + supportedModels: async () => { + connection.calls.push({ subtype: 'list_models' }) + return [{ value: 'sonnet', displayName: 'Sonnet' }] + }, + setModel: async (model) => { + connection.calls.push({ subtype: 'set_model', params: { model } }) + }, + setPermissionMode: async (mode) => { + connection.calls.push({ subtype: 'set_permission_mode', params: { mode } }) + }, + applyFlagSettings: async (settings) => { + connection.calls.push({ subtype: 'apply_flag_settings', params: { settings } }) + }, + interrupt: async () => { + connection.calls.push({ subtype: 'interrupt', params: {} }) + return undefined + }, + cancelAsyncMessage: async () => {}, + send: async (message) => { + connection.sent.push(message) + if (message.type === 'user') { + handlers.onMessage?.({ ...message, uuid: 'user-1' }) + } + }, + exitVerdict: selfExit?.exitVerdict ?? { root: 'live', tree: 'unverifiable' }, + close: async () => { + connection.closed = true + return selfExit === null + } + } + connections.push(connection) + return connection + }) as typeof openClaudeStreamJsonConnection + const live = (): FakeClaudeConnection => { + const connection = connections.at(-1) + if (!connection) { + throw new Error('no Claude connection') + } + return connection + } + return { + connections, + openConnection, + live, + setInitializeAccount: (account: unknown) => { + initializeAccount = account + }, + setSelfExit: (exit: typeof selfExit) => { + selfExit = exit + } + } +} + +let operations = 0 +// Keep IDs unique without making each assertion depend on a wall-clock tick. +const TEST_OPERATION_TIMESTAMP = Date.now().toString() + +function operationId(): string { + operations += 1 + return `${TEST_OPERATION_TIMESTAMP}-${operations.toString(16).padStart(32, '0')}` +} + +function envelope(method: string, fields: Record, fence: number | null) { + return { + sessionId: SESSION, + clientOperationId: operationId(), + expectedRuntimeFence: fence, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function createIntentParams() { + const worktree = `id:${WORKSPACE}` + const fields = { worktree, agent: 'claude' } + return { envelope: envelope('agentSession.create', fields, null), ...fields } +} + +function ensureParams(fence: number) { + const params = { + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: WORKSPACE, + workspaceKind: 'git-worktree' as const + }, + provider: 'claude' as const, + agent: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR' as const, path: join(root, 'claude-home') }, + runtimeKind: 'native' as const, + providerHandle: { + kind: 'claude' as const, + sessionId: PROVIDER_SESSION, + leafUuid: 'assistant-leaf' + } + } + const base = { + sessionId: SESSION, + clientOperationId: operationId(), + expectedRuntimeFence: fence, + payloadFingerprint: '' + } + return { + ...params, + envelope: { + ...base, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.attach', + sessionId: SESSION, + fields: attachFingerprintFields({ ...params, envelope: base } as never) + }) + } + } +} + +function leaseOf(sessionId: string): { + claimStatus: string + runtimeFence: number + handoffStage: string | null + deathEvidence: { kind: string; detail: string } | null +} { + const host = getStructuredAgentSessionHost() as unknown as { + deps: { store: { getRecord: (id: string) => { lease: ReturnType } } } + } + return host.deps.store.getRecord(sessionId).lease +} + +function handoffParams(direction: 'to-native' | 'to-tui', fence: number) { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { + envelope: envelope('agentSession.requestHandoff', fields, fence), + ...fields + } +} + +let claude: ReturnType +let root: string +let dispatcher: RpcDispatcher +let cleanups: Map void> +let tuiOwner: StructuredTuiOwner | null +let transcriptPath: string +/** Managed-account state and configured overlay this host installs, per test. */ +let claudeAuthPolicy: ClaudeStructuredAuthPolicy +let claudeLaunchEnv: Record + +async function call(method: string, params: unknown): Promise { + const replies: RpcResponse[] = [] + const request: RpcRequest = { id: `req-${operations}`, authToken: 'token', method, params } + await dispatcher.dispatchStreaming(request, (raw) => replies.push(JSON.parse(raw)), CLIENT) + if (!replies[0]) { + throw new Error(`no reply for ${method}`) + } + return replies[0] +} + +async function ok(method: string, params: unknown): Promise { + const response = await call(method, params) + expect(response, JSON.stringify(response)).toMatchObject({ ok: true }) + const result = (response as { result: { ok: boolean; value?: T } }).result + expect(result).toMatchObject({ ok: true }) + return result.value as T +} + +async function subscribe(): Promise { + const frames: AgentSessionSubscribeEvent[] = [] + await dispatcher.dispatchStreaming( + { + id: 'subscribe-1', + authToken: 'token', + method: 'agentSession.subscribe', + params: { sessionId: SESSION } + }, + (raw) => { + const response = JSON.parse(raw) as { ok: boolean; result?: AgentSessionSubscribeEvent } + if (response.ok && response.result) { + frames.push(response.result) + } + }, + CLIENT + ) + return frames +} + +function itemsOf(frames: AgentSessionSubscribeEvent[]): AgentJournalRenderItem[] { + const items = new Map() + for (const frame of frames) { + const rows = + frame.type === 'snapshot' || frame.type === 'reset' + ? frame.page.items + : frame.type === 'batch' + ? frame.batch.items + : [] + for (const row of rows) { + items.set(row.itemId, row) + } + } + return [...items.values()] +} + +function textOf(item: AgentJournalRenderItem): string { + return item.body?.kind === 'message' + ? item.body.blocks.map((block) => (block.type === 'text' ? block.text : '')).join('') + : '' +} + +beforeEach(async () => { + operations = 0 + claudeAuthPolicy = { stripAuthEnv: false } + claudeLaunchEnv = { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + } + root = await mkdtemp(join(tmpdir(), 'orca-claude-structured-integration-')) + transcriptPath = join(root, 'claude-home', 'projects', 'workspace', `${PROVIDER_SESSION}.jsonl`) + await mkdir(join(root, 'claude-home', 'projects', 'workspace'), { recursive: true }) + resolveSessionFilePath.mockResolvedValue(transcriptPath) + // The production branch proof returns the latest descendant of the prior + // cursor; mirror that contract so structured close does not regress to a + // stale mocked head. + readClaudeTranscriptLeafUuid.mockImplementation( + async (_path: string, _providerSessionId: string, previousLeafUuid?: string | null) => + previousLeafUuid ?? 'init-leaf' + ) + claude = fakeClaude() + tuiOwner = null + cleanups = new Map() + const handoffTransport: StructuredAgentSessionHandoffTransport = { + hostLabel: 'Scripted Claude host', + launchTui: async ({ record, fence, spawnToken }) => { + const head = record.providerHandleChain.at(-1)?.handle + tuiOwner = { + terminal: { + handle: 'term-claude-tui', + tabId: 'tab-claude-tui', + paneKey: 'tab-claude-tui:leaf-claude-tui', + ptyId: 'pty-claude-tui' + }, + process: { + hostId: 'local', + pid: 7331, + processStartTimeMs: 100, + spawnToken + }, + link: claudeProviderHandleLink({ + sessionId: PROVIDER_SESSION, + leafUuid: head?.provider === 'claude' ? head.leafUuid : null, + resumed: true, + fence, + observedAt: 1 + }), + transcriptPath + } + return tuiOwner + }, + reproveTuiOwner: async ({ owner }) => { + if (owner.link.handle.provider !== 'claude' || !owner.transcriptPath) { + return owner + } + return { + ...owner, + link: claudeProviderHandleLink({ + sessionId: owner.link.handle.sessionId, + leafUuid: await readClaudeTranscriptLeafUuid(owner.transcriptPath), + resumed: true, + fence: owner.link.mintedAtFence, + observedAt: 1 + }) + } + }, + recoverTuiOwner: async () => { + if (!tuiOwner) { + throw new Error('scripted TUI owner missing') + } + return tuiOwner + }, + stopRecoveredOwner: async () => {}, + waitForTuiExit: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle', + stopFailedTuiLaunch: async () => {} + } + const runtime = { + getRuntimeId: () => 'runtime-1', + getStructuredAgentSessionCreateSupport: async () => ({ supported: true }), + resolveStructuredAgentSessionCreateIntent: async (input: { envelope: unknown }) => ({ + ...ensureParams(1), + envelope: input.envelope, + providerHandle: undefined + }), + publishStructuredAgentSessionTab: vi.fn(), + ensureStructuredAgentSessionHost: () => + ensureStructuredAgentSessionHost({ + stateDirectory: root, + hostId: 'local', + claimKeyId: 'key-1', + resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, + resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeCommand: () => '/usr/local/bin/claude', + readProcessStartTime: async (pid: number) => pid * 10, + resolveClaudeLaunchEnv: () => claudeLaunchEnv, + resolveClaudeAuthPolicy: () => claudeAuthPolicy, + openClaudeConnection: claude.openConnection, + handoffTransport + }).then(() => undefined), + registerSubscriptionCleanup: (id: string, dispose: () => void) => cleanups.set(id, dispose), + cleanupSubscription: (id: string) => cleanups.get(id)?.(), + cleanupSubscriptionsByPrefix: () => {} + } + dispatcher = new RpcDispatcher({ + runtime: runtime as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }) +}) + +afterEach(async () => { + vi.unstubAllEnvs() + await stopStructuredAgentSessionRuntime() + await rm(root, { recursive: true, force: true }) +}) + +describe('a structured Claude session over agentSession.*', () => { + it('strips ambient Anthropic auth from the child once a managed account is pinned', async () => { + claudeAuthPolicy = { stripAuthEnv: true } + claudeLaunchEnv = { ANTHROPIC_BASE_URL: 'https://gateway.example.test' } + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + vi.stubEnv('ANTHROPIC_AUTH_TOKEN', 'tok-SHELL-LEAK') + + await ok<{ fence: number }>('agentSession.create', createIntentParams()) + + const env = claude.live().launch.env + expect(env).not.toHaveProperty('ANTHROPIC_API_KEY') + expect(env).not.toHaveProperty('ANTHROPIC_AUTH_TOKEN') + expect(env).toMatchObject({ + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: join(root, 'claude-home') + }) + }) + + it('refuses a create whose configured env overrides the pinned managed account auth', async () => { + claudeAuthPolicy = { stripAuthEnv: true } + // The default overlay carries ANTHROPIC_AUTH_TOKEN, which the terminal path + // refuses at spawn-env.ts:25 rather than letting it beat the pinned account. + const refused = await call('agentSession.create', createIntentParams()) + + expect(JSON.stringify(refused)).toContain('explicit Anthropic auth environment') + // Refused before spawn: no provider child was ever opened. + expect(claude.connections).toHaveLength(0) + }) + + it('durably returns actionable sign-in guidance when initialization has no credentials', async () => { + claude.setInitializeAccount({ apiProvider: 'firstParty', tokenSource: 'none' }) + const params = createIntentParams() + + const first = await call('agentSession.create', params) + const retry = await call('agentSession.create', params) + + expect(first).toMatchObject({ + ok: true, + result: { + ok: false, + refusal: { + code: 'agent_session_operation_invalid', + message: expect.stringMatching(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + } + } + }) + expect((retry as { result: unknown }).result).toEqual((first as { result: unknown }).result) + expect(claude.connections).toHaveLength(1) + }) + + it('releases a session whose CLI self-exited during create, with its diagnostic intact', async () => { + claude.setSelfExit({ + message: 'claude stream-json exited (code 1): claude: not signed in', + // The root's death is first-hand; its descendants were never snapshottable. + exitVerdict: { root: 'exited', tree: 'unverifiable' } + }) + + const failed = await call('agentSession.create', createIntentParams()) + + expect(JSON.stringify(failed)).toContain('claude: not signed in') + const lease = leaseOf(SESSION) + // Latching here would refuse every later attach with agent_session_ownership_unknown, + // wedging a user who only needs to sign in. + expect(lease).toMatchObject({ claimStatus: 'released', handoffStage: null }) + expect(lease.deathEvidence).toMatchObject({ + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable' + }) + + claude.setSelfExit(null) + // Signing in and reopening the chat works: the reservation was not latched. + await ok<{ fence: number }>('agentSession.ensure', ensureParams(lease.runtimeFence)) + }) + + it('keeps a session reserved when a descendant of the failed start was seen alive', async () => { + claude.setSelfExit({ + message: 'claude stream-json exited (code 1): claude: not signed in', + exitVerdict: { root: 'exited', tree: 'live' } + }) + + await call('agentSession.create', createIntentParams()) + + // A live descendant still holds the provider session: releasing would hand a + // second writer to it. + expect(leaseOf(SESSION)).toMatchObject({ + claimStatus: 'reserved', + handoffStage: 'manual-recovery' + }) + claude.setSelfExit(null) + }) + + it('routes a published Claude first-hand exit through fenced host reconciliation', async () => { + await ok<{ fence: number }>('agentSession.create', createIntentParams()) + const connection = claude.live() + connection.exitVerdict = { root: 'exited', tree: 'unverifiable' } + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + + for ( + let attempt = 0; + attempt < 20 && leaseOf(SESSION).claimStatus !== 'released'; + attempt += 1 + ) { + await new Promise((resolve) => setTimeout(resolve, 5)) + } + expect(leaseOf(SESSION)).toMatchObject({ claimStatus: 'released', handoffStage: null }) + }) + + it('creates, sends, streams, approves, interrupts, and resumes from the chain head', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + const created = await ok<{ fence: number }>('agentSession.create', createIntentParams()) + expect(claude.live().launch.options).toMatchObject({ sessionId: PROVIDER_SESSION }) + expect(claude.live().launch.options.resume).toBeUndefined() + expect(claude.live().launch.env).toMatchObject({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: join(root, 'claude-home'), + [CLAUDE_SPAWN_TOKEN_ENV]: expect.any(String) + }) + // System auth: the user's own shell key is their sign-in, exactly as on the + // terminal path, and the configured overlay still wins over it. + expect(claude.live().launch.env).toMatchObject({ ANTHROPIC_API_KEY: 'sk-ant-SHELL-LEAK' }) + expect(claude.live().launch.env?.PATH ?? claude.live().launch.env?.Path).toBeTruthy() + const history = await call('agentSession.history', { + sessionId: SESSION, + direction: 'tail', + limit: 1 + }) + expect(history).toMatchObject({ + ok: true, + result: { providerSession: { key: 'session_id', id: PROVIDER_SESSION } } + }) + const stream = await subscribe() + + const body = { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'List files' }] } + const sent = await ok<{ + submission: { dispatchState: string; providerItemId: string | null } + }>('agentSession.send', { + envelope: envelope('agentSession.send', { body }, created.fence), + body + }) + expect(sent.submission).toMatchObject({ + dispatchState: 'accepted', + providerItemId: `claude:${PROVIDER_SESSION}:user-1` + }) + + claude.live().handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION, + uuid: 'assistant-leaf', + event: { type: 'content_block_delta', delta: { type: 'text_delta', text: 'Two files.' } } + }) + claude.live().handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION, + uuid: 'assistant-leaf', + parent_tool_use_id: null, + message: { role: 'assistant', content: [{ type: 'text', text: 'Two files.' }] } + }) + claude.live().handlers.onMessage?.({ + type: 'result', + subtype: 'success', + session_id: PROVIDER_SESSION, + uuid: 'result-frame-uuid' + }) + claude.live().handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION, + uuid: 'stream-event-frame-uuid', + event: { type: 'message_stop' } + }) + await getStructuredAgentSessionHost()?.flushStreamedEvents(SESSION) + expect(itemsOf(stream).find((item) => textOf(item) === 'Two files.')?.itemId).toBe( + `claude:${PROVIDER_SESSION}:assistant-leaf` + ) + + const answeredPermission = Promise.resolve( + claude.live().handlers.canUseTool?.('Bash', { command: 'ls' }, { + requestId: 'permission-1', + toolUseID: 'tool-1', + signal: new AbortController().signal + } as never) + ) + await getStructuredAgentSessionHost()?.flushStreamedEvents(SESSION) + const approval = itemsOf(stream).find((item) => item.body?.kind === 'approval') + expect(approval?.body).toMatchObject({ title: 'Allow Bash?', detail: '{"command":"ls"}' }) + await ok('agentSession.respondToApproval', { + envelope: envelope( + 'agentSession.respondTo:approval', + { + itemId: approval?.itemId, + expectedRevision: approval?.revision, + optionId: 'allow' + }, + created.fence + ), + itemId: approval?.itemId, + expectedRevision: approval?.revision, + optionId: 'allow' + }) + // Answering resolves the SDK's own canUseTool callback with the allow decision. + await expect(answeredPermission).resolves.toMatchObject({ + behavior: 'allow', + toolUseID: 'tool-1' + }) + + await expect( + ok('agentSession.cancel', { + envelope: envelope('agentSession.cancel', { turnId: 'user-1' }, created.fence), + turnId: 'user-1' + }) + ).resolves.toMatchObject({ turnId: 'user-1', cancelled: true }) + expect(claude.live().calls.at(-1)).toMatchObject({ subtype: 'interrupt' }) + + const host = getStructuredAgentSessionHost() as unknown as { + deps: { + store: { + getRecord: (sessionId: string) => { + providerHandleChain: { handle: { provider: string; leafUuid?: string | null } }[] + } + } + } + } + expect(host.deps.store.getRecord(SESSION).providerHandleChain.at(-1)?.handle).toMatchObject({ + provider: 'claude', + leafUuid: null + }) + const old = claude.live() + const resumed = await ok<{ fence: number }>('agentSession.ensure', ensureParams(created.fence)) + expect(resumed.fence).toBe(created.fence + 1) + expect(old.closed).toBe(true) + expect(resolveSessionFilePath).toHaveBeenCalledWith('claude', PROVIDER_SESSION, { + claudeProjectsDir: join(root, 'claude-home', 'projects') + }) + expect(claude.live().launch.options).toMatchObject({ + resume: PROVIDER_SESSION, + resumeSessionAt: 'assistant-leaf' + }) + expect(host.deps.store.getRecord(SESSION).providerHandleChain.at(-1)).toMatchObject({ + handle: { + provider: 'claude', + sessionId: PROVIDER_SESSION, + leafUuid: 'assistant-leaf' + }, + origin: 'resumed' + }) + }) + + it('completes a scripted native to TUI to native cycle with provider-history rehydration', async () => { + const created = await ok<{ fence: number }>('agentSession.create', createIntentParams()) + await writeFile( + transcriptPath, + [ + { + type: 'user', + uuid: 'native-user', + message: { role: 'user', content: [{ type: 'text', text: 'NATIVE_USER' }] } + }, + { + type: 'assistant', + uuid: 'native-assistant', + message: { role: 'assistant', content: [{ type: 'text', text: 'NATIVE_ASSISTANT' }] } + }, + { + type: 'user', + uuid: 'tui-user', + message: { role: 'user', content: [{ type: 'text', text: 'TUI_USER' }] } + }, + { + type: 'assistant', + uuid: 'tui-assistant', + message: { role: 'assistant', content: [{ type: 'text', text: 'TUI_ASSISTANT' }] } + }, + { type: 'last-prompt', leafUuid: 'tui-assistant' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await ok('agentSession.requestHandoff', handoffParams('to-tui', created.fence)) + const host = getStructuredAgentSessionHost()! + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui', phase: 'idle' }) + ) + expect(claude.connections[0]?.closed).toBe(true) + + const tuiFence = ( + host as unknown as { + deps: { store: { getRecord: (id: string) => { lease: { runtimeFence: number } } } } + } + ).deps.store.getRecord(SESSION).lease.runtimeFence + readClaudeTranscriptLeafUuid.mockResolvedValueOnce('tui-assistant') + await ok('agentSession.requestHandoff', handoffParams('to-native', tuiFence)) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native', phase: 'idle' }) + ) + + const frames = await subscribe() + const texts = itemsOf(frames).map(textOf).filter(Boolean) + expect(texts).toEqual( + expect.arrayContaining(['NATIVE_USER', 'NATIVE_ASSISTANT', 'TUI_USER', 'TUI_ASSISTANT']) + ) + expect(new Set(texts).size).toBe(texts.length) + expect(claude.connections).toHaveLength(2) + expect(claude.live().launch.options).toMatchObject({ resume: PROVIDER_SESSION }) + const record = ( + host as unknown as { + deps: { + store: { + getRecord: (id: string) => { + providerHandleChain: { handle: { provider: string; leafUuid?: string | null } }[] + } + } + } + } + ).deps.store.getRecord(SESSION) + expect(record.providerHandleChain.at(-1)?.handle).toMatchObject({ + provider: 'claude', + leafUuid: 'tui-assistant' + }) + }) +}) diff --git a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts index afbb70fc286..f70cf033718 100644 --- a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts +++ b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts @@ -17,6 +17,9 @@ import { resolveTuiAgentLaunchArgs, resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' +import type { AgentSessionLaunchArgs } from '../../shared/agent-session-record' +import { resolveStartupShell } from '../../shared/tui-agent-startup-shell' +import { resolveAgentSessionResumeArgs } from './agent-session-resume-args' export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntimeWithResolveWorktreeRemovalTarget { protected getAgentSessionExecutionNamespace( @@ -88,7 +91,12 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim async ensureAgentSession( request: RuntimeEnsureAgentSessionRequest, _caller: RuntimeAgentSessionRpcCaller = {}, - handoffAuthority?: { spawnToken: string; providerRoot: string; sessionId: string } + handoffAuthority?: { + spawnToken: string + providerRoot: string + sessionId: string + launchArgs?: AgentSessionLaunchArgs + } ): Promise { if (request.kind === 'automatic') { // Legacy renderer sleep records are migration evidence, not host authority. @@ -134,10 +142,12 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim agent: request.agent, providerSession: identity.providerSession, cmdOverrides: settings.agentCmdOverrides ?? {}, - agentArgs: - request.agentArgs !== undefined - ? request.agentArgs - : resolveTuiAgentLaunchArgs(request.agent, settings.agentDefaultArgs), + agentArgs: resolveAgentSessionResumeArgs({ + requestArgs: request.agentArgs, + persistedArgs: handoffAuthority?.launchArgs, + defaultArgs: resolveTuiAgentLaunchArgs(request.agent, settings.agentDefaultArgs), + shell: resolveStartupShell(platform, shell) + }), agentEnv: { ...resolveTuiAgentLaunchEnv(request.agent, settings.agentDefaultEnv), ...(handoffAuthority && request.agent === 'codex' @@ -148,6 +158,7 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim }, ompResumeFilePath: request.ompResumeFilePath, sessionOptions: this.toAgentSessionOptions(request.launchPreferences), + sessionOptionsOverrideAgentArgs: Boolean(request.launchPreferences), platform, shell, isRemote diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 42c9c7ff6d3..06391e2ae83 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -10,6 +10,7 @@ import { } from './runtime-worktree-ps-activity' import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' import { compareWorktreePs } from './runtime-worktree-status-projection' +import type { AgentSessionRecord } from '../../shared/agent-session-record' import type { Repo } from '../../shared/repo-types' import { enrichMissingRepoGitRemoteIdentities } from '../repo-git-remote-identity-enrichment' import { ensureStructuredAgentSessionHost as installStructuredAgentSessionHost } from './structured-agent-session-runtime' @@ -21,9 +22,11 @@ import { resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' +import { resolveStartupShell, tokenizeStartupCommand } from '../../shared/tui-agent-startup-shell' import { resolveCodexStructuredAppServerArgs } from '../codex/codex-structured-app-server-args' import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' import { hostname } from 'node:os' +import { claudeStructuredAuthPolicyForSettings } from '../claude-accounts/claude-structured-auth-policy' import { probeAgentSessionProcessIdentity } from './agent-session-process-identity-probe' import { structuredAgentSessionTabId } from '../../shared/structured-agent-session-projection' @@ -144,13 +147,47 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent // in a plain folder lands in the folder rather than failing to resolve. resolveWorkspacePath: async (workspaceId) => (await this.resolveRuntimeFileTarget(`id:${workspaceId}`)).worktree.path, - resolveLaunchArgs: () => this.resolveConfiguredCodexStructuredArgs(), + resolveLaunchArgs: (provider) => this.resolveConfiguredStructuredLaunchArgs(provider), resolveLaunchEnvOverlay: () => resolveTuiAgentLaunchEnv('codex', this.requireStore().getSettings().agentDefaultEnv), + resolveClaudeLaunchEnv: () => + resolveTuiAgentLaunchEnv('claude', this.requireStore().getSettings().agentDefaultEnv), + resolveClaudeAuthPolicy: () => + claudeStructuredAuthPolicyForSettings(this.requireStore().getSettings()), + // Same gate and same settings as agentSession.createSupport, re-read on every acquisition. + getClaudeManagedAccountGateSettings: () => this.requireStore().getSettings(), handoffTransport: this.createStructuredAgentSessionHandoffTransport() }) } + // Why the provider is honoured rather than assumed: Codex app-server flags are not + // Claude CLI flags, and prepending them to `claude` makes it exit on an unknown option. + protected resolveConfiguredStructuredLaunchArgs( + provider: AgentSessionRecord['provider'] + ): string[] { + if (provider === 'claude') { + return this.resolveConfiguredClaudeStructuredArgs() + } + return this.resolveConfiguredCodexStructuredArgs() + } + + protected resolveConfiguredClaudeStructuredArgs(): string[] { + const settings = this.requireStore().getSettings() + const shell = resolveStartupShell( + process.platform, + resolveLocalWindowsAgentStartupShell({ + platform: process.platform, + isRemote: false, + terminalWindowsShell: settings.terminalWindowsShell + }) + ) + const tokenized = tokenizeStartupCommand( + resolveTuiAgentLaunchArgs('claude', settings.agentDefaultArgs), + shell + ) + return tokenized.ok ? tokenized.tokens : [] + } + protected resolveConfiguredCodexStructuredArgs(): string[] { const settings = this.requireStore().getSettings() const shell = resolveLocalWindowsAgentStartupShell({ @@ -195,7 +232,7 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent tuiStatus: (owner) => this.structuredTuiStatus(owner), closeTuiOwner: (owner) => this.closeStructuredTuiOwner(owner), revealNativeSession: async ({ workspaceId, sessionId, agent = 'codex', adoptedTerminal }) => { - if (adoptedTerminal || agent !== 'codex') { + if (adoptedTerminal || (agent !== 'codex' && agent !== 'claude')) { return } await this.publishStructuredAgentSessionTab({ diff --git a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts index 5700b71d9a5..aef04bde6bc 100644 --- a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts +++ b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts @@ -3,7 +3,10 @@ import { OrcaRuntimeWithStopStructuredSessionProcess } from './orca-runtime-stop import type { AgentSessionOwnerBinding } from '../../shared/agent-session-host-authority' import { agentSessionOwnerBindingsEqual } from '../../shared/claimed-agent-pty-owner-snapshot' import { resolvePinnedCodexRolloutProof } from '../codex/codex-tui-rollout-proof' +import { supportsCodexStructuredLocation } from '../codex/codex-structured-location-support' +import { supportsClaudeStructuredLocation } from '../claude/claude-structured-location-support' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { resolveStructuredAgentSessionCreateSupport } from '../native-chat/structured-agent-session-create-support' import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' import type { AgentStatusIpcPayload } from '../../shared/agent-status-types' import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-options' @@ -12,6 +15,8 @@ import { getSystemCodexHomePath } from '../codex/codex-home-paths' import { resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' import { hasPersistedStructuredAgentSessionStore as hasPersistedStructuredAgentSessionStoreOnDisk } from './structured-agent-session-runtime' import { getProfileUserDataPath } from '../orca-profiles/profile-storage-paths' +import { homedir } from 'node:os' +import { join } from 'node:path' export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends OrcaRuntimeWithStopStructuredSessionProcess { protected async resolveRecoveredStructuredTuiTranscript(input: { @@ -45,22 +50,18 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca async getStructuredAgentSessionCreateSupport( worktreeSelector: string, - agent: 'codex' + agent: 'claude' | 'codex' ): Promise<{ supported: boolean; reason?: 'agent' | 'remote' | 'wsl' }> { const location = await this.resolveStructuredAgentSessionLocation(worktreeSelector) - await this.ensureStructuredAgentSessionHost() - if (getStructuredAgentSessionHost()?.supportsCreate(location, agent)) { - return { supported: true } - } - return { - supported: false, - reason: - location.executionHostId !== LOCAL_EXECUTION_HOST_ID - ? 'remote' - : location.wslDistro - ? 'wsl' - : 'agent' - } + return resolveStructuredAgentSessionCreateSupport({ + agent, + location, + adapterSupportsCreate: + agent === 'claude' + ? supportsClaudeStructuredLocation(location) + : supportsCodexStructuredLocation(location), + getSettings: () => this.requireStore().getSettings() + }) } protected hasProviderSessionObservationSource(): boolean { @@ -108,8 +109,23 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca async resolveStructuredAgentSessionCreateIntent(input: { envelope: { sessionId: string; clientOperationId: string } worktree: string - agent: 'codex' + agent: 'claude' | 'codex' }): Promise { + if (input.agent === 'claude') { + return this.resolveStructuredAgentSessionIntent(input, async ({ launchEnv, location }) => { + return ( + launchEnv.CLAUDE_CONFIG_DIR?.trim() || + this.accounts + .getClaudeConfigDirectory( + location.wslDistro + ? { runtime: 'wsl', wslDistro: location.wslDistro } + : { runtime: 'host' } + ) + ?.trim() || + join(homedir(), '.claude') + ) + }) + } return this.resolveStructuredAgentSessionIntent(input, async ({ workspacePath, launchEnv }) => { // A create has no process yet, so the current selection is what it must follow. const preparedHome = await this.prepareCodexStructuredLaunchFn?.({ workspacePath, launchEnv }) @@ -126,11 +142,17 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca input: { envelope: { sessionId: string; clientOperationId: string } worktree: string - agent: 'codex' + agent: 'claude' | 'codex' }, resolveAccountHomePath: (context: { workspacePath: string launchEnv: NodeJS.ProcessEnv + location: { + executionHostId: string + wslDistro: string | null + workspaceId: string + workspaceKind: 'folder' | 'git-worktree' + } }) => string | Promise ): Promise { const support = await this.getStructuredAgentSessionCreateSupport(input.worktree, input.agent) @@ -152,8 +174,8 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca provider: input.agent, agent: input.agent, accountHome: { - variable: 'CODEX_HOME', - path: await resolveAccountHomePath({ workspacePath, launchEnv }) + variable: input.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', + path: await resolveAccountHomePath({ workspacePath, launchEnv, location }) }, runtimeKind: 'native' } diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index 4466594b0dc..5b2c160f2c6 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -45,7 +45,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu } this.hydrateHeadlessMobileSessionTabsFromWorkspaceSession() for (const session of host?.listSessionTabs() ?? []) { - if (session.agent !== 'codex') { + if (session.agent !== 'codex' && session.agent !== 'claude') { continue } let sessionId = session.sessionId @@ -54,7 +54,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu } await this.publishStructuredAgentSessionTab({ ...session, - agent: 'codex', + agent: session.agent, sessionId, activate: false, notify: false @@ -65,7 +65,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu async publishStructuredAgentSessionTab(input: { workspaceId: string sessionId: string - agent: 'codex' + agent: 'claude' | 'codex' activate: boolean notify?: boolean }): Promise { @@ -105,7 +105,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu const tab: RuntimeMobileSessionAgentTab = { type: 'agent-session', id, - title: 'Codex Chat', + title: input.agent === 'claude' ? 'Claude Chat' : 'Codex Chat', sessionId: input.sessionId, agent: input.agent, isActive: input.activate diff --git a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts index 083c677e39f..af1fbc20372 100644 --- a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts +++ b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts @@ -52,4 +52,108 @@ describe('structured agent-session create intent', () => { path: '/accounts/selected/home' }) }) + + it('pins the configured Claude launch home without Codex launch preparation', async () => { + const prepareCodexStructuredLaunch = vi.fn() + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + agentDefaultEnv: { + claude: { CLAUDE_CONFIG_DIR: '/configured/claude-home' } + } + }) + } as never, + undefined, + { prepareCodexStructuredLaunch } + ) + vi.spyOn(runtime, 'getStructuredAgentSessionCreateSupport').mockResolvedValue({ + supported: true + }) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise<{ + executionHostId: string + wslDistro: null + workspaceId: string + workspaceKind: 'git-worktree' + }> + resolveRuntimeFileTarget: (selector: string) => Promise<{ + worktree: { path: string } + }> + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + internal.resolveRuntimeFileTarget = vi.fn(async () => ({ + worktree: { path: '/repos/workspace-1' } + })) + + const intent = await runtime.resolveStructuredAgentSessionCreateIntent({ + envelope: { sessionId: 'session-1', clientOperationId: 'operation-1' }, + worktree: 'id:workspace-1', + agent: 'claude' + }) + + expect(prepareCodexStructuredLaunch).not.toHaveBeenCalled() + expect(intent.accountHome).toEqual({ + variable: 'CLAUDE_CONFIG_DIR', + path: '/configured/claude-home' + }) + }) + + it('uses the managed Claude launch home before falling back to ~/.claude', async () => { + const prepareCodexStructuredLaunch = vi.fn() + const getRuntimeConfigDir = vi.fn(() => '/accounts/managed/claude-home') + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + agentDefaultEnv: { claude: {} } + }) + } as never, + undefined, + { prepareCodexStructuredLaunch } + ) + runtime.setAccountServices({ + claudeAccounts: { getRuntimeConfigDir } as never, + codexAccounts: {} as never, + rateLimits: {} as never + }) + vi.spyOn(runtime, 'getStructuredAgentSessionCreateSupport').mockResolvedValue({ + supported: true + }) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise<{ + executionHostId: string + wslDistro: null + workspaceId: string + workspaceKind: 'git-worktree' + }> + resolveRuntimeFileTarget: (selector: string) => Promise<{ + worktree: { path: string } + }> + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + internal.resolveRuntimeFileTarget = vi.fn(async () => ({ + worktree: { path: '/repos/workspace-1' } + })) + + const intent = await runtime.resolveStructuredAgentSessionCreateIntent({ + envelope: { sessionId: 'session-1', clientOperationId: 'operation-1' }, + worktree: 'id:workspace-1', + agent: 'claude' + }) + + expect(getRuntimeConfigDir).toHaveBeenCalledTimes(1) + expect(intent.accountHome).toEqual({ + variable: 'CLAUDE_CONFIG_DIR', + path: '/accounts/managed/claude-home' + }) + }) }) diff --git a/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts b/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts new file mode 100644 index 00000000000..c4333fd0a4d --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' + +type InstalledDeps = { + resolveLaunchArgs: (provider: 'claude' | 'codex') => Promise | string[] + resolveLaunchEnvOverlay: () => Record + resolveClaudeLaunchEnv?: () => Record +} + +const { installStructuredAgentSessionHost } = vi.hoisted(() => ({ + installStructuredAgentSessionHost: vi.fn(async (_deps: unknown) => ({}) as never) +})) + +vi.mock('./structured-agent-session-runtime', async (importOriginal) => ({ + ...(await importOriginal()), + ensureStructuredAgentSessionHost: installStructuredAgentSessionHost +})) + +function runtimeWith(settings: Record): OrcaRuntimeService { + return new OrcaRuntimeService({ getSettings: () => settings } as never) +} + +async function installedDeps(settings: Record): Promise { + installStructuredAgentSessionHost.mockClear() + await runtimeWith(settings).ensureStructuredAgentSessionHost() + return installStructuredAgentSessionHost.mock.calls[0]?.[0] as InstalledDeps +} + +describe('structured agent-session launch args wiring', () => { + it('resolves Claude launch args from the Claude agent defaults, not Codex flags', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { + claude: '--dangerously-skip-permissions --model opus', + codex: '--dangerously-bypass-approvals-and-sandbox' + }, + agentDefaultEnv: {} + }) + + expect(await deps.resolveLaunchArgs('claude')).toEqual([ + '--dangerously-skip-permissions', + '--model', + 'opus' + ]) + }) + + it('still resolves Codex app-server args for a Codex session', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { + claude: '--dangerously-skip-permissions', + codex: '--dangerously-bypass-approvals-and-sandbox' + }, + agentDefaultEnv: {} + }) + + const codexArgs = await deps.resolveLaunchArgs('codex') + expect(codexArgs).not.toContain('--dangerously-skip-permissions') + expect(codexArgs.length).toBeGreaterThan(0) + }) + + it('never lets a broken Codex args configuration block a Claude session', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { claude: '--model opus', codex: '--not-a-real-codex-flag' }, + agentDefaultEnv: {} + }) + + expect(await deps.resolveLaunchArgs('claude')).toEqual(['--model', 'opus']) + expect(() => deps.resolveLaunchArgs('codex')).toThrow() + }) + + it('supplies the Claude env overlay so the launch resolver does not fall back to process.env', async () => { + const deps = await installedDeps({ + agentDefaultArgs: {}, + agentDefaultEnv: { + claude: { ORCA_CLAUDE_OVERLAY: 'claude-value' }, + codex: { ORCA_CODEX_OVERLAY: 'codex-value' } + } + }) + + expect(deps.resolveClaudeLaunchEnv).toBeTypeOf('function') + expect(deps.resolveClaudeLaunchEnv?.()).toMatchObject({ + ORCA_CLAUDE_OVERLAY: 'claude-value' + }) + expect(deps.resolveClaudeLaunchEnv?.()).not.toHaveProperty('ORCA_CODEX_OVERLAY') + expect(deps.resolveLaunchEnvOverlay()).toMatchObject({ ORCA_CODEX_OVERLAY: 'codex-value' }) + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts b/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts index 9e70602f91b..89836d1e027 100644 --- a/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts +++ b/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts @@ -28,7 +28,12 @@ export class OrcaRuntimeWithStructuredAgentSessionLaunchTui extends OrcaRuntimeW presentation: 'background' }, {}, - { spawnToken, providerRoot: record.accountHome.path, sessionId: record.sessionId } + { + spawnToken, + providerRoot: record.accountHome.path, + sessionId: record.sessionId, + ...(record.launchArgs !== undefined ? { launchArgs: record.launchArgs } : {}) + } ) const terminal = launched.terminal let spawnedOwner: StructuredTuiOwner | null = null diff --git a/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts b/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts new file mode 100644 index 00000000000..64b451e9ea9 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts @@ -0,0 +1,112 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' +import { setStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import type { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const WSL_ONLY: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +/** Registered Claude accounts with none selected: ambient auth, and the UI names no host identity, + * so this must reach structured rather than silently falling back to a terminal session. */ +const ACCOUNTS_PRESENT_NONE_ACTIVE: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host'), managedAccount('host-2', 'host')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +function runtimeWithAccounts(claude: ClaudeManagedAccountGateSettings | null): OrcaRuntimeService { + // No store at all is the unreadable-settings case the gate must fail closed on. + const runtime = claude + ? new OrcaRuntimeService({ getSettings: () => claude } as never) + : new OrcaRuntimeService() + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise + ensureStructuredAgentSessionHost: () => Promise + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + // The adapter's own location answer is irrelevant here; pin it supported so only the account + // gate can refuse. + internal.ensureStructuredAgentSessionHost = vi.fn(async () => {}) + setStructuredAgentSessionHost({ + supportsCreate: () => true + } as unknown as StructuredAgentSessionHost) + return runtime +} + +afterEach(() => { + setStructuredAgentSessionHost(null) +}) + +describe('structured Claude managed-account gate', () => { + it('refuses Claude under a WSL-only managed account', async () => { + const runtime = runtimeWithAccounts(WSL_ONLY) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: false }) + }) + + it('supports Claude when accounts are registered but none is selected', async () => { + const runtime = runtimeWithAccounts(ACCOUNTS_PRESENT_NONE_ACTIVE) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: true }) + }) + + it('still supports Claude under a selected host managed account', async () => { + const runtime = runtimeWithAccounts(HOST_SELECTED) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: true }) + }) + + it('fails closed for Claude when the account runtime cannot be determined', async () => { + const runtime = runtimeWithAccounts(null) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: false }) + }) + + /** The gate is Claude's alone: Codex resolves its account separately and this lane must not + * change any Codex answer. */ + it('leaves Codex supported under the same WSL-only Claude account', async () => { + const runtime = runtimeWithAccounts(WSL_ONLY) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'codex') + ).resolves.toMatchObject({ supported: true }) + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts b/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts new file mode 100644 index 00000000000..0d9b12f7c52 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts @@ -0,0 +1,64 @@ +import { describe, expect, it, vi } from 'vitest' + +const installed = vi.hoisted(() => ({ deps: null as Record | null })) + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +vi.mock('./structured-agent-session-runtime', () => ({ + ensureStructuredAgentSessionHost: vi.fn(async (deps: Record) => { + installed.deps = deps + }) +})) + +import { OrcaRuntimeService } from './orca-runtime' +import { + readClaudeManagedAccountGateSettings, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' + +const SETTINGS = { + claudeManagedAccounts: [], + activeClaudeManagedAccountId: null, + agentDefaultEnv: {}, + agentDefaultArgs: {} +} as unknown as ClaudeManagedAccountGateSettings + +function gateSettingsGetter(): (() => ClaudeManagedAccountGateSettings) | undefined { + const deps: Record = installed.deps ?? {} + const get = deps['getClaudeManagedAccountGateSettings'] + return typeof get === 'function' ? (get as () => ClaudeManagedAccountGateSettings) : undefined +} + +/** The runtime class this wiring lives on does not typecheck its own `this` calls, so a broken or + * missing gate hookup compiles clean. Pin it behaviourally instead. */ +describe('structured Claude managed-account gate wiring', () => { + it('hands the host a gate reader that resolves the live settings', async () => { + installed.deps = null + const runtime = new OrcaRuntimeService({ getSettings: () => SETTINGS } as never) + + await runtime.ensureStructuredAgentSessionHost() + + const get = gateSettingsGetter() + expect(typeof get).toBe('function') + expect(get?.()).toBe(SETTINGS) + }) + + /** The installer composes this getter with the fail-closed reader, which is the shape the + * resolver consumes; pin that composition end to end. */ + it('composes into a null answer instead of throwing when settings cannot be read', async () => { + installed.deps = null + const runtime = new OrcaRuntimeService() + + await runtime.ensureStructuredAgentSessionHost() + + const get = gateSettingsGetter() + expect(typeof get).toBe('function') + expect(() => get?.()).toThrow() + expect(readClaudeManagedAccountGateSettings(get!)).toBeNull() + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-session-restore.test.ts b/src/main/runtime/orca-runtime-structured-session-restore.test.ts index 9e752330445..d6ec1e22782 100644 --- a/src/main/runtime/orca-runtime-structured-session-restore.test.ts +++ b/src/main/runtime/orca-runtime-structured-session-restore.test.ts @@ -325,6 +325,56 @@ describe('structured session cold restoration', () => { expect(closed.tabGroups?.[0]?.tabOrder).toEqual(['terminal-tab']) }) + it('publishes restored Claude tabs with the Claude title', async () => { + const runtime = new OrcaRuntimeService() + const publish = vi.spyOn(runtime, 'publishStructuredAgentSessionTab') + const internal = runtime as unknown as { + hasPersistedStructuredAgentSessionStore(): boolean + getKnownWorkspaceSessionWorktreeIds(): Set + hydrateHeadlessMobileSessionTabsFromWorkspaceSession(): Set + refreshMobileSessionPtyRecords(): Promise | null> + ensureStructuredAgentSessionHost(): Promise + } + internal.hasPersistedStructuredAgentSessionStore = () => true + internal.getKnownWorkspaceSessionWorktreeIds = () => new Set() + internal.hydrateHeadlessMobileSessionTabsFromWorkspaceSession = () => new Set() + internal.refreshMobileSessionPtyRecords = async () => new Set() + internal.ensureStructuredAgentSessionHost = async () => undefined + setStructuredAgentSessionHost({ + reconcileRestartLeases: async () => undefined, + restoreReadableSessions: async () => undefined, + listSessionTabs: () => [ + { + sessionId: 'agent-session:agent-session:restored-claude', + workspaceId: 'workspace-1', + agent: 'claude' + } + ] + } as never) + + await runtime.restoreStructuredAgentSessionTabs() + + expect(publish).toHaveBeenCalledWith({ + workspaceId: 'workspace-1', + sessionId: 'restored-claude', + agent: 'claude', + activate: false, + notify: false + }) + + const restored = await runtime.listMobileSessionTabs('id:workspace-1') + expect(restored.tabs).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + type: 'agent-session', + id: 'agent-session:restored-claude', + title: 'Claude Chat', + agent: 'claude' + }) + ]) + ) + }) + it('commits the host close when the renderer already removed the structured tab', async () => { const runtime = new OrcaRuntimeService() runtime.setNotifier({ diff --git a/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts b/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts new file mode 100644 index 00000000000..d15f754b244 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts @@ -0,0 +1,730 @@ +import { createHash } from 'node:crypto' +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' +import { createEphemeralAgentSessionClaimSigner } from './agent-session-claim-identity' +import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' +import { OrcaRuntimeService } from './orca-runtime' + +const { + probeAgentSessionProcessIdentity, + proveCodexTuiRollout, + readClaudeTranscriptLeafUuid, + readStructuredTuiProcessIdentity, + resolveSessionFilePath, + resolvePinnedCodexRolloutProof +} = vi.hoisted(() => ({ + probeAgentSessionProcessIdentity: vi.fn(), + proveCodexTuiRollout: vi.fn(), + readClaudeTranscriptLeafUuid: vi.fn(), + readStructuredTuiProcessIdentity: vi.fn(), + resolveSessionFilePath: vi.fn(), + resolvePinnedCodexRolloutProof: vi.fn() +})) + +vi.mock('./structured-tui-process-identity', () => ({ readStructuredTuiProcessIdentity })) +vi.mock('../codex/codex-tui-rollout-proof', () => ({ + proveCodexTuiRollout, + resolvePinnedCodexRolloutProof +})) +vi.mock('../native-chat/session-file-resolver', () => ({ + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +})) +vi.mock('./agent-session-process-identity-probe', async (importOriginal) => ({ + ...(await importOriginal()), + probeAgentSessionProcessIdentity +})) + +const WORKTREE_ID = 'repo-1::/tmp/structured-handoff' + +function notifier(revealTerminalSession: ReturnType) { + return { + worktreesChanged: vi.fn(), + reposChanged: vi.fn(), + activateWorktree: vi.fn(), + createTerminal: vi.fn(), + revealTerminalSession, + splitTerminal: vi.fn(), + renameTerminal: vi.fn(), + focusTerminal: vi.fn(), + closeTerminal: vi.fn(), + sleepWorktree: vi.fn(), + terminalFitOverrideChanged: vi.fn(), + terminalDriverChanged: vi.fn() + } +} + +describe('structured TUI launch tab binding', () => { + it('recovers a live TUI from durable owner inventory in a fresh runtime', async () => { + const namespace = { + machine: 'native:test', + principal: 'uid:1', + container: 'native', + providerRoot: '/tmp/codex-home' + } + const signer = createEphemeralAgentSessionClaimSigner('profile-test') + const claim = signer.createClaim({ + namespace, + identity: { agent: 'codex', providerSession: { key: 'session_id', id: 'thread-1' } }, + canonicalWorktreeId: WORKTREE_ID + }) + const terminalHandle = 'term_cold_owner' + const leafId = '23013912-13f8-44e5-818f-d40a1ff4e8c5' + resolvePinnedCodexRolloutProof.mockResolvedValue('/tmp/codex-home/sessions/thread-1.jsonl') + const writeAgentSessionProof = vi.fn(() => false) + const runtime = new OrcaRuntimeService(undefined, undefined, { + agentSessionClaimSigner: signer + }) + runtime.setPtyController({ + listProcesses: vi.fn(async () => [ + { + id: 'pty-cold-owner', + incarnationId: 'incarnation-1', + cwd: '/tmp/structured-handoff', + title: 'codex', + worktreeId: WORKTREE_ID, + terminalHandle, + agentSessionOwners: [ + { + claim, + generation: 'generation-1', + phase: 'live' as const, + ptyId: 'pty-cold-owner', + surface: { + worktreeId: WORKTREE_ID, + tabId: 'tab-cold-owner', + leafId, + terminalHandle + } + } + ] + } + ]), + write: () => true, + kill: () => true, + writeAgentSessionProof, + getForegroundProcess: async () => null + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + refreshMobileSessionPtyRecords(): Promise | null> + listResolvedWorktrees(): Promise + resolveTerminalWorkspaceLaunchScope(): Promise<{ + id: string + path: string + connectionId: null + repo: null + folderWorkspace: null + }> + getAgentSessionExecutionNamespace(): typeof namespace + ptysById: Map< + string, + { + launchToken: string | null + launchAgent: string | null + agentSessionOwners: unknown[] + tabId?: string | null + paneKey?: string | null + } + > + } + internal.listResolvedWorktrees = vi.fn(async () => [ + { id: WORKTREE_ID, repoId: 'repo-1', path: '/tmp/structured-handoff' } + ]) + internal.resolveTerminalWorkspaceLaunchScope = vi.fn(async () => ({ + id: WORKTREE_ID, + path: '/tmp/structured-handoff', + connectionId: null, + repo: null, + folderWorkspace: null + })) + internal.getAgentSessionExecutionNamespace = () => namespace + proveCodexTuiRollout.mockResolvedValueOnce({ + transcriptPath: '/tmp/codex-home/sessions/thread-1.jsonl' + }) + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + await internal.refreshMobileSessionPtyRecords() + const coldPty = internal.ptysById.get('pty-cold-owner')! + expect(coldPty).toMatchObject({ launchToken: null, launchAgent: null }) + expect(coldPty.agentSessionOwners).toHaveLength(1) + const runtimeId = (runtime as unknown as { runtimeId: string }).runtimeId + ;( + runtime as unknown as { + handles: Map< + string, + { + handle: string + runtimeId: string + rendererGraphEpoch: number + worktreeId: string + tabId: string + leafId: string + ptyId: string + ptyGeneration: number + } + > + } + ).handles.set(terminalHandle, { + handle: terminalHandle, + runtimeId, + rendererGraphEpoch: 0, + worktreeId: WORKTREE_ID, + tabId: 'pty:pty-cold-owner', + leafId: 'pty:pty-cold-owner', + ptyId: 'pty-cold-owner', + ptyGeneration: 0 + }) + coldPty.tabId = 'tab-cold-owner' + coldPty.paneKey = `tab-cold-owner:${leafId}` + coldPty.launchToken = 'spawn-token' + coldPty.launchAgent = 'codex' + + const owner = await internal.createStructuredAgentSessionHandoffTransport().recoverTuiOwner({ + sessionId: 'session-1', + location: { workspaceId: WORKTREE_ID, executionHostId: 'local' }, + accountHome: { variable: 'CODEX_HOME', path: namespace.providerRoot }, + providerHandleChain: [{ handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 }], + lease: { + ownerProcess: { + hostId: 'local', + pid: 4243, + processStartTimeMs: 10, + spawnToken: 'spawn-token' + }, + runtimeFence: 3 + } + } as never) + + expect(owner.terminal).toEqual({ + handle: terminalHandle, + tabId: 'tab-cold-owner', + paneKey: `tab-cold-owner:${leafId}`, + ptyId: 'pty-cold-owner' + }) + expect(proveCodexTuiRollout).toHaveBeenCalledWith( + expect.objectContaining({ + codexHome: namespace.providerRoot, + threadId: 'thread-1', + readOutput: expect.any(Function), + write: expect.any(Function) + }) + ) + expect(resolvePinnedCodexRolloutProof).not.toHaveBeenCalled() + expect(writeAgentSessionProof).not.toHaveBeenCalled() + expect(agentSessionPtyWriteGate.boundSessionId('pty-cold-owner')).toBe('session-1') + agentSessionPtyWriteGate.unbindPty('pty-cold-owner') + }) + + it('rebuilds a Claude proving link from current launch-token-bound hook evidence', async () => { + const paneKey = 'tab-claude:leaf-claude' + const spawnToken = 'claude-restart-token' + const sessionId = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' + const transcriptPath = '/tmp/claude-home/projects/worktree/session.jsonl' + const attestAgentHookCompatibilityAuthority = vi.fn(() => ({ + paneKey, + source: 'hydrated_commitment' as const + })) + const runtime = new OrcaRuntimeService(null, undefined, { + attestAgentHookCompatibilityAuthority, + getAgentProviderSessionRowsForPane: () => [ + { + paneKey, + connectionId: null, + state: 'done', + prompt: '', + agentType: 'claude', + receivedAt: Date.now() + 1000, + stateStartedAt: 10, + providerSession: { key: 'session_id', id: sessionId, transcriptPath } + } + ] + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + ptysById: Map + restoredOrchestrationAuthorityByPtyId: Map + } + internal.ptysById.set('pty-claude', { + ptyId: 'pty-claude', + worktreeId: WORKTREE_ID, + connectionId: null, + tabId: 'tab-claude', + paneKey, + launchToken: spawnToken, + launchAgent: 'claude', + connected: true + }) + resolveSessionFilePath.mockResolvedValue('/tmp/claude-home/projects/worktree/session.jsonl') + readClaudeTranscriptLeafUuid.mockResolvedValue('leaf-before-resume') + const record = { + sessionId: 'session-1', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/tmp/claude-home' }, + providerHandleChain: [ + { + linkId: 'claude-old', + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-before-resume' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + lease: { + runtimeFence: 3, + ownerProcess: { + hostId: 'local', + pid: 4343, + processStartTimeMs: 20, + spawnToken + }, + provenHandleLinkId: null + } + } as never + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + const transport = internal.createStructuredAgentSessionHandoffTransport() + const recovered = await transport.recoverTuiOwner(record) + expect(recovered).toMatchObject({ + transcriptPath, + link: { + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-before-resume' }, + origin: 'resumed', + mintedAtFence: 3 + } + }) + expect(recovered.link.linkId).not.toBe('claude-old') + + const pty = internal.ptysById.get('pty-claude') as { + launchToken: string | null + } + pty.launchToken = null + const dispatchAuthority = runtime.getOrchestrationDispatchAuthority(recovered.terminal.handle)! + internal.restoredOrchestrationAuthorityByPtyId.set('pty-claude', { + ptyId: 'pty-claude', + worktreeId: WORKTREE_ID, + terminalHandle: recovered.terminal.handle, + paneKey: recovered.terminal.paneKey, + processIncarnation: dispatchAuthority.processIncarnation, + hostScope: dispatchAuthority.hostScope + }) + + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toMatchObject({ + paneKey, + terminalHandle: recovered.terminal.handle, + processIncarnation: dispatchAuthority.processIncarnation + }) + expect(attestAgentHookCompatibilityAuthority).toHaveBeenCalledWith({ + paneKey, + launchTokenHash: createHash('sha256').update(spawnToken).digest('hex'), + connectionId: null, + terminalProvenance: 'restored' + }) + attestAgentHookCompatibilityAuthority.mockReturnValueOnce(null as never) + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toBeNull() + expect(agentSessionPtyWriteGate.boundSessionId('pty-claude')).toBe('session-1') + agentSessionPtyWriteGate.unbindPty('pty-claude') + }) + + it('requires restored hook attestation after the runtime restarts', async () => { + const paneKey = 'tab-restored:leaf-restored' + const spawnToken = 'restored-token' + const sessionId = '019fd532-7c11-7a90-b6de-4e1a2c3d5f62' + const transcriptPath = '/tmp/claude-home/projects/worktree/restored.jsonl' + const attestAgentHookCompatibilityAuthority = vi.fn(() => ({ + paneKey, + source: 'hydrated_commitment' as const + })) + const runtime = new OrcaRuntimeService(null, undefined, { + attestAgentHookCompatibilityAuthority, + getAgentProviderSessionRowsForPane: () => [ + { + paneKey, + connectionId: null, + state: 'done', + prompt: '', + agentType: 'claude', + receivedAt: Date.now() + 1000, + stateStartedAt: 10, + providerSession: { key: 'session_id', id: sessionId, transcriptPath } + } + ] + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + ptysById: Map + restoredOrchestrationAuthorityByPtyId: Map + } + internal.ptysById.set('pty-restored', { + ptyId: 'pty-restored', + worktreeId: WORKTREE_ID, + connectionId: null, + tabId: 'tab-restored', + paneKey, + launchToken: spawnToken, + launchAgent: 'claude', + connected: true + }) + resolveSessionFilePath.mockResolvedValue('/tmp/claude-home/projects/worktree/restored.jsonl') + readClaudeTranscriptLeafUuid.mockResolvedValue('leaf-restored') + const record = { + sessionId: 'session-restored', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/tmp/claude-home' }, + providerHandleChain: [ + { + linkId: 'claude-restored', + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-restored' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + lease: { + runtimeFence: 4, + ownerProcess: { + hostId: 'local', + pid: 4545, + processStartTimeMs: 30, + spawnToken + }, + provenHandleLinkId: null + } + } as never + + const recovered = await internal + .createStructuredAgentSessionHandoffTransport() + .recoverTuiOwner(record) + const restoredPty = internal.ptysById.get('pty-restored') as { + launchToken: string | null + } + restoredPty.launchToken = null + const dispatchAuthority = runtime.getOrchestrationDispatchAuthority(recovered.terminal.handle)! + internal.restoredOrchestrationAuthorityByPtyId.set('pty-restored', { + ptyId: 'pty-restored', + worktreeId: WORKTREE_ID, + terminalHandle: recovered.terminal.handle, + paneKey: recovered.terminal.paneKey, + processIncarnation: dispatchAuthority.processIncarnation, + hostScope: dispatchAuthority.hostScope + }) + + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toMatchObject({ + paneKey, + terminalHandle: recovered.terminal.handle, + processIncarnation: dispatchAuthority.processIncarnation + }) + expect(attestAgentHookCompatibilityAuthority).toHaveBeenCalledWith({ + paneKey, + launchTokenHash: createHash('sha256').update(spawnToken).digest('hex'), + connectionId: null, + terminalProvenance: 'restored' + }) + attestAgentHookCompatibilityAuthority.mockReturnValueOnce(null as never) + await expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toBeNull() + agentSessionPtyWriteGate.unbindPty('pty-restored') + }) + + it('proves the published launch tab before returning its revealed renderer binding', async () => { + let explicitStatus: { + state: 'working' | 'done' + prompt: string + receivedAt: number + stateStartedAt: number + paneKey: string + terminalHandle: string + } | null = null + const revealTerminalSession = vi.fn( + (_worktreeId: string, _options: { tabId?: string; leafId?: string; ptyId?: string }) => + Promise.resolve({ tabId: 'tab-renderer' }) + ) + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + disabledTuiAgents: [], + agentCmdOverrides: {}, + agentDefaultArgs: { + codex: '-m gpt-5.6-sol -c model_reasoning_effort=high' + }, + agentDefaultEnv: {} + }) + } as never, + undefined, + { + getAgentStatusSnapshot: () => (explicitStatus ? [explicitStatus as never] : []) + } + ) + runtime.setNotifier(notifier(revealTerminalSession) as never) + const spawn = vi.fn().mockResolvedValue({ id: 'pty-structured', pid: 4242 }) + runtime.setPtyController({ + spawn, + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + }) + + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + resolveTerminalWorkspaceLaunchScope(): Promise<{ + id: string + path: string + connectionId: null + repo: null + folderWorkspace: null + }> + markLocalWorkspaceTrustedForAgent(): void + waitForTerminal(): Promise + waitForAdoptedStructuredTuiProof(): Promise<{ transcriptPath?: string }> + waitForStructuredTuiPtyExit(): Promise + closeTerminal(handle: string): Promise + handles: Map< + string, + { + rendererGraphEpoch: number + tabId: string + leafId: string + } + > + graphStatus: 'ready' + } + internal.resolveTerminalWorkspaceLaunchScope = vi.fn(async () => ({ + id: WORKTREE_ID, + path: '/tmp/structured-handoff', + connectionId: null, + repo: null, + folderWorkspace: null + })) + internal.markLocalWorkspaceTrustedForAgent = vi.fn() + const waitForTerminal = vi.fn(async () => ({})) + internal.waitForTerminal = waitForTerminal + const waitForAdoptedStructuredTuiProof = vi.fn(async () => { + const snapshot = await runtime.listMobileSessionTabs(`id:${WORKTREE_ID}`) + expect(snapshot.tabs).toContainEqual( + expect.objectContaining({ + type: 'terminal', + parentTabId: expect.any(String), + leafId: expect.any(String), + ptyId: 'pty-structured', + terminal: expect.any(String) + }) + ) + expect(revealTerminalSession).not.toHaveBeenCalled() + return { transcriptPath: '/tmp/rollout.jsonl' } + }) + internal.waitForAdoptedStructuredTuiProof = waitForAdoptedStructuredTuiProof + const waitForStructuredTuiPtyExit = vi.fn(async () => {}) + internal.waitForStructuredTuiPtyExit = waitForStructuredTuiPtyExit + const closeTerminal = vi.fn(async () => undefined) + internal.closeTerminal = closeTerminal + readStructuredTuiProcessIdentity.mockResolvedValue({ + hostId: 'local', + pid: 4243, + processStartTimeMs: 10, + spawnToken: 'spawn-token' + }) + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + const transport = internal.createStructuredAgentSessionHandoffTransport() + const onSpawned = vi.fn(async () => {}) + const owner = await transport.launchTui({ + record: { + sessionId: 'session-1', + location: { workspaceId: WORKTREE_ID, executionHostId: 'local' }, + accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex-home' }, + launchArgs: ['--search'], + options: { model: 'gpt-5.6-terra', effort: 'medium' }, + providerHandleChain: [ + { handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 } + ] + } as never, + fence: 3, + spawnToken: 'spawn-token', + onSpawned + }) + + const reveal = revealTerminalSession.mock.calls[0]?.[1] as { + tabId: string + leafId: string + } + expect(owner.terminal).toMatchObject({ + tabId: 'tab-renderer', + paneKey: `${reveal.tabId}:${reveal.leafId}`, + ptyId: 'pty-structured' + }) + expect(waitForTerminal).toHaveBeenCalledWith( + expect.any(String), + expect.objectContaining({ condition: 'tui-idle' }) + ) + expect(waitForAdoptedStructuredTuiProof).toHaveBeenCalledOnce() + expect(onSpawned).toHaveBeenCalledWith( + expect.objectContaining({ + terminal: expect.objectContaining({ ptyId: 'pty-structured' }), + process: expect.objectContaining({ spawnToken: 'spawn-token' }) + }) + ) + expect(onSpawned.mock.invocationCallOrder[0]).toBeLessThan( + waitForTerminal.mock.invocationCallOrder[0]! + ) + expect(waitForAdoptedStructuredTuiProof.mock.invocationCallOrder[0]).toBeLessThan( + revealTerminalSession.mock.invocationCallOrder[0]! + ) + const launchCommand = spawn.mock.calls[0]?.[0]?.command + expect(launchCommand).toContain("'-m' 'gpt-5.6-terra'") + expect(launchCommand).toContain("'-c' 'model_reasoning_effort=medium'") + expect(launchCommand).toContain("'--search'") + expect(launchCommand).not.toContain('gpt-5.6-sol') + expect(launchCommand).not.toContain('model_reasoning_effort=high') + + Object.assign(internal.handles.get(owner.terminal.handle)!, { + rendererGraphEpoch: -1, + tabId: 'tab-retired', + leafId: 'leaf-retired' + }) + internal.graphStatus = 'ready' + + explicitStatus = { + state: 'working', + prompt: '', + receivedAt: Date.now(), + stateStartedAt: Date.now(), + paneKey: owner.terminal.paneKey, + terminalHandle: owner.terminal.handle + } + expect(transport.tuiStatus(owner)).toBe('busy') + await expect( + transport.waitForTuiIdleOrExit(owner, new AbortController().signal) + ).resolves.toBeNull() + + explicitStatus = { ...explicitStatus, state: 'done', receivedAt: Date.now() } + expect(transport.tuiStatus(owner)).toBe('idle') + await expect(transport.waitForTuiIdleOrExit(owner, new AbortController().signal)).resolves.toBe( + 'idle' + ) + + explicitStatus = null + const livePty = ( + runtime as unknown as { + ptysById: Map< + string, + { + tailBuffer: string[] + tailPartialLine: string + preview: string + lastAgentStatus: null + lastAgentStatusObservedLive: boolean + } + > + } + ).ptysById.get('pty-structured')! + Object.assign(livePty, { + tailBuffer: [ + 'OpenAI Codex (v0.147.0)', + 'model: gpt-5.6-terra', + 'directory: /tmp/structured-handoff' + ], + tailPartialLine: '', + preview: '', + lastAgentStatus: null, + lastAgentStatusObservedLive: false + }) + expect(transport.tuiStatus(owner)).toBe('idle') + await expect(transport.waitForTuiIdleOrExit(owner, new AbortController().signal)).resolves.toBe( + 'idle' + ) + + const pty = ( + runtime as unknown as { + ptysById: Map + } + ).ptysById.get('pty-structured')! + pty.launchToken = null + const persistedRecord = { + sessionId: 'session-1', + providerHandleChain: [{ handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 }], + lease: { ownerProcess: owner.process, provenHandleLinkId: owner.link.linkId } + } as never + + const rebound = await transport.reproveTuiOwner({ record: persistedRecord, owner }) + expect(rebound.terminal).toMatchObject({ + ptyId: 'pty-structured', + tabId: owner.terminal.tabId, + paneKey: owner.terminal.paneKey + }) + expect(rebound.terminal.handle).not.toBe(owner.terminal.handle) + await transport.waitForTuiExit(rebound) + expect(waitForStructuredTuiPtyExit).toHaveBeenCalledWith('pty-structured') + expect(waitForAdoptedStructuredTuiProof).toHaveBeenCalledOnce() + + await expect(transport.closeTuiOwner?.(rebound)).resolves.toEqual({ + transcriptPath: '/tmp/rollout.jsonl' + }) + expect(closeTerminal).toHaveBeenCalledWith(rebound.terminal.handle) + + explicitStatus = null + pty.connected = false + await expect( + transport.waitForTuiIdleOrExit(rebound, new AbortController().signal) + ).resolves.toBe('exited') + await expect(transport.stopFailedTuiLaunch?.(rebound)).resolves.toBeUndefined() + }) + + it('reveals Claude structured native sessions into the mobile graph', async () => { + const runtime = new OrcaRuntimeService() + const publish = vi.spyOn(runtime, 'publishStructuredAgentSessionTab') + const focusEditorTab = vi.fn() + runtime.setNotifier({ focusEditorTab } as never) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + } + + await internal.createStructuredAgentSessionHandoffTransport().revealNativeSession?.({ + workspaceId: WORKTREE_ID, + sessionId: 'session-claude', + agent: 'claude' + }) + + expect(publish).toHaveBeenCalledWith( + expect.objectContaining({ + workspaceId: WORKTREE_ID, + sessionId: 'session-claude', + agent: 'claude', + activate: false + }) + ) + expect(focusEditorTab).toHaveBeenCalledWith( + 'structured-agent-session-session-claude', + WORKTREE_ID + ) + }) +}) diff --git a/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap b/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap index d92228517aa..8b7ffc115a7 100644 --- a/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap +++ b/src/main/runtime/orchestration/__snapshots__/preamble.test.ts.snap @@ -10,6 +10,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. === CLI COMMANDS === +\`\`\`sh # Report the terminal task outcome (REQUIRED exactly once). # # RULE: --body must be a 3-sentence executive summary (what you did, @@ -71,6 +72,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. # Check for messages from the coordinator: orca orchestration check --terminal term_WORKER +\`\`\` === AFTER YOU SEND worker_done === diff --git a/src/main/runtime/orchestration/coordinator-task-dispatch.ts b/src/main/runtime/orchestration/coordinator-task-dispatch.ts index 685552cc2e9..e058d9cdc98 100644 --- a/src/main/runtime/orchestration/coordinator-task-dispatch.ts +++ b/src/main/runtime/orchestration/coordinator-task-dispatch.ts @@ -132,7 +132,7 @@ export async function dispatchTaskToWorker(params: { let gateContext = '' if (gates.length > 0) { const latest = gates.at(-1)! - gateContext = `\n\n--- DECISION GATE RESOLVED ---\nQuestion: ${latest.question}\nResolution: ${latest.resolution}\n---\n` + gateContext = `\n\n--- DECISION GATE RESOLVED ---\nQuestion: ${latest.question}\nResolution: ${latest.resolution}\n\n---\n` } try { diff --git a/src/main/runtime/orchestration/db/database-file-permissions.ts b/src/main/runtime/orchestration/db/database-file-permissions.ts index 1bbd640ba77..dd2e49e77a5 100644 --- a/src/main/runtime/orchestration/db/database-file-permissions.ts +++ b/src/main/runtime/orchestration/db/database-file-permissions.ts @@ -1,17 +1,5 @@ -import { chmodSync, existsSync } from 'node:fs' +import { hardenSqliteDatabaseFiles } from '../../../sqlite/harden-database-files' export function hardenOrchestrationDatabaseFiles(dbPath: (string & {}) | ':memory:'): void { - if (dbPath === ':memory:' || process.platform === 'win32') { - // Why: Windows protects these files through Orca's current-user-only userData DACL; POSIX mode bits are inert there. - return - } - for (const path of [dbPath, `${dbPath}-wal`, `${dbPath}-shm`]) { - try { - if (existsSync(path)) { - chmodSync(path, 0o600) - } - } catch { - // Why: best-effort — a mount that rejects chmod (SSHFS, some network shares) must not fail DB startup. - } - } + hardenSqliteDatabaseFiles(dbPath) } diff --git a/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts b/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts index 26f82410cfe..8e36605aeb5 100644 --- a/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts +++ b/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts @@ -131,6 +131,32 @@ describe('orchestration hot-path statement compilation', () => { } }) + // Why: getTask sits on the dispatch/lifecycle path and listTasks runs several times per + // coordinator tick, so a wildcard there recompiles on every call the same way the publish did. + it('compiles the task lookup and listing SQL exactly once across repeated calls', () => { + const db = openDatabase(':memory:') + seedDispatchedWorker(db) + const seeded = db.listTasks() + expect(seeded.length).toBeGreaterThan(0) + const taskId = seeded[0].id + + const compiled = trackCompiledSql(db) + for (let call = 0; call < 5; call += 1) { + db.getTask(taskId) + db.listTasks() + db.listTasks({ ready: true }) + db.listTasks({ status: 'pending' }) + db.listTasks({ runId: seeded[0].run_id }) + } + + const compilationsPerSql = new Map() + for (const sql of compiled) { + compilationsPerSql.set(sql, (compilationsPerSql.get(sql) ?? 0) + 1) + } + expect([...compilationsPerSql].filter(([, count]) => count > 1)).toEqual([]) + expect(compiled.filter((sql) => WILDCARD_PROJECTION.test(sql))).toEqual([]) + }) + // Why: `SELECT *` is what made these statements uncacheable, and a retained wildcard is the only // way node:sqlite could build a row from stale column names after another connection's ALTER. // Seeds on one connection and publishes on a second so every compilation here is hot-path SQL. diff --git a/src/main/runtime/orchestration/db/tasks/task-store.ts b/src/main/runtime/orchestration/db/tasks/task-store.ts index 69316765a74..8bcc74ca3e5 100644 --- a/src/main/runtime/orchestration/db/tasks/task-store.ts +++ b/src/main/runtime/orchestration/db/tasks/task-store.ts @@ -79,9 +79,13 @@ export function createTask( runId, depsJson ) - return this.db.prepare('SELECT * FROM tasks WHERE id = ?').get(id) as TaskRow + return this.db.prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE id = ?`).get(id) as TaskRow } +// Why wildcard-free: SyncDatabase refuses to cache any statement containing `*`, so a `SELECT *` +// here recompiles on every call — including the hot dispatch lookups and the coordinator poll. +const TASK_COLUMN_LIST = selectColumns(TASK_COLUMNS) + // Why: hoisted and wildcard-free so the per-publish lineage lookup hits the SyncDatabase statement cache. const TASK_RUNTIME_LINEAGE_SQL = `SELECT ${selectColumns(TASK_COLUMNS, 't')}, creator.id AS creator_dispatch_id, @@ -113,7 +117,9 @@ export function getTask( dispatchRunId?: string ): TaskRow | TaskRuntimeLineageRow | undefined { if (dispatchRunId === undefined) { - return this.db.prepare('SELECT * FROM tasks WHERE id = ?').get(id) as TaskRow | undefined + return this.db.prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE id = ?`).get(id) as + | TaskRow + | undefined } return this.db.prepare(TASK_RUNTIME_LINEAGE_SQL).get(dispatchRunId, id) as | TaskRuntimeLineageRow @@ -128,20 +134,26 @@ export function listTasks( const runParams: Database.BindValue[] = filter?.runId ? [filter.runId] : [] if (filter?.ready) { return this.db - .prepare(`SELECT * FROM tasks WHERE ${runWhere}status = 'ready' ORDER BY created_at`) + .prepare( + `SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE ${runWhere}status = 'ready' ORDER BY created_at` + ) .all(...runParams) as TaskRow[] } if (filter?.status) { return this.db - .prepare(`SELECT * FROM tasks WHERE ${runWhere}status = ? ORDER BY created_at`) + .prepare( + `SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE ${runWhere}status = ? ORDER BY created_at` + ) .all(...runParams, filter.status) as TaskRow[] } if (filter?.runId) { return this.db - .prepare('SELECT * FROM tasks WHERE run_id = ? ORDER BY created_at') + .prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE run_id = ? ORDER BY created_at`) .all(filter.runId) as TaskRow[] } - return this.db.prepare('SELECT * FROM tasks ORDER BY created_at').all() as TaskRow[] + return this.db + .prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks ORDER BY created_at`) + .all() as TaskRow[] } // Why: the correlated indexed lookup avoids materializing every retained Dispatch before filtering Tasks. @@ -195,7 +207,7 @@ export function listTasksWithDispatch( // Why: runs in the status-update transaction, so a completed task never leaves its ready children unpromoted. export function promoteReadyTasks(this: OrchestrationDb, completedTaskId: string): void { const candidates = this.db - .prepare("SELECT * FROM tasks WHERE status = 'pending'") + .prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE status = 'pending'`) .all() as TaskRow[] for (const task of candidates) { diff --git a/src/main/runtime/orchestration/preamble.test.ts b/src/main/runtime/orchestration/preamble.test.ts index 79e06f5a55a..57b8b35f266 100644 --- a/src/main/runtime/orchestration/preamble.test.ts +++ b/src/main/runtime/orchestration/preamble.test.ts @@ -1,4 +1,6 @@ import { spawnSync } from 'node:child_process' +import remarkParse from 'remark-parse' +import { unified } from 'unified' import { describe, expect, it } from 'vitest' import { buildDispatchPreamble } from './preamble' @@ -23,6 +25,22 @@ function afterWorkerDoneSection(result: string) { return result.slice(sectionStart, sectionEnd) } +function cliFence(result: string): string { + const match = result.match(/=== CLI COMMANDS ===\n\n```sh\n([\s\S]*?)\n```/) + expect(match).not.toBeNull() + return match?.[1] ?? '' +} + +function markdownBlocks(result: string) { + const tree = unified().use(remarkParse).parse(result) + return { + headings: tree.children.filter((node) => node.type === 'heading'), + codeBlocks: tree.children.filter((node) => node.type === 'code') + } +} + +const driftParams = { base: 'origin/main', behind: 3, recentSubjects: ['fix: a', 'feat: b'] } + describe('buildDispatchPreamble', () => { it('substitutes template variables', () => { const result = buildDispatchPreamble(baseParams()) @@ -58,25 +76,43 @@ describe('buildDispatchPreamble', () => { { timeout: 15_000 }, () => { const result = buildDispatchPreamble(baseParams()) - // Why: feeding `bash -n` the full preamble falsely fails on apostrophes - // in the surrounding prose. Slice between the CLI markers and strip - // shell-style comment lines so we only syntax-check the commands. - const cliStart = result.indexOf('=== CLI COMMANDS ===') - const cliEnd = result.indexOf('=== AFTER YOU SEND worker_done ===') - expect(cliStart).toBeGreaterThan(-1) - expect(cliEnd).toBeGreaterThan(cliStart) - const block = result.slice(cliStart, cliEnd) - const stripped = block - .split('\n') - .filter((line) => !line.trim().startsWith('#')) - .filter((line) => !line.trim().startsWith('===')) - .join('\n') - - const check = spawnSync('bash', ['-n'], { input: stripped, encoding: 'utf8' }) + const check = spawnSync('bash', ['-n'], { input: cliFence(result), encoding: 'utf8' }) expect(check.status).toBe(0) } ) + it('fences shell comments so Markdown does not promote them to headings', () => { + const result = buildDispatchPreamble(baseParams()) + const { headings, codeBlocks } = markdownBlocks(result) + + expect(headings).toHaveLength(0) + expect(codeBlocks).toHaveLength(1) + expect(codeBlocks[0]).toMatchObject({ lang: 'sh', value: cliFence(result) }) + }) + + // Why: a `---` rule directly under a paragraph is a setext H2, so the optional + // sections' closing rules must not turn their last sentence into a heading. + it('renders no Markdown headings when the sub-dispatch and drift sections are present', () => { + const result = buildDispatchPreamble( + baseParams({ canDispatchSubWorkers: true, baseDrift: driftParams }) + ) + const { headings, codeBlocks } = markdownBlocks(result) + + expect(headings).toHaveLength(0) + expect(codeBlocks).toHaveLength(2) + expect(codeBlocks[1]).toMatchObject({ lang: 'sh' }) + expect(codeBlocks[1].value).toContain('orchestration worker-start --task ') + expect(result).toContain('able to dispatch further.\n\n---') + expect(result).toContain('before starting.\n\n---') + }) + + it('sub-dispatch fence passes bash -n', { timeout: 15_000 }, () => { + const result = buildDispatchPreamble(baseParams({ canDispatchSubWorkers: true })) + const { codeBlocks } = markdownBlocks(result) + const check = spawnSync('bash', ['-n'], { input: codeBlocks[1].value, encoding: 'utf8' }) + expect(check.status).toBe(0) + }) + it('includes heartbeat CLI block with taskId and dispatchId and 5-minute cadence', () => { const result = buildDispatchPreamble(baseParams()) expect(result).toContain('--type heartbeat') diff --git a/src/main/runtime/orchestration/preamble.ts b/src/main/runtime/orchestration/preamble.ts index 7d426184954..d4519f154b9 100644 --- a/src/main/runtime/orchestration/preamble.ts +++ b/src/main/runtime/orchestration/preamble.ts @@ -59,6 +59,7 @@ export function buildDispatchPreamble(params: PreambleParams): string { ? ` --dispatch-capability ${params.dispatchCapability}` : '' + // Why: fencing keeps shell comments executable to agents without turning them into Chat UI headings. const header = `You are working inside Orca, a multi-agent IDE. You are a dispatched worker. Your coordinator's terminal handle is: ${params.coordinatorHandle} Your task ID is: ${params.taskId} @@ -68,6 +69,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. === CLI COMMANDS === +\`\`\`sh # Report the terminal task outcome (REQUIRED exactly once). # # RULE: --body must be a 3-sentence executive summary (what you did, @@ -129,6 +131,7 @@ Slack, GitHub comments, or any other channel to reach a human during the run. # Check for messages from the coordinator: ${cli} orchestration check --terminal ${params.workerHandle} +\`\`\` ${postDoneInstructions}` @@ -193,6 +196,8 @@ work under the new Dispatch; ignore stale follow-ups from the settled task.` // Why the whole section is omitted rather than softened when nesting is off: a // worker told it "usually cannot" delegate still tries, then reports the refusal // as a blocker. +// Why fenced + blank line before the closing `---`: unfenced `` are stripped as raw +// HTML by the Chat UI, and a rule directly under a paragraph is a setext H2 (giant last sentence). function buildSubDispatchSection(cli: string): string { return ` @@ -200,13 +205,16 @@ function buildSubDispatchSection(cli: string): string { You may dispatch sub-workers for this task. Bind your own Run first, then create and start each one: +\`\`\`sh ${cli} orchestration run-create --objective "" --json ${cli} orchestration task-create --spec "" --json ${cli} orchestration worker-start --task --worktree current --agent --json +\`\`\` You own those sub-workers: wait for their worker_done, and do not report your own until they have settled. Nesting is capped, so a sub-worker of yours may not be able to dispatch further. + ---` } @@ -221,5 +229,6 @@ ${subjects} If any look relevant to your task, either pull them in (\`git pull --rebase ${drift.base}\` or equivalent) or escalate to the coordinator before starting. + ---` } diff --git a/src/main/runtime/orchestration/worker-output-archive-bounding.test.ts b/src/main/runtime/orchestration/worker-output-archive-bounding.test.ts new file mode 100644 index 00000000000..a45cc6c0531 --- /dev/null +++ b/src/main/runtime/orchestration/worker-output-archive-bounding.test.ts @@ -0,0 +1,41 @@ +import { describe, expect, it } from 'vitest' +import { boundArchiveLines } from './worker-output-archive' + +const TERMINAL_ARCHIVE_MAX_CHARS = 262_144 + +function totalCost(lines: string[]): number { + return lines.reduce((sum, line) => sum + line.length + 1, 0) +} + +describe('boundArchiveLines', () => { + it('returns the original array untouched when the tail already fits', () => { + const lines = ['one', 'two', 'three'] + const bounded = boundArchiveLines(lines) + expect(bounded.truncated).toBe(false) + expect(bounded.lines).toBe(lines) + }) + + it('keeps the newest lines in order and reports truncation', () => { + const lines = Array.from({ length: 40_000 }, (_, index) => `line ${index}`) + const bounded = boundArchiveLines(lines) + expect(bounded.truncated).toBe(true) + expect(totalCost(bounded.lines)).toBeLessThanOrEqual(TERMINAL_ARCHIVE_MAX_CHARS) + expect(bounded.lines.at(-1)).toBe(lines.at(-1)) + expect(bounded.lines).toEqual(lines.slice(lines.length - bounded.lines.length)) + }) + + it('truncates a single oversized line from its tail', () => { + const bounded = boundArchiveLines(['x'.repeat(TERMINAL_ARCHIVE_MAX_CHARS * 2)]) + expect(bounded.truncated).toBe(true) + expect(bounded.lines).toHaveLength(1) + expect(bounded.lines[0]).toHaveLength(TERMINAL_ARCHIVE_MAX_CHARS - 1) + }) + + it('bounds a blank-line flood in linear time', () => { + // The char budget admits ~262k blank lines; an unshift-per-line build was ~4.3s here. + const startedAt = performance.now() + const bounded = boundArchiveLines(Array.from({ length: 300_000 }, () => '')) + expect(bounded.lines).toHaveLength(TERMINAL_ARCHIVE_MAX_CHARS) + expect(performance.now() - startedAt).toBeLessThan(500) + }) +}) diff --git a/src/main/runtime/orchestration/worker-output-archive.ts b/src/main/runtime/orchestration/worker-output-archive.ts index f6f2b52ecb1..55d16467269 100644 --- a/src/main/runtime/orchestration/worker-output-archive.ts +++ b/src/main/runtime/orchestration/worker-output-archive.ts @@ -114,7 +114,7 @@ export async function captureWorkerOutputArchive(args: { } } -function boundArchiveLines(lines: string[]): { lines: string[]; truncated: boolean } { +export function boundArchiveLines(lines: string[]): { lines: string[]; truncated: boolean } { let total = 0 for (const line of lines) { total += line.length + 1 @@ -122,18 +122,21 @@ function boundArchiveLines(lines: string[]): { lines: string[]; truncated: boole if (total <= TERMINAL_ARCHIVE_MAX_CHARS) { return { lines, truncated: false } } - const kept: string[] = [] + // Collected newest-first and reversed once: unshift per line is O(n^2) and the + // char budget admits ~260k blank lines. + const keptReversed: string[] = [] let budget = TERMINAL_ARCHIVE_MAX_CHARS for (let index = lines.length - 1; index >= 0; index -= 1) { const cost = lines[index].length + 1 if (cost > budget) { - if (kept.length === 0 && budget > 1) { - kept.unshift(lines[index].slice(-(budget - 1))) + if (keptReversed.length === 0 && budget > 1) { + keptReversed.push(lines[index].slice(-(budget - 1))) } break } - kept.unshift(lines[index]) + keptReversed.push(lines[index]) budget -= cost } - return { lines: kept, truncated: true } + keptReversed.reverse() + return { lines: keptReversed, truncated: true } } diff --git a/src/main/runtime/relay/desktop-relay-service.ts b/src/main/runtime/relay/desktop-relay-service.ts index 55b993bd5a6..def786e7758 100644 --- a/src/main/runtime/relay/desktop-relay-service.ts +++ b/src/main/runtime/relay/desktop-relay-service.ts @@ -6,6 +6,7 @@ import type { PairingGetEndpointsResult, PairingProvisionRelayParams } from '../../../shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import { readRelayAuthContext } from './relay-auth-context' import { RelayAuthCoordinator } from './relay-auth-coordinator' import { RelaySessionBroker, type RelayBrokerStatus } from './relay-session-broker' @@ -112,7 +113,7 @@ export class DesktopRelayService { this.refreshDemand() } - fenceAndCloseNow(): void { + fenceAndCloseNow(hostCloseReason?: RelayHostCloseReason): void { // Why: a fence must be hard — a surviving liveness tick could catch the // window between the pre-sign-out fence and the profile wipe and briefly // resurrect a broker. The next auth mutation re-arms via refreshDemand. @@ -120,7 +121,7 @@ export class DesktopRelayService { clearInterval(this.livenessTimer) this.livenessTimer = null } - this.coordinator.fenceAndCloseNow() + this.coordinator.fenceAndCloseNow(hostCloseReason) } async createPairingRelay( diff --git a/src/main/runtime/relay/relay-auth-coordinator.ts b/src/main/runtime/relay/relay-auth-coordinator.ts index ae439a320d9..a7db0e3a81e 100644 --- a/src/main/runtime/relay/relay-auth-coordinator.ts +++ b/src/main/runtime/relay/relay-auth-coordinator.ts @@ -1,3 +1,7 @@ +import { + RELAY_HOST_CLOSE_REASON, + type RelayHostCloseReason +} from '../../../shared/relay-host-close-reason' import type { RelayBrokerStatus } from './relay-session-broker' import { RelayHttpError, shouldRetryRelayConnectionError } from './relay-http-client' @@ -14,7 +18,7 @@ export type RelayAuthContext = { } export type CoordinatedRelayBroker = { - closeNow(): void + closeNow(hostCloseReason?: RelayHostCloseReason): void isLive?(): boolean } @@ -78,13 +82,16 @@ export class RelayAuthCoordinator { void reconcile } - fenceAndCloseNow(): void { + // hostCloseReason names an auth loss the phone should be told about. Quit, + // relaunch and every other fence pass nothing, so the control socket dies + // abruptly exactly as before and the cell records no cause. + fenceAndCloseNow(hostCloseReason?: RelayHostCloseReason): void { ++this.authEpoch this.cancelLinger() this.cancelRetry() this.retryAttempt = 0 this.invalidatePendingOwnerships() - this.invalidateOwnership() + this.invalidateOwnership(hostCloseReason) this.options.onStatus('offline') } @@ -146,7 +153,11 @@ export class RelayAuthCoordinator { if (!context || !context.relayEntitled) { this.cancelLinger() this.retryAttempt = 0 - this.invalidateOwnership() + // Why only the null case: readContext throws on transient failures and + // returns null solely when the cloud session is gone (absent, or cleared + // by a 401). A present-but-unentitled context is still a signed-in + // desktop, and "sign in to reconnect" would be wrong advice for it. + this.invalidateOwnership(context ? undefined : RELAY_HOST_CLOSE_REASON.SIGNED_OUT) this.options.onStatus('offline') return } @@ -271,12 +282,12 @@ export class RelayAuthCoordinator { return context.accessToken } - private invalidateOwnership(): void { + private invalidateOwnership(hostCloseReason?: RelayHostCloseReason): void { const ownership = this.ownership this.ownership = null if (ownership) { ownership.valid = false - ownership.broker?.closeNow() + ownership.broker?.closeNow(hostCloseReason) } } diff --git a/src/main/runtime/relay/relay-auth-host-close-reason.test.ts b/src/main/runtime/relay/relay-auth-host-close-reason.test.ts new file mode 100644 index 00000000000..7d10e207a33 --- /dev/null +++ b/src/main/runtime/relay/relay-auth-host-close-reason.test.ts @@ -0,0 +1,122 @@ +import { describe, expect, it, vi } from 'vitest' +import { RELAY_HOST_CLOSE_REASON } from '../../../shared/relay-host-close-reason' +import { RelayAuthCoordinator, type RelayAuthContext } from './relay-auth-coordinator' + +const context: RelayAuthContext = { + identity: { userId: 'user-1', profileId: 'profile-1', organizationId: 'org-1' }, + accessToken: 'access-1', + relayEntitled: true +} + +function coordinatorOver(readContext: () => Promise) { + const broker = { closeNow: vi.fn() } + const coordinator = new RelayAuthCoordinator({ + readContext, + openBroker: async () => broker, + onStatus: vi.fn() + }) + return { broker, coordinator } +} + +describe('relay control close reason', () => { + it('names the sign-out when the cloud session is gone', async () => { + let current: RelayAuthContext | null = context + const { broker, coordinator } = coordinatorOver(async () => current) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + current = null + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(broker.closeNow).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('names the sign-out on the explicit pre-sign-out fence', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.fenceAndCloseNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + expect(broker.closeNow).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('stays silent on quit, which fences without a reason', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.fenceAndCloseNow() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + it('stays silent on stop, which is teardown rather than auth loss', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.stop() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + // A signed-in desktop that merely lost the entitlement must not tell the + // phone to sign in — the copy would be wrong and the user has nothing to do. + it('stays silent when the session survives but the entitlement is gone', async () => { + let current: RelayAuthContext = context + const { broker, coordinator } = coordinatorOver(async () => current) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + current = { ...context, relayEntitled: false } + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + it('stays silent when demand drops and the broker lingers out', async () => { + let demanded = true + const broker = { closeNow: vi.fn() } + const coordinator = new RelayAuthCoordinator({ + readContext: async () => context, + hasDemand: () => demanded, + openBroker: async () => broker, + onStatus: vi.fn(), + lingerMs: 5 + }) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + demanded = false + coordinator.reconcile() + await vi.waitFor(() => expect(broker.closeNow).toHaveBeenCalled()) + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + // Replacing a stale broker is a reconnect, not a sign-out. + it('stays silent when an identity switch replaces the broker', async () => { + let current = context + const brokers: { closeNow: ReturnType }[] = [] + const coordinator = new RelayAuthCoordinator({ + readContext: async () => current, + openBroker: async () => { + const broker = { closeNow: vi.fn() } + brokers.push(broker) + return broker + }, + onStatus: vi.fn() + }) + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + current = { ...context, identity: { ...context.identity, organizationId: 'org-2' } } + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(brokers[0]?.closeNow).toHaveBeenCalledWith(undefined) + }) +}) diff --git a/src/main/runtime/relay/relay-control-client.ts b/src/main/runtime/relay/relay-control-client.ts index 7e742173f72..968f05795b2 100644 --- a/src/main/runtime/relay/relay-control-client.ts +++ b/src/main/runtime/relay/relay-control-client.ts @@ -1,6 +1,7 @@ import { randomUUID } from 'node:crypto' import WebSocket, { type RawData } from 'ws' import { MOBILE_RELAY_CLOSE_CODE } from '../../../shared/mobile-relay-close-codes' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { E2EEKeypair } from '../e2ee-keypair' import { RelayConnectionOpenMessageSchema, @@ -8,6 +9,7 @@ import { RelayHostChallengeMessageSchema, RelayHostHelloAckMessageSchema, RelayPingMessageSchema, + encodeRelayHostHello, parseRelayControlMessage, type RelayConnectionOpenMessage, type RelayDrainMessage, @@ -21,6 +23,7 @@ import { RELAY_CONTROL_SILENCE_LIMIT_MS, RelayControlSilenceWatchdog } from './relay-control-silence-watchdog' +import { closeRelayControlSocket } from './relay-control-socket-close' import { controlWebSocketUrl } from './relay-control-url' type RelayControlState = 'idle' | 'opening' | 'proving' | 'active' | 'draining' | 'closed' @@ -166,7 +169,7 @@ export class RelayControlClient { return this.requests.confirmResume(reqId, basisConnId, (payload) => this.sendActive(payload)) } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { const wasConnecting = this.state === 'opening' || this.state === 'proving' this.state = 'closed' this.silenceWatchdog.stop() @@ -175,8 +178,9 @@ export class RelayControlClient { this.clearConnectPromise() } this.requests.rejectAll(new Error('relay_control_closed')) - this.socket?.terminate() + const socket = this.socket this.socket = null + closeRelayControlSocket(socket, hostCloseReason) } private sendHostHello(): void { @@ -185,19 +189,9 @@ export class RelayControlClient { } this.state = 'proving' this.socket.send( - JSON.stringify({ - type: 'host-hello', - v: 1, - relayHostId: this.options.relayHostId, - assignmentEpoch: this.options.assignmentEpoch, - hostPublicKeyB64: this.options.keypair.publicKeyB64, - appVersion: this.options.appVersion, - ...(this.options.previousGeneration === undefined - ? {} - : { previousGeneration: this.options.previousGeneration }), - ...(this.options.controlResumeSecret - ? { controlResumeSecret: this.options.controlResumeSecret } - : {}) + encodeRelayHostHello({ + ...this.options, + hostPublicKeyB64: this.options.keypair.publicKeyB64 }) ) } diff --git a/src/main/runtime/relay/relay-control-close-reason.test.ts b/src/main/runtime/relay/relay-control-close-reason.test.ts new file mode 100644 index 00000000000..0d9aba54f88 --- /dev/null +++ b/src/main/runtime/relay/relay-control-close-reason.test.ts @@ -0,0 +1,92 @@ +import { createHash } from 'node:crypto' +import { afterEach, describe, expect, it, vi } from 'vitest' +import nacl from 'tweetnacl' +import { WebSocketServer, type WebSocket } from 'ws' +import { RELAY_HOST_CLOSE_REASON } from '../../../shared/relay-host-close-reason' +import { RelayControlClient } from './relay-control-client' + +type ObservedClose = { code: number; reason: string } + +describe('RelayControlClient close reason', () => { + const servers: WebSocketServer[] = [] + const clients: RelayControlClient[] = [] + + afterEach(async () => { + for (const client of clients.splice(0)) { + client.closeNow() + } + await Promise.all( + servers.splice(0).map( + (server) => + new Promise((resolve) => { + for (const socket of server.clients) { + socket.terminate() + } + server.close(() => resolve()) + }) + ) + ) + }) + + async function connectedClient(): Promise<{ + client: RelayControlClient + closed: Promise + }> { + const server = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) + servers.push(server) + await new Promise((resolve) => server.once('listening', resolve)) + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('expected TCP relay test server') + } + const accepted = new Promise((resolve) => server.once('connection', resolve)) + const keypair = nacl.box.keyPair() + const client = new RelayControlClient({ + cellUrl: `http://127.0.0.1:${address.port}`, + relayJwt: 'scoped-token', + relayHostId: createHash('sha256').update(keypair.publicKey).digest('base64url').slice(0, 16), + assignmentEpoch: 1, + identity: { userId: 'user-1', profileId: 'profile-1', organizationId: 'org-1' }, + keypair: { ...keypair, publicKeyB64: Buffer.from(keypair.publicKey).toString('base64') }, + appVersion: '1.2.3', + onConnectionOpen: vi.fn(), + onDrain: vi.fn(), + onClose: vi.fn() + }) + clients.push(client) + // The handshake never completes here; only the transport close matters. + void client.connect().catch(() => {}) + const socket = await accepted + // A pong proves the client socket left CONNECTING; closeNow can only send a + // close frame from OPEN, and that is the state a real sign-out fences from. + await new Promise((resolve) => { + socket.once('pong', () => resolve()) + socket.ping() + }) + const closed = new Promise((resolve) => { + socket.once('close', (code, reason) => resolve({ code, reason: reason.toString() })) + }) + return { client, closed } + } + + it('delivers the sign-out reason to the cell', async () => { + const { client, closed } = await connectedClient() + + client.closeNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + await expect(closed).resolves.toEqual({ + code: 1000, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + }) + + // Every non-auth close keeps today's abrupt terminate, so a cell can never + // read a quit, a rotation or a sleep as a sign-out. + it('closes abruptly with no reason when none is given', async () => { + const { client, closed } = await connectedClient() + + client.closeNow() + + await expect(closed).resolves.toEqual({ code: 1006, reason: '' }) + }) +}) diff --git a/src/main/runtime/relay/relay-control-origin.ts b/src/main/runtime/relay/relay-control-origin.ts index 4d42da47977..3a33e1617e6 100644 --- a/src/main/runtime/relay/relay-control-origin.ts +++ b/src/main/runtime/relay/relay-control-origin.ts @@ -8,6 +8,7 @@ import type { RelayDrainMessage, RelayHostHelloAckMessage } from './relay-control-protocol' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { RelayIdentity } from './relay-session-broker-contract' import type { RelayAssignment } from './relay-http-client' @@ -144,7 +145,7 @@ export class RelayControlOrigin { } } - async close(): Promise { + async close(hostCloseReason?: RelayHostCloseReason): Promise { if (this.closed) { return } @@ -154,7 +155,7 @@ export class RelayControlOrigin { } this.retiredControlTimers.clear() for (const control of this.controls) { - control.closeNow() + control.closeNow(hostCloseReason) } this.controls.clear() this.activeControl = null @@ -166,8 +167,8 @@ export class RelayControlOrigin { } } - closeNow(): void { - void this.close() + closeNow(hostCloseReason?: RelayHostCloseReason): void { + void this.close(hostCloseReason) } private async openControl(overrides?: { diff --git a/src/main/runtime/relay/relay-control-protocol.ts b/src/main/runtime/relay/relay-control-protocol.ts index c94797f4815..75bf3a390e2 100644 --- a/src/main/runtime/relay/relay-control-protocol.ts +++ b/src/main/runtime/relay/relay-control-protocol.ts @@ -143,3 +143,29 @@ export function parseRelayControlMessage(raw: RawData): Record return null } } + +export type RelayHostHello = { + relayHostId: string + assignmentEpoch: number + hostPublicKeyB64: string + appVersion: string + previousGeneration?: number + controlResumeSecret?: string +} + +// Optional members are omitted rather than sent as undefined: the cell parses +// host-hello strictly and an explicit null is not the same as absent. +export function encodeRelayHostHello(hello: RelayHostHello): string { + return JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hello.relayHostId, + assignmentEpoch: hello.assignmentEpoch, + hostPublicKeyB64: hello.hostPublicKeyB64, + appVersion: hello.appVersion, + ...(hello.previousGeneration === undefined + ? {} + : { previousGeneration: hello.previousGeneration }), + ...(hello.controlResumeSecret ? { controlResumeSecret: hello.controlResumeSecret } : {}) + }) +} diff --git a/src/main/runtime/relay/relay-control-socket-close.ts b/src/main/runtime/relay/relay-control-socket-close.ts new file mode 100644 index 00000000000..2984b075425 --- /dev/null +++ b/src/main/runtime/relay/relay-control-socket-close.ts @@ -0,0 +1,26 @@ +import type WebSocket from 'ws' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' + +const NORMAL_CLOSE_CODE = 1000 +const REASONED_CLOSE_FLUSH_MS = 1_000 + +// hostCloseReason: only auth loss names itself. Every other control close +// (rotation, drain, quit, sleep) stays an abrupt terminate, so the cell learns +// nothing and can never read a restart as a sign-out. A named close has to +// reach the cell as a real close frame, but the fence must still be hard — +// bound the handshake and then terminate. +export function closeRelayControlSocket( + socket: WebSocket | null, + hostCloseReason?: RelayHostCloseReason +): void { + if (!socket) { + return + } + if (hostCloseReason && socket.readyState === socket.OPEN) { + socket.close(NORMAL_CLOSE_CODE, hostCloseReason) + const timer = setTimeout(() => socket.terminate(), REASONED_CLOSE_FLUSH_MS) + timer.unref?.() + return + } + socket.terminate() +} diff --git a/src/main/runtime/relay/relay-origin-pool.ts b/src/main/runtime/relay/relay-origin-pool.ts index a95d8f21340..acd2f90292e 100644 --- a/src/main/runtime/relay/relay-origin-pool.ts +++ b/src/main/runtime/relay/relay-origin-pool.ts @@ -4,8 +4,10 @@ import type { MobileSocketWiring } from '../rpc/mobile-socket-wiring' import { RelayControlOrigin } from './relay-control-origin' import type { RelayControlClient } from './relay-control-client' import type { RelayDrainMessage } from './relay-control-protocol' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import { RelayDrainRetrySchedule } from './relay-drain-retry-schedule' import { RelayHttpError, requestRelayAssignment, type RelayAssignment } from './relay-http-client' +import { relayRenewalDelayMs } from './relay-renewal-jitter' import type { RelayBrokerStatus, RelayIdentity } from './relay-session-broker-contract' import type { RelayRegion } from './relay-region-preference' @@ -79,7 +81,7 @@ export class RelayOriginPool { } } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { if (this.closed) { return } @@ -94,7 +96,7 @@ export class RelayOriginPool { } this.drainTimers.clear() for (const origin of this.origins) { - origin.closeNow() + origin.closeNow(hostCloseReason) } this.origins.clear() this.drainingOrigins.clear() @@ -244,8 +246,7 @@ export class RelayOriginPool { } const now = (this.options.now ?? Date.now)() const random = this.options.random ?? Math.random - const earlyMs = 60_000 + Math.floor(random() * 60_001) - const delay = Math.max(0, origin.controlLeaseExpiresAt - earlyMs - now) + const delay = relayRenewalDelayMs(origin.controlLeaseExpiresAt, now, random) this.rotationTimer = setTimeout(() => void this.rebindActiveControl(origin), delay) } diff --git a/src/main/runtime/relay/relay-renewal-jitter.test.ts b/src/main/runtime/relay/relay-renewal-jitter.test.ts new file mode 100644 index 00000000000..5bbc0994836 --- /dev/null +++ b/src/main/runtime/relay/relay-renewal-jitter.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from 'vitest' +import { + RELAY_RENEWAL_JITTER_RATIO, + RELAY_RENEWAL_SAFETY_MARGIN_MS, + relayRenewalDelayMs +} from './relay-renewal-jitter' + +const LEASE_MS = 55 * 60_000 +const latest = LEASE_MS - RELAY_RENEWAL_SAFETY_MARGIN_MS +const base = latest / (1 + RELAY_RENEWAL_JITTER_RATIO) + +describe('relay renewal jitter', () => { + it('keeps every sample inside the jitter band and before the safety margin', () => { + let seed = 1 + const random = (): number => { + seed = (seed * 1103515245 + 12345) % 2147483648 + return seed / 2147483648 + } + const samples: number[] = [] + for (let i = 0; i < 20_000; i++) { + samples.push(relayRenewalDelayMs(LEASE_MS, 0, random)) + } + for (const sample of samples) { + expect(sample).toBeGreaterThanOrEqual(Math.floor(base * (1 - RELAY_RENEWAL_JITTER_RATIO))) + expect(sample).toBeLessThanOrEqual(latest) + // The renewal never lands inside the margin, so it never races expiry. + expect(LEASE_MS - sample).toBeGreaterThanOrEqual(RELAY_RENEWAL_SAFETY_MARGIN_MS) + } + const mean = samples.reduce((total, sample) => total + sample, 0) / samples.length + expect(Math.abs(mean - base) / base).toBeLessThan(0.005) + }) + + it('spreads a same-second cohort over minutes instead of one second', () => { + const delays = Array.from({ length: 1000 }, (_, index) => + relayRenewalDelayMs(LEASE_MS, 0, () => index / 999) + ) + const spread = Math.max(...delays) - Math.min(...delays) + expect(spread).toBeGreaterThan(9 * 60_000) + }) + + it('pins the band ends to the base interval', () => { + expect(relayRenewalDelayMs(LEASE_MS, 0, () => 0)).toBe( + Math.floor(base * (1 - RELAY_RENEWAL_JITTER_RATIO)) + ) + expect(relayRenewalDelayMs(LEASE_MS, 0, () => 0.5)).toBe(Math.floor(base)) + expect(relayRenewalDelayMs(LEASE_MS, 0, () => 1)).toBe(latest) + }) + + it('renews immediately once the lease is inside the safety margin', () => { + expect(relayRenewalDelayMs(RELAY_RENEWAL_SAFETY_MARGIN_MS, 0, () => 1)).toBe(0) + expect(relayRenewalDelayMs(0, 60_000, () => 1)).toBe(0) + }) + + it('measures the delay from now, not from the epoch', () => { + expect(relayRenewalDelayMs(LEASE_MS + 1_000_000, 1_000_000, () => 0.5)).toBe(Math.floor(base)) + }) +}) diff --git a/src/main/runtime/relay/relay-renewal-jitter.ts b/src/main/runtime/relay/relay-renewal-jitter.ts new file mode 100644 index 00000000000..8b1f1ec6412 --- /dev/null +++ b/src/main/runtime/relay/relay-renewal-jitter.ts @@ -0,0 +1,25 @@ +// Why: a cell recreate reconnects a whole cohort inside one second. Every host +// in it then took its lease from the same second and, with only a 60s-wide +// spread, renewed inside the same second ~54 minutes later — a self-sustaining +// fleet-wide reconnect burst. Full +/-10% jitter spreads that cohort over +// minutes instead. +export const RELAY_RENEWAL_JITTER_RATIO = 0.1 + +// The latest jittered renewal still lands this far before expiry. +export const RELAY_RENEWAL_SAFETY_MARGIN_MS = 90_000 + +// Why: the relay accepts a rebind at any point in the lease and resets the full +// TTL from it (cloud/apps/relay/src/host-session-registry.ts:736-743), so +// renewing early is free; only renewing late is fatal (:997 drains an expired +// lease). That asymmetry is why the base is shrunk to fit the upward jitter +// rather than the jittered value being clipped at the margin. +export function relayRenewalDelayMs(expiresAt: number, now: number, random: () => number): number { + const remaining = expiresAt - now + const latest = remaining - RELAY_RENEWAL_SAFETY_MARGIN_MS + if (latest <= 0) { + return 0 + } + const base = latest / (1 + RELAY_RENEWAL_JITTER_RATIO) + const jittered = base * (1 + (random() * 2 - 1) * RELAY_RENEWAL_JITTER_RATIO) + return Math.max(0, Math.min(Math.floor(jittered), latest)) +} diff --git a/src/main/runtime/relay/relay-session-broker.ts b/src/main/runtime/relay/relay-session-broker.ts index e12e09119a0..cd83545e9ca 100644 --- a/src/main/runtime/relay/relay-session-broker.ts +++ b/src/main/runtime/relay/relay-session-broker.ts @@ -6,6 +6,7 @@ import type { MobileRelayEndpoint, PairingProvisionRelayParams } from '../../../shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { DeviceCredentialInstallAuthorization } from './relay-control-requests' import { deriveRelayHostId, @@ -15,6 +16,7 @@ import { type RelayAssignment } from './relay-http-client' import { RelayOriginPool } from './relay-origin-pool' +import { relayRenewalDelayMs } from './relay-renewal-jitter' import type { RelayBrokerStatus, RelaySessionBrokerOptions } from './relay-session-broker-contract' export type { RelayBrokerStatus } from './relay-session-broker-contract' @@ -180,7 +182,7 @@ export class RelaySessionBroker { return result } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { if (this.closed) { return } @@ -190,7 +192,7 @@ export class RelaySessionBroker { clearTimeout(this.refreshTimer) this.refreshTimer = null } - this.originPool.closeNow() + this.originPool.closeNow(hostCloseReason) if (publishOffline) { this.options.onStatus('offline') } @@ -242,8 +244,7 @@ export class RelaySessionBroker { } const now = (this.options.now ?? Date.now)() const random = this.options.random ?? Math.random - const earlyMs = 60_000 + Math.floor(random() * 60_001) - const delay = Math.max(0, authorization.expiresAt - earlyMs - now) + const delay = relayRenewalDelayMs(authorization.expiresAt, now, random) this.refreshTimer = setTimeout(() => void this.refreshAuthorization(), delay) } diff --git a/src/main/runtime/rpc/e2ee-channel-v2.test.ts b/src/main/runtime/rpc/e2ee-channel-v2.test.ts index f9e602ced41..b26b057abaf 100644 --- a/src/main/runtime/rpc/e2ee-channel-v2.test.ts +++ b/src/main/runtime/rpc/e2ee-channel-v2.test.ts @@ -156,6 +156,25 @@ describe('E2EEChannel v2', () => { }) }) + it('forwards post-auth capability-shaped frames without mutating authenticated capabilities', () => { + const ctx = setup() + const { schedule } = startV2(ctx) + const onMessage = vi.fn() + ctx.channel.onMessage(onMessage) + authenticate(ctx, schedule) + + const capabilityFrame = JSON.stringify({ + type: 'e2ee_client_capabilities', + v: 1, + clientCapabilities: ['agent-session.structured.v1'] + }) + ctx.channel.handleRawMessage(clientText(capabilityFrame, schedule, 1n)) + + expect(ctx.channel.clientCapabilities).toEqual([]) + expect(onMessage).toHaveBeenCalledOnce() + expect(onMessage.mock.calls[0]?.[0]).toBe(capabilityFrame) + }) + it('rejects legacy downgrade and runtime-only capability metadata when mobile v2 is required', () => { const legacy = setup() legacy.channel.handleRawMessage( diff --git a/src/main/runtime/rpc/methods/clipboard.test.ts b/src/main/runtime/rpc/methods/clipboard.test.ts index 118b21c766a..c0d224b85c6 100644 --- a/src/main/runtime/rpc/methods/clipboard.test.ts +++ b/src/main/runtime/rpc/methods/clipboard.test.ts @@ -1,6 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { RpcDispatcher } from '../dispatcher' -import type { RpcRequest } from '../core' +import type { RpcRequest, RpcResponse } from '../core' import type { OrcaRuntimeService } from '../../orca-runtime' import { CLIPBOARD_IMAGE_MAX_BASE64_CHARS, @@ -21,6 +21,10 @@ import { CLIPBOARD_METHODS, resetClipboardImageUploadsForTest } from './clipboard' +import { + hasMobileClipboardImagePath, + resetMobileClipboardImageProvenanceForTest +} from '../mobile-clipboard-image-provenance' function makeRequest(method: string, params?: unknown): RpcRequest { return { id: 'req-1', authToken: 'tok', method, params } @@ -31,15 +35,36 @@ function makeDispatcher(): RpcDispatcher { return new RpcDispatcher({ runtime, methods: CLIPBOARD_METHODS }) } +async function callMobile( + dispatcher: RpcDispatcher, + method: string, + params: unknown, + clientId = 'device-a' +): Promise { + const replies: RpcResponse[] = [] + await dispatcher.dispatchStreaming( + makeRequest(method, params), + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + { clientKind: 'mobile', clientId } + ) + const response = replies[0] + if (!response) { + throw new Error(`no reply for ${method}`) + } + return response +} + describe('clipboard RPC methods', () => { beforeEach(() => { saveClipboardImageBufferAsTempFile.mockReset() resetClipboardImageUploadsForTest() + resetMobileClipboardImageProvenanceForTest() }) afterEach(() => { vi.useRealTimers() resetClipboardImageUploadsForTest() + resetMobileClipboardImageProvenanceForTest() }) it('saves browser-provided clipboard image bytes on the runtime host', async () => { @@ -64,6 +89,37 @@ describe('clipboard RPC methods', () => { }) }) + it('records a successful direct mobile upload for only the authenticated client', async () => { + const path = '/tmp/orca-paste-image.png' + saveClipboardImageBufferAsTempFile.mockResolvedValue(path) + const dispatcher = makeDispatcher() + + await expect( + callMobile(dispatcher, 'clipboard.saveImageAsTempFile', { + contentBase64: Buffer.from('png-bytes').toString('base64'), + connectionId: null + }) + ).resolves.toMatchObject({ ok: true, result: path }) + + expect(hasMobileClipboardImagePath('device-a', path)).toBe(true) + expect(hasMobileClipboardImagePath('device-b', path)).toBe(false) + }) + + it('does not authorize a remote-host clipboard path for local structured delivery', async () => { + const path = '/tmp/orca-paste-image.png' + saveClipboardImageBufferAsTempFile.mockResolvedValue(path) + const dispatcher = makeDispatcher() + + await expect( + callMobile(dispatcher, 'clipboard.saveImageAsTempFile', { + contentBase64: Buffer.from('png-bytes').toString('base64'), + connectionId: 'ssh-1' + }) + ).resolves.toMatchObject({ ok: true, result: path }) + + expect(hasMobileClipboardImagePath('device-a', path)).toBe(false) + }) + it('rejects non-base64 clipboard image payloads', async () => { const dispatcher = makeDispatcher() @@ -140,6 +196,48 @@ describe('clipboard RPC methods', () => { expect(saveClipboardImageBufferAsTempFile).toHaveBeenCalledWith(Buffer.from('png-bytes'), { connectionId: 'ssh-1' }) + expect(hasMobileClipboardImagePath('device-a', '/tmp/orca-paste-image.png')).toBe(false) + }) + + it('binds chunk mutation and provenance to the mobile client that started the upload', async () => { + saveClipboardImageBufferAsTempFile.mockResolvedValue('/tmp/orca-paste-image.png') + const dispatcher = makeDispatcher() + const contentBase64 = Buffer.from('png-bytes').toString('base64') + const start = await callMobile(dispatcher, 'clipboard.startImageUpload', { + expectedBase64Length: contentBase64.length, + connectionId: null + }) + const uploadId = (start.ok ? start.result : null) as { uploadId: string } + + for (const method of [ + 'clipboard.appendImageUploadChunk', + 'clipboard.commitImageUpload', + 'clipboard.abortImageUpload' + ]) { + const params = + method === 'clipboard.appendImageUploadChunk' + ? { uploadId: uploadId.uploadId, offset: 0, contentBase64 } + : { uploadId: uploadId.uploadId } + await expect(callMobile(dispatcher, method, params, 'device-b')).resolves.toMatchObject({ + ok: false + }) + } + + await expect( + callMobile(dispatcher, 'clipboard.appendImageUploadChunk', { + uploadId: uploadId.uploadId, + offset: 0, + contentBase64 + }) + ).resolves.toMatchObject({ + ok: true, + result: { receivedBase64Length: contentBase64.length } + }) + await expect( + callMobile(dispatcher, 'clipboard.commitImageUpload', { uploadId: uploadId.uploadId }) + ).resolves.toMatchObject({ ok: true, result: '/tmp/orca-paste-image.png' }) + expect(hasMobileClipboardImagePath('device-a', '/tmp/orca-paste-image.png')).toBe(true) + expect(hasMobileClipboardImagePath('device-b', '/tmp/orca-paste-image.png')).toBe(false) }) it('rejects out-of-order chunk offsets', async () => { diff --git a/src/main/runtime/rpc/methods/clipboard.ts b/src/main/runtime/rpc/methods/clipboard.ts index 3d5212c7a52..e6b487d7761 100644 --- a/src/main/runtime/rpc/methods/clipboard.ts +++ b/src/main/runtime/rpc/methods/clipboard.ts @@ -1,11 +1,12 @@ import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod, type RpcContext, type RpcMethod } from '../core' import { saveClipboardImageBufferAsTempFile } from '../../../window/clipboard-image-temp-file' import { randomUUID } from 'node:crypto' import { CLIPBOARD_IMAGE_MAX_BASE64_CHARS, CLIPBOARD_IMAGE_TOO_LARGE_ERROR } from '../../../../shared/clipboard-image' +import { recordMobileClipboardImagePath } from '../mobile-clipboard-image-provenance' const MAX_CLIPBOARD_IMAGE_BASE64_CHARS = CLIPBOARD_IMAGE_MAX_BASE64_CHARS export const CLIPBOARD_IMAGE_UPLOAD_CHUNK_BASE64_CHARS = 512 * 1024 @@ -16,6 +17,7 @@ const BASE64_PATTERN = /^[A-Za-z0-9+/]*={0,2}$/ type ClipboardImageUpload = { expectedBase64Length: number connectionId?: string | null + mobileClientId?: string chunks: string[] receivedBase64Length: number expiresAt: number @@ -69,6 +71,28 @@ function getUpload(uploadId: string): ClipboardImageUpload { return upload } +function mobileClientId(ctx: RpcContext): string | undefined { + if (ctx.clientKind !== 'mobile') { + return undefined + } + const clientId = ctx.clientId?.trim() + if (!clientId) { + throw new Error('Clipboard image upload requires an authenticated mobile client') + } + return clientId +} + +function assertMobileUploadOwner( + upload: ClipboardImageUpload, + ctx: RpcContext +): string | undefined { + const clientId = mobileClientId(ctx) + if (clientId && upload.mobileClientId !== clientId) { + throw new Error('Clipboard image upload was not found') + } + return clientId +} + function assertValidBase64Content(value: string): void { if (!isValidBase64(value)) { throw new Error('Clipboard image content must be base64') @@ -131,15 +155,24 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.saveImageAsTempFile', params: SaveImageAsTempFile, - handler: async (params) => - saveClipboardImageBufferAsTempFile(Buffer.from(params.contentBase64, 'base64'), { - connectionId: params.connectionId - }) + handler: async (params, ctx) => { + const clientId = mobileClientId(ctx) + const path = await saveClipboardImageBufferAsTempFile( + Buffer.from(params.contentBase64, 'base64'), + { + connectionId: params.connectionId + } + ) + if (clientId && !params.connectionId) { + recordMobileClipboardImagePath(clientId, path) + } + return path + } }), defineMethod({ name: 'clipboard.startImageUpload', params: StartImageUpload, - handler: (params) => { + handler: (params, ctx) => { pruneExpiredUploads() if (clipboardImageUploads.size >= CLIPBOARD_IMAGE_UPLOAD_MAX_CONCURRENT) { throw new Error('Too many clipboard image uploads are in progress') @@ -148,6 +181,7 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ clipboardImageUploads.set(uploadId, { expectedBase64Length: params.expectedBase64Length, connectionId: params.connectionId, + mobileClientId: mobileClientId(ctx), chunks: [], receivedBase64Length: 0, expiresAt: Date.now() + CLIPBOARD_IMAGE_UPLOAD_TTL_MS, @@ -159,8 +193,9 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.appendImageUploadChunk', params: AppendImageUploadChunk, - handler: (params) => { + handler: (params, ctx) => { const upload = getUpload(params.uploadId) + assertMobileUploadOwner(upload, ctx) if (params.offset !== upload.receivedBase64Length) { throw new Error('Clipboard image chunk offset is out of order') } @@ -177,17 +212,25 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.commitImageUpload', params: CommitImageUpload, - handler: async (params) => { + handler: async (params, ctx) => { const upload = getUpload(params.uploadId) + const clientId = assertMobileUploadOwner(upload, ctx) try { if (upload.receivedBase64Length !== upload.expectedBase64Length) { throw new Error('Clipboard image upload is incomplete') } const contentBase64 = upload.chunks.join('') assertValidBase64Content(contentBase64) - return await saveClipboardImageBufferAsTempFile(Buffer.from(contentBase64, 'base64'), { - connectionId: upload.connectionId - }) + const path = await saveClipboardImageBufferAsTempFile( + Buffer.from(contentBase64, 'base64'), + { + connectionId: upload.connectionId + } + ) + if (clientId && !upload.connectionId) { + recordMobileClipboardImagePath(clientId, path) + } + return path } finally { // Why: failed SSH or filesystem commits must not leave bounded upload // memory pinned until TTL cleanup. @@ -198,7 +241,12 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.abortImageUpload', params: AbortImageUpload, - handler: (params) => { + handler: (params, ctx) => { + pruneExpiredUploads() + const upload = clipboardImageUploads.get(params.uploadId) + if (upload) { + assertMobileUploadOwner(upload, ctx) + } deleteUpload(params.uploadId) return { aborted: true } } diff --git a/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts new file mode 100644 index 00000000000..dcba8b7b64e --- /dev/null +++ b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts @@ -0,0 +1,22 @@ +import { defineMethod, type RpcAnyMethod } from '../core' +import { ActivateTab, SaveMarkdownTab } from './session-tabs-schemas' + +export const MOBILE_MARKDOWN_TAB_METHODS: RpcAnyMethod[] = [ + defineMethod({ + name: 'markdown.readTab', + params: ActivateTab, + handler: async (params, { runtime }) => + runtime.readMobileMarkdownTab(params.worktree, params.tabId) + }), + defineMethod({ + name: 'markdown.saveTab', + params: SaveMarkdownTab, + handler: async (params, { runtime }) => + runtime.saveMobileMarkdownTab( + params.worktree, + params.tabId, + params.baseVersion, + params.content + ) + }) +] diff --git a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts index 226277f6ebd..a2a93533b6f 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from 'vitest' import { + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' @@ -67,13 +68,24 @@ describe('session tab structured capability mutations', () => { expect(fixture.calls[method.runtimeMethod]).toHaveBeenCalledOnce() }) - it(`rejects ${method.name} for a legacy Claude row`, async () => { + it(`rejects ${method.name} on a Claude row the client never negotiated`, async () => { const fixture = createFixture([STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]) const response = await fixture.dispatch(method.name, method.params('claude-session')) expect(response.ok).toBe(false) expect(fixture.calls[method.runtimeMethod]).not.toHaveBeenCalled() }) + + it(`allows ${method.name} for a client that negotiated Claude rows`, async () => { + const fixture = createFixture([ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ]) + const response = await fixture.dispatch(method.name, method.params('claude-session')) + + expect(response.ok).toBe(true) + expect(fixture.calls[method.runtimeMethod]).toHaveBeenCalledOnce() + }) } it.each(['session.tabs.close', 'session.tabs.closeLifecycle'] as const)( diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts index 4a61a99bc20..7128f756d6e 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from 'vitest' import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { RuntimeMobileSessionTabsSnapshot } from '../../../../shared/runtime-types' @@ -119,44 +120,105 @@ describe('projectSessionTabAgentStatus', () => { expect(capable).toBe(snapshot) }) - it('withholds legacy Claude rows from paired structured clients', () => { - const snapshot = { - ...makeSnapshot(false), - tabs: [ - { - type: 'agent-session', - id: 'agent-session:codex', - title: 'Codex Chat', - sessionId: 'codex', - agent: 'codex', - isActive: true - }, - { - type: 'agent-session', - id: 'agent-session:claude', - title: 'Claude Chat', - sessionId: 'claude', - agent: 'claude', - isActive: false - } - ], - activeTabId: 'agent-session:codex', - activeTabType: 'agent-session' + const claudeSnapshot = { + ...makeSnapshot(false), + tabs: [ + { + type: 'agent-session', + id: 'agent-session:codex', + title: 'Codex Chat', + sessionId: 'codex', + agent: 'codex', + isActive: true + }, + { + type: 'agent-session', + id: 'agent-session:claude', + title: 'Claude Chat', + sessionId: 'claude', + agent: 'claude', + isActive: false + } + ], + activeGroupId: 'group-a', + activeTabId: 'agent-session:codex', + activeTabType: 'agent-session', + tabGroups: [ + { id: 'group-a', activeTabId: 'agent-session:codex', tabOrder: ['agent-session:codex'] }, + { id: 'group-b', activeTabId: 'agent-session:claude', tabOrder: ['agent-session:claude'] } + ], + tabGroupLayout: { + type: 'split', + direction: 'horizontal', + first: { type: 'leaf', groupId: 'group-a' }, + second: { type: 'leaf', groupId: 'group-b' } + } + } as unknown as RuntimeMobileSessionTabsSnapshot + + const structuredMobile = [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] + + it.each([ + ['mobile', 'mobile' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]], + ['runtime', 'runtime' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]] + ])( + 'withholds Claude rows from a paired %s client that never negotiated them', + (_name, clientKind, capabilities) => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, clientKind, capabilities, true) + + expect(projected.tabs.map((tab) => tab.id)).toEqual(['agent-session:codex']) + // A row pruned from `tabs` but left in the layout is its own dead tab. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a']) + expect(projected.tabGroupLayout).toEqual({ type: 'leaf', groupId: 'group-a' }) + expect(projected.activeGroupId).toBe('group-a') + expect(projected.activeTabId).toBe('agent-session:codex') + expect(projected.activeTabType).toBe('agent-session') + } + ) + + it.each([ + ['mobile', 'mobile' as const, structuredMobile], + [ + 'runtime', + 'runtime' as const, + [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] + ] + ])( + 'publishes Claude rows to a paired %s client that negotiated them', + (_name, clientKind, capabilities) => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, clientKind, capabilities, true) + + expect(projected).toBe(claudeSnapshot) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + } + ) + + it('keeps Claude rows on the local renderer, which negotiates nothing', () => { + expect(projectSessionTabAgentStatus(claudeSnapshot, undefined, undefined)).toBe(claudeSnapshot) + expect(projectSessionTabAgentStatus(claudeSnapshot, undefined, [])).toBe(claudeSnapshot) + }) + + it('leaves Codex rows untouched whether or not the Claude capability is present', () => { + const codexOnly = { + ...claudeSnapshot, + tabs: claudeSnapshot.tabs.filter((tab) => tab.id !== 'agent-session:claude'), + tabGroups: claudeSnapshot.tabGroups?.filter((group) => group.id !== 'group-b'), + tabGroupLayout: { type: 'leaf', groupId: 'group-a' } } as unknown as RuntimeMobileSessionTabsSnapshot - expect( - projectSessionTabAgentStatus(snapshot, 'runtime', [ - STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY - ]).tabs.map((tab) => tab.id) - ).toEqual(['agent-session:codex']) - expect( - projectSessionTabAgentStatus( - snapshot, - 'mobile', - [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], - true - ).tabs.map((tab) => tab.id) - ).toEqual(['agent-session:codex']) + for (const capabilities of [[STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], structuredMobile]) { + for (const clientKind of ['mobile', 'runtime'] as const) { + expect(projectSessionTabAgentStatus(codexOnly, clientKind, capabilities, true)).toBe( + codexOnly + ) + } + } + expect(projectSessionTabAgentStatus(codexOnly, undefined, undefined)).toBe(codexOnly) }) it('withholds session boundaries from legacy paired clients', () => { diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts index 375b3b499d5..0e0d9c716a5 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts @@ -1,5 +1,6 @@ import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' import type { @@ -24,7 +25,14 @@ export function projectSessionTabAgentStatus true) - if (structuredVisible && clientKind !== undefined) { + // Why: a paired client renders only codex structured tabs unless it says otherwise + // (mobile's resolveMobileNativeChat returns null for every other agent), so an + // ungated row would list and select into a pane that shows neither chat nor terminal. + if ( + structuredVisible && + clientKind !== undefined && + !clientCapabilities?.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) { projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') } // Why: only paired runtimes have legacy `done` completion side effects; mobile must keep its row without changing the exact v2 auth shape. diff --git a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts index 6a9372ed2c1..4f7c1b20cc3 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts @@ -98,7 +98,7 @@ export const CreateIntentParams = z .object({ envelope: MutationEnvelope, worktree: Identifier('Invalid worktree selector'), - agent: z.literal('codex') + agent: z.enum(['claude', 'codex']) }) .strict() @@ -107,7 +107,7 @@ export const CreateParams = z.union([AttachParams, CreateIntentParams]) export const CreateSupportParams = z .object({ worktree: Identifier('Invalid worktree selector'), - agent: z.literal('codex') + agent: z.enum(['claude', 'codex']) }) .strict() @@ -170,6 +170,15 @@ export const SetOptionParams = z }) .strict() +export const HandoffParams = z + .object({ + envelope: MutationEnvelope, + direction: z.enum(['to-tui', 'to-native']), + mode: z.enum(['now', 'after-turn', 'stop-turn']), + action: z.enum(['start', 'cancel-queued', 'retry', 'recover']).optional() + }) + .strict() + export const OptionsParams = z.object({ sessionId: SessionId }).strict() /** One surface's claim on one session. The id names the surface, not the client: two chat views diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index b65e6eff825..155d0aa6768 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -99,6 +99,22 @@ function hostStub(): StructuredAgentSessionHost { setSessionTabVisibility: vi.fn(async () => undefined), respondToPrompt: vi.fn(async () => ({ ok: true, replayed: false })), setOption: vi.fn(async () => ({ ok: true, replayed: false })), + requestHandoff: vi.fn(async () => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 0 }, + value: { + status: { + owner: 'native', + direction: null, + phase: 'idle', + stage: null, + operationId: null + } + } + })), + supportsCreate: vi.fn(() => true), handoffStatus: vi.fn(async () => ({ owner: 'native' })), readOptions: vi.fn(async () => ({ models: [{ id: 'gpt-live', label: 'GPT Live', isDefault: true, efforts: [] }], @@ -122,9 +138,12 @@ function dispatcher(runtimeOverrides: Record = {}): RpcDispatch workspaceId: 'workspace-1', workspaceKind: 'git-worktree' }, - provider: 'codex', - agent: 'codex', - accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + provider: params.agent, + agent: params.agent, + accountHome: { + variable: params.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', + path: params.agent === 'claude' ? '/host/.claude' : '/host/.codex' + }, runtimeKind: 'native' })), publishStructuredAgentSessionTab: vi.fn() @@ -221,7 +240,7 @@ describe('capability gating', () => { } // Bump deliberately: the whole agentSession.* surface is behind the structured capability, // so an additive method is invisible to old clients and needs no protocol bump. - expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(16) + expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(17) }) it('hides the surface from a declared client that did not advertise it', async () => { @@ -329,6 +348,49 @@ describe('method routing', () => { ) }) + it('routes Claude create support and create through the provider-aware runtime', async () => { + const worktree = 'id:workspace-1' + const support = await call( + 'agentSession.createSupport', + { worktree, agent: 'claude' }, + STRUCTURED_CLIENT + ) + expect(support).toMatchObject({ ok: true, result: { supported: true } }) + expect(runtimeCalls.getStructuredAgentSessionCreateSupport).toHaveBeenCalledWith( + worktree, + 'claude' + ) + + const params = { + envelope: envelope({ + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: SESSION, + fields: { worktree, agent: 'claude' } + }) + }), + worktree, + agent: 'claude' + } + const created = await call('agentSession.create', params, STRUCTURED_CLIENT) + expect(created).toMatchObject({ ok: true, result: { ok: true } }) + expect(runtimeCalls.resolveStructuredAgentSessionCreateIntent).toHaveBeenCalledWith(params) + expect(hostCalls.attach).toHaveBeenCalledWith( + expect.anything(), + expect.objectContaining({ + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/host/.claude' } + }) + ) + expect(runtimeCalls.publishStructuredAgentSessionTab).toHaveBeenCalledWith( + expect.objectContaining({ + sessionId: SESSION, + activate: true, + agent: 'claude' + }) + ) + }) + it('reports an unknown create outcome when attach commits before tab publication fails', async () => { const worktree = 'id:workspace-1' const params = { @@ -371,6 +433,25 @@ describe('method routing', () => { expect(ensured).toMatchObject({ ok: true }) }) + /** A client-supplied location skips the worktree-resolving support check, so both attach-shaped + * entries must ask the executing host directly or a host that cannot fence a provider child + * would create one anyway. */ + it.each(['agentSession.create', 'agentSession.ensure'])( + 'refuses %s for a client-supplied location the executing host does not support', + async (method) => { + hostCalls.supportsCreate.mockReturnValue(false) + + const refused = await call(method, attachParams()) + + expect(refused).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + } + ) + it('tags the prompt kind from the method name, not from the client', async () => { const params = { envelope: envelope(), @@ -386,7 +467,7 @@ describe('method routing', () => { ]) }) - it('does not register the structured handoff mutation', async () => { + it('routes the structured handoff mutation through the host', async () => { const response = await call('agentSession.requestHandoff', { envelope: envelope(), direction: 'to-tui', @@ -394,7 +475,11 @@ describe('method routing', () => { action: 'start' }) - expect(response).toMatchObject({ ok: false, error: { code: 'method_not_found' } }) + expect(response).toMatchObject({ ok: true }) + expect(hostCalls.requestHandoff).toHaveBeenCalledWith( + expect.anything(), + expect.objectContaining({ direction: 'to-tui', mode: 'now', action: 'start' }) + ) }) }) @@ -434,25 +519,6 @@ describe('parameter validation', () => { ) }) - it('rejects Claude structured create shapes', async () => { - await rejects('agentSession.createSupport', { - worktree: 'id:workspace-1', - agent: 'claude' - }) - const fields = { worktree: 'id:workspace-1', agent: 'claude' } - await rejects('agentSession.create', { - envelope: envelope({ - expectedRuntimeFence: null, - payloadFingerprint: computeAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: SESSION, - fields - }) - }), - ...fields - }) - }) - it('requires a sha256 fingerprint and a positive fence', async () => { await rejects( 'agentSession.send', diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index ffd23499a3e..3b18f6b0ef1 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -10,6 +10,7 @@ import { agentSessionFingerprintConflict, computeAgentSessionPayloadFingerprint } from '../../../../shared/agent-session-mutation-envelope' +import type { z } from 'zod' import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled as ensureHostInstalled, @@ -18,6 +19,7 @@ import { structuredCallerFor as callerFor, supportsStructuredSessions } from './structured-agent-session-gate' +import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach' import { STRUCTURED_AGENT_SESSION_HOLD_METHODS } from './structured-agent-session-hold' import { AttachParams, @@ -25,6 +27,7 @@ import { CreateParams, CreateSupportParams, HistoryParams, + HandoffParams, HandoffStatusParams, OptionsParams, RespondParams, @@ -43,6 +46,29 @@ function subscriptionIdFor(ctx: RpcContext, sessionId: string): string { return ctx.requestId ? `${base}:${ctx.requestId}` : base } +/** + * The attach-shaped entries take the location from the client instead of resolving it from a + * worktree, so they never reach the worktree-resolving create-support check. Ask the executing + * host the same question directly: the answer includes host-measured facts the client cannot see + * or forge, such as whether this machine can read a provider child's process start time. + */ +async function attachClientSuppliedLocation( + params: z.infer, + ctx: RpcContext +): Promise { + await ensureHostInstalled(ctx) + const host = requireHost(ctx) + if (!host.supportsCreate(params.location, params.agent)) { + throw new Error('structured_agent_session_unsupported') + } + const { agent: _attachAgent, provider: _attachProvider, ...attachWithoutAgent } = params + return host.attach(callerFor(ctx), { + ...attachWithoutAgent, + provider: params.provider as 'claude' | 'codex', + agent: params.agent as 'claude' | 'codex' + } as AgentSessionAttachParams) +} + export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ defineMethod({ name: 'agentSession.createSupport', @@ -86,16 +112,20 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ } }) await ensureHostInstalled(ctx) - const result = await requireHost(ctx).attach(callerFor(ctx), { - ...resolved, + const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved + const attachParams: AgentSessionAttachParams = { + ...resolvedAttach, + provider: resolved.provider as 'claude' | 'codex', + agent: resolved.agent as 'claude' | 'codex', envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } - }) - if (result.ok && resolved.agent === 'codex') { + } + const result = await requireHost(ctx).attach(callerFor(ctx), attachParams) + if (result.ok) { try { await ctx.runtime.publishStructuredAgentSessionTab({ workspaceId: resolved.location.workspaceId, sessionId: result.value.sessionId, - agent: 'codex', + agent: resolved.agent as 'claude' | 'codex', activate: true }) } catch (error) { @@ -104,24 +134,20 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ ok: false, refusal: { code: 'agent_session_operation_unknown', - message: 'The Codex chat may have been created, but its tab could not be confirmed.' + message: 'The chat may have been created, but its tab could not be confirmed.' } } } } return result } - await ensureHostInstalled(ctx) - return requireHost(ctx).attach(callerFor(ctx), params) + return attachClientSuppliedLocation(params, ctx) } }), defineMethod({ name: 'agentSession.ensure', params: AttachParams, - handler: async (params, ctx) => { - await ensureHostInstalled(ctx) - return requireHost(ctx).attach(callerFor(ctx), params) - } + handler: async (params, ctx) => attachClientSuppliedLocation(params, ctx) }), defineMethod({ name: 'agentSession.send', @@ -165,6 +191,11 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ params: SetOptionParams, handler: async (params, ctx) => requireHost(ctx).setOption(callerFor(ctx), params) }), + defineMethod({ + name: 'agentSession.requestHandoff', + params: HandoffParams, + handler: async (params, ctx) => requireHost(ctx).requestHandoff(callerFor(ctx), params) + }), defineMethod({ name: 'agentSession.handoffStatus', params: HandoffStatusParams, diff --git a/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts b/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts new file mode 100644 index 00000000000..aa1709fcecb --- /dev/null +++ b/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts @@ -0,0 +1,53 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + hasMobileClipboardImagePath, + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES, + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS, + mobileClipboardImageProvenanceSizeForTest, + recordMobileClipboardImagePath, + resetMobileClipboardImageProvenanceForTest +} from './mobile-clipboard-image-provenance' + +describe('mobile clipboard image provenance', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-01-01T00:00:00Z')) + resetMobileClipboardImageProvenanceForTest() + }) + + afterEach(() => { + resetMobileClipboardImageProvenanceForTest() + vi.useRealTimers() + }) + + it('expires records without consuming them on repeated checks', () => { + recordMobileClipboardImagePath('device-a', '/tmp/image.png') + + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(true) + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(true) + vi.advanceTimersByTime(MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS + 1) + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(false) + expect(mobileClipboardImageProvenanceSizeForTest()).toBe(0) + }) + + it('evicts the oldest record at the global bound and supports test cleanup', () => { + for (let index = 0; index <= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES; index++) { + recordMobileClipboardImagePath(`device-${index}`, `/tmp/image-${index}.png`) + vi.advanceTimersByTime(1) + } + + expect(mobileClipboardImageProvenanceSizeForTest()).toBe( + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES + ) + expect(hasMobileClipboardImagePath('device-0', '/tmp/image-0.png')).toBe(false) + expect( + hasMobileClipboardImagePath( + `device-${MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES}`, + `/tmp/image-${MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES}.png` + ) + ).toBe(true) + + resetMobileClipboardImageProvenanceForTest() + expect(mobileClipboardImageProvenanceSizeForTest()).toBe(0) + }) +}) diff --git a/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts b/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts new file mode 100644 index 00000000000..117455593c9 --- /dev/null +++ b/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts @@ -0,0 +1,90 @@ +export const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS = 60 * 60 * 1000 +export const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES = 256 +const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_PER_CLIENT = 64 + +const pathsByClient = new Map>() +let entryCount = 0 + +function deletePath(clientId: string, path: string): void { + const paths = pathsByClient.get(clientId) + if (!paths?.delete(path)) { + return + } + entryCount-- + if (paths.size === 0) { + pathsByClient.delete(clientId) + } +} + +function pruneExpired(now: number): void { + for (const [clientId, paths] of pathsByClient) { + for (const [path, expiresAt] of paths) { + if (expiresAt <= now) { + deletePath(clientId, path) + } + } + } +} + +function deleteOldestEntry(): void { + let oldest: { clientId: string; path: string; expiresAt: number } | null = null + for (const [clientId, paths] of pathsByClient) { + for (const [path, expiresAt] of paths) { + if (!oldest || expiresAt < oldest.expiresAt) { + oldest = { clientId, path, expiresAt } + } + } + } + if (oldest) { + deletePath(oldest.clientId, oldest.path) + } +} + +export function recordMobileClipboardImagePath(clientId: string | undefined, path: string): void { + const owner = clientId?.trim() + if (!owner) { + return + } + const now = Date.now() + pruneExpired(now) + let paths = pathsByClient.get(owner) + if (!paths) { + paths = new Map() + pathsByClient.set(owner, paths) + } + if (paths.delete(path)) { + entryCount-- + } + while (paths.size >= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_PER_CLIENT) { + const oldestPath = paths.keys().next().value + if (typeof oldestPath !== 'string') { + break + } + deletePath(owner, oldestPath) + } + while (entryCount >= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES) { + deleteOldestEntry() + } + paths = pathsByClient.get(owner) ?? new Map() + pathsByClient.set(owner, paths) + paths.set(path, now + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS) + entryCount++ +} + +export function hasMobileClipboardImagePath(clientId: string | undefined, path: string): boolean { + const owner = clientId?.trim() + if (!owner) { + return false + } + pruneExpired(Date.now()) + return pathsByClient.get(owner)?.has(path) ?? false +} + +export function resetMobileClipboardImageProvenanceForTest(): void { + pathsByClient.clear() + entryCount = 0 +} + +export function mobileClipboardImageProvenanceSizeForTest(): number { + return entryCount +} diff --git a/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts b/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts new file mode 100644 index 00000000000..3834a417ddb --- /dev/null +++ b/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts @@ -0,0 +1,21 @@ +import type { RuntimeCapability } from '../../../shared/protocol-version' +import { parseRemoteRuntimeJsonText } from '../../../shared/remote-runtime-request-frames' +import { parseRuntimeClientCapabilities } from './runtime-client-capabilities' + +export function parseMobileE2EEV2ClientCapabilities( + plaintext: string +): readonly RuntimeCapability[] | null { + try { + const message = parseRemoteRuntimeJsonText(plaintext) as Record + if ( + Object.keys(message).sort().join(',') !== 'clientCapabilities,type,v' || + message.type !== 'e2ee_client_capabilities' || + message.v !== 1 + ) { + return null + } + return parseRuntimeClientCapabilities(message.clientCapabilities) + } catch { + return null + } +} diff --git a/src/main/runtime/rpc/mobile-socket-wiring.test.ts b/src/main/runtime/rpc/mobile-socket-wiring.test.ts index 505cf63feb6..14d5beb8cd0 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.test.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.test.ts @@ -348,4 +348,86 @@ describe('MobileSocketWiring', () => { expect(transport.setClientId).not.toHaveBeenCalled() expect(ws.close).toHaveBeenCalledWith(4001, 'Unauthorized') }) + + it('keeps post-auth v2 capability-shaped frames on the RPC path', () => { + const desktop = nacl.box.keyPair.fromSecretKey(new Uint8Array(32).fill(1)) + const phone = nacl.box.keyPair.fromSecretKey(new Uint8Array(32).fill(2)) + const ws = new FakeSocket() + const transport = new FakeTransport() + const onText = vi.fn() + const metadata: MobileSocketTransportMetadata = { + transport: 'relay', + relayHostId: 'AbCdEf0123_-xyZ9', + relayDeviceId: 'device-1', + basisConnId: 'connection-1', + credentialKind: 'resume' + } + const wiring = new MobileSocketWiring({ + deviceRegistry: registryFor('device-1', 'valid-token'), + e2eeKeypair: { + publicKey: desktop.publicKey, + secretKey: desktop.secretKey, + publicKeyB64: Buffer.from(desktop.publicKey).toString('base64') + }, + onText, + onBinary: vi.fn(), + onClose: vi.fn() + }) + wiring.attachTransport(transport, () => metadata) + const hello: MobileE2EEV2Hello = { + type: 'e2ee_hello', + v: 2, + clientPublicKeyB64: Buffer.from(phone.publicKey).toString('base64'), + clientNonceB64: Buffer.from(new Uint8Array(32).fill(3)).toString('base64'), + capabilities: { framing: [2], payloadKinds: ['text', 'binary'] }, + context: { + protocol: 'orca-mobile-e2ee', + initiator: 'mobile', + responder: 'desktop', + transport: 'relay', + relayHostId: metadata.relayHostId + } + } + transport.receive(ws, JSON.stringify(hello)) + const ready = JSON.parse(ws.sent[0]!.toString()) as MobileE2EEV2Ready + const handshake = validateMobileE2EEV2Handshake(hello, ready)! + const schedule = deriveMobileE2EEV2KeySchedule({ + sharedSecret: deriveSharedKey(phone.secretKey, desktop.publicKey), + transcript: encodeMobileE2EEV2Transcript(handshake), + clientNonce: handshake.clientNonce, + desktopNonce: handshake.desktopNonce + }) + const send = (value: unknown, counter: bigint): void => { + const frame = sealMobileE2EEV2Frame({ + payload: new TextEncoder().encode(JSON.stringify(value)), + key: schedule.mobileToDesktopKey, + sessionId: schedule.sessionId, + direction: 'mobile-to-desktop', + payloadKind: 'text', + counter + }) + transport.receive(ws, Buffer.from(frame).toString('base64')) + } + send( + { + type: 'e2ee_auth', + v: 2, + transcriptHashB64: Buffer.from(schedule.transcriptHash).toString('base64'), + deviceToken: 'valid-token' + }, + 0n + ) + const capabilityFrame = { + type: 'e2ee_client_capabilities', + v: 1, + clientCapabilities: ['agent-session.structured.v1'] + } + send(capabilityFrame, 1n) + send({ id: 'rpc-1', method: 'agentSession.history', params: {} }, 2n) + + expect(onText).toHaveBeenCalledTimes(2) + expect(onText.mock.calls[0]?.[0].clientCapabilities).toEqual([]) + expect(JSON.parse(onText.mock.calls[0]?.[1] ?? '')).toEqual(capabilityFrame) + expect(onText.mock.calls[1]?.[0].clientCapabilities).toEqual([]) + }) }) diff --git a/src/main/runtime/rpc/mobile-socket-wiring.ts b/src/main/runtime/rpc/mobile-socket-wiring.ts index 43004be4582..384f403e9de 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.ts @@ -165,7 +165,9 @@ export class MobileSocketWiring { ws, connectionId, device, - clientCapabilities: channel.clientCapabilities, + get clientCapabilities() { + return channel.clientCapabilities + }, transport: metadata } this.authenticatedSockets.set(ws, socket) diff --git a/src/main/runtime/structured-agent-session-integration-replay.test.ts b/src/main/runtime/structured-agent-session-integration-replay.test.ts index 7bc5e66dc45..e5baa032341 100644 --- a/src/main/runtime/structured-agent-session-integration-replay.test.ts +++ b/src/main/runtime/structured-agent-session-integration-replay.test.ts @@ -21,7 +21,7 @@ import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protoc import type { AgentJournalRenderItem } from '../../shared/agent-session-journal-types' import { attachFingerprintFields } from '../native-chat/agent-session-wire/structured-agent-session-attach' import { journalDirectoryFor } from '../native-chat/agent-session-journal/journal-paths' -import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' import type { OrcaRuntimeService } from './orca-runtime' import type { RpcRequest, RpcResponse } from './rpc/core' import { RpcDispatcher } from './rpc/dispatcher' @@ -31,6 +31,8 @@ import { stopStructuredAgentSessionRuntime } from './structured-agent-session-runtime' +const journals = createTrackedJournalOpener() + const SESSION = 'session-integration-1' const THREAD = 'thread-integration' const TURN = 'turn-1' @@ -257,6 +259,7 @@ beforeEach(async () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => { bootEnvironmentReads += 1 return { @@ -285,6 +288,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await stopStructuredAgentSessionRuntime() await rm(root, { recursive: true, force: true }) }) @@ -317,6 +321,7 @@ describe('a structured codex session over agentSession.*', () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), openCodexConnection: codex.openConnection, readProcessStartTime: async () => 1_700_000_000_000 }) @@ -364,7 +369,7 @@ describe('a structured codex session over agentSession.*', () => { agent: 'codex' as const, providerHandle: { kind: 'codex' as const, threadId: THREAD } } - const reopened = await openAgentSessionJournal({ + const reopened = await journals.open({ identity, journalDir: journalDirectoryFor(root, identity) }) diff --git a/src/main/runtime/structured-agent-session-integration.test.ts b/src/main/runtime/structured-agent-session-integration.test.ts index 582bd225b40..2982a6530b2 100644 --- a/src/main/runtime/structured-agent-session-integration.test.ts +++ b/src/main/runtime/structured-agent-session-integration.test.ts @@ -25,10 +25,9 @@ import type { import { attachFingerprintFields } from '../native-chat/agent-session-wire/structured-agent-session-attach' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' import { journalDirectoryFor } from '../native-chat/agent-session-journal/journal-paths' -import { readJournalBlob } from '../native-chat/agent-session-journal/journal-blob-store' import { appendLegacyTranscriptMessages } from '../native-chat/agent-session-journal/journal-legacy-import' import type { AgentSessionJournal } from '../native-chat/agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' import type { OrcaRuntimeService } from './orca-runtime' import type { RpcRequest, RpcResponse } from './rpc/core' import { RpcDispatcher } from './rpc/dispatcher' @@ -38,6 +37,8 @@ import { stopStructuredAgentSessionRuntime } from './structured-agent-session-runtime' +const journals = createTrackedJournalOpener() + const SESSION = 'session-integration-1' const THREAD = 'thread-integration' const TURN = 'turn-1' @@ -306,6 +307,7 @@ beforeEach(async () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => { bootEnvironmentReads += 1 return { @@ -363,6 +365,7 @@ function cursorOf(frames: AgentSessionSubscribeEvent[]): { epoch: string; sequen } afterEach(async () => { + await journals.closeAll() await stopStructuredAgentSessionRuntime() await rm(root, { recursive: true, force: true }) }) @@ -376,7 +379,7 @@ describe('a structured codex session over agentSession.*', () => { agent: 'codex' as const, providerHandle: { kind: 'codex' as const, threadId: THREAD } } - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity, journalDir: journalDirectoryFor(root, identity) }) @@ -761,7 +764,7 @@ describe('a structured codex session over agentSession.*', () => { agent: 'codex' as const, providerHandle: { kind: 'codex' as const, threadId: THREAD } } - const reopened = await openAgentSessionJournal({ + const reopened = await journals.open({ identity, journalDir: journalDirectoryFor(root, identity) }) @@ -800,7 +803,6 @@ describe('a structured codex session over agentSession.*', () => { const item = journal.snapshot().items.find((candidate) => candidate.body?.kind === 'tool-call') const bounded = item?.body?.kind === 'tool-call' ? item.body.output : undefined expect(bounded).toMatchObject({ truncated: true, byteLength: Buffer.byteLength(output) }) - expect(await readJournalBlob(journal.directory, bounded?.digest ?? '')).toBe(output) }) it('keeps an answered prompt resolved after the provider exits', async () => { diff --git a/src/main/runtime/structured-agent-session-owner-probe.ts b/src/main/runtime/structured-agent-session-owner-probe.ts new file mode 100644 index 00000000000..47f92a08ca6 --- /dev/null +++ b/src/main/runtime/structured-agent-session-owner-probe.ts @@ -0,0 +1,108 @@ +import type { AgentSessionOwnerProbe } from '../../shared/agent-session-lease-adjudication' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { + probeAgentSessionProcessIdentities, + probeAgentSessionProcessIdentity, + probeAgentSessionReservation +} from './agent-session-process-identity-probe' +import { findAgentSessionSpawnTokenProcesses } from './agent-session-spawn-token-process-scan' +import { readEchoedAgentSessionSpawnToken } from './agent-session-spawn-token-readback' + +/** + * The lease's only source of truth about a previous owner. Everything it cannot + * answer PID-reuse-safely reports `indeterminate`. An exact owner stays fenced in `recovering`; + * an ownerless, unattributable reservation enters `manual-recovery`. + */ +export function createStructuredAgentSessionOwnerProbe( + hostId: string, + probe = probeAgentSessionProcessIdentity, + findSpawnTokenProcesses = findAgentSessionSpawnTokenProcesses +): (record: AgentSessionRecord) => Promise { + return async (record) => { + const owner = record.lease.ownerProcess + if (!owner) { + if (record.lease.processlessAt !== undefined && record.lease.processlessAt !== null) { + return { outcome: 'reservation-unused' } + } + const spawnToken = record.lease.reservedSpawnToken + if (spawnToken === null) { + if (record.lease.claimStatus === 'reserved') { + return { + outcome: 'indeterminate', + reason: 'reservation recorded no spawn token to scan for' + } + } + // The token is minted before the child and is the only thing a child could be carrying. + // No owner and no token means nothing on any host can be holding this lease — answering + // `indeterminate` here is what latches an already-free record into recovery forever. + return { outcome: 'reservation-unused' } + } + // Freeing a reservation needs positive proof that nothing spawned under its token. The scan + // answers null where the platform cannot read another process's environment. + return probeAgentSessionReservation({ + spawnToken, + findProcessesWithSpawnToken: (token) => findSpawnTokenProcesses(token), + hasProviderActivitySinceReservation: async () => + agentSessionReservationTouchedProvider(record) + }) + } + if (owner.hostId !== hostId) { + // Checking a remote host's pid against this machine's process table is + // exactly how a live owner gets declared dead. + return { + outcome: 'indeterminate', + reason: `owner runs on ${owner.hostId}, which this host cannot probe` + } + } + // The env read-back answers on hosts that expose it and null elsewhere, giving the + // probe a PID-reuse-safe element even when no start time was recorded. + return probe({ + identity: owner, + deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } + }) + } +} + +export function createStructuredAgentSessionOwnerProbes( + hostId: string, + probeMany: typeof probeAgentSessionProcessIdentities = probeAgentSessionProcessIdentities, + probeOne = createStructuredAgentSessionOwnerProbe(hostId) +): (records: readonly AgentSessionRecord[]) => Promise> { + return async (records) => { + const results = new Map() + const localOwners: { + record: AgentSessionRecord + owner: NonNullable + }[] = [] + for (const record of records) { + const owner = record.lease.ownerProcess + if (owner?.hostId === hostId) { + localOwners.push({ record, owner }) + } else { + results.set(record.sessionId, await probeOne(record)) + } + } + const probes = await probeMany({ + identities: localOwners.map(({ owner }) => owner), + deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } + }) + for (const [index, { record }] of localOwners.entries()) { + results.set( + record.sessionId, + probes[index] ?? { outcome: 'indeterminate', reason: 'owner probe returned no result' } + ) + } + return results + } +} + +/** + * The only provider-side trace a reservation can leave in its own record: a handle link minted at + * this fence. `proveAgentSessionOwner` refuses to append one before an identity is committed, so a + * link at the reservation's fence means a child got far enough to resume the provider thread. It + * cannot see activity the child produced without proving a handle, which is why it is paired with + * the token scan rather than trusted alone. + */ +function agentSessionReservationTouchedProvider(record: AgentSessionRecord): boolean { + return record.providerHandleChain.at(-1)?.mintedAtFence === record.lease.runtimeFence +} diff --git a/src/main/runtime/structured-agent-session-runtime-exit.test.ts b/src/main/runtime/structured-agent-session-runtime-exit.test.ts index a8419176357..5c6e43c2bc0 100644 --- a/src/main/runtime/structured-agent-session-runtime-exit.test.ts +++ b/src/main/runtime/structured-agent-session-runtime-exit.test.ts @@ -84,6 +84,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, @@ -180,6 +181,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, @@ -260,6 +262,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, diff --git a/src/main/runtime/structured-agent-session-runtime.test.ts b/src/main/runtime/structured-agent-session-runtime.test.ts index 6adf5d368fd..3b69a0a4be3 100644 --- a/src/main/runtime/structured-agent-session-runtime.test.ts +++ b/src/main/runtime/structured-agent-session-runtime.test.ts @@ -2,6 +2,10 @@ import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { agentSessionJournalCloseRetries } from '../native-chat/agent-session-journal/journal-close-retry' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' +import type { AgentSessionJournal } from '../native-chat/agent-session-journal/journal-store' import type { AgentSessionClaimStatus, AgentSessionProcessIdentity, @@ -9,7 +13,9 @@ import type { } from '../../shared/agent-session-record' import { createStructuredAgentSessionOwnerProbe, - createStructuredAgentSessionOwnerProbes, + createStructuredAgentSessionOwnerProbes +} from './structured-agent-session-owner-probe' +import { ensureStructuredAgentSessionHost, hasPersistedStructuredAgentSessionStore, stopStructuredAgentSessionRuntime @@ -222,6 +228,7 @@ describe('structured agent-session runtime install', () => { hostId: HOST_ID, claimKeyId: 'key-1', resolveWorkspacePath: async () => stateDirectory!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => ({}), reapOrphanChildren, onError @@ -248,6 +255,7 @@ describe('structured agent-session runtime install', () => { hostId: HOST_ID, claimKeyId: 'key-1', resolveWorkspacePath: async () => stateDirectory!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => ({}), reapOrphanChildren: async () => { throw failure @@ -263,3 +271,69 @@ describe('structured agent-session runtime install', () => { ) }) }) + +// A stop whose teardown fails must not forget the runtime it was tearing down. +// `installing` is cleared either way so nothing new attaches, but the host keeps +// every journal whose close rejected, and this module slot is the only handle +// onto that host once it is gone. +describe('a teardown that fails is retried by the next stop', () => { + const JOURNAL_IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-teardown-retry', + workspaceId: 'ws-1', + hostId: HOST_ID, + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + } + const journals = createTrackedJournalOpener() + let directory: string | null = null + + afterEach(async () => { + await agentSessionJournalCloseRetries.retryAll() + await journals.closeAll() + await stopStructuredAgentSessionRuntime().catch(() => undefined) + if (directory) { + await rm(directory, { recursive: true, force: true }) + directory = null + } + }) + + it('reports the failure, then releases the handle on the following stop', async () => { + directory = await mkdtemp(join(tmpdir(), 'orca-structured-runtime-')) + await ensureStructuredAgentSessionHost({ + stateDirectory: directory, + hostId: HOST_ID, + claimKeyId: 'key-1', + resolveWorkspacePath: async () => directory!, + resolveEnvironment: async () => ({}), + reapOrphanChildren: async () => [], + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }) + }) + + const journalDir = join(directory, 'stubborn-journal') + const real = await journals.open({ identity: JOURNAL_IDENTITY, journalDir }) + let closeFailures = 2 + const flaky = new Proxy(real, { + get(target, property, receiver) { + if (property !== 'close') { + return Reflect.get(target, property, receiver) + } + return async () => { + if (closeFailures > 0) { + closeFailures -= 1 + throw new Error('close rejected') + } + await target.close() + } + } + }) as AgentSessionJournal + await agentSessionJournalCloseRetries.closeOrRetain(flaky) + + // The host's teardown runs the registry retry, so this stop surfaces it. + await expect(stopStructuredAgentSessionRuntime()).rejects.toThrow() + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([journalDir]) + + // The retained runtime is what makes this a retry rather than a no-op. + await stopStructuredAgentSessionRuntime() + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + }) +}) diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index ce916bc6769..923d5627f72 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -9,29 +9,33 @@ import { existsSync } from 'node:fs' import { join } from 'node:path' -import type { AgentSessionOwnerProbe } from '../../shared/agent-session-lease-adjudication' import type { AgentSessionRecord } from '../../shared/agent-session-record' import { createCodexStructuredLaunchResolver } from '../codex/codex-structured-launch-resolution' import { CodexStructuredSessionAdapter, type CodexStructuredSessionAdapterDeps } from '../codex/codex-structured-session-adapter' +import type { ClaudeStructuredSessionAdapterDeps } from '../claude/claude-structured-session-adapter' import { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host' +import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router' import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' import { setStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { + readClaudeManagedAccountGateSettings, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' import { AgentSessionRecordStore } from './agent-session-record-store' import { agentSessionStorePath } from './agent-session-record-store-file' import { stopOrphanAgentSessionChildren } from './agent-session-orphan-child-reaper' import { - probeAgentSessionProcessIdentities, - probeAgentSessionProcessIdentity, - probeAgentSessionReservation -} from './agent-session-process-identity-probe' -import { findAgentSessionSpawnTokenProcesses } from './agent-session-spawn-token-process-scan' -import { readEchoedAgentSessionSpawnToken } from './agent-session-spawn-token-readback' + createStructuredAgentSessionOwnerProbe, + createStructuredAgentSessionOwnerProbes +} from './structured-agent-session-owner-probe' import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' import { resolveLoginShellEnvironment } from '../startup/login-shell-environment' import { recordAgentSessionProviderHandle } from './agent-session-provider-handle-transition' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { createStructuredClaudeRuntimeAdapter } from './structured-claude-runtime-adapter' /** Sibling of the journal tree rather than inside it: one file adjudicates every * session's lease, while a journal is per session. */ @@ -55,13 +59,20 @@ export type StructuredAgentSessionRuntimeDeps = { claimKeyId: string resolveWorkspacePath: (workspaceId: string) => Promise resolveCodexCommand?: (options?: { pathEnv?: string | null; homePath?: string }) => string + resolveClaudeCommand?: () => string /** Provider transports are overridden only to drive the runtime against scripted children. */ openCodexConnection?: CodexStructuredSessionAdapterDeps['openConnection'] + openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] /** Scripted app-servers carry fake pids the real start-time read cannot answer for. */ readProcessStartTime?: CodexStructuredSessionAdapterDeps['readProcessStartTime'] resolveLaunchArgs?: (provider: AgentSessionRecord['provider']) => Promise | string[] resolveLaunchEnv?: () => Promise resolveLaunchEnvOverlay?: () => Promise> | Record + resolveClaudeLaunchEnv?: () => Promise> | Record + /** Required, and asserted at install time — an absent policy must not degrade to a guess. */ + resolveClaudeAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + /** Raw settings getter; the reader that fails closed around it is built here, in checked code. */ + getClaudeManagedAccountGateSettings?: () => ClaudeManagedAccountGateSettings resolveEnvironment?: () => Promise resolveCodexOverrides?: () => NodeJS.ProcessEnv onError?: (input: { scope: string; error: unknown }) => void @@ -71,13 +82,26 @@ export type StructuredAgentSessionRuntimeDeps = { type InstalledRuntime = { host: StructuredAgentSessionHost - adapter: CodexStructuredSessionAdapter + adapter: { closeAll(): Promise } /** Resolves after every adapter-exit recovery callback has settled. */ waitForRecovery: () => Promise } let installing: Promise | null = null +/** Thrown when the host is installed without a Claude auth policy resolver. */ +export const CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED = + 'structured agent-session host requires a Claude auth policy resolver' + +/** + * Runtimes whose teardown did not finish. `installing` is cleared regardless so + * nothing new attaches, but dropping the runtime as well would strand every + * journal the host retained for a retry: `tearDownStructuredAgentSessionHost` + * deliberately keeps a failed close indexed, and only a later stop through this + * same runtime can reach those entries again. + */ +const pendingTeardown = new Set() + export function ensureStructuredAgentSessionHost( deps: StructuredAgentSessionRuntimeDeps ): Promise { @@ -90,19 +114,40 @@ export function ensureStructuredAgentSessionHost( } /** Drops the host and reaps every Codex child under it. Runtime teardown and - * test isolation take the same path, so neither can leave a live app-server. */ + * test isolation take the same path, so neither can leave a live app-server. + * + * A teardown that fails is RETRIED by the next stop rather than forgotten: the + * host keeps every journal whose close rejected, and this is the only handle + * onto that host once the module slot is cleared. */ export async function stopStructuredAgentSessionRuntime(): Promise { const pending = installing installing = null setStructuredAgentSessionHost(null) agentSessionPtyWriteGate.detachRecordLookup() - if (!pending) { - return + const outstanding = [...pendingTeardown] + pendingTeardown.clear() + const installed = pending ? await pending.catch(() => null) : null + if (installed) { + outstanding.push(installed) } - const installed = await pending.catch(() => null) - if (!installed) { - return + const failures: unknown[] = [] + for (const runtime of outstanding) { + try { + await tearDownRuntime(runtime) + } catch (error) { + pendingTeardown.add(runtime) + failures.push(error) + } } + if (failures.length === 1) { + throw failures[0] + } + if (failures.length > 1) { + throw new AggregateError(failures, 'structured agent-session runtime teardown failed') + } +} + +async function tearDownRuntime(installed: InstalledRuntime): Promise { // Drain an in-flight recovery before stopping children; recovery may still // be writing lifecycle rows or acquiring a replacement child. await installed.waitForRecovery() @@ -117,8 +162,14 @@ export async function stopStructuredAgentSessionRuntime(): Promise { } async function install(deps: StructuredAgentSessionRuntimeDeps): Promise { + // Why thrown rather than defaulted: the caller is `@ts-nocheck`, so a dropped + // field arrives here as `undefined`. Refusing to install is loud; guessing a + // policy is the silent under-strip this assertion exists to prevent. + if (typeof deps.resolveClaudeAuthPolicy !== 'function') { + throw new Error(CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED) + } const bootEnvironment = (deps.resolveEnvironment ?? resolveLoginShellEnvironment)() - const resolveEnvironment = async (): Promise => ({ + const resolveCodexEnvironment = async (): Promise => ({ ...(await bootEnvironment), ...(await deps.resolveLaunchEnv?.()), ...(await deps.resolveLaunchEnvOverlay?.()), @@ -151,7 +202,7 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise + readClaudeManagedAccountGateSettings(deps.getClaudeManagedAccountGateSettings!) + } + : {}), + onUnexpectedExit: (event) => { + recoveryChain = recoveryChain.then(async () => { + try { + await host?.handleAdapterEvent(event) + } catch (error) { + deps.onError?.({ scope: `structured-agent-session-exit:${event.sessionId}`, error }) + } + }) + }, + ...(deps.openClaudeConnection ? { openClaudeConnection: deps.openClaudeConnection } : {}), + ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) + }) + const adapter = new StructuredAgentSessionAdapterRouter({ codex, claude }, async () => { + await Promise.all([codex.closeAll(), claude.closeAll()]) + }) host = new StructuredAgentSessionHost({ store, adapter, @@ -216,102 +295,3 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise Promise { - return async (record) => { - const owner = record.lease.ownerProcess - if (!owner) { - if (record.lease.processlessAt !== undefined && record.lease.processlessAt !== null) { - return { outcome: 'reservation-unused' } - } - const spawnToken = record.lease.reservedSpawnToken - if (spawnToken === null) { - if (record.lease.claimStatus === 'reserved') { - return { - outcome: 'indeterminate', - reason: 'reservation recorded no spawn token to scan for' - } - } - // The token is minted before the child and is the only thing a child could be carrying. - // No owner and no token means nothing on any host can be holding this lease — answering - // `indeterminate` here is what latches an already-free record into recovery forever. - return { outcome: 'reservation-unused' } - } - // Freeing a reservation needs positive proof that nothing spawned under its token. The scan - // answers null where the platform cannot read another process's environment. - return probeAgentSessionReservation({ - spawnToken, - findProcessesWithSpawnToken: (token) => findSpawnTokenProcesses(token), - hasProviderActivitySinceReservation: async () => - agentSessionReservationTouchedProvider(record) - }) - } - if (owner.hostId !== hostId) { - // Checking a remote host's pid against this machine's process table is - // exactly how a live owner gets declared dead. - return { - outcome: 'indeterminate', - reason: `owner runs on ${owner.hostId}, which this host cannot probe` - } - } - // The env read-back answers on hosts that expose it and null elsewhere, giving the - // probe a PID-reuse-safe element even when no start time was recorded. - return probe({ - identity: owner, - deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } - }) - } -} - -export function createStructuredAgentSessionOwnerProbes( - hostId: string, - probeMany: typeof probeAgentSessionProcessIdentities = probeAgentSessionProcessIdentities, - probeOne = createStructuredAgentSessionOwnerProbe(hostId) -): (records: readonly AgentSessionRecord[]) => Promise> { - return async (records) => { - const results = new Map() - const localOwners: { - record: AgentSessionRecord - owner: NonNullable - }[] = [] - for (const record of records) { - const owner = record.lease.ownerProcess - if (owner?.hostId === hostId) { - localOwners.push({ record, owner }) - } else { - results.set(record.sessionId, await probeOne(record)) - } - } - const probes = await probeMany({ - identities: localOwners.map(({ owner }) => owner), - deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } - }) - for (const [index, { record }] of localOwners.entries()) { - results.set( - record.sessionId, - probes[index] ?? { outcome: 'indeterminate', reason: 'owner probe returned no result' } - ) - } - return results - } -} - -/** - * The only provider-side trace a reservation can leave in its own record: a handle link minted at - * this fence. `proveAgentSessionOwner` refuses to append one before an identity is committed, so a - * link at the reservation's fence means a child got far enough to resume the provider thread. It - * cannot see activity the child produced without proving a handle, which is why it is paired with - * the token scan rather than trusted alone. - */ -function agentSessionReservationTouchedProvider(record: AgentSessionRecord): boolean { - return record.providerHandleChain.at(-1)?.mintedAtFence === record.lease.runtimeFence -} diff --git a/src/main/runtime/structured-agent-session-support-probe.test.ts b/src/main/runtime/structured-agent-session-support-probe.test.ts new file mode 100644 index 00000000000..e393e41f3a4 --- /dev/null +++ b/src/main/runtime/structured-agent-session-support-probe.test.ts @@ -0,0 +1,174 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' +import { + getStructuredAgentSessionHost, + setStructuredAgentSessionHost +} from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' + +type InstallEffects = { + storeOpened: boolean + writeGateAttached: boolean + reaperStarted: boolean +} + +/** Stands in for `install()` by performing the three effects it performs, so a probe that + * reinstalls the host is caught by what the install *does*, not by a call count alone. */ +function stubStructuredHostInstall(runtime: OrcaRuntimeService): { + effects: InstallEffects + ensure: ReturnType +} { + const effects: InstallEffects = { + storeOpened: false, + writeGateAttached: false, + reaperStarted: false + } + // `supportsCreate` answers as the real Codex adapter would, so a probe that reinstalls the host + // still returns the right answer and fails on the install effects alone. + const host = { + reconcileRestartLeases: vi.fn(async () => {}), + supportsCreate: (location: { executionHostId: string; wslDistro: string | null }) => + location.executionHostId === 'local' && location.wslDistro === null + } + const ensure = vi.fn(async () => { + effects.storeOpened = true + effects.reaperStarted = true + agentSessionPtyWriteGate.attachRecordLookup(() => null) + effects.writeGateAttached = true + setStructuredAgentSessionHost(host as never) + }) + vi.spyOn(runtime, 'ensureStructuredAgentSessionHost').mockImplementation(ensure) + return { effects, ensure } +} + +type TestLocation = { + executionHostId: string + wslDistro: string | null + workspaceKind?: 'folder' | 'git-worktree' +} + +type SupportResult = { + supported: boolean + reason?: 'agent' | 'remote' | 'wsl' +} + +function createRuntime(location: TestLocation): OrcaRuntimeService { + const runtime = new OrcaRuntimeService({ getSettings: () => ({}) } as never) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: () => Promise + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: location.executionHostId, + wslDistro: location.wslDistro, + workspaceId: 'workspace-1', + workspaceKind: location.workspaceKind ?? 'git-worktree' + })) + return runtime +} + +async function expectSupportWithoutInstall(input: { + agent: 'claude' | 'codex' + location: TestLocation + expected: SupportResult + repetitions?: number +}): Promise { + const runtime = createRuntime(input.location) + const { effects, ensure } = stubStructuredHostInstall(runtime) + + const answers: SupportResult[] = [] + for (let index = 0; index < (input.repetitions ?? 1); index += 1) { + answers.push( + await runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', input.agent) + ) + } + + expect(answers).toEqual(Array(input.repetitions ?? 1).fill(input.expected)) + expect(ensure).not.toHaveBeenCalled() + expect(effects).toEqual({ + storeOpened: false, + writeGateAttached: false, + reaperStarted: false + }) + expect(getStructuredAgentSessionHost()).toBeNull() +} + +describe('structured agent-session create-support probe', () => { + afterEach(() => { + setStructuredAgentSessionHost(null) + agentSessionPtyWriteGate.detachRecordLookup() + vi.restoreAllMocks() + }) + + it.each(['codex', 'claude'] as const)( + 'answers %s support repeatedly without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { executionHostId: 'local', wslDistro: null }, + expected: { supported: true }, + repetitions: 3 + }) + } + ) + + it.each(['codex', 'claude'] as const)( + 'still reports an unsupported remote %s location without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { executionHostId: 'ssh-host-1', wslDistro: null }, + expected: { supported: false, reason: 'remote' } + }) + } + ) + + it.each(['codex', 'claude'] as const)( + 'still reports an unsupported WSL %s location without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { executionHostId: 'local', wslDistro: 'Ubuntu' }, + expected: { supported: false, reason: 'wsl' } + }) + } + ) + + it.each(['codex', 'claude'] as const)( + 'supports a local folder workspace for %s without installing the host', + async (agent) => { + await expectSupportWithoutInstall({ + agent, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceKind: 'folder' + }, + expected: { supported: true } + }) + } + ) + + it('still installs and reconciles on startup when a store is already persisted', async () => { + const runtime = createRuntime({ executionHostId: 'local', wslDistro: null }) + const { effects, ensure } = stubStructuredHostInstall(runtime) + const internal = runtime as unknown as { + hasPersistedStructuredAgentSessionStore: () => boolean + refreshMobileSessionPtyRecords: () => Promise + } + internal.hasPersistedStructuredAgentSessionStore = () => true + internal.refreshMobileSessionPtyRecords = vi.fn(async () => {}) + + await runtime.prepareStructuredAgentSessionStartupRestoration() + + expect(ensure).toHaveBeenCalledTimes(1) + expect(effects).toEqual({ + storeOpened: true, + writeGateAttached: true, + reaperStarted: true + }) + expect( + (getStructuredAgentSessionHost() as unknown as { reconcileRestartLeases: () => void }) + .reconcileRestartLeases + ).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/runtime/structured-claude-auth-policy-wiring.test.ts b/src/main/runtime/structured-claude-auth-policy-wiring.test.ts new file mode 100644 index 00000000000..f018dfd30da --- /dev/null +++ b/src/main/runtime/structured-claude-auth-policy-wiring.test.ts @@ -0,0 +1,59 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { afterEach, describe, expect, it } from 'vitest' +import { + CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED, + ensureStructuredAgentSessionHost, + stopStructuredAgentSessionRuntime +} from './structured-agent-session-runtime' + +/** + * The structured host's Claude auth policy has exactly one production wiring, and it + * lives in `orca-runtime-get-worktree-ps.ts` — a `@ts-nocheck` file, so neither the + * compiler nor a type test can see the field disappear. Deleting that wiring used to + * leave ~1000 tests green while every `ANTHROPIC_*` variable in the shell reached the + * child, because `stripAuthEnv` silently fell back to `false`. + * + * Two independent guards replace that silence, and this file pins both. + */ +describe('structured Claude auth policy wiring', () => { + // The behavioural version of this assertion — importing the runtime class and + // capturing the installed deps — costs 35s of module transform for the whole + // OrcaRuntime chain (measured), so the wiring itself is pinned by source and the + // policy's meaning by claude-structured-auth-policy.test.ts. + it('passes a settings-derived Claude auth policy to the host installer', () => { + const source = readFileSync(join(__dirname, 'orca-runtime-get-worktree-ps.ts'), 'utf8') + + expect(source).toContain('claudeStructuredAuthPolicyForSettings') + expect(source).toMatch( + /resolveClaudeAuthPolicy:\s*\(\)\s*=>\s*\n?\s*claudeStructuredAuthPolicyForSettings\(/ + ) + }) + + describe('installing without one', () => { + let stateDirectory: string | null = null + + afterEach(async () => { + await stopStructuredAgentSessionRuntime() + if (stateDirectory) { + await rm(stateDirectory, { recursive: true, force: true }) + stateDirectory = null + } + }) + + it('refuses loudly rather than defaulting to a guess', async () => { + stateDirectory = await mkdtemp(join(tmpdir(), 'orca-auth-policy-wiring-')) + + await expect( + ensureStructuredAgentSessionHost({ + stateDirectory, + hostId: 'local', + claimKeyId: 'key-1', + resolveWorkspacePath: async () => stateDirectory as string + } as unknown as Parameters[0]) + ).rejects.toThrow(CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED) + }) + }) +}) diff --git a/src/main/runtime/structured-claude-runtime-adapter.ts b/src/main/runtime/structured-claude-runtime-adapter.ts new file mode 100644 index 00000000000..398288562b9 --- /dev/null +++ b/src/main/runtime/structured-claude-runtime-adapter.ts @@ -0,0 +1,98 @@ +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { join } from 'node:path' +import { resolveClaudeCommand } from '../codex-cli/command' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { createClaudeStructuredLaunchResolver } from '../claude/claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps +} from '../claude/claude-structured-session-adapter' +import { claudeProviderHandleLink } from '../claude/claude-structured-owner-identity' +import type { StructuredAgentSessionLifecycleEvent } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +} from '../native-chat/session-file-resolver' +import { recordAgentSessionProviderHandle } from './agent-session-provider-handle-transition' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' +import type { AgentSessionRecordStore } from './agent-session-record-store' + +export type StructuredClaudeRuntimeAdapterDeps = { + store: AgentSessionRecordStore + resolveWorkspacePath: (workspaceId: string) => Promise + resolveClaudeCommand?: () => string + resolveClaudeLaunchEnv?: () => Promise> | Record + /** Managed-account auth state for a Claude launch, mirroring the terminal preflight. + * Required: an absent policy is what silently under-strips. */ + resolveClaudeAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + readClaudeManagedAccountGate?: () => ClaudeManagedAccountGateSettings | null + openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] + readProcessStartTime?: ClaudeStructuredSessionAdapterDeps['readProcessStartTime'] + onUnexpectedExit: (event: StructuredAgentSessionLifecycleEvent) => void +} + +export function createStructuredClaudeRuntimeAdapter( + deps: StructuredClaudeRuntimeAdapterDeps +): ClaudeStructuredSessionAdapter { + const { store } = deps + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: createClaudeStructuredLaunchResolver({ + store, + resolveWorkspacePath: deps.resolveWorkspacePath, + resolveCommand: deps.resolveClaudeCommand ?? resolveClaudeCommand, + ...(deps.resolveClaudeLaunchEnv ? { resolveEnv: deps.resolveClaudeLaunchEnv } : {}), + resolveAuthPolicy: deps.resolveClaudeAuthPolicy, + ...(deps.readClaudeManagedAccountGate + ? { readManagedAccountGate: deps.readClaudeManagedAccountGate } + : {}) + }), + persistHandle: async ({ sessionId, providerSessionId, leafUuid, fence }) => { + const currentFence = store.getRecord(sessionId)?.lease.runtimeFence ?? fence + const observedAt = Date.now() + await store.transitionHandoff(sessionId, (record: AgentSessionRecord) => + recordAgentSessionProviderHandle({ + record, + fence: currentFence, + link: claudeProviderHandleLink({ + sessionId: providerSessionId, + leafUuid, + resumed: true, + fence: currentFence, + observedAt + }), + now: observedAt + }) + ) + }, + readTranscriptLeaf: async ({ providerSessionId, previousLeafUuid, claudeConfigDir }) => { + const transcriptPath = await resolveSessionFilePath('claude', providerSessionId, { + claudeProjectsDir: join(claudeConfigDir, 'projects') + }) + return transcriptPath + ? await readClaudeTranscriptLeafUuid(transcriptPath, providerSessionId, previousLeafUuid) + : null + }, + onEvent: (event) => { + if ( + event.type === 'ended' && + event.cause === 'unexpected-exit' && + event.fence !== undefined && + event.acquisitionGeneration + ) { + deps.onUnexpectedExit({ + type: 'ended', + sessionId: event.sessionId, + reason: event.reason, + cause: event.cause, + fence: event.fence, + acquisitionGeneration: event.acquisitionGeneration, + ...(event.settlementRetryRequired + ? { settlementRetryRequired: event.settlementRetryRequired } + : {}) + }) + } + }, + ...(deps.openClaudeConnection ? { openConnection: deps.openClaudeConnection } : {}), + ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) + }) +} diff --git a/src/main/runtime/structured-tui-process-identity.test.ts b/src/main/runtime/structured-tui-process-identity.test.ts index 4324f8c5a26..fb5919824e5 100644 --- a/src/main/runtime/structured-tui-process-identity.test.ts +++ b/src/main/runtime/structured-tui-process-identity.test.ts @@ -238,6 +238,41 @@ describe('structured TUI process identity', () => { } }) + it('does not call a child absent after a single look that outlasted the budget', async () => { + // Measured on a 2,085-process host under load: one whole-machine `ps` took 6.2s while the + // shell-delivered child landed at ~3.5s. `ps` reads the table when it STARTS, so that one + // capture reported a t=0 machine and returned with the 5s budget already spent -- the loop + // answered "no exact child" without ever looking again. + let clockMs = 0 + let captures = 0 + await expect( + readStructuredTuiProcessIdentity({ + hostId: 'local', + rootPid: 100, + spawnToken: 'spawn-slow-ps', + agent: 'claude', + platform: 'darwin', + readPosixRows: async () => { + captures += 1 + const observedAtMs = clockMs + clockMs += 6_200 + return [ + { pid: 100, ppid: 1, stat: 'Ss', command: '/bin/zsh' }, + ...(observedAtMs >= 3_500 + ? [{ pid: 101, ppid: 100, stat: 'S+', command: 'claude --resume session-1' }] + : []) + ] + }, + readStartTime: async () => 1_700_000_000_000, + now: () => clockMs, + sleep: async (delayMs) => { + clockMs += delayMs + } + }) + ).resolves.toMatchObject({ pid: 101, spawnToken: 'spawn-slow-ps' }) + expect(captures).toBe(2) + }) + it('fails closed when the process snapshot omitted the PTY root', async () => { await expect( readStructuredTuiProcessIdentity({ diff --git a/src/main/runtime/structured-tui-process-identity.ts b/src/main/runtime/structured-tui-process-identity.ts index 0014351d781..f5ee019882b 100644 --- a/src/main/runtime/structured-tui-process-identity.ts +++ b/src/main/runtime/structured-tui-process-identity.ts @@ -20,6 +20,12 @@ const STRUCTURED_TUI_PROCESS_POLL_MS = 50 // window the added latency is bounded by one interval. const STRUCTURED_TUI_PROCESS_FAST_POLL_WINDOW_MS = 1_000 const STRUCTURED_TUI_PROCESS_MAX_POLL_MS = 500 +// Why a floor and not just the deadline: the first capture races the spawn it is looking for, +// so a null from it is absence of the child's arrival, not evidence the child is missing. The +// budget above assumes a look is nearly free, but one whole-machine `ps` measured 6.2s on a +// 2,085-process host under load -- long enough to spend the entire budget before the child +// (observed landing at ~3.5s) could exist, and answer "no exact child" after a single look. +const STRUCTURED_TUI_PROCESS_MIN_CAPTURES = 2 function descendants(rows: ProcessRow[], rootPid: number): (ProcessRow & { depth: number })[] { const children = new Map() @@ -175,6 +181,7 @@ export async function readStructuredTuiProcessIdentity(input: { const startedAtMs = now() const deadline = startedAtMs + (input.timeoutMs ?? STRUCTURED_TUI_PROCESS_WAIT_MS) let pollDelayMs = input.pollIntervalMs ?? STRUCTURED_TUI_PROCESS_POLL_MS + let captures = 0 while (true) { const rows: ProcessRow[] = @@ -186,6 +193,7 @@ export async function readStructuredTuiProcessIdentity(input: { foreground: false })) : posixRows(await (input.readPosixRows ?? getFreshProcessTableSnapshot)()) + captures += 1 let rootPresent = false for (const row of rows) { if (row.pid === input.rootPid) { @@ -218,11 +226,11 @@ export async function readStructuredTuiProcessIdentity(input: { } } const remainingMs = deadline - now() - if (remainingMs <= 0) { + if (remainingMs <= 0 && captures >= STRUCTURED_TUI_PROCESS_MIN_CAPTURES) { const label = input.agent === 'codex' ? 'Codex' : 'Claude' throw new Error(`The resumed terminal did not expose one exact ${label} child process.`) } - await sleep(Math.min(pollDelayMs, remainingMs)) + await sleep(Math.max(0, Math.min(pollDelayMs, remainingMs))) if (now() - startedAtMs >= STRUCTURED_TUI_PROCESS_FAST_POLL_WINDOW_MS) { // Never below the caller's interval, so an explicitly slow poll stays slow. pollDelayMs = Math.max( diff --git a/src/main/shell-wrapper-generated-file-snapshot.test.ts b/src/main/shell-wrapper-generated-file-snapshot.test.ts index ddbf1537984..36cd837e4fd 100644 --- a/src/main/shell-wrapper-generated-file-snapshot.test.ts +++ b/src/main/shell-wrapper-generated-file-snapshot.test.ts @@ -73,6 +73,7 @@ const CONTRACT_GLOBALS = new Set([ 'OPENCODE_CONFIG_DIR', 'PATH', 'PROMPT_COMMAND', + 'PS1', // Bash appends its non-printing Readline readiness marker. 'CURSOR', 'ZDOTDIR', 'precmd_functions', diff --git a/src/main/sqlite/harden-database-files.ts b/src/main/sqlite/harden-database-files.ts new file mode 100644 index 00000000000..2183f87700e --- /dev/null +++ b/src/main/sqlite/harden-database-files.ts @@ -0,0 +1,18 @@ +import { chmodSync, existsSync } from 'node:fs' + +/** Restrict a SQLite database and its sidecars to the owning user. */ +export function hardenSqliteDatabaseFiles(dbPath: (string & {}) | ':memory:'): void { + if (dbPath === ':memory:' || process.platform === 'win32') { + // Why: Windows protects these files through Orca's current-user-only userData DACL; POSIX mode bits are inert there. + return + } + for (const path of [dbPath, `${dbPath}-wal`, `${dbPath}-shm`]) { + try { + if (existsSync(path)) { + chmodSync(path, 0o600) + } + } catch { + // Why: best-effort — a mount that rejects chmod (SSHFS, some network shares) must not fail DB startup. + } + } +} diff --git a/src/main/startup/configure-process.test.ts b/src/main/startup/configure-process.test.ts index ef7ec6a9b69..7d7e2b0cc1a 100644 --- a/src/main/startup/configure-process.test.ts +++ b/src/main/startup/configure-process.test.ts @@ -1,4 +1,4 @@ -import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { homedir, tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' @@ -137,6 +137,29 @@ describe('patchPackagedProcessPath', () => { expect(segments).toContain('/usr/local/bin') }) + // Why derived, not a second literal: system-cli-install-dirs.ts documents its + // order as matching this seed's system block, and hardcoding the order in the + // fallback's own test lets a reorder here break that parity while both stay green. + it('seeds the system block in the order the install-dir fallback expects', async () => { + const { app } = await import('electron') + const { patchPackagedProcessPath } = await import('./configure-process') + const { getSystemCliInstallDirectories } = await import('../../shared/system-cli-install-dirs') + + setPlatform('linux') + Object.defineProperty(app, 'isPackaged', { configurable: true, value: true }) + process.env.HOME = '/home/tester' + process.env.PATH = '/usr/bin:/bin' + + patchPackagedProcessPath() + + const segments = (process.env.PATH ?? '').split(':') + const offsets = getSystemCliInstallDirectories('linux', '/home/tester').map((directory) => + segments.indexOf(directory) + ) + expect(offsets.every((offset) => offset >= 0)).toBe(true) + expect([...offsets].sort((a, b) => a - b)).toEqual(offsets) + }) + // Why this ordering is load-bearing (#18234): a seed exists so a GUI-launched // Electron can *find* a tool, not to re-rank tools the user already has. // `~/.local/bin` is user-writable and can hold a wrapper for any system tool. @@ -444,6 +467,43 @@ describe('configureElectronNetworkCompatibility', () => { ).toBe(false) }) + it('answers from the marker without reading the settings file', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const { writeHttp1CompatibilityMarker } = await import('./http1-compatibility-marker') + const userDataPath = createUserDataDir({ electronHttp1CompatibilityMode: false }) + writeHttp1CompatibilityMarker(userDataPath, true) + rmSync(join(userDataPath, 'orca-data.json'), { force: true }) + + expect(shouldDisableHttp2ForElectronNetworking({ env: {}, userDataPath })).toBe(true) + }) + + it('falls back to the settings file when no marker has been written yet', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const userDataPath = createUserDataDir({ electronHttp1CompatibilityMode: true }) + + expect(existsSync(join(userDataPath, 'http1-compatibility.json'))).toBe(false) + expect(shouldDisableHttp2ForElectronNetworking({ env: {}, userDataPath })).toBe(true) + }) + + it('falls back to the settings file when the marker is corrupt', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const userDataPath = createUserDataDir({ electronHttp1CompatibilityMode: true }) + writeFileSync(join(userDataPath, 'http1-compatibility.json'), '{ not json', 'utf-8') + + expect(shouldDisableHttp2ForElectronNetworking({ env: {}, userDataPath })).toBe(true) + }) + + it('lets the environment override the marker', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const { writeHttp1CompatibilityMarker } = await import('./http1-compatibility-marker') + const userDataPath = createUserDataDir({}) + writeHttp1CompatibilityMarker(userDataPath, true) + + expect( + shouldDisableHttp2ForElectronNetworking({ env: { ORCA_DISABLE_HTTP2: '0' }, userDataPath }) + ).toBe(false) + }) + it('appends Electron disable-http2 before sessions are created', async () => { const { app } = await import('electron') const { configureElectronNetworkCompatibility } = await import('./configure-process') diff --git a/src/main/startup/configure-process.ts b/src/main/startup/configure-process.ts index 6b167d7fe14..13dbd70c2c3 100644 --- a/src/main/startup/configure-process.ts +++ b/src/main/startup/configure-process.ts @@ -5,6 +5,7 @@ import { join, resolve } from 'node:path' import { getVersionManagerBinPaths } from '../codex-cli/command' import { getMainE2EConfig } from '../e2e-config' import { DISABLED_CHROMIUM_FEATURES } from './disabled-chromium-features' +import { readHttp1CompatibilityMarker } from './http1-compatibility-marker' const DEV_PARENT_SHUTDOWN_GRACE_MS = 3000 const HTTP1_COMPATIBILITY_ENV_VAR = 'ORCA_DISABLE_HTTP2' @@ -54,7 +55,13 @@ export function shouldDisableHttp2ForElectronNetworking( if (envValue !== null) { return envValue } - return readPersistedHttp1CompatibilityMode(options.userDataPath ?? app.getPath('userData')) + const userDataPath = options.userDataPath ?? app.getPath('userData') + // Why the marker first: this runs before app.whenReady(), and the settings file is the multi-MB + // orca-data.json the Store parses again moments later. The marker is refreshed whenever settings + // change, so the full read only happens on a profile that has never written one. + return ( + readHttp1CompatibilityMarker(userDataPath) ?? readPersistedHttp1CompatibilityMode(userDataPath) + ) } export function configureElectronNetworkCompatibility( diff --git a/src/main/startup/http1-compatibility-marker.ts b/src/main/startup/http1-compatibility-marker.ts new file mode 100644 index 00000000000..9e85a83be66 --- /dev/null +++ b/src/main/startup/http1-compatibility-marker.ts @@ -0,0 +1,51 @@ +import { readFileSync, writeFileSync } from 'node:fs' +import { join } from 'node:path' + +/** + * Cached copy of `settings.electronHttp1CompatibilityMode` for pre-`ready` startup. + * + * Why a standalone file (not the Store): app.commandLine.appendSwitch('disable-http2') must run + * before the first Electron session exists, which is before the settings Store is constructed. + * Reading it from the settings file meant a synchronous read + JSON.parse of the whole multi-MB + * orca-data.json on the critical path of every cold start, duplicating the parse the Store does a + * moment later. This marker is a few bytes, mirroring gpu-fallback-marker.ts. + */ + +export const HTTP1_COMPATIBILITY_MARKER_FILE = 'http1-compatibility.json' +const MARKER_SCHEME_VERSION = 1 + +type Http1CompatibilityMarker = { + schemeVersion: number + enabled: boolean +} + +function markerPath(userDataPath: string): string { + return join(userDataPath, HTTP1_COMPATIBILITY_MARKER_FILE) +} + +/** Returns null when the marker is missing or unreadable, so callers fall back to the settings file. */ +export function readHttp1CompatibilityMarker(userDataPath: string): boolean | null { + try { + const parsed = JSON.parse( + readFileSync(markerPath(userDataPath), 'utf-8') + ) as Partial + if (parsed.schemeVersion !== MARKER_SCHEME_VERSION || typeof parsed.enabled !== 'boolean') { + return null + } + return parsed.enabled + } catch { + return null + } +} + +export function writeHttp1CompatibilityMarker(userDataPath: string, enabled: boolean): void { + if (readHttp1CompatibilityMarker(userDataPath) === enabled) { + return + } + const marker: Http1CompatibilityMarker = { schemeVersion: MARKER_SCHEME_VERSION, enabled } + try { + writeFileSync(markerPath(userDataPath), JSON.stringify(marker)) + } catch { + // Best effort: a missing marker just costs the next launch the settings-file fallback. + } +} diff --git a/src/main/startup/main-process-preflight.ts b/src/main/startup/main-process-preflight.ts index 269eb2212e7..177a2357441 100644 --- a/src/main/startup/main-process-preflight.ts +++ b/src/main/startup/main-process-preflight.ts @@ -48,7 +48,7 @@ import { } from './single-instance-lock' import { setAppEnvironment } from '../../shared/app-environment' import { ElectronAppEnvironment } from '../host/electron-app-environment' -import { installProcessTreeKillBreadcrumbObserver } from '../crash-reporting/self-initiated-tree-kill-log' +import { installMainProcessTreeKillGate } from '../own-chromium-tree-kill-guard' import { setSecretStore } from '../../shared/secret-store' import { ElectronSecretStore } from '../host/electron-secret-store' import { setPtyHostBindings } from '../ipc/pty-host-bindings' @@ -163,8 +163,8 @@ export function runMainProcessPreflight(options: MainProcessPreflightOptions): b }) } // Why before any spawn: `signalProcessTree` is shared with the CLI and relay, so - // it can only reach the main-process breadcrumb store through a registered observer. - installProcessTreeKillBreadcrumbObserver() + // it can only reach the main-process guard and breadcrumb store once this is registered. + installMainProcessTreeKillGate() const isDev = is.dev configureDevUserDataPath(isDev) configureOrcaUserDataPathEnv() diff --git a/src/main/startup/main-process-ready-foundation.ts b/src/main/startup/main-process-ready-foundation.ts index 171aaf50421..3c0e01fe09e 100644 --- a/src/main/startup/main-process-ready-foundation.ts +++ b/src/main/startup/main-process-ready-foundation.ts @@ -41,6 +41,7 @@ import { registerDocPreviewGrantHandlers } from '../ipc/doc-preview-grant-ipc' import { initializeBrowserSessionsForApp } from '../browser/browser-session-startup' import { browserSessionRegistry } from '../browser/browser-session-registry' import { logStartupMilestone } from './startup-diagnostics' +import { writeHttp1CompatibilityMarker } from './http1-compatibility-marker' import { mainProcessState as state } from './main-process-state' import { recordDurableCrashBreadcrumb } from '../crash-reporting/durable-crash-breadcrumb' import { syncMacMenuBarIcon } from './main-window-actions' @@ -193,9 +194,20 @@ export async function initializeReadyFoundation(): Promise { } wslHookRelayManager.setManagedHookSettingsResolver(() => state.store?.getSettings() ?? null) logStartupMilestone('store-loaded') + // Why: pre-`ready` startup reads this flag from a marker so it never has to parse orca-data.json. + writeHttp1CompatibilityMarker( + canonicalUserDataPath, + store.getSettings().electronHttp1CompatibilityMode === true + ) // Why: apply initial fallback WSL distro from store settings for global git/CLI calls. setDefaultWslDistroOverride(store.getSettings().terminalWindowsWslDistro ?? null) store.onSettingsChanged((updates, settings) => { + if ('electronHttp1CompatibilityMode' in updates) { + writeHttp1CompatibilityMarker( + canonicalUserDataPath, + settings.electronHttp1CompatibilityMode === true + ) + } if ('terminalWindowsWslDistro' in updates) { // Why: synchronize fallback WSL distro updates to runner. setDefaultWslDistroOverride(settings.terminalWindowsWslDistro ?? null) diff --git a/src/main/startup/main-window-core-services.ts b/src/main/startup/main-window-core-services.ts index 759a4b2b100..d3ece383ee3 100644 --- a/src/main/startup/main-window-core-services.ts +++ b/src/main/startup/main-window-core-services.ts @@ -15,6 +15,7 @@ import { import { prepareCodexRuntimeHomeForLaunch } from './codex-launch-preparation' import { prepareCodexSessionResumeForLaunch } from './codex-session-resume-launch' import { isRecoveryReloadInFlight } from './main-window-lifecycle-flags' +import { RELAY_HOST_CLOSE_REASON } from '../../shared/relay-host-close-reason' export function attachMainWindowCoreServices( window: BrowserWindow, @@ -90,7 +91,10 @@ export function attachMainWindowCoreServices( }) }, onOrcaProfileAuthMutation: () => state.desktopRelayService?.authMutated(), - onBeforeOrcaProfileSignOut: () => state.desktopRelayService?.fenceAndCloseNow() + // Sign-out is the one fence a paired phone can be told about; quit and + // relaunch above stay reasonless so a restart never reads as signed out. + onBeforeOrcaProfileSignOut: () => + state.desktopRelayService?.fenceAndCloseNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) }, state.pluginService ?? undefined, state.pluginMarketplaceService && state.pluginMarketplaceInstaller diff --git a/src/main/text-generation/commit-message-text-generation-cancellation.test.ts b/src/main/text-generation/commit-message-text-generation-cancellation.test.ts index 9cfde774384..820af710fb9 100644 --- a/src/main/text-generation/commit-message-text-generation-cancellation.test.ts +++ b/src/main/text-generation/commit-message-text-generation-cancellation.test.ts @@ -101,7 +101,7 @@ describe('generateCommitMessageFromContext', () => { cancelGenerateCommitMessageLocal('/repo') - expectChildTerminated(children[0]!) + await expectChildTerminated(children[0]!) expect(children[1]?.kill).not.toHaveBeenCalled() children[0]?.listeners.get('close')?.(null) @@ -192,7 +192,7 @@ describe('generateCommitMessageFromContext', () => { cancelGeneratePullRequestFieldsLocal('/repo') expect(children[0]?.kill).not.toHaveBeenCalled() - expectChildTerminated(children[1]!) + await expectChildTerminated(children[1]!) const commitStdout = children[0]?.listeners.get('stdout:data') commitStdout?.(Buffer.from('Update README\n')) @@ -250,7 +250,7 @@ describe('generateCommitMessageFromContext', () => { cancelGeneratePullRequestFieldsLocal('/repo') listeners.get('close')?.(null) - expectChildTerminated(child) + await expectChildTerminated(child) await expect(pullRequest).resolves.toEqual({ success: false, error: 'Generation canceled.', @@ -306,7 +306,7 @@ describe('generateCommitMessageFromContext', () => { ) cancelGenerateCommitMessageLocal('/repo') - expectChildTerminated(child) + await expectChildTerminated(child) await Promise.resolve() await Promise.resolve() await Promise.resolve() @@ -344,7 +344,7 @@ describe('generateCommitMessageFromContext', () => { error: 'Generation canceled.', canceled: true }) - expectChildTerminated(firstChild) + await expectChildTerminated(firstChild) const second = generateCommitMessageFromContext(context, params, { kind: 'local', @@ -377,7 +377,7 @@ describe('generateCommitMessageFromContext', () => { await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledTimes(1)) cancelGenerateCommitMessageLocal('/descendant-repo') await expect(first).resolves.toMatchObject({ canceled: true }) - expectChildTerminated(firstChild) + await expectChildTerminated(firstChild) // SIGKILL reaches the codex process but not a grandchild that inherited its // stdout, so 'exit' arrives and 'close' never does. diff --git a/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts b/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts index c124cb85779..f73152255da 100644 --- a/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts +++ b/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts @@ -71,7 +71,7 @@ describe('generateCommitMessageFromContext', () => { error: 'agent CLI command produced too much output. Check the agent CLI configuration and try again.' }) - expectChildTerminated(child) + await expectChildTerminated(child) }) it('passes prepared provider environment to local agent subprocesses', async () => { diff --git a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts index 4bc1a720f05..9a1f932f0a0 100644 --- a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts +++ b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts @@ -325,7 +325,7 @@ describe('discoverCommitMessageModelsLocal', () => { await vi.advanceTimersByTimeAsync(60_000) await assertion - expectChildTerminated(child) + await expectChildTerminated(child) expect(child.stdout.listenerCount('data')).toBe(0) expect(child.stderr.listenerCount('data')).toBe(0) expect(child.listenerCount('error')).toBe(0) @@ -352,7 +352,7 @@ describe('discoverCommitMessageModelsLocal', () => { success: false, error: 'Codex model discovery timed out after 60s.' }) - expectChildTerminated(firstChild) + await expectChildTerminated(firstChild) expect(spawnMock).toHaveBeenCalledTimes(1) firstChild.emit('close', null) @@ -413,7 +413,7 @@ describe('discoverCommitMessageModelsLocal', () => { success: false, error: 'Cursor returned too much model data.' }) - expectChildTerminated(child) + await expectChildTerminated(child) expect(child.stdout.listenerCount('data')).toBe(0) expect(child.stderr.listenerCount('data')).toBe(0) expect(child.listenerCount('error')).toBe(0) diff --git a/src/main/text-generation/commit-message-text-generation-test-harness.ts b/src/main/text-generation/commit-message-text-generation-test-harness.ts index 21103d71f40..8ba925ef087 100644 --- a/src/main/text-generation/commit-message-text-generation-test-harness.ts +++ b/src/main/text-generation/commit-message-text-generation-test-harness.ts @@ -33,15 +33,17 @@ export function withPlatform(platform: NodeJS.Platform, fn: () => T): T { // expectChildTerminated(child) with no extra argument. export function createChildTerminationExpectation( terminateWindowsProcessTreeMock: ReturnType -): (child: { pid: number; kill: ReturnType }) => void { - return (child) => { +): (child: { pid: number; kill: ReturnType }) => Promise { + return async (child) => { if (process.platform === 'win32') { expect(terminateWindowsProcessTreeMock).toHaveBeenCalledWith(child.pid, { site: 'source-control-text-generation' }) - expect(child.kill).not.toHaveBeenCalled() - return } - expect(child.kill).toHaveBeenCalledWith('SIGKILL') + // Every platform kills the root by its own handle. On win32 that is not a + // duplicate of the tree walk: it is what keeps a refused walk from resolving + // having killed nothing while the caller releases the managed-home lock. It + // runs after the walk there, so it can be a tick behind the caller. + await vi.waitFor(() => expect(child.kill).toHaveBeenCalledWith('SIGKILL')) } } diff --git a/src/main/text-generation/source-control-local-process.ts b/src/main/text-generation/source-control-local-process.ts index 170dead6b52..3f046494f0f 100644 --- a/src/main/text-generation/source-control-local-process.ts +++ b/src/main/text-generation/source-control-local-process.ts @@ -22,22 +22,25 @@ import type { TextGenerationOperation } from './source-control-text-generation-types' -export function killSourceControlAgentProcess( +export async function killSourceControlAgentProcess( child: SpawnedSourceControlAgentProcess ): Promise { const pid = child.pid if (!pid) { - return Promise.resolve() + return } if (process.platform === 'win32') { - return terminateWindowsProcessTree(pid, { site: 'source-control-text-generation' }) + // taskkill owns the tree, but the own-Chromium gate can refuse the + // pid-addressed walk; the handle-addressed root kill below cannot reach the + // recycled pid it refused, and callers release the managed-home lock on this + // promise, so it must not resolve having killed nothing. + await terminateWindowsProcessTree(pid, { site: 'source-control-text-generation' }) } try { child.kill('SIGKILL') } catch { // The process may exit between the PID check and kill. } - return Promise.resolve() } export function runLocalSourceControlPlan(input: { diff --git a/src/main/windows-descendant-exit-verification.test.ts b/src/main/windows-descendant-exit-verification.test.ts new file mode 100644 index 00000000000..392c44399e7 --- /dev/null +++ b/src/main/windows-descendant-exit-verification.test.ts @@ -0,0 +1,175 @@ +import { describe, expect, it, vi } from 'vitest' +import { + captureWindowsDescendantSnapshot, + terminateIdentifiedWindowsProcessTree, + verifyWindowsDescendantSnapshotExit, + type WindowsDescendantSnapshot +} from './windows-descendant-exit-verification' + +function snapshot( + descendants: { pid: number; creationTimeMs: number }[], + unidentifiedCount = 0 +): WindowsDescendantSnapshot { + return { + root: { pid: 100, creationTimeMs: 5 }, + descendants, + unidentifiedCount, + capturedAtMs: 1_700_000_000_000 + } +} + +describe('captureWindowsDescendantSnapshot', () => { + it('walks the whole subtree and keeps only rows a later read can re-identify', async () => { + const captured = await captureWindowsDescendantSnapshot(100, { + // 400 is a grandchild; 300 denied a creation-time query, so no later read + // could tell it from a recycled pid and signalling it would risk a stranger. + readTable: vi.fn(async () => [ + { pid: 100, ppid: 1, creationTimeMs: 5 }, + { pid: 200, ppid: 100, creationTimeMs: 7 }, + { pid: 300, ppid: 100 }, + { pid: 400, ppid: 200, creationTimeMs: 9 }, + { pid: 500, ppid: 1, creationTimeMs: 11 } + ]), + now: () => 42 + }) + + expect(captured).toEqual({ + root: { pid: 100, creationTimeMs: 5 }, + descendants: [ + { pid: 400, creationTimeMs: 9 }, + { pid: 200, creationTimeMs: 7 } + ], + // Seen but not re-identifiable: counted, so no later read can prove it gone. + unidentifiedCount: 1, + capturedAtMs: 42 + }) + }) + + it('reports an unreadable or rootless table as no snapshot rather than an empty one', async () => { + await expect( + captureWindowsDescendantSnapshot(100, { + readTable: vi.fn(async () => { + throw new Error('table unavailable') + }) + }) + ).resolves.toBeNull() + // A snapshot without the root is stale or filtered; only an observed root + // can authoritatively have no descendants. + await expect( + captureWindowsDescendantSnapshot(100, { + readTable: vi.fn(async () => [{ pid: 999, ppid: 1, creationTimeMs: 5 }]) + }) + ).resolves.toBeNull() + }) + + it('refuses an invalid root pid', async () => { + const readTable = vi.fn() + await expect(captureWindowsDescendantSnapshot(0, { readTable })).resolves.toBeNull() + expect(readTable).not.toHaveBeenCalled() + }) +}) + +describe('verifyWindowsDescendantSnapshotExit', () => { + it('proves an empty tree without reading the table', async () => { + const readTable = vi.fn() + await expect(verifyWindowsDescendantSnapshotExit(snapshot([]), { readTable })).resolves.toBe( + 'exited' + ) + expect(readTable).not.toHaveBeenCalled() + }) + + it('never proves a tree that held a descendant it could not identify', async () => { + // A descendant that denied the creation-time query was seen in the table; + // being unable to re-identify it is "could not look", never "it is gone". + const readTable = vi.fn() + await expect(verifyWindowsDescendantSnapshotExit(snapshot([], 1), { readTable })).resolves.toBe( + 'unverifiable' + ) + expect(readTable).not.toHaveBeenCalled() + + // The identified sibling leaving proves nothing about the unidentified one. + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }], 1), { + readTable: vi.fn(async () => []), + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(1) + }) + ).resolves.toBe('unverifiable') + }) + + it('reports exited once no identity-matched row remains', async () => { + const readTable = vi + .fn() + .mockResolvedValueOnce([{ pid: 200, ppid: 100, creationTimeMs: 7 }]) + // The pid came back on a different process; that is a recycle, not a survivor. + .mockResolvedValueOnce([{ pid: 200, ppid: 100, creationTimeMs: 99 }]) + + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable, + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(1) + }) + ).resolves.toBe('exited') + expect(readTable).toHaveBeenCalledTimes(2) + }) + + it('reports live for a descendant still matched at the deadline', async () => { + let clock = 0 + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable: vi.fn(async () => [{ pid: 200, ppid: 100, creationTimeMs: 7 }]), + wait: async () => { + clock += 100 + }, + now: () => clock, + verifyMs: 250 + }) + ).resolves.toBe('live') + }) + + it('reports unverifiable when the table cannot be read at the deadline', async () => { + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable: vi.fn(async () => { + throw new Error('table unavailable') + }), + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(9_999) + }) + ).resolves.toBe('unverifiable') + }) +}) + +describe('terminateIdentifiedWindowsProcessTree', () => { + it('never taskkills a replacement that reused the captured root pid', async () => { + const terminateTree = vi.fn(async () => {}) + + await expect( + terminateIdentifiedWindowsProcessTree( + { pid: 100, creationTimeMs: 5 }, + { + readTable: vi.fn(async () => [{ pid: 100, ppid: 1, creationTimeMs: 99 }]), + terminateTree + } + ) + ).resolves.toBe(false) + expect(terminateTree).not.toHaveBeenCalled() + }) + + it('rechecks retained-child ownership after the identity read settles', async () => { + const terminateTree = vi.fn(async () => {}) + + await expect( + terminateIdentifiedWindowsProcessTree( + { pid: 100, creationTimeMs: 5 }, + { + readTable: vi.fn(async () => [{ pid: 100, ppid: 1, creationTimeMs: 5 }]), + ownsRoot: () => false, + terminateTree + } + ) + ).resolves.toBe(false) + expect(terminateTree).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/windows-descendant-exit-verification.ts b/src/main/windows-descendant-exit-verification.ts new file mode 100644 index 00000000000..079833a2bd6 --- /dev/null +++ b/src/main/windows-descendant-exit-verification.ts @@ -0,0 +1,156 @@ +import type { DescendantTreeVerdict } from './pty-descendant-exit-verification' +import { windowsDescendantsFromRows } from './providers/windows-foreground-process-rows' +import { readWindowsProcessTableFresh } from './windows/windows-process-table' +import { terminateWindowsProcessTree } from './windows-process-tree-kill' + +export const WINDOWS_DESCENDANT_KILL_VERIFY_MS = 3_500 +const WINDOWS_DESCENDANT_POLL_MS = 100 + +/** + * A Windows descendant tree captured while its root was alive, with the + * PID-reuse guard the POSIX snapshot gets from ps lstart: a row only counts as + * the same process when its creation time still matches. Rows without a + * creation time are never signalled, because a bare pid cannot be re-identified, + * but they are counted: a descendant that was seen and denied identification + * is one no later read can prove gone. + */ +export type WindowsProcessIdentity = { pid: number; creationTimeMs: number } + +export type WindowsDescendantSnapshot = { + root: WindowsProcessIdentity + descendants: WindowsProcessIdentity[] + /** Descendants seen in the walk that denied the creation-time query. */ + unidentifiedCount: number + capturedAtMs: number + /** Per-PID boundaries retained when close refreshes merge snapshots. */ + capturedAtMsByPid?: Readonly> +} + +export type WindowsDescendantVerificationDeps = { + readTable?: () => Promise<{ pid: number; ppid: number; creationTimeMs?: number }[]> + now?: () => number + wait?: (ms: number) => Promise + verifyMs?: number +} + +/** Revalidate a Windows PID/creation-time identity immediately before a kill. */ +export async function verifyWindowsProcessIdentity( + target: WindowsProcessIdentity, + deps: Pick = {} +): Promise { + if (!Number.isInteger(target.pid) || target.pid <= 0 || !Number.isFinite(target.creationTimeMs)) { + return false + } + const table = await (deps.readTable ?? readWindowsProcessTableFresh)().catch(() => null) + const current = table?.filter((row) => row.pid === target.pid) ?? [] + return current.length === 1 && current[0]?.creationTimeMs === target.creationTimeMs +} + +function delay(ms: number): Promise { + return new Promise((resolve) => { + const timer = setTimeout(resolve, ms) + timer.unref?.() + }) +} + +/** + * Snapshot a Windows root's descendants while it is still alive. Resolves null + * (never rejects) when the table is unreadable or the root is absent — the same + * contract as the POSIX walk, because "cannot see" is never "nothing is there". + */ +export async function captureWindowsDescendantSnapshot( + rootPid: number, + deps: WindowsDescendantVerificationDeps = {} +): Promise { + if (!Number.isInteger(rootPid) || rootPid <= 0) { + return null + } + const capturedAtMs = (deps.now ?? Date.now)() + // One table read, not a walk plus an identity read: each is bounded in + // seconds, and this runs inside the close ladder's budget. + const table = await (deps.readTable ?? readWindowsProcessTableFresh)().catch(() => null) + const descendants = table && windowsDescendantsFromRows(table, rootPid) + const root = table?.find((row) => row.pid === rootPid) + if (!descendants || typeof root?.creationTimeMs !== 'number') { + return null + } + return { + root: { pid: root.pid, creationTimeMs: root.creationTimeMs }, + descendants: descendants.flatMap((row) => + // A descendant that denied a creation-time query cannot be told from a + // recycled pid later, so it is never signalled on a bare pid. + typeof row.creationTimeMs === 'number' + ? [{ pid: row.pid, creationTimeMs: row.creationTimeMs }] + : [] + ), + unidentifiedCount: descendants.filter((row) => typeof row.creationTimeMs !== 'number').length, + capturedAtMs + } +} + +export type IdentifiedWindowsTreeTerminationDeps = { + readTable?: WindowsDescendantVerificationDeps['readTable'] + terminateTree?: (target: WindowsProcessIdentity) => Promise + ownsRoot?: () => boolean +} + +/** Revalidate the captured root at the last async boundary before taskkill. */ +export async function terminateIdentifiedWindowsProcessTree( + target: WindowsProcessIdentity, + deps: IdentifiedWindowsTreeTerminationDeps = {} +): Promise { + if (!(await verifyWindowsProcessIdentity(target, { readTable: deps.readTable }))) { + return false + } + if (deps.ownsRoot?.() === false) { + return false + } + await ( + deps.terminateTree ?? + ((identified: WindowsProcessIdentity) => terminateWindowsProcessTree(identified.pid)) + )(target) + return true +} + +/** + * Whether a snapshotted Windows tree is gone, polled to a bounded deadline. + * + * Why a verification pass at all: `taskkill /T /F` resolves the same way on a + * timeout, an access denial and a recycled root as it does on a successful + * kill, so its completion is never evidence. Only a table read that no longer + * shows an identity-matched row is. + */ +export async function verifyWindowsDescendantSnapshotExit( + snapshot: WindowsDescendantSnapshot, + deps: WindowsDescendantVerificationDeps = {} +): Promise { + // The most a read can prove: a descendant that denied identification was seen + // and can never be matched gone, so "could not look" caps the verdict. + const proven: DescendantTreeVerdict = snapshot.unidentifiedCount > 0 ? 'unverifiable' : 'exited' + if (snapshot.descendants.length === 0) { + return proven + } + const now = deps.now ?? Date.now + const readTable = deps.readTable ?? readWindowsProcessTableFresh + const deadline = now() + (deps.verifyMs ?? WINDOWS_DESCENDANT_KILL_VERIFY_MS) + let verdict: DescendantTreeVerdict = 'unverifiable' + do { + const table = await readTable().catch(() => null) + if (!table) { + verdict = 'unverifiable' + } else { + const live = new Map(table.map((row) => [row.pid, row.creationTimeMs])) + verdict = snapshot.descendants.some((row) => live.get(row.pid) === row.creationTimeMs) + ? 'live' + : proven + if (verdict === proven) { + return verdict + } + } + if (now() >= deadline) { + return verdict + } + await (deps.wait ?? delay)(WINDOWS_DESCENDANT_POLL_MS) + } while (now() < deadline) + return verdict +} diff --git a/src/main/windows-live-tree-kill.win32.test.ts b/src/main/windows-live-tree-kill.win32.test.ts new file mode 100644 index 00000000000..45563e82c89 --- /dev/null +++ b/src/main/windows-live-tree-kill.win32.test.ts @@ -0,0 +1,197 @@ +import { existsSync, mkdtempSync, readFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { spawn, ChildProcess } from 'node:child_process' +import { subscribe, unsubscribe } from 'node:diagnostics_channel' +import { afterAll, afterEach, beforeEach, describe, expect, it } from 'vitest' +import { setAppEnvironment, type AppEnvironment } from '../shared/app-environment' +import { setProcessTreeKillGate } from '../shared/child-process/process-tree-kill-gate' +import { signalProcessTree } from '../shared/child-process/process-tree-termination' +import { removeTreeSync } from '../shared/windows-transient-lock-removal' +import { + findSelfInitiatedTreeKills, + resetSelfInitiatedTreeKillLogForTest +} from './crash-reporting/self-initiated-tree-kill-log' +import { installMainProcessTreeKillGate } from './own-chromium-tree-kill-guard' +import { terminateWindowsProcessTree } from './windows-process-tree-kill' + +/** + * The unit tests pin the gate's decision against a mocked `taskkill`; this pins + * what that decision does to real Windows processes. + * + * Both are needed. Every claim the gate makes is about a mechanism the mocks + * cannot show: that `taskkill /T /F` actually reaps a detached grandchild, that + * a refusal actually leaves that tree standing, and that the handle-addressed + * root kill the refusal path falls back to actually reaps the root while + * orphaning its descendants — the asymmetry the PR discloses rather than fixes. + * + * Runs only on win32; skipped elsewhere. + */ +const describeOnWindows = process.platform === 'win32' ? describe : describe.skip + +/** Read live by the guard on every kill, so a case can flip it mid-test. */ +let orcaChromiumPids: number[] = [] + +function appEnvironment(): AppEnvironment { + return { + getPath: () => process.cwd(), + getAppPath: () => process.cwd(), + getVersion: () => '0.0.0-live', + isPackaged: () => false, + onWillQuit: () => {}, + exit: () => {}, + getAppMetrics: (() => + orcaChromiumPids.map((pid) => ({ + pid, + type: 'Tab' + }))) as unknown as AppEnvironment['getAppMetrics'] + } +} + +function isAlive(pid: number): boolean { + try { + process.kill(pid, 0) + return true + } catch { + return false + } +} + +const sleep = (ms: number): Promise => new Promise((resolve) => setTimeout(resolve, ms)) + +async function waitFor(predicate: () => boolean, timeoutMs = 10_000): Promise { + const deadline = Date.now() + timeoutMs + while (Date.now() < deadline && !predicate()) { + await sleep(100) + } + return predicate() +} + +let markerDirectory = '' +let markerSequence = 0 +const spawnedRoots: ChildProcess[] = [] +const spawnedLeaves: number[] = [] +const observedSpawns: ChildProcess[] = [] + +function observeSpawn(message: unknown): void { + if ( + typeof message === 'object' && + message !== null && + 'process' in message && + message.process instanceof ChildProcess + ) { + observedSpawns.push(message.process) + } +} + +/** A real root with a real grandchild; the grandchild reports its pid on disk. */ +async function spawnLiveTree(): Promise<{ + child: ChildProcess + rootPid: number + leafPid: number +}> { + const marker = join(markerDirectory, `leaf-${markerSequence++}.pid`) + const leafSource = `require('node:fs').writeFileSync(${JSON.stringify(marker)}, String(process.pid)); setTimeout(() => {}, 600000)` + // Non-detached Windows children can die with the root's libuv Job Object. + const rootSource = `require('node:child_process').spawn(process.execPath, ['-e', ${JSON.stringify(leafSource)}], { stdio: 'ignore', detached: true, windowsHide: true }); setTimeout(() => {}, 600000)` + const child = spawn(process.execPath, ['-e', rootSource], { + stdio: 'ignore', + windowsHide: true + }) + spawnedRoots.push(child) + const rootPid = child.pid as number + expect(rootPid).toBeGreaterThan(0) + expect(await waitFor(() => existsSync(marker))).toBe(true) + const leafPid = Number(readFileSync(marker, 'utf8')) + spawnedLeaves.push(leafPid) + expect(await waitFor(() => isAlive(leafPid))).toBe(true) + return { child, rootPid, leafPid } +} + +describeOnWindows('own-Chromium gate against real Windows process trees', () => { + beforeEach(() => { + markerDirectory ||= mkdtempSync(join(tmpdir(), 'orca-live-tree-kill-')) + resetSelfInitiatedTreeKillLogForTest() + orcaChromiumPids = [] + setAppEnvironment(appEnvironment()) + installMainProcessTreeKillGate() + observedSpawns.length = 0 + subscribe('child_process', observeSpawn) + }) + + afterEach(async () => { + unsubscribe('child_process', observeSpawn) + orcaChromiumPids = [] + for (const leafPid of spawnedLeaves.splice(0)) { + await terminateWindowsProcessTree(leafPid, { site: 'live-tree-kill-cleanup' }) + } + for (const root of spawnedRoots.splice(0)) { + root.kill('SIGKILL') + } + setProcessTreeKillGate(null) + }) + + afterAll(() => { + if (markerDirectory) { + removeTreeSync(markerDirectory) + } + }) + + it('admitted: taskkill reaps the root and its detached grandchild, and the kill is recorded', async () => { + const { rootPid, leafPid } = await spawnLiveTree() + + await terminateWindowsProcessTree(rootPid, { site: 'live-tree-kill-admit' }) + + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + expect(await waitFor(() => !isAlive(leafPid))).toBe(true) + expect( + findSelfInitiatedTreeKills(Date.now()).some( + (kill) => kill.pid === rootPid && kill.site === 'live-tree-kill-admit' + ) + ).toBe(true) + }) + + it('refused: the tree survives, nothing is recorded, and the handle kill still reaps the root', async () => { + const { child, rootPid, leafPid } = await spawnLiveTree() + orcaChromiumPids = [rootPid] + + await terminateWindowsProcessTree(rootPid, { site: 'live-tree-kill-refuse' }) + + await sleep(1_000) + expect(isAlive(rootPid)).toBe(true) + expect(isAlive(leafPid)).toBe(true) + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([]) + + // The fallback every gated site runs after a refusal. + child.kill('SIGKILL') + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + // Let root-owned job cleanup finish before asserting independent survival. + await sleep(250) + // Disclosed asymmetry: a refusal orphans descendants rather than reaping them. + expect(isAlive(leafPid)).toBe(true) + }) + + it('signalProcessTree refused: the root goes by handle and the barrier reports unverified', async () => { + const { child, rootPid, leafPid } = await spawnLiveTree() + orcaChromiumPids = [rootPid] + observedSpawns.length = 0 + + await expect(signalProcessTree(child, 'SIGKILL')).resolves.toBe(false) + + expect(observedSpawns).toHaveLength(0) + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + await sleep(250) + expect(isAlive(leafPid)).toBe(true) + }) + + it('signalProcessTree admitted: the whole tree goes and the barrier reports verified', async () => { + const { child, rootPid, leafPid } = await spawnLiveTree() + observedSpawns.length = 0 + + await expect(signalProcessTree(child, 'SIGKILL')).resolves.toBe(true) + + expect(observedSpawns.map((child) => child.spawnfile)).toEqual(['taskkill']) + expect(await waitFor(() => !isAlive(rootPid))).toBe(true) + expect(await waitFor(() => !isAlive(leafPid))).toBe(true) + }) +}) diff --git a/src/main/windows-process-tree-kill.ts b/src/main/windows-process-tree-kill.ts index ea7b756ada4..31e6b18da2a 100644 --- a/src/main/windows-process-tree-kill.ts +++ b/src/main/windows-process-tree-kill.ts @@ -11,8 +11,9 @@ export const WINDOWS_PROCESS_TREE_KILL_TIMEOUT_MS = 5_000 * Best-effort: missing/already-dead roots still resolve so callers can finish * their own handle cleanup via killRoot. * - * Nearly every main-process taskkill runs through here; the two account-login - * teardowns keep their own spawn but share the same gate, so the refusal and the + * Most main-process taskkills run through here; the families that keep their own + * spawn (account-login teardowns, codex app-server deadline, git-command abort, + * notebook and precheck timeouts) share the same gate, so the refusal and the * breadcrumb live in `admitSelfInitiatedTreeKill` rather than in this function. */ export function terminateWindowsProcessTree( diff --git a/src/preload/api/orca-profile-api.ts b/src/preload/api/orca-profile-api.ts index 16c2a575078..80e9f08fc8d 100644 --- a/src/preload/api/orca-profile-api.ts +++ b/src/preload/api/orca-profile-api.ts @@ -28,6 +28,8 @@ import type { export type OrcaProfileApi = { list: () => Promise authStatus: () => Promise + /** Fires when main changed the stored auth status on its own (e.g. a revoked session). */ + onAuthStatusChanged: (callback: () => void) => () => void createLocal: (args?: CreateLocalOrcaProfileArgs) => Promise createCloudLinked: ( args?: CreateCloudLinkedOrcaProfileArgs diff --git a/src/preload/api/orca-profiles-bridge.ts b/src/preload/api/orca-profiles-bridge.ts index 0b2897f8ab1..da58b2d9def 100644 --- a/src/preload/api/orca-profiles-bridge.ts +++ b/src/preload/api/orca-profiles-bridge.ts @@ -1,9 +1,15 @@ import { ipcRenderer } from 'electron' import type { PreloadApi } from '../api-types' +import { ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL } from '../../shared/orca-profiles' export const orcaProfilesApi = { list: () => ipcRenderer.invoke('orcaProfiles:list'), authStatus: () => ipcRenderer.invoke('orcaProfiles:authStatus'), + onAuthStatusChanged: (callback: () => void): (() => void) => { + const listener = (): void => callback() + ipcRenderer.on(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL, listener) + return () => ipcRenderer.removeListener(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL, listener) + }, createLocal: (args) => ipcRenderer.invoke('orcaProfiles:createLocal', args), createCloudLinked: (args) => ipcRenderer.invoke('orcaProfiles:createCloudLinked', args), switchProfile: (args) => ipcRenderer.invoke('orcaProfiles:switch', args), diff --git a/src/relay/pty-handler-ownership-attestation.test.ts b/src/relay/pty-handler-ownership-attestation.test.ts index 94f462c46d2..ee1c144df09 100644 --- a/src/relay/pty-handler-ownership-attestation.test.ts +++ b/src/relay/pty-handler-ownership-attestation.test.ts @@ -33,7 +33,8 @@ import { endPtyHandlerTest, type MockDispatcher } from './pty-handler-test-harness' -import { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } from '../shared/process-table-snapshot-reader' +import * as processTableSnapshotReader from '../shared/process-table-snapshot-reader' +import { RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS } from '../shared/ssh-relay-pty-ownership-proof' const PANE_KEY = 'tab-agent:22222222-2222-4222-8222-222222222222' @@ -139,16 +140,41 @@ describe('PtyHandler publishes host-attested PTY ownership', () => { expect(entry?.ownerClientInstanceId).toBe('client-A') }) - it('dates the foreground observation instead of stamping it fresh', async () => { - // `capturedAgeMs` used to be a hardcoded 0 with no reader anywhere, so the one field that - // exists to bound staleness asserted the evidence was never stale. It now carries the - // actual age of the TTL-shared capture the record was derived from. - const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + it('publishes the age the capture reported, rather than restamping it fresh', async () => { + // This assertion used to read `capturedAgeMs <= PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS`, which + // could not fail: `beginPtyHandlerTest` installs fake timers, so `Date.now()` is frozen, the + // real reader reports exactly +0, and `0 <= 500` held identically for a hardcoded zero, for + // completion-stamping and for start-stamping. The one test guarding this field was blind to + // every change to it, while the real reader on a 2,002-process host returns thousands of ms. + // + // So drive a real age in from the reader. That the reader MEASURES the age correctly is + // pinned separately, against a controllable clock, by process-table-snapshot.test.ts; what + // belongs here is that the handler publishes what it was given instead of restamping. + const capturedAgeMs = 6_140 + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockResolvedValue({ rows: [], capturedAgeMs }) + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) const entry = (await listProcesses()).find((process) => process.id === id) - expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBeLessThanOrEqual( - PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + expect(snapshot).toHaveBeenCalled() + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBe(capturedAgeMs) + }) + + it('publishes an age a destructive consumer will refuse, rather than one it will trust', async () => { + // The point of the field, stated as the consumer sees it: an observation this old cannot + // authorize a stop, and the whole bug was that it used to arrive claiming it could. + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockResolvedValue({ rows: [], capturedAgeMs: 6_140 }) + + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(snapshot).toHaveBeenCalled() + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBeGreaterThan( + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS ) }) }) diff --git a/src/relay/pty-handler-spawn-admission.test.ts b/src/relay/pty-handler-spawn-admission.test.ts index 045ee2e412d..07fe8e9d88f 100644 --- a/src/relay/pty-handler-spawn-admission.test.ts +++ b/src/relay/pty-handler-spawn-admission.test.ts @@ -126,6 +126,46 @@ describe('PtyHandler', () => { expect(hasChildren).toHaveBeenLastCalledWith(mockPtyInstance.pid, { fresh: true }) }) + it('does not re-enter the shared capture after the evidence read gave up on it', async () => { + // The budget is worthless if the compatibility fields answer by joining the very capture the + // evidence read just abandoned: `inspectPtyChildProcesses` and `getForegroundProcessName` + // read the same TTL-shared table with no budget of their own, so on a slow host this call + // would still block for the whole capture -- once, then once per managed PTY in the listing. + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockRejectedValue(new Error('process table unreadable: capture_over_budget')) + const hasChildren = vi.spyOn(ptyChildProcessInspection, 'inspectPtyChildProcesses') + const foregroundName = vi.spyOn(ptyShellUtils, 'getForegroundProcessName') + + const { id } = (await spawnPty({ cols: 80, rows: 24 })) as { id: string } + hasChildren.mockClear() + foregroundName.mockClear() + + const inspection = (await dispatcher.callRequest('pty.inspectProcess', { id })) as { + hasChildProcesses: boolean + childProcessEvidence?: string + foregroundProcessEvidence?: { verdict: string; reason?: string } + } + + expect(snapshot).toHaveBeenCalled() + expect(hasChildren).not.toHaveBeenCalled() + // The verdict the gates already handle, reached promptly instead of late. + expect(inspection.foregroundProcessEvidence?.verdict).toBe('unverifiable') + expect(inspection.foregroundProcessEvidence?.reason).toBe('process_table_unreadable') + // The honest verdict rather than a fabricated negative, reached without the wait. The + // compatibility boolean still spells `unverifiable` as `false` for older clients. + expect(inspection.childProcessEvidence).toBe('unverifiable') + expect(inspection.hasChildProcesses).toBe(false) + + const listing = (await dispatcher.callRequest('pty.listProcesses', {})) as { + id: string + title: string + }[] + + expect(foregroundName).not.toHaveBeenCalled() + expect(listing.find((entry) => entry.id === id)?.title).toBeTruthy() + }) + it('rejects strict process inspection for a missing relay PTY', async () => { await expect(dispatcher.callRequest('pty.inspectProcess', { id: 'missing' })).rejects.toThrow( 'terminal_gone' diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 190bcabb3f9..4b7c6dac2d6 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -2663,6 +2663,9 @@ export class PtyHandler { } } let rows: readonly ProcessTableRow[] | null = null + // Set only when the budgeted evidence read gave up, so the compatibility fields below do not + // turn around and ask the same unreadable table again with no budget at all. + let tableUnavailable = false let evidence: RemoteForegroundEvidence | undefined if (process.platform === 'win32') { // Why SSH-to-Windows is always unverifiable: POSIX has a real foreground primitive @@ -2701,6 +2704,7 @@ export class PtyHandler { rows ) } catch { + tableUnavailable = true evidence = { authorityGeneration: this.ptyIdMintEpoch, observationEpoch: ++this.foregroundEvidenceEpoch, @@ -2727,13 +2731,19 @@ export class PtyHandler { // 1.36s CIM scan, and polling that would reinstate exactly the fork storm the shared table // exists to prevent (#15209, #15036). Close and cleanup decisions ask for the scan by name; // a poll gets the honest `unverifiable` instead of a fabricated negative. + // Why `tableUnavailable` first: it means the budgeted evidence read already gave up. Without + // this arm `inspectPtyChildProcesses` re-enters `getProcessTableSnapshot()` and joins the very + // capture this call just abandoned, blocking for all of it and spending the whole latency the + // budget exists to avoid. The destructive `pty.hasChildProcesses` RPC keeps its fresh probe. const childProcessEvidence: PtyChildProcessVerdict = rows ? rows.some((row) => row.ppid === managed.pty.pid) ? 'children' : 'no-children' - : process.platform === 'win32' && params.scanChildProcesses !== true + : tableUnavailable ? 'unverifiable' - : await inspectPtyChildProcesses(managed.pty.pid) + : process.platform === 'win32' && params.scanChildProcesses !== true + ? 'unverifiable' + : await inspectPtyChildProcesses(managed.pty.pid) return { foregroundProcess, // `unverifiable` keeps spelling itself `false` on the compatibility field, which is what @@ -2758,6 +2768,10 @@ export class PtyHandler { // process-table work on the host. const includeForegroundProcessEvidence = params.includeForegroundProcessEvidence !== false let evidenceRows: readonly ProcessTableRow[] | null = null + // Same reason as `inspectProcess`: once the budgeted read has given up, the per-PTY title + // fallback below must not re-enter the same capture without a budget -- and here it would do + // so once per managed PTY. + let evidenceTableUnavailable = false let evidenceResults: BatchedForegroundProcessResult[] = [] const evidenceEpoch = ++this.foregroundEvidenceEpoch // Worst-case capture time for the snapshot below, not the instant its await settled: the @@ -2783,6 +2797,7 @@ export class PtyHandler { } catch { // An unreadable capture is represented as unverifiable evidence below; // existing inventory fields remain available for old clients. + evidenceTableUnavailable = true } } for (const [entryIndex, [id, managed]] of managedEntries.entries()) { @@ -2797,7 +2812,7 @@ export class PtyHandler { const title = (evidenceRows ? (evidenceResults[entryIndex]?.processName ?? managed.pty.process ?? null) - : includeForegroundProcessEvidence + : includeForegroundProcessEvidence && !evidenceTableUnavailable ? await getForegroundProcessName(managed.pty.pid, managed.pty.process || null) : managed.pty.process || null) || 'shell' const foregroundProcessEvidence = diff --git a/src/renderer/src/assets/main.css b/src/renderer/src/assets/main.css index 8187fe496c7..e3d267cb353 100644 --- a/src/renderer/src/assets/main.css +++ b/src/renderer/src/assets/main.css @@ -392,9 +392,9 @@ z-index: 40 !important; } -/* Keep interruption controls above unrelated updater/onboarding chrome. */ +/* Above the z-40 updater/onboarding chrome, below the floating workspace panel's z-45. */ .native-chat-pane-shell:has([data-native-chat-working='true']) { - z-index: 50; + z-index: 44; } [data-sonner-toaster] [data-sonner-toast][data-styled='true'] { diff --git a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx new file mode 100644 index 00000000000..f2fd2ad533b --- /dev/null +++ b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx @@ -0,0 +1,310 @@ +// @vitest-environment happy-dom +import { act, cleanup, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { BrowserPage } from '../../../../shared/browser-workspace-types' + +const mocks = vi.hoisted(() => ({ + attach: vi.fn(), + detach: vi.fn(), + recordBreadcrumb: vi.fn() +})) + +vi.mock('./browser-client-page-renderer-installation', () => ({ + attachBrowserClientPageToViewport: mocks.attach +})) +vi.mock('@/lib/crash-breadcrumb-recorder', () => ({ + recordRendererCrashBreadcrumb: mocks.recordBreadcrumb +})) +vi.mock('sonner', () => ({ + toast: { error: vi.fn(), success: vi.fn(), loading: vi.fn(), message: vi.fn() } +})) + +import { TooltipProvider } from '@/components/ui/tooltip' +import { installClientHostedPaneApi } from './client-hosted-browser-pane-test-rig' +import { ClientHostedBrowserPagePane } from './ClientHostedBrowserPagePane' + +const PLACEMENT = { + kind: 'client' as const, + browserHostClientId: 'host-a', + browserHostGeneration: 3, + pageHostGeneration: 7 +} + +/** Verbatim from Electron 43.4.1: main destroyed the guest, the tag still holds its id. */ +function invalidGuestInstanceId(): Error { + return new Error('Invalid guestInstanceId: 7') +} + +/** Verbatim from Electron 43.4.1: focus() after the retained tag left the DOM. */ +function nullContentWindowFocus(): TypeError { + return new TypeError("Cannot read properties of null (reading 'focus')") +} + +function page(overrides?: Partial): BrowserPage { + return { + id: 'page-a', + workspaceId: 'workspace-a', + worktreeId: 'worktree-a', + url: 'https://example.internal/', + title: 'Example', + loading: false, + faviconUrl: null, + canGoBack: false, + canGoForward: false, + loadError: null, + createdAt: 1, + ...overrides + } +} + +function createGuest(): Electron.WebviewTag & { + getURL: ReturnType + reload: ReturnType +} { + const webview = document.createElement('webview') as Electron.WebviewTag & { + getURL: ReturnType + reload: ReturnType + } + Object.assign(webview, { + getURL: vi.fn(() => 'https://example.internal/'), + getTitle: vi.fn(() => 'Example'), + isLoading: vi.fn(() => false), + canGoBack: vi.fn(() => false), + canGoForward: vi.fn(() => false), + focus: vi.fn(), + blur: vi.fn(), + goBack: vi.fn(), + goForward: vi.fn(), + reload: vi.fn(), + loadURL: vi.fn(async () => {}) + }) + mocks.attach.mockReturnValue({ + webview, + detach: mocks.detach, + nextMetadataRevision: vi.fn(() => 1) + }) + return webview +} + +function paneElement( + isActive: boolean, + options?: { browserTab?: BrowserPage; onUpdatePageState?: (id: string, state: unknown) => void } +): React.JSX.Element { + return ( + + + + ) +} + +let webview: ReturnType + +beforeEach(() => { + mocks.attach.mockReset() + mocks.detach.mockReset() + mocks.recordBreadcrumb.mockReset() + installClientHostedPaneApi() + webview = createGuest() +}) + +afterEach(() => { + cleanup() + vi.clearAllMocks() +}) + +describe('client-hosted browser pane over a dead guest', () => { + it('degrades to the unavailable notice when the guest was destroyed in main', () => { + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + expect(() => render(paneElement(true))).not.toThrow() + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.detach).toHaveBeenCalled() + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_unavailable', { + browserPageId: 'page-a', + pageHostGeneration: PLACEMENT.pageHostGeneration, + reason: 'unreadable', + tagConnected: false + }) + // Why: the catch is total, so the swallowed error must stay visible to diagnostics. + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_read_failed', { + errorName: 'Error', + errorMessage: 'Invalid guestInstanceId: 7' + }) + }) + + it('stops the spinner it inherited from a page that died mid-load', () => { + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + const onUpdatePageState = vi.fn() + + render(paneElement(true, { browserTab: page({ loading: true }), onUpdatePageState })) + + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + }) + + it('flips to the unavailable notice when the guest renderer goes away after attach', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { browserTab: page({ loading: true }), onUpdatePageState })) + expect(screen.queryByText('Client-hosted browser unavailable')).toBeNull() + onUpdatePageState.mockClear() + + // The registry pulls the tag out of the DOM on this event without telling the pane. + webview.remove() + act(() => { + webview.dispatchEvent(new Event('render-process-gone')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.detach).toHaveBeenCalled() + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_unavailable', { + browserPageId: 'page-a', + pageHostGeneration: PLACEMENT.pageHostGeneration, + reason: 'render-process-gone', + tagConnected: false + }) + }) + + it('flips to the unavailable notice when main destroys the guest after attach', () => { + render(paneElement(true)) + + act(() => { + webview.dispatchEvent(new Event('destroyed')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith( + 'browser_client_page_guest_unavailable', + expect.objectContaining({ reason: 'destroyed' }) + ) + // The chrome must not keep driving the dead tag: Reload routes to the notice, not a throw. + webview.reload.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + expect(() => act(() => screen.getByRole('button', { name: 'Reload' }).click())).not.toThrow() + expect(webview.reload).not.toHaveBeenCalled() + }) + + it('does not freeze silently when a navigation event finds the guest gone', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + onUpdatePageState.mockClear() + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => { + webview.dispatchEvent(new Event('did-navigate')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + }) + + it('stops the spinner when the guest dies as a load starts', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + onUpdatePageState.mockClear() + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => { + webview.dispatchEvent(new Event('did-start-loading')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + // did-start-loading writes loading:true first; the loss must be the last word. + expect(onUpdatePageState.mock.calls.at(-1)).toEqual(['page-a', { loading: false }]) + }) + + it('ignores queued load events after guest loss', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + act(() => webview.dispatchEvent(new Event('destroyed'))) + onUpdatePageState.mockClear() + + act(() => webview.dispatchEvent(new Event('did-start-loading'))) + + expect(onUpdatePageState).not.toHaveBeenCalled() + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + }) + + it('uses the guarded title snapshot if the guest dies immediately afterward', () => { + render(paneElement(true)) + webview.getTitle = vi.fn(() => { + webview.getTitle = vi.fn(() => { + throw invalidGuestInstanceId() + }) + return 'Last live title' + }) + + expect(() => act(() => webview.dispatchEvent(new Event('did-navigate')))).not.toThrow() + expect(webview.getTitle).not.toHaveBeenCalled() + }) + + it('shows unavailability when a load-failure fallback finds the guest gone', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => webview.dispatchEvent(new Event('did-fail-load'))) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(onUpdatePageState.mock.calls.at(-1)).toEqual(['page-a', { loading: false }]) + }) + + it('removes loss listeners when the initial guest read fails', () => { + const removeListener = vi.spyOn(webview, 'removeEventListener') + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + render(paneElement(true)) + + expect(removeListener).toHaveBeenCalledWith('destroyed', expect.any(Function)) + expect(removeListener).toHaveBeenCalledWith('render-process-gone', expect.any(Function)) + }) + + it('stops listening for guest loss once the pane lets go of the tag', () => { + const onUpdatePageState = vi.fn() + const view = render(paneElement(true, { onUpdatePageState })) + view.unmount() + onUpdatePageState.mockClear() + mocks.recordBreadcrumb.mockClear() + + webview.dispatchEvent(new Event('destroyed')) + + expect(onUpdatePageState).not.toHaveBeenCalled() + expect(mocks.recordBreadcrumb).not.toHaveBeenCalled() + }) + + it('survives activation focus after the retained tag left the DOM', () => { + const view = render(paneElement(false)) + webview.focus = vi.fn(() => { + throw nullContentWindowFocus() + }) + + expect(() => + act(() => { + view.rerender(paneElement(true)) + }) + ).not.toThrow() + expect(webview.focus).toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx index f4f698f516b..349886a570f 100644 --- a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx +++ b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx @@ -7,7 +7,10 @@ import type { } from '../../../../shared/browser-workspace-types' import { toHttpsRecoveryUrl } from '../../../../shared/browser-url' import type { RuntimeBrowserClientPlacement } from '../../../../shared/runtime-browser-placement' -import { readBrowserClientPageGuestMetadata } from './browser-client-page-guest-metadata' +import { + readBrowserClientPageGuestMetadataIfLive, + createBrowserClientPageLoadFailureHandler +} from './browser-client-page-guest-metadata' import { forgetBrowserClientPageMetadataReports, startBrowserClientPageMetadataPublisher @@ -18,6 +21,7 @@ import { useBrowserClientHostedPopupNotices } from './browser-client-hosted-popu import { useBrowserClientHostedPermissionNotices } from './browser-client-hosted-permission-notices' import { useClientHostedBrowserIntroTour } from './use-client-hosted-browser-intro-tour' import { ClientHostedBrowserUnavailableNotice } from './client-hosted-browser-unavailable-notice' +import { watchBrowserClientPageGuestLoss } from './host-guest/browser-client-page-guest-loss' import { useRestoredClientHostedRecoveryWindow } from './restored-client-hosted-recovery-window' import BrowserFind from './assemble-chrome/BrowserFind' import { BrowserNavigationControlRow } from './assemble-chrome/browser-navigation-control-row' @@ -36,7 +40,6 @@ import { BrowserLoadFailureOverlay } from './navigate/browser-load-failure-overl import { useClientHostedPageUrlSubmission } from './navigate/use-client-hosted-page-url-submission' import { convertBrowserPageToWorkspaceDoc } from '@/lib/file-preview' import { useBrowserPageReloadActions } from './navigate/use-browser-page-reload-actions' -import { resolveBrowserWebviewLoadFailure } from './navigate/browser-webview-load-failure' import { resolveActiveBrowserLoadFailure } from './navigate/browser-load-failure-for-url' import { consumeBrowserPageDeferredNavigation } from './navigate/browser-page-deferred-navigation' import { @@ -46,7 +49,6 @@ import { } from './describe-page/browser-page-url-display' import type { BrowserChromeShortcutScope, - BrowserPageFailLoadEvent, BrowserPageUrlSetter, BrowserTabPageState } from './describe-page/browser-page-types' @@ -174,9 +176,7 @@ export function ClientHostedBrowserPagePane({ useLayoutEffect(() => { const viewport = viewportRef.current - // Why: no placement means the host has not minted this page yet. Attaching would throw for an - // id the retained registry has never seen and strand the pane on the unavailable notice, whose - // only exit is reopening on the server — so mount quiet and wait for adoption to supply it. + // Wait for host adoption before attaching an optimistic page the registry has not seen. if ( !viewport || pageHostGeneration === null || @@ -200,6 +200,24 @@ export function ClientHostedBrowserPagePane({ return } const webview = attachment.webview + // Guest loss uses the existing recovery notice and clears pending loading state. + let releaseGuest = (): void => attachment.detach() + const guestLoss = watchBrowserClientPageGuestLoss({ + webview, + webviewRef, + browserPageId: browserTab.id, + pageHostGeneration, + onLost: () => { + releaseGuest() + retryGuestRecoveryRef.current() + } + }) + // Main can destroy the guest while its tag still holds the stale id. + const attachedMetadata = readBrowserClientPageGuestMetadataIfLive(webview) + if (!attachedMetadata) { + guestLoss.lose('unreadable') + return guestLoss.dispose() + } const publisher = startBrowserClientPageMetadataPublisher({ browserPageId: browserTab.id, environmentId: runtimeEnvironmentId, @@ -213,23 +231,21 @@ export function ClientHostedBrowserPagePane({ }) webviewRef.current = webview setAttachmentError(null) - // Why: the failure carried in from the store is hearsay — this pane may be remounting over a - // guest that navigated on while nothing was listening — so it is checked once against where - // the guest actually is. Failures this session observes are trusted as they arrive, because a - // navigation that fails outright often never commits and leaves the guest on the old URL. + // Reconcile restored failures once; failed navigations this session may never commit a URL. activeLoadFailureRef.current = resolveActiveBrowserLoadFailure( activeLoadFailureRef.current, - readBrowserClientPageGuestMetadata(webview).url + attachedMetadata.url ) const syncNavigation = (event?: Event): void => { const eventUrl = (event as (Event & { url?: string }) | undefined)?.url - const metadata = readBrowserClientPageGuestMetadata(webview, eventUrl) - // Why: did-stop-loading fires after did-fail-load, so an unconditional null here would - // wipe the failure the overlay is about to show. + const metadata = readBrowserClientPageGuestMetadataIfLive(webview, eventUrl) + if (!metadata) { + guestLoss.lose('unreadable') + return + } + // did-stop-loading must preserve the preceding did-fail-load overlay. const activeLoadFailure = activeLoadFailureRef.current - // Why: a URL write drops the page's certificate challenge by design (challenges are - // transient across navigation), so a standing failure must not run through one — the - // local pane returns before its own setUrl for the same reason. + // URL writes clear certificate challenges, so preserve them while a failure stands. if (!activeLoadFailure) { setUrlFromGuest(browserTab.id, metadata.url, { preserveLoadError: true @@ -243,26 +259,41 @@ export function ClientHostedBrowserPagePane({ loadError: activeLoadFailure }) publisher.publish(metadata) - // Why: the address bar's suggestions read the client's shared URL history, so a page - // hosted here has to file its navigations there like a local guest does. - recordHistoryFromGuest(metadata.url, getBrowserDisplayTitle(webview.getTitle(), metadata.url)) + // Address-bar suggestions use the client's URL history, including client-hosted pages. + recordHistoryFromGuest(metadata.url, getBrowserDisplayTitle(metadata.title, metadata.url)) setAddressBarValueFromPage(toDisplayUrl(metadata.url)) } const onStart = (): void => { activeLoadFailureRef.current = null updatePageStateFromGuest(browserTab.id, { loading: true, loadError: null }) - publisher.publish(readBrowserClientPageGuestMetadata(webview, undefined, true)) - } - const onFailLoad = (event: Event): void => { - const loadError = resolveBrowserWebviewLoadFailure(event as BrowserPageFailLoadEvent, { - fallbackUrl: webview.getURL() - }) - if (!loadError) { + const startMetadata = readBrowserClientPageGuestMetadataIfLive(webview, undefined, true) + if (!startMetadata) { + guestLoss.lose('unreadable') return } - activeLoadFailureRef.current = loadError - updatePageStateFromGuest(browserTab.id, { loading: false, loadError }) + publisher.publish(startMetadata) } + const onFailLoad = createBrowserClientPageLoadFailureHandler( + webview, + () => guestLoss.lose('unreadable'), + (loadError) => { + activeLoadFailureRef.current = loadError + updatePageStateFromGuest(browserTab.id, { loading: false, loadError }) + } + ) + const cleanupGuest = (): void => { + webview.removeEventListener('did-start-loading', onStart) + webview.removeEventListener('did-stop-loading', syncNavigation) + webview.removeEventListener('did-navigate', syncNavigation) + webview.removeEventListener('did-navigate-in-page', syncNavigation) + webview.removeEventListener('page-title-updated', syncNavigation) + webview.removeEventListener('did-fail-load', onFailLoad) + guestLoss.dispose() + publisher.dispose() + forgetBrowserClientPageMetadataReports(browserTab.id) + attachment.detach() + } + releaseGuest = cleanupGuest webview.addEventListener('did-start-loading', onStart) webview.addEventListener('did-stop-loading', syncNavigation) webview.addEventListener('did-navigate', syncNavigation) @@ -270,26 +301,12 @@ export function ClientHostedBrowserPagePane({ webview.addEventListener('page-title-updated', syncNavigation) webview.addEventListener('did-fail-load', onFailLoad) syncNavigation() - // Why: the user pressed Enter while this page was still an optimistic stage, so the navigation - // was parked rather than sent to a host page that did not exist yet. The guest exists now. + // Resume navigation submitted before host adoption. const deferredUrl = consumeBrowserPageDeferredNavigation(browserTab.id) if (deferredUrl) { runDeferredNavigation(deferredUrl) } - return () => { - webview.removeEventListener('did-start-loading', onStart) - webview.removeEventListener('did-stop-loading', syncNavigation) - webview.removeEventListener('did-navigate', syncNavigation) - webview.removeEventListener('did-navigate-in-page', syncNavigation) - webview.removeEventListener('page-title-updated', syncNavigation) - webview.removeEventListener('did-fail-load', onFailLoad) - if (webviewRef.current === webview) { - webviewRef.current = null - } - publisher.dispose() - forgetBrowserClientPageMetadataReports(browserTab.id) - attachment.detach() - } + return cleanupGuest }, [ browserTab.id, browserHostClientId, @@ -299,7 +316,7 @@ export function ClientHostedBrowserPagePane({ setAddressBarValueFromPage ]) - useClientHostedGuestActivationFocus({ isActive, webviewRef, keepAddressBarFocusRef }) + useClientHostedGuestActivationFocus({ isActive, guestFocus, keepAddressBarFocusRef }) const showFailureOverlay = !attachmentError && Boolean(browserTab.loadError) // Why: the failure is about the URL that failed, not whatever page is still loaded — feeding diff --git a/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts b/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts index 5e4446f978a..93e62de4dd7 100644 --- a/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts +++ b/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts @@ -1,24 +1,67 @@ +import type { BrowserLoadError } from '../../../../shared/browser-workspace-types' +import type { BrowserPageFailLoadEvent } from './describe-page/browser-page-types' +import { resolveBrowserWebviewLoadFailure } from './navigate/browser-webview-load-failure' +import { recordRendererCrashBreadcrumb } from '@/lib/crash-breadcrumb-recorder' import { redactKagiSessionToken } from '../../../../shared/browser-url' import type { BrowserClientPageMetadataSnapshot } from './browser-client-page-metadata-publisher' /** - * What a client-hosted guest currently is, read straight off the webview. + * What a client-hosted guest currently is, read straight off the webview, or null once the tag + * can no longer reach its guest. * * `eventUrl` wins when a navigation event carries one: the tag's own getURL() can still report the * previous page while the event is being delivered. `loading` is forced for did-start-loading, * which fires before isLoading() flips. + * + * Why total rather than throwing: a guest destroyed in main leaves the tag holding its id, so + * every method on it throws `Invalid guestInstanceId` from then on — and every caller reads from + * a React effect, where that unwinds the whole workbench error boundary. */ -export function readBrowserClientPageGuestMetadata( +export function readBrowserClientPageGuestMetadataIfLive( webview: Electron.WebviewTag, eventUrl?: string, loading?: boolean -): BrowserClientPageMetadataSnapshot { - const url = redactKagiSessionToken(eventUrl || webview.getURL() || 'about:blank') - return { - url, - title: webview.getTitle() || url || 'Browser', - loading: loading ?? webview.isLoading(), - canGoBack: webview.canGoBack(), - canGoForward: webview.canGoForward() +): BrowserClientPageMetadataSnapshot | null { + try { + const url = redactKagiSessionToken(eventUrl || webview.getURL() || 'about:blank') + return { + url, + title: webview.getTitle() || url || 'Browser', + loading: loading ?? webview.isLoading(), + canGoBack: webview.canGoBack(), + canGoForward: webview.canGoForward() + } + } catch (error) { + // Why recorded: the catch is total, so a read failure that is NOT guest death would otherwise + // be indistinguishable from one — the breadcrumb carries the error text the console cannot. + console.warn('[browser-client-page] guest read failed, treating the page as gone:', error) + recordRendererCrashBreadcrumb('browser_client_page_guest_read_failed', { + errorName: error instanceof Error ? error.name : typeof error, + errorMessage: error instanceof Error ? error.message : String(error) + }) + return null + } +} + +export function createBrowserClientPageLoadFailureHandler( + webview: Electron.WebviewTag, + onUnavailable: () => void, + onFailure: (error: BrowserLoadError) => void +): (event: Event) => void { + return (event) => { + let guestUnavailable = false + const loadError = resolveBrowserWebviewLoadFailure(event as BrowserPageFailLoadEvent, { + // Discarded ERR_ABORTED/subframe events must not read the guest. + fallbackUrl: () => { + const metadata = readBrowserClientPageGuestMetadataIfLive(webview) + guestUnavailable = metadata === null + return metadata?.url ?? null + } + }) + if (guestUnavailable) { + onUnavailable() + } else if (loadError) { + onFailure(loadError) + } } } diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts b/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts new file mode 100644 index 00000000000..b9e2f6a21f6 --- /dev/null +++ b/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts @@ -0,0 +1,54 @@ +import type { MutableRefObject } from 'react' +import { recordRendererCrashBreadcrumb } from '@/lib/crash-breadcrumb-recorder' + +export type BrowserClientPageGuestLossReason = 'unreadable' | 'destroyed' | 'render-process-gone' + +/** + * Tells a client-hosted pane, once, that its guest is gone. The retained registry fences the tag on + * `destroyed` / `render-process-gone` without telling the pane, which would otherwise sit mute or + * spinning over a tag whose every method throws; a failed guest read is the same verdict. + */ +export function watchBrowserClientPageGuestLoss(options: { + webview: Electron.WebviewTag + /** Released on loss and dispose: every chrome action null-checks it, so a dead tag is never driven. */ + webviewRef: MutableRefObject + browserPageId: string + pageHostGeneration: number + onLost: () => void +}): { lose(reason: BrowserClientPageGuestLossReason): void; dispose(): void } { + const { webview } = options + const releaseWebviewRef = (): void => { + if (options.webviewRef.current === webview) { + options.webviewRef.current = null + } + } + let lost = false + const lose = (reason: BrowserClientPageGuestLossReason): void => { + if (lost) { + return + } + lost = true + // Why the breadcrumb: the crash report this replaces was the only field signal for guest death. + recordRendererCrashBreadcrumb('browser_client_page_guest_unavailable', { + browserPageId: options.browserPageId, + pageHostGeneration: options.pageHostGeneration, + reason, + tagConnected: webview.isConnected + }) + releaseWebviewRef() + options.onLost() + } + const onDestroyed = (): void => lose('destroyed') + const onRendererGone = (): void => lose('render-process-gone') + webview.addEventListener('destroyed', onDestroyed) + webview.addEventListener('render-process-gone', onRendererGone) + return { + lose, + dispose: () => { + lost = true + releaseWebviewRef() + webview.removeEventListener('destroyed', onDestroyed) + webview.removeEventListener('render-process-gone', onRendererGone) + } + } +} diff --git a/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts b/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts index 95494fb043e..ac9fcf87d8b 100644 --- a/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts +++ b/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts @@ -1,4 +1,5 @@ import { useEffect, useRef, type RefObject } from 'react' +import type { BrowserPageGuestFocus } from '../assemble-chrome/browser-page-guest-focus' import { useWebviewDragPassthroughActive } from './use-webview-drag-passthrough-active' /** @@ -11,11 +12,12 @@ import { useWebviewDragPassthroughActive } from './use-webview-drag-passthrough- */ export function useClientHostedGuestActivationFocus({ isActive, - webviewRef, + guestFocus, keepAddressBarFocusRef }: { isActive: boolean - webviewRef: RefObject + /** Not the raw tag: a retired page's is out of the DOM, where focus() throws (STA-3448). */ + guestFocus: BrowserPageGuestFocus keepAddressBarFocusRef: RefObject }): void { const dragPassthroughActive = useWebviewDragPassthroughActive() @@ -41,6 +43,6 @@ export function useClientHostedGuestActivationFocus({ if (keepAddressBarFocusRef.current) { return } - webviewRef.current?.focus() - }, [dragPassthroughActive, isActive, keepAddressBarFocusRef, webviewRef]) + guestFocus.focus() + }, [dragPassthroughActive, guestFocus, isActive, keepAddressBarFocusRef]) } diff --git a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts index 44e59fdbedd..1595b69f6aa 100644 --- a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts +++ b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { resolveBrowserWebviewLoadFailure } from './browser-webview-load-failure' describe('resolveBrowserWebviewLoadFailure', () => { @@ -49,6 +49,18 @@ describe('resolveBrowserWebviewLoadFailure', () => { ).toMatchObject({ validatedUrl: 'https://example.com/current' }) }) + it('never reads a lazy fallback URL for an event it discards', () => { + const fallbackUrl = vi.fn(() => 'https://example.com/current') + expect(resolveBrowserWebviewLoadFailure({ errorCode: -3 }, { fallbackUrl })).toBeNull() + expect(fallbackUrl).not.toHaveBeenCalled() + expect( + resolveBrowserWebviewLoadFailure( + { errorCode: -105, errorDescription: 'ERR_NAME_NOT_RESOLVED', validatedURL: '' }, + { fallbackUrl } + ) + ).toMatchObject({ validatedUrl: 'https://example.com/current' }) + }) + it('keeps a usable description when Chromium reports an empty one', () => { expect( resolveBrowserWebviewLoadFailure({ errorCode: -105, errorDescription: '' }) diff --git a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts index c39c5afe53b..03228c3e1d6 100644 --- a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts +++ b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts @@ -8,20 +8,23 @@ import type { BrowserPageFailLoadEvent } from '../describe-page/browser-page-typ * cannot forget the ignore rules or build a differently-shaped BrowserLoadError. * * `fallbackUrl` covers failures that arrive without a validatedURL — pass the webview's - * current URL so the overlay names the page instead of about:blank. + * current URL so the overlay names the page instead of about:blank. Pass it as a function when + * reading it costs anything: discarded events never ask for it. */ export function resolveBrowserWebviewLoadFailure( event: BrowserPageFailLoadEvent, - options: { fallbackUrl?: string | null } = {} + options: { fallbackUrl?: string | null | (() => string | null) } = {} ): BrowserLoadError | null { // Why: Chromium reports redirect/cancel races as ERR_ABORTED (-3) even when the // replacement navigation succeeds; subframe failures never blank the page. if (event.isMainFrame === false || event.errorCode === -3) { return null } + const fallbackUrl = + typeof options.fallbackUrl === 'function' ? options.fallbackUrl() : options.fallbackUrl return { code: event.errorCode ?? -1, description: event.errorDescription || 'Unknown load failure', - validatedUrl: redactKagiSessionToken(event.validatedURL || options.fallbackUrl || 'about:blank') + validatedUrl: redactKagiSessionToken(event.validatedURL || fallbackUrl || 'about:blank') } } diff --git a/src/renderer/src/components/cmd-j/native-chat-split-quick-actions.ts b/src/renderer/src/components/cmd-j/native-chat-split-quick-actions.ts new file mode 100644 index 00000000000..d2019fb429a --- /dev/null +++ b/src/renderer/src/components/cmd-j/native-chat-split-quick-actions.ts @@ -0,0 +1,61 @@ +import { PanelBottomClose, PanelRightClose } from 'lucide-react' +import { translate } from '@/i18n/i18n' +import type { CmdJQuickAction } from './quick-actions' +import type { CmdJQuickActionContext } from './quick-action-context' +import type { NativeChatSplitDirection } from '@/components/native-chat/native-chat-split-shortcut' + +function availability(ctx: CmdJQuickActionContext) { + return ctx.canSplitActiveChat + ? ({ available: true } as const) + : ({ available: false, reason: 'no-active-chat' } as const) +} + +function splitAction( + direction: NativeChatSplitDirection, + action: Pick +): CmdJQuickAction { + return { + ...action, + kind: 'action', + isAvailable: availability, + run: async (ctx) => { + if (!availability(ctx).available || !ctx.splitActiveChat?.(direction)) { + return { status: 'unavailable', reason: 'no-active-chat' } + } + return { status: 'ok' } + } + } +} + +export function getNativeChatSplitQuickActions(): CmdJQuickAction[] { + return [ + splitAction('right', { + id: 'split-chat-right', + title: translate('auto.components.cmd.j.quick.actions.splitChatRight', 'Split Chat Right'), + description: translate( + 'auto.components.cmd.j.quick.actions.splitChatRightDescription', + 'Open the active chat in a split pane to the right.' + ), + icon: PanelRightClose, + verbKeywords: [ + translate('auto.components.cmd.j.quick.actions.verbs.splitChatRight', 'split chat right'), + translate('auto.components.cmd.j.quick.actions.verbs.moveChatRight', 'move chat right'), + translate('auto.components.cmd.j.quick.actions.verbs.chatPaneRight', 'chat pane right') + ] + }), + splitAction('down', { + id: 'split-chat-down', + title: translate('auto.components.cmd.j.quick.actions.splitChatDown', 'Split Chat Down'), + description: translate( + 'auto.components.cmd.j.quick.actions.splitChatDownDescription', + 'Open the active chat in a split pane below.' + ), + icon: PanelBottomClose, + verbKeywords: [ + translate('auto.components.cmd.j.quick.actions.verbs.splitChatDown', 'split chat down'), + translate('auto.components.cmd.j.quick.actions.verbs.moveChatDown', 'move chat down'), + translate('auto.components.cmd.j.quick.actions.verbs.chatPaneBelow', 'chat pane below') + ] + }) + ] +} diff --git a/src/renderer/src/components/cmd-j/quick-action-context.test.ts b/src/renderer/src/components/cmd-j/quick-action-context.test.ts index 325252c4dc2..843634309df 100644 --- a/src/renderer/src/components/cmd-j/quick-action-context.test.ts +++ b/src/renderer/src/components/cmd-j/quick-action-context.test.ts @@ -254,6 +254,62 @@ describe('Cmd+J quick action context', () => { expect(context.isLoading).toBe(true) }) + it('exposes chat split actions only on the workspace surface', () => { + const worktree = { + id: 'wt-1', + repoId: 'repo-1', + path: '/repo/wt', + displayName: 'Workspace', + branch: 'main', + createdAt: 0 + } as Worktree + const state = { + activeWorktreeId: 'wt-1', + worktreesByRepo: { 'repo-1': [worktree] }, + repos: [{ id: 'repo-1', path: '/repo', displayName: 'Repo', addedAt: 0 }], + sshConnectionStates: new Map(), + activeGroupIdByWorktree: { 'wt-1': 'group-1' }, + groupsByWorktree: { + 'wt-1': [ + { + id: 'group-1', + worktreeId: 'wt-1', + activeTabId: 'chat-1', + tabOrder: ['chat-1', 'other-1'] + } + ] + }, + unifiedTabsByWorktree: { + 'wt-1': [ + { + id: 'chat-1', + entityId: 'session-1', + groupId: 'group-1', + worktreeId: 'wt-1', + contentType: 'agent-session' + } + ] + }, + activeView: 'settings', + settings: null + } as unknown as AppState + + const buildContext = (activeView: AppState['activeView']) => + buildCmdJQuickActionContext({ + state: { ...state, activeView }, + activeGroupSnapshot: null, + openNewBrowserTab: async () => {}, + openNewMarkdownFile: async () => {}, + openNewTerminalTab: async () => {}, + openCreateWorkspace: () => {}, + deleteActiveWorkspace: () => {}, + openAddQuickCommand: () => {} + }) + + expect(buildContext('terminal').canSplitActiveChat).toBe(true) + expect(buildContext('settings').canSplitActiveChat).toBe(false) + }) + it('runtime re-check returns unavailable without invoking the action helper', async () => { const calls: string[] = [] const action = getCmdJQuickActions().find((entry) => entry.id === 'new-terminal-tab') @@ -335,4 +391,33 @@ describe('Cmd+J quick action context', () => { await expect(action?.run(context)).resolves.toEqual({ status: 'ok' }) expect(calls).toEqual(['delete']) }) + + it('offers and runs split actions only for an active movable chat', async () => { + const calls: string[] = [] + const action = getCmdJQuickActions().find((entry) => entry.id === 'split-chat-right') + const context = { + ...ctx({}), + activeWorktree: null, + runtimeMode: 'local-desktop' as const, + openNewBrowserTab: async () => {}, + openNewMarkdownFile: async () => {}, + openNewTerminalTab: async () => {}, + openCreateWorkspace: () => {}, + deleteActiveWorkspace: () => {}, + openAddQuickCommand: () => {}, + canSplitActiveChat: true, + splitActiveChat: (direction: string) => { + calls.push(direction) + return true + } + } satisfies CmdJQuickActionContext + + expect(action?.isAvailable(context)).toEqual({ available: true }) + await expect(action?.run(context)).resolves.toEqual({ status: 'ok' }) + expect(calls).toEqual(['right']) + expect(action?.isAvailable({ ...context, canSplitActiveChat: false })).toEqual({ + available: false, + reason: 'no-active-chat' + }) + }) }) diff --git a/src/renderer/src/components/cmd-j/quick-action-context.ts b/src/renderer/src/components/cmd-j/quick-action-context.ts index 88a8cfad533..bbed9671a4e 100644 --- a/src/renderer/src/components/cmd-j/quick-action-context.ts +++ b/src/renderer/src/components/cmd-j/quick-action-context.ts @@ -3,12 +3,19 @@ import { findWorktreeById } from '@/store/slices/worktree-helpers' import type { Worktree } from '../../../../shared/worktree/types' import type { SshConnectionStatus } from '../../../../shared/ssh-types' import { getClientCreationActionPolicy } from '@/lib/client-creation-action-policy' +import { + canRunNativeChatSplitTarget, + resolveActiveNativeChatSplitTarget, + runActiveNativeChatSplit +} from '@/components/native-chat/native-chat-layout-actions' +import type { NativeChatSplitDirection } from '@/components/native-chat/native-chat-split-shortcut' export type CmdJUnavailableReason = | 'loading' | 'no-active-workspace' | 'ssh-disconnected' | 'no-active-group' + | 'no-active-chat' | 'client-action-unsupported' export type CmdJQuickActionAvailability = @@ -35,6 +42,8 @@ export type CmdJQuickActionContext = { openCreateWorkspace: () => void deleteActiveWorkspace: () => void openAddQuickCommand: () => void + canSplitActiveChat?: boolean + splitActiveChat?: (direction: NativeChatSplitDirection) => boolean } export function resolveCmdJActiveGroupId( @@ -166,6 +175,10 @@ export function buildCmdJQuickActionContext(args: { const managedBrowserCreationEnabled = getClientCreationActionPolicy(args.state, activeWorktreeId)['managed-browser'].state === 'enabled' + const activeChatTarget = + args.state.activeView === 'terminal' + ? resolveActiveNativeChatSplitTarget(args.state, activeWorktreeId, activeGroupId) + : null return { activeView: args.state.activeView, @@ -181,7 +194,10 @@ export function buildCmdJQuickActionContext(args: { openNewTerminalTab: args.openNewTerminalTab, openCreateWorkspace: args.openCreateWorkspace, deleteActiveWorkspace: args.deleteActiveWorkspace, - openAddQuickCommand: args.openAddQuickCommand + openAddQuickCommand: args.openAddQuickCommand, + canSplitActiveChat: canRunNativeChatSplitTarget(args.state, activeChatTarget), + splitActiveChat: (direction) => + runActiveNativeChatSplit(activeWorktreeId, activeGroupId, direction) } } @@ -198,6 +214,8 @@ export function getUnavailableQuickActionMessage( return `Can't ${actionTitle.toLowerCase()} — workspace is disconnected.` case 'no-active-group': return `Can't ${actionTitle.toLowerCase()} — no tab group is available.` + case 'no-active-chat': + return `Can't ${actionTitle.toLowerCase()} — no movable chat is active.` case 'client-action-unsupported': return `Can't ${actionTitle.toLowerCase()} — this client and runtime do not support it.` } diff --git a/src/renderer/src/components/cmd-j/quick-actions.ts b/src/renderer/src/components/cmd-j/quick-actions.ts index 70f0e5f6859..00ccf127bad 100644 --- a/src/renderer/src/components/cmd-j/quick-actions.ts +++ b/src/renderer/src/components/cmd-j/quick-actions.ts @@ -8,6 +8,7 @@ import { } from './quick-action-context' import { translate } from '@/i18n/i18n' import { createLocalizedCatalog } from '@/i18n/localized-catalog' +import { getNativeChatSplitQuickActions } from './native-chat-split-quick-actions' export type CmdJQuickActionRunResult = | { status: 'ok' } @@ -125,6 +126,7 @@ export const getCmdJQuickActions = createLocalizedCatalog((): CmdJQuickAction[] isAvailable: workspaceActionAvailability, run: (ctx) => runWorkspaceAction(ctx, ctx.openNewTerminalTab) }, + ...getNativeChatSplitQuickActions(), { id: CREATE_WORKSPACE_QUICK_ACTION_ID, kind: 'action', diff --git a/src/renderer/src/components/editor/details-markdown-html.ts b/src/renderer/src/components/editor/details-markdown-html.ts index 109005460dc..964c08f35d6 100644 --- a/src/renderer/src/components/editor/details-markdown-html.ts +++ b/src/renderer/src/components/editor/details-markdown-html.ts @@ -94,7 +94,7 @@ export function renderDetailsAttributes(attrs: Record | undefin function markdownFenceRanges(content: string): MarkdownFenceRanges { const ranges: [number, number][] = [] let offset = 0 - let openFence: { marker: '`' | '~'; length: number; start: number } | null = null + let openFence: { closingPattern: RegExp; start: number } | null = null for (const lineMatch of content.matchAll(/[^\r\n]*(?:\r\n|\n|\r|$)/g)) { const line = lineMatch[0] @@ -104,11 +104,8 @@ function markdownFenceRanges(content: string): MarkdownFenceRanges { const lineText = line.replace(/(?:\r\n|\n|\r)$/u, '') if (openFence) { - const closingFencePattern = - openFence.marker === '`' - ? new RegExp(`^ {0,3}\`{${openFence.length},}\\s*$`) - : new RegExp(`^ {0,3}~{${openFence.length},}\\s*$`) - if (closingFencePattern.test(lineText)) { + // Built once per fence: rebuilding it per line recompiled the same regex for every fenced line. + if (openFence.closingPattern.test(lineText)) { ranges.push([openFence.start, offset + line.length]) openFence = null } @@ -116,8 +113,9 @@ function markdownFenceRanges(content: string): MarkdownFenceRanges { const openingFenceMatch = lineText.match(/^ {0,3}(`{3,}|~{3,})/u) if (openingFenceMatch?.[1]) { openFence = { - marker: openingFenceMatch[1][0] as '`' | '~', - length: openingFenceMatch[1].length, + closingPattern: new RegExp( + `^ {0,3}${openingFenceMatch[1][0]}{${openingFenceMatch[1].length},}\\s*$` + ), start: offset } } diff --git a/src/renderer/src/components/mobile/MobilePage.test.tsx b/src/renderer/src/components/mobile/MobilePage.test.tsx index 2ef1cf5e965..2149cadc284 100644 --- a/src/renderer/src/components/mobile/MobilePage.test.tsx +++ b/src/renderer/src/components/mobile/MobilePage.test.tsx @@ -18,6 +18,7 @@ type StoreState = { mobilePairingCustomAddresses?: string[] } updateSettings: () => Promise + fetchOrcaProfileAuthStatus: () => Promise } const mocks = vi.hoisted(() => ({ @@ -146,7 +147,8 @@ describe('MobilePage pairing connection mode', () => { closeMobilePage: vi.fn(), orcaProfileAuthStatus: { state: 'connected' }, settings: { showMobileButton: true }, - updateSettings: vi.fn().mockResolvedValue(undefined) + updateSettings: vi.fn().mockResolvedValue(undefined), + fetchOrcaProfileAuthStatus: vi.fn().mockResolvedValue(null) } Object.defineProperty(window, 'api', { configurable: true, diff --git a/src/renderer/src/components/mobile/MobilePage.tsx b/src/renderer/src/components/mobile/MobilePage.tsx index ccb9a9e093f..434f0298626 100644 --- a/src/renderer/src/components/mobile/MobilePage.tsx +++ b/src/renderer/src/components/mobile/MobilePage.tsx @@ -39,6 +39,7 @@ export default function MobilePage(): React.JSX.Element { const [relayMintFailure, setRelayMintFailure] = useState(null) const [pairLoading, setPairLoading] = useState(false) const signedIn = useAppStore((state) => state.orcaProfileAuthStatus?.state === 'connected') + const refreshAuthStatus = useAppStore((state) => state.fetchOrcaProfileAuthStatus) const [connectionMode, setConnectionMode] = useMobilePairingConnectionMode() const [networkInterfaces, setNetworkInterfaces] = useState([]) const pairingAddressChangeRef = useRef<(change: MobilePairingAddressChange) => void>(() => {}) @@ -93,7 +94,8 @@ export default function MobilePage(): React.JSX.Element { setPairingUrl, setPairingQrError, setPairLoading, - setRelayMintFailure + setRelayMintFailure, + refreshAuthStatus }) useLayoutEffect(() => { pairingAddressChangeRef.current = ({ address, source }) => { diff --git a/src/renderer/src/components/mobile/mobile-platform-copy.ts b/src/renderer/src/components/mobile/mobile-platform-copy.ts index fd70176609d..cd6669891a3 100644 --- a/src/renderer/src/components/mobile/mobile-platform-copy.ts +++ b/src/renderer/src/components/mobile/mobile-platform-copy.ts @@ -22,7 +22,7 @@ const IOS_CHANNEL_COPY: Record = { const ANDROID_COPY: InstallCopy = { ctaLabel: 'Download APK', - url: 'https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk' + url: 'https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk' } export function getInstallCopy(platform: Platform, iosChannel: IosChannel): InstallCopy { diff --git a/src/renderer/src/components/mobile/mobile-relay-mint-failure-notice.tsx b/src/renderer/src/components/mobile/mobile-relay-mint-failure-notice.tsx index d5e929152fb..ee3d82a51d1 100644 --- a/src/renderer/src/components/mobile/mobile-relay-mint-failure-notice.tsx +++ b/src/renderer/src/components/mobile/mobile-relay-mint-failure-notice.tsx @@ -3,6 +3,7 @@ import { CircleAlert, Loader2 } from 'lucide-react' import { Button } from '../ui/button' import { translate } from '@/i18n/i18n' import type { MobileRelayMintFailure } from '../../../../shared/mobile-relay-mint-failure' +import { useAppStore } from '@/store' import { cn } from '@/lib/utils' export function MobileRelayMintFailureNotice({ @@ -23,6 +24,10 @@ export function MobileRelayMintFailureNotice({ busy?: boolean }): React.JSX.Element { const providerMissing = failure.stage === 'provider_missing' + // Why: a revoked cloud session fails every mint; "retry or use LAN" hides the one action that works. + const reconnectRequired = useAppStore( + (state) => state.orcaProfileAuthStatus?.state === 'reconnect-required' + ) const [showBusyFeedback, setShowBusyFeedback] = useState(false) useEffect(() => { if (!busy) { @@ -43,10 +48,15 @@ export function MobileRelayMintFailureNotice({ 'auto.components.mobile.MobileRelayMintFailureNotice.unavailableTitle', 'Orca Relay isn’t available on this desktop.' ) - : translate( - 'auto.components.mobile.MobileRelayMintFailureNotice.title', - 'Couldn’t create a Relay pairing code.' - ) + : reconnectRequired + ? translate( + 'auto.components.mobile.MobileRelayMintFailureNotice.reconnectTitle', + 'Your Orca account session expired.' + ) + : translate( + 'auto.components.mobile.MobileRelayMintFailureNotice.title', + 'Couldn’t create a Relay pairing code.' + ) const body = visibleBusy ? translate( 'auto.components.mobile.MobileRelayMintFailureNotice.retryingBody', @@ -57,10 +67,15 @@ export function MobileRelayMintFailureNotice({ 'auto.components.mobile.MobileRelayMintFailureNotice.unavailableBody', 'Use LAN to pair over Tailscale or the same Wi‑Fi.' ) - : translate( - 'auto.components.mobile.MobileRelayMintFailureNotice.body', - 'Retry, or use LAN to pair over Tailscale or the same Wi‑Fi.' - ) + : reconnectRequired + ? translate( + 'auto.components.mobile.MobileRelayMintFailureNotice.reconnectBody', + 'Sign in again to use Orca Relay, or use LAN to pair over Tailscale or the same Wi‑Fi.' + ) + : translate( + 'auto.components.mobile.MobileRelayMintFailureNotice.body', + 'Retry, or use LAN to pair over Tailscale or the same Wi‑Fi.' + ) return (
{translate('auto.components.mobile.MobileRelayMintFailureNotice.useLan', 'Use LAN')} - {!providerMissing ? ( + {!providerMissing && !reconnectRequired ? (
) } diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx index 41a70a8457d..965ed0ad176 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx @@ -99,7 +99,9 @@ describe('NativeChatToolRun', () => { const { container } = render() - expect(screen.getByText('Running cat package.json')).toBeInTheDocument() + const activeLabel = screen.getByText('Running cat package.json') + expect(activeLabel).toBeInTheDocument() + expect(activeLabel).toHaveClass('animate-pulse', 'motion-reduce:animate-none') expect(screen.queryByText('Running date')).toBeNull() expect(screen.queryByText('Running pwd')).toBeNull() expect(screen.queryByText('Ran 3 commands and used 1 tool')).toBeNull() diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx index 716d293838e..f16d01331f8 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx @@ -227,7 +227,7 @@ export function NativeChatToolRun({ - + {activeToolLabel(latestActiveCall)} {open ? : null} diff --git a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx index a883429f658..21145e94c94 100644 --- a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx @@ -3,6 +3,21 @@ import { ChevronRight } from 'lucide-react' import { translate } from '@/i18n/i18n' import { useNow } from '@/hooks/use-now' +/** Format turn time without exposing an ever-growing raw seconds count. */ +export function formatNativeChatDuration(seconds: number): string { + const totalSeconds = Number.isFinite(seconds) ? Math.max(0, Math.floor(seconds)) : 0 + if (totalSeconds < 60) { + return `${totalSeconds}s` + } + const minutes = Math.floor(totalSeconds / 60) + const remainingSeconds = totalSeconds % 60 + if (minutes < 60) { + return `${minutes}m ${remainingSeconds}s` + } + const hours = Math.floor(minutes / 60) + return `${hours}h ${minutes % 60}m ${remainingSeconds}s` +} + export function NativeChatWorkingStatus({ startedAt, thinking, @@ -30,13 +45,13 @@ export function NativeChatWorkingStatus({ const label = workedSeconds != null - ? translate('components.native-chat.status.workedFor', 'Worked for {{value0}} seconds', { - value0: workedSeconds + ? translate('components.native-chat.status.workedFor', 'Worked for {{value0}}', { + value0: formatNativeChatDuration(workedSeconds) }) : thinking ? translate('components.native-chat.status.thinking', 'Thinking') - : translate('components.native-chat.status.workingFor', 'Working for {{value0}} seconds', { - value0: elapsedSeconds + : translate('components.native-chat.status.workingFor', 'Working for {{value0}}', { + value0: formatNativeChatDuration(elapsedSeconds) }) const className = `flex min-h-8 items-center gap-1 text-sm text-muted-foreground${thinking ? '' : ' border-b border-border'}` diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.test.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.test.tsx new file mode 100644 index 00000000000..4a185a3d630 --- /dev/null +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.test.tsx @@ -0,0 +1,58 @@ +// @vitest-environment happy-dom + +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionHandoffStatus } from '../../../../shared/agent-session-wire' +import { StructuredAgentSessionHandoffChrome } from './StructuredAgentSessionHandoffChrome' + +const IDLE_NATIVE: AgentSessionHandoffStatus = { + owner: 'native', + direction: null, + phase: 'idle', + stage: null, + operationId: null +} + +afterEach(cleanup) + +describe('StructuredAgentSessionHandoffChrome', () => { + it('uses queued-safe admission when the native view still appears idle', () => { + const onRequest = vi.fn() + render( + + ) + + fireEvent.click(screen.getByRole('button', { name: 'Open agent TUI' })) + + expect(onRequest).toHaveBeenCalledWith('to-tui', 'after-turn') + }) + + it('offers one Retry action for a recoverable dead TUI owner', () => { + const onRequest = vi.fn() + render( + + ) + + expect(screen.queryByRole('button', { name: 'Return to chat' })).toBeNull() + fireEvent.click(screen.getByRole('button', { name: 'Retry' })) + expect(onRequest).toHaveBeenCalledWith('to-native', 'now', 'retry') + }) +}) diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.tsx new file mode 100644 index 00000000000..840039564d8 --- /dev/null +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.tsx @@ -0,0 +1,225 @@ +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffMode, + AgentSessionHandoffStatus +} from '../../../../shared/agent-session-wire' +import { Button } from '@/components/ui/button' +import { Badge } from '@/components/ui/badge' +import { translate } from '@/i18n/i18n' + +type Props = { + status: AgentSessionHandoffStatus | null + isWorking: boolean + onRequest: ( + direction: AgentSessionHandoffDirection, + mode: AgentSessionHandoffMode, + action?: 'start' | 'cancel-queued' | 'retry' | 'recover' + ) => void +} + +function handoffStageCopy(status: AgentSessionHandoffStatus): string { + if (status.stage === 'preparing') { + return status.direction === 'to-tui' + ? translate('components.native-chat.handoff.stage.finishingChat', 'Finishing chat session…') + : translate( + 'components.native-chat.handoff.stage.finishingTerminal', + 'Finishing agent terminal…' + ) + } + if (status.stage === 'old-owner-stopped') { + return status.direction === 'to-tui' + ? translate('components.native-chat.handoff.stage.openingTerminal', 'Opening agent terminal…') + : translate('components.native-chat.handoff.stage.resumingChat', 'Resuming chat session…') + } + if (status.stage === 'new-owner-proving') { + return status.direction === 'to-tui' + ? translate( + 'components.native-chat.handoff.stage.verifyingTerminal', + 'Verifying agent terminal…' + ) + : translate('components.native-chat.handoff.stage.verifyingChat', 'Verifying chat session…') + } + if (status.stage === 'recovering') { + return translate('components.native-chat.handoff.stage.recovering', 'Recovering agent session…') + } + if (status.stage === 'manual-recovery') { + return translate( + 'components.native-chat.handoff.stage.manualRecovery', + 'Agent session needs recovery' + ) + } + return translate('components.native-chat.handoff.switchingOwner', 'Switching session owner…') +} + +export function StructuredAgentSessionHandoffChrome({ + status, + isWorking, + onRequest +}: Props): React.JSX.Element | null { + if (!status) { + return null + } + const owner = status?.owner ?? 'native' + const phase = status?.phase ?? 'idle' + const switching = phase === 'switching' || phase === 'waiting-for-exit' + return ( + <> +
+ + {switching + ? translate('components.native-chat.handoff.mode.switching', 'Switching') + : owner === 'tui' + ? translate('components.native-chat.handoff.mode.terminal', 'Terminal') + : translate('components.native-chat.handoff.mode.chat', 'Chat')} + +
+ {phase === 'queued' && status?.direction ? ( + <> + + {status.direction === 'to-tui' + ? translate( + 'components.native-chat.handoff.switchingAfterTurn', + 'Switching after this turn' + ) + : translate( + 'components.native-chat.handoff.returningAfterTurn', + 'Returning after this turn' + )} + + + + ) : owner === 'native' && phase === 'idle' ? ( + isWorking ? ( + <> + + + + ) : ( + + ) + ) : owner === 'tui' && phase === 'idle' ? ( + + ) : null} +
+
+ {owner === 'tui' && phase === 'idle' ? ( +
+ + {status?.hostLabel + ? translate( + 'components.native-chat.handoff.agentOpenOnHost', + 'Agent is open in terminal on {{value0}}.', + { value0: status.hostLabel } + ) + : translate('components.native-chat.handoff.agentOpen', 'Agent is open in terminal.')} + + +
+ ) : null} + {switching ? ( +
+ {phase === 'waiting-for-exit' + ? translate( + 'components.native-chat.handoff.exitTerminal', + 'Exit the agent terminal to continue in chat.' + ) + : status?.stage + ? handoffStageCopy(status) + : translate( + 'components.native-chat.handoff.switchingOwner', + 'Switching session owner…' + )} +
+ ) : null} + {phase === 'failed' && status?.error ? ( +
+
+ {status.error.message} + {status.direction && status.error.canRetryProof ? ( + + ) : status.direction && status.error.recoverableOwner !== 'none' ? ( + + ) : null} +
+ {status.error.details ? ( +
+ {translate('components.native-chat.handoff.details', 'Details')} +

{status.error.details}

+
+ ) : null} +
+ ) : null} + + ) +} diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx index 25991eac3ff..3cf25030aa1 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.test.tsx @@ -16,7 +16,8 @@ const mocks = vi.hoisted(() => ({ store: null as null | { setState: (state: Partial) => void }, focusGroup: vi.fn(), mountsByTabId: new Map(), - unmountsByTabId: new Map() + unmountsByTabId: new Map(), + groupIdByTabId: new Map() })) vi.mock('@/store', async () => { @@ -53,11 +54,14 @@ vi.mock('./NativeChatView', async () => { return { default: function MockNativeChatView({ tabId, + groupId, isVisible }: { tabId: string + groupId?: string isVisible: boolean }) { + mocks.groupIdByTabId.set(tabId, groupId) useEffect(() => { mocks.mountsByTabId.set(tabId, (mocks.mountsByTabId.get(tabId) ?? 0) + 1) return () => { @@ -87,6 +91,7 @@ describe('StructuredAgentSessionPaneOverlayLayer', () => { mocks.focusGroup.mockClear() mocks.mountsByTabId.clear() mocks.unmountsByTabId.clear() + mocks.groupIdByTabId.clear() mocks.store?.setState(createState(FIRST_TAB_ID)) }) @@ -125,6 +130,12 @@ describe('StructuredAgentSessionPaneOverlayLayer', () => { expect(mocks.mountsByTabId.get(FIRST_TAB_ID)).toBe(1) expect(mocks.mountsByTabId.get(SECOND_TAB_ID)).toBe(1) expect(mocks.unmountsByTabId.size).toBe(0) + expect(mocks.groupIdByTabId).toEqual( + new Map([ + [FIRST_TAB_ID, GROUP_ID], + [SECOND_TAB_ID, GROUP_ID] + ]) + ) }) it('routes overlay interaction back to the owning split group', () => { diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx index 555d23b9ec4..cf6a0e0b4d1 100644 --- a/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionPaneOverlayLayer.tsx @@ -69,6 +69,7 @@ const StructuredAgentSessionOverlaySlot = memo(function StructuredAgentSessionOv , tabOrder = ['chat', 'other']) { + return { + groupsByWorktree: { + workspace: [{ id: 'group', worktreeId: 'workspace', activeTabId: 'chat', tabOrder }] + }, + unifiedTabsByWorktree: { + workspace: [ + { + id: 'chat', + entityId: 'session', + groupId: 'group', + worktreeId: 'workspace', + ...tab + } + ] + } + } as unknown as Pick +} + +describe('native chat layout actions', () => { + it('resolves structured chats to the reusable workspace-tab move path', () => { + const state = stateWithActiveTab({ contentType: 'agent-session' }) + const target = resolveActiveNativeChatSplitTarget(state, 'workspace', 'group') + + expect(target).toEqual({ kind: 'workspace-tab', unifiedTabId: 'chat', groupId: 'group' }) + expect(canRunNativeChatSplitTarget(state, target)).toBe(true) + expect( + canRunNativeChatSplitTarget( + stateWithActiveTab({ contentType: 'agent-session' }, ['chat']), + target + ) + ).toBe(false) + }) + + it('resolves terminal-backed chat mode to the existing pane split path', () => { + const state = stateWithActiveTab({ contentType: 'terminal', viewMode: 'chat' }) + + expect(resolveActiveNativeChatSplitTarget(state, 'workspace', 'group')).toEqual({ + kind: 'terminal-pane', + terminalTabId: 'session' + }) + expect( + resolveActiveNativeChatSplitTarget( + stateWithActiveTab({ contentType: 'terminal', viewMode: 'terminal' }), + 'workspace', + 'group' + ) + ).toBeNull() + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-layout-actions.ts b/src/renderer/src/components/native-chat/native-chat-layout-actions.ts new file mode 100644 index 00000000000..4f24a8a663c --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-layout-actions.ts @@ -0,0 +1,74 @@ +import type { AppState } from '@/store/types' +import { useAppStore } from '@/store' +import { + canMoveTabToNewPaneColumnFromState, + moveTabToNewPaneColumn +} from '@/components/tab-bar/tab-move-to-pane-column' +import { requestActiveTerminalPaneSplit } from '@/components/tab-bar/request-active-terminal-pane-split' +import type { NativeChatSplitDirection } from './native-chat-split-shortcut' + +export type NativeChatSplitTarget = + | { kind: 'terminal-pane'; terminalTabId: string } + | { kind: 'workspace-tab'; unifiedTabId: string; groupId: string } + +export function resolveActiveNativeChatSplitTarget( + state: Pick, + worktreeId: string | null, + groupId: string | null +): NativeChatSplitTarget | null { + if (!worktreeId || !groupId) { + return null + } + const group = (state.groupsByWorktree?.[worktreeId] ?? []).find((entry) => entry.id === groupId) + const tab = (state.unifiedTabsByWorktree?.[worktreeId] ?? []).find( + (entry) => entry.id === group?.activeTabId && entry.groupId === groupId + ) + if (tab?.contentType === 'agent-session') { + return { kind: 'workspace-tab', unifiedTabId: tab.id, groupId } + } + if (tab?.contentType === 'terminal' && tab.viewMode === 'chat') { + return { kind: 'terminal-pane', terminalTabId: tab.entityId } + } + return null +} + +export function canRunNativeChatSplitTarget( + state: Pick, + target: NativeChatSplitTarget | null +): boolean { + if (!target) { + return false + } + return ( + target.kind === 'terminal-pane' || + canMoveTabToNewPaneColumnFromState(state, target.unifiedTabId, target.groupId) + ) +} + +export function runNativeChatSplitTarget( + target: NativeChatSplitTarget, + direction: NativeChatSplitDirection +): boolean { + if (target.kind === 'terminal-pane') { + requestActiveTerminalPaneSplit({ + tabId: target.terminalTabId, + direction: direction === 'right' ? 'vertical' : 'horizontal' + }) + return true + } + return moveTabToNewPaneColumn({ + unifiedTabId: target.unifiedTabId, + groupId: target.groupId, + direction + }) +} + +export function runActiveNativeChatSplit( + worktreeId: string | null, + groupId: string | null, + direction: NativeChatSplitDirection +): boolean { + const state = useAppStore.getState() + const target = resolveActiveNativeChatSplitTarget(state, worktreeId, groupId) + return target ? runNativeChatSplitTarget(target, direction) : false +} diff --git a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts index 9dae22da197..8cf82b69ca1 100644 --- a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts @@ -117,6 +117,7 @@ describe('native chat PTY session options', () => { expect(effortResult.snapshot.map(({ id }) => id)).toEqual(['model', 'effort', 'fastMode']) expect(effortResult.snapshot.find(({ id }) => id === 'effort')).toMatchObject({ valueSource: 'dispatched', + transport: 'catalog', kind: { currentValue: 'high' } }) expect(listener).toHaveBeenCalledOnce() diff --git a/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts b/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts index aa7562b5944..3e3587e68d1 100644 --- a/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts +++ b/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts @@ -98,7 +98,8 @@ export function createNativeChatPtySessionOptions( catalog, models: activeModels(), record, - mode: args.mode + mode: args.mode, + liveTransport: 'catalog' }) const listeners = new Set<(value: SessionOptionDescriptor[]) => void>() @@ -108,7 +109,8 @@ export function createNativeChatPtySessionOptions( catalog, models: activeModels(), record, - mode: args.mode + mode: args.mode, + liveTransport: 'catalog' }) for (const listener of listeners) { listener(snapshot) diff --git a/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts b/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts index 60d0ffe6aba..7cdf621b272 100644 --- a/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts @@ -18,6 +18,7 @@ function modelDescriptor( id: 'model', label: 'Model', valueSource, + transport: 'catalog', settable: true, kind: { type: 'select', diff --git a/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts b/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts index 9010aa66f27..4acb1a1932a 100644 --- a/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts +++ b/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts @@ -7,6 +7,7 @@ import { buildNativeChatSessionOptionSnapshot as buildSharedSnapshot, resolveEffectiveNativeChatModelId, withTrackedNativeChatModel, + type NativeChatLiveOptionTransport, type NativeChatSessionOptionMode } from '../../../../shared/native-chat-session-option-snapshot' import { @@ -15,7 +16,7 @@ import { } from '../../../../shared/native-chat-session-option-state' import { translate } from '@/i18n/i18n' -export type { NativeChatSessionOptionMode } +export type { NativeChatLiveOptionTransport, NativeChatSessionOptionMode } export { flattenNativeChatSessionOptionRecord, resolveEffectiveNativeChatModelId, @@ -27,6 +28,7 @@ export function buildNativeChatSessionOptionSnapshot(args: { models: readonly CatalogModel[] record: NativeChatSessionOptionRecord mode: NativeChatSessionOptionMode + liveTransport: NativeChatLiveOptionTransport }): SessionOptionDescriptor[] { return buildSharedSnapshot({ ...args, diff --git a/src/renderer/src/components/native-chat/native-chat-split-shortcut.test.ts b/src/renderer/src/components/native-chat/native-chat-split-shortcut.test.ts new file mode 100644 index 00000000000..0a5b2bd6688 --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-split-shortcut.test.ts @@ -0,0 +1,38 @@ +import { describe, expect, it } from 'vitest' +import { matchNativeChatSplitShortcut } from './native-chat-split-shortcut' + +describe('matchNativeChatSplitShortcut', () => { + it('uses the existing platform split bindings', () => { + expect( + matchNativeChatSplitShortcut( + { key: 'd', metaKey: true, ctrlKey: false, altKey: false, shiftKey: false }, + 'darwin', + {} + ) + ).toBe('right') + expect( + matchNativeChatSplitShortcut( + { key: 'd', metaKey: false, ctrlKey: true, altKey: false, shiftKey: true }, + 'win32', + {} + ) + ).toBe('right') + expect( + matchNativeChatSplitShortcut( + { key: 'd', metaKey: false, ctrlKey: false, altKey: true, shiftKey: true }, + 'linux', + {} + ) + ).toBe('down') + }) + + it('respects customized bindings', () => { + expect( + matchNativeChatSplitShortcut( + { key: 'ArrowRight', metaKey: false, ctrlKey: true, altKey: true, shiftKey: false }, + 'linux', + { 'terminal.splitRight': ['Ctrl+Alt+Right'] } + ) + ).toBe('right') + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-split-shortcut.ts b/src/renderer/src/components/native-chat/native-chat-split-shortcut.ts new file mode 100644 index 00000000000..5ded10fa6bb --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-split-shortcut.ts @@ -0,0 +1,21 @@ +import { + keybindingMatchesAction, + type KeybindingInput, + type KeybindingOverrides +} from '../../../../shared/keybindings' + +export type NativeChatSplitDirection = 'right' | 'down' + +export function matchNativeChatSplitShortcut( + input: KeybindingInput, + platform: NodeJS.Platform, + keybindings: KeybindingOverrides +): NativeChatSplitDirection | null { + if (keybindingMatchesAction('terminal.splitRight', input, platform, keybindings)) { + return 'right' + } + if (keybindingMatchesAction('terminal.splitDown', input, platform, keybindings)) { + return 'down' + } + return null +} diff --git a/src/renderer/src/components/native-chat/native-chat-stop-layering.test.ts b/src/renderer/src/components/native-chat/native-chat-stop-layering.test.ts index 716f1dc3182..46bee5d6172 100644 --- a/src/renderer/src/components/native-chat/native-chat-stop-layering.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-stop-layering.test.ts @@ -6,6 +6,15 @@ function source(path: string): string { return readFileSync(join(process.cwd(), path), 'utf8') } +function workingChatZIndex(css: string): number { + const match = + /\.native-chat-pane-shell:has\(\[data-native-chat-working='true'\]\)[^{]*\{[^}]*z-index:\s*(\d+);/s.exec( + css + ) + expect(match, 'working native-chat z-index rule not found in main.css').not.toBeNull() + return Number(match?.[1]) +} + describe('native chat Stop layering', () => { it('keeps a working chat pane above bottom-right product chrome', () => { const css = source('src/renderer/src/assets/main.css') @@ -15,9 +24,24 @@ describe('native chat Stop layering', () => { expect(terminalPane).toContain('native-chat-pane-shell absolute inset-0 z-10') expect(css).toMatch(/\[data-sonner-toaster\][^{]*\{[^}]*z-index:\s*40\s*!important;/s) - expect(css).toMatch( - /\.native-chat-pane-shell:has\(\[data-native-chat-working='true'\]\)[^{]*\{[^}]*z-index:\s*50;/s + expect(workingChatZIndex(css)).toBeGreaterThan(40) + }) + + // Why both bounds: raising the working pane over the panel hides a summoned + // floating workspace behind the chat column while an agent streams. + it('stays under the floating workspace panel while working', () => { + // Comments stripped first: the surrounding layering comment cites bare z-40/z-50 + // tiers, and a reworded one could otherwise be read as the panel's own class. + const panel = source( + 'src/renderer/src/components/floating-terminal/FloatingTerminalPanelSurface.tsx' + ).replace(/\/\*[\s\S]*?\*\/|\/\/[^\n]*/g, '') + const panelZIndex = Number( + /data-floating-terminal-panel[\s\S]*?className=[\s\S]*?z-\[(\d+)\]/.exec(panel)?.[1] ) + + // FloatingTerminalPanel.bounds.test.tsx pins this same 45 through a real render. + expect(panelZIndex).toBe(45) + expect(workingChatZIndex(source('src/renderer/src/assets/main.css'))).toBeLessThan(panelZIndex) }) it('publishes working state from both structured and bridge chat roots', () => { diff --git a/src/renderer/src/components/native-chat/native-chat-view-types.ts b/src/renderer/src/components/native-chat/native-chat-view-types.ts index 106b29b429d..de86ecbe228 100644 --- a/src/renderer/src/components/native-chat/native-chat-view-types.ts +++ b/src/renderer/src/components/native-chat/native-chat-view-types.ts @@ -37,11 +37,13 @@ export type NativeChatBridgeViewProps = NativeChatOrchestrationProps & { export type NativeChatStructuredViewProps = NativeChatOrchestrationProps & { mode: 'structured' tabId: string + groupId?: string sessionId: string target: RuntimeClientTarget agent: AgentType isVisible: boolean allowFileUriLinks: boolean + contextMenuActions?: Omit } export type NativeChatResolvedViewProps = NativeChatOrchestrationProps & { diff --git a/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx index dfea16ae0e4..e4bae292004 100644 --- a/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx +++ b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx @@ -3,7 +3,7 @@ import { act } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' +import { formatNativeChatDuration, NativeChatWorkingStatus } from './NativeChatWorkingStatus' let container: HTMLDivElement let root: Root @@ -49,34 +49,50 @@ afterEach(() => { }) describe('native chat working status elapsed clock', () => { + it.each([ + [0, '0s'], + [59, '59s'], + [60, '1m 0s'], + [69, '1m 9s'], + [3_725, '1h 2m 5s'] + ])('formats %s seconds as %s', (seconds, expected) => { + expect(formatNativeChatDuration(seconds)).toBe(expected) + }) + + it('renders the compact duration in the completed status label', () => { + act(() => { + root.render( + + ) + }) + + expect(elapsedLabels()).toEqual(['Worked for 1m 9s']) + }) + it('collapses every in-flight turn onto one shared visibility-gated timer', () => { renderTurns(3, 1_000_000) // One shared 1s clock for all three turns, not one interval per turn. expect(vi.getTimerCount()).toBe(1) act(() => vi.advanceTimersByTime(3_000)) - expect(elapsedLabels()).toEqual([ - 'Working for 3 seconds', - 'Working for 3 seconds', - 'Working for 3 seconds' - ]) + expect(elapsedLabels()).toEqual(['Working for 3s', 'Working for 3s', 'Working for 3s']) }) it('stops ticking while hidden and re-syncs the elapsed value on return', () => { renderTurns(1, 1_000_000) act(() => vi.advanceTimersByTime(3_000)) - expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + expect(elapsedLabels()).toEqual(['Working for 3s']) setDocumentVisibility('hidden') expect(vi.getTimerCount()).toBe(0) // A minute of hidden wall-clock: no callbacks, no commits, label frozen. act(() => vi.advanceTimersByTime(60_000)) - expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + expect(elapsedLabels()).toEqual(['Working for 3s']) // Returning re-derives elapsed from startedAt, so nothing was lost. setDocumentVisibility('visible') - expect(elapsedLabels()).toEqual(['Working for 63 seconds']) + expect(elapsedLabels()).toEqual(['Working for 1m 3s']) expect(vi.getTimerCount()).toBe(1) }) diff --git a/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx b/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx index f4e8b3efc72..14841e6c56d 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx @@ -3,7 +3,8 @@ */ import React, { createRef, type ReactNode } from 'react' import { renderToStaticMarkup } from 'react-dom/server' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { cleanup, render } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { emptyNativeChatContextMenuActions, useNativeChatContextMenu, @@ -52,6 +53,10 @@ vi.mock('@/i18n/i18n', () => ({ translate: (_key: string, fallback: string) => fallback })) +vi.mock('@/components/tab-bar/TabWorkspaceLayoutMenuSection', () => ({ + TabWorkspaceLayoutMenuSection: () => 'Move Tab to Split' +})) + function childrenText(children: ReactNode): string { return React.Children.toArray(children) .map((child) => { @@ -65,11 +70,22 @@ function childrenText(children: ReactNode): string { .join('') } -function Harness({ onSwitchToTerminal }: { onSwitchToTerminal?: () => void }) { +function Harness({ + onSwitchToTerminal, + structured = false, + enabled = true +}: { + onSwitchToTerminal?: () => void + structured?: boolean + enabled?: boolean +}) { const rootRef = createRef() const { menu } = useNativeChatContextMenu({ rootRef, + enabled, onSwitchToTerminal, + showTerminalPaneActions: !structured, + workspaceLayout: structured ? { unifiedTabId: 'chat-tab', groupId: 'group-1' } : undefined, actions: { ...emptyNativeChatContextMenuActions, onPaste: vi.fn() @@ -83,6 +99,11 @@ describe('useNativeChatContextMenu', () => { items.list = [] }) + afterEach(() => { + cleanup() + vi.restoreAllMocks() + }) + it('restores the bridge switch-to-terminal action when supplied', () => { const onSwitchToTerminal = vi.fn() @@ -107,4 +128,31 @@ describe('useNativeChatContextMenu', () => { items.list.some((candidate) => childrenText(candidate.children) === 'Switch to terminal view') ).toBe(false) }) + + it('reuses workspace layout actions without terminal-only pane commands', () => { + const markup = renderToStaticMarkup() + + expect(markup).toContain('Move Tab to Split') + expect(markup).not.toContain('Split Terminal Right') + expect(markup).not.toContain('Fork Agent Session') + }) + + it('subscribes to selection changes only while its retained chat is visible', () => { + const getSelection = vi.spyOn(window, 'getSelection').mockReturnValue(null) + const view = render() + + getSelection.mockClear() + document.dispatchEvent(new Event('selectionchange')) + expect(getSelection).not.toHaveBeenCalled() + + view.rerender() + getSelection.mockClear() + document.dispatchEvent(new Event('selectionchange')) + expect(getSelection).toHaveBeenCalledOnce() + + view.rerender() + getSelection.mockClear() + document.dispatchEvent(new Event('selectionchange')) + expect(getSelection).not.toHaveBeenCalled() + }) }) diff --git a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx index a2d7dae4c93..48d896d4aea 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx @@ -30,6 +30,8 @@ import { } from '@/components/ui/dropdown-menu' import { translate } from '@/i18n/i18n' import { isMacPlatform, nativeChatToggleShortcutLabel } from './native-chat-shortcut' +import { TabWorkspaceLayoutMenuSection } from '@/components/tab-bar/TabWorkspaceLayoutMenuSection' +import type { TabSplitDirection } from '@/store/slices/tabs' type NativeChatContextMenuState = { open: boolean @@ -39,9 +41,17 @@ type NativeChatContextMenuState = { type UseNativeChatContextMenuArgs = { rootRef: RefObject - /** Bridge-only escape hatch; structured sessions never mount this menu. */ + enabled?: boolean + /** Bridge-only escape hatch; structured sessions have no terminal view. */ onSwitchToTerminal?: () => void actions: NativeChatContextMenuActions + showTerminalPaneActions?: boolean + splitShortcutLabels?: { right: string; down: string } + workspaceLayout?: { + unifiedTabId: string + groupId: string + shortcutLabels?: Partial> + } } export type NativeChatContextMenuActions = { @@ -88,8 +98,12 @@ export const emptyNativeChatContextMenuActions: Omit onSelectionCapture: () => void @@ -112,9 +126,18 @@ export function useNativeChatContextMenu({ }, [rootRef]) useEffect(() => { + if (!enabled) { + return + } document.addEventListener('selectionchange', rememberCurrentSelection) return () => document.removeEventListener('selectionchange', rememberCurrentSelection) - }, [rememberCurrentSelection]) + }, [enabled, rememberCurrentSelection]) + + useEffect(() => { + if (!enabled) { + setState((current) => (current.open ? { ...current, open: false } : current)) + } + }, [enabled]) const onContextMenuCapture = useCallback( (event: React.MouseEvent) => { @@ -142,7 +165,7 @@ export function useNativeChatContextMenu({ onContextMenuCapture, onSelectionCapture: rememberCurrentSelection, menu: ( - + + + + + + ) +} + +let container: HTMLDivElement +let root: Root + +beforeEach(() => { + // Radix Presence retains closing content when the production exit animation runs. + const getStyle = window.getComputedStyle.bind(window) + vi.spyOn(window, 'getComputedStyle').mockImplementation((element, ...args) => { + const style = getStyle(element, ...args) + if (element.getAttribute('data-slot') === 'popover-content') { + return new Proxy(style, { + get: (target, property) => + property === 'animationName' + ? element.getAttribute('data-state') === 'closed' + ? 'exit' + : 'enter' + : Reflect.get(target, property) + }) + } + return style + }) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) +}) + +afterEach(async () => { + await act(async () => root.unmount()) + container.remove() + vi.restoreAllMocks() +}) + +function projectField(): HTMLInputElement { + return document.querySelector('[role="combobox"][aria-label="Project"]')! +} + +function addOption(): HTMLElement { + return Array.from(document.querySelectorAll('[role="option"]')).find((option) => + option.textContent?.includes('Add a new project') + )! +} + +describe.each([false, true])( + 'project selector to nested dialog handoff (populated: %s)', + (populated) => { + it.each(['mouse', 'keyboard'] as const)( + 'removes the selector before the creation dialog is active via %s', + async (input) => { + await act(async () => root.render()) + await act(async () => projectField().click()) + expect(document.querySelector('[role="listbox"]')).not.toBeNull() + + await act(async () => { + if (input === 'mouse') { + addOption().dispatchEvent( + new MouseEvent('mousedown', { bubbles: true, cancelable: true }) + ) + addOption().click() + } else if (populated) { + projectField().dispatchEvent( + new KeyboardEvent('keydown', { key: 'ArrowDown', bubbles: true, cancelable: true }) + ) + } + }) + if (input === 'keyboard') { + await act(async () => { + projectField().dispatchEvent( + new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }) + ) + }) + } + + expect(document.querySelector('[role="listbox"]')).toBeNull() + expect(document.querySelector('[data-slot="popover-content"]')).toBeNull() + expect(projectField().getAttribute('aria-expanded')).toBe('false') + expect(projectField().hasAttribute('aria-activedescendant')).toBe(false) + const path = document.querySelector('[aria-label="Project path"]')! + expect(document.activeElement).toBe(path) + expect(path.closest('[aria-hidden="true"]')).toBeNull() + expect(projectField().closest('[aria-hidden="true"]')).not.toBeNull() + + await act(async () => { + Array.from(document.querySelectorAll('button')) + .find((b) => b.textContent === 'Cancel')! + .click() + }) + await vi.waitFor(() => + expect(document.activeElement).toBe( + document.querySelector('[aria-label="Workspace name"]') + ) + ) + await act(async () => projectField().click()) + expect(projectField().getAttribute('aria-expanded')).toBe('true') + expect(document.querySelector('[role="listbox"]')).not.toBeNull() + } + ) + } +) diff --git a/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx b/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx index 3b7931238b3..aa6b8df7b2d 100644 --- a/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx +++ b/src/renderer/src/components/new-workspace/ProjectCombobox.test.tsx @@ -72,7 +72,7 @@ function field(): HTMLInputElement { function openList(): void { act(() => { - field().dispatchEvent(new FocusEvent('focus', { bubbles: true })) + field().focus() }) } diff --git a/src/renderer/src/components/new-workspace/ProjectCombobox.tsx b/src/renderer/src/components/new-workspace/ProjectCombobox.tsx index c305964ca0e..8561465d57e 100644 --- a/src/renderer/src/components/new-workspace/ProjectCombobox.tsx +++ b/src/renderer/src/components/new-workspace/ProjectCombobox.tsx @@ -229,129 +229,132 @@ export default function ProjectCombobox({ - event.preventDefault()} - onCloseAutoFocus={(event) => event.preventDefault()} - // Why: the field lives in the anchor, not inside the content, so Radix - // sees a focus/pointer event "outside" the layer and dismisses it the - // instant you tab in. Keep the layer open whenever the interaction is - // within this control; genuine outside events still close it. - onFocusOutside={(event) => { - if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { - event.preventDefault() - } - }} - onInteractOutside={(event) => { - if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { - event.preventDefault() - } - }} - > - {/* The listbox wraps a scrolling pane plus the pinned Add row, so both - stay `option` children of one listbox. */} -
event.preventDefault()} + onCloseAutoFocus={(event) => event.preventDefault()} + // Why: the field lives in the anchor, not inside the content, so Radix + // sees a focus/pointer event "outside" the layer and dismisses it the + // instant you tab in. Keep the layer open whenever the interaction is + // within this control; genuine outside events still close it. + onFocusOutside={(event) => { + if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { + event.preventDefault() + } + }} + onInteractOutside={(event) => { + if (isWithinComboboxRoot(event.target, ROOT_ATTRIBUTE)) { + event.preventDefault() + } + }} > - {/* Why: `presentation` — this element exists to scroll, and an + {/* The listbox wraps a scrolling pane plus the pinned Add row, so both + stay `option` children of one listbox. */} +
+ {/* Why: `presentation` — this element exists to scroll, and an unroled div between a listbox and its options breaks the ownership relationship assistive tech relies on. */} -
- {matches.length === 0 ? ( - // Why: row-height rather than a tall centred block — a 60px panel - // next to 32px rows reads as a different kind of surface and - // makes an empty result feel like an error. -

- {options.length === 0 - ? translate( - 'auto.components.new.workspace.ProjectCombobox.noProjects', - 'No projects yet.' - ) - : translate( - 'auto.components.new.workspace.ProjectCombobox.empty', - 'No projects match your search.' - )} -

- ) : null} - {sections.map((section) => ( - // Why: `role="group"` — a bare div between a listbox and its - // options breaks the ownership relationship for screen readers, - // which is the only thing that makes the headings announceable. -
- {section.heading ? ( - - ) : null} - {section.items.map((scored) => ( - arm(scored.option.id)} - onCommit={() => commit(scored.option.id)} - /> - ))} -
- ))} -
- {onAddProject ? (
event.preventDefault()} - onMouseMove={() => arm(ADD_PROJECT_KEY)} - onClick={() => commit(ADD_PROJECT_KEY)} - className={cn( - 'flex h-9 shrink-0 cursor-default items-center gap-2 border-t border-border px-2 text-sm', - armedKey === ADD_PROJECT_KEY && 'bg-accent text-accent-foreground' - )} + ref={setListNode} + role="presentation" + className="max-h-72 min-h-0 flex-1 overflow-y-auto p-1 scrollbar-sleek" > - - - {translate( - 'auto.components.new.workspace.ProjectCombobox.addProject', - 'Add a new project' - )} - + {matches.length === 0 ? ( + // Why: row-height rather than a tall centred block — a 60px panel + // next to 32px rows reads as a different kind of surface and + // makes an empty result feel like an error. +

+ {options.length === 0 + ? translate( + 'auto.components.new.workspace.ProjectCombobox.noProjects', + 'No projects yet.' + ) + : translate( + 'auto.components.new.workspace.ProjectCombobox.empty', + 'No projects match your search.' + )} +

+ ) : null} + {sections.map((section) => ( + // Why: `role="group"` — a bare div between a listbox and its + // options breaks the ownership relationship for screen readers, + // which is the only thing that makes the headings announceable. +
+ {section.heading ? ( + + ) : null} + {section.items.map((scored) => ( + arm(scored.option.id)} + onCommit={() => commit(scored.option.id)} + /> + ))} +
+ ))}
- ) : null} -
- + {onAddProject ? ( +
event.preventDefault()} + onMouseMove={() => arm(ADD_PROJECT_KEY)} + onClick={() => commit(ADD_PROJECT_KEY)} + className={cn( + 'flex h-9 shrink-0 cursor-default items-center gap-2 border-t border-border px-2 text-sm', + armedKey === ADD_PROJECT_KEY && 'bg-accent text-accent-foreground' + )} + > + + + {translate( + 'auto.components.new.workspace.ProjectCombobox.addProject', + 'Add a new project' + )} + +
+ ) : null} +
+
+ ) : null} ) } diff --git a/src/renderer/src/components/right-sidebar/active-checks-status.test.ts b/src/renderer/src/components/right-sidebar/active-checks-status.test.ts index 2686937c345..5bf09276c72 100644 --- a/src/renderer/src/components/right-sidebar/active-checks-status.test.ts +++ b/src/renderer/src/components/right-sidebar/active-checks-status.test.ts @@ -1,5 +1,9 @@ -import { describe, expect, it } from 'vitest' -import { getActiveChecksStatus } from './active-checks-status' +import { beforeEach, describe, expect, it } from 'vitest' +import { + ACTIVE_CHECKS_STATUS_INPUT_KEYS, + clearActiveChecksStatusCacheForTests, + getActiveChecksStatus +} from './active-checks-status' import type { AppState } from '../../store/types' import type { PRInfo } from '../../../../shared/github/pull-request-types' @@ -16,6 +20,10 @@ function makePR(status: PRInfo['checksStatus']): PRInfo { } describe('getActiveChecksStatus', () => { + beforeEach(() => { + clearActiveChecksStatusCacheForTests() + }) + it('prefers repo-id scoped status over stale path-scoped status for the active worktree', () => { const state = { activeWorktreeId: 'wt-1', @@ -125,3 +133,67 @@ describe('getActiveChecksStatus', () => { expect(getActiveChecksStatus(state)).toBeNull() }) }) + +describe('getActiveChecksStatus caching', () => { + beforeEach(() => { + clearActiveChecksStatusCacheForTests() + }) + + function makeState(prCache: Record) { + return { + activeWorktreeId: 'wt-1', + repos: [{ id: 'repo-1', path: '/repo' }], + worktreesByRepo: { + 'repo-1': [{ id: 'wt-1', repoId: 'repo-1', branch: 'refs/heads/feature/test' }] + }, + prCache + } as unknown as Pick + } + + it('reuses the cached status when every input reference is unchanged', () => { + const prCache = { 'repo-1::feature/test': { data: makePR('success'), fetchedAt: 2 } } + const state = makeState(prCache) + + expect(getActiveChecksStatus(state)).toBe('success') + // A different state object carrying the same field references must still hit the cache. + expect(getActiveChecksStatus({ ...state })).toBe('success') + }) + + it('recomputes when a keyed input reference changes', () => { + expect( + getActiveChecksStatus( + makeState({ 'repo-1::feature/test': { data: makePR('success'), fetchedAt: 2 } }) + ) + ).toBe('success') + expect( + getActiveChecksStatus( + makeState({ 'repo-1::feature/test': { data: makePR('failure'), fetchedAt: 3 } }) + ) + ).toBe('failure') + }) + + it('reads no store field the cache is not keyed on', () => { + // Guards against a cast or widened type bypassing the derived key list: every property the + // computation touches on the state object must invalidate the cache. + const reads = new Set() + const keyed = new Set(ACTIVE_CHECKS_STATUS_INPUT_KEYS) + const state = new Proxy( + makeState({ 'repo-1::feature/test': { data: makePR('success'), fetchedAt: 2 } }), + { + get(target, prop, receiver) { + reads.add(prop) + return Reflect.get(target, prop, receiver) + }, + has(target, prop) { + reads.add(prop) + return Reflect.has(target, prop) + } + } + ) + + expect(getActiveChecksStatus(state)).toBe('success') + expect([...reads].filter((prop) => !keyed.has(prop))).toEqual([]) + // The read set must also be non-trivial, or the guard proves nothing. + expect(reads.has('prCache')).toBe(true) + }) +}) diff --git a/src/renderer/src/components/right-sidebar/active-checks-status.ts b/src/renderer/src/components/right-sidebar/active-checks-status.ts index d6f29e80658..9f1e08a03ba 100644 --- a/src/renderer/src/components/right-sidebar/active-checks-status.ts +++ b/src/renderer/src/components/right-sidebar/active-checks-status.ts @@ -5,17 +5,69 @@ import { getGitHubPRCacheKey } from '../../store/slices/github-cache-key' import { getHostedReviewCacheKey } from '../../store/slices/hosted-review-cache-identity' import { isGitHubPRSuppressed } from '../../../../shared/worktree/github-pr-suppression' -type ActiveChecksStatusState = Pick< - AppState, - 'activeWorktreeId' | 'worktreesByRepo' | 'repos' | 'prCache' -> & - Partial> +// Why one list: the parameter type and the cache key both derive from it, so a new store field +// cannot be read here (TS rejects it) without also invalidating the cache on it. +const REQUIRED_INPUT_KEYS = [ + 'activeWorktreeId', + 'worktreesByRepo', + 'repos', + 'prCache' +] as const satisfies readonly (keyof AppState)[] +const OPTIONAL_INPUT_KEYS = [ + 'settings', + 'hostedReviewCache' +] as const satisfies readonly (keyof AppState)[] +/** @internal Every store field getActiveChecksStatus may read; the cache invalidates on any of them. */ +export const ACTIVE_CHECKS_STATUS_INPUT_KEYS = [ + ...REQUIRED_INPUT_KEYS, + ...OPTIONAL_INPUT_KEYS +] as const + +type ActiveChecksStatusState = Pick & + Partial> +type ActiveChecksStatusInputs = { + [K in (typeof ACTIVE_CHECKS_STATUS_INPUT_KEYS)[number]]: ActiveChecksStatusState[K] +} function branchDisplayName(branch: string): string { return branch.replace(/^refs\/heads\//, '') } +// Why cached: the right sidebar is always mounted, so this ran on every store write and rebuilt two +// cache-key strings each time. Same single-entry, reference-keyed shape as selectFloatingVisibleTabCount. +let activeChecksStatusCache: { + inputs: ActiveChecksStatusInputs + status: CheckStatus | null +} | null = null + +/** @internal */ +export function clearActiveChecksStatusCacheForTests(): void { + activeChecksStatusCache = null +} + +function hasSameInputs(inputs: ActiveChecksStatusInputs, state: ActiveChecksStatusState): boolean { + for (const key of ACTIVE_CHECKS_STATUS_INPUT_KEYS) { + if (inputs[key] !== state[key]) { + return false + } + } + return true +} + export function getActiveChecksStatus(state: ActiveChecksStatusState): CheckStatus | null { + const cached = activeChecksStatusCache + if (cached && hasSameInputs(cached.inputs, state)) { + return cached.status + } + const status = computeActiveChecksStatus(state) + const inputs = Object.fromEntries( + ACTIVE_CHECKS_STATUS_INPUT_KEYS.map((key) => [key, state[key]]) + ) as ActiveChecksStatusInputs + activeChecksStatusCache = { inputs, status } + return status +} + +function computeActiveChecksStatus(state: ActiveChecksStatusState): CheckStatus | null { const activeWorktree = state.activeWorktreeId ? (getWorktreeMapFromState(state).get(state.activeWorktreeId) ?? null) : null diff --git a/src/renderer/src/components/right-sidebar/source-control/listing/use-selection.ts b/src/renderer/src/components/right-sidebar/source-control/listing/use-selection.ts index 635ec2102f3..c76576d50dd 100644 --- a/src/renderer/src/components/right-sidebar/source-control/listing/use-selection.ts +++ b/src/renderer/src/components/right-sidebar/source-control/listing/use-selection.ts @@ -1,4 +1,4 @@ -import { useState, useCallback, useEffect, useRef, type RefObject } from 'react' +import { useState, useCallback, useEffect, useMemo, useRef, type RefObject } from 'react' import type { GitStatusEntry } from '../../../../../../shared/git-status-types' import type { SourceControlRowOpenEvent } from './split-open' @@ -34,6 +34,10 @@ export function reconcileSourceControlSelectionState(args: { flatEntries: FlatEntry[] }): { selectedKeys: ReadonlySet; anchorKey: string | null } { const { anchorKey, flatEntries, selectedKeys } = args + // Nothing to prune and no anchor to invalidate: skip building the key set over every visible row. + if (selectedKeys.size === 0 && anchorKey === null) { + return { selectedKeys, anchorKey } + } const validKeys = new Set(flatEntries.map((e) => e.key)) const nextSelected = new Set() let selectedChanged = false @@ -111,11 +115,12 @@ export function useSourceControlSelection({ shouldOpenAsSplitRef.current = shouldOpenAsSplit }, [shouldOpenAsSplit]) - const reconciledSelection = reconcileSourceControlSelectionState({ - selectedKeys, - anchorKey, - flatEntries - }) + // Memoized: this hook re-runs on every Source Control panel render (commit-message keystrokes, + // status polls), but the reconciliation only moves when one of these three references does. + const reconciledSelection = useMemo( + () => reconcileSourceControlSelectionState({ selectedKeys, anchorKey, flatEntries }), + [selectedKeys, anchorKey, flatEntries] + ) if (reconciledSelection.selectedKeys !== selectedKeys) { // Why: visible source-control rows can disappear after filtering, staging, // or status refresh; prune stale bulk-action keys before children see them. diff --git a/src/renderer/src/components/right-sidebar/useSourceControlSelection.test.ts b/src/renderer/src/components/right-sidebar/useSourceControlSelection.test.ts index 7821b7f07f5..574d2c5c708 100644 --- a/src/renderer/src/components/right-sidebar/useSourceControlSelection.test.ts +++ b/src/renderer/src/components/right-sidebar/useSourceControlSelection.test.ts @@ -38,6 +38,41 @@ describe('reconcileSelectionKeys', () => { }) describe('reconcileSourceControlSelectionState', () => { + it('returns the same references without scanning rows when nothing is selected', () => { + const selectedKeys: ReadonlySet = new Set() + let keyReads = 0 + const flatEntries = Array.from({ length: 50 }, (_, index) => ({ + get key() { + keyReads += 1 + return `file-${index}.ts` + } + })) as unknown as Parameters[0]['flatEntries'] + + const result = reconcileSourceControlSelectionState({ + selectedKeys, + anchorKey: null, + flatEntries + }) + + expect(keyReads).toBe(0) + expect(result.selectedKeys).toBe(selectedKeys) + expect(result.anchorKey).toBeNull() + }) + + it('still drops an anchor that is no longer visible when nothing is selected', () => { + const selectedKeys: ReadonlySet = new Set() + const result = reconcileSourceControlSelectionState({ + selectedKeys, + anchorKey: 'gone.ts', + flatEntries: [{ key: 'kept.ts' }] as unknown as Parameters< + typeof reconcileSourceControlSelectionState + >[0]['flatEntries'] + }) + + expect(result.anchorKey).toBeNull() + expect(result.selectedKeys).toBe(selectedKeys) + }) + it('keeps selected keys and anchor identity when all keys are still visible', () => { const flatEntries = [ makeEntry('unstaged::a.ts', 'unstaged', 'a.ts'), diff --git a/src/renderer/src/components/settings/ArtifactsSettingsPane.tsx b/src/renderer/src/components/settings/ArtifactsSettingsPane.tsx index 60dd3e0c0c9..7e8b87029d4 100644 --- a/src/renderer/src/components/settings/ArtifactsSettingsPane.tsx +++ b/src/renderer/src/components/settings/ArtifactsSettingsPane.tsx @@ -1,8 +1,8 @@ -import { useEffect } from 'react' import { ArrowRight, Files } from 'lucide-react' import type { GlobalSettings } from '../../../../shared/global-settings-types' import { Button } from '@/components/ui/button' import { SettingsSwitchRow } from './SettingsFormControls' +import { useOrcaProfileAuthStatusRefresh } from '@/hooks/use-orca-profile-auth-status-refresh' import { useAppStore } from '@/store' import { isWebClientLocation } from '@/lib/web-client-location' import { translate } from '@/i18n/i18n' @@ -20,18 +20,13 @@ export function ArtifactsSettingsPane({ const authStatus = useAppStore((state) => state.orcaProfileAuthStatus) const connecting = useAppStore((state) => state.orcaProfileConnecting) const connect = useAppStore((state) => state.connectCurrentOrcaProfile) - const fetchAuthStatus = useAppStore((state) => state.fetchOrcaProfileAuthStatus) const signedIn = authStatus?.state === 'connected' // Why: the capability lives in the desktop host's store and is deliberately absent from the // settings.update allowlist, so a web client can only mirror it — never grant it. const isWebClient = isWebClientLocation() const sharingEnabled = settings.artifactSharingEnabled === true - useEffect(() => { - if (!authStatus) { - void fetchAuthStatus() - } - }, [authStatus, fetchAuthStatus]) + useOrcaProfileAuthStatusRefresh() const howToSteps: HowToStep[] = [ ...(sharingEnabled diff --git a/src/renderer/src/components/settings/DebouncedSettingsTextInput.tsx b/src/renderer/src/components/settings/DebouncedSettingsTextInput.tsx new file mode 100644 index 00000000000..3461133d299 --- /dev/null +++ b/src/renderer/src/components/settings/DebouncedSettingsTextInput.tsx @@ -0,0 +1,39 @@ +import type React from 'react' +import { Input } from '../ui/input' +import { useDebouncedSettingsTextDraft } from './use-debounced-settings-text-draft' + +type DebouncedSettingsTextInputProps = Omit< + React.ComponentProps, + 'value' | 'onChange' | 'onBlur' +> & { + value: string + commit: (next: string) => void + onEdit?: () => void +} + +/** + * Text input for a free-text setting, committed on a debounce instead of per keystroke. + * + * Why a component and not a hook at the call site: the account sections are render functions the + * settings search calls conditionally, so hooks cannot live in them. Rendering this as JSX gives + * the draft its own component to mount and unmount with. + */ +export function DebouncedSettingsTextInput({ + value, + commit, + onEdit, + ...inputProps +}: DebouncedSettingsTextInputProps): React.JSX.Element { + const draft = useDebouncedSettingsTextDraft({ value, commit }) + return ( + { + onEdit?.() + draft.onChange(event.target.value) + }} + onBlur={draft.onBlur} + /> + ) +} diff --git a/src/renderer/src/components/settings/DevToolsPane.tsx b/src/renderer/src/components/settings/DevToolsPane.tsx index 11bfb36d933..5084dbfb4de 100644 --- a/src/renderer/src/components/settings/DevToolsPane.tsx +++ b/src/renderer/src/components/settings/DevToolsPane.tsx @@ -5,6 +5,7 @@ import { Badge } from '../ui/badge' import { SettingsSubsectionHeader } from './SettingsFormControls' import { showDeleteWorktreeFailureToast } from '../sidebar/delete-worktree-failure-toast' import { showLocalBaseRefUpdateSuggestionToast } from '../sidebar/local-base-ref-suggestion-toast' +import { useOrcaProfileAuthStatusRefresh } from '@/hooks/use-orca-profile-auth-status-refresh' import { translate } from '@/i18n/i18n' import { useAppStore } from '@/store' import type { AppState } from '@/store/types' @@ -137,6 +138,7 @@ function OrcaCloudDevSubsection(): React.JSX.Element { const refresh = useAppStore((s) => s.fetchOrcaProfileAuthStatus) const configured = authStatus?.configured === true const connected = authStatus?.state === 'connected' + useOrcaProfileAuthStatusRefresh() return (
diff --git a/src/renderer/src/components/settings/ExperimentalPane.test.tsx b/src/renderer/src/components/settings/ExperimentalPane.test.tsx index b421b77bfc2..ba8e1518921 100644 --- a/src/renderer/src/components/settings/ExperimentalPane.test.tsx +++ b/src/renderer/src/components/settings/ExperimentalPane.test.tsx @@ -258,8 +258,12 @@ describe('ExperimentalPane', () => { }) expect(container.textContent).toContain('Use updated structured native chat') + // The one opt-in gates both providers, so its copy must not name only Codex. expect(container.textContent).toContain( - 'Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.' + 'Opt in to the host-owned structured chat runtime for Codex and Claude.' + ) + expect(container.textContent).toContain( + 'Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.' ) expect(container.textContent).toContain('Default view') root.unmount() diff --git a/src/renderer/src/components/settings/MobilePairingConnectionOptions.test.tsx b/src/renderer/src/components/settings/MobilePairingConnectionOptions.test.tsx index 6ff64efa82a..a08c220038c 100644 --- a/src/renderer/src/components/settings/MobilePairingConnectionOptions.test.tsx +++ b/src/renderer/src/components/settings/MobilePairingConnectionOptions.test.tsx @@ -2,6 +2,7 @@ import '@testing-library/jest-dom/vitest' +import { StrictMode, useSyncExternalStore } from 'react' import { cleanup, render, screen, waitFor, within } from '@testing-library/react' import userEvent from '@testing-library/user-event' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' @@ -17,13 +18,28 @@ type MobileRelayStoreState = { } const mocks = vi.hoisted(() => ({ - state: {} as MobileRelayStoreState + state: {} as MobileRelayStoreState, + listeners: new Set<() => void>() })) +// Why subscribable: a mid-mount auth re-read has to reach the rendered tree, which a +// plain selector-over-a-mutable-object mock silently swallows. vi.mock('../../store', () => ({ - useAppStore: (selector: (state: MobileRelayStoreState) => unknown) => selector(mocks.state) + useAppStore: (selector: (state: MobileRelayStoreState) => unknown) => + useSyncExternalStore( + (onStoreChange) => { + mocks.listeners.add(onStoreChange) + return () => mocks.listeners.delete(onStoreChange) + }, + () => selector(mocks.state) + ) })) +function publishStoreState(next: MobileRelayStoreState): void { + mocks.state = next + mocks.listeners.forEach((listener) => listener()) +} + vi.mock('../../i18n/i18n', () => ({ translate: (_key: string, fallback: string) => fallback })) @@ -251,4 +267,39 @@ describe('MobilePairingConnectionOptions', () => { await user.click(lan) expect(onChange).toHaveBeenCalledWith('local-only') }) + + it('re-reads a session revoked since startup and offers Sign in again', async () => { + // Regression: the store cached "connected" at startup and the pane only + // fetched when it was empty, so a revoked session stayed invisible. + const connectedState: MobileRelayStoreState = { + ...mocks.state, + orcaProfileAuthStatus: { + activeProfileId: 'profile-1', + configured: true, + state: 'connected', + persistence: 'encrypted' + } + } + mocks.state = connectedState + fetchAuthStatus.mockImplementation(async () => { + const revoked: OrcaProfileAuthStatus = { + activeProfileId: 'profile-1', + configured: true, + state: 'reconnect-required', + persistence: 'encrypted' + } + publishStoreState({ ...connectedState, orcaProfileAuthStatus: revoked }) + return revoked + }) + + // StrictMode double-invokes the effect: a fetch keyed on what it writes would loop. + render( + + + + ) + + expect(await screen.findByRole('button', { name: 'Sign in again for Relay' })).toBeVisible() + expect(fetchAuthStatus).toHaveBeenCalledTimes(2) + }) }) diff --git a/src/renderer/src/components/settings/MobilePairingConnectionOptions.tsx b/src/renderer/src/components/settings/MobilePairingConnectionOptions.tsx index 1af58c45827..7874bc5047f 100644 --- a/src/renderer/src/components/settings/MobilePairingConnectionOptions.tsx +++ b/src/renderer/src/components/settings/MobilePairingConnectionOptions.tsx @@ -4,6 +4,7 @@ import { Badge } from '../ui/badge' import { Button } from '../ui/button' import { translate } from '../../i18n/i18n' import { useAppStore } from '../../store' +import { useOrcaProfileAuthStatusRefresh } from '@/hooks/use-orca-profile-auth-status-refresh' import { cn } from '@/lib/utils' import type { MobileRelayStatus } from '../../../../shared/mobile-relay-status' import type { MobilePairingConnectionMode } from '../../../../shared/mobile-pairing-connection-mode' @@ -54,7 +55,6 @@ export function MobilePairingConnectionOptions({ const authStatus = useAppStore((state) => state.orcaProfileAuthStatus) const connecting = useAppStore((state) => state.orcaProfileConnecting) const connect = useAppStore((state) => state.connectCurrentOrcaProfile) - const fetchAuthStatus = useAppStore((state) => state.fetchOrcaProfileAuthStatus) const [relayStatus, setRelayStatus] = useState('offline') const signedIn = authStatus?.state === 'connected' const reconnectRequired = authStatus?.state === 'reconnect-required' @@ -93,11 +93,7 @@ export function MobilePairingConnectionOptions({ optionRefs.current[next]?.focus() } - useEffect(() => { - if (!authStatus) { - void fetchAuthStatus() - } - }, [authStatus, fetchAuthStatus]) + useOrcaProfileAuthStatusRefresh() useEffect(() => { let receivedEvent = false diff --git a/src/renderer/src/components/settings/MobilePane.test.tsx b/src/renderer/src/components/settings/MobilePane.test.tsx index 1695263a740..28f83ccf697 100644 --- a/src/renderer/src/components/settings/MobilePane.test.tsx +++ b/src/renderer/src/components/settings/MobilePane.test.tsx @@ -32,6 +32,7 @@ type StoreState = { } updateSettings: (patch: Record) => Promise recordFeatureInteraction: (feature: string) => void + fetchOrcaProfileAuthStatus: () => Promise } const mocks = vi.hoisted(() => { @@ -188,7 +189,8 @@ describe('MobilePane pairing connection mode', () => { settingsSearchQuery: '', settings: { mobileAutoRestoreFitMs: null }, updateSettings, - recordFeatureInteraction: vi.fn() + recordFeatureInteraction: vi.fn(), + fetchOrcaProfileAuthStatus: vi.fn().mockResolvedValue(null) } Object.defineProperty(window, 'api', { configurable: true, @@ -797,7 +799,8 @@ describe('MobilePane', () => { settingsSearchQuery: '', settings: { mobileAutoRestoreFitMs: null }, updateSettings: mocks.updateSettings, - recordFeatureInteraction: vi.fn() + recordFeatureInteraction: vi.fn(), + fetchOrcaProfileAuthStatus: vi.fn().mockResolvedValue(null) } Object.defineProperty(window, 'api', { configurable: true, diff --git a/src/renderer/src/components/settings/MobilePane.tsx b/src/renderer/src/components/settings/MobilePane.tsx index 39c6df203c0..0e51b677ffc 100644 --- a/src/renderer/src/components/settings/MobilePane.tsx +++ b/src/renderer/src/components/settings/MobilePane.tsx @@ -218,6 +218,9 @@ export function MobilePane(): React.JSX.Element { setEndpoint(null) if (result.reason === 'relay_mint_failed' && result.relayFailure) { setRelayMintFailure(result.relayFailure) + // Why: a revoked session is the likeliest cause; re-read it so the + // notice can offer sign-in instead of a retry that cannot succeed. + void useAppStore.getState().fetchOrcaProfileAuthStatus() } else { setRelayMintFailure(null) // Why: IPC now forwards reason/guidance for all unavailability paths; diff --git a/src/renderer/src/components/settings/MobileSettingsPane.tsx b/src/renderer/src/components/settings/MobileSettingsPane.tsx index 2e98a5ccb43..ff8de2b16e5 100644 --- a/src/renderer/src/components/settings/MobileSettingsPane.tsx +++ b/src/renderer/src/components/settings/MobileSettingsPane.tsx @@ -13,7 +13,7 @@ export { getMobileSettingsPaneSearchEntries } const ORCA_IOS_APP_STORE_URL = 'https://apps.apple.com/app/orca-ide/id6766130217' const ORCA_ANDROID_APK_URL = - 'https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk' + 'https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk' export function MobileSettingsPane(): React.JSX.Element { const showMobileButton = useAppStore((s) => s.settings?.showMobileButton !== false) diff --git a/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx b/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx index 93c4b1c899d..85d27dae2e3 100644 --- a/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx +++ b/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx @@ -126,13 +126,13 @@ export function NativeChatExperimentalSetting({

{translate( 'auto.components.settings.ExperimentalPane.nativeChat.structuredCopy', - 'Opt in to the host-owned structured Codex runtime. Off keeps the existing terminal-backed chat path.' + 'Opt in to the host-owned structured chat runtime for Codex and Claude. Off keeps the existing terminal-backed chat path.' )}

{translate( 'auto.components.settings.ExperimentalPane.nativeChat.structuredScope', - 'Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.' + 'Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.' )}

diff --git a/src/renderer/src/components/settings/OrcaAccountSettingsPane.tsx b/src/renderer/src/components/settings/OrcaAccountSettingsPane.tsx index cf6aec97946..8d7ef62f086 100644 --- a/src/renderer/src/components/settings/OrcaAccountSettingsPane.tsx +++ b/src/renderer/src/components/settings/OrcaAccountSettingsPane.tsx @@ -1,7 +1,8 @@ -import { useEffect, useState } from 'react' +import { useState } from 'react' import { BookOpen, Check, CircleUserRound, Files, Smartphone } from 'lucide-react' import { Badge } from '@/components/ui/badge' import { Button } from '@/components/ui/button' +import { useOrcaProfileAuthStatusRefresh } from '@/hooks/use-orca-profile-auth-status-refresh' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' import { useAppStore } from '@/store' @@ -61,18 +62,13 @@ export function OrcaAccountSettingsPane(): React.JSX.Element { const authStatus = useAppStore((state) => state.orcaProfileAuthStatus) const connecting = useAppStore((state) => state.orcaProfileConnecting) const connect = useAppStore((state) => state.connectCurrentOrcaProfile) - const fetchAuthStatus = useAppStore((state) => state.fetchOrcaProfileAuthStatus) const signOut = useAppStore((state) => state.signOutCurrentOrcaProfile) const [signOutOpen, setSignOutOpen] = useState(false) const [signingOut, setSigningOut] = useState(false) const connected = authStatus?.state === 'connected' const canConnect = authStatus?.configured === true - useEffect(() => { - if (!authStatus) { - void fetchAuthStatus() - } - }, [authStatus, fetchAuthStatus]) + useOrcaProfileAuthStatusRefresh() const confirmSignOut = async (): Promise => { if (signingOut) { diff --git a/src/renderer/src/components/settings/ShareSkillsSettingsPane.tsx b/src/renderer/src/components/settings/ShareSkillsSettingsPane.tsx index 72da1ad4880..ddbf35734c9 100644 --- a/src/renderer/src/components/settings/ShareSkillsSettingsPane.tsx +++ b/src/renderer/src/components/settings/ShareSkillsSettingsPane.tsx @@ -1,6 +1,6 @@ -import { useEffect } from 'react' import { ArrowRight, BookOpen } from 'lucide-react' import { Button } from '@/components/ui/button' +import { useOrcaProfileAuthStatusRefresh } from '@/hooks/use-orca-profile-auth-status-refresh' import { translate } from '@/i18n/i18n' import { isWebClientLocation } from '@/lib/web-client-location' import { useAppStore } from '@/store' @@ -16,16 +16,11 @@ export function ShareSkillsSettingsPane(): React.JSX.Element { const authStatus = useAppStore((state) => state.orcaProfileAuthStatus) const connecting = useAppStore((state) => state.orcaProfileConnecting) const connect = useAppStore((state) => state.connectCurrentOrcaProfile) - const fetchAuthStatus = useAppStore((state) => state.fetchOrcaProfileAuthStatus) const signedIn = authStatus?.state === 'connected' const isWebClient = isWebClientLocation() const agentSharingEnabled = settings?.agentSkillSharingEnabled === true - useEffect(() => { - if (!authStatus) { - void fetchAuthStatus() - } - }, [authStatus, fetchAuthStatus]) + useOrcaProfileAuthStatusRefresh() const steps: HowToStep[] = [ { diff --git a/src/renderer/src/components/settings/accounts-pane-minimax-section.tsx b/src/renderer/src/components/settings/accounts-pane-minimax-section.tsx index 96321584483..ce72bc0a16d 100644 --- a/src/renderer/src/components/settings/accounts-pane-minimax-section.tsx +++ b/src/renderer/src/components/settings/accounts-pane-minimax-section.tsx @@ -10,6 +10,7 @@ import { Popover, PopoverContent, PopoverTrigger } from '../ui/popover' import { MiniMaxIcon } from '../status-bar/icons' import { SearchableSetting } from './SearchableSetting' import type { AccountsPaneSectionModel } from './accounts-pane-types' +import { DebouncedSettingsTextInput } from './DebouncedSettingsTextInput' const MINIMAX_CONSOLE_URL = 'https://platform.minimax.io/console/usage' @@ -265,10 +266,10 @@ export function renderMiniMaxAccountsSection(model: AccountsPaneSectionModel): R - updateSettings({ minimaxGroupId: e.target.value })} + commit={(minimaxGroupId) => updateSettings({ minimaxGroupId })} placeholder={translate( 'auto.components.settings.AccountsPane.0747d6391a', 'Use group ID from cookie' @@ -290,10 +291,10 @@ export function renderMiniMaxAccountsSection(model: AccountsPaneSectionModel): R - updateSettings({ minimaxUsageModels: e.target.value })} + commit={(minimaxUsageModels) => updateSettings({ minimaxUsageModels })} placeholder={translate('auto.components.settings.AccountsPane.3c92b0d31c', 'general')} spellCheck={false} className="text-xs" diff --git a/src/renderer/src/components/settings/accounts-pane-provider-setting-sections.tsx b/src/renderer/src/components/settings/accounts-pane-provider-setting-sections.tsx index 258dea971ae..6bf03df51f4 100644 --- a/src/renderer/src/components/settings/accounts-pane-provider-setting-sections.tsx +++ b/src/renderer/src/components/settings/accounts-pane-provider-setting-sections.tsx @@ -1,11 +1,11 @@ import { translate } from '@/i18n/i18n' import { Button } from '../ui/button' -import { Input } from '../ui/input' import { Label } from '../ui/label' import { Switch } from '../ui/switch' import { GeminiIcon, OpenCodeGoIcon } from '../status-bar/icons' import { SearchableSetting } from './SearchableSetting' import type { AccountsPaneSectionModel } from './accounts-pane-types' +import { DebouncedSettingsTextInput } from './DebouncedSettingsTextInput' export function renderGeminiAccountsSection(model: AccountsPaneSectionModel): React.JSX.Element { const { localAccountRuntimeSentenceLabel, recordFeatureInteraction, settings, updateSettings } = @@ -114,13 +114,11 @@ export function renderOpenCodeAccountsSection(model: AccountsPaneSectionModel): )}
- { - recordOpenCodeSettingEdit('cookie') - updateSettings({ opencodeSessionCookie: e.target.value }) - }} + onEdit={() => recordOpenCodeSettingEdit('cookie')} + commit={(opencodeSessionCookie) => updateSettings({ opencodeSessionCookie })} placeholder={translate( 'auto.components.settings.AccountsPane.a7e38affcd', 'Fe26.2**… token or auth=Fe26.2**… header' @@ -180,13 +178,11 @@ export function renderOpenCodeAccountsSection(model: AccountsPaneSectionModel): {translate('auto.components.settings.AccountsPane.dbdb0b0bd8', 'Workspace ID override')}
- { - recordOpenCodeSettingEdit('workspaceId') - updateSettings({ opencodeWorkspaceId: e.target.value }) - }} + onEdit={() => recordOpenCodeSettingEdit('workspaceId')} + commit={(opencodeWorkspaceId) => updateSettings({ opencodeWorkspaceId })} placeholder={translate( 'auto.components.settings.AccountsPane.a122332371', 'wrk_… (leave blank for automatic lookup)' diff --git a/src/renderer/src/components/settings/use-debounced-settings-text-draft.test.ts b/src/renderer/src/components/settings/use-debounced-settings-text-draft.test.ts new file mode 100644 index 00000000000..da12fd96a26 --- /dev/null +++ b/src/renderer/src/components/settings/use-debounced-settings-text-draft.test.ts @@ -0,0 +1,222 @@ +// @vitest-environment happy-dom + +import { StrictMode } from 'react' +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { useDebouncedSettingsTextDraft } from './use-debounced-settings-text-draft' + +beforeEach(() => { + vi.useFakeTimers() +}) + +afterEach(() => { + cleanup() + vi.useRealTimers() +}) + +describe('useDebouncedSettingsTextDraft', () => { + it('shows every keystroke immediately but commits once', () => { + const commit = vi.fn() + const { result } = renderHook(() => useDebouncedSettingsTextDraft({ value: '', commit })) + + for (const next of ['w', 'wr', 'wrk']) { + act(() => result.current.onChange(next)) + } + + expect(result.current.value).toBe('wrk') + expect(commit).not.toHaveBeenCalled() + + act(() => { + vi.advanceTimersByTime(700) + }) + + expect(commit).toHaveBeenCalledTimes(1) + expect(commit).toHaveBeenCalledWith('wrk') + }) + + it('commits immediately on blur without waiting for the debounce', () => { + const commit = vi.fn() + const { result } = renderHook(() => useDebouncedSettingsTextDraft({ value: '', commit })) + + act(() => result.current.onChange('abc')) + act(() => result.current.onBlur()) + + expect(commit).toHaveBeenCalledExactlyOnceWith('abc') + + act(() => { + vi.advanceTimersByTime(700) + }) + + // The pending timer must not fire a second, duplicate commit. + expect(commit).toHaveBeenCalledTimes(1) + }) + + it('commits a pending edit when the field unmounts', () => { + const commit = vi.fn() + const { result, unmount } = renderHook(() => + useDebouncedSettingsTextDraft({ value: '', commit }) + ) + + act(() => result.current.onChange('half-typed')) + unmount() + + expect(commit).toHaveBeenCalledExactlyOnceWith('half-typed') + }) + + it('adopts an external value while the field is untouched', () => { + const commit = vi.fn() + const { result, rerender } = renderHook( + ({ value }) => useDebouncedSettingsTextDraft({ value, commit }), + { initialProps: { value: 'first' } } + ) + + rerender({ value: 'from-another-window' }) + + expect(result.current.value).toBe('from-another-window') + expect(commit).not.toHaveBeenCalled() + }) + + it('does not let an external value overwrite an in-progress edit', () => { + const commit = vi.fn() + const { result, rerender } = renderHook( + ({ value }) => useDebouncedSettingsTextDraft({ value, commit }), + { initialProps: { value: 'first' } } + ) + + act(() => result.current.onChange('typing')) + rerender({ value: 'from-another-window' }) + + expect(result.current.value).toBe('typing') + }) + + it('does not commit when nothing was edited', () => { + const commit = vi.fn() + const { result } = renderHook(() => useDebouncedSettingsTextDraft({ value: 'x', commit })) + + act(() => result.current.onBlur()) + + expect(commit).not.toHaveBeenCalled() + }) +}) + +describe('useDebouncedSettingsTextDraft flush paths', () => { + it('commits a pending edit on beforeunload, since a window close never unmounts the tree', () => { + const commit = vi.fn() + const { result, unmount } = renderHook(() => + useDebouncedSettingsTextDraft({ value: '', commit }) + ) + + act(() => result.current.onChange('quit-mid-word')) + act(() => { + window.dispatchEvent(new Event('beforeunload', { cancelable: true })) + }) + + expect(commit).toHaveBeenCalledExactlyOnceWith('quit-mid-word') + + // The later unmount and timer must not commit the same value again. + unmount() + act(() => { + vi.advanceTimersByTime(700) + }) + expect(commit).toHaveBeenCalledTimes(1) + }) + + it('does not commit on beforeunload when nothing is pending', () => { + const commit = vi.fn() + renderHook(() => useDebouncedSettingsTextDraft({ value: 'x', commit })) + + act(() => { + window.dispatchEvent(new Event('beforeunload', { cancelable: true })) + }) + + expect(commit).not.toHaveBeenCalled() + }) + + it('stops listening for beforeunload after unmount', () => { + const commit = vi.fn() + const { result, unmount } = renderHook(() => + useDebouncedSettingsTextDraft({ value: '', commit }) + ) + + act(() => result.current.onChange('abc')) + unmount() + act(() => { + window.dispatchEvent(new Event('beforeunload', { cancelable: true })) + }) + + expect(commit).toHaveBeenCalledExactlyOnceWith('abc') + }) + + it('commits every edit burst, not only the first', () => { + const commit = vi.fn() + const { result } = renderHook(() => useDebouncedSettingsTextDraft({ value: '', commit })) + + act(() => result.current.onChange('one')) + act(() => { + vi.advanceTimersByTime(700) + }) + act(() => result.current.onChange('one two')) + act(() => { + vi.advanceTimersByTime(700) + }) + + expect(commit).toHaveBeenNthCalledWith(1, 'one') + expect(commit).toHaveBeenNthCalledWith(2, 'one two') + }) + + it('adopts external values again once a pending edit has been committed', () => { + const commit = vi.fn() + const { result, rerender } = renderHook( + ({ value }) => useDebouncedSettingsTextDraft({ value, commit }), + { initialProps: { value: '' } } + ) + + act(() => result.current.onChange('typed')) + act(() => { + vi.advanceTimersByTime(700) + }) + expect(commit).toHaveBeenCalledExactlyOnceWith('typed') + + // The store echoes the commit, then another window writes a different value. + rerender({ value: 'typed' }) + rerender({ value: 'from-another-window' }) + + expect(result.current.value).toBe('from-another-window') + }) + + it('keeps a keystroke typed while the previous commit is still in flight', () => { + const commit = vi.fn() + const { result, rerender } = renderHook( + ({ value }) => useDebouncedSettingsTextDraft({ value, commit }), + { initialProps: { value: '' } } + ) + + act(() => result.current.onChange('abc')) + act(() => { + vi.advanceTimersByTime(700) + }) + act(() => result.current.onChange('abcd')) + // The store echoes the first commit after the user has already typed more. + rerender({ value: 'abc' }) + + expect(result.current.value).toBe('abcd') + + act(() => result.current.onBlur()) + expect(commit).toHaveBeenLastCalledWith('abcd') + expect(commit).toHaveBeenCalledTimes(2) + }) + + it('does not spuriously commit under StrictMode effect replay', () => { + const commit = vi.fn() + const { result } = renderHook(() => useDebouncedSettingsTextDraft({ value: 'x', commit }), { + wrapper: StrictMode + }) + + expect(commit).not.toHaveBeenCalled() + + act(() => result.current.onChange('xy')) + act(() => result.current.onBlur()) + + expect(commit).toHaveBeenCalledExactlyOnceWith('xy') + }) +}) diff --git a/src/renderer/src/components/settings/use-debounced-settings-text-draft.ts b/src/renderer/src/components/settings/use-debounced-settings-text-draft.ts new file mode 100644 index 00000000000..1f9fe235ead --- /dev/null +++ b/src/renderer/src/components/settings/use-debounced-settings-text-draft.ts @@ -0,0 +1,86 @@ +import { useCallback, useEffect, useRef, useState } from 'react' + +// Matches the repository-hook script draft, the established debounce for settings text in this pane. +const SETTINGS_TEXT_COMMIT_DEBOUNCE_MS = 700 + +export type DebouncedSettingsTextDraft = { + value: string + onChange: (next: string) => void + onBlur: () => void +} + +/** + * Local draft for a free-text setting, committed on a debounce and flushed on blur, unmount, and + * window unload. + * + * Why: binding an `` straight to `updateSettings` sends one IPC round trip per keystroke, + * and each one replaces the `settings` object identity in every other window, re-rendering every + * component subscribed to it. The committed value is unchanged — only the number of commits is. + * + * A pending timer is the single source of truth for "the draft has uncommitted edits": `onChange` + * is the only place that arms it and `flush` the only place that clears it, so there is no separate + * dirty flag to fall out of sync. + */ +export function useDebouncedSettingsTextDraft(args: { + value: string + commit: (next: string) => void +}): DebouncedSettingsTextDraft { + const { value, commit } = args + const [draft, setDraft] = useState(value) + const draftRef = useRef(draft) + const commitRef = useRef(commit) + const timerRef = useRef | null>(null) + + // Why an effect, not a render-time write: render must stay pure, and React can replay it. + useEffect(() => { + commitRef.current = commit + }, [commit]) + + // Why gated on a pending commit: an external write (another window, a reset) should land in the + // field, but must not yank characters out from under someone mid-edit. + useEffect(() => { + if (timerRef.current !== null) { + return + } + draftRef.current = value + setDraft(value) + }, [value]) + + const flush = useCallback(() => { + if (timerRef.current === null) { + return + } + clearTimeout(timerRef.current) + timerRef.current = null + commitRef.current(draftRef.current) + }, []) + + const onChange = useCallback( + (next: string) => { + draftRef.current = next + setDraft(next) + if (timerRef.current !== null) { + clearTimeout(timerRef.current) + } + timerRef.current = setTimeout(flush, SETTINGS_TEXT_COMMIT_DEBOUNCE_MS) + }, + [flush] + ) + + // Why unmount: closing the pane (or the settings search hiding the section) mid-word must persist + // the same value typing it would have. `flush` has no dependencies, so this cleanup only ever runs + // on unmount. + // Why beforeunload: a window close or app quit never unmounts the tree, so the cleanup cannot run. + // The close coordinator dispatches a synthetic beforeunload while the tree is still mounted so + // listeners like this one can flush; `updateSettings` issues its IPC synchronously, ahead of the + // close confirmation, so main persists the value before it flushes the store on quit. + useEffect(() => { + window.addEventListener('beforeunload', flush) + return () => { + window.removeEventListener('beforeunload', flush) + flush() + } + }, [flush]) + + return { value: draft, onChange, onBlur: flush } +} diff --git a/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx b/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx index 5709dc87f90..13f3a0c883d 100644 --- a/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx +++ b/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx @@ -17,8 +17,9 @@ import { markOnboardingProjectAdded } from '@/lib/onboarding-project-checklist' import { translate } from '@/i18n/i18n' import { upsertAddedRepoWithProjectHostSetup } from './add-repo-store-upsert' import { worktreeRefreshOptions } from './add-repo-runtime-owner' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' const NonGitFolderDialog = React.memo(function NonGitFolderDialog() { const activeModal = useAppStore((s) => s.activeModal) @@ -101,8 +102,11 @@ const NonGitFolderDialog = React.memo(function NonGitFolderDialog() { ...(launch.startup ? { startup: launch.startup } : {}), ...(launch.route === 'structured-native-chat' ? { providesInitialSurface: true } : {}) }) - if (launch.route === 'structured-native-chat' && launch.agent === 'codex') { - const structured = startStructuredCodexLaunch(folderWorktree.id) + if ( + launch.route === 'structured-native-chat' && + isAgentSessionHandleProvider(launch.agent) + ) { + const structured = startStructuredAgentLaunch(folderWorktree.id, launch.agent) const fallback = structured.claimDefinitiveRefusalFallback(() => { activateAndRevealWorktree(folderWorktree.id, { sidebarRevealBehavior: 'auto', diff --git a/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts b/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts index 661bf623880..b6f1a1654f3 100644 --- a/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts +++ b/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts @@ -28,8 +28,9 @@ import { resolveAgentLaunchRoute } from '@/lib/agent-launch-routing' import { readLocalRuntimeCapabilities } from '@/runtime/local-runtime-capabilities' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { useAppStore } from '@/store' import { buildFolderWorkspaceLinkedStartupPlan, @@ -232,8 +233,8 @@ export async function submitFolderWorkspaceCreate({ runtimeEnvironmentId }) let structuredLaunchAccepted = structuredLaunch - if (structuredLaunch && quickAgent === 'codex') { - const launch = startStructuredCodexLaunch(folderWorkspaceKey(workspace.id), { + if (structuredLaunch && isAgentSessionHandleProvider(quickAgent)) { + const launch = startStructuredAgentLaunch(folderWorkspaceKey(workspace.id), quickAgent, { prompt: launchDraftPrompt ?? note }) const refusalFallback = launch.claimDefinitiveRefusalFallback(async () => { diff --git a/src/renderer/src/components/sidebar/natural-worktree-ids.test.ts b/src/renderer/src/components/sidebar/natural-worktree-ids.test.ts new file mode 100644 index 00000000000..b36fc2aa262 --- /dev/null +++ b/src/renderer/src/components/sidebar/natural-worktree-ids.test.ts @@ -0,0 +1,45 @@ +import { describe, expect, it } from 'vitest' +import { PINNED_GROUP_KEY } from './worktree-list/grouping/group-keys' +import { getNaturalWorktreeIds } from './natural-worktree-ids' + +const item = (id: string, sectionKey: string) => ({ + type: 'item' as const, + sectionKey, + worktree: { id } +}) + +describe('getNaturalWorktreeIds', () => { + it('collects item rows outside the pinned section', () => { + expect([...getNaturalWorktreeIds([item('a', 'group-1'), item('b', 'group-2')])]).toEqual([ + 'a', + 'b' + ]) + }) + + it('excludes a pinned duplicate that also renders in its natural group', () => { + const rows = [item('a', PINNED_GROUP_KEY), item('a', 'group-1'), item('b', PINNED_GROUP_KEY)] + + const ids = getNaturalWorktreeIds(rows) + + expect(ids.has('a')).toBe(true) + expect(ids.has('b')).toBe(false) + }) + + it('ignores every non-item row type', () => { + const rows = [ + { type: 'header', key: 'k' }, + { type: 'host-header' }, + { type: 'imported-worktrees-card' }, + { type: 'new-external-worktrees-inbox' }, + { type: 'pending-creation' }, + { type: 'folder-workspace' }, + item('a', 'group-1') + ] + + expect([...getNaturalWorktreeIds(rows)]).toEqual(['a']) + }) + + it('returns an empty set for no rows', () => { + expect(getNaturalWorktreeIds([]).size).toBe(0) + }) +}) diff --git a/src/renderer/src/components/sidebar/natural-worktree-ids.ts b/src/renderer/src/components/sidebar/natural-worktree-ids.ts new file mode 100644 index 00000000000..7ac21ba1f9f --- /dev/null +++ b/src/renderer/src/components/sidebar/natural-worktree-ids.ts @@ -0,0 +1,24 @@ +import { PINNED_GROUP_KEY } from './worktree-list/grouping/group-keys' + +/** The row shape both drag models share: only `item` rows carry a worktree and a section. */ +type NaturalWorktreeIdRow = { type: string } & Partial<{ + sectionKey: string + worktree: { id: string } +}> + +/** + * Ids of worktrees rendered in their own group. + * + * Why: a pinned duplicate of a worktree that also renders in its natural group is not its own drag + * slot. Shared by every drag model so the rule lives in one place, and built with a loop rather + * than `flatMap` — that allocated a throwaway array per row, four times per row-model rebuild. + */ +export function getNaturalWorktreeIds(rows: readonly NaturalWorktreeIdRow[]): Set { + const ids = new Set() + for (const row of rows) { + if (row.type === 'item' && row.sectionKey !== PINNED_GROUP_KEY && row.worktree) { + ids.add(row.worktree.id) + } + } + return ids +} diff --git a/src/renderer/src/components/sidebar/worktree-drag-units.ts b/src/renderer/src/components/sidebar/worktree-drag-units.ts index e53d663a9f0..9b900ab72fc 100644 --- a/src/renderer/src/components/sidebar/worktree-drag-units.ts +++ b/src/renderer/src/components/sidebar/worktree-drag-units.ts @@ -1,5 +1,6 @@ import type { WorktreeDragGroup } from './worktree-manual-order' import { ALL_GROUP_KEY, PINNED_GROUP_KEY } from './worktree-list/grouping/group-keys' +import { getNaturalWorktreeIds } from './natural-worktree-ids' export type WorktreeDragUnitGroup = WorktreeDragGroup & { units: { worktreeId: string; worktreeIds: string[] }[] @@ -19,11 +20,7 @@ export function getWorktreeDragUnitGroups( ): WorktreeDragUnitGroup[] { const groups: WorktreeDragUnitGroup[] = [] let current: { key: string; units: WorktreeDragUnitGroup['units'] } | null = null - const naturalWorktreeIds = new Set( - rows.flatMap((row) => - row.type === 'item' && row.sectionKey !== PINNED_GROUP_KEY ? [row.worktree.id] : [] - ) - ) + const naturalWorktreeIds = getNaturalWorktreeIds(rows) for (const row of rows) { if (row.type === 'header') { diff --git a/src/renderer/src/components/sidebar/worktree-list/drag/groups.ts b/src/renderer/src/components/sidebar/worktree-list/drag/groups.ts index f8ee60c495e..7f1a8e19f49 100644 --- a/src/renderer/src/components/sidebar/worktree-list/drag/groups.ts +++ b/src/renderer/src/components/sidebar/worktree-list/drag/groups.ts @@ -1,16 +1,8 @@ import { ALL_GROUP_KEY, PINNED_GROUP_KEY } from '../grouping/group-keys' +import { getNaturalWorktreeIds } from '../../natural-worktree-ids' import type { HostSectionRow } from '../../host-section-rows' import type { WorktreeDragGroup } from '../../worktree-manual-order' -// A pinned duplicate of a worktree that also renders in its natural group is not its own drag slot. -function getNaturalWorktreeIds(rows: readonly HostSectionRow[]): Set { - return new Set( - rows.flatMap((row) => - row.type === 'item' && row.sectionKey !== PINNED_GROUP_KEY ? [row.worktree.id] : [] - ) - ) -} - export function getWorktreeDragGroups(rows: HostSectionRow[]): WorktreeDragGroup[] { const groups: WorktreeDragGroup[] = [] let current: { key: string; ids: string[] } | null = null diff --git a/src/renderer/src/components/sidebar/worktree-list/drag/use-session.ts b/src/renderer/src/components/sidebar/worktree-list/drag/use-session.ts index 78b25b25f38..92e23ec4928 100644 --- a/src/renderer/src/components/sidebar/worktree-list/drag/use-session.ts +++ b/src/renderer/src/components/sidebar/worktree-list/drag/use-session.ts @@ -24,6 +24,7 @@ import { } from '../../worktree-sidebar-drop-preview' import { getWorktreeDragGroups, getWorktreeDragIndexes } from './groups' import type { WorktreeItemRow } from '../listing/renderable-rows' +import { getNaturalWorktreeIds } from '../../natural-worktree-ids' export type WorktreeStatusDropRequest = { pointerY: number @@ -48,15 +49,7 @@ export function useWorktreeDragSession(args: { const worktreeDragGroups = useMemo(() => getWorktreeDragGroups(rows), [rows]) const worktreeDragUnitGroups = useMemo(() => getWorktreeDragUnitGroups(rows), [rows]) - const naturalDragWorktreeIds = useMemo( - () => - new Set( - rows.flatMap((row) => - row.type === 'item' && row.sectionKey !== PINNED_GROUP_KEY ? [row.worktree.id] : [] - ) - ), - [rows] - ) + const naturalDragWorktreeIds = useMemo(() => getNaturalWorktreeIds(rows), [rows]) const worktreeLineageDragRows = useMemo( () => rows diff --git a/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.test.ts b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.test.ts new file mode 100644 index 00000000000..7411dc00d6a --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, it } from 'vitest' +import type { ExecutionHostId } from '../../../../../../shared/execution-host' +import type { Repo } from '../../../../../../shared/repo-types' +import type { Worktree } from '../../../../../../shared/worktree/types' +import { getHostWorktreeCounts, getHostWorktreeIds } from './host-labels' + +const LOCAL = 'local' as ExecutionHostId + +function makeRepos(hostByRepo: Record): Map { + return new Map( + Object.entries(hostByRepo).map(([id, executionHostId]) => [ + id, + { id, path: `/${id}`, ...(executionHostId ? { executionHostId } : {}) } as Repo + ]) + ) +} + +function makeWorktrees(rows: { id: string; repoId: string }[]): Worktree[] { + return rows.map((row) => row as Worktree) +} + +describe('host worktree counts and ids', () => { + it('reports a count equal to the length of each host id list', () => { + const repoMap = makeRepos({ + 'repo-local': undefined, + 'repo-remote': 'ssh:other' as ExecutionHostId + }) + const worktrees = makeWorktrees([ + { id: 'a', repoId: 'repo-local' }, + { id: 'b', repoId: 'repo-local' }, + { id: 'c', repoId: 'repo-remote' } + ]) + + const counts = getHostWorktreeCounts(worktrees, repoMap, LOCAL) + const ids = getHostWorktreeIds(worktrees, repoMap, LOCAL) + + expect(ids).toBeDefined() + expect(counts).toBeDefined() + for (const [hostId, hostIds] of ids ?? []) { + expect(counts?.get(hostId), `count for ${hostId}`).toBe(hostIds.length) + } + expect([...(counts?.keys() ?? [])].sort()).toEqual([...(ids?.keys() ?? [])].sort()) + }) + + it('counts a repeated host identity once', () => { + const repoMap = makeRepos({ 'repo-local': undefined }) + const worktrees = makeWorktrees([ + { id: 'a', repoId: 'repo-local' }, + { id: 'a', repoId: 'repo-local' }, + { id: 'b', repoId: 'repo-local' } + ]) + + expect(getHostWorktreeCounts(worktrees, repoMap, LOCAL)?.get(LOCAL)).toBe(2) + expect(getHostWorktreeIds(worktrees, repoMap, LOCAL)?.get(LOCAL)).toEqual(['a', 'b']) + }) + + it('returns undefined for an empty lane', () => { + const repoMap = makeRepos({}) + expect(getHostWorktreeCounts([], repoMap, LOCAL)).toBeUndefined() + expect(getHostWorktreeIds([], repoMap, LOCAL)).toBeUndefined() + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts index d1c5c8be2d4..925b04eed41 100644 --- a/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts +++ b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts @@ -153,17 +153,17 @@ export function getHostWorktreeCounts( if (worktrees.length === 0) { return undefined } - const counts = new Map() + // Derived from the id map rather than repeating its dedupe walk: every caller asks for both, + // and a host's count is exactly the length of its id list by construction. // Dedup by host, not by bare id: the same id on two hosts is two workspaces // and has to be counted under each of them (STA-4343). - const seenIdentities = new Set() - for (const worktree of worktrees) { - if (seenIdentities.has(getWorktreeHostIdentity(worktree))) { - continue - } - seenIdentities.add(getWorktreeHostIdentity(worktree)) - const hostId = getWorktreeExecutionHostId(worktree, repoMap.get(worktree.repoId), defaultHostId) - counts.set(hostId, (counts.get(hostId) ?? 0) + 1) + const idsByHost = getHostWorktreeIds(worktrees, repoMap, defaultHostId) + if (!idsByHost) { + return undefined + } + const counts = new Map() + for (const [hostId, ids] of idsByHost) { + counts.set(hostId, ids.length) } return counts } @@ -179,10 +179,12 @@ export function getHostWorktreeIds( const idsByHost = new Map() const seenIdentities = new Set() for (const worktree of worktrees) { - if (seenIdentities.has(getWorktreeHostIdentity(worktree))) { + // Hoisted: the identity string was built twice per row, once to test and once to record. + const identity = getWorktreeHostIdentity(worktree) + if (seenIdentities.has(identity)) { continue } - seenIdentities.add(getWorktreeHostIdentity(worktree)) + seenIdentities.add(identity) const hostId = getWorktreeExecutionHostId(worktree, repoMap.get(worktree.repoId), defaultHostId) const ids = idsByHost.get(hostId) ?? [] ids.push(worktree.id) diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.test.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.test.tsx new file mode 100644 index 00000000000..5ce0c7a950b --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.test.tsx @@ -0,0 +1,102 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { TooltipProvider } from '@/components/ui/tooltip' +import { useAppStore } from '@/store' +import type { Repo } from '../../../../../../shared/repo-types' +import { makeDetectedResult } from '@/store/slices/worktrees-detected-listing-fixtures' +import { RepoScanUnavailableIndicator } from './RepoScanUnavailableIndicator' + +const repo = { + id: 'repo-1', + path: 'C:\\repo', + displayName: 'repo', + badgeColor: '#000', + addedAt: 0 +} as Repo + +const initialState = useAppStore.getInitialState() +const roots: Root[] = [] + +async function render(): Promise { + const container = document.createElement('div') + document.body.appendChild(container) + const root = createRoot(container) + roots.push(root) + await act(async () => { + root.render( + + + + ) + }) + return container +} + +describe('RepoScanUnavailableIndicator', () => { + beforeEach(() => { + globalThis.IS_REACT_ACT_ENVIRONMENT = true + useAppStore.setState(initialState, true) + }) + + afterEach(async () => { + for (const root of roots.splice(0)) { + await act(async () => root.unmount()) + } + document.body.innerHTML = '' + useAppStore.setState(initialState, true) + }) + + it('renders nothing for an authoritative listing', async () => { + useAppStore.setState({ + detectedWorktreesByRepo: { [repo.id]: makeDetectedResult(repo.id, []) } + }) + + const container = await render() + + expect(container.querySelector('button')).toBeNull() + }) + + // Why: a non-authoritative listing without a reason is the disconnected-SSH shape, which the + // host header already explains; this marker is only for a scan that failed with a cause. + it('renders nothing for a non-authoritative listing that carries no reason', async () => { + useAppStore.setState({ + detectedWorktreesByRepo: { + [repo.id]: makeDetectedResult(repo.id, [], { + authoritative: false, + source: 'metadata-fallback' + }) + } + }) + + const container = await render() + + expect(container.querySelector('button')).toBeNull() + }) + + it('marks a failed scan and re-runs it on click', async () => { + const fetchWorktrees = vi.fn(async () => true) + useAppStore.setState({ + fetchWorktrees: fetchWorktrees as never, + detectedWorktreesByRepo: { + [repo.id]: makeDetectedResult(repo.id, [], { + authoritative: false, + source: 'metadata-fallback', + unavailableReason: 'wsl.exe host failure (distro "kali-linux"): WSL_E_DISTRO_NOT_FOUND' + }) + } + }) + + const container = await render() + const button = container.querySelector('button') + + expect(button?.getAttribute('aria-label')).toContain('Worktree scan failed for repo') + expect(button?.className).toContain('text-destructive') + await act(async () => { + button?.click() + }) + expect(fetchWorktrees).toHaveBeenCalledWith(repo.id, { executionHostId: 'local' }) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.tsx new file mode 100644 index 00000000000..a956598140e --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/rows/RepoScanUnavailableIndicator.tsx @@ -0,0 +1,75 @@ +import React from 'react' +import { TriangleAlert } from 'lucide-react' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import { useAppStore } from '@/store' +import type { Repo } from '../../../../../../shared/repo-types' +import { getRepoExecutionHostId } from '../../../../../../shared/execution-host' +import { + handleRepoHeaderActionPointerDown, + stopRepoHeaderKeyboardToggle +} from './header-event-guards' + +/** + * Marks a repo whose worktree scan failed, so its rows are retained but cannot be trusted. + * Click re-runs the scan: the failure is otherwise re-tried only by the next incidental refresh. + */ +export function RepoScanUnavailableIndicator({ repo }: { repo: Repo }): React.JSX.Element | null { + const detected = useAppStore((s) => s.detectedWorktreesByRepo[repo.id]) + const fetchWorktrees = useAppStore((s) => s.fetchWorktrees) + const [pending, setPending] = React.useState(false) + if (!detected || detected.authoritative || !detected.unavailableReason) { + return null + } + const title = translate( + 'auto.components.sidebar.RepoScanUnavailableIndicator.title', + 'Worktree scan failed for {{value0}}', + { value0: repo.displayName } + ) + const retryLabel = translate( + 'auto.components.sidebar.RepoScanUnavailableIndicator.retry', + 'Retry scan' + ) + return ( + + + + + +
+
{title}
+
{detected.unavailableReason}
+
+ {translate( + 'auto.components.sidebar.RepoScanUnavailableIndicator.retained', + 'Existing worktrees are kept until a scan succeeds. Click to retry.' + )} +
+
+
+
+ ) +} diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx index b4ed149712a..db0c7ec1371 100644 --- a/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx +++ b/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx @@ -26,6 +26,7 @@ import { WORKTREE_SECTION_HEADER_PADDING_LEFT } from './indentation' import { FolderPathStatusIndicator } from './FolderPathStatusIndicator' +import { RepoScanUnavailableIndicator } from './RepoScanUnavailableIndicator' import { ProjectGroupCreateWorkspaceButton, ProjectGroupHeaderMenu @@ -334,6 +335,7 @@ export function renderWorktreeSectionHeaderRow(args: {
+ {isRepoHeader ? : null}
diff --git a/src/renderer/src/components/sidebar/worktree-list/viewport/use-row-measurement.ts b/src/renderer/src/components/sidebar/worktree-list/viewport/use-row-measurement.ts index ff812a950be..a6f795cf392 100644 --- a/src/renderer/src/components/sidebar/worktree-list/viewport/use-row-measurement.ts +++ b/src/renderer/src/components/sidebar/worktree-list/viewport/use-row-measurement.ts @@ -51,10 +51,6 @@ export function useVirtualRowMeasurementSync(args: { const { virtualizer, isCurrentVirtualRowElement } = virtualization const prCacheLen = useAppStore((s) => countRecordKeysByReference(s.prCache)) const issueCacheLen = useAppStore((s) => countRecordKeysByReference(s.issueCache)) - const renderRowKeySignature = useMemo( - () => renderRows.map(getRenderRowKey).join('\n'), - [renderRows] - ) const activeRenderRowKeys = useMemo(() => new Set(renderRows.map(getRenderRowKey)), [renderRows]) const lineageRowRekeys = useMemo(() => buildLineageRowRekeyMap(renderRows), [renderRows]) const totalSize = virtualizer.getTotalSize() @@ -100,14 +96,7 @@ export function useVirtualRowMeasurementSync(args: { measureMountedRows() const frameId = window.requestAnimationFrame(measureMountedRows) return () => window.cancelAnimationFrame(frameId) - }, [ - activeRenderRowKeys, - prCacheLen, - issueCacheLen, - measureMountedRows, - renderRowKeySignature, - virtualizer - ]) + }, [activeRenderRowKeys, prCacheLen, issueCacheLen, measureMountedRows, virtualizer]) useVirtualizedScrollAnchor({ anchorRef: scrollAnchorRef, diff --git a/src/renderer/src/components/status-bar/ClaudeSwitcherMenu.tsx b/src/renderer/src/components/status-bar/ClaudeSwitcherMenu.tsx index c5414d66d9a..2e159c67018 100644 --- a/src/renderer/src/components/status-bar/ClaudeSwitcherMenu.tsx +++ b/src/renderer/src/components/status-bar/ClaudeSwitcherMenu.tsx @@ -34,6 +34,7 @@ import { import { AccountRuntimeToggle } from './StatusBarAccountControls' import { InlineUsageBars, InlineUsageSkeleton } from './InlineProviderUsage' import { ProviderDetailsMenu } from './ProviderDetailsMenu' +import { getClaudeAccountSyncKey } from './provider-account-sync-key' // Exported so its account-switch/reset logic is preserved for row drill-in even // though the footer now opens the consolidated UsageRosterPanel first. @@ -83,13 +84,7 @@ export function ClaudeSwitcherMenu({ getWindowsTerminalCapabilityOwnerKey(settings?.activeRuntimeEnvironmentId), runtimeTarget ) - const claudeAccountSyncKey = useAppStore((s) => { - const settings = s.settings - if (!settings) { - return 'no-settings' - } - return `${settings.activeRuntimeEnvironmentId?.trim() || 'local'}:${settings.activeClaudeManagedAccountId ?? 'system'}:${JSON.stringify(settings.activeClaudeManagedAccountIdsByRuntime ?? null)}:${settings.claudeManagedAccounts.map((account) => `${account.id}:${account.updatedAt}`).join('|')}` - }) + const claudeAccountSyncKey = useAppStore((s) => getClaudeAccountSyncKey(s.settings)) const accountState = resolveClaudeStatusAccountState(settings, accounts) useEffect(() => { diff --git a/src/renderer/src/components/status-bar/codex-switcher-projection.ts b/src/renderer/src/components/status-bar/codex-switcher-projection.ts index 192f8d76fd9..5bddcce53f2 100644 --- a/src/renderer/src/components/status-bar/codex-switcher-projection.ts +++ b/src/renderer/src/components/status-bar/codex-switcher-projection.ts @@ -1,13 +1,7 @@ -import type { GlobalSettings } from '../../../../shared/global-settings-types' import type { ProviderRateLimits } from '../../../../shared/rate-limit-types' import { formatResetCreditExpiry } from './tooltip' -export function getCodexAccountSyncKey(settings: GlobalSettings | null | undefined): string { - if (!settings) { - return 'no-settings' - } - return `${settings.activeRuntimeEnvironmentId?.trim() || 'local'}:${settings.activeCodexManagedAccountId ?? 'system'}:${JSON.stringify(settings.activeCodexManagedAccountIdsByRuntime ?? null)}:${settings.codexManagedAccounts.map((account) => `${account.id}:${account.updatedAt}`).join('|')}` -} +export { getCodexAccountSyncKey } from './provider-account-sync-key' export function getCodexResetProjection( codex: ProviderRateLimits, diff --git a/src/renderer/src/components/status-bar/provider-account-sync-key.test.ts b/src/renderer/src/components/status-bar/provider-account-sync-key.test.ts new file mode 100644 index 00000000000..7931bb1598b --- /dev/null +++ b/src/renderer/src/components/status-bar/provider-account-sync-key.test.ts @@ -0,0 +1,42 @@ +import { describe, expect, it } from 'vitest' +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { getClaudeAccountSyncKey, getCodexAccountSyncKey } from './provider-account-sync-key' + +function makeSettings(overrides: Partial = {}): GlobalSettings { + return { + activeRuntimeEnvironmentId: null, + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: null, + claudeManagedAccounts: [{ id: 'a1', updatedAt: 5 }], + activeCodexManagedAccountId: null, + activeCodexManagedAccountIdsByRuntime: null, + codexManagedAccounts: [{ id: 'c1', updatedAt: 7 }], + ...overrides + } as unknown as GlobalSettings +} + +describe.each([ + ['claude', getClaudeAccountSyncKey], + ['codex', getCodexAccountSyncKey] +])('%s account sync key', (_provider, getSyncKey) => { + it('returns the same string for the same settings identity', () => { + const settings = makeSettings() + expect(getSyncKey(settings)).toBe(getSyncKey(settings)) + }) + + it('answers no-settings when settings are absent', () => { + expect(getSyncKey(null)).toBe('no-settings') + expect(getSyncKey(undefined)).toBe('no-settings') + }) + + it('recomputes for a new settings identity', () => { + const first = getSyncKey(makeSettings()) + const second = getSyncKey( + makeSettings({ + claudeManagedAccounts: [{ id: 'a1', updatedAt: 6 }], + codexManagedAccounts: [{ id: 'c1', updatedAt: 8 }] + } as unknown as Partial) + ) + expect(second).not.toBe(first) + }) +}) diff --git a/src/renderer/src/components/status-bar/provider-account-sync-key.ts b/src/renderer/src/components/status-bar/provider-account-sync-key.ts new file mode 100644 index 00000000000..acc8e3e9867 --- /dev/null +++ b/src/renderer/src/components/status-bar/provider-account-sync-key.ts @@ -0,0 +1,42 @@ +import type { GlobalSettings } from '../../../../shared/global-settings-types' + +// Why memoized on settings identity: these run inside useAppStore selectors, which Zustand re-runs +// on every store write. Building the key stringifies a map and joins the whole managed-account +// roster, and its inputs only move when `settings` is replaced. +const claudeKeyBySettings = new WeakMap() +const codexKeyBySettings = new WeakMap() + +function memoizeSyncKey( + cache: WeakMap, + settings: GlobalSettings | null | undefined, + build: (settings: GlobalSettings) => string +): string { + if (!settings) { + return 'no-settings' + } + const cached = cache.get(settings) + if (cached !== undefined) { + return cached + } + const key = build(settings) + cache.set(settings, key) + return key +} + +export function getClaudeAccountSyncKey(settings: GlobalSettings | null | undefined): string { + return memoizeSyncKey( + claudeKeyBySettings, + settings, + (resolved) => + `${resolved.activeRuntimeEnvironmentId?.trim() || 'local'}:${resolved.activeClaudeManagedAccountId ?? 'system'}:${JSON.stringify(resolved.activeClaudeManagedAccountIdsByRuntime ?? null)}:${resolved.claudeManagedAccounts.map((account) => `${account.id}:${account.updatedAt}`).join('|')}` + ) +} + +export function getCodexAccountSyncKey(settings: GlobalSettings | null | undefined): string { + return memoizeSyncKey( + codexKeyBySettings, + settings, + (resolved) => + `${resolved.activeRuntimeEnvironmentId?.trim() || 'local'}:${resolved.activeCodexManagedAccountId ?? 'system'}:${JSON.stringify(resolved.activeCodexManagedAccountIdsByRuntime ?? null)}:${resolved.codexManagedAccounts.map((account) => `${account.id}:${account.updatedAt}`).join('|')}` + ) +} diff --git a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx index 85d978e3bf6..6d5b523c2af 100644 --- a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx +++ b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx @@ -8,6 +8,7 @@ import { useAgentDetectionTargetForWorktree } from '@/hooks/useAgentDetectionTar import { useDetectedAgents } from '@/hooks/useDetectedAgents' import { useOptionalShortcutLabel } from '@/hooks/useShortcutLabel' import { launchAgentInNewTab } from '@/lib/launch-agent-in-new-tab' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import type { TuiAgent } from '../../../../shared/tui-agent' import type { LaunchSource } from '../../../../shared/telemetry-events' import { @@ -15,7 +16,7 @@ import { filterEnabledTuiAgents } from '../../../../shared/tui-agent-selection' import { translate } from '@/i18n/i18n' -import { useStructuredCodexLaunchStatus } from '@/lib/structured-agent-session-launch' +import { useStructuredAgentLaunchStatus } from '@/lib/structured-agent-session-launch' export type QuickLaunchAgentMenuItemsProps = { worktreeId: string @@ -117,7 +118,12 @@ function QuickLaunchAgentMenuItemsInner({ const openSettingsPage = useAppStore((s) => s.openSettingsPage) const openSettingsTarget = useAppStore((s) => s.openSettingsTarget) const newAgentShortcut = useOptionalShortcutLabel('tab.newAgent') - const structuredCodexLaunchStatus = useStructuredCodexLaunchStatus(worktreeId) + // One hook per structured provider: the launch registry is keyed by agent, and hooks cannot run + // inside the agent list's render loop. + const structuredLaunchStatusByAgent = { + claude: useStructuredAgentLaunchStatus(worktreeId, 'claude'), + codex: useStructuredAgentLaunchStatus(worktreeId, 'codex') + } const openAgentSettings = useCallback(() => { openSettingsTarget({ pane: 'agents', repoId: null }) @@ -199,26 +205,33 @@ function QuickLaunchAgentMenuItemsInner({ {agents.map((agent) => { const entry = getCatalogEntry(agent) const label = entry?.label ?? agent - const isStructuredCodexPending = - agent === 'codex' && structuredCodexLaunchStatus === 'pending' - const menuLabel = isStructuredCodexPending ? 'Starting Codex chat…' : label + const isStructuredLaunchPending = + isAgentSessionHandleProvider(agent) && structuredLaunchStatusByAgent[agent] === 'pending' + const pendingLabel = translate( + 'components.native-chat.structuredSessionLaunchPending', + 'Starting {{value0}} chat…', + { value0: label } + ) + const menuLabel = isStructuredLaunchPending ? pendingLabel : label const showsDefaultAgentShortcut = newAgentShortcut !== null && defaultAgent !== 'blank' && agent === defaultAgent return ( runLaunch(agent)} className="gap-2 rounded-[7px] px-2 py-1.5 text-[12px] leading-5 font-medium" - title={translate( - 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', - isStructuredCodexPending - ? 'Starting Codex chat…' - : 'Launch {{value0}} in a new terminal', - isStructuredCodexPending ? undefined : { value0: label } - )} + title={ + isStructuredLaunchPending + ? pendingLabel + : translate( + 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', + 'Launch {{value0}} in a new terminal', + { value0: label } + ) + } > - {isStructuredCodexPending ? ( + {isStructuredLaunchPending ? ( ))} diff --git a/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx b/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx index 1b620e35dbe..a2226036f27 100644 --- a/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx +++ b/src/renderer/src/components/tab-bar/TerminalTabSplitMenuSection.tsx @@ -21,6 +21,7 @@ export function TerminalTabSplitMenuSection({ onActivate, splitRightShortcut, splitDownShortcut, + showTerminalSplit = true, trailingSeparator = false }: { unifiedTabId: string @@ -30,6 +31,7 @@ export function TerminalTabSplitMenuSection({ onActivate: (tabId: string) => void splitRightShortcut: string splitDownShortcut: string + showTerminalSplit?: boolean trailingSeparator?: boolean }): React.JSX.Element { const splitActiveTerminalPane = (direction: 'vertical' | 'horizontal'): void => { @@ -42,33 +44,37 @@ export function TerminalTabSplitMenuSection({ return ( <> - - - - {translate( - 'auto.components.tab.bar.TerminalTabSplitMenuSection.splitTerminal', - 'Split terminal' - )} - - - splitActiveTerminalPane('vertical')}> - + {showTerminalSplit ? ( + + + {translate( - 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalRight', - 'Split terminal right' + 'auto.components.tab.bar.TerminalTabSplitMenuSection.splitTerminal', + 'Split terminal' )} - {splitRightShortcut} - - splitActiveTerminalPane('horizontal')}> - - {translate( - 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalDown', - 'Split terminal down' - )} - {splitDownShortcut} - - - + + + splitActiveTerminalPane('vertical')}> + + {translate( + 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalRight', + 'Split terminal right' + )} + {splitRightShortcut} + + splitActiveTerminalPane('horizontal')}> + + {translate( + 'auto.components.tab.bar.SortableTabContextMenu.splitTerminalDown', + 'Split terminal down' + )} + {splitDownShortcut} + + + + ) : null} {trailingSeparator ? : null} ) diff --git a/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx b/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx index 24114f42a16..95e9a98e7bb 100644 --- a/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx +++ b/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx @@ -266,6 +266,7 @@ export function renderTabBarItems({ onSetTabColor={onSetTabColor} onTogglePin={() => togglePinned(item)} onToggleExpand={() => {}} + canSplitTerminal={false} dragData={dragData} dropIndicator={dropIndicatorByVisibleId.get(item.id) ?? null} includeTopTabBorder={includeTopTabBorder} diff --git a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts index 20791dd3dde..d1feb923255 100644 --- a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts +++ b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.structured-session.test.ts @@ -3,7 +3,7 @@ import type * as ReactModule from 'react' const mocks = vi.hoisted(() => ({ callRuntimeRpc: vi.fn(), - cancelStructuredCodexLaunch: vi.fn(), + cancelStructuredAgentLaunch: vi.fn(), closeBrowserTab: vi.fn(), closeFile: vi.fn(), closeStructuredAgentSession: vi.fn(), @@ -73,7 +73,7 @@ vi.mock('@/runtime/structured-agent-session-close', () => ({ })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - cancelStructuredCodexLaunch: mocks.cancelStructuredCodexLaunch + cancelStructuredAgentLaunch: mocks.cancelStructuredAgentLaunch })) vi.mock('@/runtime/runtime-worktree-selector', () => ({ @@ -129,7 +129,7 @@ describe('structured agent-session close ordering', () => { closeItem(AGENT_TAB.id) await vi.waitFor(() => expect(order).toEqual(['agent-close', 'tab-close', 'local-remove'])) - expect(mocks.cancelStructuredCodexLaunch).toHaveBeenCalledWith('wt-1', 'session-1') + expect(mocks.cancelStructuredAgentLaunch).toHaveBeenCalledWith('wt-1', 'session-1') }) it('keeps the tab available when owner disposal fails, so close can be retried', async () => { @@ -154,7 +154,7 @@ describe('structured agent-session close ordering', () => { closeMany([AGENT_TAB.id]) - expect(mocks.cancelStructuredCodexLaunch).toHaveBeenCalledWith('wt-1', 'session-1') + expect(mocks.cancelStructuredAgentLaunch).toHaveBeenCalledWith('wt-1', 'session-1') await vi.waitFor(() => expect(mocks.closeUnifiedTab).toHaveBeenCalledWith(AGENT_TAB.id)) }) }) diff --git a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts index 0c8c71d6d7c..a5a9a65455d 100644 --- a/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts +++ b/src/renderer/src/components/tab-group/useTabGroupTabCloseCommands.ts @@ -10,7 +10,7 @@ import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner import { closeBrowserWorkspaceTabOnHosts } from '@/runtime/browser-workspace-tab-close' import { callRuntimeRpc, getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' import { closeStructuredAgentSession } from '@/runtime/structured-agent-session-close' -import { cancelStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' +import { cancelStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' import { translate } from '@/i18n/i18n' @@ -18,7 +18,7 @@ function reportStructuredSessionCloseError(error: unknown): void { toast.error( translate( 'components.native-chat.structuredSessionCloseFailed', - 'Could not close this Codex chat' + 'Could not close this chat session' ), { description: error instanceof Error ? error.message : String(error) } ) @@ -121,12 +121,12 @@ export function useTabGroupTabCloseCommands({ worktreeId ) if (item.contentType === 'agent-session') { - cancelStructuredCodexLaunch(worktreeId, item.entityId) + cancelStructuredAgentLaunch(worktreeId, item.entityId) // Why: the structured session lives on the host, so the local tab close must also // retire the host's canonical row or it reappears on the next sync. // Cancel a still-reconciling create before closing its owner; otherwise a missing // post-create snapshot is mistaken for an unknown outcome and retried after close. - cancelStructuredCodexLaunch(worktreeId, item.entityId) + cancelStructuredAgentLaunch(worktreeId, item.entityId) const target = getActiveRuntimeTarget({ activeRuntimeEnvironmentId: runtimeEnvironmentId }) @@ -198,7 +198,7 @@ export function useTabGroupTabCloseCommands({ worktreeId ) if (item.contentType === 'agent-session') { - cancelStructuredCodexLaunch(worktreeId, item.entityId) + cancelStructuredAgentLaunch(worktreeId, item.entityId) const target = getActiveRuntimeTarget({ activeRuntimeEnvironmentId: runtimeEnvironmentId }) diff --git a/src/renderer/src/components/terminal-pane/StructuredAgentSessionTerminalReturnButton.tsx b/src/renderer/src/components/terminal-pane/StructuredAgentSessionTerminalReturnButton.tsx new file mode 100644 index 00000000000..bba08bd914c --- /dev/null +++ b/src/renderer/src/components/terminal-pane/StructuredAgentSessionTerminalReturnButton.tsx @@ -0,0 +1,25 @@ +import { Button } from '@/components/ui/button' +import { translate } from '@/i18n/i18n' + +export function StructuredAgentSessionTerminalReturnButton(props: { + enabled: boolean + onReturn?: () => void +}): React.JSX.Element | null { + if (!props.enabled) { + return null + } + return ( + + ) +} diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx index 1b5e94b4a11..259037afea9 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx @@ -41,6 +41,31 @@ export function TerminalPaneNativeChatPortal({ return null } + const contextMenuActions = { + onSplitRight: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitRight), + onSplitDown: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitDown), + canEqualizePaneSizes: managedPanes.length > 1 && expandedPaneId === null, + onEqualizePaneSizes: () => contextMenu.runForPane(chatPane.id, contextMenu.onEqualizePaneSizes), + canExpandPane: managedPanes.length > 1, + isPaneExpanded: expandedPaneId === chatPane.id, + onToggleExpand: () => contextMenu.runForPane(chatPane.id, contextMenu.onToggleExpand), + canContinueAgentSessionInNewSession: canContinueAgentSessionInNewSession( + resolveAgentForLeaf(chatPane.leafId) + ), + onContinueAgentSessionInNewSession: () => + contextMenu.runForPane(chatPane.id, contextMenu.onContinueAgentSessionInNewSession), + onForkAgentSession: () => + void contextMenu.runForPane(chatPane.id, contextMenu.onForkAgentSession), + onSetTitle: () => contextMenu.runForPane(chatPane.id, contextMenu.onSetTitle), + onCopyTerminalId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyTerminalId), + onCopyPaneId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyPaneId), + canCopyAgentSessionId: chatPaneSessionId !== null, + onCopyAgentSessionId: () => + void contextMenu.runForPane(chatPane.id, contextMenu.onCopyAgentSessionId), + canClosePane: managedPanes.length > 1, + onClosePane: () => contextMenu.runForPane(chatPane.id, contextMenu.onClosePane) + } + return createPortal(
{structuredSessionId && structuredChatAgent ? ( @@ -52,6 +77,7 @@ export function TerminalPaneNativeChatPortal({ isVisible={isRendererVisible} target={structuredChatTarget} allowFileUriLinks + contextMenuActions={contextMenuActions} orchestrationDispatchStatus={chatPaneDispatchStatus} /> ) : ( @@ -65,32 +91,7 @@ export function TerminalPaneNativeChatPortal({ ownsTabWideLaunchDraft={chatPaneOwnsTabWideLaunchDraft} onSwitchToTerminal={switchNativeChatToTerminal} readTerminalScreen={readNativeChatTerminalScreen} - contextMenuActions={{ - onSplitRight: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitRight), - onSplitDown: () => contextMenu.runForPane(chatPane.id, contextMenu.onSplitDown), - canEqualizePaneSizes: managedPanes.length > 1 && expandedPaneId === null, - onEqualizePaneSizes: () => - contextMenu.runForPane(chatPane.id, contextMenu.onEqualizePaneSizes), - canExpandPane: managedPanes.length > 1, - isPaneExpanded: expandedPaneId === chatPane.id, - onToggleExpand: () => contextMenu.runForPane(chatPane.id, contextMenu.onToggleExpand), - canContinueAgentSessionInNewSession: canContinueAgentSessionInNewSession( - resolveAgentForLeaf(chatPane.leafId) - ), - onContinueAgentSessionInNewSession: () => - contextMenu.runForPane(chatPane.id, contextMenu.onContinueAgentSessionInNewSession), - onForkAgentSession: () => - void contextMenu.runForPane(chatPane.id, contextMenu.onForkAgentSession), - onSetTitle: () => contextMenu.runForPane(chatPane.id, contextMenu.onSetTitle), - onCopyTerminalId: () => - void contextMenu.runForPane(chatPane.id, contextMenu.onCopyTerminalId), - onCopyPaneId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyPaneId), - canCopyAgentSessionId: chatPaneSessionId !== null, - onCopyAgentSessionId: () => - void contextMenu.runForPane(chatPane.id, contextMenu.onCopyAgentSessionId), - canClosePane: managedPanes.length > 1, - onClosePane: () => contextMenu.runForPane(chatPane.id, contextMenu.onClosePane) - }} + contextMenuActions={contextMenuActions} orchestrationDispatchStatus={chatPaneDispatchStatus} /> )} diff --git a/src/renderer/src/components/terminal-pane/agent-completion-stale-evidence-backoff.test.ts b/src/renderer/src/components/terminal-pane/agent-completion-stale-evidence-backoff.test.ts new file mode 100644 index 00000000000..82714edce6e --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-completion-stale-evidence-backoff.test.ts @@ -0,0 +1,136 @@ +// The second consumer of the capture budget, at the point where a user feels it. +// +// A whole-machine `ps` costs seconds on a large or loaded host: 2.5-9.0s on an idle 2,002-process +// laptop, 4.0-18.6s at load 46. Publishing that as a truthful-but-late `live` record does not help +// this pane. `admitRemoteForegroundEvidence` refuses it, every refusal increments +// `consecutiveInspectionErrors`, and the poll scheduler's backoff then stretches the cadence to its +// 10s floor -- so agent-completion detection degrades on exactly the hosts where a capture is +// slowest, which are the hosts where agents take longest to finish. +// +// Giving up on the capture and publishing a prompt `unverifiable` instead costs one poll and +// nothing else: the record is admitted, so no error is counted. +import { describe, expect, it, vi } from 'vitest' +import { handleAgentCompletionInspectionResult } from './agent-completion-inspection-result' +import type { RemoteInspectionState } from './agent-completion-inspection-result' +import type { ProcessMonitorState } from './agent-completion-process-types' +import type { AgentCompletionCoordinatorOptions } from './agent-completion-coordinator-types' +import type { RuntimeTerminalProcessInspection } from '@/runtime/runtime-terminal-inspection' +import { toAppSshPtyId } from '../../../../shared/ssh-pty-id' +import { REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS } from '../../../../shared/remote-foreground-evidence-admission' + +const SSH_PTY_ID = toAppSshPtyId('target-1', 'pty-1') +const INCARNATION = 'inc-1' + +function liveRecord(capturedAgeMs: number): RuntimeTerminalProcessInspection { + return { + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'live', + processName: 'claude', + authorityGeneration: 'gen-1', + observationEpoch: 1, + capturedAgeMs, + ptyId: 'pty-1', + ptyIncarnationId: INCARNATION, + fence: { + platform: 'posix', + shellPid: 10, + shellStartTime: '100', + tty: '/dev/pts/2', + foregroundPgid: 11, + process: { pid: 11, startTime: '101' } + } + } + } +} + +/** What both relay call sites publish when the capture misses its budget. */ +function unreadableTableRecord(): RuntimeTerminalProcessInspection { + return { + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'unverifiable', + reason: 'process_table_unreadable', + authorityGeneration: 'gen-1', + observationEpoch: 1, + capturedAgeMs: 0, + ptyId: 'pty-1', + ptyIncarnationId: INCARNATION + } + } +} + +function inspect(result: RuntimeTerminalProcessInspection, roundTripMs = 20): ProcessMonitorState { + const state: ProcessMonitorState = { + disposed: false, + inspectionInFlight: false, + inspectionGeneration: 0, + consecutiveInspectionErrors: 0, + pollTrackingStarted: true, + pollTimer: null, + pollTimerTier: null, + lastPaneActivityAt: null, + hasAgentRunEvidence: false, + pendingProcessExitAgent: null, + lastForegroundAgent: null, + processSession: 1 + } + const remoteInspection: RemoteInspectionState = { + authorityGeneration: null, + observationEpoch: -1, + bindingKey: null, + knownAuthorityGenerations: new Set() + } + const started = performance.now() + vi.spyOn(performance, 'now').mockReturnValue(started + roundTripMs) + handleAgentCompletionInspectionResult({ + result, + requestStartedAtMonotonic: started, + options: { + paneKey: 'tab-1:leaf-1', + getPtyId: () => SSH_PTY_ID, + getSettings: () => null, + isRemotePtyId: () => true, + getExpectedIncarnationId: () => INCARNATION + } as unknown as AgentCompletionCoordinatorOptions, + state, + identityScope: {} as never, + clearAgentRunEvidence: vi.fn(), + hasPendingHookDone: () => false, + hasPendingCodexAttention: () => false, + scheduleNextPoll: vi.fn(), + handleRecognizedProcess: vi.fn(), + dispatchCompletion: vi.fn(), + remoteInspection + }) + vi.restoreAllMocks() + return state +} + +describe('agent completion polling under a host capture it cannot use', () => { + it('counts no error for the prompt unverifiable a capture over budget produces', () => { + expect(inspect(unreadableTableRecord()).consecutiveInspectionErrors).toBe(0) + }) + + it('counts an error for the late live record the same capture would have produced', () => { + // One of the measured captures. Every poll refusing this way is what drives the cadence to + // its 10s backoff floor and stops completion detection for the pane. + expect(inspect(liveRecord(6_140)).consecutiveInspectionErrors).toBe(1) + }) + + it.each([ + ['at the ceiling', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS, 0], + ['one step past it', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1, 1] + ])('counts %s as %s errors', (_label, capturedAgeMs, errors) => { + expect(inspect(liveRecord(capturedAgeMs), 0).consecutiveInspectionErrors).toBe(errors) + }) + + it('admits a capture that lands inside the evidence budget', () => { + // The reason the budget is 1,200ms rather than lower: a capture inside it must still clear the + // 2,000ms ceiling once its duration is counted once instead of twice. Under the double count + // this same record was refused, because 1,200 + a 1,300ms round trip read as 2,500. + expect(inspect(liveRecord(1_200), 1_300).consecutiveInspectionErrors).toBe(0) + }) +}) diff --git a/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts b/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts index 7c95bd7e46c..508ab978101 100644 --- a/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts +++ b/src/renderer/src/components/terminal-pane/codex-backfill-error-detector.ts @@ -9,6 +9,8 @@ const ANSI_ESCAPE_PATTERN = // eslint-disable-next-line no-control-regex -- terminal escape sequences contain control bytes /\u001b(?:\[[0-9;?]*[ -/]*[@-~]|\][^\u0007\u001b]*(?:\u0007|\u001b\\)?)/g const DETECTOR_BUFFER_MAX_CHARS = 4096 +// Why a regex over toLowerCase(): the case-folded copy allocated the whole 4KB carry on every chunk. +const CODEX_BACKFILL_TIMEOUT_PATTERN = new RegExp(CODEX_BACKFILL_TIMEOUT_SIGNATURE, 'i') export type CodexBackfillErrorDetector = { observe(chunk: string): string | null } @@ -23,7 +25,7 @@ export function createCodexBackfillErrorDetector(): CodexBackfillErrorDetector { } const normalized = (tail + chunk).replace(ANSI_ESCAPE_PATTERN, '').replace(/\r/g, '') tail = normalized.slice(-DETECTOR_BUFFER_MAX_CHARS) - if (!tail.toLowerCase().includes(CODEX_BACKFILL_TIMEOUT_SIGNATURE)) { + if (!CODEX_BACKFILL_TIMEOUT_PATTERN.test(tail)) { return null } armed = false diff --git a/src/renderer/src/components/terminal-pane/git-bash-console-capacity.test.ts b/src/renderer/src/components/terminal-pane/git-bash-console-capacity.test.ts index 27067b16828..9b6258efa6d 100644 --- a/src/renderer/src/components/terminal-pane/git-bash-console-capacity.test.ts +++ b/src/renderer/src/components/terminal-pane/git-bash-console-capacity.test.ts @@ -11,6 +11,29 @@ describe('Git Bash console capacity detection', () => { expect(detector.detected()).toBe(true) }) + it('detects the marker at every chunk boundary, in any case', () => { + const noisyPrefix = 'x'.repeat(200) + const stream = `${noisyPrefix}Console device allocation failure - TOO MANY CONSOLES In Use, Max Consoles Is 128\r\n` + + for (let split = 0; split <= stream.length; split += 1) { + const detector = createGitBashConsoleCapacityDetector() + detector.observe(stream.slice(0, split)) + detector.observe(stream.slice(split)) + expect(detector.detected(), `split at ${split}`).toBe(true) + } + }) + + it('still matches when the carry is rebuilt by chunks that skip the fast path', () => { + const detector = createGitBashConsoleCapacityDetector() + + // None of these chunks contain the marker's final character, so each takes the fast path. + detector.observe('too many consoles in use, ') + detector.observe('max consoles is ') + detector.observe('128') + + expect(detector.detected()).toBe(true) + }) + it('ignores unrelated shell failures', () => { const detector = createGitBashConsoleCapacityDetector() detector.observe('bash: command not found') diff --git a/src/renderer/src/components/terminal-pane/git-bash-console-capacity.ts b/src/renderer/src/components/terminal-pane/git-bash-console-capacity.ts index 6f70a8b2a2e..4753c817ca8 100644 --- a/src/renderer/src/components/terminal-pane/git-bash-console-capacity.ts +++ b/src/renderer/src/components/terminal-pane/git-bash-console-capacity.ts @@ -1,4 +1,8 @@ const GIT_BASH_CONSOLE_CAPACITY_MARKER = 'too many consoles in use, max consoles is 128' +const CARRY_LENGTH = GIT_BASH_CONSOLE_CAPACITY_MARKER.length - 1 +// Why this char: a match that was not already found must end inside the new chunk, so the chunk has +// to contain the marker's final character. It is a digit, so the test needs no case folding. +const MARKER_FINAL_CHAR = GIT_BASH_CONSOLE_CAPACITY_MARKER.at(-1) as string export type GitBashConsoleCapacityDetector = { observe: (data: string) => void @@ -14,9 +18,17 @@ export function createGitBashConsoleCapacityDetector(): GitBashConsoleCapacityDe if (matched || data.length === 0) { return } + if (!data.includes(MARKER_FINAL_CHAR)) { + // Fold only the carry instead of copying the whole chunk: this runs per PTY chunk per pane. + tail = + data.length >= CARRY_LENGTH + ? data.slice(-CARRY_LENGTH).toLowerCase() + : (tail + data).slice(-CARRY_LENGTH).toLowerCase() + return + } const candidate = (tail + data).toLowerCase() matched = candidate.includes(GIT_BASH_CONSOLE_CAPACITY_MARKER) - tail = candidate.slice(-(GIT_BASH_CONSOLE_CAPACITY_MARKER.length - 1)) + tail = candidate.slice(-CARRY_LENGTH) }, detected: () => matched } diff --git a/src/renderer/src/components/terminal-pane/terminal-paste-executor-default-yield.test.ts b/src/renderer/src/components/terminal-pane/terminal-paste-executor-default-yield.test.ts new file mode 100644 index 00000000000..4adc98a585b --- /dev/null +++ b/src/renderer/src/components/terminal-pane/terminal-paste-executor-default-yield.test.ts @@ -0,0 +1,101 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' + +// Why: the executor must yield through the shared helper by default, not a local setTimeout(0) +// that Chromium clamps once nested. Mocking the module pins the default's identity without +// depending on the helper's Vitest-only setTimeout fallback. +const { events, yieldToEventLoop } = vi.hoisted(() => { + const events: string[] = [] + return { + events, + yieldToEventLoop: vi.fn(async () => { + events.push('yield') + }) + } +}) + +vi.mock('../../../../shared/event-loop-yield', () => ({ yieldToEventLoop })) + +import { planTerminalPaste, type TerminalPasteTarget } from './terminal-paste-coordinator' +import { executeTerminalPastePlan } from './terminal-paste-executor' + +const target: TerminalPasteTarget = { + kind: 'terminal', + paneId: 1, + leafId: 'leaf-1', + ptyId: 'pty-default-yield', + runtime: { platform: 'linux', runtimeKey: 'local:linux', kind: 'local' } +} + +function chunkedPlan() { + return planTerminalPaste({ + text: '0123456789abcdef', + source: 'keyboard', + target, + maxDirectBytes: 4, + maxChunkBytes: 4 + }) +} + +afterEach(() => { + events.length = 0 + yieldToEventLoop.mockClear() + vi.restoreAllMocks() +}) + +describe('terminal paste executor default yield', () => { + it('yields after every chunk through the shared event-loop helper, never a timer', async () => { + const setTimeoutSpy = vi.spyOn(globalThis, 'setTimeout') + const writePty = vi.fn((chunk: string) => { + events.push(`write:${chunk}`) + return true + }) + const plan = chunkedPlan() + expect(plan.mode).toBe('chunked') + + const result = await executeTerminalPastePlan(plan, { + pasteText: vi.fn(), + writePty, + isTargetCurrent: () => true, + canContinue: () => true, + // Why: 0 disables the per-operation timeout timer, so any setTimeout call would be a yield. + operationTimeoutMs: 0 + }) + + expect(result.status).toBe('pasted') + expect(events).toEqual([ + 'write:0123', + 'yield', + 'write:4567', + 'yield', + 'write:89ab', + 'yield', + 'write:cdef', + 'yield' + ]) + expect(setTimeoutSpy).not.toHaveBeenCalled() + }) + + it('re-checks the target after a default yield before writing the next chunk', async () => { + let current = true + yieldToEventLoop.mockImplementationOnce(async () => { + events.push('yield') + current = false + }) + const writePty = vi.fn((chunk: string) => { + events.push(`write:${chunk}`) + return true + }) + + const result = await executeTerminalPastePlan(chunkedPlan(), { + pasteText: vi.fn(), + writePty, + isTargetCurrent: () => current, + canContinue: () => true, + operationTimeoutMs: 0 + }) + + expect(result.status).toBe('cancelled') + expect(result.reason).toBe('stale-target') + expect(events).toEqual(['write:0123', 'yield']) + }) +}) diff --git a/src/renderer/src/components/terminal-pane/terminal-paste-executor.ts b/src/renderer/src/components/terminal-pane/terminal-paste-executor.ts index 39a05f4dd24..747bc7b8bed 100644 --- a/src/renderer/src/components/terminal-pane/terminal-paste-executor.ts +++ b/src/renderer/src/components/terminal-pane/terminal-paste-executor.ts @@ -1,3 +1,4 @@ +import { yieldToEventLoop as yieldToEventLoopTask } from '../../../../shared/event-loop-yield' import { BRACKETED_PASTE_END, BRACKETED_PASTE_START } from './terminal-bracketed-paste' import { iterateTerminalPastePlanChunks } from './terminal-paste-chunks' import { createRedactedPasteExecutionDiagnostic } from './terminal-paste-diagnostics' @@ -40,7 +41,7 @@ async function executeTerminalPastePlanNow( writePty, isTargetCurrent, canContinue, - yieldToEventLoop = defaultYieldToEventLoop, + yieldToEventLoop = yieldToEventLoopTask, operationTimeoutMs = getTerminalPasteOperationTimeoutMs(plan), now = defaultNow }: ExecuteTerminalPastePlanArgs @@ -190,7 +191,3 @@ function result( function defaultNow(): number { return globalThis.performance?.now?.() ?? Date.now() } - -function defaultYieldToEventLoop(): Promise { - return new Promise((resolve) => setTimeout(resolve, 0)) -} diff --git a/src/renderer/src/components/terminal/terminal-tab-actions.ts b/src/renderer/src/components/terminal/terminal-tab-actions.ts index 49bc3d11fef..92eb85b49e5 100644 --- a/src/renderer/src/components/terminal/terminal-tab-actions.ts +++ b/src/renderer/src/components/terminal/terminal-tab-actions.ts @@ -157,7 +157,7 @@ export function closeTerminalTab( toast.error( translate( 'components.native-chat.structuredSessionCloseFailed', - 'Could not close this Codex chat' + 'Could not close this chat session' ), { description: translate( diff --git a/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts b/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts index 0783a7e5d23..32cddf92a17 100644 --- a/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts +++ b/src/renderer/src/components/use-worktree-jump-palette-quick-actions.ts @@ -56,6 +56,7 @@ export function useWorktreeJumpPaletteQuickActions({ sshConnectionStates, activeGroupIdByWorktree, groupsByWorktree, + unifiedTabsByWorktree, isLoading, settings, runtimeStatusByEnvironmentId, @@ -129,6 +130,7 @@ export function useWorktreeJumpPaletteQuickActions({ void sshConnectionStates void activeGroupIdByWorktree void groupsByWorktree + void unifiedTabsByWorktree void isLoading void settings?.activeRuntimeEnvironmentId void runtimeStatusByEnvironmentId @@ -144,6 +146,7 @@ export function useWorktreeJumpPaletteQuickActions({ sshConnectionStates, activeGroupIdByWorktree, groupsByWorktree, + unifiedTabsByWorktree, isLoading, settings?.activeRuntimeEnvironmentId, runtimeStatusByEnvironmentId diff --git a/src/renderer/src/hooks/agent-hook-completion-store-sync.ts b/src/renderer/src/hooks/agent-hook-completion-store-sync.ts index fc47a20205a..12c7d9102d6 100644 --- a/src/renderer/src/hooks/agent-hook-completion-store-sync.ts +++ b/src/renderer/src/hooks/agent-hook-completion-store-sync.ts @@ -48,7 +48,10 @@ function terminalTabLivenessMatches( return false } - for (const [worktreeIndex, worktreeId] of currentWorktreeIds.entries()) { + // Indexed loops, not .entries(): this runs on every tab write including title frames, and the + // iterator allocated a [index, value] tuple per worktree and per tab in each changed bucket. + for (let worktreeIndex = 0; worktreeIndex < currentWorktreeIds.length; worktreeIndex += 1) { + const worktreeId = currentWorktreeIds[worktreeIndex] // Why: duplicate tab ids use first-worktree-wins lookup semantics, so a // worktree-key reorder is a liveness change even when every array is reused. if (previousWorktreeIds[worktreeIndex] !== worktreeId) { @@ -62,7 +65,8 @@ function terminalTabLivenessMatches( if (!currentTabs || !previousTabs || currentTabs.length !== previousTabs.length) { return false } - for (const [tabIndex, currentTab] of currentTabs.entries()) { + for (let tabIndex = 0; tabIndex < currentTabs.length; tabIndex += 1) { + const currentTab = currentTabs[tabIndex] visitTab?.() const previousTab = previousTabs[tabIndex] if ( diff --git a/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts b/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts index e968e499afa..707b9916c5c 100644 --- a/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts +++ b/src/renderer/src/hooks/composer-state/composer-submit-orchestration.ts @@ -140,27 +140,14 @@ export function useComposerSubmitOrchestration( workspaceSeedName: target.derivedComposerState.workspaceSeedName }) const multipleCreateReset = useMultipleCreateReset({ + handleClearSmartNameSelection: source.issueSourceActions.handleClearSmartNameSelection, lastAutoNameRef: target.asyncComposerState.lastAutoNameRef, nameInputRef: target.asyncComposerState.nameInputRef, setAgentPrompt: target.sourceContextState.setAgentPrompt, setAttachmentPaths: target.sourceContextState.setAttachmentPaths, - setBranchNameOverride: target.workspaceIdentityState.setBranchNameOverride, - setBranchNameOverridePreservesNameEdits: - target.workspaceIdentityState.setBranchNameOverridePreservesNameEdits, - setCompareBaseRef: target.workspaceIdentityState.setCompareBaseRef, setCreateError: target.asyncComposerState.setCreateError, - setForkPushWarning: target.workspaceIdentityState.setForkPushWarning, - setLinkedGitLabIssue: target.workspaceIdentityState.setLinkedGitLabIssue, - setLinkedGitLabMR: target.workspaceIdentityState.setLinkedGitLabMR, - setLinkedIssue: target.workspaceIdentityState.setLinkedIssue, - setLinkedPR: target.workspaceIdentityState.setLinkedPR, - setLinkedTaskSourceContext: target.sourceContextState.setLinkedTaskSourceContext, - setLinkedWorkItem: target.sourceContextState.setLinkedWorkItem, setName: target.sourceContextState.setName, - setNote: target.sourceContextState.setNote, - setPushTarget: target.workspaceIdentityState.setPushTarget, - setReuseSelectedBranch: target.workspaceIdentityState.setReuseSelectedBranch, - setStartFromResetHint: target.workspaceIdentityState.setStartFromResetHint + setNote: target.sourceContextState.setNote }) const quickSubmitSourcePreparation = useQuickSubmitSourcePreparation({ baseBranch: target.workspaceIdentityState.baseBranch, diff --git a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts index 2176a1863ed..77c94652a52 100644 --- a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts +++ b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.test.ts @@ -1,23 +1,23 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' const mocks = vi.hoisted(() => ({ - startStructuredCodexLaunch: vi.fn(), + startStructuredAgentLaunch: vi.fn(), activateStructuredAgentSessionById: vi.fn() })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - startStructuredCodexLaunch: mocks.startStructuredCodexLaunch + startStructuredAgentLaunch: mocks.startStructuredAgentLaunch })) vi.mock('@/lib/structured-agent-session-tab-activation', () => ({ activateStructuredAgentSessionById: mocks.activateStructuredAgentSessionById })) -vi.mock('@/lib/launch-structured-codex-session', () => ({ +vi.mock('@/lib/launch-structured-agent-session', () => ({ StructuredAgentSessionCreateRefusalError: class extends Error {} })) -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { settleFullCreationStructuredLaunch } from './full-creation-structured-launch' describe('settleFullCreationStructuredLaunch', () => { @@ -26,7 +26,7 @@ describe('settleFullCreationStructuredLaunch', () => { it('runs the legacy terminal fallback after a definitive refusal', async () => { const fallbackActivation = { primaryTabId: 'fallback-tab' } const onDefinitiveRefusal = vi.fn().mockResolvedValue(fallbackActivation) - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new StructuredAgentSessionCreateRefusalError('unsupported')), isVisibilityUnknown: () => false, claimDefinitiveRefusalFallback: (fallback: () => Promise) => @@ -54,7 +54,7 @@ describe('settleFullCreationStructuredLaunch', () => { it('reports an unknown outcome without starting a fallback terminal', async () => { const onDefinitiveRefusal = vi.fn() - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new Error('connection lost')), isVisibilityUnknown: () => true, claimDefinitiveRefusalFallback: vi.fn(() => Promise.resolve(false)) diff --git a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts index f352724d841..90ecb7c6722 100644 --- a/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts +++ b/src/renderer/src/hooks/composer-state/full-creation-structured-launch.ts @@ -1,7 +1,8 @@ +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import type { TuiAgent } from '../../../../shared/tui-agent' import type { ActivateAndRevealResult } from '@/lib/worktree-activation' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { activateStructuredAgentSessionById } from '@/lib/structured-agent-session-tab-activation' type Activation = ActivateAndRevealResult | false @@ -20,11 +21,11 @@ export async function settleFullCreationStructuredLaunch(args: { }> { let activation = args.initialActivation let structuredLaunchAccepted = args.structuredLaunch - if (!args.structuredLaunch || args.agent !== 'codex') { + if (!args.structuredLaunch || !isAgentSessionHandleProvider(args.agent)) { return { structuredLaunchAccepted, visibilityUnknown: false, activation } } - const launch = startStructuredCodexLaunch(args.worktreeId, { prompt: args.prompt }) + const launch = startStructuredAgentLaunch(args.worktreeId, args.agent, { prompt: args.prompt }) const refusalFallback = launch.claimDefinitiveRefusalFallback(async () => { structuredLaunchAccepted = false activation = await args.onDefinitiveRefusal() diff --git a/src/renderer/src/hooks/composer-state/multiple-create-reset.test.ts b/src/renderer/src/hooks/composer-state/multiple-create-reset.test.ts new file mode 100644 index 00000000000..910143cace8 --- /dev/null +++ b/src/renderer/src/hooks/composer-state/multiple-create-reset.test.ts @@ -0,0 +1,168 @@ +// @vitest-environment happy-dom + +import { useRef, useState } from 'react' +import { act, renderHook } from '@testing-library/react' +import { describe, expect, it, vi } from 'vitest' +import type { LinkedWorkItemSummary } from '@/lib/new-workspace' +import { useIssueSourceActions } from './issue-source-actions' +import { useMultipleCreateReset } from './multiple-create-reset' +import type { SmartGitHubPrStartPointSelection } from './source-selection-decisions' + +const sources: LinkedWorkItemSummary[] = [ + { + provider: 'github', + type: 'pr', + number: 42, + title: 'Fix checkout', + url: 'https://github.com/acme/app/pull/42' + }, + { + provider: 'github', + type: 'issue', + number: 43, + title: 'Fix checkout', + url: 'https://github.com/acme/app/issues/43' + }, + { + provider: 'gitlab', + type: 'mr', + number: 44, + title: 'Fix checkout', + url: 'https://gitlab.com/acme/app/-/merge_requests/44' + } +] + +function useSelectedSourceReset( + initialItem: LinkedWorkItemSummary | null, + isProjectGroupTarget = false, + initialBaseBranch: string | undefined = '1234567890abcdef1234567890abcdef12345678' +) { + const [linkedWorkItem, setLinkedWorkItem] = useState(initialItem) + const [baseBranch, setBaseBranch] = useState(initialBaseBranch) + const [name, setName] = useState('fix-checkout') + const [note, setNote] = useState('User note') + const lastAutoNameRef = useRef(name) + const branchAutoNameRef = useRef('fix-checkout') + const lastAutoNoteRef = useRef('Generated note') + const smartGitHubPrStartPointSelectionRef = useRef( + initialItem?.provider === 'github' && initialItem.type === 'pr' + ? { + repoId: 'repo-1', + item: { + ...initialItem, + type: 'pr', + id: 'pr-42', + repoId: 'repo-1', + state: 'open', + labels: [], + updatedAt: '2026-09-01T00:00:00Z', + author: null + } + } + : null + ) + const source = useIssueSourceActions({ + baseBranch, + branchAutoNameRef, + isProjectGroupTarget, + lastAutoNameRef, + lastAutoNoteRef, + linkedWorkItem, + name, + noteRef: useRef(note), + setBaseBranch, + setBranchNameOverride: vi.fn(), + setBranchNameOverridePreservesNameEdits: vi.fn(), + setCompareBaseRef: vi.fn(), + setForkPushWarning: vi.fn(), + setLinkedGitLabIssue: vi.fn(), + setLinkedGitLabMR: vi.fn(), + setLinkedIssue: vi.fn(), + setLinkedPR: vi.fn(), + setLinkedTaskSourceContext: vi.fn(), + setLinkedWorkItem, + setName, + setNote, + setPushTarget: vi.fn(), + setReuseEligibleBranch: vi.fn(), + setReuseSelectedBranch: vi.fn(), + setStartFromResetHint: vi.fn(), + smartGitHubPrStartPointSelectionRef + }) + const reset = useMultipleCreateReset({ + handleClearSmartNameSelection: source.handleClearSmartNameSelection, + lastAutoNameRef, + nameInputRef: useRef(null), + setAgentPrompt: vi.fn(), + setAttachmentPaths: vi.fn(), + setCreateError: vi.fn(), + setName, + setNote + }) + return { + ...reset, + selection: source.smartNameSelection, + linkedWorkItem, + baseBranch, + name, + note, + branchAutoNameRef, + smartGitHubPrStartPointSelectionRef + } +} + +describe('create more source reset', () => { + it.each(sources)( + 'clears $provider $type and its checkout source before the next create', + (item) => { + const { result } = renderHook(() => useSelectedSourceReset(item)) + expect(result.current.selection?.label).toContain('Fix checkout') + + if (item.provider === 'github' && item.type === 'pr') { + expect(result.current.smartGitHubPrStartPointSelectionRef.current).not.toBeNull() + } + + act(() => result.current.resetForNextCreate()) + + expect(result.current.smartGitHubPrStartPointSelectionRef.current).toBeNull() + expect(result.current.selection).toBeNull() + expect(result.current.linkedWorkItem).toBeNull() + expect(result.current.baseBranch).toBeUndefined() + expect(result.current.name).toBe('') + expect(result.current.note).toBe('') + expect(result.current.branchAutoNameRef.current).toBe('') + } + ) + + it.each(['linear', 'jira'] as const)('clears a %s task on a folder target', (provider) => { + const item: LinkedWorkItemSummary = { + provider, + type: 'issue', + number: 0, + title: 'Fix checkout', + url: + provider === 'linear' + ? 'https://linear.app/acme/issue/APP-45' + : 'https://acme.atlassian.net/browse/APP-45' + } + const { result } = renderHook(() => useSelectedSourceReset(item, true)) + expect(result.current.selection?.kind).toBe(provider) + + act(() => result.current.resetForNextCreate()) + + expect(result.current.selection).toBeNull() + expect(result.current.linkedWorkItem).toBeNull() + expect(result.current.name).toBe('') + expect(result.current.note).toBe('') + }) + + it('clears a plain branch selection before the next create', () => { + const { result } = renderHook(() => useSelectedSourceReset(null, false, 'feature/checkout')) + expect(result.current.selection).toEqual({ kind: 'branch', label: 'feature/checkout' }) + + act(() => result.current.resetForNextCreate()) + + expect(result.current.selection).toBeNull() + expect(result.current.baseBranch).toBeUndefined() + }) +}) diff --git a/src/renderer/src/hooks/composer-state/multiple-create-reset.ts b/src/renderer/src/hooks/composer-state/multiple-create-reset.ts index b1b77720934..e078d692717 100644 --- a/src/renderer/src/hooks/composer-state/multiple-create-reset.ts +++ b/src/renderer/src/hooks/composer-state/multiple-create-reset.ts @@ -2,96 +2,48 @@ import type { ComposerModel } from './composer-model' type MultipleCreateResetInput = Pick< ComposerModel, + | 'handleClearSmartNameSelection' | 'lastAutoNameRef' | 'nameInputRef' | 'setAgentPrompt' | 'setAttachmentPaths' - | 'setBranchNameOverride' - | 'setBranchNameOverridePreservesNameEdits' - | 'setCompareBaseRef' | 'setCreateError' - | 'setForkPushWarning' - | 'setLinkedGitLabIssue' - | 'setLinkedGitLabMR' - | 'setLinkedIssue' - | 'setLinkedPR' - | 'setLinkedTaskSourceContext' - | 'setLinkedWorkItem' | 'setName' | 'setNote' - | 'setPushTarget' - | 'setReuseSelectedBranch' - | 'setStartFromResetHint' > import { useCallback } from 'react' export function useMultipleCreateReset(input: MultipleCreateResetInput) { const { + handleClearSmartNameSelection, lastAutoNameRef, nameInputRef, setAgentPrompt, setAttachmentPaths, - setBranchNameOverride, - setBranchNameOverridePreservesNameEdits, - setCompareBaseRef, setCreateError, - setForkPushWarning, - setLinkedGitLabIssue, - setLinkedGitLabMR, - setLinkedIssue, - setLinkedPR, - setLinkedTaskSourceContext, - setLinkedWorkItem, setName, - setNote, - setPushTarget, - setReuseSelectedBranch, - setStartFromResetHint + setNote } = input const resetForNextCreate = useCallback(() => { - // Why: clear identity fields derived from a PR pick while retaining repo, base, agent, and group context for sequential creates. + // Clear the checkout source too, so a PR's resolved SHA cannot become the next selection. + handleClearSmartNameSelection() setName('') lastAutoNameRef.current = '' setAgentPrompt('') setNote('') setAttachmentPaths([]) - setLinkedWorkItem(null) - setLinkedTaskSourceContext(null) - setLinkedIssue('') - setLinkedPR(null) - setLinkedGitLabIssue(null) - setLinkedGitLabMR(null) - setBranchNameOverride(undefined) - setBranchNameOverridePreservesNameEdits(false) - setCompareBaseRef(undefined) - setPushTarget(undefined) - setReuseSelectedBranch(false) - setStartFromResetHint(null) - setForkPushWarning(null) setCreateError(null) requestAnimationFrame(() => nameInputRef.current?.focus()) }, [ + handleClearSmartNameSelection, lastAutoNameRef, nameInputRef, setAgentPrompt, setAttachmentPaths, - setBranchNameOverride, - setBranchNameOverridePreservesNameEdits, - setCompareBaseRef, setCreateError, - setForkPushWarning, - setLinkedGitLabIssue, - setLinkedGitLabMR, - setLinkedIssue, - setLinkedPR, - setLinkedTaskSourceContext, - setLinkedWorkItem, setName, - setNote, - setPushTarget, - setReuseSelectedBranch, - setStartFromResetHint + setNote ]) return { diff --git a/src/renderer/src/hooks/ipc-events/app-lifetime-ipc-bridge.ts b/src/renderer/src/hooks/ipc-events/app-lifetime-ipc-bridge.ts index 17a5357071d..fa683b7188f 100644 --- a/src/renderer/src/hooks/ipc-events/app-lifetime-ipc-bridge.ts +++ b/src/renderer/src/hooks/ipc-events/app-lifetime-ipc-bridge.ts @@ -12,6 +12,7 @@ import { createDirectSshBridgeRuntime } from './direct-ssh-bridge-runtime' import { registerDirectSshStateIpcBridge } from './direct-ssh-state-ipc-bridge' import { registerMobileAndTerminalCloseIpcBridge } from './mobile-terminal-close-ipc-bridge' import { registerMobileDriverIpcBridge } from './mobile-driver-ipc-bridge' +import { registerOrcaProfileAuthIpcBridge } from './orca-profile-auth-ipc-bridge' import { registerOsMarkdownFileOpenBridge } from './os-markdown-file-open-bridge' import { registerProjectCatalogIpcBridge } from './project-catalog-ipc-bridge' import { registerRateLimitIpcBridge } from './rate-limit-ipc-bridge' @@ -78,6 +79,7 @@ export function installAppLifetimeIpcEvents( remountTerminalTabsAwaitingHostHydration ) registerSettingsAndSidebarIpcBridge(unsubs) + registerOrcaProfileAuthIpcBridge(unsubs) registerWorkspaceShortcutIpcBridge(unsubs) registerOsMarkdownFileOpenBridge(unsubs) unsubs.push( diff --git a/src/renderer/src/hooks/ipc-events/orca-profile-auth-ipc-bridge.test.ts b/src/renderer/src/hooks/ipc-events/orca-profile-auth-ipc-bridge.test.ts new file mode 100644 index 00000000000..593ace5336d --- /dev/null +++ b/src/renderer/src/hooks/ipc-events/orca-profile-auth-ipc-bridge.test.ts @@ -0,0 +1,82 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { OrcaProfileAuthStatus } from '../../../../shared/orca-profiles' +import { createTestStore } from '../../store/slices/store-test-helpers' + +const { storeHolder } = vi.hoisted(() => ({ + storeHolder: { current: null as { getState: () => unknown } | null } +})) + +vi.mock('../../store', () => ({ + useAppStore: { getState: () => storeHolder.current?.getState() } +})) + +vi.mock('sonner', () => ({ + toast: { error: vi.fn(), info: vi.fn(), success: vi.fn(), warning: vi.fn() } +})) + +import { registerOrcaProfileAuthIpcBridge } from './orca-profile-auth-ipc-bridge' + +const connectedAuthStatus: OrcaProfileAuthStatus = { + activeProfileId: 'local-default', + configured: true, + state: 'connected', + persistence: 'encrypted' +} + +const reconnectRequiredAuthStatus: OrcaProfileAuthStatus = { + activeProfileId: 'local-default', + configured: true, + state: 'reconnect-required', + persistence: 'encrypted' +} + +describe('orca profile auth IPC bridge', () => { + let listener: (() => void) | null = null + const unsubscribe = vi.fn() + const authStatus = vi.fn() + + beforeEach(() => { + listener = null + unsubscribe.mockClear() + authStatus.mockReset() + vi.stubGlobal('window', { + api: { + orcaProfiles: { + authStatus, + onAuthStatusChanged: (callback: () => void) => { + listener = callback + return unsubscribe + } + } + } + }) + }) + + it('re-fetches auth status on the push, flipping connected to reconnect-required', async () => { + authStatus.mockResolvedValue(connectedAuthStatus) + const store = createTestStore() + storeHolder.current = store + await store.getState().fetchOrcaProfileAuthStatus() + expect(store.getState().orcaProfileAuthStatus).toEqual(connectedAuthStatus) + + const unsubs: (() => void)[] = [] + registerOrcaProfileAuthIpcBridge(unsubs) + authStatus.mockResolvedValue(reconnectRequiredAuthStatus) + listener?.() + await vi.waitFor(() => + expect(store.getState().orcaProfileAuthStatus).toEqual(reconnectRequiredAuthStatus) + ) + + unsubs.forEach((dispose) => dispose()) + expect(unsubscribe).toHaveBeenCalledTimes(1) + }) + + it('skips registration when the preload bridge does not expose the event', () => { + vi.stubGlobal('window', { api: { orcaProfiles: { authStatus } } }) + const unsubs: (() => void)[] = [] + + registerOrcaProfileAuthIpcBridge(unsubs) + + expect(unsubs).toHaveLength(0) + }) +}) diff --git a/src/renderer/src/hooks/ipc-events/orca-profile-auth-ipc-bridge.ts b/src/renderer/src/hooks/ipc-events/orca-profile-auth-ipc-bridge.ts new file mode 100644 index 00000000000..95397b64a62 --- /dev/null +++ b/src/renderer/src/hooks/ipc-events/orca-profile-auth-ipc-bridge.ts @@ -0,0 +1,14 @@ +import { useAppStore } from '../../store' + +/** Re-reads auth status when main clears a revoked cloud session behind the renderer's back. */ +export function registerOrcaProfileAuthIpcBridge(unsubs: (() => void)[]): void { + const subscribe = window.api.orcaProfiles?.onAuthStatusChanged + if (typeof subscribe !== 'function') { + return + } + unsubs.push( + subscribe(() => { + void useAppStore.getState().fetchOrcaProfileAuthStatus() + }) + ) +} diff --git a/src/renderer/src/hooks/use-orca-profile-auth-status-refresh.ts b/src/renderer/src/hooks/use-orca-profile-auth-status-refresh.ts new file mode 100644 index 00000000000..692ed9bcfc1 --- /dev/null +++ b/src/renderer/src/hooks/use-orca-profile-auth-status-refresh.ts @@ -0,0 +1,14 @@ +import { useEffect } from 'react' +import { useAppStore } from '../store' + +/** + * Re-reads the cloud auth status whenever a surface that renders it mounts. The + * store caches the startup value, so without this a session revoked since launch + * still reads as connected. The cached value stays rendered while the fetch runs. + */ +export function useOrcaProfileAuthStatusRefresh(): void { + const fetchAuthStatus = useAppStore((state) => state.fetchOrcaProfileAuthStatus) + useEffect(() => { + void fetchAuthStatus() + }, [fetchAuthStatus]) +} diff --git a/src/renderer/src/hooks/useIpcEvents-lifecycle.test.ts b/src/renderer/src/hooks/useIpcEvents-lifecycle.test.ts index 6e0e7238c05..5991c40c4de 100644 --- a/src/renderer/src/hooks/useIpcEvents-lifecycle.test.ts +++ b/src/renderer/src/hooks/useIpcEvents-lifecycle.test.ts @@ -19,6 +19,7 @@ const EXPECTED_DIRECT_CALLBACK_METHODS = [ 'emulator.onPaneFocus', 'gh.onPRRefreshEvent', 'keybindings.onChanged', + 'orcaProfiles.onAuthStatusChanged', 'pty.onExit', 'rateLimits.onUpdate', 'remoteWorkspace.onChanged', @@ -126,6 +127,7 @@ const EXPECTED_CALLBACK_REGISTRATION_SEQUENCE = [ 'ui.onToggleWorktreePalette', 'ui.onToggleFloatingTerminal', 'ui.onTerminalShortcutCaptured', + 'orcaProfiles.onAuthStatusChanged', 'ui.onOpenQuickOpen', 'ui.onToggleQuickCommandsMenu', 'ui.onOpenNewWorkspace', diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index d9fb5b898b0..f61042c3510 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -6231,6 +6231,11 @@ "projectOnly": "Added in this project only.", "useGlobalFor": "Use global for {{value0}}", "useGlobal": "Use global" + }, + "RepoScanUnavailableIndicator": { + "title": "Worktree scan failed for {{value0}}", + "retry": "Retry scan", + "retained": "Existing worktrees are kept until a scan succeeds. Click to retry." } }, "shared": { @@ -6964,8 +6969,8 @@ "defaultViewTerminal": "Terminal chat", "defaultViewNative": "Chat UI", "structuredTitle": "Use updated structured native chat", - "structuredCopy": "Opt in to the host-owned structured Codex runtime. Off keeps the existing terminal-backed chat path.", - "structuredScope": "Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.", + "structuredCopy": "Opt in to the host-owned structured chat runtime for Codex and Claude. Off keeps the existing terminal-backed chat path.", + "structuredScope": "Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.", "structuredToggleLabel": "Toggle updated structured native chat" }, "agentDashboard": { @@ -13771,7 +13776,9 @@ "useLan": "Use LAN", "retrying": "Retrying…", "retry": "Retry Relay", - "copyDiagnostics": "Copy diagnostics" + "copyDiagnostics": "Copy diagnostics", + "reconnectTitle": "Your Orca account session expired.", + "reconnectBody": "Sign in again to use Orca Relay, or use LAN to pair over Tailscale or the same Wi‑Fi." } }, "gitlab": { @@ -15297,8 +15304,18 @@ "removeWorktree": "remove worktree", "trashWorktree": "trash worktree", "addQuickCommand": "add quick command", - "newQuickCommand": "new quick command" - } + "newQuickCommand": "new quick command", + "splitChatRight": "split chat right", + "moveChatRight": "move chat right", + "chatPaneRight": "chat pane right", + "splitChatDown": "split chat down", + "moveChatDown": "move chat down", + "chatPaneBelow": "chat pane below" + }, + "splitChatRight": "Split Chat Right", + "splitChatRightDescription": "Open the active chat in a split pane to the right.", + "splitChatDown": "Split Chat Down", + "splitChatDownDescription": "Open the active chat in a split pane below." } }, "palette": { @@ -16950,8 +16967,8 @@ "responding": "Agent is responding", "working": "Working…", "thinking": "Thinking", - "workingFor": "Working for {{value0}} seconds", - "workedFor": "Worked for {{value0}} seconds", + "workingFor": "Working for {{value0}}", + "workedFor": "Worked for {{value0}}", "toggleDetails": "Toggle turn details" }, "jumpToLatest": "Jump to latest", @@ -17007,11 +17024,44 @@ "message": "Structured Chat blocks terminal prompts and sends. Orchestration messages remain queued; switch to Terminal, then check the Orca inbox with", "command": "orca orchestration check" }, - "structuredSessionCloseFailed": "Could not close this Codex chat", - "structuredSessionLaunchFailed": "Could not open Codex chat", + "structuredSessionCloseFailed": "Could not close this chat session", + "structuredSessionLaunchFailed": "Could not open {{value0}} chat", + "structuredSessionLaunchPending": "Starting {{value0}} chat…", "structuredSessionCloseFailedDescription": "The terminal stayed open so the provider remains recoverable.", "monitoringStatus": { "label": "Monitoring background tasks" + }, + "handoff": { + "stage": { + "finishingChat": "Finishing chat session…", + "finishingTerminal": "Finishing agent terminal…", + "openingTerminal": "Opening agent terminal…", + "resumingChat": "Resuming chat session…", + "verifyingTerminal": "Verifying agent terminal…", + "verifyingChat": "Verifying chat session…", + "recovering": "Recovering agent session…", + "manualRecovery": "Agent session needs recovery" + }, + "switchingOwner": "Switching session owner…", + "mode": { + "switching": "Switching", + "terminal": "Terminal", + "chat": "Chat" + }, + "switchingAfterTurn": "Switching after this turn", + "returningAfterTurn": "Returning after this turn", + "cancel": "Cancel", + "switchAfterTurn": "Switch after this turn", + "stopTurnAndSwitch": "Stop turn and switch", + "openAgentTui": "Open agent TUI", + "returnAfterTurn": "Return after this turn", + "returnToChat": "Return to chat", + "agentOpenOnHost": "Agent is open in terminal on {{value0}}.", + "agentOpen": "Agent is open in terminal.", + "exitTerminal": "Exit the agent terminal to continue in chat.", + "retryProof": "Retry proof", + "retry": "Retry", + "details": "Details" } }, "tab": { diff --git a/src/renderer/src/lib/agent-launch-routing.test.ts b/src/renderer/src/lib/agent-launch-routing.test.ts index a16ed1d5ff2..dd33a8357d9 100644 --- a/src/renderer/src/lib/agent-launch-routing.test.ts +++ b/src/renderer/src/lib/agent-launch-routing.test.ts @@ -27,6 +27,68 @@ function route(overrides: Partial[0]> } describe('resolveAgentLaunchRoute', () => { + it.each(['claude', 'codex'] as const)( + 'routes a supported local %s launch to structured native chat', + (agent) => { + expect(route({ agent })).toBe('structured-native-chat') + expect( + route({ agent, launchText: 'explain this change', promptDelivery: 'auto-submit' }) + ).toBe('structured-native-chat') + } + ) + + /** Boundary guard between this lane and the one that owns Windows Codex. Codex's win32 refusal is + * deliberate, so it is asserted against whatever currently lets Claude through rather than + * against one host answer — a future gate swap must not be able to flip Codex on quietly. */ + describe("Codex's Windows refusal", () => { + it('holds in the exact situation that routes Claude to structured', () => { + const onWindows = { platform: 'win32' } as const + expect(route({ ...onWindows, agent: 'claude' })).toBe('structured-native-chat') + expect(route({ ...onWindows, agent: 'codex' })).toBe('legacy-native-chat') + }) + + it('holds for every host capability set, including ones that carry extra gates', () => { + for (const hostCapabilities of [ + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, 'agent-session.structured.claude.v1'], + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, 'agent-session.structured.hold.v1'] + ]) { + expect(route({ agent: 'codex', platform: 'win32', hostCapabilities })).toBe( + 'legacy-native-chat' + ) + } + }) + + it('holds for prompted and folder-workspace launches too', () => { + expect( + route({ + agent: 'codex', + platform: 'win32', + launchText: 'go', + promptDelivery: 'auto-submit' + }) + ).toBe('legacy-native-chat') + expect(route({ agent: 'codex', platform: 'win32', workspaceKind: 'folder' })).toBe( + 'legacy-native-chat' + ) + }) + }) + + /** Pins Codex's whole platform answer, not just win32, so no platform silently changes here. */ + it.each([ + ['darwin', 'structured-native-chat'], + ['linux', 'structured-native-chat'], + ['win32', 'legacy-native-chat'] + ] as const)('leaves Codex routing on %s unchanged', (platform, expected) => { + expect(route({ agent: 'codex', platform })).toBe(expected) + }) + + /** Claude's Windows answer is not a client-side platform guess: the route lets it through and the + * executing host settles it with agentSession.createSupport at create time. */ + it('lets a Windows Claude launch reach the host-measured create support check', () => { + expect(route({ agent: 'claude', platform: 'win32' })).toBe('structured-native-chat') + }) + it('routes a supported local Codex launch to structured native chat', () => { expect(route()).toBe('structured-native-chat') expect(route({ launchText: 'explain this change', promptDelivery: 'auto-submit' })).toBe( @@ -52,7 +114,9 @@ describe('resolveAgentLaunchRoute', () => { it('fails closed for missing capability, unsupported providers, and explicit TUI options', () => { expect(route({ hostCapabilities: [] })).toBe('legacy-native-chat') - expect(route({ agent: 'claude' })).toBe('legacy-native-chat') + // openclaude and grok render native chat but have no structured adapter. + expect(route({ agent: 'openclaude' })).toBe('legacy-native-chat') + expect(route({ agent: 'grok' })).toBe('legacy-native-chat') expect(route({ requiresTuiLaunchCustomization: true })).toBe('legacy-native-chat') expect(route({ initialSessionOptions: { model: 'gpt-5.6-sol' } })).toBe('legacy-native-chat') }) @@ -71,9 +135,11 @@ describe('resolveAgentLaunchRoute', () => { } ) - it('keeps floating, Windows, WSL, and repair-required launches terminal-backed', () => { + it('keeps floating, WSL, and repair-required launches terminal-backed', () => { expect(route({ workspaceKind: 'floating' })).toBe('legacy-native-chat') - expect(route({ platform: 'win32' })).toBe('legacy-native-chat') + expect(route({ agent: 'claude', workspaceKind: 'floating', platform: 'win32' })).toBe( + 'legacy-native-chat' + ) expect( route({ projectRuntime: { diff --git a/src/renderer/src/lib/agent-launch-routing.ts b/src/renderer/src/lib/agent-launch-routing.ts index dd72cb29794..243788f8117 100644 --- a/src/renderer/src/lib/agent-launch-routing.ts +++ b/src/renderer/src/lib/agent-launch-routing.ts @@ -1,5 +1,6 @@ import type { GlobalSettings } from '../../../shared/global-settings-types' import type { ProjectExecutionRuntimeResolution } from '../../../shared/project-execution-runtime' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../shared/protocol-version' import type { TuiAgent } from '../../../shared/tui-agent' import { @@ -92,13 +93,16 @@ export function resolveAgentLaunchRoute(input: AgentLaunchRoutingInput): AgentLa input.initialSessionOptions && Object.keys(input.initialSessionOptions).length > 0 ) const structuredSupported = - input.agent === 'codex' && + isAgentSessionHandleProvider(input.agent) && input.promptDelivery !== 'draft' && input.workspaceKind !== 'floating' && input.requiresTuiLaunchCustomization !== true && !hasInitialSessionOptions && input.executionHostId === 'local' && - input.platform !== 'win32' && + // Codex's Windows refusal is deliberate and settled elsewhere, so it stays a client-side + // answer. Claude's is measured by the executing host at create time (agentSession.createSupport) + // because only that host knows whether it can read a provider child's start time. + (input.agent !== 'codex' || input.platform !== 'win32') && !runtimeRefused && input.hostCapabilities.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) diff --git a/src/renderer/src/lib/launch-agent-in-new-tab.ts b/src/renderer/src/lib/launch-agent-in-new-tab.ts index 5259a054bfe..118fcdbeda8 100644 --- a/src/renderer/src/lib/launch-agent-in-new-tab.ts +++ b/src/renderer/src/lib/launch-agent-in-new-tab.ts @@ -31,7 +31,8 @@ import type { LaunchSource } from '../../../shared/telemetry-events' import { getConnectionIdFromState } from '@/lib/connection-context' import { resolveInitialNativeChatSessionOptions } from '@/components/native-chat/native-chat-launch-session-options' import { seedNativeChatAppliedSessionOptions } from '@/components/native-chat/native-chat-session-option-cache' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import { hasExplicitTuiLaunchCustomization, hasExplicitTuiAgentArgs, @@ -226,8 +227,8 @@ function launchAgentInNewTabInternal( hasExplicitTuiLaunchCustomization(store.settings, agent), initialSessionOptions: startupPlan.sessionOptions }) - if (launchRoute === 'structured-native-chat' && agent === 'codex') { - const structuredLaunch = startStructuredCodexLaunch(worktreeId, { + if (launchRoute === 'structured-native-chat' && isAgentSessionHandleProvider(agent)) { + const structuredLaunch = startStructuredAgentLaunch(worktreeId, agent, { prompt: trimmedPrompt, ...(promptDelivery === 'submit-after-ready' ? { promptDelivery } : {}), onPromptDelivered diff --git a/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts b/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts index 35fdcc1d261..5753651c051 100644 --- a/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts +++ b/src/renderer/src/lib/launch-agent-structured-chat-guard.test.ts @@ -13,6 +13,8 @@ const mockLaunchStructuredCodexSession = vi.fn() const mockRefreshLocalStructuredSessionTabs = vi.fn() const mockToastError = vi.fn() const mockCallStructuredAgentSession = vi.fn() +const STRUCTURED_HOST_CAPABILITIES = ['agent-session.structured.v1'] +let hostCapabilities: readonly string[] = STRUCTURED_HOST_CAPABILITIES function structuredLaunchIntent(worktreeId: string, sessionId = 'codex-session-1') { return { @@ -91,12 +93,12 @@ vi.mock('@/runtime/web-runtime-session', () => ({ isWebRuntimeSessionActive: vi.fn(() => false), isWebTerminalSurfaceTabId: vi.fn(() => false) })) -vi.mock('@/lib/launch-structured-codex-session', () => { +vi.mock('@/lib/launch-structured-agent-session', () => { class StructuredAgentSessionCreateRefusalError extends Error {} return { - createStructuredCodexSessionLaunchIntent: mockCreateStructuredCodexSessionLaunchIntent, + createStructuredAgentSessionLaunchIntent: mockCreateStructuredCodexSessionLaunchIntent, abandonStructuredAgentSessionLaunchIntent: mockAbandonStructuredAgentSessionLaunchIntent, - launchStructuredCodexSession: mockLaunchStructuredCodexSession, + launchStructuredAgentSession: mockLaunchStructuredCodexSession, StructuredAgentSessionCreateRefusalError } }) @@ -105,7 +107,7 @@ vi.mock('@/runtime/local-structured-session-tabs-sync', () => ({ LOCAL_STRUCTURED_SESSION_OWNER: 'local-structured-session' })) vi.mock('@/runtime/local-runtime-capabilities', () => ({ - readLocalRuntimeCapabilities: () => ['agent-session.structured.v1'] + readLocalRuntimeCapabilities: () => hostCapabilities })) vi.mock('@/lib/worktree-runtime-owner', () => ({ getExecutionHostIdForWorktree: () => @@ -149,6 +151,7 @@ describe('structured chat adoption guard on the launch path', () => { } ]) mockToastError.mockReset() + hostCapabilities = STRUCTURED_HOST_CAPABILITIES store.settings.openAgentTabsInChatByDefault = true }) @@ -164,7 +167,7 @@ describe('structured chat adoption guard on the launch path', () => { focusAfterMenuClose: 'structured-session' }) expect(shouldQueueTerminalFocusAfterMenuClose(result!)).toBe(false) - expect(mockCreateStructuredCodexSessionLaunchIntent).toHaveBeenCalledWith('wt-1') + expect(mockCreateStructuredCodexSessionLaunchIntent).toHaveBeenCalledWith('wt-1', 'codex') expect(mockLaunchStructuredCodexSession).toHaveBeenCalledWith( expect.objectContaining({ worktreeId: 'wt-1' }) ) @@ -172,6 +175,51 @@ describe('structured chat adoption guard on the launch path', () => { expect(mockWaitForAgentReady).not.toHaveBeenCalled() }) + it('takes the structured path for Claude, naming Claude as the create provider', async () => { + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + const result = launchAgentInNewTab({ agent: 'claude', worktreeId: 'wt-1' }) + + expect(result).toMatchObject({ tabId: null, focusAfterMenuClose: 'structured-session' }) + expect(mockCreateStructuredCodexSessionLaunchIntent).toHaveBeenCalledWith('wt-1', 'claude') + expect(mockCreateTab).not.toHaveBeenCalled() + }) + + it('keeps a native-chat agent with no structured adapter on the terminal-backed path', async () => { + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + launchAgentInNewTab({ agent: 'openclaude', worktreeId: 'wt-1' }) + + expect(mockCreateStructuredCodexSessionLaunchIntent).not.toHaveBeenCalled() + expect(mockCreateTab).toHaveBeenCalled() + }) + + it('fails a Claude launch closed to the terminal when the host declines create support', async () => { + const { StructuredAgentSessionCreateRefusalError } = + await import('./launch-structured-agent-session') + mockLaunchStructuredCodexSession.mockRejectedValueOnce( + new StructuredAgentSessionCreateRefusalError('structured_agent_session_unsupported') + ) + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + const result = launchAgentInNewTab({ agent: 'claude', worktreeId: 'wt-1' }) + + expect(result).toMatchObject({ tabId: null }) + await vi.waitFor(() => expect(mockCreateTab).toHaveBeenCalledOnce()) + expect(mockToastError).not.toHaveBeenCalled() + }) + + it('routes every structured launch through the shared host capability gate', async () => { + hostCapabilities = [] + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + + launchAgentInNewTab({ agent: 'claude', worktreeId: 'wt-1' }) + launchAgentInNewTab({ agent: 'codex', worktreeId: 'wt-1' }) + + expect(mockCreateStructuredCodexSessionLaunchIntent).not.toHaveBeenCalled() + expect(mockCreateTab).toHaveBeenCalledTimes(2) + }) + /** The toggle is hidden under Terminal chat but its persisted value survives, so the launch * path must re-check the default view rather than trust a stale opt-in. */ it('ignores a stale structured opt-in while the default view is Terminal chat', async () => { @@ -192,7 +240,7 @@ describe('structured chat adoption guard on the launch path', () => { it('falls back to the preserved terminal launch on a definitive refusal', async () => { const { StructuredAgentSessionCreateRefusalError } = - await import('./launch-structured-codex-session') + await import('./launch-structured-agent-session') mockLaunchStructuredCodexSession.mockRejectedValueOnce( new StructuredAgentSessionCreateRefusalError('provider unavailable') ) @@ -207,7 +255,7 @@ describe('structured chat adoption guard on the launch path', () => { it('reports prompt delivery from the definitive-refusal terminal fallback', async () => { const { StructuredAgentSessionCreateRefusalError } = - await import('./launch-structured-codex-session') + await import('./launch-structured-agent-session') mockLaunchStructuredCodexSession.mockRejectedValueOnce( new StructuredAgentSessionCreateRefusalError('provider unavailable') ) diff --git a/src/renderer/src/lib/launch-structured-agent-session.test.ts b/src/renderer/src/lib/launch-structured-agent-session.test.ts new file mode 100644 index 00000000000..5f46d5b60d2 --- /dev/null +++ b/src/renderer/src/lib/launch-structured-agent-session.test.ts @@ -0,0 +1,244 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' +import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' +import { + createStructuredAgentSessionLaunchIntent, + launchStructuredAgentSession, + StructuredAgentSessionCreateRefusalError +} from './launch-structured-agent-session' + +vi.mock('@/runtime/structured-agent-session-client', () => ({ + callStructuredAgentSession: vi.fn() +})) + +describe('structured agent session launch', () => { + beforeEach(() => { + vi.mocked(callStructuredAgentSession).mockReset() + }) + + it('creates a native session with a host-verifiable launch intent', async () => { + vi.mocked(callStructuredAgentSession).mockImplementation(async (_target, _method, params) => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-1', sequence: 0 }, + value: { + sessionId: (params as { envelope: { sessionId: string } }).envelope.sessionId, + fence: 1, + page: { + sessionId: 'session-1', + epoch: 'epoch-1', + direction: 'tail', + items: [], + removedItemIds: [], + submissions: [], + window: { + oldest: null, + newest: null, + nextCursor: { epoch: 'epoch-1', sequence: 0 } + }, + liveCursor: { epoch: 'epoch-1', sequence: 0 }, + hasOlder: false, + hasNewer: false + }, + unconfirmedClientMessageIds: [] + } + })) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'codex') + const receipt = await launchStructuredAgentSession(intent) + const params = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] as { + envelope: { sessionId: string; payloadFingerprint: string } + worktree: string + agent: 'codex' + } + + expect(receipt).toEqual({ + sessionId: expect.stringMatching(/^codex_[A-Za-z0-9_]{36}$/), + fence: 1 + }) + expect(callStructuredAgentSession).toHaveBeenCalledWith( + { kind: 'local' }, + 'agentSession.create', + expect.objectContaining({ worktree: 'id:workspace-1', agent: 'codex' }) + ) + expect(params.envelope.payloadFingerprint).toBe( + structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: params.envelope.sessionId, + fields: { worktree: 'id:workspace-1', agent: 'codex' } + }) + ) + expect(params).toBe(intent.params) + }) + + it('names Claude as the create provider and in the session id', () => { + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + expect(intent.sessionId).toMatch(/^claude_[A-Za-z0-9_]{36}$/) + expect(intent.params.agent).toBe('claude') + expect(intent.params.envelope.payloadFingerprint).toBe( + structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: intent.sessionId, + fields: { worktree: 'id:workspace-1', agent: 'claude' } + }) + ) + }) + + it('asks the executing host for create support before creating a Claude session', async () => { + vi.mocked(callStructuredAgentSession).mockImplementation(async (_target, method) => + method === 'agentSession.createSupport' + ? { supported: true } + : { ok: true, replayed: false, value: { sessionId: 'claude_1', fence: 1 } } + ) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + await launchStructuredAgentSession(intent) + + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create' + ]) + expect(callStructuredAgentSession).toHaveBeenNthCalledWith( + 1, + { kind: 'local' }, + 'agentSession.createSupport', + { worktree: 'id:workspace-1', agent: 'claude' } + ) + }) + + it('refuses a Claude launch the host says it cannot support, without creating', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ supported: false, reason: 'agent' }) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + + await expect(launchStructuredAgentSession(intent)).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport' + ]) + }) + + it('fails closed when the create support probe cannot be answered', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('runtime unreachable')) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + + await expect(launchStructuredAgentSession(intent)).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + }) + + /** A worktree is not resolvable for a beat after createWorktree resolves, so the probe fails with + * selector_not_found instead of answering. That is "not ready", not "no". */ + it('retries a probe the host cannot answer yet, then creates', async () => { + const notResolvableYet = Object.assign(new Error('selector_not_found'), { + code: 'selector_not_found' + }) + vi.mocked(callStructuredAgentSession) + .mockRejectedValueOnce(notResolvableYet) + .mockRejectedValueOnce(notResolvableYet) + .mockImplementation(async (_target, method) => + method === 'agentSession.createSupport' + ? { supported: true } + : { ok: true, replayed: false, value: { sessionId: 'claude_1', fence: 1 } } + ) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + await expect(launchStructuredAgentSession(intent)).resolves.toMatchObject({ + sessionId: 'claude_1' + }) + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.create' + ]) + }) + + it('refuses once the retry budget for an unresolvable selector is spent', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValue( + Object.assign(new Error('selector_not_found'), { code: 'selector_not_found' }) + ) + + const intent = createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + + await expect(launchStructuredAgentSession(intent)).rejects.toBeInstanceOf( + StructuredAgentSessionCreateRefusalError + ) + // Bounded: the first ask plus the retry delays, and never agentSession.create. + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.createSupport', + 'agentSession.createSupport' + ]) + }) + + it('does not retry a host that answered no, or an unrelated failure', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ supported: false, reason: 'wsl' }) + await expect( + launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + ) + ).rejects.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + + vi.mocked(callStructuredAgentSession).mockReset() + vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('runtime unreachable')) + await expect( + launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + ) + ).rejects.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + }) + + /** A message-wrapped token must not be confused with prose that merely mentions it. */ + it('does not retry a failure that only mentions the token in passing', async () => { + vi.mocked(callStructuredAgentSession).mockRejectedValue( + new Error('Access denied after a prior selector_not_found') + ) + + await expect( + launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'claude') + ) + ).rejects.toBeInstanceOf(StructuredAgentSessionCreateRefusalError) + expect(callStructuredAgentSession).toHaveBeenCalledOnce() + }) + + /** Codex's support answer is settled by the launch route and owned elsewhere; this pins that the + * Claude probe did not change Codex's wire traffic. */ + it('does not probe create support for Codex', async () => { + vi.mocked(callStructuredAgentSession).mockResolvedValue({ + ok: true, + replayed: false, + value: { sessionId: 'codex_1', fence: 1 } + }) + + await launchStructuredAgentSession( + createStructuredAgentSessionLaunchIntent('workspace-1', 'codex') + ) + + expect(vi.mocked(callStructuredAgentSession).mock.calls.map(([, method]) => method)).toEqual([ + 'agentSession.create' + ]) + }) + + it('replays the exact create envelope when an unknown outcome is retried', async () => { + const intent = createStructuredAgentSessionLaunchIntent('workspace-retry', 'codex') + vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('response lost')) + + await expect(launchStructuredAgentSession(intent)).rejects.toThrow('response lost') + await expect(launchStructuredAgentSession(intent)).rejects.toThrow('response lost') + + const first = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] + const second = vi.mocked(callStructuredAgentSession).mock.calls[1]?.[2] + expect(first).toBe(intent.params) + expect(second).toBe(first) + expect(intent.params.envelope.clientOperationId).toMatch(/^\d{13}-[0-9a-f]{32}$/) + }) +}) diff --git a/src/renderer/src/lib/launch-structured-agent-session.ts b/src/renderer/src/lib/launch-structured-agent-session.ts new file mode 100644 index 00000000000..0d12ca91d40 --- /dev/null +++ b/src/renderer/src/lib/launch-structured-agent-session.ts @@ -0,0 +1,159 @@ +import type { AgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' +import type { + AgentSessionAttachResult, + AgentSessionMutationEnvelope, + AgentSessionMutationResult +} from '../../../shared/agent-session-wire' +import { + createStructuredAgentSessionOperationId, + structuredAgentSessionPayloadFingerprint +} from '../../../shared/structured-agent-session-mutation' +import { hasRuntimeRpcErrorCode } from '../../../shared/runtime-rpc-error-code' +import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' +import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' +import { useAppStore } from '@/store' +import { + clearWebSessionFocusIntentIfMatches, + recordWebSessionFocusIntent, + resolveWebSessionVisibleTabId +} from '@/runtime/web-session-focus-intent' +import { LOCAL_STRUCTURED_SESSION_OWNER } from '@/runtime/local-structured-session-tabs-sync' + +type StructuredAgentSessionCreateParams = { + envelope: AgentSessionMutationEnvelope + worktree: string + agent: AgentSessionHandleProvider +} + +export type StructuredAgentSessionLaunchIntent = { + sessionId: string + worktreeId: string + agent: AgentSessionHandleProvider + params: StructuredAgentSessionCreateParams +} + +export class StructuredAgentSessionCreateRefusalError extends Error {} + +export function createStructuredAgentSessionLaunchIntent( + worktreeId: string, + agent: AgentSessionHandleProvider +): StructuredAgentSessionLaunchIntent { + const sessionId = `${agent}_${crypto.randomUUID().replaceAll('-', '_')}` + const fields = { worktree: toRuntimeWorktreeSelector(worktreeId), agent } + const state = useAppStore.getState() + recordWebSessionFocusIntent( + { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, + worktreeId, + `agent-session:${sessionId}`, + undefined, + resolveWebSessionVisibleTabId(state, worktreeId) + ) + return { + sessionId, + worktreeId, + agent, + params: { + envelope: { + sessionId, + clientOperationId: createStructuredAgentSessionOperationId(() => crypto.randomUUID()), + expectedRuntimeFence: null, + payloadFingerprint: structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId, + fields + }) + }, + ...fields + } + } +} + +export function abandonStructuredAgentSessionLaunchIntent( + intent: StructuredAgentSessionLaunchIntent +): void { + clearWebSessionFocusIntentIfMatches( + { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, + intent.worktreeId, + `agent-session:${intent.sessionId}` + ) +} + +/** The host answers a worktree selector it cannot resolve yet with this rather than a verdict. */ +const SELECTOR_NOT_RESOLVABLE_CODE = 'selector_not_found' + +/** + * A worktree is not resolvable for a beat after `createWorktree` resolves, so a probe fired + * immediately after creation fails instead of answering. Measured window: under ~250ms. These + * delays cover it with margin and bound the wait when the selector is genuinely absent. + */ +const CREATE_SUPPORT_RETRY_DELAYS_MS: readonly number[] = [50, 150, 300] + +function delay(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)) +} + +/** + * Whether the executing host supports creating this session — retrying only while the host cannot + * yet resolve the worktree. + * + * "Could not answer" and "answered no" are different states and only the second is a verdict. + * Collapsing them sends a launch to the terminal because a selector was a beat late, which is + * indistinguishable to the user from the gate refusing them. The retry is narrowed to that one + * transient code so every other failure still refuses on the first ask. + */ +async function hostSupportsCreate(intent: StructuredAgentSessionLaunchIntent): Promise { + for (let attempt = 0; ; attempt += 1) { + try { + const support = await callStructuredAgentSession<{ supported: boolean; reason?: string }>( + { kind: 'local' }, + 'agentSession.createSupport', + { worktree: intent.params.worktree, agent: intent.agent } + ) + return support.supported === true + } catch (error) { + const retryDelayMs = CREATE_SUPPORT_RETRY_DELAYS_MS[attempt] + if ( + retryDelayMs === undefined || + !hasRuntimeRpcErrorCode(error, SELECTOR_NOT_RESOLVABLE_CODE) + ) { + // An unanswered probe is still not a yes. + return false + } + await delay(retryDelayMs) + } + } +} + +/** + * Only the host that will execute the session can answer whether it supports creating one there — + * on Windows that means reading the provider child's process start time, which a client cannot + * observe. + * + * Codex is absent on purpose: its answer is settled by the launch route and owned elsewhere, so + * probing here would change Codex's wire traffic. Note that this early return is also why the + * unresolvable-selector race above has never been able to refuse a Codex launch — the race is + * identical for Codex, nothing asks. Whoever gives Codex a probe inherits it. + */ +async function requireHostCreateSupport(intent: StructuredAgentSessionLaunchIntent): Promise { + if (intent.agent !== 'claude') { + return + } + if (!(await hostSupportsCreate(intent))) { + abandonStructuredAgentSessionLaunchIntent(intent) + throw new StructuredAgentSessionCreateRefusalError('structured_agent_session_unsupported') + } +} + +export async function launchStructuredAgentSession( + intent: StructuredAgentSessionLaunchIntent +): Promise> { + await requireHostCreateSupport(intent) + const result = await callStructuredAgentSession< + AgentSessionMutationResult + >({ kind: 'local' }, 'agentSession.create', intent.params) + if (!result.ok) { + abandonStructuredAgentSessionLaunchIntent(intent) + throw new StructuredAgentSessionCreateRefusalError(result.refusal.message) + } + return { sessionId: result.value.sessionId, fence: result.value.fence } +} diff --git a/src/renderer/src/lib/launch-structured-codex-session.test.ts b/src/renderer/src/lib/launch-structured-codex-session.test.ts deleted file mode 100644 index ea498c81719..00000000000 --- a/src/renderer/src/lib/launch-structured-codex-session.test.ts +++ /dev/null @@ -1,87 +0,0 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' -import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' -import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' -import { - createStructuredCodexSessionLaunchIntent, - launchStructuredCodexSession -} from './launch-structured-codex-session' - -vi.mock('@/runtime/structured-agent-session-client', () => ({ - callStructuredAgentSession: vi.fn() -})) - -describe('structured Codex launch', () => { - beforeEach(() => { - vi.mocked(callStructuredAgentSession).mockReset() - }) - - it('creates a native session with a host-verifiable launch intent', async () => { - vi.mocked(callStructuredAgentSession).mockImplementation(async (_target, _method, params) => ({ - ok: true, - replayed: false, - fence: 1, - cursor: { epoch: 'epoch-1', sequence: 0 }, - value: { - sessionId: (params as { envelope: { sessionId: string } }).envelope.sessionId, - fence: 1, - page: { - sessionId: 'session-1', - epoch: 'epoch-1', - direction: 'tail', - items: [], - removedItemIds: [], - submissions: [], - window: { - oldest: null, - newest: null, - nextCursor: { epoch: 'epoch-1', sequence: 0 } - }, - liveCursor: { epoch: 'epoch-1', sequence: 0 }, - hasOlder: false, - hasNewer: false - }, - unconfirmedClientMessageIds: [] - } - })) - - const intent = createStructuredCodexSessionLaunchIntent('workspace-1') - const receipt = await launchStructuredCodexSession(intent) - const params = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] as { - envelope: { sessionId: string; payloadFingerprint: string } - worktree: string - agent: 'codex' - } - - expect(receipt).toEqual({ - sessionId: expect.stringMatching(/^codex_[A-Za-z0-9_]{36}$/), - fence: 1 - }) - expect(callStructuredAgentSession).toHaveBeenCalledWith( - { kind: 'local' }, - 'agentSession.create', - expect.objectContaining({ worktree: 'id:workspace-1', agent: 'codex' }) - ) - expect(params.envelope.payloadFingerprint).toBe( - structuredAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: params.envelope.sessionId, - fields: { worktree: 'id:workspace-1', agent: 'codex' } - }) - ) - expect(params).toBe(intent.params) - }) - - it('replays the exact create envelope when an unknown outcome is retried', async () => { - const intent = createStructuredCodexSessionLaunchIntent('workspace-retry') - vi.mocked(callStructuredAgentSession).mockRejectedValue(new Error('response lost')) - - await expect(launchStructuredCodexSession(intent)).rejects.toThrow('response lost') - await expect(launchStructuredCodexSession(intent)).rejects.toThrow('response lost') - - const first = vi.mocked(callStructuredAgentSession).mock.calls[0]?.[2] - const second = vi.mocked(callStructuredAgentSession).mock.calls[1]?.[2] - expect(first).toBe(intent.params) - expect(second).toBe(first) - expect(intent.params.envelope.clientOperationId).toMatch(/^\d{13}-[0-9a-f]{32}$/) - }) -}) diff --git a/src/renderer/src/lib/launch-structured-codex-session.ts b/src/renderer/src/lib/launch-structured-codex-session.ts deleted file mode 100644 index 32a3feb19b4..00000000000 --- a/src/renderer/src/lib/launch-structured-codex-session.ts +++ /dev/null @@ -1,87 +0,0 @@ -import type { - AgentSessionAttachResult, - AgentSessionMutationEnvelope, - AgentSessionMutationResult -} from '../../../shared/agent-session-wire' -import { - createStructuredAgentSessionOperationId, - structuredAgentSessionPayloadFingerprint -} from '../../../shared/structured-agent-session-mutation' -import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' -import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' -import { useAppStore } from '@/store' -import { - clearWebSessionFocusIntentIfMatches, - recordWebSessionFocusIntent, - resolveWebSessionVisibleTabId -} from '@/runtime/web-session-focus-intent' -import { LOCAL_STRUCTURED_SESSION_OWNER } from '@/runtime/local-structured-session-tabs-sync' - -type StructuredAgentSessionCreateParams = { - envelope: AgentSessionMutationEnvelope - worktree: string - agent: 'codex' -} - -export type StructuredAgentSessionLaunchIntent = { - sessionId: string - worktreeId: string - params: StructuredAgentSessionCreateParams -} - -export class StructuredAgentSessionCreateRefusalError extends Error {} - -export function createStructuredCodexSessionLaunchIntent( - worktreeId: string -): StructuredAgentSessionLaunchIntent { - const sessionId = `codex_${crypto.randomUUID().replaceAll('-', '_')}` - const fields = { worktree: toRuntimeWorktreeSelector(worktreeId), agent: 'codex' as const } - const state = useAppStore.getState() - recordWebSessionFocusIntent( - { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, - worktreeId, - `agent-session:${sessionId}`, - undefined, - resolveWebSessionVisibleTabId(state, worktreeId) - ) - return { - sessionId, - worktreeId, - params: { - envelope: { - sessionId, - clientOperationId: createStructuredAgentSessionOperationId(() => crypto.randomUUID()), - expectedRuntimeFence: null, - payloadFingerprint: structuredAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId, - fields - }) - }, - ...fields - } - } -} - -export function abandonStructuredAgentSessionLaunchIntent( - intent: StructuredAgentSessionLaunchIntent -): void { - clearWebSessionFocusIntentIfMatches( - { environmentId: LOCAL_STRUCTURED_SESSION_OWNER }, - intent.worktreeId, - `agent-session:${intent.sessionId}` - ) -} - -export async function launchStructuredCodexSession( - intent: StructuredAgentSessionLaunchIntent -): Promise> { - const result = await callStructuredAgentSession< - AgentSessionMutationResult - >({ kind: 'local' }, 'agentSession.create', intent.params) - if (!result.ok) { - abandonStructuredAgentSessionLaunchIntent(intent) - throw new StructuredAgentSessionCreateRefusalError(result.refusal.message) - } - return { sessionId: result.value.sessionId, fence: result.value.fence } -} diff --git a/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts b/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts index f0e9cf5b783..7d3e6522bc8 100644 --- a/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts +++ b/src/renderer/src/lib/launch-work-item-direct-agent-routing.test.ts @@ -1,13 +1,13 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' const mocks = vi.hoisted(() => ({ - startStructuredCodexLaunch: vi.fn(), + startStructuredAgentLaunch: vi.fn(), activateAndRevealWorktree: vi.fn(), preflightAgentTrust: vi.fn() })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - startStructuredCodexLaunch: mocks.startStructuredCodexLaunch + startStructuredAgentLaunch: mocks.startStructuredAgentLaunch })) vi.mock('@/lib/worktree-activation', () => ({ @@ -18,7 +18,7 @@ vi.mock('@/lib/agent-trust-preflight', () => ({ preflightAgentTrust: mocks.preflightAgentTrust })) -vi.mock('@/lib/launch-structured-codex-session', () => ({ +vi.mock('@/lib/launch-structured-agent-session', () => ({ StructuredAgentSessionCreateRefusalError: class extends Error {} })) @@ -26,7 +26,7 @@ vi.mock('@/lib/native-chat-transcript-readability', () => ({ isNativeChatTranscriptLocalReadable: vi.fn(() => true) })) -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { settleDirectWorkItemStructuredLaunch } from './launch-work-item-direct-agent-routing' const baseArgs = { @@ -47,7 +47,7 @@ describe('settleDirectWorkItemStructuredLaunch', () => { it('runs the legacy terminal fallback after a definitive refusal', async () => { mocks.activateAndRevealWorktree.mockReturnValue({ primaryTabId: 'fallback-tab' }) - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new StructuredAgentSessionCreateRefusalError('unsupported')), isVisibilityUnknown: () => false, claimDefinitiveRefusalFallback: (fallback: () => Promise) => @@ -65,7 +65,7 @@ describe('settleDirectWorkItemStructuredLaunch', () => { }) it('reports an unknown outcome without starting a fallback terminal', async () => { - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ launchResult: Promise.reject(new Error('connection lost')), isVisibilityUnknown: () => true, claimDefinitiveRefusalFallback: vi.fn(() => Promise.resolve(false)) diff --git a/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts b/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts index 43f6a6e700d..7d77b8d0d18 100644 --- a/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts +++ b/src/renderer/src/lib/launch-work-item-direct-agent-routing.ts @@ -1,3 +1,4 @@ +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import type { TuiAgent } from '../../../shared/tui-agent' import type { AgentStartupPlan } from '@/lib/tui-agent-startup' import type { LaunchSource } from '../../../shared/telemetry-events' @@ -9,8 +10,8 @@ import { buildDirectWorkItemAgentStartupPlan, buildDirectWorkItemStartupOpts } from '@/lib/launch-work-item-direct-agent' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { isNativeChatTranscriptLocalReadable } from '@/lib/native-chat-transcript-readability' import { resolveSourceControlLaunchPlatform } from '@/lib/source-control-launch-platform' import { preflightAgentTrust } from '@/lib/agent-trust-preflight' @@ -127,11 +128,11 @@ export async function settleDirectWorkItemStructuredLaunch(args: { primaryTabId: string | null }> { let { structuredLaunch, primaryTabId } = args - if (!structuredLaunch || args.agent !== 'codex') { + if (!structuredLaunch || !isAgentSessionHandleProvider(args.agent)) { return { completed: false, structuredLaunch, visibilityUnknown: false, primaryTabId } } - const launch = startStructuredCodexLaunch(args.worktreeId, { + const launch = startStructuredAgentLaunch(args.worktreeId, args.agent, { prompt: args.draftContent, ...(args.promptDelivery === 'submit-after-ready' ? { promptDelivery: args.promptDelivery } : {}) }) diff --git a/src/renderer/src/lib/structured-agent-session-launch-callers.ts b/src/renderer/src/lib/structured-agent-session-launch-callers.ts index 7097b085cba..db9715c0c1a 100644 --- a/src/renderer/src/lib/structured-agent-session-launch-callers.ts +++ b/src/renderer/src/lib/structured-agent-session-launch-callers.ts @@ -1,6 +1,6 @@ -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { - settleStructuredCodexLaunchPrompt, + settleStructuredAgentLaunchPrompt, type StructuredPromptDeliveryResult } from '@/lib/structured-agent-session-launch-prompt' import type { StructuredAgentSessionOutboxEntry } from '../../../shared/structured-agent-session-outbox' @@ -10,7 +10,7 @@ export type StructuredRefusalFallback = () => | StructuredPromptDeliveryResult | Promise -export type StructuredCodexLaunchOptions = { +export type StructuredAgentLaunchOptions = { prompt?: string promptDelivery?: 'auto-submit' | 'submit-after-ready' onPromptDelivered?: () => void @@ -137,7 +137,7 @@ function trackPromptDelivery( export function addStructuredLaunchCaller(args: { group: StructuredLaunchCallerGroup launchResult: Promise<{ sessionId: string; fence: number }> - options: StructuredCodexLaunchOptions + options: StructuredAgentLaunchOptions stagedEntry: StructuredAgentSessionOutboxEntry | null }): StructuredLaunchCaller { const fallback = Promise.withResolvers() @@ -156,7 +156,7 @@ export function addStructuredLaunchCaller(args: { } } args.group.entries.add(caller) - const promptDeliveryResult = settleStructuredCodexLaunchPrompt({ + const promptDeliveryResult = settleStructuredAgentLaunchPrompt({ launchResult: args.launchResult, options: args.options, stagedEntry: args.stagedEntry diff --git a/src/renderer/src/lib/structured-agent-session-launch-prompt.ts b/src/renderer/src/lib/structured-agent-session-launch-prompt.ts index 513d3055c73..d52105057cd 100644 --- a/src/renderer/src/lib/structured-agent-session-launch-prompt.ts +++ b/src/renderer/src/lib/structured-agent-session-launch-prompt.ts @@ -78,7 +78,7 @@ async function dispatchStructuredLaunchPrompt( } } -export function settleStructuredCodexLaunchPrompt(args: { +export function settleStructuredAgentLaunchPrompt(args: { launchResult: Promise options: StructuredLaunchPromptOptions stagedEntry: StructuredAgentSessionOutboxEntry | null diff --git a/src/renderer/src/lib/structured-agent-session-launch-recovery.ts b/src/renderer/src/lib/structured-agent-session-launch-recovery.ts index 1c4347281d1..1af33421652 100644 --- a/src/renderer/src/lib/structured-agent-session-launch-recovery.ts +++ b/src/renderer/src/lib/structured-agent-session-launch-recovery.ts @@ -1,18 +1,18 @@ import type { AgentSessionHistoryResult } from '../../../shared/agent-session-wire' import { - launchStructuredCodexSession, + launchStructuredAgentSession, StructuredAgentSessionCreateRefusalError, type StructuredAgentSessionLaunchIntent -} from '@/lib/launch-structured-codex-session' +} from '@/lib/launch-structured-agent-session' import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' import { refreshLocalStructuredSessionTabs } from '@/runtime/local-structured-session-tabs-sync' import { useAppStore } from '@/store' -export type StructuredCodexLaunchReceipt = { sessionId: string; fence: number } +export type StructuredAgentLaunchReceipt = { sessionId: string; fence: number } export type StructuredLaunchRecoveryState = { intent: StructuredAgentSessionLaunchIntent - promise: Promise + promise: Promise visibilityUnknown: boolean cancelled: boolean onVisibilityChanged?: () => void @@ -64,7 +64,7 @@ function hasAdoptedStructuredSession(intent: StructuredAgentSessionLaunchIntent) async function recoverPublishedSessionReceipt( state: StructuredLaunchRecoveryState -): Promise { +): Promise { await verifyPublishedSession(state) const history = await callStructuredAgentSession( { kind: 'local' }, @@ -82,10 +82,10 @@ async function recoverPublishedSessionReceipt( async function retrySameIntent( state: StructuredLaunchRecoveryState, priorError: unknown -): Promise { +): Promise { throwIfLaunchCancelled(state) try { - const receipt = await launchStructuredCodexSession(state.intent) + const receipt = await launchStructuredAgentSession(state.intent) throwIfLaunchCancelled(state) await verifyPublishedSession(state) return receipt @@ -111,11 +111,11 @@ async function retrySameIntent( export async function launchAndReconcile( state: StructuredLaunchRecoveryState -): Promise { +): Promise { throwIfLaunchCancelled(state) - let receipt: StructuredCodexLaunchReceipt + let receipt: StructuredAgentLaunchReceipt try { - receipt = await launchStructuredCodexSession(state.intent) + receipt = await launchStructuredAgentSession(state.intent) } catch (error) { if (state.cancelled) { throw new StructuredAgentSessionLaunchCancelledError() @@ -143,7 +143,7 @@ export async function launchAndReconcile( export async function reconcileUnknownLaunch( state: StructuredLaunchRecoveryState -): Promise { +): Promise { throwIfLaunchCancelled(state) state.visibilityUnknown = false state.onVisibilityChanged?.() diff --git a/src/renderer/src/lib/structured-agent-session-launch.test.ts b/src/renderer/src/lib/structured-agent-session-launch.test.ts index 812248e7a49..f86f2432b99 100644 --- a/src/renderer/src/lib/structured-agent-session-launch.test.ts +++ b/src/renderer/src/lib/structured-agent-session-launch.test.ts @@ -20,12 +20,12 @@ vi.mock('sonner', () => ({ } })) -vi.mock('@/lib/launch-structured-codex-session', () => { +vi.mock('@/lib/launch-structured-agent-session', () => { class StructuredAgentSessionCreateRefusalError extends Error {} return { - createStructuredCodexSessionLaunchIntent: mocks.createIntent, + createStructuredAgentSessionLaunchIntent: mocks.createIntent, abandonStructuredAgentSessionLaunchIntent: mocks.abandonIntent, - launchStructuredCodexSession: mocks.launch, + launchStructuredAgentSession: mocks.launch, StructuredAgentSessionCreateRefusalError } }) @@ -51,17 +51,25 @@ vi.mock('@/store', () => ({ })) vi.mock('@/i18n/i18n', () => ({ - translate: (_key: string, fallback: string) => fallback + translate: (_key: string, fallback: string, options?: { value0?: string }) => + fallback.replace('{{value0}}', options?.value0 ?? '') +})) + +vi.mock('@/lib/agent-catalog', () => ({ + getAgentCatalog: () => [ + { id: 'claude', label: 'Claude' }, + { id: 'codex', label: 'Codex' } + ] })) import { StructuredAgentSessionCreateRefusalError, type StructuredAgentSessionLaunchIntent -} from '@/lib/launch-structured-codex-session' +} from '@/lib/launch-structured-agent-session' import { refreshLocalStructuredSessionTabs } from '@/runtime/local-structured-session-tabs-sync' import { - cancelStructuredCodexLaunch, - startStructuredCodexLaunch + cancelStructuredAgentLaunch, + startStructuredAgentLaunch } from './structured-agent-session-launch' import { readOutbox } from '@/components/native-chat/structured-agent-session-outbox-storage' @@ -72,6 +80,7 @@ function launchIntent( return { worktreeId, sessionId, + agent: 'codex', params: { envelope: { sessionId, @@ -112,13 +121,16 @@ async function flushLaunchSettlement(): Promise { } } -describe('startStructuredCodexLaunch', () => { +describe('startStructuredAgentLaunch', () => { beforeEach(() => { vi.clearAllMocks() localStorage.clear() mocks.rendererTabs = {} mocks.listeners.clear() - mocks.createIntent.mockImplementation((worktreeId: string) => launchIntent(worktreeId)) + mocks.createIntent.mockImplementation((worktreeId: string, agent: 'claude' | 'codex') => { + const intent = launchIntent(worktreeId, `${agent}-session-${worktreeId}`) + return { ...intent, agent, params: { ...intent.params, agent } } + }) mocks.callStructuredAgentSession.mockResolvedValue({ ok: true, page: { fence: 1 } @@ -138,7 +150,7 @@ describe('startStructuredCodexLaunch', () => { value: { submission: { dispatchState: 'accepted' } } }) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.launch).toHaveBeenCalledOnce() @@ -147,6 +159,42 @@ describe('startStructuredCodexLaunch', () => { expect(toast.error).not.toHaveBeenCalled() }) + it('keeps a Claude and a Codex launch in the same worktree apart', async () => { + const worktreeId = 'wt-two-agents' + mocks.launch.mockImplementation(async (intent: StructuredAgentSessionLaunchIntent) => { + mocks.rendererTabs[worktreeId] = [ + ...(mocks.rendererTabs[worktreeId] ?? []), + { contentType: 'agent-session', entityId: intent.sessionId, worktreeId } + ] + return { sessionId: intent.sessionId, fence: 1 } + }) + + const claude = startStructuredAgentLaunch(worktreeId, 'claude') + const codex = startStructuredAgentLaunch(worktreeId, 'codex') + await flushLaunchSettlement() + + expect(mocks.createIntent).toHaveBeenNthCalledWith(1, worktreeId, 'claude') + expect(mocks.createIntent).toHaveBeenNthCalledWith(2, worktreeId, 'codex') + expect(mocks.launch).toHaveBeenCalledTimes(2) + expect(vi.mocked(mocks.launch).mock.calls.map(([intent]) => intent.params.agent)).toEqual([ + 'claude', + 'codex' + ]) + expect(claude.sessionId).not.toBe(codex.sessionId) + expect(toast.error).not.toHaveBeenCalled() + }) + + it('names the refused agent in the launch failure toast', async () => { + const worktreeId = 'wt-claude-toast' + mocks.launch.mockRejectedValue(new Error('boom')) + vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) + + startStructuredAgentLaunch(worktreeId, 'claude') + await flushLaunchSettlement() + + expect(toast.error).toHaveBeenCalledWith('Could not open Claude chat', expect.anything()) + }) + it('completes from the host-emitted projection without listing inventory', async () => { const worktreeId = 'wt-host-frame' const intent = launchIntent(worktreeId, 'session-host-frame') @@ -161,7 +209,7 @@ describe('startStructuredCodexLaunch', () => { return { sessionId: intent.sessionId, fence: 1 } }) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.launch).toHaveBeenCalledOnce() @@ -186,8 +234,8 @@ describe('startStructuredCodexLaunch', () => { value: { submission: { dispatchState: 'accepted' } } }) - startStructuredCodexLaunch(worktreeId) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') + startStructuredAgentLaunch(worktreeId, 'codex') expect(mocks.createIntent).toHaveBeenCalledOnce() expect(mocks.launch).toHaveBeenCalledOnce() @@ -213,8 +261,8 @@ describe('startStructuredCodexLaunch', () => { ok: true, value: { submission: { dispatchState: 'accepted' } } }) - startStructuredCodexLaunch(worktreeId) - const second = startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex') + const second = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) resolveLaunch({ sessionId: intent.sessionId, fence: 1 }) await expect(second.promptDeliveryResult).resolves.toEqual({ @@ -251,12 +299,12 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((resolve) => (resolveDelivery = resolve)) ) - startStructuredCodexLaunch(worktreeId) - const coalesced = startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex') + const coalesced = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) resolveLaunch({ sessionId: intent.sessionId, fence: 1 }) await vi.waitFor(() => expect(mocks.callStructuredAgentSession).toHaveBeenCalledOnce()) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') expect(mocks.createIntent).toHaveBeenCalledOnce() expect(mocks.launch).toHaveBeenCalledOnce() @@ -274,9 +322,9 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - startStructuredCodexLaunch(worktreeId, { prompt: 'first prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'first prompt' }) await flushLaunchSettlement() - startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -291,7 +339,7 @@ describe('startStructuredCodexLaunch', () => { publishedSnapshot(worktreeId, intent.sessionId) ]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -310,7 +358,7 @@ describe('startStructuredCodexLaunch', () => { .mockResolvedValueOnce([]) .mockResolvedValueOnce([publishedSnapshot(worktreeId, intent.sessionId)]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.launch).toHaveBeenCalledTimes(2) @@ -327,14 +375,14 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(toast.error).toHaveBeenCalledOnce() vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([ publishedSnapshot(worktreeId, intent.sessionId) ]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -357,7 +405,7 @@ describe('startStructuredCodexLaunch', () => { }) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValueOnce([]).mockResolvedValueOnce([]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledOnce() @@ -374,7 +422,7 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - const first = startStructuredCodexLaunch(worktreeId, { prompt: 'only once' }) + const first = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'only once' }) const firstFallbackResult = first.claimDefinitiveRefusalFallback(firstFallback) await expect(first.launchResult).rejects.toThrow('offline') expect(first.releaseCallerAfterUnknownOutcome()).toBe(true) @@ -382,7 +430,7 @@ describe('startStructuredCodexLaunch', () => { vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([ publishedSnapshot(worktreeId, intent.sessionId) ]) - const retry = startStructuredCodexLaunch(worktreeId) + const retry = startStructuredAgentLaunch(worktreeId, 'codex') await expect(retry.launchResult).resolves.toEqual({ sessionId: intent.sessionId, fence: 1 }) await expect(firstFallbackResult).resolves.toBe(false) @@ -408,7 +456,7 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValue(new Error('offline')) vi.mocked(refreshLocalStructuredSessionTabs).mockResolvedValue([]) - const first = startStructuredCodexLaunch(worktreeId) + const first = startStructuredAgentLaunch(worktreeId, 'codex') const firstFallbackResult = first.claimDefinitiveRefusalFallback(firstFallback) await expect(first.launchResult).rejects.toThrow('offline') expect(first.releaseCallerAfterUnknownOutcome()).toBe(true) @@ -416,7 +464,7 @@ describe('startStructuredCodexLaunch', () => { mocks.launch.mockRejectedValueOnce( new StructuredAgentSessionCreateRefusalError('structured launch disabled') ) - const retry = startStructuredCodexLaunch(worktreeId) + const retry = startStructuredAgentLaunch(worktreeId, 'codex') const retryFallbackResult = retry.claimDefinitiveRefusalFallback(retryFallback) await expect(retry.launchResult).rejects.toBeInstanceOf( @@ -440,9 +488,9 @@ describe('startStructuredCodexLaunch', () => { publishedSnapshot(worktreeId, second.sessionId) ]) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await flushLaunchSettlement() expect(mocks.createIntent).toHaveBeenCalledTimes(2) @@ -460,7 +508,7 @@ describe('startStructuredCodexLaunch', () => { throw new Error('storage unavailable') }) - const result = startStructuredCodexLaunch(worktreeId, { prompt: 'start this task' }) + const result = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'start this task' }) const fallbackResult = result.claimDefinitiveRefusalFallback(fallback) await expect(result.launchResult).rejects.toBeInstanceOf( @@ -481,8 +529,8 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((_resolve, reject) => (rejectLaunch = reject)) ) - const first = startStructuredCodexLaunch(worktreeId, { prompt: 'first prompt' }) - const second = startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + const first = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'first prompt' }) + const second = startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) const firstFallback = vi.fn().mockResolvedValue({ delivered: true, failureNotified: false @@ -525,9 +573,9 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((resolve) => (resolveRefresh = resolve)) ) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await vi.waitFor(() => expect(refreshLocalStructuredSessionTabs).toHaveBeenCalledOnce()) - expect(cancelStructuredCodexLaunch(worktreeId, intent.sessionId)).toBe(true) + expect(cancelStructuredAgentLaunch(worktreeId, intent.sessionId)).toBe(true) resolveRefresh([]) await flushLaunchSettlement() @@ -546,12 +594,12 @@ describe('startStructuredCodexLaunch', () => { () => new Promise((resolve) => (resolveRefresh = resolve)) ) - startStructuredCodexLaunch(worktreeId, { prompt: 'first prompt' }) - startStructuredCodexLaunch(worktreeId, { prompt: 'second prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'first prompt' }) + startStructuredAgentLaunch(worktreeId, 'codex', { prompt: 'second prompt' }) await vi.waitFor(() => expect(refreshLocalStructuredSessionTabs).toHaveBeenCalledOnce()) expect(readOutbox(intent.sessionId)).toHaveLength(2) - expect(cancelStructuredCodexLaunch(worktreeId, intent.sessionId)).toBe(true) + expect(cancelStructuredAgentLaunch(worktreeId, intent.sessionId)).toBe(true) expect(readOutbox(intent.sessionId)).toEqual([]) resolveRefresh([]) await flushLaunchSettlement() @@ -569,9 +617,9 @@ describe('startStructuredCodexLaunch', () => { .mockResolvedValueOnce([]) .mockImplementationOnce(() => new Promise((resolve) => (resolveRetryRefresh = resolve))) - startStructuredCodexLaunch(worktreeId) + startStructuredAgentLaunch(worktreeId, 'codex') await vi.waitFor(() => expect(refreshLocalStructuredSessionTabs).toHaveBeenCalledTimes(2)) - expect(cancelStructuredCodexLaunch(worktreeId, intent.sessionId)).toBe(true) + expect(cancelStructuredAgentLaunch(worktreeId, intent.sessionId)).toBe(true) resolveRetryRefresh([]) await flushLaunchSettlement() diff --git a/src/renderer/src/lib/structured-agent-session-launch.ts b/src/renderer/src/lib/structured-agent-session-launch.ts index e5e3083e8fd..3ad8b880a49 100644 --- a/src/renderer/src/lib/structured-agent-session-launch.ts +++ b/src/renderer/src/lib/structured-agent-session-launch.ts @@ -1,11 +1,13 @@ import { useSyncExternalStore } from 'react' import { toast } from 'sonner' +import type { AgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' +import { getAgentCatalog } from '@/lib/agent-catalog' import { translate } from '@/i18n/i18n' import { abandonStructuredAgentSessionLaunchIntent, - createStructuredCodexSessionLaunchIntent, + createStructuredAgentSessionLaunchIntent, StructuredAgentSessionCreateRefusalError -} from '@/lib/launch-structured-codex-session' +} from '@/lib/launch-structured-agent-session' import { discardStructuredAgentSessionLaunchOutbox, enqueueStructuredAgentSessionLaunchPrompt @@ -14,7 +16,7 @@ import { launchAndReconcile, reconcileUnknownLaunch, StructuredAgentSessionLaunchCancelledError, - type StructuredCodexLaunchReceipt, + type StructuredAgentLaunchReceipt, type StructuredLaunchRecoveryState } from '@/lib/structured-agent-session-launch-recovery' import type { StructuredPromptDeliveryResult } from '@/lib/structured-agent-session-launch-prompt' @@ -26,13 +28,13 @@ import { settleStructuredLaunchCallersWithFallback, settleStructuredLaunchCallersWithoutFallback, structuredLaunchCallersHavePendingWork, - type StructuredCodexLaunchOptions, + type StructuredAgentLaunchOptions, type StructuredLaunchCaller, type StructuredLaunchCallerGroup, type StructuredRefusalFallback } from '@/lib/structured-agent-session-launch-callers' -export type { StructuredCodexLaunchOptions, StructuredCodexLaunchReceipt } +export type { StructuredAgentLaunchOptions, StructuredAgentLaunchReceipt } type StructuredLaunchState = StructuredLaunchRecoveryState & { identity: string @@ -44,16 +46,20 @@ type StructuredLaunchStateResult = { caller: StructuredLaunchCaller } -export type StructuredCodexLaunchResult = { +export type StructuredAgentLaunchResult = { sessionId: string - launchResult: Promise + launchResult: Promise promptDeliveryResult?: Promise isVisibilityUnknown: () => boolean releaseCallerAfterUnknownOutcome: () => boolean claimDefinitiveRefusalFallback: (fallback: StructuredRefusalFallback) => Promise } -export type StructuredCodexLaunchStatus = 'idle' | 'pending' | 'unknown' +export type StructuredAgentLaunchStatus = 'idle' | 'pending' | 'unknown' + +function structuredAgentLabel(agent: AgentSessionHandleProvider): string { + return getAgentCatalog().find((entry) => entry.id === agent)?.label ?? agent +} const pendingStructuredLaunchesByIdentity = new Map() const structuredLaunchListeners = new Set<() => void>() @@ -64,29 +70,37 @@ function notifyStructuredLaunchListeners(): void { } } -export function subscribeStructuredCodexLaunchStatus(listener: () => void): () => void { +export function subscribeStructuredAgentLaunchStatus(listener: () => void): () => void { structuredLaunchListeners.add(listener) return () => structuredLaunchListeners.delete(listener) } -export function getStructuredCodexLaunchStatus(worktreeId: string): StructuredCodexLaunchStatus { - const state = pendingStructuredLaunchesByIdentity.get(worktreeId) +export function getStructuredAgentLaunchStatus( + worktreeId: string, + agent: AgentSessionHandleProvider +): StructuredAgentLaunchStatus { + const state = pendingStructuredLaunchesByIdentity.get(launchIdentity(worktreeId, agent)) if (!state) { return 'idle' } return state.visibilityUnknown ? 'unknown' : 'pending' } -export function useStructuredCodexLaunchStatus(worktreeId: string): StructuredCodexLaunchStatus { +export function useStructuredAgentLaunchStatus( + worktreeId: string, + agent: AgentSessionHandleProvider +): StructuredAgentLaunchStatus { return useSyncExternalStore( - subscribeStructuredCodexLaunchStatus, - () => getStructuredCodexLaunchStatus(worktreeId), + subscribeStructuredAgentLaunchStatus, + () => getStructuredAgentLaunchStatus(worktreeId, agent), () => 'idle' ) } -function launchIdentity(worktreeId: string): string { - return worktreeId +// Why keyed by agent too: one worktree can hold a Claude and a Codex launch at once, and a shared +// key would hand the second caller the first agent's intent. +function launchIdentity(worktreeId: string, agent: AgentSessionHandleProvider): string { + return `${agent}:${worktreeId}` } function cleanupLaunchState(state: StructuredLaunchState): void { @@ -114,7 +128,7 @@ function settleDefinitiveRefusalFallback(state: StructuredLaunchState): void { function trackLaunchSettlement( state: StructuredLaunchState, - promise: Promise + promise: Promise ): void { void promise.then( () => { @@ -155,18 +169,20 @@ function trackLaunchFailureToast(state: StructuredLaunchState): void { toast.error( translate( 'components.native-chat.structuredSessionLaunchFailed', - 'Could not open Codex chat' + 'Could not open {{value0}} chat', + { value0: structuredAgentLabel(state.intent.agent) } ), { description: error instanceof Error ? error.message : String(error) } ) }) } -function structuredCodexLaunchState( +function structuredAgentLaunchState( worktreeId: string, - options: StructuredCodexLaunchOptions + agent: AgentSessionHandleProvider, + options: StructuredAgentLaunchOptions ): StructuredLaunchStateResult { - const identity = launchIdentity(worktreeId) + const identity = launchIdentity(worktreeId, agent) const existing = pendingStructuredLaunchesByIdentity.get(identity) if (existing) { if (existing.visibilityUnknown) { @@ -192,7 +208,7 @@ function structuredCodexLaunchState( } } - const intent = createStructuredCodexSessionLaunchIntent(worktreeId) + const intent = createStructuredAgentSessionLaunchIntent(worktreeId, agent) const text = options.prompt?.trim() ?? '' const stagedPrompt = text ? enqueueStructuredAgentSessionLaunchPrompt(intent.sessionId, text) @@ -212,7 +228,7 @@ function structuredCodexLaunchState( text && !stagedPrompt ? Promise.reject( new StructuredAgentSessionCreateRefusalError( - 'Could not durably stage the Codex launch prompt.' + `Could not durably stage the ${structuredAgentLabel(agent)} launch prompt.` ) ) : launchAndReconcile(state) @@ -232,7 +248,7 @@ function structuredCodexLaunchState( } } -export function cancelStructuredCodexLaunch(worktreeId: string, sessionId: string): boolean { +export function cancelStructuredAgentLaunch(worktreeId: string, sessionId: string): boolean { const state = [...pendingStructuredLaunchesByIdentity.values()].find( (candidate) => candidate.intent.worktreeId === worktreeId && candidate.intent.sessionId === sessionId @@ -249,11 +265,12 @@ export function cancelStructuredCodexLaunch(worktreeId: string, sessionId: strin return true } -export function startStructuredCodexLaunch( +export function startStructuredAgentLaunch( worktreeId: string, - options: StructuredCodexLaunchOptions = {} -): StructuredCodexLaunchResult { - const { state, caller } = structuredCodexLaunchState(worktreeId, options) + agent: AgentSessionHandleProvider, + options: StructuredAgentLaunchOptions = {} +): StructuredAgentLaunchResult { + const { state, caller } = structuredAgentLaunchState(worktreeId, agent, options) return { sessionId: state.intent.sessionId, launchResult: state.promise, diff --git a/src/renderer/src/lib/worktree-creation-flow-execute.ts b/src/renderer/src/lib/worktree-creation-flow-execute.ts index b6c48da719c..9b6ba569b1e 100644 --- a/src/renderer/src/lib/worktree-creation-flow-execute.ts +++ b/src/renderer/src/lib/worktree-creation-flow-execute.ts @@ -13,6 +13,7 @@ import { formatWorkspaceCreateError, getWorkspaceCreateErrorToastMessage } from '@/lib/workspace-create-error-format' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import type { CreateWorktreeResult } from '../../../shared/worktree/create-types' import type { WorktreeCreationRequest } from '@/lib/pending-worktree-creation' import { createBrowserUuid } from '@/lib/browser-uuid' @@ -208,7 +209,7 @@ export async function executeWorktreeCreation( } let structuredLaunchAccepted = structuredLaunch - if (structuredLaunch && preparedRequest.agent === 'codex') { + if (structuredLaunch && isAgentSessionHandleProvider(preparedRequest.agent)) { const structuredSession = await launchStructuredWorktreeSession({ creationId, request: preparedRequest, diff --git a/src/renderer/src/lib/worktree-creation-structured-session.test.ts b/src/renderer/src/lib/worktree-creation-structured-session.test.ts index 0fc91c18393..c58fcd40dce 100644 --- a/src/renderer/src/lib/worktree-creation-structured-session.test.ts +++ b/src/renderer/src/lib/worktree-creation-structured-session.test.ts @@ -6,8 +6,8 @@ const mocks = vi.hoisted(() => ({ }, listener: null as ((state: { pendingWorktreeCreations: Record }) => void) | null, unsubscribe: vi.fn(), - startStructuredCodexLaunch: vi.fn(), - cancelStructuredCodexLaunch: vi.fn(), + startStructuredAgentLaunch: vi.fn(), + cancelStructuredAgentLaunch: vi.fn(), closeStructuredAgentSession: vi.fn(), callRuntimeRpc: vi.fn(), activateStructuredAgentSessionById: vi.fn() @@ -26,8 +26,8 @@ vi.mock('@/store', () => ({ })) vi.mock('@/lib/structured-agent-session-launch', () => ({ - startStructuredCodexLaunch: mocks.startStructuredCodexLaunch, - cancelStructuredCodexLaunch: mocks.cancelStructuredCodexLaunch + startStructuredAgentLaunch: mocks.startStructuredAgentLaunch, + cancelStructuredAgentLaunch: mocks.cancelStructuredAgentLaunch })) vi.mock('@/runtime/structured-agent-session-close', () => ({ @@ -58,7 +58,7 @@ vi.mock('@/lib/agent-trust-preflight', () => ({ preflightAgentTrust: vi.fn() })) -vi.mock('@/lib/launch-structured-codex-session', () => ({ +vi.mock('@/lib/launch-structured-agent-session', () => ({ StructuredAgentSessionCreateRefusalError: class extends Error {} })) @@ -78,7 +78,7 @@ describe('launchStructuredWorktreeSession', () => { const launchResult = new Promise<{ sessionId: string; fence: number }>((resolve) => { resolveLaunch = resolve }) - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ sessionId: 'session-1', launchResult, isVisibilityUnknown: () => false, @@ -117,7 +117,7 @@ describe('launchStructuredWorktreeSession', () => { activation: false, primaryTabId: null }) - expect(mocks.cancelStructuredCodexLaunch).toHaveBeenCalledWith('worktree-1', 'session-1') + expect(mocks.cancelStructuredAgentLaunch).toHaveBeenCalledWith('worktree-1', 'session-1') expect(mocks.closeStructuredAgentSession).toHaveBeenCalledWith({ kind: 'local' }, 'session-1') expect(mocks.callRuntimeRpc).toHaveBeenCalledWith({ kind: 'local' }, 'session.tabs.close', { worktree: { id: 'worktree-1' }, @@ -130,7 +130,7 @@ describe('launchStructuredWorktreeSession', () => { it('reports an unknown launch without claiming a visible surface', async () => { const releaseCallerAfterUnknownOutcome = vi.fn() - mocks.startStructuredCodexLaunch.mockReturnValue({ + mocks.startStructuredAgentLaunch.mockReturnValue({ sessionId: 'session-unknown', launchResult: Promise.reject(new Error('connection lost')), isVisibilityUnknown: () => true, diff --git a/src/renderer/src/lib/worktree-creation-structured-session.ts b/src/renderer/src/lib/worktree-creation-structured-session.ts index d163852e3ac..2e2cd07698a 100644 --- a/src/renderer/src/lib/worktree-creation-structured-session.ts +++ b/src/renderer/src/lib/worktree-creation-structured-session.ts @@ -2,10 +2,11 @@ import { useAppStore } from '@/store' import { ensureWorktreeHasInitialTerminal } from '@/lib/worktree-initial-terminal-seeding' import { activateAndRevealWorktree, type ActivateAndRevealResult } from '@/lib/worktree-activation' import { - cancelStructuredCodexLaunch, - startStructuredCodexLaunch + cancelStructuredAgentLaunch, + startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' +import { isAgentSessionHandleProvider } from '../../../shared/agent-session-provider-handle' import { activateStructuredAgentSessionById } from '@/lib/structured-agent-session-tab-activation' import { preflightAgentTrust } from '@/lib/agent-trust-preflight' import type { WorktreeCreationRequest } from '@/lib/pending-worktree-creation' @@ -48,15 +49,17 @@ export async function launchStructuredWorktreeSession(args: { let { activation, primaryTabId } = args let accepted = true let visibilityUnknown = false - if (args.request.agent !== 'codex') { + const agent = args.request.agent + if (!isAgentSessionHandleProvider(agent)) { return { accepted, cancelled: false, visibilityUnknown, activation, primaryTabId } } if (!useAppStore.getState().pendingWorktreeCreations[args.creationId]) { return { accepted, cancelled: true, visibilityUnknown, activation, primaryTabId } } - const launch = startStructuredCodexLaunch( + const launch = startStructuredAgentLaunch( args.worktreeId, + agent, args.recoverUnknownLaunch ? {} : { prompt: args.request.launchDraftPrompt ?? args.request.quickPrompt } @@ -67,7 +70,7 @@ export async function launchStructuredWorktreeSession(args: { return } cancelled = true - cancelStructuredCodexLaunch(args.worktreeId, launch.sessionId) + cancelStructuredAgentLaunch(args.worktreeId, launch.sessionId) } const unsubscribe = useAppStore.subscribe((state) => { if (!state.pendingWorktreeCreations[args.creationId]) { diff --git a/src/renderer/src/runtime/structured-agent-session-handoff-store.ts b/src/renderer/src/runtime/structured-agent-session-handoff-store.ts new file mode 100644 index 00000000000..8800f349802 --- /dev/null +++ b/src/renderer/src/runtime/structured-agent-session-handoff-store.ts @@ -0,0 +1,50 @@ +import { useSyncExternalStore } from 'react' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' + +export type TerminalStructuredHandoff = { + sessionId: string + fence: number + status: AgentSessionHandoffStatus +} + +const byTerminalTabId = new Map() +const listeners = new Set<() => void>() + +export function publishStructuredHandoff(input: TerminalStructuredHandoff): void { + for (const [tabId, current] of byTerminalTabId) { + if (current.sessionId === input.sessionId && tabId !== input.status.terminal?.tabId) { + byTerminalTabId.delete(tabId) + } + } + if (input.status.terminal) { + byTerminalTabId.set(input.status.terminal.tabId, input) + } + for (const listener of listeners) { + listener() + } +} + +export function clearStructuredHandoff(sessionId: string): void { + let changed = false + for (const [tabId, current] of byTerminalTabId) { + if (current.sessionId === sessionId) { + byTerminalTabId.delete(tabId) + changed = true + } + } + if (changed) { + for (const listener of listeners) { + listener() + } + } +} + +export function useTerminalStructuredHandoff(tabId: string): TerminalStructuredHandoff | null { + return useSyncExternalStore( + (listener) => { + listeners.add(listener) + return () => listeners.delete(listener) + }, + () => byTerminalTabId.get(tabId) ?? null + ) +} diff --git a/src/renderer/src/runtime/use-worktree-runtime-target.test.ts b/src/renderer/src/runtime/use-worktree-runtime-target.test.ts new file mode 100644 index 00000000000..6e59c6e035d --- /dev/null +++ b/src/renderer/src/runtime/use-worktree-runtime-target.test.ts @@ -0,0 +1,34 @@ +// @vitest-environment happy-dom + +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, expect, it } from 'vitest' +import { useAppStore } from '@/store' +import { useWorktreeRuntimeTarget } from './use-worktree-runtime-target' + +const initialState = useAppStore.getInitialState() + +afterEach(() => { + cleanup() + useAppStore.setState(initialState, true) +}) + +it('keeps the runtime target identity stable across unrelated store writes', () => { + let renders = 0 + const { result } = renderHook(() => { + renders += 1 + return useWorktreeRuntimeTarget('worktree-1') + }) + + const first = result.current + const rendersAfterMount = renders + + act(() => { + for (let index = 0; index < 50; index += 1) { + useAppStore.setState({ activeRepoId: `repo-${index}` }) + } + }) + + // A selector that built the target object inline re-rendered on every write. + expect(renders).toBe(rendersAfterMount) + expect(result.current).toBe(first) +}) diff --git a/src/renderer/src/store/repos/repo-add-actions.ts b/src/renderer/src/store/repos/repo-add-actions.ts index d9aa5d6d968..02d8440dc03 100644 --- a/src/renderer/src/store/repos/repo-add-actions.ts +++ b/src/renderer/src/store/repos/repo-add-actions.ts @@ -3,6 +3,7 @@ import { toast } from 'sonner' import type { AppState } from '../types' import type { Repo } from '../../../../shared/repo-types' import { isGitRepoKind } from '../../../../shared/repo-kind' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import { getRepoHostIdentity } from '../slices/repo-host-identity' import { callRuntimeRpc, getActiveRuntimeTarget } from '../../runtime/runtime-rpc-client' import { resolveDismissedOnboardingFolderAgentLaunch } from '@/lib/onboarding-folder-agent-startup' @@ -202,13 +203,16 @@ export function createRepoAddActions( ...(launch.startup ? { startup: launch.startup } : {}), ...(launch.route === 'structured-native-chat' ? { providesInitialSurface: true } : {}) }) - if (launch.route === 'structured-native-chat' && launch.agent === 'codex') { - const [{ startStructuredCodexLaunch }, { StructuredAgentSessionCreateRefusalError }] = + if ( + launch.route === 'structured-native-chat' && + isAgentSessionHandleProvider(launch.agent) + ) { + const [{ startStructuredAgentLaunch }, { StructuredAgentSessionCreateRefusalError }] = await Promise.all([ import('@/lib/structured-agent-session-launch'), - import('@/lib/launch-structured-codex-session') + import('@/lib/launch-structured-agent-session') ]) - const structured = startStructuredCodexLaunch(folderWorktree.id) + const structured = startStructuredAgentLaunch(folderWorktree.id, launch.agent) const fallback = structured.claimDefinitiveRefusalFallback(() => { activateAndRevealWorktree(folderWorktree.id, { sidebarRevealBehavior: 'auto', diff --git a/src/renderer/src/store/slices/agent-status-provider-session.test.ts b/src/renderer/src/store/slices/agent-status-provider-session.test.ts index f854517d94f..0d60d023956 100644 --- a/src/renderer/src/store/slices/agent-status-provider-session.test.ts +++ b/src/renderer/src/store/slices/agent-status-provider-session.test.ts @@ -16,6 +16,27 @@ function makePiCompatibleProviderSession(agent: 'pi' | 'omp' | 'prime-agent') { } describe('recordAgentProviderSession', () => { + it('does not capture a structured native owner for terminal resume on restart', () => { + const store = createTestStore() + const paneKey = 'structured-tab:leaf-1' + const providerSession = { key: 'session_id' as const, id: 'provider-session-uuid' } + + store + .getState() + .setAgentStatus( + paneKey, + { state: 'working', prompt: 'keep going', agentType: 'claude' }, + 'Claude Chat', + undefined, + { tabId: 'structured-tab', worktreeId: 'wt-1' }, + { providerSession, terminalResumeEligible: false } + ) + store.getState().captureAllSleepingAgentSessions('quit') + + expect(store.getState().agentStatusByPaneKey[paneKey]?.providerSession).toEqual(providerSession) + expect(store.getState().sleepingAgentSessionsByPaneKey[paneKey]).toBeUndefined() + }) + it('preserves the root session while a child permission hook moves Codex to waiting', () => { const store = createTestStore() const providerSession = { key: 'session_id' as const, id: 'root-session' } diff --git a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts index 073bd712f32..f8078a9925c 100644 --- a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts +++ b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts @@ -22,6 +22,7 @@ export function mergeDetectedWorktreesForHost( current.repoId === refreshed.repoId && current.authoritative === refreshed.authoritative && current.source === refreshed.source && + current.unavailableReason === refreshed.unavailableReason && current.worktrees === worktrees ) { return current diff --git a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts new file mode 100644 index 00000000000..da44f24c0d3 --- /dev/null +++ b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from 'vitest' +import { makeDetectedResult } from '../../worktrees-detected-listing-fixtures' +import { mergeDetectedWorktreesForHost } from './detected-worktree-host-merge' +import { areDetectedWorktreeResultsEqual } from './worktree-catalog-visibility' + +const failed = (unavailableReason?: string) => + makeDetectedResult('repo-1', [], { + authoritative: false, + source: 'metadata-fallback', + ...(unavailableReason ? { unavailableReason } : {}) + }) + +// Why: two failed scans differ only by cause; dropping that from equality would freeze the first +// reason on the header until the listing's rows or authority changed. +describe('detected listing unavailable reason', () => { + it('is part of listing equality', () => { + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed('distro gone'))).toBe(true) + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed('mount hung'))).toBe(false) + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed())).toBe(false) + }) + + it('survives the host merge when only the reason changed', () => { + const merged = mergeDetectedWorktreesForHost( + failed('distro gone'), + failed('mount hung'), + 'local' + ) + + expect(merged.unavailableReason).toBe('mount hung') + }) +}) diff --git a/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts b/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts index ce7a13c4b25..c6d3a9636f7 100644 --- a/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts +++ b/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts @@ -14,6 +14,7 @@ export function areDetectedWorktreeResultsEqual( current.repoId === next.repoId && current.authoritative === next.authoritative && current.source === next.source && + current.unavailableReason === next.unavailableReason && catalogRowsEqual(current.worktrees, next.worktrees) ) } diff --git a/src/renderer/src/web/preload-api/web-orca-profiles-api.ts b/src/renderer/src/web/preload-api/web-orca-profiles-api.ts index 0b552bf15c8..e59d805ce3d 100644 --- a/src/renderer/src/web/preload-api/web-orca-profiles-api.ts +++ b/src/renderer/src/web/preload-api/web-orca-profiles-api.ts @@ -3,6 +3,7 @@ import { DEFAULT_LOCAL_ORCA_PROFILE_ID, createDefaultLocalOrcaProfile } from '../../../../shared/orca-profiles' +import { noopUnsubscribe } from './web-storage' export function createWebOrcaProfilesApi(): Partial { const webOrcaProfileAuthStatus = () => @@ -22,6 +23,7 @@ export function createWebOrcaProfilesApi(): Partial { multiProfileUi: false }), authStatus: webOrcaProfileAuthStatus, + onAuthStatusChanged: () => noopUnsubscribe, createLocal: () => Promise.resolve({ activeProfileId: DEFAULT_LOCAL_ORCA_PROFILE_ID, diff --git a/src/shared/agent-cli-install-dir-fallback.test.ts b/src/shared/agent-cli-install-dir-fallback.test.ts new file mode 100644 index 00000000000..793c0861453 --- /dev/null +++ b/src/shared/agent-cli-install-dir-fallback.test.ts @@ -0,0 +1,276 @@ +import { delimiter, join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { detectCommandsInInstallDirs } from './local-agent-install-dir-detection' +import { + getVersionManagerBinPaths, + resolveCliCommand, + resolveCliCommands +} from './node-cli-command-resolution' +import { buildPosixFallbackPathPrelude } from './posix-version-manager-bin-dirs' +import { getSystemCliInstallDirectories } from './system-cli-install-dirs' + +/** + * The install-dir fallback answers "is this agent CLI installed?" whenever the + * login-shell PATH probe does not land. Homebrew, npm's default global prefix + * and opencode's own installer are absolute paths, so they cannot be staged + * under a temp home -- hence a synthetic fs rather than a fixture tree. + * + * Every staged path goes through `join`, because the lookup builds candidates + * with the host's `join`: a literal `/opt/homebrew/bin/codex` would never match + * on a Windows dev machine. + */ +const fsFixture = vi.hoisted(() => ({ executables: new Set() })) + +const MOCK_HOME = '/home/tester' + +vi.mock('node:os', () => ({ homedir: () => MOCK_HOME })) + +vi.mock('node:fs', () => ({ + constants: { X_OK: 1 }, + statSync: (target: string) => { + if (!fsFixture.executables.has(target)) { + throw new Error(`ENOENT: ${target}`) + } + return { isFile: () => true } + }, + accessSync: (target: string) => { + if (!fsFixture.executables.has(target)) { + throw new Error(`EACCES: ${target}`) + } + }, + // No nvm install in any of these cases; the nvm walk is covered by nvm-default-alias.test.ts. + existsSync: () => false, + readdirSync: () => { + throw new Error('ENOENT') + }, + readFileSync: () => { + throw new Error('ENOENT') + } +})) + +// The PATH a Finder/Dock-launched macOS app inherits with no login shell. +const GUI_LAUNCH_PATH = ['/usr/bin', '/bin', '/usr/sbin', '/sbin'].join(delimiter) + +function stage(...paths: string[]): void { + for (const path of paths) { + fsFixture.executables.add(path) + } +} + +function resolveAll( + commands: string[], + options: { platform: NodeJS.Platform; homePath: string } +): Record { + return Object.fromEntries( + resolveCliCommands(commands, { ...options, pathEnv: GUI_LAUNCH_PATH }) + ) as Record +} + +beforeEach(() => { + fsFixture.executables.clear() + // Why: the no-options entry point reads the ambient PATH, where a dev box's + // real /usr/local/bin would answer before the fallback ever runs. + vi.stubEnv('PATH', GUI_LAUNCH_PATH) +}) + +afterEach(() => { + vi.unstubAllEnvs() +}) + +describe('agent CLI install-dir fallback', () => { + it('finds macOS CLIs installed outside a version manager', () => { + const home = '/Users/tester' + stage( + join(home, '.local', 'bin', 'claude'), + join('/opt/homebrew/bin', 'codex'), + join('/usr/local/bin', 'cursor-agent'), + join(home, '.opencode', 'bin', 'opencode') + ) + expect( + resolveAll(['claude', 'codex', 'cursor-agent', 'opencode'], { + platform: 'darwin', + homePath: home + }) + ).toEqual({ + claude: join(home, '.local', 'bin', 'claude'), + codex: join('/opt/homebrew/bin', 'codex'), + 'cursor-agent': join('/usr/local/bin', 'cursor-agent'), + opencode: join(home, '.opencode', 'bin', 'opencode') + }) + }) + + it('finds Linux CLIs in Linuxbrew, snap and nix prefixes, not the macOS brew prefix', () => { + const home = '/home/tester' + stage( + join('/home/linuxbrew/.linuxbrew/bin', 'codex'), + join('/snap/bin', 'cursor-agent'), + join(home, '.nix-profile', 'bin', 'opencode'), + join('/opt/homebrew/bin', 'claude') + ) + expect( + resolveAll(['codex', 'cursor-agent', 'opencode', 'claude'], { + platform: 'linux', + homePath: home + }) + ).toEqual({ + codex: join('/home/linuxbrew/.linuxbrew/bin', 'codex'), + 'cursor-agent': join('/snap/bin', 'cursor-agent'), + opencode: join(home, '.nix-profile', 'bin', 'opencode'), + // Why unresolved: /opt/homebrew is an Apple Silicon prefix; Linuxbrew uses another. + claude: 'claude' + }) + }) + + it('leaves the win32 branch on its own install dirs', () => { + const home = 'C:/Users/tester' + stage(join(home, 'AppData', 'Roaming', 'npm', 'codex.cmd'), join('/usr/local/bin', 'claude')) + expect(resolveAll(['codex', 'claude'], { platform: 'win32', homePath: home })).toEqual({ + codex: join(home, 'AppData', 'Roaming', 'npm', 'codex.cmd'), + claude: 'claude' + }) + }) + + // Why pinned: patchPackagedProcessPath seeds these onto PATH in this order and + // the POSIX guest prelude appends them in it, so a divergence here would spawn + // a different binary than the packaged PATH scan for the same install. + it('ranks system install dirs in the same order as the PATH seed', () => { + const home = '/home/tester' + const dirs = [ + '/usr/local/bin', + '/snap/bin', + '/home/linuxbrew/.linuxbrew/bin', + '/nix/var/nix/profiles/default/bin', + join(home, '.nix-profile', 'bin'), + join(home, '.opencode', 'bin'), + join(home, '.vite-plus', 'bin') + ] + stage(...dirs.map((dir) => join(dir, 'opencode'))) + for (const expected of dirs) { + expect(resolveAll(['opencode'], { platform: 'linux', homePath: home })).toEqual({ + opencode: join(expected, 'opencode') + }) + fsFixture.executables.delete(join(expected, 'opencode')) + } + }) + + // Why both resolvers and both platforms: resolveCliCommand is what every + // spawn site (codex login, app-server, session-index heal) calls, and its + // list was once spelled separately from resolveCliCommands'. A same-named + // binary in /usr/local/bin must never shadow the one a version manager owns. + describe.each([ + { platform: 'darwin' as const, home: '/Users/tester', systemDir: '/opt/homebrew/bin' }, + { + platform: 'linux' as const, + home: '/home/tester', + systemDir: '/home/linuxbrew/.linuxbrew/bin' + } + ])('$platform: system dirs stay last', ({ platform, home, systemDir }) => { + it('lets a version-manager install outrank a system one', () => { + const managed = join(home, '.volta', 'bin', 'codex') + stage(managed, join(systemDir, 'codex'), join('/usr/local/bin', 'codex')) + expect(resolveCliCommand('codex', { platform, homePath: home })).toBe(managed) + expect(resolveAll(['codex'], { platform, homePath: home })).toEqual({ codex: managed }) + }) + + it('lets an npm --user (~/.local/bin) install outrank a system one', () => { + const managed = join(home, '.local', 'bin', 'codex') + stage(managed, join(systemDir, 'codex')) + expect(resolveCliCommand('codex', { platform, homePath: home })).toBe(managed) + expect(resolveAll(['codex'], { platform, homePath: home })).toEqual({ codex: managed }) + }) + + it('lets a copy already on PATH outrank every install dir', () => { + const onPath = join('/custom/bin', 'codex') + const pathEnv = [GUI_LAUNCH_PATH, '/custom/bin'].join(delimiter) + stage(onPath, join(home, '.volta', 'bin', 'codex'), join(systemDir, 'codex')) + expect(resolveCliCommand('codex', { platform, homePath: home, pathEnv })).toBe(onPath) + expect(resolveCliCommands(['codex'], { platform, homePath: home, pathEnv })).toEqual( + new Map([['codex', onPath]]) + ) + }) + }) + + // Why this guard: getVersionManagerBinPaths is PREPENDED onto PATH by + // patchPackagedProcessPath and the CLI's addAgentNodePaths, so a system dir + // leaking into it would re-rank binaries the user already has (#18234). + it('keeps system install dirs out of the PATH seed list', () => { + for (const platform of ['darwin', 'linux'] as const) { + const home = platform === 'darwin' ? '/Users/tester' : '/home/tester' + const seeded = getVersionManagerBinPaths({ platform, homePath: home }) + // Spelled out, not derived from the list under test: a guard that iterates + // getSystemCliInstallDirectories passes vacuously if that list is emptied + // into getBaseVersionManagerDirectories, which is the leak it guards. + for (const directory of [ + '/opt/homebrew/bin', + '/usr/local/bin', + '/snap/bin', + '/home/linuxbrew/.linuxbrew/bin', + '/nix/var/nix/profiles/default/bin', + join(home, '.nix-profile', 'bin'), + join(home, '.opencode', 'bin'), + join(home, '.vite-plus', 'bin') + ]) { + expect(seeded).not.toContain(directory) + } + } + }) + + // Why through this entry point: it is what the `orca` CLI's agent detection + // calls, and the "absolute path means installed" contract lives here. + it.skipIf(process.platform === 'win32')( + 'reports a system-installed CLI as detected, not just resolved', + () => { + stage( + join('/usr/local/bin', 'codex'), + join(MOCK_HOME, '.opencode', 'bin', 'opencode'), + // Why pi: it is a probed detect command on every runtime (tui-agent-config.ts, + // no detectUnsupportedRuntimes) and its installer defaults to ~/.vite-plus/bin, + // the second dir #829 named and seeded alongside ~/.opencode/bin. + join(MOCK_HOME, '.vite-plus', 'bin', 'pi') + ) + // All three come from the fallback: the stubbed PATH holds no system dir. + expect(detectCommandsInInstallDirs(['codex', 'opencode', 'pi', 'cursor-agent'])).toEqual( + new Set(['codex', 'opencode', 'pi']) + ) + } + ) + + it('carries the system install dirs into the POSIX guest fallback prelude', () => { + const prelude = buildPosixFallbackPathPrelude() + const systemDirs = [ + '"/usr/local/bin"', + '"/snap/bin"', + '"/home/linuxbrew/.linuxbrew/bin"', + '"/nix/var/nix/profiles/default/bin"', + '"$HOME/.nix-profile/bin"', + '"$HOME/.opencode/bin"', + '"$HOME/.vite-plus/bin"' + ] + const offsets = systemDirs.map((dir) => prelude.indexOf(dir)) + expect(offsets.every((offset) => offset >= 0)).toBe(true) + expect([...offsets].sort((a, b) => a - b)).toEqual(offsets) + // Why after: the guest prelude appends, so a version manager must still win. + expect(prelude.indexOf('.nvm/versions/node/*/bin')).toBeLessThan(offsets[0]) + // Why absent: a WSL guest is Linux, so /opt/homebrew is never its brew prefix. + expect(prelude).not.toContain('/opt/homebrew') + }) + + // Why derived: the native and guest lists drifted apart once by hand. Every + // version-manager dir the native resolver knows must precede the guest's + // first system dir, and the guest's system block must be the native one. + it('keeps the WSL guest prelude in step with the native Linux lists', () => { + const prelude = buildPosixFallbackPathPrelude() + const asGuest = (dir: string): string => `"${dir.split('\\').join('/')}"` + const systemDirs = getSystemCliInstallDirectories('linux', '$HOME').map(asGuest) + const firstSystemOffset = prelude.indexOf(systemDirs[0]) + expect(firstSystemOffset).toBeGreaterThan(0) + for (const dir of getVersionManagerBinPaths({ platform: 'linux', homePath: '$HOME' })) { + const offset = prelude.indexOf(asGuest(dir)) + expect(offset, dir).toBeGreaterThanOrEqual(0) + expect(offset, dir).toBeLessThan(firstSystemOffset) + } + const systemOffsets = systemDirs.map((dir) => prelude.indexOf(dir)) + expect(systemOffsets.every((offset) => offset >= firstSystemOffset)).toBe(true) + expect([...systemOffsets].sort((a, b) => a - b)).toEqual(systemOffsets) + }) +}) diff --git a/src/shared/agent-hook-spool-read.test.ts b/src/shared/agent-hook-spool-read.test.ts new file mode 100644 index 00000000000..d589f5b2274 --- /dev/null +++ b/src/shared/agent-hook-spool-read.test.ts @@ -0,0 +1,65 @@ +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { readSpoolFile } from './agent-hook-spool' + +let dir: string + +beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'orca-spool-read-')) +}) + +afterEach(() => { + rmSync(dir, { recursive: true, force: true }) +}) + +function write(contents: string): string { + const file = join(dir, 'spool.jsonl') + writeFileSync(file, contents) + return file +} + +function record(paneKey: string): string { + return JSON.stringify({ paneKey, source: 'PreToolUse', payload: {}, receivedAt: Date.now() }) +} + +describe('readSpoolFile', () => { + it('leaves a trailing line without its newline unconsumed', () => { + const complete = `${record('a')}\n` + const result = readSpoolFile(write(`${complete}${record('b')}`)) + + expect(result.records.map((entry) => entry.paneKey)).toEqual(['a']) + // The in-flight final record must stay replayable: consumed stops at the last newline. + expect(result.consumed).toBe(Buffer.byteLength(complete)) + }) + + it('consumes through the final newline when every line is complete', () => { + const contents = `${record('a')}\n${record('b')}\n` + const result = readSpoolFile(write(contents)) + + expect(result.records.map((entry) => entry.paneKey)).toEqual(['a', 'b']) + expect(result.consumed).toBe(Buffer.byteLength(contents)) + }) + + it('skips blank lines without consuming less than the bytes they occupy', () => { + const contents = `${record('a')}\n\n\n${record('b')}\n` + const result = readSpoolFile(write(contents)) + + expect(result.records.map((entry) => entry.paneKey)).toEqual(['a', 'b']) + expect(result.consumed).toBe(Buffer.byteLength(contents)) + }) + + it('returns nothing for an empty file', () => { + expect(readSpoolFile(write(''))).toEqual({ records: [], consumed: 0 }) + }) + + it('returns nothing for a file that is one torn line', () => { + expect(readSpoolFile(write(record('a'))).records).toEqual([]) + expect(readSpoolFile(write(record('a'))).consumed).toBe(0) + }) + + it('returns nothing for a missing file', () => { + expect(readSpoolFile(join(dir, 'absent.jsonl'))).toEqual({ records: [], consumed: 0 }) + }) +}) diff --git a/src/shared/agent-hook-spool.ts b/src/shared/agent-hook-spool.ts index 3ad478ed8a5..275e6f5ce52 100644 --- a/src/shared/agent-hook-spool.ts +++ b/src/shared/agent-hook-spool.ts @@ -67,19 +67,17 @@ export function readSpoolFile( const records: SpoolRecord[] = [] let consumed = 0 let start = 0 - for (let end = 0; end <= bytes.length; end += 1) { - if (end !== bytes.length && bytes[end] !== 0x0a) { - continue - } + // indexOf, not a per-byte loop: this runs over every spooled file before the hook listener binds, + // and Buffer.indexOf finds the newline with memchr instead of an interpreted scan. + for (;;) { + const end = bytes.indexOf(0x0a, start) // A final line without its newline may still be in flight from a hook writer. // Leave it untouched until the writer terminates the record explicitly. - if (end === bytes.length && (end === 0 || bytes[end - 1] !== 0x0a)) { + if (end === -1) { break } const lineBytes = bytes.subarray(start, end) - if (end !== bytes.length) { - consumed = end + 1 - } + consumed = end + 1 start = end + 1 if (lineBytes.length === 0) { continue diff --git a/src/shared/agent-session-journal-schemas.ts b/src/shared/agent-session-journal-schemas.ts index 2b1ab5404fc..ee200cdcd94 100644 --- a/src/shared/agent-session-journal-schemas.ts +++ b/src/shared/agent-session-journal-schemas.ts @@ -1,11 +1,12 @@ // ─── Canonical runtime schemas for the journal render model ───────────────── -// The journal admits JSON it did not just write — snapshot files and log rows -// re-enter from disk and are republished to clients — while the reducer, the +// The journal admits JSON it did not just write — persisted rows re-enter from +// SQLite on replay and are republished to clients — while the reducer, the // shared projection, and the prompt surfaces dereference nested fields without // guards. These schemas are the single deep validators for that render model: // admission must reject a JSON-valid but structurally wrong item (a question -// whose `options` are null, a prompt without its `resolution`) so corruption -// lands in quarantine instead of throwing mid-render. +// whose `options` are null, a prompt without its `resolution`) so the row is +// rejected at replay, where a repair can delete it, instead of throwing +// mid-render. // // Discriminants (`kind`, known block `type`s) are validated deeply. Open string // fields (roles, dispatch/tool states) stay type-checked, never enum-checked, @@ -63,7 +64,24 @@ const Block = z.union([ z.object({ type: z.string() }).refine((block) => !KNOWN_BLOCK_TYPES.has(block.type)) ]) -const PromptOption = z.object({ id: z.string(), label: z.string() }) +const PromptOption = z + .object({ + id: z.string(), + label: z.string(), + description: z.string().optional() + }) + .strict() + +const Question = z + .object({ + id: z.string(), + question: z.string(), + header: z.string().optional(), + multiSelect: z.boolean(), + options: z.array(PromptOption), + freeTextQuestionId: z.string().optional() + }) + .strict() const Resolution = z.object({ state: z.string().min(1), @@ -100,6 +118,7 @@ export const AgentJournalItemBodySchema = z.discriminatedUnion('kind', [ kind: z.literal('question'), question: z.string(), options: z.array(PromptOption), + questions: z.array(Question).optional(), freeTextQuestionId: z.string().optional(), resolution: Resolution }), @@ -154,8 +173,8 @@ export function isAdmissibleAgentJournalSubmission( return AgentJournalSubmissionSchema.safeParse(value).success } -/** Compile-time proof that every canonical value is admissible, so admission - * can never quarantine a row a writer in this build produced. The schemas are +/** Compile-time proof that every canonical value is admissible, so replay can + * never reject a row a writer in this build produced. The schemas are * deliberately wider on open string fields, so only this direction holds. */ type Admits = T export type CanonicalJournalShapesAreAdmissible = [ diff --git a/src/shared/agent-session-journal-types.ts b/src/shared/agent-session-journal-types.ts index f5dabfdec23..cf6d89070bf 100644 --- a/src/shared/agent-session-journal-types.ts +++ b/src/shared/agent-session-journal-types.ts @@ -58,13 +58,15 @@ export type AgentJournalItemIdentity = // ─── Bounded payloads ─────────────────────────────────────────────────────── -/** A tool output or diff body clipped to a head plus a content-addressed - * remainder. Crossing a bound sets `truncated`; it never silently drops. */ +/** A tool output or diff body clipped to a head. The remainder is DISCARDED, + * never stored: crossing a bound sets `truncated` and the two fields below + * describe what was dropped, so it is marked rather than silently lost. */ export type AgentJournalBoundedPayload = { head: string /** Byte length of the ORIGINAL payload, not of `head`. */ byteLength: number - /** sha256 of the original payload, and the blob store key when `truncated`. */ + /** sha256 of the original payload — identification only; nothing stores or + * retrieves the discarded remainder by it. */ digest: string truncated: boolean } @@ -111,6 +113,17 @@ export type AgentJournalResolution = { export type AgentJournalPromptOption = { id: string label: string + description?: string +} + +export type AgentJournalQuestion = { + id: string + question: string + header?: string + multiSelect: boolean + options: AgentJournalPromptOption[] + /** Present when the provider accepts an answer outside the offered options. */ + freeTextQuestionId?: string } export type AgentJournalApprovalItem = { @@ -125,6 +138,7 @@ export type AgentJournalQuestionItem = { kind: 'question' question: string options: AgentJournalPromptOption[] + questions?: AgentJournalQuestion[] /** Present when the provider accepts an answer outside the offered options. */ freeTextQuestionId?: string resolution: AgentJournalResolution diff --git a/src/shared/agent-session-question-answer.test.ts b/src/shared/agent-session-question-answer.test.ts new file mode 100644 index 00000000000..529bfdfd72d --- /dev/null +++ b/src/shared/agent-session-question-answer.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from 'vitest' +import { + decodeAgentSessionQuestionAnswers, + encodeAgentSessionQuestionAnswers, + isValidAgentSessionQuestionAnswers, + type AgentSessionQuestionAnswer +} from './agent-session-question-answer' + +describe('agent-session grouped question answers', () => { + const answers: AgentSessionQuestionAnswer[] = [ + { questionId: 'q1', optionIds: ['target-web', 'target-mobile'] }, + { questionId: 'q2', optionIds: [], other: 'SSH host' } + ] + + it('round-trips grouped multi-select and free-text answers', () => { + expect(decodeAgentSessionQuestionAnswers(encodeAgentSessionQuestionAnswers(answers))).toEqual( + answers + ) + }) + + it('validates each grouped answer against its question shape', () => { + const questions = [ + { + id: 'q1', + question: 'Targets', + multiSelect: true, + options: [ + { id: 'target-web', label: 'Web' }, + { id: 'target-mobile', label: 'Mobile' } + ] + }, + { + id: 'q2', + question: 'Host', + multiSelect: false, + options: [], + freeTextQuestionId: 'q2' + } + ] + + expect(isValidAgentSessionQuestionAnswers(questions, answers)).toBe(true) + expect( + isValidAgentSessionQuestionAnswers(questions, [ + { questionId: 'q1', optionIds: ['unknown'] }, + answers[1]! + ]) + ).toBe(false) + }) +}) diff --git a/src/shared/agent-session-question-answer.ts b/src/shared/agent-session-question-answer.ts new file mode 100644 index 00000000000..f90df30ddff --- /dev/null +++ b/src/shared/agent-session-question-answer.ts @@ -0,0 +1,84 @@ +import type { AgentJournalQuestion } from './agent-session-journal-types' + +const GROUP_ANSWER_PREFIX = 'question-group:' + +export type AgentSessionQuestionAnswer = { + questionId: string + optionIds: string[] + other?: string +} + +export function encodeAgentSessionQuestionAnswers( + answers: readonly AgentSessionQuestionAnswer[] +): string { + return `${GROUP_ANSWER_PREFIX}${encodeURIComponent(JSON.stringify(answers))}` +} + +export function decodeAgentSessionQuestionAnswers( + encoded: string +): AgentSessionQuestionAnswer[] | null { + if (!encoded.startsWith(GROUP_ANSWER_PREFIX)) { + return null + } + try { + const parsed: unknown = JSON.parse( + decodeURIComponent(encoded.slice(GROUP_ANSWER_PREFIX.length)) + ) + if (!Array.isArray(parsed)) { + return null + } + const answers = parsed.flatMap((value): AgentSessionQuestionAnswer[] => { + if (!value || typeof value !== 'object' || Array.isArray(value)) { + return [] + } + const record = value as Record + if ( + typeof record.questionId !== 'string' || + !Array.isArray(record.optionIds) || + !record.optionIds.every((optionId) => typeof optionId === 'string') || + (record.other !== undefined && typeof record.other !== 'string') + ) { + return [] + } + return [ + { + questionId: record.questionId, + optionIds: record.optionIds, + ...(typeof record.other === 'string' ? { other: record.other } : {}) + } + ] + }) + return answers.length === parsed.length ? answers : null + } catch { + return null + } +} + +export function isValidAgentSessionQuestionAnswers( + questions: readonly AgentJournalQuestion[], + answers: readonly AgentSessionQuestionAnswer[] +): boolean { + if (answers.length !== questions.length) { + return false + } + const byId = new Map(answers.map((answer) => [answer.questionId, answer])) + if (byId.size !== answers.length) { + return false + } + return questions.every((question) => { + const answer = byId.get(question.id) + if (!answer) { + return false + } + const offered = new Set(question.options.map((option) => option.id)) + if (answer.optionIds.some((optionId) => !offered.has(optionId))) { + return false + } + const other = answer.other?.trim() ?? '' + if (other && !question.freeTextQuestionId) { + return false + } + const answerCount = answer.optionIds.length + (other ? 1 : 0) + return answerCount > 0 && (question.multiSelect || answerCount === 1) + }) +} diff --git a/src/shared/agent-session-wire.ts b/src/shared/agent-session-wire.ts index 893534f7f10..6c2b3791024 100644 --- a/src/shared/agent-session-wire.ts +++ b/src/shared/agent-session-wire.ts @@ -254,5 +254,11 @@ export type AgentSessionOptionsResult = { current: { model: string effort?: string + /** + * Option ids whose value the provider reported back, not merely accepted. + * Optional: a host that predates it sends nothing and the client keeps + * treating the value as unconfirmed, which is what it was before. + */ + confirmed?: readonly string[] } } diff --git a/src/shared/child-process/process-tree-kill-gate.ts b/src/shared/child-process/process-tree-kill-gate.ts new file mode 100644 index 00000000000..420fbfeda9b --- /dev/null +++ b/src/shared/child-process/process-tree-kill-gate.ts @@ -0,0 +1,43 @@ +/** + * Seam that lets the main process decide, and record, the tree-kills issued + * from code it does not own. + * + * Why a seam and not a direct call: `signalProcessTree` is the choke point every + * `runProcess` termination funnels through, and the codex app-server and + * ephemeral-VM kills are shared with the CLI — all of them live outside + * `src/main` and cannot import the own-Chromium guard or the crash breadcrumb + * store. Main registers the guard at startup; everywhere else this admits every + * kill and records nothing. + */ + +/** Blast radius, not mechanism: `win-taskkill-tree` is addressed by pid and walks + * whatever tree that pid has *now*, so it can land on a recycled pid that is + * since one of Orca's own Chromium processes. A process group can only contain + * processes Orca itself put there. */ +export type ProcessTreeKillScope = 'win-taskkill-tree' | 'posix-process-group' + +export type ProcessTreeKill = { + pid: number + site: string + scope: ProcessTreeKillScope +} + +/** False means the caller must not walk that pid's tree — main is accounting for + * it. Killing the root through its own child handle stays correct and required: + * a handle cannot land on the recycled pid the refusal is about. */ +type ProcessTreeKillGate = (kill: ProcessTreeKill) => boolean + +let gate: ProcessTreeKillGate | null = null + +export function setProcessTreeKillGate(next: ProcessTreeKillGate | null): void { + gate = next +} + +export function admitProcessTreeKill(kill: ProcessTreeKill): boolean { + try { + return gate?.(kill) ?? true + } catch { + // Diagnostics must never turn a successful termination into a failed one. + return true + } +} diff --git a/src/shared/child-process/process-tree-kill-observer.ts b/src/shared/child-process/process-tree-kill-observer.ts deleted file mode 100644 index 8abf971798a..00000000000 --- a/src/shared/child-process/process-tree-kill-observer.ts +++ /dev/null @@ -1,37 +0,0 @@ -/** - * Seam that lets the main process record the tree-kills issued from here. - * - * Why a seam and not a direct call: `signalProcessTree` is the choke point every - * `runProcess` termination funnels through, but it lives in `src/shared` and so - * runs in the CLI and relay too — it cannot import the main-process crash - * breadcrumb store. Main registers the recorder at startup; everywhere else this - * stays a no-op. - */ - -/** Blast radius, not mechanism: `win-taskkill-tree` is addressed by pid and walks - * whatever tree that pid has *now*, so it can land on a recycled pid that is - * since one of Orca's own Chromium processes. A process group can only contain - * processes Orca itself put there. */ -export type ProcessTreeKillScope = 'win-taskkill-tree' | 'posix-process-group' - -export type ProcessTreeKill = { - pid: number - site: string - scope: ProcessTreeKillScope -} - -type ProcessTreeKillObserver = (kill: ProcessTreeKill) => void - -let observer: ProcessTreeKillObserver | null = null - -export function setProcessTreeKillObserver(next: ProcessTreeKillObserver | null): void { - observer = next -} - -export function notifyProcessTreeKill(kill: ProcessTreeKill): void { - try { - observer?.(kill) - } catch { - // Diagnostics must never turn a successful termination into a failed one. - } -} diff --git a/src/shared/child-process/process-tree-termination.test.ts b/src/shared/child-process/process-tree-termination.test.ts index cc3394fb5a3..10a4c725089 100644 --- a/src/shared/child-process/process-tree-termination.test.ts +++ b/src/shared/child-process/process-tree-termination.test.ts @@ -7,7 +7,7 @@ const { spawnMock } = vi.hoisted(() => ({ spawnMock: vi.fn() })) vi.mock('node:child_process', () => ({ spawn: spawnMock })) import { forceTerminateProcessTree, signalProcessTree } from './process-tree-termination' -import { setProcessTreeKillObserver, type ProcessTreeKill } from './process-tree-kill-observer' +import { setProcessTreeKillGate, type ProcessTreeKill } from './process-tree-kill-gate' function mockProcess(pid: number): ChildProcess { const child = new EventEmitter() as EventEmitter & { @@ -103,11 +103,14 @@ describe('process-tree-kill breadcrumb seam', () => { beforeEach(() => { observed.length = 0 - setProcessTreeKillObserver((kill) => observed.push(kill)) + setProcessTreeKillGate((kill) => { + observed.push(kill) + return true + }) }) afterEach(() => { - setProcessTreeKillObserver(null) + setProcessTreeKillGate(null) spawnMock.mockReset() vi.restoreAllMocks() }) diff --git a/src/shared/child-process/process-tree-termination.ts b/src/shared/child-process/process-tree-termination.ts index 15264b2f6ce..ffcb93b2245 100644 --- a/src/shared/child-process/process-tree-termination.ts +++ b/src/shared/child-process/process-tree-termination.ts @@ -1,5 +1,5 @@ import { spawn as nodeSpawn, type ChildProcess } from 'node:child_process' -import { notifyProcessTreeKill } from './process-tree-kill-observer' +import { admitProcessTreeKill } from './process-tree-kill-gate' const PROBE_INTERVAL_MS = 25 const SUBPROCESS_TIMEOUT_MS = 2_000 @@ -13,8 +13,17 @@ const MAX_PS_OUTPUT_BYTES = 8 * 1024 * 1024 * leader would hand it to whatever group it inherited instead. * * Runs in every host — Electron main, the daemon, the relay, the CLI — so the - * main-process own-Chromium guard cannot reach here; the exit check below is - * what keeps the Windows branch off a pid that is no longer ours. + * own-Chromium guard arrives through `process-tree-kill-gate`, which main + * installs and every other host leaves admitting. The exit check below is what + * keeps the Windows branch off a pid that is no longer ours on those hosts. + * + * Both arms ask before walking the tree, so a refused pid never gets a + * pid-addressed kill; that also means the recorded crumb says "about to kill", + * not "killed". The root is still killed through its handle, which cannot reach + * the recycled pid the refusal was about, so a refusal is never a leak. Main's + * gate only ever refuses the `win-taskkill-tree` scope — a POSIX group holds + * only what Orca put in it — so the POSIX refusal arm is the seam's contract, + * not something any installed gate exercises today. */ export function signalProcessTree(child: ChildProcess, signal?: NodeJS.Signals): Promise { if (!child.pid) { @@ -39,13 +48,20 @@ export function signalProcessTree(child: ChildProcess, signal?: NodeJS.Signals): } return taskkillTree(child, child.pid, signal) } - try { - process.kill(-child.pid, signal) - notifyProcessTreeKill({ + if ( + !admitProcessTreeKill({ pid: child.pid, site: 'run-process-tree', scope: 'posix-process-group' }) + ) { + // Same shape as the reaped-pid skip above: refuse the group, still kill the + // root by handle, and report unverified. + killRoot(child, signal) + return Promise.resolve(false) + } + try { + process.kill(-child.pid, signal) return Promise.resolve(true) } catch { return Promise.resolve(!processGroupExists(child.pid)) @@ -73,6 +89,13 @@ function taskkillTree( rootPid: number, signal?: NodeJS.Signals ): Promise { + // Asked before the spawn, not after: a refusal has to prevent the taskkill. + if ( + !admitProcessTreeKill({ pid: rootPid, site: 'run-process-tree', scope: 'win-taskkill-tree' }) + ) { + killRoot(child, signal) + return Promise.resolve(false) + } return new Promise((resolve) => { let killer: ChildProcess try { @@ -86,7 +109,6 @@ function taskkillTree( resolve(false) return } - notifyProcessTreeKill({ pid: rootPid, site: 'run-process-tree', scope: 'win-taskkill-tree' }) let settled = false const finish = (fallback: boolean): void => { if (settled) { diff --git a/src/shared/command-code-output-status.test.ts b/src/shared/command-code-output-status.test.ts index 77a8005056e..4ca0be25bbc 100644 --- a/src/shared/command-code-output-status.test.ts +++ b/src/shared/command-code-output-status.test.ts @@ -49,6 +49,22 @@ describe('createCommandCodeOutputStatusDetector', () => { expect(onWorking).toHaveBeenCalledWith('') }) + it('detects the banner after chunks that carry no banner characters at all', () => { + const onWorking = vi.fn() + const detector = createCommandCodeOutputStatusDetector({ + startupCommand: null, + onWorking + }) + + // Ordinary agent output with no '#': the raw prefilter skips these without building a window. + expect(detector.observe('Codex is editing the file and rendering a diff\r\n')).toBe(false) + expect(detector.observe('Codex wrote 12 lines\r\n')).toBe(false) + expect(detector.observe('# Command Code v0.27.2\r\n')).toBe(false) + expect(detector.observe('⌘ Parsing...')).toBe(true) + + expect(onWorking).toHaveBeenCalledWith('') + }) + it('detects the Command Code banner when ANSI styling splits the words', () => { const onWorking = vi.fn() const detector = createCommandCodeOutputStatusDetector({ diff --git a/src/shared/command-code-output-status.ts b/src/shared/command-code-output-status.ts index 16942674219..e98f72d857b 100644 --- a/src/shared/command-code-output-status.ts +++ b/src/shared/command-code-output-status.ts @@ -142,6 +142,21 @@ function rawTextMayContainCommandCodeBanner(rawText: string): boolean { return rawText.includes('C') && rawText.includes('o') && rawText.includes('d') } +// COMMAND_CODE_BANNER_RE requires a literal '#', and stripTerminalControl only ever removes +// characters, so raw bytes without one cannot produce a banner match. +const BANNER_REQUIRED_RAW_CHAR = '#' + +/** + * Prefilter run against the raw chunk before the scan windows are built. Testing the whole carry + * plus chunk over-admits relative to the window test (the windows drop middle content), so it can + * never turn a match into a miss. + */ +function rawChunkMayContainCommandCodeBanner(previousRawText: string, data: string): boolean { + return ( + data.includes(BANNER_REQUIRED_RAW_CHAR) || previousRawText.includes(BANNER_REQUIRED_RAW_CHAR) + ) +} + function appendRecentRawText(previousRawText: string, data: string): string { if (data.length >= RECENT_TEXT_LIMIT) { return data.slice(-RECENT_TEXT_LIMIT) @@ -233,6 +248,11 @@ export function createCommandCodeOutputStatusDetector(args: { observe(data: string): boolean { const previousRawText = recentRawText recentRawText = appendRecentRawText(previousRawText, data) + // Why before the windows: a non-Command-Code pane pays two ~4KB string builds per chunk + // otherwise, only to fail the same prefilter a few lines later. + if (!hasSeenCommandCodeUi && !rawChunkMayContainCommandCodeBanner(previousRawText, data)) { + return false + } const scanRawText = buildStatusScanRawText(previousRawText, data) const scanRawTextWithChunkBoundary = previousRawText ? buildStatusScanRawText(`${previousRawText}\n`, data) diff --git a/src/shared/ephemeral-vm-recipe-process.ts b/src/shared/ephemeral-vm-recipe-process.ts index b258a1b2704..30d44ac1dbc 100644 --- a/src/shared/ephemeral-vm-recipe-process.ts +++ b/src/shared/ephemeral-vm-recipe-process.ts @@ -1,5 +1,6 @@ import { spawn, type ChildProcessWithoutNullStreams } from 'node:child_process' import type { EphemeralVmRecipeContext } from './ephemeral-vm-recipe-runner' +import { admitProcessTreeKill } from './child-process/process-tree-kill-gate' const DEFAULT_MAX_CAPTURE_BYTES = 1024 * 1024 const CANCEL_FORCE_KILL_DELAY_MS = 5_000 @@ -132,13 +133,26 @@ export async function runRecipeCommand(args: { }) } -function killRecipeProcess(child: ChildProcessWithoutNullStreams, force = false): void { +/** Exported for the refusal-fallback test; the abort path is otherwise unreachable. */ +export function killRecipeProcess(child: ChildProcessWithoutNullStreams, force = false): void { const signal = force ? 'SIGKILL' : 'SIGTERM' if (process.platform === 'win32') { // Recipes run through `cmd.exe /c` (shell: true), so child.kill() would only // terminate the wrapper and orphan the actual recipe subprocess (e.g. a cloud // CLI mid-provision). taskkill /T walks and kills the whole tree. if (child.pid) { + if ( + !admitProcessTreeKill({ + pid: child.pid, + site: 'ephemeral-vm-recipe', + scope: 'win-taskkill-tree' + }) + ) { + // Refusal blocks the tree walk, not the termination: the root kill is + // handle-addressed, so it cannot reach the recycled pid we refused. + child.kill(signal) + return + } const killer = spawn('taskkill', ['/pid', String(child.pid), '/t', '/f'], { windowsHide: true, stdio: 'ignore' diff --git a/src/shared/native-chat-session-option-snapshot.test.ts b/src/shared/native-chat-session-option-snapshot.test.ts index 5286ad9154b..c9980be3b3b 100644 --- a/src/shared/native-chat-session-option-snapshot.test.ts +++ b/src/shared/native-chat-session-option-snapshot.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from 'vitest' -import type { SessionOptionDescriptor } from './native-chat-session-options' +import { + sessionOptionDispatchUnconfirmed, + type SessionOptionDescriptor +} from './native-chat-session-options' import { mergeDiscoveredAuthoritativeModels } from './agent-session-option-catalog' import { CLAUDE_SESSION_OPTION_CATALOG, @@ -22,13 +25,37 @@ function claudeRecord(): NativeChatSessionOptionRecord { } describe('buildNativeChatSessionOptionSnapshot', () => { + // The producer names its lane once, here; `dispatched` is emitted by both and + // is not evidence of which one, so the descriptor has to carry the answer. + it.each(['catalog', 'agent-session'] as const)( + 'stamps every descriptor with the %s transport it was built for', + (liveTransport) => { + const record = claudeRecord() + record.model = { value: 'sonnet', source: 'dispatched' } + const snapshot = buildNativeChatSessionOptionSnapshot({ + catalog: CLAUDE_SESSION_OPTION_CATALOG, + models: CLAUDE_SESSION_OPTION_CATALOG.models, + record, + mode: 'live', + modelLabel: 'Model', + liveTransport + }) + expect(snapshot.length).toBeGreaterThan(1) + expect(snapshot.every((descriptor) => descriptor.transport === liveTransport)).toBe(true) + const dispatched = snapshot.filter((descriptor) => descriptor.valueSource === 'dispatched') + expect(dispatched.length).toBeGreaterThan(0) + expect(dispatched.every(sessionOptionDispatchUnconfirmed)).toBe(liveTransport === 'catalog') + } + ) + it('offers every catalog model with the current value unknown', () => { const snapshot = buildNativeChatSessionOptionSnapshot({ catalog: CLAUDE_SESSION_OPTION_CATALOG, models: CLAUDE_SESSION_OPTION_CATALOG.models, record: claudeRecord(), mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot).toHaveLength(1) const model = snapshot[0]! @@ -50,7 +77,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot.map((descriptor) => descriptor.id)).toEqual(['model', 'effort']) expect(snapshot[0]).toMatchObject({ valueSource: 'dispatched' }) @@ -66,7 +94,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const model = snapshot[0]! if (model.kind.type !== 'select') { @@ -84,7 +113,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: [], record: claudeRecord(), mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) ).toEqual([]) }) @@ -136,7 +166,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: reconciled, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const model = snapshot[0]! if (model.kind.type !== 'select') { @@ -197,7 +228,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: reconciled, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot.map((descriptor) => descriptor.id)).toEqual(['model', 'effort']) expect(resolveAgentSessionOptionLaunch('grok', { model: 'grok-4.5' }).args).toEqual([ @@ -215,7 +247,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CODEX_SESSION_OPTION_CATALOG.models, record: createNativeChatSessionOptionRecord('codex'), mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot[0]).toMatchObject({ settable: true }) expect(snapshot[0]?.action).toEqual({ type: 'agent-picker' }) @@ -233,7 +266,8 @@ describe('buildNativeChatSessionOptionSnapshot', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const fastMode = snapshot.find((descriptor) => descriptor.id === 'fastMode') expect(fastMode).toMatchObject({ action: { type: 'toggle-command' } }) @@ -250,7 +284,8 @@ describe('defaults on load', () => { models, record: createNativeChatSessionOptionRecord('grok'), mode, - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) it('shows the default model before anything is picked', () => { @@ -299,7 +334,8 @@ describe('defaults on load', () => { ], record, mode: 'draft', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(snapshot[0]).toMatchObject({ valueSource: 'dispatched' }) expect(snapshot[0]!.kind.type === 'select' ? snapshot[0]!.kind.currentValue : null).toBe( @@ -315,7 +351,8 @@ describe('defaults on load', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record: claudeRecord(), mode: 'draft', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) expect(CLAUDE_SESSION_OPTION_CATALOG.models.some((model) => model.isDefault)).toBe(true) expect(CLAUDE_SESSION_OPTION_CATALOG.defaultModelIsCliDefault).toBeUndefined() @@ -332,7 +369,8 @@ describe('defaults on load', () => { models: CLAUDE_SESSION_OPTION_CATALOG.models, record, mode: 'draft', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) const effort = snapshot.find((descriptor) => descriptor.id === 'effort') expect(effort).toMatchObject({ valueSource: 'default' }) diff --git a/src/shared/native-chat-session-option-snapshot.ts b/src/shared/native-chat-session-option-snapshot.ts index 76556211ab0..9fba42cf897 100644 --- a/src/shared/native-chat-session-option-snapshot.ts +++ b/src/shared/native-chat-session-option-snapshot.ts @@ -5,6 +5,7 @@ import type { CatalogOption } from './agent-session-option-catalog' import type { + NativeChatLiveOptionTransport, SessionOptionDescriptor, SessionOptionSelectChoice } from './native-chat-session-options' @@ -15,7 +16,7 @@ import { } from './native-chat-session-option-state' export type NativeChatSessionOptionMode = 'draft' | 'live' -export type NativeChatLiveOptionTransport = 'catalog' | 'agent-session' +export type { NativeChatLiveOptionTransport } function choiceWithCurrent( choices: readonly SessionOptionSelectChoice[], @@ -107,6 +108,7 @@ function optionDescriptor(args: { choices }, valueSource, + transport: liveTransport, ...settable, ...(action ? { action } : {}) } @@ -127,6 +129,7 @@ function optionDescriptor(args: { ...(currentValue === undefined ? {} : { currentValue }) }, valueSource, + transport: liveTransport, ...settable, ...(action ? { action } : {}) } @@ -202,9 +205,11 @@ export function buildNativeChatSessionOptionSnapshot(args: { record: NativeChatSessionOptionRecord mode: NativeChatSessionOptionMode modelLabel: string - liveTransport?: NativeChatLiveOptionTransport + /** Required, not defaulted: this is the only place a descriptor is built, so a + * producer that must state its lane here cannot silently inherit the other's. */ + liveTransport: NativeChatLiveOptionTransport }): SessionOptionDescriptor[] { - const { catalog, models, record, mode, modelLabel, liveTransport = 'catalog' } = args + const { catalog, models, record, mode, modelLabel, liveTransport } = args if (models.length === 0) { return [] } @@ -232,6 +237,7 @@ export function buildNativeChatSessionOptionSnapshot(args: { choices: modelChoices }, valueSource: modelTracked?.source ?? (defaultModelId ? 'default' : 'unknown'), + transport: liveTransport, ...settableState({ mode, liveTransport, apply: catalog.modelApply }), ...(modelAction ? { action: modelAction } : {}) } diff --git a/src/shared/native-chat-session-option-state.ts b/src/shared/native-chat-session-option-state.ts index 78b70980270..5d070969775 100644 --- a/src/shared/native-chat-session-option-state.ts +++ b/src/shared/native-chat-session-option-state.ts @@ -122,25 +122,30 @@ export function flattenNativeChatSessionOptionRecord( export function applyNativeChatReportedSessionOptions( record: NativeChatSessionOptionRecord, - values: Record + values: Record, + /** Ids the provider reported back. Omitted means every value is a report, which + * is what a surface that only ever learns values by reading them sends. */ + confirmed?: readonly string[] ): boolean { + const sourceFor = (id: string): TrackedNativeChatSessionOption['source'] => + confirmed === undefined || confirmed.includes(id) ? 'reported' : 'dispatched' const modelId = typeof values.model === 'string' ? values.model : null if (!modelId) { return false } const modelChanged = record.model?.value !== modelId - let changed = modelChanged || record.model?.source !== 'reported' - record.model = { value: modelId, source: 'reported' } + let changed = modelChanged || record.model?.source !== sourceFor('model') + record.model = { value: modelId, source: sourceFor('model') } const modelValues = modelChanged ? {} : { ...record.valuesByModel[modelId] } for (const [id, value] of Object.entries(values)) { if (id === 'model') { continue } const current = modelValues[id] - if (current?.value !== value || current.source !== 'reported') { + if (current?.value !== value || current.source !== sourceFor(id)) { changed = true } - modelValues[id] = { value, source: 'reported' } + modelValues[id] = { value, source: sourceFor(id) } } record.valuesByModel[modelId] = modelValues return changed diff --git a/src/shared/native-chat-session-options.ts b/src/shared/native-chat-session-options.ts index 84b23dcba06..33567d39131 100644 --- a/src/shared/native-chat-session-options.ts +++ b/src/shared/native-chat-session-options.ts @@ -7,9 +7,17 @@ export type SessionOptionSelectChoice = { } /** `default` is the catalog's own value shown before anything is observed — - * truthful to display, but never evidence about a running agent. */ + * truthful to display, but never evidence about a running agent. `dispatched` + * is sent-but-unread: the pill shows it, and a later report that disagrees is + * what corrects it. Both transports emit it, so it alone names neither — see + * `transport` on the descriptor. */ export type SessionOptionValueSource = 'applied' | 'dispatched' | 'reported' | 'default' | 'unknown' +/** How a live value reaches the agent. `catalog` types the catalog's command into + * the agent's terminal and can only learn the outcome by parsing the screen back; + * `agent-session` writes over the structured protocol, which reports every turn. */ +export type NativeChatLiveOptionTransport = 'catalog' | 'agent-session' + /** Closed set of reasons an option is not settable in the current mode. A key * (not free English) so the producer and the localized label stay in sync — * an exhaustive switch turns any drift into a type error instead of leaking @@ -34,6 +42,9 @@ export type SessionOptionDescriptor = { currentValue?: boolean } valueSource: SessionOptionValueSource + /** Required so a new producer cannot inherit the wrong lane's rendering by + * omission — `dispatched` is emitted identically by both and cannot discriminate. */ + transport: NativeChatLiveOptionTransport settable: boolean disabledReason?: SessionOptionDisabledReason /** Why: picker-only and toggle-only PTY commands cannot be represented as @@ -41,6 +52,15 @@ export type SessionOptionDescriptor = { action?: { type: 'agent-picker' | 'toggle-command' } } +/** A value we typed at the agent and have never read back. Only the terminal + * transport can be in this state: the structured lane's own per-turn report is + * what moves a value off `dispatched`, and until it lands nothing else has. */ +export function sessionOptionDispatchUnconfirmed( + descriptor: Pick +): boolean { + return descriptor.valueSource === 'dispatched' && descriptor.transport === 'catalog' +} + export type SessionOptionSetResult = { snapshot: SessionOptionDescriptor[] } diff --git a/src/shared/node-cli-command-resolution.ts b/src/shared/node-cli-command-resolution.ts index 6931cc8fc7e..c7407ac5eb1 100644 --- a/src/shared/node-cli-command-resolution.ts +++ b/src/shared/node-cli-command-resolution.ts @@ -1,6 +1,7 @@ import { accessSync, constants, existsSync, readFileSync, readdirSync, statSync } from 'node:fs' import { homedir } from 'node:os' import { delimiter, dirname, isAbsolute, join } from 'node:path' +import { getSystemCliInstallDirectories } from './system-cli-install-dirs' type ResolveCommandOptions = { pathEnv?: string | null @@ -245,6 +246,17 @@ function getVersionManagerDirectories( return directories } +// Why one list for both resolvers: the system block must stay LAST so a +// version-manager install always outranks a Homebrew/npm/snap one, and two +// hand-spelled spreads is how the native and WSL lists drifted apart before. +function getCliInstallDirectories(platform: NodeJS.Platform, homePath: string): string[] { + return [ + ...getNvmVersionDirectories(homePath), + ...getBaseVersionManagerDirectories(platform, homePath), + ...getSystemCliInstallDirectories(platform, homePath) + ] +} + export function resolveCliCommand( commandName: string, options: ResolveCommandOptions = {} @@ -258,19 +270,12 @@ export function resolveCliCommand( } const homePath = options.homePath ?? homedir() - const nvmCandidate = findFirstExecutable( + const installCandidate = findFirstExecutable( platform, - getNvmVersionDirectories(homePath), + getCliInstallDirectories(platform, homePath), executableNames ) - const versionManagerCandidate = - nvmCandidate ?? - findFirstExecutable( - platform, - getBaseVersionManagerDirectories(platform, homePath), - executableNames - ) - return versionManagerCandidate ?? commandName + return installCandidate ?? commandName } export function resolveCliCommands( @@ -281,10 +286,7 @@ export function resolveCliCommands( const pathEnv = options.pathEnv ?? process.env.PATH ?? process.env.Path ?? null const pathDirectories = splitPath(pathEnv) const homePath = options.homePath ?? homedir() - const installDirectories = [ - ...getNvmVersionDirectories(homePath), - ...getBaseVersionManagerDirectories(platform, homePath) - ] + const installDirectories = getCliInstallDirectories(platform, homePath) const resolved = new Map() for (const commandName of new Set(commandNames)) { diff --git a/src/shared/orca-profiles.ts b/src/shared/orca-profiles.ts index 851021a0779..298cfc28312 100644 --- a/src/shared/orca-profiles.ts +++ b/src/shared/orca-profiles.ts @@ -4,6 +4,8 @@ import type { ExecutionHostId } from './execution-host' export const ORCA_PROFILE_INDEX_SCHEMA_VERSION = 1 export const DEFAULT_LOCAL_ORCA_PROFILE_ID = 'local-default' export const DEFAULT_LOCAL_ORCA_PROFILE_NAME = 'Personal' +/** Main -> renderer push when the stored auth status changed without the renderer asking. */ +export const ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL = 'orcaProfiles:authStatusChanged' const LEGACY_ORCA_BROWSER_SESSION_PARTITION_PREFIX = 'persist:orca-browser-session-' export type OrcaProfileAvatar = { diff --git a/src/shared/posix-version-manager-bin-dirs.ts b/src/shared/posix-version-manager-bin-dirs.ts index 698b0f7f5b1..85eb581ccb1 100644 --- a/src/shared/posix-version-manager-bin-dirs.ts +++ b/src/shared/posix-version-manager-bin-dirs.ts @@ -8,8 +8,20 @@ * installed, which is #9725. * * Kept in step with `getBaseVersionManagerDirectories` in - * node-cli-command-resolution.ts: a WSL user on asdf, mise, volta or fnm would - * otherwise still hit #9725 while the same user on native does not. + * node-cli-command-resolution.ts and `getSystemCliInstallDirectories` in + * system-cli-install-dirs.ts: a WSL user on asdf, mise, volta or fnm -- or on + * Linuxbrew, snap or nix -- would otherwise still hit #9725 while the same user + * on native does not. `/opt/homebrew` stays out because a WSL guest is Linux, + * where Homebrew installs to the Linuxbrew prefix below. + * + * Version-manager dirs lead and the system block trails, which took the one + * behavior change here: `/usr/local/bin` moved from before the nvm glob to + * after it, so the guest ranks a version manager over a system install the way + * native does. Bounded, not free: every entry is APPENDED behind a resolved + * login PATH, so this can only re-rank a command that BOTH consumers would + * otherwise miss, and both only test presence. Not full parity either -- the + * glob expands lexicographically, while native orders nvm dirs + * default-alias-first (#10932). * * Each entry is quoted so a `$HOME` containing a space cannot word-split into * a relative path -- except the nvm glob, where only the prefix is quoted so @@ -24,8 +36,15 @@ const POSIX_VERSION_MANAGER_BIN_DIRS = [ '"$HOME/.asdf/shims"', '"$HOME/.fnm/aliases/default/bin"', '"$HOME/.local/share/mise/shims"', + '"$HOME"/.nvm/versions/node/*/bin', '"/usr/local/bin"', - '"$HOME"/.nvm/versions/node/*/bin' + '"/snap/bin"', + '"/home/linuxbrew/.linuxbrew/bin"', + '"/nix/var/nix/profiles/default/bin"', + '"$HOME/.nix-profile/bin"', + // Why both: the opencode and Pi installers' own defaults, which no version manager owns (#829). + '"$HOME/.opencode/bin"', + '"$HOME/.vite-plus/bin"' ].join(' ') /** diff --git a/src/shared/process-table-snapshot-reader.ts b/src/shared/process-table-snapshot-reader.ts index 8e1655d400d..a40b934c0e1 100644 --- a/src/shared/process-table-snapshot-reader.ts +++ b/src/shared/process-table-snapshot-reader.ts @@ -14,11 +14,15 @@ export { PS_ARGS, PS_MAX_BUFFER_BYTES } const execFile = promisify(execFileCb) -/** Columns used by the evidence reader. Keep command last so its spaces survive parsing. */ -export const PS_TIMEOUT_MS = 3000 +// Why 15s: the `command=` column costs a per-pid argv read (measured 1.15s for 1,948 +// processes; 0.03s without it), and CPU contention multiplies that -- at load 27 the same +// capture measured 1.3-6.0s, so a 3s budget timed out on 6 of 20 consecutive tries and the +// whole subsystem answered "unverifiable" about a table it could read. This keeps a wedged +// `ps` bounded while staying out of reach of a host that is merely busy. +export const PS_TIMEOUT_MS = 15_000 const DEFAULT_SNAPSHOT_TTL_MS = 500 -type Snapshot = { value: T; capturedAtMs: number } +type Snapshot = { value: T; capturedAtMs: number; completedAtMs: number } type ProcessTableSnapshotReaderDeps = { runPs: () => Promise @@ -42,11 +46,16 @@ export function createProcessTableSnapshotReader( let freshQueued: { promise: Promise; startSequence: number | null } | null = null async function runSnapshot(): Promise { + // Two stamps because they answer different questions: `capturedAtMs` is when `ps` read the + // kernel table, which is what a destructive consumer bounds staleness against, while the TTL + // keys on completion so a capture slower than the TTL still coalesces instead of forking a + // whole-machine `ps` per caller on exactly the loaded host that can least afford it. + const capturedAtMs = deps.now() const promise = deps.runPs() inFlight = promise try { const value = await promise - cached = { value, capturedAtMs: deps.now() } + cached = { value, capturedAtMs, completedAtMs: deps.now() } return value } finally { if (inFlight === promise) { @@ -56,7 +65,7 @@ export function createProcessTableSnapshotReader( } async function getSnapshot(): Promise { - if (cached && deps.now() - cached.capturedAtMs < ttlMs) { + if (cached && deps.now() - cached.completedAtMs < ttlMs) { return cached.value } if (inFlight) { @@ -269,15 +278,55 @@ export async function getStrictProcessTableSnapshot(): Promise(pending: Promise): Promise { + let timer: ReturnType | undefined + try { + return await Promise.race([ + pending, + new Promise((_resolve, reject) => { + timer = setTimeout( + () => reject(new ProcessTableCaptureError('capture_over_budget')), + PROCESS_TABLE_EVIDENCE_BUDGET_MS + ) + }) + ]) + } finally { + clearTimeout(timer) + } +} + export async function getStrictProcessTableSnapshotWithAge(): Promise<{ rows: ProcessTableRow[] capturedAgeMs: number }> { - const snapshot = await processTableReader.getSnapshotWithAge() + const snapshot = await withEvidenceBudget(processTableReader.getSnapshotWithAge()) return { rows: snapshot.value.strict(), capturedAgeMs: snapshot.capturedAgeMs } } -/** How much older than its own await a TTL-cached capture may be. */ +/** How much older than its own await a TTL-cached capture may be, on top of the capture's own + * duration. Reported ages carry both, so this alone is not the staleness bound. */ export const PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS = DEFAULT_SNAPSHOT_TTL_MS export function resetProcessTableSnapshotForTests(): void { diff --git a/src/shared/process-table-snapshot.test.ts b/src/shared/process-table-snapshot.test.ts index 80f9ed1ab20..fea3d891b6d 100644 --- a/src/shared/process-table-snapshot.test.ts +++ b/src/shared/process-table-snapshot.test.ts @@ -1,6 +1,6 @@ import { readFileSync } from 'node:fs' import { join } from 'node:path' -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const { execFileMock } = vi.hoisted(() => ({ execFileMock: vi.fn() })) @@ -10,7 +10,10 @@ import { createProcessTableSnapshotReader, getProcessTableSnapshot, getStrictProcessTableSnapshot, + getStrictProcessTableSnapshotWithAge, + PROCESS_TABLE_EVIDENCE_BUDGET_MS, PS_MAX_BUFFER_BYTES, + PS_TIMEOUT_MS, resetProcessTableSnapshotForTests } from './process-table-snapshot-reader' import { @@ -124,6 +127,26 @@ describe('process-table-snapshot reader', () => { expect(scans).toBe(1) }) + it('reports the age of the capture instant, not of the moment ps finished', async () => { + let clock = 0 + const gate = deferred() + const reader = createProcessTableSnapshotReader({ + runPs: () => gate.promise, + now: () => clock, + ttlMs: 500 + }) + + const first = reader.getSnapshot() + // A capture that took 4s of wall clock describes the machine as it was 4s ago. + clock = 4_000 + gate.resolve('scan-1') + await first + + // Age used to start at the callback, so a capture older than the destructive + // consumer's ceiling was admitted as if it had just been taken. + expect((await reader.getSnapshotWithAge()).capturedAgeMs).toBe(4_000) + }) + it('does not cache failures and retries on the next call', async () => { let scans = 0 const reader = createProcessTableSnapshotReader({ @@ -599,4 +622,118 @@ describe('process-table capture completeness', () => { await expect(getProcessTableSnapshot()).rejects.toThrow(ProcessTableCaptureError) await expect(getProcessTableSnapshot()).rejects.toThrow('empty_capture') }) + + /** + * Emulate Node's own execFile deadline: past `timeout` it SIGTERMs the child and calls + * back with `Command failed: ` and no stderr -- indistinguishable from a broken + * `ps` unless the budget itself is under test. + */ + function mockPsTakingMs(durationMs: number, stdout: string): void { + execFileMock.mockImplementation( + (command: string, args: string[], options: unknown, callback: unknown) => { + const done = callback as (err: unknown, result: { stdout: string; stderr: string }) => void + const timeout = (options as { timeout?: number })?.timeout + if (timeout !== undefined && durationMs > timeout) { + const error = Object.assign( + new Error(`Command failed: ${[command, ...args].join(' ')}\n`), + { code: null, killed: true, signal: 'SIGTERM' } + ) + done(error, { stdout: '', stderr: '' }) + return + } + done(null, { stdout, stderr: '' }) + } + ) + } + + it('reads a loaded host whose whole-machine ps runs for seconds', async () => { + // Measured on a 1,948-process host at load 27: the `command=` argv read alone costs + // 1.15s, and contention stretched the same capture to 6.0s. The old 3s budget killed + // 6 of 20 consecutive captures, so a readable table answered "unverifiable". + mockPsTakingMs(6_000, busyHostTable(200)) + + expect(PS_TIMEOUT_MS).toBeGreaterThan(6_000) + await expect(getProcessTableSnapshot()).resolves.toHaveLength(200) + }) + + it('still bounds a ps that never returns', async () => { + mockPsTakingMs(PS_TIMEOUT_MS + 1, busyHostTable(200)) + + await expect(getProcessTableSnapshot()).rejects.toThrow('Command failed: ps') + }) +}) + +/** + * The evidence-publishing read has a far shorter budget than identity proof, and the two are not + * interchangeable. Identity proof asks whether a process exists and must not read a slow capture + * as an absent one, so it waits out `PS_TIMEOUT_MS`. These consumers ask whether an observation + * describes NOW: past this budget it cannot, so the answer they need is a prompt `unverifiable`, + * which both relay call sites already produce from a rejection. + */ +describe('evidence-publishing capture budget', () => { + beforeEach(() => { + execFileMock.mockReset() + resetProcessTableSnapshotForTests() + vi.useFakeTimers() + }) + + afterEach(() => { + vi.useRealTimers() + }) + + /** A `ps` that answers after `durationMs` of wall clock, the way a loaded host does. */ + function mockPsTaking(durationMs: number, stdout = '1 0 1 1 S+ ?? Jan 1 00:00:00 2026 bash\n') { + execFileMock.mockImplementation( + (_command: string, _args: string[], _options: unknown, callback: unknown) => { + const done = callback as (err: unknown, result: { stdout: string; stderr: string }) => void + setTimeout(() => done(null, { stdout, stderr: '' }), durationMs) + } + ) + } + + it('answers from a capture that lands one tick inside the budget', async () => { + mockPsTaking(PROCESS_TABLE_EVIDENCE_BUDGET_MS - 1) + const pending = getStrictProcessTableSnapshotWithAge() + + await vi.advanceTimersByTimeAsync(PROCESS_TABLE_EVIDENCE_BUDGET_MS) + + // The age carries the capture's own duration, which is the whole reason the budget is this + // far under the 2,000ms admission ceiling rather than under PS_TIMEOUT_MS. + expect((await pending).capturedAgeMs).toBe(PROCESS_TABLE_EVIDENCE_BUDGET_MS - 1) + }) + + it('gives up on a capture one tick past the budget instead of waiting out PS_TIMEOUT_MS', async () => { + mockPsTaking(PROCESS_TABLE_EVIDENCE_BUDGET_MS + 1) + const pending = getStrictProcessTableSnapshotWithAge() + const settled = pending.then( + () => 'resolved', + (error: Error) => error.message + ) + + await vi.advanceTimersByTimeAsync(PROCESS_TABLE_EVIDENCE_BUDGET_MS) + + expect(await settled).toContain('process table unreadable') + }) + + it('leaves the abandoned capture running to fill the cache for the next reader', async () => { + // Why this matters: a whole-machine `ps` is the most expensive thing the host does, and the + // host that blows the budget is by definition the one that can least afford a second one. + mockPsTaking(PROCESS_TABLE_EVIDENCE_BUDGET_MS + 100) + const abandoned = getStrictProcessTableSnapshotWithAge().catch(() => null) + + await vi.advanceTimersByTimeAsync(PROCESS_TABLE_EVIDENCE_BUDGET_MS) + expect(await abandoned).toBeNull() + + await vi.advanceTimersByTimeAsync(100) + expect(await getStrictProcessTableSnapshotWithAge()).toMatchObject({ + capturedAgeMs: PROCESS_TABLE_EVIDENCE_BUDGET_MS + 100 + }) + expect(execFileMock).toHaveBeenCalledTimes(1) + }) + + it('keeps the identity budget far above its own', () => { + // A slow capture must still not read as an absent process; only the consumers that need the + // observation to describe NOW gave up waiting for it. + expect(PROCESS_TABLE_EVIDENCE_BUDGET_MS).toBeLessThan(PS_TIMEOUT_MS) + }) }) diff --git a/src/shared/protocol-version.ts b/src/shared/protocol-version.ts index ed235094473..d78c6223c99 100644 --- a/src/shared/protocol-version.ts +++ b/src/shared/protocol-version.ts @@ -126,6 +126,10 @@ export const AGENT_SESSION_OMP_RESUME_PATH_RUNTIME_CAPABILITY = // can neither display nor drive. The host also refuses every agentSession.* // method from a connection that does not advertise this. export const STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY = 'agent-session.structured.v1' as const +// Why: mobile clients advertise Claude-structured session support during E2EE +// pairing so the desktop can keep structured-specific affordances enabled. +export const CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY = + 'agent-session.structured.claude.v1' as const // Why: paired structured clients explicitly hold every visible session surface, allowing the host // to stop provider children after the last surface closes without tying lifetime to a transport. export const STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY = diff --git a/src/shared/relay-host-close-reason.ts b/src/shared/relay-host-close-reason.ts new file mode 100644 index 00000000000..460aa704a34 --- /dev/null +++ b/src/shared/relay-host-close-reason.ts @@ -0,0 +1,19 @@ +// Why a WebSocket close reason and not a JSON field: every relay-hello and +// director-resolve schema on the phone is zod `.strict()`, so an added key is a +// hard parse failure on already-shipped phones. The close reason is a wire slot +// old peers never read, which makes it the only additive channel here. +export const RELAY_HOST_CLOSE_REASON = { + // The desktop lost its Orca Cloud session (cleared, or refused with 401). + SIGNED_OUT: 'signed-out' +} as const + +export type RelayHostCloseReason = + (typeof RELAY_HOST_CLOSE_REASON)[keyof typeof RELAY_HOST_CLOSE_REASON] + +const REASONS: readonly string[] = Object.values(RELAY_HOST_CLOSE_REASON) + +// Close reasons are attacker-adjacent free text; only exact known members count. +export function relayHostCloseReasonFrom(value: unknown): RelayHostCloseReason | null { + const text = typeof value === 'string' ? value : (value?.toString() ?? '') + return REASONS.includes(text) ? (text as RelayHostCloseReason) : null +} diff --git a/src/shared/remote-foreground-evidence-admission.ts b/src/shared/remote-foreground-evidence-admission.ts index 335d724d7b8..b9f060a346b 100644 --- a/src/shared/remote-foreground-evidence-admission.ts +++ b/src/shared/remote-foreground-evidence-admission.ts @@ -40,7 +40,16 @@ export function admitRemoteForegroundEvidence( 0, admission.receivedAtMonotonic - admission.requestStartedAtMonotonic ) - if (value.capturedAgeMs + receiveDelay > REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS) { + // The larger of the two, never their sum: `ps` runs INSIDE this round trip, so its duration is + // already in `receiveDelay`, and `capturedAgeMs` -- stamped at capture start -- is that same + // duration measured on the host's clock. Adding them charged the capture twice and halved the + // budget this ceiling actually grants a host, from ~2.0s of `ps` to ~1.0s: a 1.2s capture + // stamped 1200 and arrived at 1300, summed to 2500, and was refused as too old at 1.3s. + // + // Not the same shape as the sweep's gate, which sums deliberately and correctly: + // `evidenceAgeSinceListingMs` is stamped AFTER the listing arrives, so it measures only + // planning time and overlaps nothing. + if (Math.max(value.capturedAgeMs, receiveDelay) > REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS) { return null } if ( diff --git a/src/shared/remote-foreground-evidence.test.ts b/src/shared/remote-foreground-evidence.test.ts index 10b9371b7c0..2fc1f8f93fc 100644 --- a/src/shared/remote-foreground-evidence.test.ts +++ b/src/shared/remote-foreground-evidence.test.ts @@ -59,14 +59,82 @@ describe('remote foreground evidence contract', () => { expect( admitRemoteForegroundEvidence(live, { ...base, expectedIncarnationId: 'inc-2' }) ).toBeNull() + // `+ 1`, not the ceiling itself: the ceiling used to be crossed by the ceiling plus this + // admission's 10ms round trip, which was the capture being charged a second time. expect( admitRemoteForegroundEvidence( - { ...live, capturedAgeMs: REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS }, + { ...live, capturedAgeMs: REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1 }, base ) ).toBeNull() }) + // The capture runs inside the round trip, so `capturedAgeMs` and `receiveDelay` measure + // overlapping intervals on two clocks. Summing them charged `ps` twice and halved the budget + // this ceiling grants a host; these pin the boundary on the surviving measurement. + describe('does not charge the capture twice', () => { + const admission = { + expectedPtyId: 'pty-1', + expectedIncarnationId: 'inc-1', + requestStartedAtMonotonic: 0, + receivedAtMonotonic: 0, + lastAuthorityGeneration: 'host-a', + lastObservationEpoch: 3 + } + + it('admits a capture whose duration alone is inside the ceiling', () => { + // 1,200ms on the host, 1,300ms round trip: the same 1.2s of `ps` seen twice. The sum said + // 2,500 and refused it; the observation is 1.3s old. + expect( + admitRemoteForegroundEvidence( + { ...live, capturedAgeMs: 1_200 }, + { ...admission, receivedAtMonotonic: 1_300 } + ) + ).not.toBeNull() + }) + + it.each([ + ['host-measured', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS - 1, 0], + ['client-measured', 0, REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS - 1] + ])('admits at the %s boundary', (_label, capturedAgeMs, receiveDelay) => { + expect( + admitRemoteForegroundEvidence( + { ...live, capturedAgeMs }, + { ...admission, receivedAtMonotonic: receiveDelay } + ) + ).not.toBeNull() + }) + + it.each([ + ['host-measured', REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1, 0], + ['client-measured', 0, REMOTE_FOREGROUND_EVIDENCE_MAX_AGE_MS + 1] + ])('refuses one step past the %s boundary', (_label, capturedAgeMs, receiveDelay) => { + expect( + admitRemoteForegroundEvidence( + { ...live, capturedAgeMs }, + { ...admission, receivedAtMonotonic: receiveDelay } + ) + ).toBeNull() + }) + + it('still admits the prompt unverifiable a capture over its budget produces', () => { + // The reason the evidence path gives up on a slow capture rather than publishing a late + // truthful one: a REFUSED record is what bumps `consecutiveInspectionErrors` and stalls the + // completion poller, while an admitted `unverifiable` costs a poll and nothing else. + expect( + admitRemoteForegroundEvidence( + { + ...live, + verdict: 'unverifiable' as const, + reason: 'process_table_unreadable', + capturedAgeMs: 0 + }, + admission + ) + ).not.toBeNull() + }) + }) + it('rejects delayed observations from a previously accepted host generation', () => { const knownAuthorityGenerations = new Set(['host-a', 'host-b']) const admission = { diff --git a/src/shared/runtime-mobile-session-tab-contracts.ts b/src/shared/runtime-mobile-session-tab-contracts.ts index d43c43dd4e0..01a7b1ba4da 100644 --- a/src/shared/runtime-mobile-session-tab-contracts.ts +++ b/src/shared/runtime-mobile-session-tab-contracts.ts @@ -93,7 +93,7 @@ export type RuntimeMobileSessionAgentTab = { id: string title: string sessionId: string - agent: 'codex' + agent: 'claude' | 'codex' color?: string | null isPinned?: boolean isActive: boolean diff --git a/src/shared/ssh-relay-pty-ownership-proof.test.ts b/src/shared/ssh-relay-pty-ownership-proof.test.ts index b17f96808f9..92e523d3e25 100644 --- a/src/shared/ssh-relay-pty-ownership-proof.test.ts +++ b/src/shared/ssh-relay-pty-ownership-proof.test.ts @@ -174,6 +174,52 @@ describe('planRelayPtySweep', () => { expect(reasonFor(plan, 'pty-1')).toBe('host foreground observation is malformed') }) + // The kill gate's own boundary. Unlike the renderer's admission this sum is correct -- + // `evidenceAgeSinceListingMs` is stamped after the listing ARRIVES, so it measures planning time + // and overlaps the host's capture window not at all. + it.each([ + ['at the ceiling', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, 0], + ['at the ceiling once planning is counted', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS - 10, 10] + ])('still sweeps on an observation %s', (_label, capturedAgeMs, evidenceAgeSinceListingMs) => { + const plan = planRelayPtySweep( + [orphan({ foregroundProcessEvidence: { ...idleShell(), capturedAgeMs } })], + context({ evidenceAgeSinceListingMs }) + ) + + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) + + it.each([ + ['the capture alone', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS + 1, 0], + ['the capture plus planning', RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, 1] + ])('refuses the stop one step past the ceiling on %s', (_label, capturedAgeMs, sinceListing) => { + const plan = planRelayPtySweep( + [orphan({ foregroundProcessEvidence: { ...idleShell(), capturedAgeMs } })], + context({ evidenceAgeSinceListingMs: sinceListing }) + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe( + 'host foreground observation is too old to authorize a stop' + ) + }) + + it('refuses the stop on a capture slow enough to be worth waiting out', () => { + // Measured `ps` with PS_ARGS: 2.5-9.0s on a 2,002-process laptop, 4.0-18.6s at load 46. This + // gate tolerates the fast end and refuses the rest, so waiting out a slow capture cannot buy a + // sweep -- which is why the evidence path gives up on one instead of blocking a connect for it. + for (const capturedAgeMs of [6_140, 9_010, 18_600]) { + const plan = planRelayPtySweep( + [orphan({ foregroundProcessEvidence: { ...idleShell(), capturedAgeMs } })], + context() + ) + expect(plan.sweep, `capturedAgeMs=${capturedAgeMs}`).toEqual([]) + expect(reasonFor(plan, 'pty-1'), `capturedAgeMs=${capturedAgeMs}`).toBe( + 'host foreground observation is too old to authorize a stop' + ) + } + }) + it('never sweeps on an evidence record whose other host stamps are malformed', () => { const plan = planRelayPtySweep( [ diff --git a/src/shared/structured-agent-session-composer.test.ts b/src/shared/structured-agent-session-composer.test.ts new file mode 100644 index 00000000000..38fed5ddbc1 --- /dev/null +++ b/src/shared/structured-agent-session-composer.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from 'vitest' +import { + isStructuredAgentSessionComposerCommand, + structuredSlashCommands +} from './structured-agent-session-composer' + +describe('structuredSlashCommands', () => { + // The composer menu and the dispatcher read this one list. When they disagreed, + // a Claude session was offered Codex-only tokens that missed the command guard + // and reached the model as literal prompt text instead of erroring. + it.each(['codex', 'claude'] as const)('offers %s only commands it also accepts', (agent) => { + const offered = structuredSlashCommands(agent) + expect(offered.length).toBeGreaterThan(0) + for (const command of offered) { + expect(isStructuredAgentSessionComposerCommand(`/${command.name}`, agent)).toBe(true) + } + }) + + it('offers each agent its own catalog', () => { + const claude = structuredSlashCommands('claude').map((command) => command.name) + expect(claude).toContain('compact') + expect(claude).not.toContain('vim') + expect(structuredSlashCommands('codex').map((command) => command.name)).toContain('vim') + }) + + it('offers effort to every structured agent', () => { + for (const agent of ['codex', 'claude'] as const) { + expect(structuredSlashCommands(agent).map((command) => command.name)).toContain('effort') + } + }) +}) diff --git a/src/shared/structured-agent-session-composer.ts b/src/shared/structured-agent-session-composer.ts index 420651aba5b..18bdeab001d 100644 --- a/src/shared/structured-agent-session-composer.ts +++ b/src/shared/structured-agent-session-composer.ts @@ -35,7 +35,10 @@ function commandParts(text: string): { name: string; argument: string } | null { return match ? { name: match[1]!.toLowerCase(), argument: match[2]?.trim() ?? '' } : null } -function structuredSlashCommands(agent: AgentType): readonly SlashCommandSuggestion[] { +/** The command catalog a structured session offers and accepts. The composer menu + * and the dispatcher must read the same list, or a menu pick falls through the + * command guard and reaches the model as literal prompt text. */ +export function structuredSlashCommands(agent: AgentType): readonly SlashCommandSuggestion[] { if (agent === 'codex') { return STRUCTURED_AGENT_SESSION_SLASH_COMMANDS } diff --git a/src/shared/structured-agent-session-mutation.ts b/src/shared/structured-agent-session-mutation.ts index ccc80475c93..79c82095f29 100644 --- a/src/shared/structured-agent-session-mutation.ts +++ b/src/shared/structured-agent-session-mutation.ts @@ -26,6 +26,33 @@ export function structuredAgentSessionPayloadFingerprint(input: { return Array.from(bytes, (byte) => byte.toString(16).padStart(2, '0')).join('') } +export function structuredAgentSessionCreateFingerprint(input: { + sessionId: string + worktree: string + agent: 'claude' | 'codex' +}): string { + return structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: input.sessionId, + fields: { + worktree: input.worktree, + agent: input.agent + } + }) +} + +export function showStructuredAgentSessionChoice(input: { + hostCapability: boolean + workspaceSupport: boolean + agent: string +}): boolean { + return ( + input.hostCapability && + input.workspaceSupport && + (input.agent === 'claude' || input.agent === 'codex') + ) +} + export function createStructuredAgentSessionOperationId( randomUuid: () => string, now: number = Date.now() diff --git a/src/shared/structured-agent-session-options.test.ts b/src/shared/structured-agent-session-options.test.ts index 72f817c7646..0f21efadd98 100644 --- a/src/shared/structured-agent-session-options.test.ts +++ b/src/shared/structured-agent-session-options.test.ts @@ -49,8 +49,12 @@ describe('structured agent session options', () => { models: CODEX_SESSION_OPTION_CATALOG.models, record: bridgeRecord, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) + // Same catalog, same `dispatched` vocabulary — only the transport separates them. + expect(structured.every((descriptor) => descriptor.transport === 'agent-session')).toBe(true) + expect(bridge.every((descriptor) => descriptor.transport === 'catalog')).toBe(true) expect(bridge[0]).toMatchObject({ action: { type: 'agent-picker' } }) expect(bridge.find((descriptor) => descriptor.id === 'effort')).toMatchObject({ action: { type: 'agent-picker' } diff --git a/src/shared/structured-agent-session-options.ts b/src/shared/structured-agent-session-options.ts index 94f5f45351a..d746a52025c 100644 --- a/src/shared/structured-agent-session-options.ts +++ b/src/shared/structured-agent-session-options.ts @@ -76,10 +76,14 @@ export function applyStructuredAgentSessionOptions( seed: AgentSessionOptionCatalog, result: AgentSessionOptionsResult ): StructuredAgentSessionOptionState { - applyNativeChatReportedSessionOptions(state.record, { - model: result.current.model, - ...(result.current.effort ? { effort: result.current.effort } : {}) - }) + applyNativeChatReportedSessionOptions( + state.record, + { + model: result.current.model, + ...(result.current.effort ? { effort: result.current.effort } : {}) + }, + result.current.confirmed ?? [] + ) return { ...state, catalog: structuredAgentSessionOptionCatalog(seed, result) } } diff --git a/src/shared/system-cli-install-dirs.ts b/src/shared/system-cli-install-dirs.ts new file mode 100644 index 00000000000..fe070e761e0 --- /dev/null +++ b/src/shared/system-cli-install-dirs.ts @@ -0,0 +1,60 @@ +import { join } from 'node:path' + +/** + * Where an agent CLI lands when no version manager installed it: Homebrew (both + * prefixes), npm's default global prefix, snap, nix, or the CLI's own installer + * (#829 named `~/.opencode/bin` and `~/.vite-plus/bin` as the motivating cases, + * but only for the login-shell probe; the fallback used when that probe fails + * never gained either). + * + * Ordered to match the system block `patchPackagedProcessPath` appends to PATH, + * so a CLI present in two of *these* dirs resolves to the same binary here, in + * the packaged PATH scan, and in `POSIX_VERSION_MANAGER_BIN_DIRS`. That parity + * stops at the block boundary and is not claimed across it: the seed appends + * `~/.local/bin` after this block, while here it arrives ahead of it from + * `getBaseVersionManagerDirectories`, so a `claude` installed in both + * `~/.local/bin` and `/opt/homebrew/bin` resolves to the former via this + * fallback and the latter via the seeded PATH. Pre-existing, and left alone + * because closing it means hoisting a system dir over a version-manager one. + * + * Deliberate gaps vs that seed: the `sbin` dirs, the generic `~/bin`, and + * `/opt/homebrew` off darwin -- the seed does push that prefix on every posix, + * but Linux Homebrew installs to the Linuxbrew prefix below, so off darwin it + * is a directory no brew install can occupy. + * + * Lookup-only, deliberately outside `getBaseVersionManagerDirectories`: that + * list is PREPENDED to PATH by `getVersionManagerBinPaths` callers, and hoisting + * a system dir over the inherited PATH re-ranks binaries the user already has + * (#18234). One bounded exception: when a hit here ships a sibling `node`, + * `withCliRuntimeOnPath` prepends that dir onto the *spawned child's* PATH + * (#10932 runtime pairing) -- only for a command PATH did not contain at all. + */ +export function getSystemCliInstallDirectories( + platform: NodeJS.Platform, + homePath: string +): string[] { + // Why nothing here: the PATH seed's system block is POSIX-only too, so + // Windows installs outside a version manager (`%USERPROFILE%\.opencode\bin`) + // have never had install-dir coverage in either list. Unchanged, not fixed. + if (platform === 'win32') { + return [] + } + const directories: string[] = [] + if (platform === 'darwin') { + // Apple Silicon Homebrew; Intel Homebrew shares /usr/local with npm's prefix. + directories.push('/opt/homebrew/bin') + } + directories.push('/usr/local/bin') + if (platform === 'linux') { + // Gated like the seed: snap and Linuxbrew ship on Linux only, so elsewhere they are phantom stats. + directories.push('/snap/bin', '/home/linuxbrew/.linuxbrew/bin') + } + directories.push( + '/nix/var/nix/profiles/default/bin', + join(homePath, '.nix-profile', 'bin'), + // Why both: the opencode and Pi installers' own defaults, which no version manager owns (#829). + join(homePath, '.opencode', 'bin'), + join(homePath, '.vite-plus', 'bin') + ) + return directories +} diff --git a/src/shared/worktree/host-context-labels.test.ts b/src/shared/worktree/host-context-labels.test.ts new file mode 100644 index 00000000000..ccea06a25ed --- /dev/null +++ b/src/shared/worktree/host-context-labels.test.ts @@ -0,0 +1,53 @@ +import { describe, expect, it } from 'vitest' +import type { ExecutionHostId } from '../execution-host' +import { getMixedHostContextLabels } from './host-context-labels' + +type Item = { id: string; hostId: ExecutionHostId } + +function argsFor(counters: { hostIdReads: number; identityReads: number }) { + return { + getHostId: (item: Item): ExecutionHostId => { + counters.hostIdReads += 1 + return item.hostId + }, + getIdentity: (item: Item): string => { + counters.identityReads += 1 + return item.id + } + } +} + +describe('getMixedHostContextLabels', () => { + it('returns undefined without building any labels when every item shares one host', () => { + const items: Item[] = Array.from({ length: 50 }, (_, index) => ({ + id: `wt-${index}`, + hostId: 'local' as ExecutionHostId + })) + const counters = { hostIdReads: 0, identityReads: 0 } + + expect(getMixedHostContextLabels(items, argsFor(counters))).toBeUndefined() + // The label map was built for all 50 and thrown away before; now nothing is built. + expect(counters.identityReads).toBe(0) + }) + + it('labels every item once a second host appears, wherever it appears', () => { + for (const lateIndex of [1, 25, 49]) { + const items: Item[] = Array.from({ length: 50 }, (_, index) => ({ + id: `wt-${index}`, + hostId: (index === lateIndex ? 'ssh:other' : 'local') as ExecutionHostId + })) + const counters = { hostIdReads: 0, identityReads: 0 } + + const labels = getMixedHostContextLabels(items, argsFor(counters)) + + expect(labels, `second host at index ${lateIndex}`).toBeDefined() + expect(labels?.size).toBe(50) + expect(labels?.has(`wt-${lateIndex}`)).toBe(true) + } + }) + + it('returns undefined for an empty list', () => { + const counters = { hostIdReads: 0, identityReads: 0 } + expect(getMixedHostContextLabels([], argsFor(counters))).toBeUndefined() + }) +}) diff --git a/src/shared/worktree/host-context-labels.ts b/src/shared/worktree/host-context-labels.ts index 4b9481c441b..d3ab9ed2ed7 100644 --- a/src/shared/worktree/host-context-labels.ts +++ b/src/shared/worktree/host-context-labels.ts @@ -83,12 +83,30 @@ export function getMixedHostContextLabels( sources?: HostContextLabelSources } ): Map | undefined { - const labelsByIdentity = new Map() - const hostIds = new Set() + // Why two passes: on a single-host install the map is built for every item and then discarded. + // Deciding first costs one getHostId per item and skips the label work entirely. + let firstHostId: ExecutionHostId | undefined + let isMixed = false for (const item of items) { const hostId = args.getHostId(item) - hostIds.add(hostId) - labelsByIdentity.set(args.getIdentity(item), getHostContextLabel(hostId, args.sources)) + if (firstHostId === undefined) { + firstHostId = hostId + continue + } + if (hostId !== firstHostId) { + isMixed = true + break + } } - return hostIds.size > 1 ? labelsByIdentity : undefined + if (!isMixed) { + return undefined + } + const labelsByIdentity = new Map() + for (const item of items) { + labelsByIdentity.set( + args.getIdentity(item), + getHostContextLabel(args.getHostId(item), args.sources) + ) + } + return labelsByIdentity } diff --git a/src/shared/worktree/types.ts b/src/shared/worktree/types.ts index 3e5c05a65bc..e368716dc01 100644 --- a/src/shared/worktree/types.ts +++ b/src/shared/worktree/types.ts @@ -221,4 +221,6 @@ export type DetectedWorktreeListResult = { authoritative: boolean source: DetectedWorktreeListSource worktrees: DetectedWorktree[] + /** Why a non-authoritative listing could not be scanned; additive, older hosts omit it. */ + unavailableReason?: string } diff --git a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts index d79c99b679d..2ad8bd0647f 100644 --- a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts +++ b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts @@ -77,6 +77,11 @@ const STRUCTURED_CALLS: { hostMethod: 'setOption', result: { ok: true, replayed: false } }, + { + method: 'agentSession.requestHandoff', + hostMethod: 'requestHandoff', + result: { status: { owner: 'native' } } + }, { method: 'agentSession.handoffStatus', hostMethod: 'handoffStatus', @@ -197,6 +202,14 @@ function paramsFor(method: string): unknown { const fields = { itemId: 'item-1', expectedRevision: 1, optionId: 'allow' } return { envelope: envelope({ method, fields, fence }), ...fields } } + case 'agentSession.requestHandoff': { + const fields = { + direction: 'to-tui' as const, + mode: 'now' as const, + action: 'start' as const + } + return { envelope: envelope({ method, fields, fence }), ...fields } + } case 'agentSession.setOption': { const fields = { key: 'model', value: 'gpt-5' } return { envelope: envelope({ method, fields, fence }), ...fields } @@ -286,6 +299,10 @@ async function callBuild( function structuredHostStub(): Record> { return { attach: vi.fn(async () => ({ ok: true, replayed: false, value: { sessionId: SESSION } })), + // Attach-shaped entries take a client-supplied location, so the host is asked whether it + // supports creating there. A real host always answers; leaving it unstubbed made every + // `ensure` refuse for the harness's own reason rather than the location's. + supportsCreate: vi.fn(() => true), send: vi.fn(async () => ({ ok: true, replayed: false })), cancel: vi.fn(async () => ({ ok: true, replayed: false })), close: vi.fn(async () => undefined), @@ -723,6 +740,10 @@ describe('cross-version structured agent sessions', () => { /** Phase 2 owns provider processes; the adapter is the only stub here. */ function adapter(): StructuredAgentSessionAdapter { return { + // Every real adapter answers this; without it adapterSupportsCreate falls through to + // `supportsLocation`, which this fake also lacks, so the client-supplied-location gate + // refused for the fake's silence rather than for the location. + supportsCreate: () => true, acquire: async ({ fence }) => ({ process: { hostId: 'local', diff --git a/tests/e2e/new-workspace-create-more.spec.ts b/tests/e2e/new-workspace-create-more.spec.ts new file mode 100644 index 00000000000..b166d61dfe5 --- /dev/null +++ b/tests/e2e/new-workspace-create-more.spec.ts @@ -0,0 +1,106 @@ +import { writeFileSync } from 'node:fs' +import { execFileSync } from 'node:child_process' +import { test, expect } from './helpers/orca-app' +import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' + +test.use({ orcaAppExtraEnv: { ORCA_BACKGROUND_LAUNCH: '1' } }) + +test('Create more clears the GitHub PR source before the next worktree', async ({ + electronApp, + orcaPage, + testRepoPath +}, testInfo) => { + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const sha = execFileSync('git', ['rev-parse', 'HEAD'], { + cwd: testRepoPath, + encoding: 'utf8' + }).trim() + await electronApp.evaluate(({ ipcMain }, baseBranch) => { + ipcMain.removeHandler('worktrees:resolvePrBase') + ipcMain.handle('worktrees:resolvePrBase', () => ({ baseBranch })) + }, sha) + await orcaPage.evaluate(() => { + const store = window.__store! + const state = store.getState() + store.setState({ settings: { ...state.settings!, defaultTuiAgent: 'blank' } }) + }) + await orcaPage.getByRole('button', { name: 'New workspace', exact: true }).click() + await orcaPage.evaluate(() => { + const store = window.__store! + const repoId = store.getState().repos[0].id + const item = { + id: 'pr-4242', + provider: 'github' as const, + type: 'pr' as const, + number: 4242, + title: 'Fix workspace task reset', + state: 'open' as const, + url: 'https://github.com/acme/app/pull/4242', + labels: [], + updatedAt: '2026-09-01T00:00:00Z', + author: 'e2e', + repoId + } + store.setState({ + getCachedWorkItems: () => [item], + fetchWorkItems: async () => [item], + fetchWorkItemsAcrossRepos: async () => ({ + items: [item], + failedCount: 0, + githubUnavailable: false + }) + }) + }) + const dialog = orcaPage.getByRole('dialog', { name: /Create (Workspace|Worktree)/i }) + const input = dialog.locator('[data-workspace-name-input="true"]') + await input.click() + await orcaPage + .getByRole('option', { name: '#4242 Fix workspace task reset', exact: true }) + .click() + const pill = dialog.locator('[data-workspace-source-pill="true"]') + await expect(pill).toContainText('Fix workspace task reset') + await dialog.getByRole('switch', { name: 'Create more' }).click() + await dialog.getByRole('button', { name: /^Create/ }).click() + await expect(dialog).toBeVisible() + await expect(input).toHaveValue('') + await expect + .poll(() => + orcaPage.evaluate(() => + window + .__store!.getState() + .allWorktrees() + .some((worktree) => worktree.linkedPR === 4242) + ) + ) + .toBe(true) + const cdp = await orcaPage.context().newCDPSession(orcaPage) + const screenshot = await cdp.send('Page.captureScreenshot') + const proofPath = testInfo.outputPath('create-more-result.png') + writeFileSync(proofPath, Buffer.from(screenshot.data, 'base64')) + await testInfo.attach('create-more-result.png', { + path: proofPath, + contentType: 'image/png' + }) + await cdp.detach() + await expect(pill).toHaveCount(0) + await expect(dialog.getByRole('switch', { name: 'Create more' })).toHaveAttribute( + 'aria-checked', + 'true' + ) + await input.fill('next-independent-worktree') + await dialog.getByRole('button', { name: /^Create/ }).click() + await expect + .poll(() => + orcaPage.evaluate(() => { + const worktree = window + .__store!.getState() + .allWorktrees() + .find((entry) => entry.displayName === 'next-independent-worktree') + return worktree ? { linkedPR: worktree.linkedPR, linkedIssue: worktree.linkedIssue } : null + }) + ) + .toEqual({ linkedPR: null, linkedIssue: null }) + await expect(input).toHaveValue('') + await expect(pill).toHaveCount(0) +})