diff --git a/README.md b/README.md index 7a3cbe2360c..2ae59035da8 100644 --- a/README.md +++ b/README.md @@ -238,9 +238,9 @@ Pair with your desktop app to monitor and steer your agents from your phone. - **Discord:** Join the community on **[Discord](https://discord.gg/fzjDKHxv8Q)**. - **Twitter / X:** Follow **[@orca_build](https://x.com/orca_build)** for updates and announcements. -- **WeChat:** Scan to join the Orca community WeChat group 8. +- **WeChat:** Scan to join the Orca community WeChat group 8. Group 8 may be full; if so, scan the Group 9 QR code instead. - WeChat group 8 QR code for the Orca community + WeChat group 8 QR code for the Orca community  WeChat group 9 QR code for the Orca community - **Feedback & Ideas:** We ship fast. Missing something? [Request a new feature](https://github.com/stablyai/orca/issues). - **Privacy:** See the [privacy & telemetry docs](https://www.onorca.dev/docs/telemetry) for what anonymous usage data Orca collects and how to opt out. diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 61a73b64dbe..4e1da9fab26 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -111,14 +111,19 @@ describe('incident monitor evaluator', () => { }) }) - it('freezes when postgres retries exceed the recalibrated ceiling', () => { - const sample = healthySample() - sample.sources['relay-logs']!.signals['relay.postgres_retries'] = - signal(INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries + 1) - expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + // Why: the global relay_cells lock made retries a steady-state rate (24 h p99 + // 1320/5min on 2026-09-04); the bar fences only unbounded growth beyond that. + it('tolerates the measured healthy retry rate and freezes above the bar', () => { + const healthy = healthySample() + healthy.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(1504) + expect(evaluateIncidentSample(healthy, startedAt).status).toBe('green') + + const incident = healthySample() + incident.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(2001) + expect(evaluateIncidentSample(incident, startedAt)).toMatchObject({ status: 'freeze', failures: [ - expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 300 }) + expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 2000 }) ] }) }) diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index 868bb86fb93..a121568d918 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -32,11 +32,20 @@ export const INCIDENT_MONITOR_THRESHOLDS = { relayPoolWaiting: 800, relayPoolWaitMs: 2_500, // Why: successful lock retries are the contention machinery working, not harm. - // Healthy 2026-08-26 baseline bursts to 234/5min (26% of windows crossed the old - // bar of 20, set unmeasured at the monitor's 2026-07-28 birth); the 2026-08-23 - // incident ran ~2,200-3,000/5min. 300 clears healthy bursts with ~10x incident - // margin; relayPostgresRetryExhausted below bounds the terminally failed share. - relayPostgresRetries: 300, + // Recalibrated 2026-09-04 from 300, which was set 2026-08-26 when healthy bursts + // reached 234/5min. The global relay_cells FOR UPDATE lock has since become the + // fleet's steady state: measured fleet-wide (director + cells, summed per five + // minutes) 2026-09-03T05Z..2026-09-04T05Z p50 430 / p90 924 / p99 1320 / max + // 1504, with 55% of windows over 300 and only 22% of 15-minute gates clean, so + // the bar blocked the very cell roll that carries the 500 ms lock wait (#18521) + // and the beginProof crash guard to the cells. The 2026-08-23 lock incident on + // this same metric peaked at 1510 in one window and 646 in the next, so it is + // not separable from today's contention by retries alone; it is caught by + // relayPostgresRetryExhausted (467 at the peak vs a 300 bar), director + // concurrency, and the pool bars. 2000 passes every healthy 15-minute window + // measured in the last 24 h and still fences unbounded growth. Re-tighten once + // the fleet is on the 500 ms lock wait and the baseline is re-measured. + relayPostgresRetries: 2000, // Why: 300 per five minutes, recalibrated 2026-09-04 from a bar of zero that no // production window has cleared since #18521 shipped to the director. That // change cut the request-path cell-inventory wait from the 1 s pool lock_timeout @@ -48,7 +57,7 @@ export const INCIDENT_MONITOR_THRESHOLDS = { // quiet hours p50 2 / max 36; pre-#18521 daytime p50 10 / p90 25 / max 87; // post-#18521 p50 42 / p90 147 / max 220. The 2026-08-23 lock incident peaked // at 467. 300 clears every measured healthy window and still sits below the - // incident shape; relayPostgresRetries above stays the ~10x discriminator. + // incident shape; retries above fence only unbounded growth. // User-facing /v1/assign 503 share did not move with #18521 (13.9% old image // vs 12.3% new, same evening), so exhaustion is not a proxy for user harm. relayPostgresRetryExhausted: 300, diff --git a/cloud/apps/relay-ops/src/resource-inventory.test.ts b/cloud/apps/relay-ops/src/resource-inventory.test.ts index 6d8b3070c98..4bf43fe9a7a 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.test.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.test.ts @@ -23,8 +23,10 @@ describe('readResourceInventory', () => { calls += 1 return new Response(null, { status: 200 }) }, - async () => { - waits += 1 + { + wait: async () => { + waits += 1 + } } ) @@ -45,8 +47,10 @@ describe('readResourceInventory', () => { calls.set(path, call) return new Response(null, { status: path === '/ready' && call === 1 ? 503 : 200 }) }, - async (ms) => { - waits.push(ms) + { + wait: async (ms) => { + waits.push(ms) + } } ) @@ -65,17 +69,130 @@ describe('readResourceInventory', () => { calls += 1 return new Response(null, { status: 503 }) }, - async (ms) => { - waits.push(ms) + { + wait: async (ms) => { + waits.push(ms) + } } ) expect(result.health).toBe(false) expect(result.ready).toBe(false) expect(calls).toBe(4) + // A refusing endpoint is a reading, so only the independent retry runs. expect(waits).toEqual([11_000]) }) + it('treats a thrown fetch as no reading and re-asks that path once', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + if (path === '/health' && calls.filter((call) => call === '/health').length === 1) { + throw new TypeError('fetch failed') + } + return new Response(null, { status: 200 }) + }, + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(true) + expect(calls.filter((call) => call === '/health')).toEqual(['/health', '/health']) + expect(waits).toEqual([1_000]) + }) + + it('fails closed when both attempts of a path throw', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + if (path === '/health') throw new TypeError('fetch failed') + return new Response(null, { status: 200 }) + }, + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(false) + expect(calls.filter((call) => call === '/health')).toHaveLength(4) + expect(waits).toEqual([1_000, 11_000, 1_000]) + }) + + it('accepts an auth-shaped endpoint that serves no readiness path', async () => { + const calls: string[] = [] + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://login.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + calls.push(path) + return new Response(null, { status: path === '/ready' ? 404 : 200 }) + }, + { + requiresReady: false, + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBeNull() + expect(calls).toEqual(['/health']) + expect(waits).toEqual([]) + }) + + it('still requires readiness for the director and cells', async () => { + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://relay.onorca.dev', + async (input) => new Response(null, { + status: new URL(String(input)).pathname === '/ready' ? 503 : 200 + }), + { + wait: async (ms) => { + waits.push(ms) + } + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(false) + expect(waits).toEqual([11_000]) + }) + + it('measures latency as the answering round trip, not the retry delay', async () => { + let healthCalls = 0 + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + if (new URL(String(input)).pathname !== '/health') return new Response(null, { status: 200 }) + healthCalls += 1 + if (healthCalls === 1) throw new TypeError('fetch failed') + return new Response(null, { status: 200 }) + }, + { wait: async (ms) => await new Promise((resolve) => setTimeout(resolve, Math.min(ms, 60))) } + ) + + expect(result.health).toBe(true) + expect(result.latencyMs).not.toBeNull() + expect(result.latencyMs!).toBeLessThan(60) + }) + it('uses aggregate REST inventory without probing sleeping staging endpoints', async () => { const gcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } let publicProbeCalls = 0 diff --git a/cloud/apps/relay-ops/src/resource-inventory.ts b/cloud/apps/relay-ops/src/resource-inventory.ts index 62ed3fd862b..c1d01191baf 100644 --- a/cloud/apps/relay-ops/src/resource-inventory.ts +++ b/cloud/apps/relay-ops/src/resource-inventory.ts @@ -102,6 +102,7 @@ export type ResourceInventory = { const unavailableEndpoint = (): EndpointHealth => ({ health: null, ready: null, latencyMs: null }) const independentEndpointRetryDelayMs = 11_000 +const transientProbeRetryDelayMs = 1_000 function finalSegment(value: string): string { return value.split('/').at(-1) ?? value @@ -138,41 +139,80 @@ async function googleRequest( return await response.json() } -async function endpointProbe(origin: string, fetchImpl: typeof fetch): Promise { - const startedAt = performance.now() - const check = async (path: '/health' | '/ready'): Promise => { +// A reading the endpoint actually produced: ok is its answer, latencyMs is that answer's round trip. +type PathReading = { ok: boolean; latencyMs: number | null } + +async function probePath( + origin: string, + path: '/health' | '/ready', + fetchImpl: typeof fetch, + wait: (ms: number) => Promise +): Promise { + // null means the request never produced an answer (DNS/TCP/TLS failure or the 8s abort). + const attempt = async (): Promise => { + const startedAt = performance.now() try { const response = await fetchImpl(`${origin}${path}`, { redirect: 'error', signal: AbortSignal.timeout(8_000) }) - return response.ok + return { ok: response.ok, latencyMs: Math.round(performance.now() - startedAt) } } catch { - return false + return null } } - const [health, ready] = await Promise.all([check('/health'), check('/ready')]) - return { health, ready, latencyMs: Math.round(performance.now() - startedAt) } + const first = await attempt() + if (first) return first + // A thrown fetch is the absence of a reading, not an unhealthy answer, so re-ask before concluding. + await wait(transientProbeRetryDelayMs) + return (await attempt()) ?? { ok: false, latencyMs: null } +} + +async function endpointProbe( + origin: string, + fetchImpl: typeof fetch, + requiresReady: boolean, + wait: (ms: number) => Promise +): Promise { + const [health, ready] = await Promise.all([ + probePath(origin, '/health', fetchImpl, wait), + requiresReady ? probePath(origin, '/ready', fetchImpl, wait) : null + ]) + // Latency is the slowest answering round trip in this probe; retry delays are not serving latency. + const latencies = [health.latencyMs, ready?.latencyMs ?? null].filter( + (value): value is number => value !== null + ) + return { + health: health.ok, + ready: ready ? ready.ok : null, + latencyMs: latencies.length > 0 ? Math.max(...latencies) : null + } +} + +export type EndpointProbeOptions = { + // Auth serves no /ready by design, so it is judged on /health and latency alone. + requiresReady?: boolean + wait?: (ms: number) => Promise } export async function probeEndpointHealth( origin: string, fetchImpl: typeof fetch, - wait: (ms: number) => Promise = async (ms) => - await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)) + options: EndpointProbeOptions = {} ): Promise { - const first = await endpointProbe(origin, fetchImpl) - if ( - first.health && - first.ready && - first.latencyMs !== null && - first.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs - ) { - return first - } + const requiresReady = options.requiresReady ?? true + const wait = options.wait ?? + (async (ms: number) => await new Promise((resolvePromise) => setTimeout(resolvePromise, ms))) + const accepted = (probe: EndpointHealth): boolean => + probe.health === true && + (!requiresReady || probe.ready === true) && + probe.latencyMs !== null && + probe.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs + const first = await endpointProbe(origin, fetchImpl, requiresReady, wait) + if (accepted(first)) return first // Outwait Relay's ten-second readiness cache before treating the retry as independent. await wait(independentEndpointRetryDelayMs) - return await endpointProbe(origin, fetchImpl) + return await endpointProbe(origin, fetchImpl, requiresReady, wait) } function imageDigest(template: z.infer): string | null { @@ -338,7 +378,8 @@ export async function readResourceInventory( ? [unavailableEndpoint(), unavailableEndpoint()] : await Promise.all([ probeEndpointHealth(environment.directorOrigin, fetchImpl), - probeEndpointHealth(environment.authOrigin, fetchImpl) + // The auth service exposes no /ready, so requiring it would fail every first probe. + probeEndpointHealth(environment.authOrigin, fetchImpl, { requiresReady: false }) ]) const cells = await Promise.all(environment.cells.map((cell, index) => readCell(environment, cell, migValues[index] ?? null, token, fetchImpl) diff --git a/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts index 6ac9521c3d6..80a74a47eeb 100644 --- a/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts +++ b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts @@ -44,6 +44,12 @@ describePostgres('PostgreSQL assignment connection headroom', () => { `DELETE FROM relay_assignments WHERE user_id LIKE 'connection-headroom-postgres-%'` ) + // A snapshot left by an aborted run rejects the replayed watermark + // with stale_connection_snapshot. + await database.query( + `DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, + [cell.id] + ) await database.query( `DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id] diff --git a/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts index 10193b78cc6..cf8819686b5 100644 --- a/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts +++ b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts @@ -38,6 +38,10 @@ describePostgres('PostgreSQL control supersession', () => { [identity.userId] ) await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [identity.userId]) + // A snapshot left by an aborted run rejects the replayed watermark with stale_connection_snapshot. + await database.query(`DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, [ + cell.id + ]) await database.query(`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id]) await database.query(`DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, [cell.id]) await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts index d0517d46746..226df9b3984 100644 --- a/cloud/apps/relay/src/assignment-store.ts +++ b/cloud/apps/relay/src/assignment-store.ts @@ -645,9 +645,17 @@ export class RelayAssignmentStore { ): Promise { const now = this.now() return await this.database.transaction(async (transaction) => { - const lockedCells = inventoryFirst - ? await this.lockCellInventory(transaction, lockMode) + // Why: the retry exists to take a cell row before the assignment row, the + // order placement uses. It only ever needs the one cell this host is + // pinned to, so read the pin unlocked and lock that row alone; taking all + // 23 queued every sticky refresh in the fleet behind every other one. + const pinnedCellId = inventoryFirst + ? await this.pinnedCellId(transaction, identity) : undefined + const lockedCells = + pinnedCellId === undefined + ? undefined + : await this.lockCellRows(transaction, [pinnedCellId], lockMode) const existing = await this.assignmentRow(transaction, identity, inventoryFirst) if (!existing) return null const activityLeases = await this.lockAssignmentActivities(transaction, identity, true) @@ -661,6 +669,11 @@ export class RelayAssignmentStore { } const currentCellId = text(existing, 'cell_id') + // The pin moved between the unlocked read and the assignment lock, so the + // row held is the wrong one. Same recovery as losing the lock: retry. + if (pinnedCellId !== undefined && pinnedCellId !== currentCellId) { + throw new Error('database_lock_unavailable') + } const hadControl = holdsControlLease( activityLeases, currentCellId, @@ -701,14 +714,9 @@ export class RelayAssignmentStore { if (hadControl) { await this.touchAssignment(transaction, identity, leaseExpiresAt, now) } else { - const nextReservation = integer(currentRow, 'reserved_requests') + 1 - if (nextReservation > integer(currentRow, 'capacity_requests')) { - throw new Error('relay_capacity_exhausted') - } - await transaction.query( - `UPDATE relay_cells SET reserved_requests = ?, updated_at = ? WHERE cell_id = ?`, - [nextReservation, now, currentCellId] - ) + // Delta, not the value read from the snapshot: an absolute write here + // would clobber any concurrent movement of the same counter. + await this.adjustCellReservationAtomically(transaction, currentCellId, 1) await this.adjustActivityCount(transaction, identity, 'control', 1, leaseExpiresAt, now) await this.insertPendingControlLease( transaction, @@ -3202,8 +3210,7 @@ export class RelayAssignmentStore { ) const requestDelta = ACTIVITY_REQUEST_UNITS[kind] * (after - before) if (requestDelta !== 0) { - await this.lockCellInventory(transaction, 'request') - await this.adjustCellReservation(transaction, text(row, 'cell_id'), requestDelta) + await this.adjustCellReservationAtomically(transaction, text(row, 'cell_id'), requestDelta) } }) }) @@ -3263,9 +3270,12 @@ export class RelayAssignmentStore { } const units = ACTIVITY_REQUEST_UNITS[input.kind] if (existing) { - await this.lockCellInventory(transaction, 'request') + // Why: a client-chosen activity id can move between cells, so lock the + // one or two rows this path touches in cell_id order, the same order + // placement takes the inventory in, and no cycle can form. + await this.lockCellRows(transaction, [text(existing, 'cell_id'), input.cellId]) await this.removeActivityLease(transaction, identity, existing, now) - await this.adjustCellReservation(transaction, input.cellId, units) + await this.adjustCellReservationAtomically(transaction, input.cellId, units) } await this.adjustActivityCount(transaction, identity, input.kind, 1, expiresAt, now) await transaction.query( @@ -3580,8 +3590,7 @@ export class RelayAssignmentStore { ) await this.touchAssignment(transaction, identity, expiresAt, now) } else { - await this.lockCellInventory(transaction, 'request') - await this.adjustCellReservation(transaction, input.cellId, 1) + await this.adjustCellReservationAtomically(transaction, input.cellId, 1) await this.adjustActivityCount(transaction, identity, 'control', 1, expiresAt, now) await transaction.query( `INSERT INTO relay_assignment_activity_leases @@ -6863,13 +6872,24 @@ export class RelayAssignmentStore { targetCellId ] ) - const cells = await this.lockCellInventory(transaction, 'pool-default') + // Only the two cells this repairs need holding. The id set below is an + // existence check against a table that only reconcileCells writes, so it + // reads unlocked instead of dragging the other 21 rows into the section. + const cellIds = new Set( + (await transaction.query(`SELECT cell_id FROM relay_cells`)).map((row) => + text(row, 'cell_id') + ) + ) + const cells = await this.lockCellRows( + transaction, + [sourceCellId, targetCellId], + 'pool-default' + ) const assignmentKeys = new Set( assignments.map((row) => assignmentKey(text(row, 'user_id'), text(row, 'relay_host_id')) ) ) - const cellIds = new Set(cells.map((row) => text(row, 'cell_id'))) const assignmentCounts = new Map< string, { counts: Record; leaseExpiresAt: number } @@ -6923,9 +6943,7 @@ export class RelayAssignmentStore { ) } - for (const row of cells.filter((cell) => - [sourceCellId, targetCellId].includes(text(cell, 'cell_id')) - )) { + for (const row of cells) { const cellId = text(row, 'cell_id') const expected = cellUnits.get(cellId) ?? 0 if (expected > integer(row, 'capacity_requests')) { @@ -6954,6 +6972,43 @@ export class RelayAssignmentStore { return rows } + // Per-connection paths touch one or two cells. Locking exactly those rows, + // in the same ascending order the inventory lock uses (ORDER BY fixes the + // row-lock order), keeps them off the fleet-wide lock without a cycle. + // The wait policy follows the caller for the same reason the inventory lock's + // does: a sweep must not fail terminally on ordinary contention. Hold time is + // deliberately not sampled here — the metric tracks the fleet-wide lock these + // rows replace, and mixing in short single-row holds would flatter it. + private async lockCellRows( + database: RelayDatabase, + cellIds: string[], + mode: CellInventoryLockMode = 'request' + ): Promise { + const distinct = [...new Set(cellIds)] + const { measureHoldMs: _sampled, ...wait } = cellInventoryLockOptions(mode) + return await database.queryLocked( + `SELECT * FROM relay_cells WHERE cell_id IN (${distinct.map(() => '?').join(', ')}) + ORDER BY cell_id ASC`, + distinct, + wait + ) + } + + // Unlocked on purpose: this only names the row to lock next, and the caller + // re-checks the pin once the assignment row is held. + private async pinnedCellId( + database: RelayDatabase, + identity: AssignmentIdentity + ): Promise { + const row = ( + await database.query( + `SELECT cell_id FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + )[0] + return row ? text(row, 'cell_id') : undefined + } + private async lockGeneralCellInventory( database: RelayDatabase, mode: CellInventoryLockMode @@ -6972,10 +7027,11 @@ export class RelayAssignmentStore { private async leastLoadedCell( database: RelayDatabase, - lockedCells: SqlRow[] | undefined, + // Required: the one caller has already locked the inventory it selects from, + // and an optional parameter left a second fleet-wide lock reachable here. + rows: SqlRow[], preferredRegion: RelayRegion ): Promise { - const rows = lockedCells ?? (await this.lockCellInventory(database, 'pool-default')) const regions = new Map( (await database.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ text(row, 'cell_id'), @@ -7590,7 +7646,10 @@ export class RelayAssignmentStore { ) { throw new Error('activity_lease_shape_mismatch') } - const cells = await this.lockCellInventory(database, 'request') + // Why: this recomputes one cell's reservation from its leases, so only that + // row needs to be held; the 23-row inventory lock here serialised every + // desktop control rebind in the fleet behind every other one. + const cellRow = (await this.lockCellRows(database, [cellId]))[0] await database.query( `DELETE FROM relay_assignment_activity_leases WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control' @@ -7611,7 +7670,6 @@ export class RelayAssignmentStore { [cellId] ) )[0]! - const cellRow = cells.find((cell) => text(cell, 'cell_id') === cellId) const cellUnits = integer(cellUnitsRow, 'request_units') if (!cellRow) throw new Error('assigned_cell_missing') if (cellUnits > integer(cellRow, 'capacity_requests')) { diff --git a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts index 8ca7cee55f5..0ac4c8225e3 100644 --- a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts +++ b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts @@ -18,17 +18,21 @@ type CensusEntry = { method: string; mode: CensusMode; reach: Reachability } // assignment-store.ts, in source order. A new site fails this test until it is // classified here, which is the point. const CENSUS: CensusEntry[] = [ - { method: 'assignStickyOnce', mode: 'caller', reach: 'both' }, + // assignStickyOnce is gone from this list: its retry now locks only the row + // the host is pinned to (lockCellRows), which is what a sticky refresh + // touches. Placement below is the one genuinely fleet-wide decision left. { method: 'assignOnce', mode: 'caller', reach: 'both' }, { method: 'assignOnce', mode: 'caller', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'refreshDrainMigrationLeasesOnce', mode: 'request', reach: 'request' }, - // Reachable from neither: changeActivity has no production callers, only tests. - { method: 'changeActivity', mode: 'request', reach: 'orphan' }, - { method: 'acquireActivity', mode: 'request', reach: 'request' }, - { method: 'activateControl', mode: 'request', reach: 'request' }, + // changeActivity, acquireActivity, activateControl and + // removeSupersededSameCellControls no longer take the inventory: they lock + // only the one or two cell rows they touch, in cell_id order (lockCellRows), + // so they cannot cycle with placement's ordered inventory lock, and the + // 23-row lock there had serialised every reconnect in the fleet behind every + // other one. { method: 'startEvacuation', mode: 'request', reach: 'request' }, { method: 'completeEvacuationFromDeadSourceOnce', mode: 'request', reach: 'request' }, { method: 'completeEvacuationFromDeadSourceOnce', mode: 'nowait', reach: 'request' }, @@ -47,9 +51,33 @@ const CENSUS: CensusEntry[] = [ { method: 'abortExpiredEvacuations', mode: 'nowait', reach: 'sweep' }, { method: 'releaseExpiredActivityLeases', mode: 'nowait', reach: 'sweep' }, { method: 'releaseExpiredActivity', mode: 'nowait', reach: 'sweep' }, - { method: 'reconcileReservationAccounting', mode: 'pool-default', reach: 'both' }, - { method: 'leastLoadedCell', mode: 'pool-default', reach: 'both' }, - { method: 'removeSupersededSameCellControls', mode: 'request', reach: 'request' } + // reconcileReservationAccounting and leastLoadedCell are gone too: the first + // repairs exactly two cells' counters and now holds only those rows, and the + // second selects from the inventory its single caller has already locked. +] + +// Every inline `FROM relay_cells ... FOR UPDATE` outside the named lock helpers, +// in source order: whole-table locks in reconciliation and sticky placement, +// and single-row locks for a cell the method is already scoped to (heartbeat, +// fence, drain generation, configuration, or a reservation adjust that runs +// under a lock its caller already holds). A new inline lock fails the census +// below until it is listed here; per-connection paths that touch more than one +// cell go through lockCellRows so the order is fixed. +const NAMED_LOCK_HELPERS = ['lockCellInventory', 'lockGeneralCellInventory', 'lockCellRows'] + +const INLINE_CELL_LOCK_SITES = [ + 'reconcileCellsWithOptions', + 'assignStickyOnce', + 'recordCellHeartbeat', + 'attestCellFence', + 'adoptLegacyCellFence', + 'commitLegacyCellFenceAdoption', + 'prepareCellFenceAttempt', + 'attestCellFenceAttempt', + 'attestCellFenceAttempt', + 'configureCell', + 'assertDrainCellGeneration', + 'adjustCellReservation' ] // The background sweeps, and nothing else. A method reachable from one of these @@ -151,6 +179,42 @@ describe('cell inventory lock call-site census', () => { ) }) + // Why: the census only sees lockCellInventory calls, so a hand-written + // `relay_cells ... FOR UPDATE` would escape classification entirely. + it('routes every relay_cells row lock through a named lock helper', () => { + const lines = storeSource() + const rawSites: string[] = [] + // Whole statements, not a fixed window: a wide column list or a raw + // FOR UPDATE inside query() must not slip past. + const source = lines.join('\n') + const bounds: { name: string; start: number }[] = [] + lines.forEach((line, index) => { + const declaration = DECLARATION.exec(line) + if (declaration) bounds.push({ name: declaration[1]!, start: index }) + }) + const methodAt = (offset: number): string => { + const lineIndex = source.slice(0, offset).split('\n').length - 1 + let name = '' + for (const bound of bounds) if (bound.start <= lineIndex) name = bound.name + return name + } + const tick = String.fromCharCode(96) + const statementCall = new RegExp( + '\\.(queryLocked|query)\\(\\s*' + tick + '([^' + tick + ']*)' + tick, + 'g' + ) + for (const call of source.matchAll(statementCall)) { + const statement = call[2]! + if (!/\bFROM\s+relay_cells\b/.test(statement)) continue + const locks = call[1] === 'queryLocked' || /\bFOR\s+UPDATE\b/.test(statement) + if (!locks) continue + const method = methodAt(call.index) + if (NAMED_LOCK_HELPERS.includes(method)) continue + rawSites.push(method) + } + expect(rawSites).toEqual(INLINE_CELL_LOCK_SITES) + }) + it('leaves no call site taking the inventory without naming a mode', () => { const source = readFileSync(new URL('./assignment-store.ts', import.meta.url), 'utf8') const unclassified = source diff --git a/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts b/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts new file mode 100644 index 00000000000..8c0ebd73f32 --- /dev/null +++ b/cloud/apps/relay/src/cell-inventory-per-cell-locking-postgres.test.ts @@ -0,0 +1,206 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +// Sorted ascending, and the host is pinned to the LAST id on purpose: the +// fleet-wide lock is one ordered scan, so it holds every earlier row while it +// waits on the pinned one. Pinning to the first id would make the two locking +// models indistinguishable. +const cells = ['a', 'b', 'c'].map((suffix) => ({ + id: `percell-postgres-${suffix}`, + url: `https://percell-postgres-${suffix}.example.com`, + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 +})) +const [cellA, cellB, cellC] = cells as [(typeof cells)[0], (typeof cells)[0], (typeof cells)[0]] +const identity = { userId: 'percell-postgres-user', relayHostId: 'percellhost00001' } + +function heartbeat(cell: (typeof cells)[number]) { + return { + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 1, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +} + +describePostgres('PostgreSQL per-cell inventory locking', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + for (let index = 0; index < 3; index++) { + databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' })) + } + }) + + async function removeTestRows(database: RelayDatabase): Promise { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id LIKE 'percell-postgres-%'` + ) + for (const table of [ + 'relay_assignment_activity_leases', + 'relay_post_drain_migration_pins', + 'relay_assignment_migration_incarnations', + 'relay_assignment_migrations', + 'relay_assignment_region_preferences', + 'relay_assignments' + ]) { + await database.query(`DELETE FROM ${table} WHERE user_id LIKE 'percell-postgres-%'`) + } + for (const cell of cells) { + for (const table of [ + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_connection_limits', + 'relay_cell_runtime', + 'relay_cells' + ]) { + await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id]) + } + } + } + + afterAll(async () => { + if (databases[0]) await removeTestRows(databases[0]) + for (const connection of databases) await connection.close() + }) + + async function pinHostToLastCell(store: RelayAssignmentStore): Promise { + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cellC.id) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + } + + async function lockWaiterAppeared(database: RelayDatabase): Promise { + const deadline = Date.now() + 4_000 + while (Date.now() < deadline) { + const rows = await database.query( + `SELECT count(*) AS waiting FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock'` + ) + if (Number(rows[0]!.waiting) > 0) return true + await new Promise((resolve) => setTimeout(resolve, 10)) + } + return false + } + + // Why: a sticky refresh whose first NOWAIT probe loses retries by taking a + // cell row before the assignment row. That retry used to take the whole + // inventory, so one busy cell stalled every other cell's reconnects. + it('waits only on the pinned cell row while refreshing a sticky assignment', async () => { + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await pinHostToLastCell(store) + // A host whose control lease was already reaped still holds its pin; that + // is the shape that reaches the cell-row probe instead of touchAssignment. + await databases[0]!.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + [identity.userId] + ) + + let releaseRow!: () => void + const rowReleased = new Promise((resolve) => { + releaseRow = resolve + }) + let rowHeld!: () => void + const rowHeldPromise = new Promise((resolve) => { + rowHeld = resolve + }) + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellC.id]) + rowHeld() + await rowReleased + }) + await rowHeldPromise + + const refresh = store.assign(identity) + expect(await lockWaiterAppeared(databases[2]!)).toBe(true) + // The refresh is blocked on cell C. Every earlier row must still be free: + // the ordered fleet-wide scan would be holding both of them by now. + const heldWhileRefreshWaits: string[] = [] + await databases[2]!.transaction(async (transaction) => { + for (const cell of [cellA, cellB]) { + try { + await transaction.queryLocked( + `SELECT * FROM relay_cells WHERE cell_id = ?`, + [cell.id], + { failIfUnavailable: true } + ) + } catch { + heldWhileRefreshWaits.push(cell.id) + } + } + }) + releaseRow() + await holder + + expect(heldWhileRefreshWaits).toEqual([]) + expect((await refresh).cellId).toBe(cellC.id) + }, 15_000) + + // Why: the counter moves by a delta now instead of an absolute value read + // from a snapshot, so concurrent movement on the same cell must still sum. + it('keeps a cell reservation exact under concurrent same-cell activity', async () => { + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + + const hosts = Array.from({ length: 6 }, (_, index) => ({ + userId: `percell-postgres-user-${index}`, + relayHostId: `percellhost0000${index}` + })) + const stores = databases.map((database) => new RelayAssignmentStore(database, () => 100)) + await Promise.all(hosts.map((host, index) => stores[index % stores.length]!.assign(host))) + + // One splice each (2 units) on the same cell, from three connections at once. + await Promise.all( + hosts.map((host, index) => + stores[index % stores.length]!.acquireActivity(host, { + activityId: `splice:percell-${index}`, + kind: 'splice', + cellId: cellC.id + }) + ) + ) + const afterAcquire = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cellC.id] + ) + // 6 pending control grants + 6 splices at 2 units each. + expect(Number(afterAcquire[0]!.reserved_requests)).toBe(6 + 12) + + await Promise.all( + hosts.map((host, index) => + stores[index % stores.length]!.releaseActivity(host, `splice:percell-${index}`) + ) + ) + const afterRelease = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cellC.id] + ) + expect(Number(afterRelease[0]!.reserved_requests)).toBe(6) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts b/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts new file mode 100644 index 00000000000..e990ac1ed1a --- /dev/null +++ b/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts @@ -0,0 +1,260 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +// Three cells: the inventory lock covers more than the rows a move touches, and +// a high-to-low move exposes any lock taken out of cell_id order. +const cells = [ + { + id: 'rebind-inventory-postgres-a', + url: 'https://rebind-inventory-postgres-a.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'rebind-inventory-postgres-b', + url: 'https://rebind-inventory-postgres-b.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'rebind-inventory-postgres-c', + url: 'https://rebind-inventory-postgres-c.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +] +const identity = { userId: 'rebind-inventory-postgres-user', relayHostId: 'rebindinvhost001' } + +function heartbeat(cell: (typeof cells)[number]) { + return { + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 1, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +} + +// Why: every desktop control rebind used to take the fleet-wide relay_cells +// FOR UPDATE lock, so a rebind on one cell queued behind whatever held any +// other cell's row, until COMMIT (55P03 at the request bound). A rebind only +// touches its own cell row, so it must proceed while another cell's row is +// held elsewhere. +describePostgres('PostgreSQL control rebind under a held cell row', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push( + await openRelayDatabase({ databaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl, dataDir: '' }) + ) + }) + + async function removeTestRows(database: RelayDatabase): Promise { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, + [identity.userId] + ) + for (const table of [ + 'relay_assignment_activity_leases', + 'relay_post_drain_migration_pins', + 'relay_assignment_migration_incarnations', + 'relay_assignment_migrations', + 'relay_assignments' + ]) { + await database.query(`DELETE FROM ${table} WHERE user_id = ?`, [identity.userId]) + } + for (const cell of cells) { + for (const table of [ + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_connection_limits', + 'relay_cell_runtime', + 'relay_cells' + ]) { + await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id]) + } + } + } + + afterAll(async () => { + if (databases[0]) await removeTestRows(databases[0]) + for (const connection of databases) await connection.close() + }) + + it("rebinds and supersedes a control while another cell's row is held", async () => { + // A prior aborted run leaves connection snapshots that reject a replayed watermark. + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + // Pin the host to cell A so placement is deterministic. + await store.setCellEnabled(cells[1]!.id, false) + await store.setCellEnabled(cells[2]!.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cells[0]!.id) + await store.setCellEnabled(cells[1]!.id, true) + await store.setCellEnabled(cells[2]!.id, true) + await store.activateControl(identity, { + cellId: cells[0]!.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 10 + }) + + // Hold only cell B's row on a second connection, the way a rebind on B + // does, for longer than the request-path lock bound. + let releaseInventory!: () => void + const inventoryReleased = new Promise((resolve) => { + releaseInventory = resolve + }) + let inventoryHeld!: () => void + const inventoryHeldPromise = new Promise((resolve) => { + inventoryHeld = resolve + }) + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cells[1]!.id]) + inventoryHeld() + await inventoryReleased + }) + await inventoryHeldPromise + + // A generation-2 rebind on cell A supersedes generation 1. It must not + // wait on cell B's row. + const startedAt = Date.now() + const blockedStatement = async (): Promise => { + const rows = await databases[1]!.query( + `SELECT left(query, 160) AS q FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock'` + ) + return rows.map((row) => String(row.q)).join(' | ') + } + const timeout = new Promise((_, reject) => + setTimeout( + () => + void blockedStatement().then((statement) => + reject(new Error(`rebind on cell A blocked behind cell B's row: ${statement}`)) + ), + 2_000 + ) + ) + const rebound = await Promise.race([ + store.activateControl(identity, { + cellId: cells[0]!.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 2, + connectionInclusionWatermark: 11 + }), + timeout + ]) + const elapsedMs = Date.now() - startedAt + releaseInventory() + await holder + + expect(rebound).toBe(`control:${cells[0]!.id}:2`) + expect(elapsedMs).toBeLessThan(2_000) + const controls = await databases[0]!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND activity_kind = 'control' ORDER BY activity_id`, + [identity.userId] + ) + expect(controls).toEqual([{ activity_id: `control:${cells[0]!.id}:2` }]) + const reserved = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cells[0]!.id] + ) + expect(Number(reserved[0]!.reserved_requests)).toBe(1) + }, 15_000) + + // Why: a phone's activity id is client-chosen and can follow the host across + // a migration, so acquireActivity may touch two cell rows. Moving from the + // higher cell to the lower one is where an unordered lock cycles with + // placement's ascending inventory lock (reproduced live before this fix). + it('moves an activity from a higher cell to a lower one in cell_id order', async () => { + await removeTestRows(databases[0]!) + const [cellA, cellB, cellC] = cells as [typeof cells[0], typeof cells[0], typeof cells[0]] + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cellC.id) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + const activityId = 'splice:rebind-inventory-postgres' + await store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellC.id }) + // The migration makes B authoritative; the lease still sits on C. + const migration = await store.startEvacuation(identity, cellB.id) + expect(migration.targetCellId).toBe(cellB.id) + + // Hold B elsewhere. An ordered move locks B first and queues here holding + // nothing else. Locking C first (the old lease's row, as an unordered move + // does) or the whole inventory (which takes A) shows up as a held row. + let releaseRow!: () => void + const rowReleased = new Promise((resolve) => { + releaseRow = resolve + }) + let rowHeld!: () => void + const rowHeldPromise = new Promise((resolve) => { + rowHeld = resolve + }) + const heldWhileMoverWaits: string[] = [] + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellB.id]) + rowHeld() + await rowReleased + for (const cell of [cellA, cellC]) { + try { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id], { + failIfUnavailable: true + }) + } catch { + heldWhileMoverWaits.push(cell.id) + } + } + }) + await rowHeldPromise + const move = store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellB.id }) + let moved = false + void move.then(() => { + moved = true + }) + await new Promise((resolve) => setTimeout(resolve, 250)) + expect(moved).toBe(false) + releaseRow() + await holder + await move + expect(heldWhileMoverWaits).toEqual([]) + + const reservations = await databases[0]!.query( + `SELECT cell_id, reserved_requests FROM relay_cells + WHERE cell_id IN (?, ?, ?) ORDER BY cell_id ASC`, + [cellA.id, cellB.id, cellC.id] + ) + const reserved = reservations.map((row) => [String(row.cell_id), Number(row.reserved_requests)]) + expect(reserved).toEqual([ + [cellA.id, 0], + // Migration grant plus the moved splice, as in the SQLite origin-scoped + // reservation case: the lock change did not alter accounting. + [cellB.id, 6], + // The sticky grant stays on the source until the migration completes. + [cellC.id, 1] + ]) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/database-postgres-timeout.test.ts b/cloud/apps/relay/src/database-postgres-timeout.test.ts index 7aba1234f7f..c9021a9ef18 100644 --- a/cloud/apps/relay/src/database-postgres-timeout.test.ts +++ b/cloud/apps/relay/src/database-postgres-timeout.test.ts @@ -2,7 +2,10 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const fakes = vi.hoisted(() => ({ configs: [] as Array>, - query: vi.fn(async () => ({ rows: [], rowCount: 0 })), + // Pool construction and pool shutdown interleaved, so "the schema pool is + // gone before the serving pool opens" is checkable rather than assumed. + lifecycle: [] as string[], + query: vi.fn(async (_sql: string) => ({ rows: [], rowCount: 0 })), release: vi.fn(), end: vi.fn(async () => undefined) })) @@ -13,20 +16,37 @@ vi.mock('pg', () => ({ totalCount = 1 idleCount = 1 waitingCount = 0 - end = fakes.end on = vi.fn() connect = vi.fn(async () => ({ query: fakes.query, release: fakes.release })) + private readonly label: string constructor(config: Record) { fakes.configs.push(config) + this.label = `max=${String(config.max)} statement_timeout=${String(config.statement_timeout)}` + fakes.lifecycle.push(`open ${this.label}`) + } + + async end(): Promise { + fakes.lifecycle.push(`end ${this.label}`) + await fakes.end() } } } })) -import { openRelayDatabase } from './database.js' +import { openRelayDatabase, relayPostgresStatementTimeoutMs } from './database.js' import { applyPostgresSchema } from './postgres-schema-startup.js' +const SCHEMA_POOL = { + max: 1, + application_name: 'orca-relay/director/director/schema', + connectionTimeoutMillis: 2_000, + // Why: DDL must not inherit the request deadline. + statement_timeout: 0, + lock_timeout: 1_000, + idle_in_transaction_session_timeout: 5_000 +} + afterEach(() => { vi.restoreAllMocks() }) @@ -34,9 +54,11 @@ afterEach(() => { describe('PostgreSQL relay deadlines', () => { beforeEach(() => { fakes.configs.length = 0 + fakes.lifecycle.length = 0 fakes.query.mockClear() fakes.release.mockClear() fakes.end.mockClear() + delete process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS }) it('bounds pool acquisition, statements, locks, and abandoned transactions', async () => { @@ -48,6 +70,7 @@ describe('PostgreSQL relay deadlines', () => { }) expect(fakes.configs).toEqual([ + expect.objectContaining(SCHEMA_POOL), expect.objectContaining({ max: 3, application_name: 'orca-relay/director/director', @@ -59,6 +82,110 @@ describe('PostgreSQL relay deadlines', () => { ]) await database.close() }) + + // Why: an untimed session left open would be a standing way for request work + // to escape the deadline this whole pool config exists to enforce. + it('closes the untimed schema pool before the serving pool opens', async () => { + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused', + poolMax: 3, + applicationName: 'orca-relay/director/director' + }) + + expect(fakes.lifecycle).toEqual([ + 'open max=1 statement_timeout=0', + 'end max=1 statement_timeout=0', + 'open max=3 statement_timeout=5000' + ]) + await database.close() + }) + + it('applies the schema on the untimed pool, never on the serving one', async () => { + fakes.query.mockClear() + const ddl: string[] = [] + fakes.query.mockImplementation(async (sql: string) => { + // Every statement issued before the serving pool exists is schema work. + if (fakes.lifecycle.length === 1) ddl.push(sql) + return { rows: [], rowCount: 0 } + }) + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + + expect(ddl.length).toBeGreaterThan(0) + // Statements can open with a leading `--` rationale comment. + const body = (statement: string): string => + statement.replace(/^(?:\s*--[^\n]*\n)*\s*/, '') + expect(ddl.every((statement) => /^CREATE\b/i.test(body(statement)))).toBe(true) + // The backfill is DML, so it stays on the deadline-bearing serving pool. + expect(ddl.some((statement) => statement.includes('INSERT INTO'))).toBe(false) + await database.close() + }) + + it('takes the serving statement deadline from the environment', async () => { + process.env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS = '2500' + + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + + expect(fakes.configs).toEqual([ + expect.objectContaining({ statement_timeout: 0 }), + expect.objectContaining({ statement_timeout: 2_500 }) + ]) + await database.close() + }) + + it.each(['0', '-1', '2.5', 'soon', ' '])( + 'refuses %s as a statement deadline instead of running unbounded', + (value) => { + expect(() => + relayPostgresStatementTimeoutMs({ ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value }) + ).toThrow('invalid_statement_timeout') + } + ) + + it.each([undefined, ''])('defaults to 5s when the environment says %s', (value) => { + expect( + relayPostgresStatementTimeoutMs( + value === undefined ? {} : { ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS: value } + ) + ).toBe(5_000) + }) + + // Why: a statement deadline that reaches the caller as a crash converts a + // transient stall into a failed assignment. It aborts the transaction exactly + // as a lock timeout does, so it belongs on the same bounded retry. + it('retries a statement timeout on a fresh client', async () => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + let attempts = 0 + + const result = await database.transaction(async (transaction) => { + attempts += 1 + if (attempts === 1) { + await transaction.query('SELECT 1') + throw Object.assign(new Error('canceling statement due to statement timeout'), { + code: '57014' + }) + } + return 'committed' + }) + + expect(result).toBe('committed') + expect(attempts).toBe(2) + expect(console.warn).toHaveBeenCalledWith( + expect.stringContaining('"event":"orca_relay_postgres_transaction_retry"') + ) + expect(console.warn).toHaveBeenCalledWith(expect.stringContaining('"code":"57014"')) + await database.close() + }) }) describe('PostgreSQL schema startup', () => { diff --git a/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts b/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts new file mode 100644 index 00000000000..6b7ebb0334e --- /dev/null +++ b/cloud/apps/relay/src/database-statement-timeout-postgres.test.ts @@ -0,0 +1,98 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const applicationName = 'orca-relay/statement-timeout-postgres' + +describePostgres('PostgreSQL statement deadline', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push(await openRelayDatabase({ databaseUrl, dataDir: '' })) + }) + + afterAll(async () => { + for (const database of databases) await database.close() + }) + + it('serves requests under the configured deadline', async () => { + const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 }) + databases.push(database) + + expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([ + { statement_timeout: '300ms' } + ]) + }) + + // Why: a real 57014 aborts the transaction exactly as a lock timeout does. If + // it escapes the bounded retry it becomes a failed assignment instead of a + // slow one. + it('retries a real statement timeout on a fresh client', async () => { + const database = await openRelayDatabase({ databaseUrl, dataDir: '', statementTimeoutMs: 300 }) + databases.push(database) + let attempts = 0 + + const result = await database.transaction(async (transaction) => { + attempts += 1 + if (attempts === 1) await transaction.query(`SELECT pg_sleep(2)`) + return attempts + }) + + expect(result).toBe(2) + }, 15_000) + + // Why: DDL runs on its own untimed connection. relay_invites carries a + // CREATE INDEX IF NOT EXISTS, which (unlike CREATE TABLE IF NOT EXISTS) + // really does queue behind an ACCESS EXCLUSIVE lock on the table. + it('applies the schema behind a held ACCESS EXCLUSIVE lock', async () => { + let releaseTable!: () => void + const tableReleased = new Promise((resolve) => { + releaseTable = resolve + }) + let tableHeld!: () => void + const tableHeldPromise = new Promise((resolve) => { + tableHeld = resolve + }) + const holder = databases[0]!.transaction(async (transaction) => { + await transaction.query(`LOCK TABLE relay_invites IN ACCESS EXCLUSIVE MODE`) + tableHeld() + await tableReleased + }) + await tableHeldPromise + + const opening = openRelayDatabase({ + databaseUrl, + dataDir: '', + applicationName, + // Far too short for a blocked DDL; the serving pool wears it, the schema + // connection must not. + statementTimeoutMs: 200 + }) + const blockedOnSchemaConnection = async (): Promise => { + const deadline = Date.now() + 4_000 + while (Date.now() < deadline) { + const rows = await databases[0]!.query( + `SELECT count(*) AS waiting FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock' + AND application_name = ?`, + [`${applicationName}/schema`] + ) + if (Number(rows[0]!.waiting) > 0) return true + await new Promise((resolve) => setTimeout(resolve, 10)) + } + return false + } + const blocked = await blockedOnSchemaConnection() + releaseTable() + await holder + + const database = await opening + databases.push(database) + expect(blocked).toBe(true) + // The serving pool still carries the short deadline it was opened with. + expect(await database.query(`SELECT current_setting('statement_timeout') AS statement_timeout`)).toEqual([ + { statement_timeout: '200ms' } + ]) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/database.ts b/cloud/apps/relay/src/database.ts index 326ab010ccb..2558831ca64 100644 --- a/cloud/apps/relay/src/database.ts +++ b/cloud/apps/relay/src/database.ts @@ -796,12 +796,34 @@ class PostgresTransaction implements RelayDatabase { const POSTGRES_TRANSACTION_ATTEMPTS = 3 const POSTGRES_RETRY_MAX_DELAY_MS = 25 const POSTGRES_CONNECTION_TIMEOUT_MS = 2_000 -const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 +// Derivation: a control renewal must land inside its own 30s tick +// (RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2), and a transaction gets +// POSTGRES_TRANSACTION_ATTEMPTS tries, so the worst case a renewal can spend in +// Postgres is attempts * timeout. 5s keeps that at 15s, half the tick, and still +// leaves room for the connect timeout above. +export const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 const POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 5_000 +export function relayPostgresStatementTimeoutMs( + env: NodeJS.ProcessEnv = process.env +): number { + const configured = env.ORCA_RELAY_POSTGRES_STATEMENT_TIMEOUT_MS + if (configured === undefined || configured === '') return POSTGRES_STATEMENT_TIMEOUT_MS + const milliseconds = Number(configured) + // 0 is PostgreSQL's "no timeout"; refusing it keeps the deadline this exists + // to enforce from being disabled by a typo in an environment variable. + if (!Number.isInteger(milliseconds) || milliseconds < 1) { + throw new Error('invalid_statement_timeout') + } + return milliseconds +} + function retryablePostgresTransactionError(error: unknown): boolean { const code = String((error as { code?: unknown }).code) - return code === '40P01' || code === '40001' || code === '55P03' + // 57014 is the pool statement_timeout firing. It aborts the transaction the + // same way a lock timeout does, so it belongs on the bounded retry path + // rather than surfacing as a terminal failure to the caller. + return code === '40P01' || code === '40001' || code === '55P03' || code === '57014' } export function isRelayDatabaseTransientError(error: unknown): boolean { @@ -963,11 +985,36 @@ async function applySchema(database: RelayDatabase): Promise { } } -async function applySchemaWithPostgresRetries(database: RelayDatabase): Promise { - await applyPostgresSchema( - SCHEMA.split(';').filter((statement) => statement.trim()), - async (statement) => await database.query(statement) - ) +// Why: DDL is not a request. A CREATE INDEX on a grown table legitimately runs +// longer than the request statement_timeout, and inheriting that timeout would +// make every startup fail at the same statement instead of finishing once. One +// short-lived connection of its own, ended before the serving pool opens, keeps +// the untimed session off the request path entirely. +async function applySchemaOnUntimedPool( + databaseUrl: string, + applicationName: string | undefined +): Promise { + const pool = new pg.Pool({ + connectionString: databaseUrl, + max: 1, + application_name: applicationName ? `${applicationName}/schema` : undefined, + connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, + statement_timeout: 0, + // Kept: a DDL blocked behind another director's ACCESS EXCLUSIVE lock must + // yield to the bounded schema retry instead of holding the connection. + lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, + idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS + }) + absorbPostgresIdleClientErrors(pool) + const database = new PostgresDatabase(pool) + try { + await applyPostgresSchema( + SCHEMA.split(';').filter((statement) => statement.trim()), + async (statement) => await database.query(statement) + ) + } finally { + await database.close().catch(() => undefined) + } } async function backfillRelayCellRegions(database: RelayDatabase): Promise { @@ -983,15 +1030,17 @@ export async function openRelayDatabase(input: { dataDir: string poolMax?: number applicationName?: string + statementTimeoutMs?: number }): Promise { let database: RelayDatabase if (input.databaseUrl) { + await applySchemaOnUntimedPool(input.databaseUrl, input.applicationName) const pool = new pg.Pool({ connectionString: input.databaseUrl, max: input.poolMax ?? 10, application_name: input.applicationName, connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, - statement_timeout: POSTGRES_STATEMENT_TIMEOUT_MS, + statement_timeout: input.statementTimeoutMs ?? relayPostgresStatementTimeoutMs(), lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS }) @@ -1004,8 +1053,7 @@ export async function openRelayDatabase(input: { database = new SqliteDatabase(sqlite) } try { - if (input.databaseUrl) await applySchemaWithPostgresRetries(database) - else await applySchema(database) + if (!input.databaseUrl) await applySchema(database) await backfillRelayCellRegions(database) return database } catch (error) { diff --git a/cloud/apps/relay/src/host-close-reason-memory.test.ts b/cloud/apps/relay/src/host-close-reason-memory.test.ts new file mode 100644 index 00000000000..2985e6f1a1d --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.test.ts @@ -0,0 +1,82 @@ +import { ASSIGNMENT_LIMITS, RELAY_HOST_CLOSE_REASON } from '@orca-cloud/relay-contract' +import { describe, expect, it } from 'vitest' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' + +function memoryAt(clock: { now: number }): HostCloseReasonMemory { + return new HostCloseReasonMemory(() => clock.now) +} + +describe('HostCloseReasonMemory', () => { + it('remembers only reasons it knows', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', 'quitting') + memory.record('c', Buffer.alloc(0)) + memory.record('d', undefined) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + expect(memory.read('b')).toBeNull() + expect(memory.read('c')).toBeNull() + expect(memory.read('d')).toBeNull() + }) + + it('accepts the reason as the Buffer a ws close delivers', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', Buffer.from(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('expires an entry once its host may have been rebalanced away', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += ASSIGNMENT_LIMITS.dormantTtlMs - 1 + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += 1 + expect(memory.read('a')).toBeNull() + expect(memory.size()).toBe(0) + }) + + it('forgets on demand', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + memory.forget('a') + + expect(memory.read('a')).toBeNull() + }) + + it('drops the oldest survivors rather than growing without bound', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + for (let index = 0; index < 50_050; index++) { + memory.record(`host-${index}`, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + } + + expect(memory.size()).toBe(50_000) + expect(memory.read('host-0')).toBeNull() + expect(memory.read('host-50049')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('re-recording refreshes recency so a live host is not evicted first', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + expect([...['a', 'b'].map((key) => memory.read(key))]).toEqual([ + RELAY_HOST_CLOSE_REASON.SIGNED_OUT, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ]) + expect(memory.size()).toBe(2) + }) +}) diff --git a/cloud/apps/relay/src/host-close-reason-memory.ts b/cloud/apps/relay/src/host-close-reason-memory.ts new file mode 100644 index 00000000000..ed01aacd666 --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.ts @@ -0,0 +1,72 @@ +import { + ASSIGNMENT_LIMITS, + relayHostCloseReasonFrom, + type RelayHostCloseReason +} from '@orca-cloud/relay-contract' + +// Retention matches the dormant assignment TTL: past it the host may have been +// rebalanced onto another cell, so this cell is no longer the one a phone asks. +const RETENTION_MS = ASSIGNMENT_LIMITS.dormantTtlMs +// A fleet-wide auth outage signs out every host at once; the cap bounds that +// burst well above any single cell's host count without becoming a leak. +const MAX_ENTRIES = 50_000 + +// Why in-memory and not Postgres: a phone reaches the cell its host's assignment +// row already names, which is the same cell that watched the control socket +// close. Losing this on a cell restart degrades to the pre-existing generic +// verdict, so the failure mode is the old behaviour rather than a wrong one. +export class HostCloseReasonMemory { + private readonly entries = new Map() + + constructor(private readonly now: () => number = Date.now) {} + + // Silently ignores anything that is not a known reason, which is every close + // from a host that predates the field and every abrupt 1006. + record(key: string, reason: unknown): void { + const parsed = relayHostCloseReasonFrom(reason) + if (!parsed) { + return + } + this.entries.delete(key) + this.entries.set(key, { reason: parsed, expiresAt: this.now() + RETENTION_MS }) + this.evict() + } + + forget(key: string): void { + this.entries.delete(key) + } + + read(key: string): RelayHostCloseReason | null { + const entry = this.entries.get(key) + if (!entry) { + return null + } + if (entry.expiresAt <= this.now()) { + this.entries.delete(key) + return null + } + return entry.reason + } + + size(): number { + return this.entries.size + } + + private evict(): void { + const now = this.now() + for (const [key, entry] of this.entries) { + if (entry.expiresAt > now) { + break + } + this.entries.delete(key) + } + // Insertion order is recency order (record deletes before setting), so the + // head is always the oldest survivor. + for (const key of this.entries.keys()) { + if (this.entries.size <= MAX_ENTRIES) { + break + } + this.entries.delete(key) + } + } +} diff --git a/cloud/apps/relay/src/host-session-registry.ts b/cloud/apps/relay/src/host-session-registry.ts index 11b7d1de030..5c53041e7ff 100644 --- a/cloud/apps/relay/src/host-session-registry.ts +++ b/cloud/apps/relay/src/host-session-registry.ts @@ -14,7 +14,8 @@ import { HostHelloSchema, InviteCreateSchema, RELAY_PROTOCOL_LIMITS, - RELAY_CLOSE_CODE + RELAY_CLOSE_CODE, + type RelayHostCloseReason } from '@orca-cloud/relay-contract' import nacl from 'tweetnacl' import type WebSocket from 'ws' @@ -25,6 +26,7 @@ import { RelayCredentialStore, type CredentialReservation } from './credential-store.js' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' import { relayHostLogDigest } from './relay-host-log-digest.js' import type { RelayTokenClaims } from './relay-token-verifier.js' import type { RelayRuntimeObserver } from './relay-observability.js' @@ -130,6 +132,10 @@ const ACTIVATION_QUEUE_WAIT_MS = 30_000 export class HostSessionRegistry { private readonly sessions = new Map() private readonly activationQueues = new Map>() + // Why it outlives `sessions`: the orphan grace deletes the session within 30s, + // but a signed-out desktop never comes back, so the phone that asks minutes + // later would otherwise find nothing to explain its rejection with. + private readonly hostCloseReasons = new HostCloseReasonMemory(() => this.now()) private draining = false constructor( @@ -175,7 +181,8 @@ export class HostSessionRegistry { return } this.observer.recordAuth(true) - const session = this.sessions.get(this.key(reservation.userId, hostId)) + const sessionKey = this.key(reservation.userId, hostId) + const session = this.sessions.get(sessionKey) if ( !session || session.state !== 'active' || @@ -184,7 +191,13 @@ export class HostSessionRegistry { ) { capacityReservation?.release() await this.store.failReservation(reservation) - this.rejectClient(socket, RELAY_CLOSE_CODE.HOST_OFFLINE) + // The only rejection that can name a cause: the host is genuinely absent. + // The attach-deadline 4404 below fires while control is still connected. + this.rejectClient( + socket, + RELAY_CLOSE_CODE.HOST_OFFLINE, + this.hostCloseReasons.read(sessionKey) + ) return } if (session.activeConnIds.size + session.pendingConns.size >= 8) { @@ -793,7 +806,10 @@ export class HostSessionRegistry { regionalDrainTimer: null, regionalDrainExpiresAt: null } - this.sessions.set(this.key(identity.sub, identity.relayHostId), session) + const sessionKey = this.key(identity.sub, identity.relayHostId) + // A host that proved itself again is not signed out, whatever it said last. + this.hostCloseReasons.forget(sessionKey) + this.sessions.set(sessionKey, session) this.wireActiveControl(session) this.sendHelloAck(session) } @@ -813,6 +829,11 @@ export class HostSessionRegistry { }) socket.once('close', (code, reason) => { this.observer.recordControlClose?.(code) + // Guarded on identity: a predecessor retired by a rebind must not stamp a + // cause onto the live session that replaced it. + if (session.socket === socket) { + this.hostCloseReasons.record(this.key(session.identity.sub, session.relayHostId), reason) + } // One line per control close makes reconnect churners attributable by // host digest without exposing the raw relay host id. console.warn( @@ -1187,9 +1208,16 @@ export class HostSessionRegistry { if (session.socket) send(session.socket, 'control-error', { ...(reqId ? { reqId } : {}), code }) } - private rejectClient(socket: WebSocket, code: number): void { + // hostCloseReason rides the WebSocket close reason, never relay-hello: every + // shipped phone parses relay-hello with a strict schema that rejects an + // unknown key, and none of them read the close reason at all. + private rejectClient( + socket: WebSocket, + code: number, + hostCloseReason?: RelayHostCloseReason | null + ): void { send(socket, 'relay-hello', { ok: false, code }) - closeRelayWebSocket(socket, code, 'relay connection rejected') + closeRelayWebSocket(socket, code, hostCloseReason ?? 'relay connection rejected') } private releaseControlActivity(session: HostSession): void { diff --git a/cloud/apps/relay/src/host-signed-out-rejection.test.ts b/cloud/apps/relay/src/host-signed-out-rejection.test.ts new file mode 100644 index 00000000000..0f8542c960f --- /dev/null +++ b/cloud/apps/relay/src/host-signed-out-rejection.test.ts @@ -0,0 +1,206 @@ +import { EventEmitter } from 'node:events' +import { + CONTROL_CONTINUITY_LIMITS, + RELAY_CLOSE_CODE, + RELAY_HOST_CLOSE_REASON +} from '@orca-cloud/relay-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { RelayCredentialStore } from './credential-store.js' +import { HostSessionRegistry } from './host-session-registry.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close', 1006, Buffer.alloc(0)) + }) +} + +const config = { + port: 8080, + publicUrl: 'https://relay-c3.example.com', + cellUrl: 'https://relay-c3.example.com', + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [] +} as unknown as RelayConfig + +const identity = { + sub: 'user-1', + prof: 'profile-1', + org: 'org-1', + relayHostId: 'AbCdEf0123_-xyZ9' +} as unknown as RelayTokenClaims + +const reservation = { + userId: identity.sub, + relayHostId: identity.relayHostId, + credentialKind: 'resume', + relayDeviceId: 'device-1', + leaseExpiresAt: Date.now() + 60_000 +} + +function createRegistry() { + const store = { + resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }), + reserveCredential: vi.fn().mockResolvedValue(reservation), + failReservation: vi.fn().mockResolvedValue(undefined) + } + const assignments = { + activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'), + markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined), + resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }), + acquireActivity: vi.fn().mockResolvedValue(undefined), + renewControlActivity: vi.fn().mockResolvedValue(undefined), + releaseActivity: vi.fn().mockResolvedValue(true) + } as unknown as RelayAssignmentStore + const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn(), + recordControlClose: vi.fn(), + recordSpliceClose: vi.fn() + } satisfies RelayRuntimeObserver + const registry = new HostSessionRegistry( + config, + vi.fn(), + store as unknown as RelayCredentialStore, + assignments, + new ProcessQueuedByteBudget(), + observer + ) + const activate = (socket: WebSocket, generation: number): Promise => + ( + registry as unknown as { + activate: ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string + ) => Promise + } + ).activate(socket, identity, null, generation, false, 1, '1.4.173') + return { registry, activate } +} + +async function dialPhone(registry: HostSessionRegistry): Promise { + const phone = new FakeSocket() + await registry.acceptClient(phone as unknown as WebSocket, identity.relayHostId, 'credential') + return phone +} + +// The 4404 hello body is unchanged: every shipped phone parses it with a strict +// schema, so the cause has to ride the close frame instead. +const HOST_OFFLINE_HELLO = JSON.stringify({ + type: 'relay-hello', + ok: false, + code: RELAY_CLOSE_CODE.HOST_OFFLINE +}) + +describe('host sign-out reason on phone rejection', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('names the sign-out to a phone that arrives after the host is gone', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.send).toHaveBeenCalledWith(HOST_OFFLINE_HELLO) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ) + }) + + it('says nothing when the host died without naming a cause', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('ignores a close reason the host invented', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, 'signed-out-ish') + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('forgets the sign-out once the host proves itself again', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const reconnected = new FakeSocket() + await activate(reconnected as unknown as WebSocket, 2) + // Drop it abruptly, as a network death would, so only the stale memory + // could still name a cause. + reconnected.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + // A live host is present: the 4404 there is an attach deadline, not absence. + it('never names a cause while the host control is connected', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + const phone = await dialPhone(registry) + expect(phone.close).not.toHaveBeenCalled() + expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"')) + }) +}) diff --git a/cloud/apps/relay/src/postgres-transaction-recovery.test.ts b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts index a49c07d2e7d..ae9d52a7c86 100644 --- a/cloud/apps/relay/src/postgres-transaction-recovery.test.ts +++ b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts @@ -503,12 +503,19 @@ describePostgres('PostgreSQL transaction recovery', () => { const directorLockOrder: string[] = [] const assignmentDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { if (phase === 'before') { - if (sql.includes('FROM relay_assignments WHERE user_id = ?')) { + // Only locked statements reach this hook, so classifying the pin read + // is what proves it stays unlocked: if it ever grows a FOR UPDATE it + // shows up in the order below instead of silently joining the queue. + if (sql.includes('SELECT cell_id FROM relay_assignments')) { + directorLockOrder.push('pin-read') + } else if (sql.includes('FROM relay_assignments WHERE user_id = ?')) { directorLockOrder.push('assignment') } else if (sql.includes('FROM relay_assignment_activity_leases')) { directorLockOrder.push('activity') } else if (sql.includes('FROM relay_cells ORDER BY')) { directorLockOrder.push('cell-inventory') + } else if (sql.includes('FROM relay_cells WHERE cell_id IN')) { + directorLockOrder.push('cell-rows') } else if (sql.includes('FROM relay_cells WHERE cell_id = ?')) { directorLockOrder.push('cell') } @@ -530,11 +537,14 @@ describePostgres('PostgreSQL transaction recovery', () => { }) await expect(legacyTransaction).resolves.toBeUndefined() expect(assignmentDatabase.attempts).toBe(2) + // The retry still takes a cell row before the assignment row — the order + // that avoids the legacy cycle — but only the pinned row, never the + // inventory. expect(directorLockOrder).toEqual([ 'assignment', 'activity', 'cell', - 'cell-inventory', + 'cell-rows', 'assignment', 'activity' ]) diff --git a/cloud/dev/fixtures/terraform-root-partition/families.json b/cloud/dev/fixtures/terraform-root-partition/families.json index 9664c1eb299..dfe100fd2dd 100644 --- a/cloud/dev/fixtures/terraform-root-partition/families.json +++ b/cloud/dev/fixtures/terraform-root-partition/families.json @@ -133,10 +133,15 @@ "google_logging_metric.relay_snapshot", "google_monitoring_alert_policy.relay_assignment_5xx", "google_monitoring_alert_policy.relay_assignment_edge_429", + "google_monitoring_alert_policy.relay_cell_process_exit", + "google_monitoring_alert_policy.relay_cloud_nat_port_drops", "google_monitoring_alert_policy.relay_cloud_sql_backends", + "google_monitoring_alert_policy.relay_cloud_sql_checkpoint_loop", + "google_monitoring_alert_policy.relay_cloud_sql_disk", "google_monitoring_alert_policy.relay_custom", "google_monitoring_alert_policy.relay_gce_connection_headroom", "google_monitoring_alert_policy.relay_postgres_retry_exhausted", + "google_monitoring_dashboard.relay_incident", "google_project_iam_custom_role.github_production_relay_capacity_mutation", "google_project_iam_custom_role.github_relay_asia_topology_mutation", "google_project_iam_custom_role.github_relay_asia_topology_read", diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 337d3f1b20f..870c95dd413 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -99,7 +99,7 @@ durably marked consumed before mutation and cannot authorize another run. | Cloud SQL deadlocks | over 0 | | Relay pool waiters | over 800 | | Relay pool wait | over 2,500 ms | -| PostgreSQL retries in five minutes | over 300 | +| PostgreSQL retries in five minutes | over 2,000 | | Exhausted PostgreSQL retries in five minutes | over 300 | | Director instances | outside 5–6 | | Director CPU or memory | over 80% | @@ -138,7 +138,23 @@ heartbeats, and matching live admission. `jsonPayload.event="orca_relay_postgres_transaction_retry"` in production logs: healthy-day bursts reach 234/5min with zero exhausted retries and 26% of five-minute windows over 20, while the 2026-08-23 lock-contention - incident ran roughly 2,200–3,000/5min. + incident ran roughly 2,200–3,000/5min by raw log-line count (the gate's + own `orca_relay_postgres_retries` metric read 1,510 for that window; see the + 2026-09-04 entry). +- Recalibrated the PostgreSQL-retry freeze from 300 to 2,000 per five minutes + (2026-09-04). Basis: the global `relay_cells FOR UPDATE` lock made + successful retries a steady-state rate. Measured fleet-wide (director + + cells, summed per five minutes from the `orca_relay_postgres_retries` + log metric) over 2026-09-03T05Z..2026-09-04T05Z: p50 430 / p90 924 / + p99 1,320 / max 1,504; 55% of windows over 300; only 22% of 15-minute gates + clean at 300 versus 100% at 2,000. Three read-only dry-runs on 2026-09-04 + froze on this bar (runs 33836470590, 33838698725) or on a genuine six-cell + crash storm (33837160275), blocking the same-cap roll that carries #18521 + and the `beginProof` crash guard to the 23 cells. The 2026-08-23 incident + on this metric peaked at 1,510 then 646, so retries alone no longer + separate it from today's baseline; the exhausted-retry bar (incident peak + 467 vs bar 300), director concurrency, and the pool bars carry that role. + Re-tighten after the fleet is on the 500 ms lock wait. - Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters diff --git a/cloud/infra/terraform/relay-gce-cells.tf b/cloud/infra/terraform/relay-gce-cells.tf index d6b7f3351f9..a4505ba2e37 100644 --- a/cloud/infra/terraform/relay-gce-cells.tf +++ b/cloud/infra/terraform/relay-gce-cells.tf @@ -242,6 +242,7 @@ resource "google_compute_instance_template" "relay_gce_cell" { artifact_registry_host = "${var.region}-docker.pkg.dev" relay_image = each.value.image cloud_sql_proxy_image = var.relay_gce_cloud_sql_proxy_image + cloud_sql_private_ip = var.relay_cloud_sql_private_ip # Keep cell-only plans independent from unrelated database configuration drift. cloud_sql_connection_name = local.relay_database_connection_name }) diff --git a/cloud/infra/terraform/relay-gce-foundation.tf b/cloud/infra/terraform/relay-gce-foundation.tf index aab3b4579eb..a8d64b3fcea 100644 --- a/cloud/infra/terraform/relay-gce-foundation.tf +++ b/cloud/infra/terraform/relay-gce-foundation.tf @@ -42,6 +42,12 @@ resource "google_compute_router_nat" "relay_gce" { router = google_compute_router.relay_gce[0].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce[0].id @@ -85,6 +91,12 @@ resource "google_compute_router_nat" "relay_gce_additional" { router = google_compute_router.relay_gce_additional[each.key].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce_additional[each.key].id diff --git a/cloud/infra/terraform/relay-gce-startup.sh.tftpl b/cloud/infra/terraform/relay-gce-startup.sh.tftpl index f593d94e9e5..a77466f2169 100644 --- a/cloud/infra/terraform/relay-gce-startup.sh.tftpl +++ b/cloud/infra/terraform/relay-gce-startup.sh.tftpl @@ -109,6 +109,9 @@ docker run --detach \ --user 0:0 \ --volume "$${cloudsql_dir}:/cloudsql" \ '${cloud_sql_proxy_image}' \ +%{ if cloud_sql_private_ip ~} + --private-ip \ +%{ endif ~} --unix-socket=/cloudsql \ '${cloud_sql_connection_name}' diff --git a/cloud/infra/terraform/relay-observability.tf b/cloud/infra/terraform/relay-observability.tf index a8fa4276eb2..964fc1060e6 100644 --- a/cloud/infra/terraform/relay-observability.tf +++ b/cloud/infra/terraform/relay-observability.tf @@ -37,6 +37,16 @@ locals { description = "Relay PostgreSQL transactions that exhausted bounded retry." filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR resource.type=\"gce_instance\") AND jsonPayload.event=\"orca_relay_postgres_transaction_exhausted\"" } + cell_process_exit = { + # The docker event stream is the only per-exit line: the relay's own crash footer only + # appears for unhandled rejections, and `container start` also counts healthy first boots. + description = "Relay cell container exits, one Docker `container die` event per process exit." + filter = "resource.type=\"gce_instance\" AND logName=\"projects/${var.project_id}/logs/cos_system\" AND jsonPayload.SYSLOG_IDENTIFIER=\"docker\" AND jsonPayload.MESSAGE:\"container die\" AND jsonPayload.MESSAGE:\"name=orca-relay)\"" + } + cloud_sql_wal_checkpoint = { + description = "Cloud SQL checkpoints triggered by WAL volume instead of the timed schedule; a sustained run is the fsync loop that stalled every relay process at once on 2026-09-04." + filter = "resource.type=\"cloudsql_database\" AND resource.labels.database_id=\"${var.project_id}:${local.relay_database_instance_name}\" AND textPayload:\"checkpoint starting: wal\"" + } } relay_runtime_metrics = { @@ -523,3 +533,280 @@ resource "google_monitoring_alert_policy" "relay_cloud_sql_backends" { mime_type = "text/markdown" } } + +resource "google_monitoring_alert_policy" "relay_cloud_sql_checkpoint_loop" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL checkpoint loop" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "WAL-triggered checkpoints above 3 in 5 minutes" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\"" + comparison = "COMPARISON_GT" + threshold_value = 3 + duration = "300s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Healthy operation is one timed checkpoint every 5 minutes. Repeated `checkpoint starting: wal` lines mean WAL is outrunning `max_wal_size` and every checkpoint fsync stalls all relay SQL for seconds. Check `checkpoint complete` sync= times and disk write throughput against the PD-SSD ceiling; the fix is disk size and `max_wal_size` in the Terraform root that owns the instance (orca-cloud `infra/terraform-foundation`)." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +resource "google_monitoring_alert_policy" "relay_cloud_sql_disk" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL disk utilization" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cloud SQL disk above 70%" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND resource.label.\"database_id\"=\"${var.project_id}:${local.relay_database_instance_name}\" AND metric.type=\"cloudsql.googleapis.com/database/disk/utilization\"" + comparison = "COMPARISON_GT" + threshold_value = 0.7 + duration = "600s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_MAX" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "The shared auth/relay Cloud SQL disk is filling. `refresh_tokens` is the largest table and grows without pruning; grow the disk (IOPS scale with size) before it reaches the WAL checkpoint loop, and prune revoked token rows." + mime_type = "text/markdown" + } +} + +resource "google_monitoring_alert_policy" "relay_cloud_nat_port_drops" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + display_name = "Orca Relay: Cloud NAT port exhaustion" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "NAT packets dropped for lack of ports" + + condition_threshold { + filter = "resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\") AND metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND metric.label.\"reason\"=\"OUT_OF_RESOURCES\"" + comparison = "COMPARISON_GT" + threshold_value = 0 + duration = "120s" + + aggregations { + alignment_period = "60s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"gateway_name\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Relay cells reach Cloud SQL's public IP through this NAT. Port exhaustion makes every cell's Cloud SQL Auth Proxy dial time out at once, which reads as a fleet-wide SQL stall with a healthy database. Check `nat/port_usage` per VM and raise `max_ports_per_vm` in `relay-gce-foundation.tf`, or move the database to a private IP." + mime_type = "text/markdown" + } +} + +resource "google_monitoring_alert_policy" "relay_cell_process_exit" { + project = var.project_id + display_name = "Orca Relay: cell process exits" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cell container exits above 3 in 15 minutes" + + condition_threshold { + filter = "resource.type=\"gce_instance\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cell_process_exit\"" + comparison = "COMPARISON_GT" + threshold_value = 3 + duration = "0s" + + aggregations { + alignment_period = "900s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"instance_id\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "A Relay GCE cell restarted its container more than three times in 15 minutes. Each exit drops every host and phone on that cell, and 201 exits went unpaged over 48 h on 2026-09-04. The instance hostname is `relay--`; read `jsonPayload.MESSAGE` on `cos_system` for the exit code and the container's own stderr for the stack before blaming MIG autoheal or load. A same-capacity roll is the remedy when the running image is behind." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +# Why: the four signals that had to be assembled by hand during the 2026-09-04 incident. +resource "google_monitoring_dashboard" "relay_incident" { + project = var.project_id + + dashboard_json = jsonencode({ + displayName = "Orca Relay: incident overview" + mosaicLayout = { + columns = 12 + tiles = [ + { + xPos = 0 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Cloud SQL WAL checkpoints" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\" AND resource.type=\"cloudsql_database\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "checkpoints" + scale = "LINEAR" + } + } + } + }, + { + xPos = 3 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Cloud NAT dropped packets" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\")" + aggregation = { + alignmentPeriod = "60s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + groupByFields = ["resource.label.\"gateway_name\"", "metric.label.\"reason\""] + } + } + } + }] + yAxis = { + label = "packets" + scale = "LINEAR" + } + } + } + }, + { + xPos = 6 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Auth refresh 401s" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + filter = "metric.type=\"logging.googleapis.com/user/orca_auth_refresh_401\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_SUM" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "rejections" + scale = "LINEAR" + } + } + } + }, + { + xPos = 9 + yPos = 0 + width = 3 + height = 4 + widget = { + title = "Standing desktop controls (fleet sum)" + xyChart = { + dataSets = [{ + plotType = "LINE" + targetAxis = "Y1" + timeSeriesQuery = { + timeSeriesFilter = { + # ALIGN_MEAN, not ALIGN_SUM: each process reports its standing control count once per interval. + filter = "metric.type=\"logging.googleapis.com/user/orca_relay_controls\"" + aggregation = { + alignmentPeriod = "300s" + perSeriesAligner = "ALIGN_MEAN" + crossSeriesReducer = "REDUCE_SUM" + } + } + } + }] + yAxis = { + label = "controls" + scale = "LINEAR" + } + } + } + } + ] + } + }) + + depends_on = [google_logging_metric.relay_incident, google_logging_metric.relay_snapshot] +} diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf index 68f6c555bd3..91f67e8ebe0 100644 --- a/cloud/infra/terraform/variables.tf +++ b/cloud/infra/terraform/variables.tf @@ -468,6 +468,12 @@ variable "relay_gce_fenced_cells" { default = [] } +variable "relay_cloud_sql_private_ip" { + type = bool + description = "Dial Cloud SQL over its private IP inside this VPC instead of its public IP through Cloud NAT. Requires the foundation root's private services access peering to be applied first; a cell that cannot reach the private IP never becomes ready." + default = false +} + variable "relay_gce_cloud_sql_proxy_image" { type = string description = "Digest-pinned Cloud SQL Auth Proxy image used by private relay workers." diff --git a/cloud/packages/relay-contract/src/host-close-reason.ts b/cloud/packages/relay-contract/src/host-close-reason.ts new file mode 100644 index 00000000000..3a5abde3f00 --- /dev/null +++ b/cloud/packages/relay-contract/src/host-close-reason.ts @@ -0,0 +1,18 @@ +// Mirror of src/shared/relay-host-close-reason.ts in the Orca app repo half. +// A host control socket may close with one of these as its WebSocket close +// reason; the cell records it so a later phone rejection can name the cause. +// Anything else (including the empty reason of an abrupt 1006) means "unknown", +// which is what every peer that predates this file sends. +export const RELAY_HOST_CLOSE_REASON = { + SIGNED_OUT: 'signed-out' +} as const + +export type RelayHostCloseReason = + (typeof RELAY_HOST_CLOSE_REASON)[keyof typeof RELAY_HOST_CLOSE_REASON] + +const REASONS: readonly string[] = Object.values(RELAY_HOST_CLOSE_REASON) + +export function relayHostCloseReasonFrom(value: unknown): RelayHostCloseReason | null { + const text = typeof value === 'string' ? value : (value?.toString() ?? '') + return REASONS.includes(text) ? (text as RelayHostCloseReason) : null +} diff --git a/cloud/packages/relay-contract/src/index.ts b/cloud/packages/relay-contract/src/index.ts index 2a7d7d0feda..aab3b53b5f3 100644 --- a/cloud/packages/relay-contract/src/index.ts +++ b/cloud/packages/relay-contract/src/index.ts @@ -5,6 +5,7 @@ export * from './control-messages.js' export * from './control-continuity.js' export * from './credential-messages.js' export * from './director-messages.js' +export * from './host-close-reason.js' export * from './host-proof-transcript.js' export * from './persistence-invariants.js' export * from './protocol-limits.js' diff --git a/config/patches/node-pty@1.1.0.patch b/config/patches/node-pty@1.1.0.patch index ff474f7d95e..8f5045b932a 100644 --- a/config/patches/node-pty@1.1.0.patch +++ b/config/patches/node-pty@1.1.0.patch @@ -603,7 +603,7 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..2ae787c5bd4f3eba470584dc658a01a5 } #endif diff --git a/src/win/conpty.cc b/src/win/conpty.cc -index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a6a4082ce 100644 +index 7b286d3d644c26141df516929703aa6e129df4b2..4aed260dd68e6a171dcfd349e9a7c5c97209248e 100644 --- a/src/win/conpty.cc +++ b/src/win/conpty.cc @@ -18,6 +18,7 @@ @@ -614,7 +614,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a #include #include #include -@@ -44,12 +45,29 @@ struct pty_baton { +@@ -44,12 +45,39 @@ struct pty_baton { HANDLE hOut; HPCON hpc; @@ -630,22 +630,32 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + // refused to create or assign one (an outer job without breakaway rights), + // in which case callers fall back to their pre-job behaviour. + HANDLE hJob = nullptr; ++ ++ // Orca: teardown needs BOTH the shell's death and an explicit kill() before ++ // the baton can be freed, so each side records that it has run. Whichever ++ // arrives second frees it. Freeing on the shell's death alone -- what this ++ // file did before -- destroyed the only record of `hpc` while ++ // ClosePseudoConsole was still owed, which is why a self-exiting shell ++ // leaked its pseudoconsole and the console host it reaps (#18601 / F24). ++ bool shellExited = false; ++ bool consoleClosed = false; pty_baton(int _id, HANDLE _hIn, HANDLE _hOut, HPCON _hpc) : id(_id), hIn(_hIn), hOut(_hOut), hpc(_hpc) {}; }; static std::vector> ptyHandles; -+// Orca: guards the job accessors below against the exit watcher thread. It does -+// NOT make the whole table safe -- PtyResize/PtyClear/PtyKill read it unlocked, -+// as they always have -- but it closes the window this patch opened, where the -+// watcher can close hShell/hJob and free the baton between a lookup and its use. ++// Orca: guards the job accessors below, and PtyKill, against the exit watcher ++// thread. It does NOT make the whole table safe -- PtyResize and PtyClear still ++// read it unlocked, as they always have -- but it closes the window this patch ++// opened, where the watcher can close hShell/hJob and free the baton between a ++// lookup and its use. +// Handle VALUES are recycled aggressively, so an unguarded read could pass the +// shell-pid check against an unrelated process and terminate the wrong job. +static std::mutex ptyJobMutex; static volatile LONG ptyCounter; static pty_baton* get_pty_baton(int id) { -@@ -102,8 +120,27 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) { +@@ -102,8 +130,31 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) { // Get process exit code. GetExitCodeProcess(baton->hShell, (LPDWORD)(&exit_event->exit_code)); // Clean up handles @@ -665,9 +675,13 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + // Why inside the lock: erasing frees the baton the job accessors hold a + // pointer to. Note remove_pty_baton must not be an assert() argument -- + // NDEBUG would compile the call away and leak every baton. -+ const bool removed = remove_pty_baton(baton->id); -+ assert(removed); -+ (void)removed; ++ baton->shellExited = true; ++ if (baton->consoleClosed) { ++ const bool removed = remove_pty_baton(baton->id); ++ assert(removed); ++ (void)removed; ++ } ++ // Else PtyKill has not run yet and still owns hpc. It frees the baton. + } + // Why the lock ends here: BlockingCall below waits on the JS thread, and the + // JS thread can be waiting on ptyJobMutex inside PtyTerminateJob. Holding @@ -675,7 +689,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a auto status = tsfn.BlockingCall(exit_event, callback); // In main thread switch (status) { -@@ -409,6 +446,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -409,6 +460,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { throw errorWithCode(info, "UpdateProcThreadAttribute failed"); } @@ -691,7 +705,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a PROCESS_INFORMATION piClient{}; fSuccess = !!CreateProcessW( nullptr, -@@ -416,7 +462,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -416,7 +476,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { nullptr, // lpProcessAttributes nullptr, // lpThreadAttributes false, // bInheritHandles VERY IMPORTANT that this is false @@ -703,7 +717,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a envArg, // lpEnvironment mutableCwd.get(), // lpCurrentDirectory &siEx.StartupInfo, // lpStartupInfo -@@ -426,8 +475,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -426,8 +489,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { throw errorWithCode(info, "Cannot create process"); } @@ -753,7 +767,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a if (useConptyDll && fLoadedDll) { PFNRELEASEPSEUDOCONSOLE const pfnReleasePseudoConsole = (PFNRELEASEPSEUDOCONSOLE)GetProcAddress( -@@ -440,6 +528,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -440,6 +542,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { // Update handle handle->hShell = piClient.hProcess; @@ -762,7 +776,91 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a // Close the thread handle to avoid resource leak CloseHandle(piClient.hThread); -@@ -567,6 +657,143 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) { +@@ -544,29 +648,215 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) { + int id = info[0].As().Int32Value(); + const bool useConptyDll = info[1].As().Value(); + +- const pty_baton* handle = get_pty_baton(id); ++ // Orca: resolve the DLL BEFORE touching any baton state, for the same reason ++ // PtyConnect does it before creating anything. LoadConptyDll throws when ++ // conpty.dll is missing, and a throw after consoleClosed was set would strand ++ // the pseudoconsole permanently: the retry would find the work already ++ // claimed and do nothing. Only the useConptyDll path can throw here; the ++ // other returns kernel32. ++ HANDLE hLibrary = LoadConptyDll(info, useConptyDll); ++ PFNCLOSEPSEUDOCONSOLE pfnClosePseudoConsole = nullptr; ++ if (hLibrary != nullptr) { ++ pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress( ++ (HMODULE)hLibrary, ++ useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole"); ++ } + +- if (handle != nullptr) { +- HANDLE hLibrary = LoadConptyDll(info, useConptyDll); +- bool fLoadedDll = hLibrary != nullptr; +- if (fLoadedDll) +- { +- PFNCLOSEPSEUDOCONSOLE const pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress( +- (HMODULE)hLibrary, +- useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole"); +- if (pfnClosePseudoConsole) +- { +- pfnClosePseudoConsole(handle->hpc); ++ // Orca: the baton now outlives the shell, so this runs on a self-exited pty ++ // too -- that is the whole point. Take what we need under the lock: the ++ // watcher thread nulls hShell the moment the shell dies, and TerminateProcess ++ // on a handle it just closed is an invalid-handle operation. Duplicating ++ // rather than reordering keeps upstream's close-then-terminate sequence. ++ HPCON hpc = nullptr; ++ HANDLE hShellDup = nullptr; ++ bool owed = false; ++ { ++ std::lock_guard guard(ptyJobMutex); ++ pty_baton* handle = get_pty_baton(id); ++ // Why the consoleClosed check: a second kill() would otherwise close the ++ // same pseudoconsole twice. Upstream relied on the baton being gone. ++ if (handle != nullptr && !handle->consoleClosed) { ++ hpc = handle->hpc; ++ owed = true; ++ handle->consoleClosed = true; ++ // Null hShell means a self-exited pty, where there is nothing to kill. ++ if (useConptyDll && handle->hShell != nullptr) { ++ if (!DuplicateHandle(GetCurrentProcess(), handle->hShell, GetCurrentProcess(), ++ &hShellDup, 0, FALSE, DUPLICATE_SAME_ACCESS)) { ++ // Why terminate here instead of skipping: a failed duplication leaves ++ // hShellDup null, which is indistinguishable from the self-exit case, ++ // and skipping would leave the shell RUNNING after its pane closed -- ++ // a worse outcome than the leak this all exists to fix. hShell is ++ // valid under this lock and TerminateProcess does not block, so the ++ // only cost is that this rare path kills before the console closes. ++ hShellDup = nullptr; ++ TerminateProcess(handle->hShell, 1); ++ } ++ } ++ if (handle->shellExited) { ++ const bool removed = remove_pty_baton(id); ++ assert(removed); ++ (void)removed; + } ++ // Else the shell is still running and the watcher frees the baton. + } +- if (useConptyDll) { +- TerminateProcess(handle->hShell, 1); ++ } ++ ++ // Why outside the lock: ClosePseudoConsole blocks until the conout side has ++ // drained, and the watcher must be able to take the lock while it does. ++ if (owed) { ++ if (pfnClosePseudoConsole) ++ { ++ pfnClosePseudoConsole(hpc); ++ } ++ if (hShellDup != nullptr) { ++ TerminateProcess(hShellDup, 1); ++ CloseHandle(hShellDup); + } + } + return env.Undefined(); } @@ -808,9 +906,11 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + * Orca: the pids still alive in this pty's tree, straight from the kernel. + * + * Descendant liveness for a tree that is still tracked, including children that -+ * detached from the console. Once the shell exits the baton is gone, so this -+ * returns null rather than an empty list -- null means "no answer", never -+ * "they died". Also returns null when no job was assigned. ++ * detached from the console. Once the shell exits the watcher nulls hJob, which ++ * ownsShell rejects, so this returns null rather than an empty list -- null ++ * means "no answer", never "they died". (The baton itself now outlives the ++ * shell, until kill() runs; hJob is what makes the answer null.) Also returns ++ * null when no job was assigned. + * + * Does not include the ConPTY console host: CreatePseudoConsole spawns it + * before this job exists, so it is not a member and ClosePseudoConsole is what @@ -906,7 +1006,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a /** * Init */ -@@ -577,6 +804,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) { +@@ -577,6 +867,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) { exports.Set("resize", Napi::Function::New(env, PtyResize)); exports.Set("clear", Napi::Function::New(env, PtyClear)); exports.Set("kill", Napi::Function::New(env, PtyKill)); @@ -917,7 +1017,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a }; diff --git a/lib/windowsPtyAgent.js b/lib/windowsPtyAgent.js -index a358ffb..fb3a96f 100644 +index a358ffb177357e177661033c1b092f9c9d0e5f5a..26c2a4c58799ce649f5113131e4c52f7ed2d87ad 100644 --- a/lib/windowsPtyAgent.js +++ b/lib/windowsPtyAgent.js @@ -136,6 +136,9 @@ var WindowsPtyAgent = /** @class */ (function () { @@ -930,6 +1030,20 @@ index a358ffb..fb3a96f 100644 this._outSocket.readable = false; this._getConsoleProcessList().then(function (consoleProcessList) { consoleProcessList.forEach(function (pid) { +@@ -154,9 +157,10 @@ var WindowsPtyAgent = /** @class */ (function () { + // Close the input write handle to signal the end of session. + this._inSocket.destroy(); + this._ptyNative.kill(this._pty, this._useConptyDll); +- this._outSocket.on('data', function () { +- _this._conoutSocketWorker.dispose(); +- }); ++ // Orca: dispose unconditionally, as the non-DLL branch above does. ++ // Waiting for another 'data' event leaks the conout worker on every ++ // self-exiting shell, because no more data ever arrives (F24). ++ this._conoutSocketWorker.dispose(); + } + } + else { diff --git a/lib/windowsTerminal.js b/lib/windowsTerminal.js index 3c38f89..e20b3e6 100644 --- a/lib/windowsTerminal.js @@ -1015,7 +1129,7 @@ index 3c38f89..e20b3e6 100644 \ No newline at end of file +//# sourceMappingURL=windowsTerminal.js.map diff --git a/src/windowsPtyAgent.ts b/src/windowsPtyAgent.ts -index d705444..ce611b8 100644 +index d7054449516f0c9a62af351c2caa17331206d530..0c28a32e2e1db2b3f208ddde8443cd4e67bb1ad6 100644 --- a/src/windowsPtyAgent.ts +++ b/src/windowsPtyAgent.ts @@ -143,6 +143,9 @@ export class WindowsPtyAgent { @@ -1028,6 +1142,20 @@ index d705444..ce611b8 100644 this._outSocket.readable = false; this._getConsoleProcessList().then(consoleProcessList => { consoleProcessList.forEach((pid: number) => { +@@ -159,9 +162,10 @@ export class WindowsPtyAgent { + // Close the input write handle to signal the end of session. + this._inSocket.destroy(); + (this._ptyNative as IConptyNative).kill(this._pty, this._useConptyDll); +- this._outSocket.on('data', () => { +- this._conoutSocketWorker.dispose(); +- }); ++ // Orca: dispose unconditionally, as the non-DLL branch above does. ++ // Waiting for another 'data' event leaks the conout worker on every ++ // self-exiting shell, because no more data ever arrives (F24). ++ this._conoutSocketWorker.dispose(); + } + } else { + // Because pty.kill closes the handle, it will kill most processes by itself. diff --git a/src/windowsTerminal.ts b/src/windowsTerminal.ts index 13f6c6d..eda63c8 100644 --- a/src/windowsTerminal.ts diff --git a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs new file mode 100644 index 00000000000..dd26784ee46 --- /dev/null +++ b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs @@ -0,0 +1,158 @@ +const { createHash } = require('node:crypto') +const { readFileSync, renameSync, rmSync, writeFileSync } = require('node:fs') +const { join, resolve } = require('node:path') + +/** + * Release the ConPTY teardown handles a relay's npm-installed node-pty never releases. + * + * Two files, and the ORDER of one of the edits is the whole fix. + * + * `windowsPtyAgent.js` -- `kill()` flips `readable` on both sockets and destroys neither. + * `_cleanUpProcess` destroys `_outSocket`, so the conout handle comes back; nothing ever destroys + * `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`. + * Every terminal leaks one File handle for the life of the host process. + * + * The obvious fix -- and the one the desktop patch ships -- releases it at the TOP of the branch, + * before `_getConsoleProcessList()` forks and before the native kill. That is measurably worse than + * leaving the leak alone: teardown aborts partway, the forked console-list agent is never reaped, + * and both pipe handles stay alive instead of one. This asset releases it at the END of the branch + * instead, after the fork and the kill have already happened. + * + * Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type + * (identical numbers standalone and through a real relay): + * + * published node-pty File +1/terminal, Process flat + * desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE + * released last (here) File flat, Process flat + * + * `windowsTerminal.js` carries the desktop's error-listener hunks verbatim. The conin listener is + * what keeps a pipe error retiring one terminal instead of the host -- its own comment names the + * failure mode: "Without a listener, Node promotes errors such as write EAGAIN to uncaughtException". + * It is not what fixes the leak (adding it changed nothing on its own), but it is the guard that + * makes destroying conin safe at all. + * + * Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm + * patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there. + * + * DELIBERATE DIVERGENCE FROM THE DESKTOP: the desktop patch has the early placement and therefore + * the +2 File / +1 Process regression, measured against its exact installed tree. Correcting it + * there is a separate change with its own verification, so the two trees differ on this one hunk on + * purpose, and the test pins that so a future "sync the patches" does not copy the bug back. + * + * NOT ADDRESSED, AND A SEPARATE DEFECT THAT IS STILL OPEN: a terminal that exits on its own is + * still torn down through `kill()` -- both hosts call `destroy()` on natural exit and + * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, and the + * ordering this patch relies on does not hold. Measured over 20 self-exit cycles with that + * `destroy()` issued: published +3 File/+1 Process per terminal, desktop-patched +2/+1, this tree + * +2/+1. So this patch does not close it and the desktop patch does not either. It is reachable + * for every Windows user, local and relay, on every terminal closed by typing `exit`. + */ + +const EXPECTED_NODE_PTY_VERSION = '1.1.0' + +/** Each entry is one published file, its patched form, and the edits between them. */ +const PATCH_TARGETS = [ + { + relativePath: ['lib', 'windowsPtyAgent.js'], + originalSha256: '8636d16b38266112204061a22b135734177c242837982fd3a4055be726efa64a', + patchedSha256: '1e23ef480569e73706e3ab4f5482c7e553c76f51414ae8e7b0bdcc2fd75f7280', + replacements: [ + [ + ' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n', + ' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n // Orca: released AFTER the console-list fork and the native kill, not before them.\n // Destroying conin first aborts teardown partway -- measured on a Windows SSH relay\n // as +2 File and +1 Process handles per terminal, against +1 File unpatched.\n this._inSocket.destroy();\n' + ] + ] + }, + { + relativePath: ['lib', 'windowsTerminal.js'], + originalSha256: 'c3a65716f53fed0135a8a633373d5f9c2ab092544d651f27ef0a67096dd3bcd9', + patchedSha256: '8247ecd69be8b18257050fb026b290024612c5ffc6d492ff1d46f81e613be2cf', + replacements: [ + [ + ' _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;', + " _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Attach before readiness so a broken ConPTY output pipe cannot be unhandled.\n _this._socket.on('error', function (err) {\n var code = err && err.code;\n // PTY output can report EPIPE before `_close()` wins the race.\n _this._close();\n if (code === 'EPIPE' || code === 'ERR_STREAM_PUSH_AFTER_EOF' || code === 'ERR_STREAM_DESTROYED') {\n return;\n }\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (typeof code === 'string') {\n if (~code.indexOf('errno 5') || ~code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;" + ], + [ + " }\n });\n // Shutdown if `error` event is emitted.\n _this._socket.on('error', function (err) {\n // Close terminal session.\n _this._close();\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (err.code) {\n if (~err.code.indexOf('errno 5') || ~err.code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {", + " }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {" + ], + [ + ' _this._readable = true;\n _this._writable = true;\n _this._forwardEvents();\n return _this;', + " _this._readable = true;\n _this._writable = true;\n // A ConPTY input-pipe error must retire only this terminal. Without a listener, Node promotes\n // errors such as write EAGAIN to uncaughtException and kills every PTY in the daemon.\n _this._agent.inSocket.on('error', function () {\n if (!_this._writable) {\n return;\n }\n _this._close();\n try {\n _this._agent.kill();\n }\n catch (_a) {\n // The failing pipe may have raced process exit; the terminal is already unwritable.\n }\n });\n _this._forwardEvents();\n return _this;" + ], + [ + 'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map', + 'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map\n' + ] + ] + } +] + +function inspectTarget(relayDir, target) { + const nodePtyDir = resolve(relayDir, 'node_modules', 'node-pty') + const packageJson = JSON.parse(readFileSync(join(nodePtyDir, 'package.json'), 'utf8')) + if (packageJson.version !== EXPECTED_NODE_PTY_VERSION) { + throw new Error( + `Refusing to patch node-pty ${packageJson.version}; expected ${EXPECTED_NODE_PTY_VERSION}` + ) + } + const filePath = join(nodePtyDir, ...target.relativePath) + return { filePath, source: readFileSync(filePath, 'utf8') } +} + +function assertPatchedNodePtyWindowsTeardown(relayDir = process.cwd()) { + for (const target of PATCH_TARGETS) { + const inspected = inspectTarget(relayDir, target) + if (sourceSha256(inspected.source) !== target.patchedSha256) { + throw new Error( + `node-pty ConPTY teardown release is not installed in ${target.relativePath.join('/')}` + ) + } + } +} + +function patchNodePtyWindowsTeardown(relayDir = process.cwd()) { + for (const target of PATCH_TARGETS) { + const inspected = inspectTarget(relayDir, target) + const sourceHash = sourceSha256(inspected.source) + if (sourceHash === target.patchedSha256) { + continue + } + if (sourceHash !== target.originalSha256) { + throw new Error( + `Refusing to patch unexpected node-pty source in ${target.relativePath.join('/')}` + ) + } + let patchedSource = inspected.source + for (const [from, to] of target.replacements) { + // Why the count check: an anchor that matched twice would patch the wrong site silently, and + // the hash below would then reject a tree this script had already rewritten. + if (patchedSource.split(from).length - 1 !== 1) { + throw new Error(`Refusing to patch ${target.relativePath.join('/')}; anchor is not unique`) + } + patchedSource = patchedSource.replace(from, to) + } + const temporaryPath = `${inspected.filePath}.orca-patch-${process.pid}` + // Why: a terminated remote install must leave either known source version recoverable on reconnect. + try { + writeFileSync(temporaryPath, patchedSource) + renameSync(temporaryPath, inspected.filePath) + } finally { + rmSync(temporaryPath, { force: true }) + } + } + assertPatchedNodePtyWindowsTeardown(relayDir) +} + +function sourceSha256(source) { + return createHash('sha256').update(source).digest('hex') +} + +if (require.main === module) { + patchNodePtyWindowsTeardown() +} + +module.exports = { + assertPatchedNodePtyWindowsTeardown, + patchNodePtyWindowsTeardown +} diff --git a/config/scripts/build-relay.mjs b/config/scripts/build-relay.mjs index 289c7a957bd..4d408712f97 100644 --- a/config/scripts/build-relay.mjs +++ b/config/scripts/build-relay.mjs @@ -57,6 +57,13 @@ const NODE_PTY_CONSOLE_LIST_PATCH_SOURCE = join( 'relay-assets', NODE_PTY_CONSOLE_LIST_PATCH_FILENAME ) +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME = 'node-pty-1.1.0-windows-pty-teardown-patch.cjs' +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_SOURCE = join( + ROOT, + 'config', + 'relay-assets', + NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME +) const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' const NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE = join( ROOT, @@ -132,6 +139,10 @@ for (const platform of RELAY_BUILD_PLATFORMS) { NODE_PTY_CONSOLE_LIST_PATCH_SOURCE, join(outDir, NODE_PTY_CONSOLE_LIST_PATCH_FILENAME) ) + copyFileSync( + NODE_PTY_WINDOWS_TEARDOWN_PATCH_SOURCE, + join(outDir, NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME) + ) } copyFileSync( NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE, diff --git a/config/scripts/locale-ko-key-overrides.json b/config/scripts/locale-ko-key-overrides.json index f368ecc3cbc..bf5f62d1fa5 100644 --- a/config/scripts/locale-ko-key-overrides.json +++ b/config/scripts/locale-ko-key-overrides.json @@ -492,7 +492,7 @@ "ko": "agent CLI를 찾지 못했습니다. 하나를 설치하거나 설정에서 기본 agent를 선택하세요." }, "auto.components.Terminal.7958465754": { - "ko": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" + "ko": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" }, "auto.components.Terminal.cdc9ac4b2d": { "ko": "편집기" diff --git a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs new file mode 100644 index 00000000000..64fb1b056b8 --- /dev/null +++ b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs @@ -0,0 +1,213 @@ +// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with the +// desktop's own node-pty patch. pnpm patches do not cross the SSH boundary, so a relay runs the tree +// `npm install` put there; the desktop had this fix and the relay did not, and every terminal on a +// Windows SSH host leaked one File handle for the life of the relay process. +// +// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- what the +// desktop patch does -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a new +// Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +import { createRequire } from 'node:module' +import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const { + assertPatchedNodePtyWindowsTeardown, + patchNodePtyWindowsTeardown +} = require('../relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs') +const projectDir = resolve(import.meta.dirname, '..', '..') +const cleanupDirs = [] + +const PATCHED_FILES = ['windowsPtyAgent.js', 'windowsTerminal.js'] + +/** The hunks config/patches/node-pty@1.1.0.patch adds to the installed desktop tree. */ +const DESKTOP_HUNKS = { + 'windowsPtyAgent.js': [ + [ + [ + ' this._inSocket.readable = false;', + ' // The non-DLL path previously only flipped `readable`, leaving the', + ' // conin PipeWrap alive until the host exited (#947).', + ' this._inSocket.destroy();', + ' this._outSocket.readable = false;', + '' + ].join('\n'), + [ + ' this._inSocket.readable = false;', + ' this._outSocket.readable = false;', + '' + ].join('\n') + ], + // The useConptyDll branch, which only the DESKTOP runs -- the relay takes the + // non-DLL branch above, where the dispose is already unconditional. Listed here + // so un-applying still yields published; the relay asset needs no counterpart. + [ + [ + ' // Orca: dispose unconditionally, as the non-DLL branch above does.', + " // Waiting for another 'data' event leaks the conout worker on every", + ' // self-exiting shell, because no more data ever arrives (F24).', + ' this._conoutSocketWorker.dispose();', + '' + ].join('\n'), + [ + " this._outSocket.on('data', function () {", + ' _this._conoutSocketWorker.dispose();', + ' });', + '' + ].join('\n') + ] + ], + 'windowsTerminal.js': [ + [ + ' // Attach before readiness so a broken ConPTY output pipe cannot be unhandled.', + null + ], + [' // A ConPTY input-pipe error must retire only this terminal.', null] + ] +} + +function desktopPath(file) { + return join(projectDir, 'node_modules', 'node-pty', 'lib', file) +} + +afterEach(() => { + for (const dir of cleanupDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +describe('Windows SSH relay node-pty ConPTY teardown patch', () => { + // Why reconstruct rather than vendor upstream: the installed tree IS the published file plus the + // desktop's hunks, so un-applying them yields upstream exactly -- and pinning that against this + // asset's own hashes is what fails loudly if either side of the pair moves. + it('takes the desktop error listeners verbatim', () => { + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + + expect(readFileSync(join(fixture.libDir, 'windowsTerminal.js'), 'utf8')).toBe( + readFileSync(desktopPath('windowsTerminal.js'), 'utf8') + ) + }) + + // The one hunk that must NOT match the desktop, and the reason is measured, not stylistic: + // releasing conin before `_getConsoleProcessList()` forks aborts teardown partway. + it('releases conin after the console-list fork, not before it like the desktop patch', () => { + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + const patched = readFileSync(join(fixture.libDir, 'windowsPtyAgent.js'), 'utf8') + + const branch = patched.slice( + patched.indexOf('if (!this._useConptyDll) {'), + patched.indexOf('else {', patched.indexOf('if (!this._useConptyDll) {')) + ) + expect(branch).toContain('this._inSocket.destroy();') + expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( + branch.indexOf('this._conoutSocketWorker.dispose();') + ) + expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( + branch.indexOf('this._getConsoleProcessList()') + ) + // Pinned so a future "sync the relay asset to config/patches" cannot copy the regression back. + expect(patched).not.toBe(readFileSync(desktopPath('windowsPtyAgent.js'), 'utf8')) + }) + + it('installs and verifies idempotently', () => { + const fixture = writeNodePtyFixture('1.1.0') + + patchNodePtyWindowsTeardown(fixture.root) + const once = PATCHED_FILES.map((file) => readFileSync(join(fixture.libDir, file), 'utf8')) + for (const file of PATCHED_FILES) { + expect(existsSync(`${join(fixture.libDir, file)}.orca-patch-${process.pid}`)).toBe(false) + } + expect(() => assertPatchedNodePtyWindowsTeardown(fixture.root)).not.toThrow() + + patchNodePtyWindowsTeardown(fixture.root) + expect(PATCHED_FILES.map((file) => readFileSync(join(fixture.libDir, file), 'utf8'))).toEqual( + once + ) + }) + + it('refuses a different package version or unexpected source', () => { + const wrongVersion = writeNodePtyFixture('1.2.0-beta.11') + expect(() => patchNodePtyWindowsTeardown(wrongVersion.root)).toThrow('expected 1.1.0') + + for (const file of PATCHED_FILES) { + const drifted = writeNodePtyFixture('1.1.0') + const path = join(drifted.libDir, file) + writeFileSync(path, `${readFileSync(path, 'utf8')}\n// drift`) + expect(() => patchNodePtyWindowsTeardown(drifted.root)).toThrow('unexpected node-pty') + } + }) + + it('refuses a half-applied tree, so one file cannot pass for both', () => { + for (const file of PATCHED_FILES) { + const partial = writeNodePtyFixture('1.1.0') + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + writeFileSync(join(partial.libDir, file), readFileSync(join(fixture.libDir, file), 'utf8')) + expect(() => assertPatchedNodePtyWindowsTeardown(partial.root)).toThrow('is not installed') + } + }) +}) + +/** A published node-pty tree, rebuilt by un-applying the desktop hunks from the installed one. */ +function writeNodePtyFixture(version) { + const root = mkdtempSync(join(projectDir, '.node-pty-teardown-patch-test-')) + cleanupDirs.push(root) + const libDir = join(root, 'node_modules', 'node-pty', 'lib') + mkdirSync(libDir, { recursive: true }) + writeFileSync(join(root, 'node_modules', 'node-pty', 'package.json'), JSON.stringify({ version })) + for (const file of PATCHED_FILES) { + const desktop = readFileSync(desktopPath(file), 'utf8') + for (const [marker] of DESKTOP_HUNKS[file]) { + expect(desktop).toContain(marker) + } + writeFileSync(join(libDir, file), unapplyDesktopHunks(file, desktop)) + } + return { root, libDir } +} + +/** + * Reverse of the published-to-desktop transform. + * + * `windowsTerminal.js` is taken verbatim from the desktop, so the asset's own replacement table is + * the transform and reversing it is exact. `windowsPtyAgent.js` deliberately diverges, so its + * published form is rebuilt from the desktop hunk instead -- which is also what makes this file the + * place that notices if the desktop hunk itself ever moves. + */ +function unapplyDesktopHunks(file, desktop) { + if (file === 'windowsPtyAgent.js') { + let published = desktop + for (const [patched, original] of DESKTOP_HUNKS[file]) { + expect(published.split(patched).length - 1).toBe(1) + published = published.replace(patched, original) + } + return published + } + const asset = readFileSync( + join(projectDir, 'config', 'relay-assets', 'node-pty-1.1.0-windows-pty-teardown-patch.cjs'), + 'utf8' + ) + const { PATCH_TARGETS } = loadPatchTargets(asset) + const target = PATCH_TARGETS.find((entry) => entry.relativePath.at(-1) === file) + expect(target).toBeDefined() + let published = desktop + for (const [from, to] of target.replacements.toReversed()) { + expect(published.split(to).length - 1).toBe(1) + published = published.replace(to, from) + } + return published +} + +function loadPatchTargets(assetSource) { + const module = { exports: {} } + const factory = new Function( + 'module', + 'exports', + 'require', + `${assetSource}\nmodule.exports.PATCH_TARGETS = PATCH_TARGETS` + ) + factory(module, module.exports, require) + return module.exports +} diff --git a/config/scripts/skill-description-length.test.mjs b/config/scripts/skill-description-length.test.mjs new file mode 100644 index 00000000000..e7a9db79541 --- /dev/null +++ b/config/scripts/skill-description-length.test.mjs @@ -0,0 +1,39 @@ +import { readdirSync, readFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' + +const skillsDir = resolve(import.meta.dirname, '../../skills') +// Why: the Agent Skills spec caps `description` at 1024 chars and conforming installers +// reject the whole skill (#17935); the frontmatter is what the installer parses, so check it. +const MAX_DESCRIPTION_LENGTH = 1024 + +function readDescription(skillName) { + const skillMarkdown = readFileSync(join(skillsDir, skillName, 'SKILL.md'), 'utf8') + const frontmatter = /^---\r?\n([\s\S]*?)\r?\n---\r?\n/u.exec(skillMarkdown)?.[1] + + expect(frontmatter, `${skillName}: missing frontmatter`).toBeDefined() + + return parse(frontmatter ?? '').description +} + +describe('bundled skill descriptions', () => { + const skillNames = readdirSync(skillsDir, { withFileTypes: true }) + .filter((entry) => entry.isDirectory()) + .map((entry) => entry.name) + + it('discovers the bundled skills', () => { + expect(skillNames).toContain('orchestration') + }) + + it.each(skillNames)('%s keeps description within the Agent Skills spec limit', (name) => { + const description = readDescription(name) + + expect(typeof description, `${name}: description must be a string`).toBe('string') + expect(description.trim().length, `${name}: description is empty`).toBeGreaterThan(0) + expect( + description.length, + `${name}: description is ${description.length} chars` + ).toBeLessThanOrEqual(MAX_DESCRIPTION_LENGTH) + }) +}) diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index ef8ebb61bb4..c240c965fee 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 38m + + downloads: 39m @@ -15,7 +15,7 @@ downloads downloads - 38m - 38m + 39m + 39m diff --git a/docs/assets/wechat-qr-group9.jpg b/docs/assets/wechat-qr-group9.jpg new file mode 100644 index 00000000000..2bf46a28c3d Binary files /dev/null and b/docs/assets/wechat-qr-group9.jpg differ diff --git a/docs/readme/README.fr.md b/docs/readme/README.fr.md index 97c78d4e713..e601abc2344 100644 --- a/docs/readme/README.fr.md +++ b/docs/readme/README.fr.md @@ -243,9 +243,9 @@ Associez-la à l'app de bureau pour surveiller et piloter vos agents depuis votr - **Discord :** Rejoignez la communauté sur **[Discord](https://discord.gg/fzjDKHxv8Q)**. - **Twitter / X :** Suivez **[@orca_build](https://x.com/orca_build)** pour les news et annonces. -- **WeChat :** Scannez pour rejoindre le groupe WeChat 8 de la communauté Orca. +- **WeChat :** Scannez pour rejoindre le groupe WeChat 8 de la communauté Orca. Le groupe 8 est peut-être complet ; dans ce cas, scannez plutôt le QR code du groupe 9. - QR code WeChat groupe 8 de la communauté Orca + QR code WeChat groupe 8 de la communauté Orca  QR code WeChat groupe 9 de la communauté Orca - **Feedback & idées :** On ship vite. Il manque quelque chose ? [Demandez une feature](https://github.com/stablyai/orca/issues). - **Confidentialité :** Voir la [doc confidentialité & télémétrie](https://www.onorca.dev/docs/telemetry) pour ce qu'Orca collecte en anonyme et comment désactiver la télémétrie. diff --git a/docs/readme/README.ko.md b/docs/readme/README.ko.md index 4a75722ff8c..837ecf2133f 100644 --- a/docs/readme/README.ko.md +++ b/docs/readme/README.ko.md @@ -238,9 +238,9 @@ yay -S stably-orca-bin - **Discord:** **[Discord](https://discord.gg/fzjDKHxv8Q)** 커뮤니티에 참여하세요. - **Twitter / X:** 업데이트와 공지는 **[@orca_build](https://x.com/orca_build)** 를 팔로우하세요. -- **WeChat:** QR 코드를 스캔해 Orca 커뮤니티 WeChat 그룹 8에 참여하세요. +- **WeChat:** QR 코드를 스캔해 Orca 커뮤니티 WeChat 그룹 8에 참여하세요. 그룹 8이 가득 찼을 수 있으니, 그런 경우 그룹 9 QR 코드를 스캔하세요. - Orca 커뮤니티 WeChat 그룹 8 QR 코드 + Orca 커뮤니티 WeChat 그룹 8 QR 코드  Orca 커뮤니티 WeChat 그룹 9 QR 코드 - **피드백과 아이디어:** 우리는 빠르게 출시합니다. 필요한 기능이 있나요? [새 기능을 요청](https://github.com/stablyai/orca/issues)하세요. - **개인정보 보호:** Orca가 수집하는 익명 사용 데이터와 수집 거부 방법은 [개인정보 및 텔레메트리 문서](https://www.onorca.dev/docs/telemetry)를 참고하세요. diff --git a/docs/readme/README.zh-CN.md b/docs/readme/README.zh-CN.md index d7bae3fba9e..10f47e20fe6 100644 --- a/docs/readme/README.zh-CN.md +++ b/docs/readme/README.zh-CN.md @@ -235,9 +235,9 @@ yay -S stably-orca-bin - **Discord:** 加入 **[Discord](https://discord.gg/fzjDKHxv8Q)** 社区。 - **Twitter / X:** 关注 **[@orca_build](https://x.com/orca_build)** 获取更新和公告。 -- **微信:** 扫码加入 Orca 社区微信第 8 群。 +- **微信:** 扫码加入 Orca 社区微信第 8 群。第 8 群可能已满,如遇这种情况请扫描第 9 群二维码。 - Orca 社区微信第 8 群二维码 + Orca 社区微信第 8 群二维码  Orca 社区微信第 9 群二维码 - **反馈与想法:** 我们发布很快。缺少什么功能?[提交功能请求](https://github.com/stablyai/orca/issues)。 - **隐私:** 查看[隐私与遥测文档](https://www.onorca.dev/docs/telemetry),了解 Orca 收集哪些匿名使用数据以及如何退出。 diff --git a/mobile/src/home/MobileHomeHostList.tsx b/mobile/src/home/MobileHomeHostList.tsx index 3907df16f03..41d1f07156d 100644 --- a/mobile/src/home/MobileHomeHostList.tsx +++ b/mobile/src/home/MobileHomeHostList.tsx @@ -19,6 +19,7 @@ type MobileHomeHostListProps = { hostAttempts: Record hostLastConnected: Record hostPairingRejected: Record + hostSignedOut: Record hostPaths: Record hostPendingPaths: Record hosts: HostCatalogEntry[] @@ -40,6 +41,7 @@ export function MobileHomeHostList(props: MobileHomeHostListProps) { hostAttempts={props.hostAttempts} hostLastConnected={props.hostLastConnected} hostPairingRejected={props.hostPairingRejected} + hostSignedOut={props.hostSignedOut} hostPaths={props.hostPaths} hostPendingPaths={props.hostPendingPaths} hostStates={props.hostStates} @@ -54,6 +56,7 @@ export function MobileHomeHostList(props: MobileHomeHostListProps) { props.hostAttempts, props.hostLastConnected, props.hostPairingRejected, + props.hostSignedOut, props.hostPaths, props.hostPendingPaths, props.hostStates, @@ -91,6 +94,7 @@ type MobileHomeHostRowProps = Pick< | 'hostAttempts' | 'hostLastConnected' | 'hostPairingRejected' + | 'hostSignedOut' | 'hostPaths' | 'hostPendingPaths' | 'hostStates' @@ -113,7 +117,8 @@ const MobileHomeHostRow = memo(function MobileHomeHostRow(props: MobileHomeHostR lastConnectedAt: props.hostLastConnected[item.id] ?? null, endpoint: item.endpoint, pendingPath: props.hostPendingPaths[item.id] ?? null, - pairingRejected: props.hostPairingRejected[item.id] ?? false + pairingRejected: props.hostPairingRejected[item.id] ?? false, + hostSignedOut: props.hostSignedOut[item.id] ?? false }) const open = useCallback(() => onOpen(item), [item, onOpen]) const longPress = useCallback(() => onLongPress(item), [item, onLongPress]) diff --git a/mobile/src/home/MobileHomeScreen.tsx b/mobile/src/home/MobileHomeScreen.tsx index 99164243dcf..317a41171de 100644 --- a/mobile/src/home/MobileHomeScreen.tsx +++ b/mobile/src/home/MobileHomeScreen.tsx @@ -151,6 +151,7 @@ export function MobileHomeScreen() { hostAttempts={data.hostAttempts} hostLastConnected={data.hostLastConnected} hostPairingRejected={data.hostPairingRejected} + hostSignedOut={data.hostSignedOut} hostPaths={data.hostPaths} hostPendingPaths={data.hostPendingPaths} hosts={data.sortedHostCatalog} diff --git a/mobile/src/home/home-host-connection-projection.ts b/mobile/src/home/home-host-connection-projection.ts index f8fdfd4bcdf..9a49f6186c4 100644 --- a/mobile/src/home/home-host-connection-projection.ts +++ b/mobile/src/home/home-host-connection-projection.ts @@ -5,12 +5,14 @@ export type HomeHostConnectionProjectionEntry = { path: MobileConnectionPath pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean } export type HomeHostConnectionProjection = { hostPaths: Record hostPendingPaths: Record hostPairingRejected: Record + hostSignedOut: Record } /** Build all host lookup maps while reading each connection entry once. */ @@ -22,16 +24,19 @@ export function projectHomeHostConnections( const hostPaths = Object.create(null) as Record const hostPendingPaths = Object.create(null) as Record const hostPairingRejected = Object.create(null) as Record + const hostSignedOut = Object.create(null) as Record - for (const { hostId, path, pendingPath, pairingRejected } of entries) { + for (const { hostId, path, pendingPath, pairingRejected, hostSignedOut: signedOut } of entries) { hostPaths[hostId] = path hostPendingPaths[hostId] = pendingPath hostPairingRejected[hostId] = pairingRejected + hostSignedOut[hostId] = signedOut } Object.setPrototypeOf(hostPaths, Object.prototype) Object.setPrototypeOf(hostPendingPaths, Object.prototype) Object.setPrototypeOf(hostPairingRejected, Object.prototype) + Object.setPrototypeOf(hostSignedOut, Object.prototype) - return { hostPaths, hostPendingPaths, hostPairingRejected } + return { hostPaths, hostPendingPaths, hostPairingRejected, hostSignedOut } } diff --git a/mobile/src/home/use-mobile-home-data.ts b/mobile/src/home/use-mobile-home-data.ts index c77d024158e..6b28354a86d 100644 --- a/mobile/src/home/use-mobile-home-data.ts +++ b/mobile/src/home/use-mobile-home-data.ts @@ -179,6 +179,7 @@ export function useMobileHomeData() { connectedHosts, hostCatalog, hostPairingRejected: hostConnectionProjection.hostPairingRejected, + hostSignedOut: hostConnectionProjection.hostSignedOut, hostPaths: hostConnectionProjection.hostPaths, hostPendingPaths: hostConnectionProjection.hostPendingPaths, primaryHost, diff --git a/mobile/src/mobile-web/disabled-page-client-context.tsx b/mobile/src/mobile-web/disabled-page-client-context.tsx index 5cce6a701bc..c99c03b079a 100644 --- a/mobile/src/mobile-web/disabled-page-client-context.tsx +++ b/mobile/src/mobile-web/disabled-page-client-context.tsx @@ -24,6 +24,7 @@ const value: RpcClientContextValue = { getActivePath: () => 'lan', getPendingPath: () => null, isPairingRejected: () => false, + isHostSignedOut: () => false, subscribeHostState: () => noop, getAllClients: () => [], subscribeAllHosts: () => noop, diff --git a/mobile/src/mobile-web/mobile-web-native-chat-image-operations.ts b/mobile/src/mobile-web/mobile-web-native-chat-image-operations.ts index 896b15cf7d0..7438a42ce70 100644 --- a/mobile/src/mobile-web/mobile-web-native-chat-image-operations.ts +++ b/mobile/src/mobile-web/mobile-web-native-chat-image-operations.ts @@ -86,6 +86,7 @@ export async function executeMobileWebNativeChatImageOperation(args: { terminal: binding.hostTerminalId!, deviceToken: args.terminalClientId, imagePaths, + followedByText: payload.followedByText === true, deadline: payload.deadline, assertCurrent: () => assertCurrentMobileWebNativeChatPageBinding( diff --git a/mobile/src/mobile-web/mobile-web-native-chat-terminal-operations.ts b/mobile/src/mobile-web/mobile-web-native-chat-terminal-operations.ts index 795f95cbd71..9112947d815 100644 --- a/mobile/src/mobile-web/mobile-web-native-chat-terminal-operations.ts +++ b/mobile/src/mobile-web/mobile-web-native-chat-terminal-operations.ts @@ -82,6 +82,7 @@ export async function executeMobileWebNativeChatTerminalOperation(args: { terminal: binding.hostTerminalId!, deviceToken: args.terminalClientId, imagePaths: [], + followedByText: false, deadline: payload.deadline, assertCurrent: () => assertCurrentBinding(args, payload, binding) }) diff --git a/mobile/src/session/host-session-native-chat-operations.ts b/mobile/src/session/host-session-native-chat-operations.ts index 5090b442e9e..60788eede2e 100644 --- a/mobile/src/session/host-session-native-chat-operations.ts +++ b/mobile/src/session/host-session-native-chat-operations.ts @@ -69,7 +69,8 @@ export type HostSessionNativeChatOperations = { pasteImages?( target: HostSessionNativeChatTarget, references: readonly string[], - deadline?: number + deadline?: number, + followedByText?: boolean ): Promise releaseImages?(target: HostSessionNativeChatTarget, references: readonly string[]): Promise searchFiles(target: HostSessionNativeChatTarget, query: string): Promise diff --git a/mobile/src/session/mobile-image-attachment.test.ts b/mobile/src/session/mobile-image-attachment.test.ts index 9d725d5fe60..eead9691303 100644 --- a/mobile/src/session/mobile-image-attachment.test.ts +++ b/mobile/src/session/mobile-image-attachment.test.ts @@ -50,7 +50,9 @@ describe('attachMobileImageToTerminal', () => { const sendCall = client.calls.find((c) => c.method === 'terminal.send') expect(sendCall?.params).toEqual({ terminal: 'term-1', - text: '\x1b[200~/tmp/orca-attach.png\x1b[201~', + // Trailing space: the user types on this same line next, so a bare + // `…\x1b[201~` would arrive as `…pngadd` (STA-4847). + text: '\x1b[200~/tmp/orca-attach.png\x1b[201~ ', enter: false, client: { id: 'device-9', type: 'mobile' } }) diff --git a/mobile/src/session/mobile-image-attachment.ts b/mobile/src/session/mobile-image-attachment.ts index 567be99a9a1..9cb7d60aa8e 100644 --- a/mobile/src/session/mobile-image-attachment.ts +++ b/mobile/src/session/mobile-image-attachment.ts @@ -1,4 +1,5 @@ import type { RpcClient } from '../transport/rpc-client' +import { separateImagePasteFromFollowingText } from '../../../src/shared/image-paste-following-text' import { buildMobileImagePastePayload, saveMobileClipboardImageAsTempFile @@ -47,7 +48,10 @@ export async function attachMobileImageToTerminal( }) // Why: a generated image path is terminal image injection, so it's always // bracketed (matching desktop paste) regardless of terminal mode. - const payload = buildMobileImagePastePayload(imagePath) + // Always separated: attach-then-type is the whole interaction here, so the user's + // next keystroke would otherwise glue onto the path (`…pngadd`). Unlike native + // chat there is no batch to look ahead in, and a trailing space is inert. + const payload = separateImagePasteFromFollowingText(buildMobileImagePastePayload(imagePath), true) if (beforeTerminalSend && !(await beforeTerminalSend(terminal))) { return false } diff --git a/mobile/src/session/mobile-native-chat-image-send.test.ts b/mobile/src/session/mobile-native-chat-image-send.test.ts index cf1c59adf0f..a41b3fca3f1 100644 --- a/mobile/src/session/mobile-native-chat-image-send.test.ts +++ b/mobile/src/session/mobile-native-chat-image-send.test.ts @@ -38,7 +38,8 @@ describe('pasteMobileNativeChatImagePaths', () => { client, terminal: 'term-1', deviceToken: 'device-9', - imagePaths: ['/tmp/a.png', '/tmp/b.png', '/tmp/c.png'] + imagePaths: ['/tmp/a.png', '/tmp/b.png', '/tmp/c.png'], + followedByText: true }) expect(ok).toBe(true) @@ -55,7 +56,7 @@ describe('pasteMobileNativeChatImagePaths', () => { }) expect(client.calls[1]?.params.text).toBe('\x1b[200~/tmp/a.png\x1b[201~') expect(client.calls[2]?.params.text).toBe('\x1b[200~/tmp/b.png\x1b[201~') - expect(client.calls[3]?.params.text).toBe('\x1b[200~/tmp/c.png\x1b[201~') + expect(client.calls[3]?.params.text).toBe('\x1b[200~/tmp/c.png\x1b[201~ ') }) it('stops and reports failure as soon as a paste is rejected', async () => { @@ -66,7 +67,8 @@ describe('pasteMobileNativeChatImagePaths', () => { client, terminal: 'term-1', deviceToken: null, - imagePaths: ['/tmp/a.png', '/tmp/b.png'] + imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: true }) expect(ok).toBe(false) @@ -94,7 +96,8 @@ describe('pasteMobileNativeChatImagePaths', () => { client, terminal: 'term-1', deviceToken: null, - imagePaths: ['/tmp/a.png', '/tmp/b.png'] + imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: true }) expect(ok).toBe(false) @@ -121,6 +124,7 @@ describe('clearing a parked multi-line launch draft before the image paste', () terminal: 'term-1', deviceToken: null, imagePaths: ['/tmp/a.png'], + followedByText: true, clearInput }) @@ -137,6 +141,7 @@ describe('clearing a parked multi-line launch draft before the image paste', () terminal: 'term-1', deviceToken: null, imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: true, clearInput }) @@ -151,9 +156,27 @@ describe('clearing a parked multi-line launch draft before the image paste', () client, terminal: 'term-1', deviceToken: null, - imagePaths: ['/tmp/a.png'] + imagePaths: ['/tmp/a.png'], + followedByText: true }) expect(client.calls[0]?.params.text).toBe('\x15') }) + + it('keeps image writes byte-clean when no text or submit follows', async () => { + const client = clientWithResponses([sendResult(true), sendResult(true), sendResult(true)]) + + await pasteMobileNativeChatImagePaths({ + client, + terminal: 'term-1', + deviceToken: null, + imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: false + }) + + expect(client.calls.slice(1).map((call) => call.params.text)).toEqual([ + '\x1b[200~/tmp/a.png\x1b[201~', + '\x1b[200~/tmp/b.png\x1b[201~' + ]) + }) }) diff --git a/mobile/src/session/mobile-native-chat-image-send.ts b/mobile/src/session/mobile-native-chat-image-send.ts index 63c59dfa3d5..9b81200dd96 100644 --- a/mobile/src/session/mobile-native-chat-image-send.ts +++ b/mobile/src/session/mobile-native-chat-image-send.ts @@ -1,4 +1,5 @@ import type { RpcClient } from '../transport/rpc-client' +import { imagePasteWritesFollowedByText } from '../../../src/shared/image-paste-following-text' import { buildMobileImagePastePayload } from './mobile-clipboard-image' import { MOBILE_NATIVE_CHAT_MIN_WRITE_TIMEOUT_MS, @@ -23,6 +24,7 @@ type PasteImagesArgs = { readonly terminal: string readonly deviceToken: string | null readonly imagePaths: readonly string[] + readonly followedByText: boolean /** Budget shared with the rest of the user action (the text body that follows, or * the send this is healing for). Omit to open a fresh one for this paste alone. */ readonly deadline?: number @@ -41,6 +43,7 @@ export async function pasteMobileNativeChatImagePaths({ terminal, deviceToken, imagePaths, + followedByText, deadline: sharedDeadline, clearInput, assertCurrent = () => {} @@ -55,7 +58,7 @@ export async function pasteMobileNativeChatImagePaths({ const deadline = sharedDeadline ?? openMobileNativeChatSendBudget() for (const text of [ clearInput ?? MOBILE_NATIVE_CHAT_CLEAR_UNSUBMITTED_INPUT, - ...imagePaths.map(buildMobileImagePastePayload) + ...imagePasteWritesFollowedByText(imagePaths.map(buildMobileImagePastePayload), followedByText) ]) { const remainingMs = deadline - Date.now() // Why: the budget is the whole sequence's — starting a write it can't fund would diff --git a/mobile/src/session/mobile-native-chat-image-submit.ts b/mobile/src/session/mobile-native-chat-image-submit.ts index a561d2d2cc8..377813f654d 100644 --- a/mobile/src/session/mobile-native-chat-image-submit.ts +++ b/mobile/src/session/mobile-native-chat-image-submit.ts @@ -65,6 +65,7 @@ export async function sendMobileNativeChatWithImages(args: { } try { const references = args.pendingImages.map((attachment) => attachment.path) + const followedByText = args.text.trim().length > 0 const seededLaunchDraft = args.readSeededLaunchDraft() const pasted = args.client ? await pasteMobileNativeChatImagePaths({ @@ -72,12 +73,13 @@ export async function sendMobileNativeChatWithImages(args: { terminal: handle, deviceToken: args.deviceTokenRef.current, imagePaths: references, + followedByText, deadline, ...(seededLaunchDraft ? { clearInput: buildAgentTuiClearInputForText(seededLaunchDraft) } : {}) }) - : await args.operations!.pasteImages!(target!, references, deadline) + : await args.operations!.pasteImages!(target!, references, deadline, followedByText) if (!pasted) { markMobileNativeChatInputStale(handle) args.onError?.() diff --git a/mobile/src/session/mobile-native-chat-stale-input.ts b/mobile/src/session/mobile-native-chat-stale-input.ts index 18d84d2e503..cdc21067683 100644 --- a/mobile/src/session/mobile-native-chat-stale-input.ts +++ b/mobile/src/session/mobile-native-chat-stale-input.ts @@ -55,6 +55,7 @@ export async function healMobileNativeChatStaleInput(args: { terminal: args.terminal, deviceToken: args.deviceToken, imagePaths: [], + followedByText: false, ...(args.deadline === undefined ? {} : { deadline: args.deadline }) }) } catch { diff --git a/mobile/src/session/use-mobile-native-chat-hosted-image-attachments.test.ts b/mobile/src/session/use-mobile-native-chat-hosted-image-attachments.test.ts index bcc0ef66133..27e8f3ea219 100644 --- a/mobile/src/session/use-mobile-native-chat-hosted-image-attachments.test.ts +++ b/mobile/src/session/use-mobile-native-chat-hosted-image-attachments.test.ts @@ -82,7 +82,7 @@ describe('hosted native-chat image attachments', () => { await act(async () => { expect(await hook!.sendNativeChat('inspect')).toBe(true) }) - expect(pasteImages).toHaveBeenCalledWith(TARGET, [REFERENCE], expect.any(Number)) + expect(pasteImages).toHaveBeenCalledWith(TARGET, [REFERENCE], expect.any(Number), true) expect(baseSend).toHaveBeenCalledWith('inspect', [PREVIEW], expect.any(Number)) expect(releaseImages).toHaveBeenCalledWith(TARGET, [REFERENCE]) expect(hook!.attachments).toEqual([]) diff --git a/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts b/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts index 3219588b305..79f842ec809 100644 --- a/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts +++ b/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts @@ -184,9 +184,12 @@ describe('useMobileNativeChatImageAttachments', () => { expect(sendCalls).toHaveLength(2) expect(sendCalls[0]?.params).toMatchObject({ text: '\x15', enter: false }) expect(sendCalls[1]?.params).toMatchObject({ - text: '\x1b[200~/tmp/a.png\x1b[201~', + text: '\x1b[200~/tmp/a.png\x1b[201~ ', enter: false }) + const combined = String(sendCalls[1]?.params.text ?? '') + 'look at this' + expect(combined).toContain('.png\x1b[201~ look') + expect(combined).not.toContain('.png\x1b[201~look') // Clear, then paste, then settle, then the text send — in that order. expect(order).toEqual(['clear', 'paste', 'settle', 'text:look at this']) // The local preview URI rides along so the sent bubble shows the photo. @@ -269,7 +272,10 @@ describe('useMobileNativeChatImageAttachments', () => { } }) - it('routes an attachments-only send through baseSend with empty text so the echo still shows the photo', async () => { + it.each([ + ['empty', ''], + ['whitespace-only', ' '] + ])('routes an attachments-only send through baseSend with %s text', async (_label, text) => { pick.mockResolvedValue([{ base64: 'AAAA', uri: 'file:///a.jpg' }]) const client = makeClient([ methodNotFound('start'), @@ -285,16 +291,17 @@ describe('useMobileNativeChatImageAttachments', () => { }) let accepted = false await act(async () => { - accepted = await hook!.sendNativeChat('') + accepted = await hook!.sendNativeChat(text) }) expect(accepted).toBe(true) - // Empty text still goes through baseSend (which submits the bare Enter) so the + // Attachment-only text still goes through baseSend (which submits Enter) so the // optimistic echo carries the preview URI. - expect(baseSend).toHaveBeenCalledWith('', ['file:///a.jpg'], expect.any(Number)) + expect(baseSend).toHaveBeenCalledWith(text, ['file:///a.jpg'], expect.any(Number)) const sendCalls = client.calls.filter((c) => c.method === 'terminal.send') // Only the clear + image paste hit the wire here; baseSend owns the submit. expect(sendCalls).toHaveLength(2) + expect(sendCalls[1]?.params.text).toBe('\x1b[200~/tmp/a.png\x1b[201~') expect(hook!.attachments).toEqual([]) }) diff --git a/mobile/src/session/web-host-session-native-chat-deadlines.test.ts b/mobile/src/session/web-host-session-native-chat-deadlines.test.ts index 6a5cece3a38..22c90298732 100644 --- a/mobile/src/session/web-host-session-native-chat-deadlines.test.ts +++ b/mobile/src/session/web-host-session-native-chat-deadlines.test.ts @@ -80,6 +80,44 @@ describe('hosted native-chat deadlines', () => { expect(pasteImages).not.toHaveBeenCalled() }) + it('forwards the following-text hint only when typed text follows the paste', async () => { + vi.useFakeTimers() + vi.setSystemTime(10_000) + const pasteImages = vi.fn().mockResolvedValue({ pasted: true }) + const client = { nativeChat: { pasteImages } } as unknown as MobileWebBridgeClient + const operations = webHostSessionNativeChatOperations(client) + + await expect(operations.pasteImages!(TARGET, ['opaque-image'], 20_000, true)).resolves.toBe( + true + ) + await expect(operations.pasteImages!(TARGET, ['opaque-image'], 20_000, false)).resolves.toBe( + true + ) + + // Why: an older strict shell rejects unknown keys, so the flag rides only when it is true. + expect(pasteImages).toHaveBeenNthCalledWith( + 1, + { + workspaceId: 'workspace', + sessionId: 'native_chat_session', + references: ['opaque-image'], + deadline: 20_000, + followedByText: true + }, + { timeoutMs: 10_000 } + ) + expect(pasteImages).toHaveBeenNthCalledWith( + 2, + { + workspaceId: 'workspace', + sessionId: 'native_chat_session', + references: ['opaque-image'], + deadline: 20_000 + }, + { timeoutMs: 10_000 } + ) + }) + it('clears a stale hosted composer only after the shell accepts preparation', async () => { vi.useFakeTimers() vi.setSystemTime(10_000) diff --git a/mobile/src/session/web-host-session-native-chat-operations.ts b/mobile/src/session/web-host-session-native-chat-operations.ts index 00c82cf3963..843813badcf 100644 --- a/mobile/src/session/web-host-session-native-chat-operations.ts +++ b/mobile/src/session/web-host-session-native-chat-operations.ts @@ -128,7 +128,7 @@ export function webHostSessionNativeChatOperations( } return { status: result.status } }, - async pasteImages(target, references, deadline) { + async pasteImages(target, references, deadline, followedByText) { const budget = bridgeBudget(deadline) if (!budget) { return false @@ -138,7 +138,8 @@ export function webHostSessionNativeChatOperations( await client.nativeChat.pasteImages( bridgeTarget(target, { references: [...references], - deadline: budget.deadline + deadline: budget.deadline, + ...(followedByText ? { followedByText } : {}) }), { timeoutMs: budget.timeoutMs } ) diff --git a/mobile/src/transport/client-context-connection-metrics.ts b/mobile/src/transport/client-context-connection-metrics.ts index 1e2ff6f7919..1fe72687bd2 100644 --- a/mobile/src/transport/client-context-connection-metrics.ts +++ b/mobile/src/transport/client-context-connection-metrics.ts @@ -30,14 +30,16 @@ export function useConnectionPathStatus(hostId: string | undefined): { export function useRelayRecoveryStatus(hostId: string | undefined): { pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean } { return useHostMetric( hostId, (context, id) => ({ pendingPath: context.getPendingPath(id), - pairingRejected: context.isPairingRejected(id) + pairingRejected: context.isPairingRejected(id), + hostSignedOut: context.isHostSignedOut(id) }), - { pendingPath: null, pairingRejected: false } + { pendingPath: null, pairingRejected: false, hostSignedOut: false } ) } diff --git a/mobile/src/transport/client-context.test.ts b/mobile/src/transport/client-context.test.ts index 4a7d5d7b0e8..56b227bdff7 100644 --- a/mobile/src/transport/client-context.test.ts +++ b/mobile/src/transport/client-context.test.ts @@ -533,12 +533,12 @@ describe('useAllHostClients', () => { await Promise.resolve() }) act(() => client.emitPendingPath('relay')) - expect(status).toEqual({ pendingPath: 'relay', pairingRejected: false }) + expect(status).toEqual({ pendingPath: 'relay', pairingRejected: false, hostSignedOut: false }) // Why: the desktop refusing the credential is a status-only change — no // transport state moves, so only the connection-path signal can carry it. act(() => client.emitPairingRejected(true)) - expect(status).toEqual({ pendingPath: 'relay', pairingRejected: true }) + expect(status).toEqual({ pendingPath: 'relay', pairingRejected: true, hostSignedOut: false }) act(() => renderer.unmount()) }) diff --git a/mobile/src/transport/connection-health.ts b/mobile/src/transport/connection-health.ts index 858b13a8b24..1a9e282047f 100644 --- a/mobile/src/transport/connection-health.ts +++ b/mobile/src/transport/connection-health.ts @@ -29,6 +29,10 @@ const STALE_SINCE_LAST_CONNECT_MS = 60_000 // instead of leaving the user staring at a generic "Can't connect". const TAILSCALE_HINT = 'check Tailscale' +// No hint field: the remedy is the label, and appending "— check Tailscale" to +// it would be wrong advice for a desktop that is reachable but signed out. +const SIGNED_OUT_LABEL = 'Desktop signed out — sign in to Orca on your desktop to reconnect' + export type ConnectionVerdict = | { kind: 'normal'; label: string } | { kind: 'warning'; label: string; hint?: string } // "Can't connect" @@ -54,6 +58,10 @@ export function classifyConnection(args: { // The desktop has repeatedly refused this device's relay credential — retrying // cannot fix it, so it outranks any "still connecting" reading (STA-4681). pairingRejected?: boolean + // The relay says the desktop's last control close named its own Orca Cloud + // sign-out. Retrying is still correct and still happens on the same cadence, + // but only the desktop's owner can end it, so the label has to say so. + hostSignedOut?: boolean nowMs?: number }): ConnectionVerdict { const { state, reconnectAttempts, lastConnectedAt } = args @@ -70,6 +78,17 @@ export function classifyConnection(args: { return { kind: 'normal', label: 'Connected' } } + // Ahead of the attempt thresholds: this is evidence, not an inference from a + // failure streak, and waiting twelve dials to show it wastes the whole point. + // Below auth-failed because a revoked pairing cannot be fixed by signing in. + if (args.hostSignedOut) { + return { + kind: 'unreachable', + label: SIGNED_OUT_LABEL, + reason: lastConnectedAt == null ? 'never-connected' : 'stale' + } + } + // A disconnected pending path can survive a cleared retry timer during a // lifecycle race. Only narrate Relay while dialing or after a retry has // recorded progress; otherwise the idle transport must read Disconnected. diff --git a/mobile/src/transport/host-client-context-state.ts b/mobile/src/transport/host-client-context-state.ts index 859e5d847f9..db4c7908b41 100644 --- a/mobile/src/transport/host-client-context-state.ts +++ b/mobile/src/transport/host-client-context-state.ts @@ -89,10 +89,16 @@ export function createHostClientSelectors( getPendingPath: (hostId: string): MobileConnectionPath | null => clientPendingPath(entries.get(hostId)?.client), isPairingRejected: (hostId: string): boolean => - clientPairingRejected(entries.get(hostId)?.client) + clientPairingRejected(entries.get(hostId)?.client), + isHostSignedOut: (hostId: string): boolean => clientHostSignedOut(entries.get(hostId)?.client) } } +export function clientHostSignedOut(client: RpcClient | undefined): boolean { + const logical = client as Partial | undefined + return logical?.isHostSignedOut?.() ?? false +} + export function clientPairingRejected(client: RpcClient | undefined): boolean { const logical = client as Partial | undefined return logical?.isPairingRejected?.() ?? false diff --git a/mobile/src/transport/logical-client-connection-path.ts b/mobile/src/transport/logical-client-connection-path.ts index 55c6d1d28f3..b7f02b40b88 100644 --- a/mobile/src/transport/logical-client-connection-path.ts +++ b/mobile/src/transport/logical-client-connection-path.ts @@ -5,6 +5,7 @@ export class LogicalClientConnectionPath { private recovery: MobileConnectionPath | null = null private recoveryAttempt = 0 private pairingRejected = false + private hostSignedOut = false private readonly listeners = new Set<() => void>() constructor(private readonly isConnected: () => boolean) {} @@ -35,12 +36,23 @@ export class LogicalClientConnectionPath { }) } + isHostSignedOut(): boolean { + return this.hostSignedOut + } + + setHostSignedOut(signedOut: boolean): void { + this.update(() => { + this.hostSignedOut = signedOut + }) + } + clearAfterConnected(): void { this.migration = null this.recovery = null this.recoveryAttempt = 0 // Why: an authenticated session is the desktop accepting this device. this.pairingRejected = false + this.hostSignedOut = false } setRecovery(path: MobileConnectionPath | null, attempt?: number): void { @@ -69,11 +81,13 @@ export class LogicalClientConnectionPath { const previousPath = this.pending() const previousAttempt = this.reconnectAttempt(0) const previousRejected = this.pairingRejected + const previousSignedOut = this.hostSignedOut apply() if ( previousPath === this.pending() && previousAttempt === this.reconnectAttempt(0) && - previousRejected === this.pairingRejected + previousRejected === this.pairingRejected && + previousSignedOut === this.hostSignedOut ) { return } diff --git a/mobile/src/transport/mobile-endpoint-lifecycle.ts b/mobile/src/transport/mobile-endpoint-lifecycle.ts index 8ee8df6948e..7ec5f28b945 100644 --- a/mobile/src/transport/mobile-endpoint-lifecycle.ts +++ b/mobile/src/transport/mobile-endpoint-lifecycle.ts @@ -86,7 +86,7 @@ function createSupervisor( ): MobileEndpointSupervisor { return new MobileEndpointSupervisor(logical, host, { openDirect: (endpoint) => connect(endpoint, host.deviceToken, host.publicKeyB64, { onLog }), - openRelay: (relay, credential, confirmReqId) => + openRelay: (relay, credential, confirmReqId, onHostCloseReason) => connectMobileRelayRpcSession({ relay, resumeToken: credential.token, @@ -94,6 +94,7 @@ function createSupervisor( resumeConfirmReqId: confirmReqId, deviceToken: host.deviceToken, desktopPublicKeyB64: host.publicKeyB64, + onHostCloseReason, onLog }), resolveRelay: resolveMobileRelayEndpoint, diff --git a/mobile/src/transport/mobile-endpoint-supervisor-contract.ts b/mobile/src/transport/mobile-endpoint-supervisor-contract.ts index 0098de6e079..2a784fd8895 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-contract.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-contract.ts @@ -1,4 +1,5 @@ import type { MobileRelayEndpoint } from '../../../src/shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../src/shared/relay-host-close-reason' import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bundle' import type { MobileRelayRpcSession } from './mobile-relay-rpc-session' import type { resolveMobileRelayEndpoint } from './mobile-relay-resume-director' @@ -10,7 +11,8 @@ export type MobileEndpointSupervisorDependencies = { openRelay: ( relay: MobileRelayEndpoint, credential: { token: string; version: number }, - confirmReqId: string + confirmReqId: string, + onHostCloseReason?: (reason: RelayHostCloseReason) => void ) => MobileRelayRpcSession resolveRelay: typeof resolveMobileRelayEndpoint readBundle: (hostId: string) => Promise diff --git a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts index 9a66a58aaef..fb7eaa777db 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts @@ -135,10 +135,22 @@ export class FakeLogicalClient extends FakeSession implements StableLogicalRpcCl } }) isPairingRejected = () => this.pairingRejected + private hostSignedOut = false + setHostSignedOut = vi.fn((signedOut: boolean) => { + if (this.hostSignedOut === signedOut) { + return + } + this.hostSignedOut = signedOut + for (const listener of this.pathListeners) { + listener() + } + }) + isHostSignedOut = () => this.hostSignedOut // Mirrors LogicalClientConnectionPath.clearAfterConnected. publishState(state: ConnectionState): void { if (state === 'connected') { this.pairingRejected = false + this.hostSignedOut = false } super.publishState(state) } diff --git a/mobile/src/transport/mobile-endpoint-supervisor.test.ts b/mobile/src/transport/mobile-endpoint-supervisor.test.ts index aeb9cddef63..10ef892a479 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor.test.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor.test.ts @@ -185,7 +185,12 @@ describe('mobile endpoint supervisor', () => { await supervisor.start() expect(deps.resolveRelay).toHaveBeenCalledOnce() - expect(openRelay).toHaveBeenLastCalledWith(resolved, expect.any(Object), expect.any(String)) + expect(openRelay).toHaveBeenLastCalledWith( + resolved, + expect.any(Object), + expect.any(String), + expect.any(Function) + ) expect(deps.saveHost).toHaveBeenCalledWith( expect.objectContaining({ relay: resolved, endpoint: host.endpoint }) ) @@ -556,7 +561,8 @@ describe('mobile endpoint supervisor', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) supervisor.stop() }) @@ -603,7 +609,8 @@ describe('mobile endpoint supervisor', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) supervisor.stop() }) diff --git a/mobile/src/transport/mobile-relay-e2ee-link.ts b/mobile/src/transport/mobile-relay-e2ee-link.ts index 7743deb23dd..9b1f7a9a355 100644 --- a/mobile/src/transport/mobile-relay-e2ee-link.ts +++ b/mobile/src/transport/mobile-relay-e2ee-link.ts @@ -2,6 +2,10 @@ import { RelayPhoneHelloSchema, type RelayPhoneHello } from '../../../src/shared/mobile-relay-phone-protocol' +import { + relayHostCloseReasonFrom, + type RelayHostCloseReason +} from '../../../src/shared/relay-host-close-reason' import { MobileE2EEV2ClientSession } from './mobile-e2ee-v2-client-session' import { MobileE2EEV2PhysicalChannel } from './mobile-e2ee-v2-physical-channel' import { websocketPayloadToUint8 } from './websocket-payload-bytes' @@ -26,6 +30,12 @@ type MobileRelayE2eeLinkOptions = { onText: (plaintext: string) => void onBinary: (plaintext: Uint8Array) => void onHello?: (hello: Extract) => void + // The cell's account of why the desktop is absent, read off the close frame. + // Reported separately from onError because a rejection is delivered as both a + // relay-hello and a close, and which one the runtime dispatches first is not + // ordered — only the close carries the reason, and it must not be lost to + // that race. + onHostCloseReason?: (reason: RelayHostCloseReason) => void // Fired once relay-auth is on the wire: from here the cell owns the wait. onOpen?: () => void onError: (error: Error) => void @@ -129,6 +139,11 @@ export class MobileRelayE2eeLink { clearTimeout(this.transportErrorTimer) this.transportErrorTimer = null } + // Ahead of fail(), which no-ops once the hello already reported this close. + const hostCloseReason = relayHostCloseReasonFrom(event.reason) + if (hostCloseReason) { + this.options.onHostCloseReason?.(hostCloseReason) + } this.fail(new RelayOuterError(event.code || 1006)) } } diff --git a/mobile/src/transport/mobile-relay-rpc-session.ts b/mobile/src/transport/mobile-relay-rpc-session.ts index ee2f6b11d62..bf5bfc16735 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.ts @@ -12,6 +12,7 @@ import { isRpcResponse } from './rpc-response-shape' import { RelayDialStageTracker, type RelayDialStageSource } from './relay-dial-stage' import { RelayPendingRequests } from './relay-pending-requests' import { settleMobileRuntimeCapabilities } from './mobile-runtime-capability-negotiation' +import type { RelayHostCloseReason } from '../../../src/shared/relay-host-close-reason' import type { RpcClient } from './rpc-client' import type { ConnectionLogSink, ConnectionState, RpcResponse } from './types' import { encodeTerminalStreamFrame } from './terminal-stream-protocol' @@ -36,6 +37,7 @@ export function connectMobileRelayRpcSession(args: { desktopPublicKeyB64: string requestTimeoutMs?: number createSocket?: (url: string) => WebSocket + onHostCloseReason?: (reason: RelayHostCloseReason) => void onLog?: ConnectionLogSink }): MobileRelayRpcSession { const requestTimeoutMs = args.requestTimeoutMs ?? 30_000 @@ -63,6 +65,7 @@ export function connectMobileRelayRpcSession(args: { deviceToken: args.deviceToken, desktopPublicKeyB64: args.desktopPublicKeyB64, createSocket: args.createSocket, + onHostCloseReason: args.onHostCloseReason, onOpen: () => dialStage.advance('awaiting-hello'), onHello: (hello) => { if ( diff --git a/mobile/src/transport/mobile-relay-runtime-failover.test.ts b/mobile/src/transport/mobile-relay-runtime-failover.test.ts index 01f4d45feb0..ce7cca3fd9f 100644 --- a/mobile/src/transport/mobile-relay-runtime-failover.test.ts +++ b/mobile/src/transport/mobile-relay-runtime-failover.test.ts @@ -145,10 +145,22 @@ class FakeLogicalClient extends FakeSession implements StableLogicalRpcClient { } }) isPairingRejected = () => this.pairingRejected + private hostSignedOut = false + setHostSignedOut = vi.fn((signedOut: boolean) => { + if (this.hostSignedOut === signedOut) { + return + } + this.hostSignedOut = signedOut + for (const listener of this.pathListeners) { + listener() + } + }) + isHostSignedOut = () => this.hostSignedOut // Mirrors LogicalClientConnectionPath.clearAfterConnected. publishState(state: ConnectionState): void { if (state === 'connected') { this.pairingRejected = false + this.hostSignedOut = false } super.publishState(state) } @@ -264,7 +276,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() @@ -353,7 +366,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(deps.openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 2 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() @@ -382,7 +396,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 1 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() diff --git a/mobile/src/transport/mobile-relay-session-establisher.ts b/mobile/src/transport/mobile-relay-session-establisher.ts index 7a8ce372155..9a04ae44137 100644 --- a/mobile/src/transport/mobile-relay-session-establisher.ts +++ b/mobile/src/transport/mobile-relay-session-establisher.ts @@ -10,6 +10,7 @@ import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bund import type { RelayReconnectController } from './mobile-relay-reconnect-controller' import type { StableLogicalRpcClient } from './stable-logical-rpc-client' import type { MobileRelayEndpoint } from '../../../src/shared/mobile-relay-credential-contract' +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' import type { HostProfile } from './types' type EstablishResult = { ok: true } | { ok: false; error: Error } @@ -100,7 +101,16 @@ export class MobileRelaySessionEstablisher { const session = args.openRelay( relay, credential, - `confirm-${encodeBase64Url(args.randomBytes(16))}` + `confirm-${encodeBase64Url(args.randomBytes(16))}`, + // Latched on the logical client, not on the dial result: the close that + // carries the reason can land after this dial has already reported its + // failure. Clearing is clearAfterConnected's job, so any path that + // reaches connected retires it. + (reason) => { + if (reason === RELAY_HOST_CLOSE_REASON.SIGNED_OUT) { + args.logical.setHostSignedOut(true) + } + } ) try { // Why: backgrounding or a direct winner withdraws this dial before cutover. diff --git a/mobile/src/transport/relay-host-signed-out-supervisor.test.ts b/mobile/src/transport/relay-host-signed-out-supervisor.test.ts new file mode 100644 index 00000000000..cc7a9ac6c35 --- /dev/null +++ b/mobile/src/transport/relay-host-signed-out-supervisor.test.ts @@ -0,0 +1,60 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { MOBILE_RELAY_CLOSE_CODE } from '../../../src/shared/mobile-relay-close-codes' +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' +import { RelayOuterError } from './mobile-relay-e2ee-link' +import { + dependencies, + FakeLogicalClient, + FakeRelaySession, + host +} from './mobile-endpoint-supervisor-test-fakes' +import { MobileEndpointSupervisor } from './mobile-endpoint-supervisor' + +vi.mock('react-native', () => ({ Platform: { OS: 'ios' } })) +vi.mock('expo-secure-store', () => ({ WHEN_UNLOCKED_THIS_DEVICE_ONLY: 'when-unlocked' })) +vi.mock('expo-crypto', () => ({ getRandomBytes: (length: number) => new Uint8Array(length) })) + +// The reason travels from the cell's close frame to the screens. This covers +// the production wiring between them: the supervisor's own openRelay callback. +describe('a signed-out desktop reaches the phone verdict', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-07-13T12:00:00Z')) + }) + afterEach(() => vi.useRealTimers()) + + function supervisorOver(closeReason: string | null) { + const logical = new FakeLogicalClient('disconnected', 'lan') + const deps = dependencies({ + openDirect: vi.fn(() => new FakeRelaySession('disconnected')), + openRelay: vi.fn((_relay, _credential, _confirmReqId, onHostCloseReason) => { + if (closeReason) { + onHostCloseReason?.(closeReason as never) + } + return new FakeRelaySession( + 'disconnected', + new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE) + ) + }) + }) + return { logical, supervisor: new MobileEndpointSupervisor(logical, host, deps) } + } + + it('latches the sign-out the cell reported', async () => { + const { logical, supervisor } = supervisorOver(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + await supervisor.start() + await vi.waitFor(() => expect(logical.isHostSignedOut()).toBe(true)) + + supervisor.stop() + }) + + it('stays quiet for an ordinary host-offline rejection', async () => { + const { logical, supervisor } = supervisorOver(null) + + await supervisor.start() + + expect(logical.isHostSignedOut()).toBe(false) + supervisor.stop() + }) +}) diff --git a/mobile/src/transport/relay-host-signed-out-verdict.test.ts b/mobile/src/transport/relay-host-signed-out-verdict.test.ts new file mode 100644 index 00000000000..2607b922b58 --- /dev/null +++ b/mobile/src/transport/relay-host-signed-out-verdict.test.ts @@ -0,0 +1,216 @@ +import { describe, expect, it, vi } from 'vitest' + +vi.mock('./mobile-e2ee-v2-client-session', () => ({ + MobileE2EEV2ClientSession: { create: () => ({}) } +})) + +vi.mock('./mobile-e2ee-v2-physical-channel', () => ({ + MobileE2EEAuthenticationError: class extends Error {}, + MobileE2EEV2PhysicalChannel: class { + start = vi.fn() + handleMessage = vi.fn(async () => {}) + sendText = vi.fn(() => true) + sendBinary = vi.fn(() => true) + dispose = vi.fn() + } +})) + +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' +import { MOBILE_RELAY_CLOSE_CODE } from '../../../src/shared/mobile-relay-close-codes' +import { classifyConnection, verdictDisplayLabel } from './connection-health' +import { MobileRelayE2eeLink, RelayOuterError } from './mobile-relay-e2ee-link' +import { LogicalClientConnectionPath } from './logical-client-connection-path' +import { RelayReconnectController } from './mobile-relay-reconnect-controller' + +const SIGNED_OUT_LABEL = 'Desktop signed out — sign in to Orca on your desktop to reconnect' + +class FakeSocket { + static readonly OPEN = 1 + readonly OPEN = FakeSocket.OPEN + readyState = FakeSocket.OPEN + bufferedAmount = 0 + onopen: (() => void) | null = null + onmessage: ((event: { data: unknown }) => void) | null = null + onerror: (() => void) | null = null + onclose: ((event: { code: number; reason: string }) => void) | null = null + send = vi.fn() + close = vi.fn() +} + +function linkOver( + socket: FakeSocket, + onHostCloseReason: (reason: string) => void, + onError: (error: Error) => void +): MobileRelayE2eeLink { + return new MobileRelayE2eeLink({ + endpoint: { cellUrl: 'https://relay-c1.onorca.dev', relayHostId: 'AbCdEf0123_-xyZ9' }, + credential: 'credential', + expectedCredentialKind: 'resume', + deviceToken: 'device-token', + desktopPublicKeyB64: 'desktop-key', + onAuthenticated: vi.fn(), + onText: vi.fn(), + onBinary: vi.fn(), + onHostCloseReason, + onError, + createSocket: () => socket as unknown as WebSocket + }) +} + +describe('relay close reason on the phone', () => { + it('reports the cell close reason and still fails with 4404', () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + const onError = vi.fn() + linkOver(socket, onHostCloseReason, onError) + + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + + expect(onHostCloseReason).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + expect(onError).toHaveBeenCalledWith(new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE)) + }) + + // An old cell sends its constant, and every other close sends nothing. + it('reports nothing for a reason it does not know', () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + linkOver(socket, onHostCloseReason, vi.fn()) + + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: 'relay connection rejected' + }) + + expect(onHostCloseReason).not.toHaveBeenCalled() + }) + + // The rejection arrives as a relay-hello AND a close, in an unordered pair. + // Whichever lands first, the reason must survive. + it('still reports the reason when the hello already failed the link', async () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + linkOver(socket, onHostCloseReason, vi.fn()) + + socket.onmessage?.({ + data: JSON.stringify({ + type: 'relay-hello', + ok: false, + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE + }) + }) + await Promise.resolve() + await Promise.resolve() + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + + expect(onHostCloseReason).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) +}) + +describe('the signed-out signal on the logical client', () => { + it('publishes on change and retires when any path reaches connected', () => { + const path = new LogicalClientConnectionPath(() => false) + const changes = vi.fn() + path.subscribe(changes) + + path.setHostSignedOut(true) + path.setHostSignedOut(true) + expect(path.isHostSignedOut()).toBe(true) + expect(changes).toHaveBeenCalledTimes(1) + + path.clearAfterConnected() + expect(path.isHostSignedOut()).toBe(false) + }) +}) + +describe('RelayReconnectController cadence', () => { + // The reason changes no recovery decision; 4404 keeps the host-offline + // backoff it has always had, so a phone on this build retries exactly as + // often as one that never hears the reason. + it('keeps the host-offline retry delay for a 4404', () => { + const delays: number[] = [] + const controller = new RelayReconnectController( + { + now: () => 0, + randomBytes: () => new Uint8Array([0, 0]), + setTimer: ((callback: () => void, delay: number) => { + delays.push(delay) + return 1 as unknown as ReturnType + }) as unknown as typeof setTimeout, + clearTimer: (() => {}) as unknown as typeof clearTimeout + }, + vi.fn() + ) + + controller.registerFailure(new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE)) + + // hostOfflineDelayMs' 5s floor, not the 250ms transport-backoff floor. + expect(delays.at(-1)).toBe(5_000) + }) +}) + +describe('classifyConnection with a signed-out desktop', () => { + const base = { reconnectAttempts: 0, lastConnectedAt: null, hostSignedOut: true } + + it('says so from the first failed dial instead of "Connecting via Relay…"', () => { + const verdict = classifyConnection({ + ...base, + state: 'connecting', + pendingPath: 'relay' + }) + + expect(verdict).toEqual({ + kind: 'unreachable', + label: SIGNED_OUT_LABEL, + reason: 'never-connected' + }) + expect(verdictDisplayLabel(verdict)).toBe(SIGNED_OUT_LABEL) + }) + + it('replaces "Can\'t reach desktop" on the direct path too', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', reconnectAttempts: 20 }).label + ).toBe(SIGNED_OUT_LABEL) + }) + + it('reads as stale once this session had been connected', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', lastConnectedAt: 1, nowMs: 2 }).reason + ).toBe('stale') + }) + + // A Tailscale endpoint cannot make "sign in on your desktop" better advice. + it('never appends the Tailscale hint', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', endpoint: '100.64.0.1' }) + ).not.toHaveProperty('hint') + }) + + it('never outranks a connected session', () => { + expect(classifyConnection({ ...base, state: 'connected' }).label).toBe('Connected') + }) + + // Re-pairing, not signing in, is the remedy when the pairing itself is dead. + it('never outranks a revoked pairing', () => { + expect(classifyConnection({ ...base, state: 'reconnecting', pairingRejected: true }).kind).toBe( + 'auth-failed' + ) + }) + + it('leaves every other verdict alone when the desktop is not signed out', () => { + expect( + classifyConnection({ + state: 'connecting', + reconnectAttempts: 0, + lastConnectedAt: null, + pendingPath: 'relay', + hostSignedOut: false + }).label + ).toBe('Connecting via Relay…') + }) +}) diff --git a/mobile/src/transport/rpc-client-context-contract.ts b/mobile/src/transport/rpc-client-context-contract.ts index 54e25973c7f..65262a6fc15 100644 --- a/mobile/src/transport/rpc-client-context-contract.ts +++ b/mobile/src/transport/rpc-client-context-contract.ts @@ -24,6 +24,7 @@ export type RpcClientContextValue = { getActivePath: (hostId: string) => MobileConnectionPath getPendingPath: (hostId: string) => MobileConnectionPath | null isPairingRejected: (hostId: string) => boolean + isHostSignedOut: (hostId: string) => boolean subscribeHostState: (hostId: string, listener: (state: ConnectionState) => void) => () => void getAllClients: () => { hostId: string; client: RpcClient }[] subscribeAllHosts: (listener: () => void) => () => void diff --git a/mobile/src/transport/stable-logical-rpc-client.ts b/mobile/src/transport/stable-logical-rpc-client.ts index 206b40deaa4..e249940ce4a 100644 --- a/mobile/src/transport/stable-logical-rpc-client.ts +++ b/mobile/src/transport/stable-logical-rpc-client.ts @@ -55,6 +55,9 @@ export type StableLogicalRpcClient = RpcClient & { // Latched when the desktop has repeatedly refused this device's relay credential. setPairingRejected(rejected: boolean): void isPairingRejected(): boolean + // Latched when the relay named the desktop's own sign-out as the reason it is absent. + setHostSignedOut(signedOut: boolean): void + isHostSignedOut(): boolean // Recovery attempts share this signal so status-only changes rerender. onConnectionPathChange(listener: () => void): () => void getGeneration(): number @@ -286,6 +289,8 @@ export function createStableLogicalRpcClient( setRecoveryAttempt: (attempt) => connectionPath.setRecoveryAttempt(attempt), setPairingRejected: (rejected) => connectionPath.setPairingRejected(rejected), isPairingRejected: () => connectionPath.isPairingRejected(), + setHostSignedOut: (signedOut) => connectionPath.setHostSignedOut(signedOut), + isHostSignedOut: () => connectionPath.isHostSignedOut(), onConnectionPathChange: (listener) => connectionPath.subscribe(listener), getGeneration: () => generation } diff --git a/mobile/src/transport/use-all-host-clients.ts b/mobile/src/transport/use-all-host-clients.ts index 03ac5890015..70c709d965f 100644 --- a/mobile/src/transport/use-all-host-clients.ts +++ b/mobile/src/transport/use-all-host-clients.ts @@ -138,6 +138,7 @@ export function useAllHostClients(hostIds: string[], options?: UseAllHostClients path: MobileConnectionPath pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean }>((hostId) => { const client = clientsByHostId.get(hostId) return client @@ -148,7 +149,8 @@ export function useAllHostClients(hostIds: string[], options?: UseAllHostClients state: ctx.getState(hostId), path: ctx.getActivePath(hostId), pendingPath: ctx.getPendingPath(hostId), - pairingRejected: ctx.isPairingRejected(hostId) + pairingRejected: ctx.isPairingRejected(hostId), + hostSignedOut: ctx.isHostSignedOut(hostId) } ] : [] diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 5a63b5a9377..4f00b78214f 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -116,7 +116,7 @@ patchedDependencies: '@xterm/addon-webgl@0.20.0-beta.299': 94687e89a0115e6e6aa102837f986debdc029c091527ee5eb4a4e17ceaf9473e '@xterm/xterm@6.1.0-beta.303': 98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d lint-staged@16.4.0: 7333b3837f80a7fbd045964db6d76ba4fc118e49134bdbabb00585b6b7b60673 - node-pty@1.1.0: e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa + node-pty@1.1.0: 7cc9d45f3d2c38f142490d0805e75db55f0eef5174ad41c4b52abc5fbe079ad1 importers: @@ -157,7 +157,7 @@ importers: version: 3.3.1 node-pty: specifier: ^1.1.0 - version: 1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa) + version: 1.1.0(patch_hash=7cc9d45f3d2c38f142490d0805e75db55f0eef5174ad41c4b52abc5fbe079ad1) posthog-node: specifier: ^5.33.3 version: 5.33.3 @@ -12214,7 +12214,7 @@ snapshots: node-int64@0.4.0: {} - node-pty@1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa): + node-pty@1.1.0(patch_hash=7cc9d45f3d2c38f142490d0805e75db55f0eef5174ad41c4b52abc5fbe079ad1): dependencies: node-addon-api: 7.1.1 diff --git a/resources/skills/current-manifest.json b/resources/skills/current-manifest.json index c8b204f00c5..fdb54016a8f 100644 --- a/resources/skills/current-manifest.json +++ b/resources/skills/current-manifest.json @@ -131,17 +131,17 @@ "name": "orchestration", "sourcePath": "skills/orchestration", "releaseRevision": 29, - "packageDigest": "7a386ce558ba54abe02b4a0de5d71fe3d63c944ef0888ddde130c729b37f7cc8", - "gitTreeSha": "4199ec6988801dd491631706cba62631b4bed8fb", + "packageDigest": "689e31d84256aded123c801eaa87413474943a9a30d96bff9a19d0a321aefb54", + "gitTreeSha": "902cc33dd65730b32ac234dd0ae7166d75498b46", "files": [ { "path": "SKILL.md", - "size": 4451, + "size": 4398, "executable": false, "classification": "text", - "exactSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "textNormalizedSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "identitySha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f" + "exactSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "textNormalizedSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "identitySha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18" } ] } diff --git a/resources/skills/snapshot-registry.json b/resources/skills/snapshot-registry.json index 5842b48858a..5b3412a497b 100644 --- a/resources/skills/snapshot-registry.json +++ b/resources/skills/snapshot-registry.json @@ -1046,17 +1046,17 @@ }, { "releaseRevision": 29, - "packageDigest": "7a386ce558ba54abe02b4a0de5d71fe3d63c944ef0888ddde130c729b37f7cc8", - "gitTreeSha": "4199ec6988801dd491631706cba62631b4bed8fb", + "packageDigest": "689e31d84256aded123c801eaa87413474943a9a30d96bff9a19d0a321aefb54", + "gitTreeSha": "902cc33dd65730b32ac234dd0ae7166d75498b46", "files": [ { "path": "SKILL.md", - "size": 4451, + "size": 4398, "executable": false, "classification": "text", - "exactSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "textNormalizedSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "identitySha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f" + "exactSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "textNormalizedSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "identitySha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18" } ] } diff --git a/skill-guides/orchestration.md b/skill-guides/orchestration.md index b9b7ca79442..0878532c447 100644 --- a/skill-guides/orchestration.md +++ b/skill-guides/orchestration.md @@ -3,18 +3,18 @@ name: orchestration description: >- Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation - waits, task DAGs, decision gates, coordinator loops, or decomposing work - across agents. Use `orca-cli` instead for full ownership handoffs, including - requests phrased as "hand off", "handoff", "handover", "give this to another - agent", or "another worktree" when the user did not explicitly ask to - supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for - terminal control, lightweight terminal prompts, shell commands, Orca - worktree management, reading or waiting on terminals, and automation of the - browser embedded inside Orca. Use Computer Use for external browser windows, - webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when - the task requires OS/window-level control such as focus, menus, dialogs, - coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a - page-automation tool such as Playwright or CDP for external pages. + waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli` + instead for full ownership handoffs, including requests phrased as "hand + off", "handoff", "handover", "give this to another agent", or "another + worktree" when the user did not explicitly ask to supervise, monitor, wait + for results, or coordinate a DAG. Use `orca-cli` for terminal control, + lightweight terminal prompts, shell commands, Orca worktree management, + reading or waiting on terminals, and the Orca embedded browser. Use Computer + Use for external browser windows, webviews, Orca app UI, or desktop UI + outside Orca's embedded browser only when the task requires OS/window-level + control such as focus, menus, dialogs, coordinates, or screenshots. Use + `orca-cli` for Orca's embedded pages and a page-automation tool such as + Playwright or CDP for external pages. --- # Orca Inter-Agent Orchestration diff --git a/skills/orchestration/SKILL.md b/skills/orchestration/SKILL.md index 85a0ff8c4b0..5725a8f5512 100644 --- a/skills/orchestration/SKILL.md +++ b/skills/orchestration/SKILL.md @@ -3,18 +3,18 @@ name: orchestration description: >- Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation - waits, task DAGs, decision gates, coordinator loops, or decomposing work - across agents. Use `orca-cli` instead for full ownership handoffs, including - requests phrased as "hand off", "handoff", "handover", "give this to another - agent", or "another worktree" when the user did not explicitly ask to - supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for - terminal control, lightweight terminal prompts, shell commands, Orca - worktree management, reading or waiting on terminals, and automation of the - browser embedded inside Orca. Use Computer Use for external browser windows, - webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when - the task requires OS/window-level control such as focus, menus, dialogs, - coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a - page-automation tool such as Playwright or CDP for external pages. + waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli` + instead for full ownership handoffs, including requests phrased as "hand + off", "handoff", "handover", "give this to another agent", or "another + worktree" when the user did not explicitly ask to supervise, monitor, wait + for results, or coordinate a DAG. Use `orca-cli` for terminal control, + lightweight terminal prompts, shell commands, Orca worktree management, + reading or waiting on terminals, and the Orca embedded browser. Use Computer + Use for external browser windows, webviews, Orca app UI, or desktop UI + outside Orca's embedded browser only when the task requires OS/window-level + control such as focus, menus, dialogs, coordinates, or screenshots. Use + `orca-cli` for Orca's embedded pages and a page-automation tool such as + Playwright or CDP for external pages. --- # Orca Orchestration diff --git a/src/cli/bundled-skill-guides.ts b/src/cli/bundled-skill-guides.ts index 66625fc9a1f..06aae7bdd7d 100644 --- a/src/cli/bundled-skill-guides.ts +++ b/src/cli/bundled-skill-guides.ts @@ -30,7 +30,7 @@ const ORCA_LINEAR_MARKDOWN = "---\nname: orca-linear\ndescription: >-\n Use Orc const ORCA_PER_WORKSPACE_ENV_MARKDOWN = "---\nname: orca-per-workspace-env\ndescription: >-\n Set up, review, debug, or validate Orca per-workspace environment recipes —\n on-demand, disposable runtimes (cloud sandboxes, VMs, or local) created fresh\n for each workspace. Covers first-time setup (provider prerequisites, the\n reusable base snapshot, the coding-agent auth snapshot, credentials, and\n state), not just the per-workspace lifecycle scripts. Use to stand up\n per-workspace environments, fix an `environmentRecipes` entry in `orca.yaml`, scaffold\n provider lifecycle scripts, or resolve an `orca vm recipe doctor` failure.\n---\n\n# Per-Workspace Environments\n\nHelp a user stand up and maintain a repo-owned per-workspace environment recipe end to end. Each\nworkspace gets its own on-demand, disposable runtime (a cloud sandbox, a VM, or a local one),\ncreated fresh and torn down after.\n\nOrca is a **thin wrapper**: you guide, detect, and scaffold; you never own the user's cloud account,\nbilling, images, or credentials.\n\n- **You DO:** sequence the setup, detect what's detectable (provider CLI present/logged-in? recipe\n present? `doctor` passing?), scaffold provider-templated scripts the user fills in, drive the slow\n snapshot/auth phases with the user, and always show the next action.\n- **You DO NOT:** create accounts, choose plans/regions, invent org/project/scope ids, store or print\n secrets, or run anything that spends money without an explicit user OK.\n\nFirst-time setup has **four phases before the per-workspace recipe runs** — easy to miss, so walk\nthem in order:\n\n1. **Prerequisites** — cloud account, provider CLI, scope/project, plan limits, git token (§2).\n2. **Base snapshot** — reusable image: tools + repo + headless build, snapshotted once (§3).\n3. **Agent-auth snapshot** — boot the base, run interactive device-auth, re-snapshot (§4).\n4. **State** — thread snapshot id / scope / project / port between phases via a state file (§6).\n\nThen the **per-workspace contract** (create/suspend/resume/destroy) runs fast (§8).\n\n**The one branch that shapes everything — connection mode:** **Orca-server** (`create` runs `orca serve`\nin the env and emits a `pairingCode`; §7c/§7f) vs **SSH** (`create` runs no server and emits a\n`connection.type:\"ssh\"` block Orca dials into; §7g/§7h). Settle this first — it changes the `create`\noutput shape and half the templates.\n\nKeep Orca's checkout behavior unchanged by default: omit `checkoutMode`, emit schema version 1, and\nlet Orca create a linked worktree. Only use `checkoutMode: provisioned-root` when the user explicitly\nwants one ephemeral machine to clone the finished workspace itself. This niche mode currently requires\ndirect SSH, an ordinary non-bare/non-sparse primary checkout at `projectRoot`, and schema version 2.\n\n**Quick-start (happy path):** interview the user (connection mode Orca-server vs SSH, provider, agent CLI,\ngit auth — §1.2) + read the provider's CLI docs → scaffold `scripts/orca-vm/` from §7 → run the\nbase-snapshot script, then the auth script (you invoke these by hand; not via `orca.yaml`) → wire\n`environmentRecipes` in `orca.yaml` → `orca vm recipe doctor --json` (free) → then the `--provision`\nself-test loop (§9) until it passes.\n\n---\n\n## 1. Setup workflow\n\nDrive these with the user. **[CHECKPOINT]** steps need explicit confirmation — they spend money, take\na long time, or need the user at the keyboard. Never create an Orca workspace or commit unless asked.\n\n1. **Inspect the repo** for an existing `environmentRecipes` entry, `scripts/orca-vm/`, a state file, or setup\n notes. If a working recipe exists, jump to Doctor (§9) instead of rebuilding.\n2. **Interview the user up front** — gather these choices and confirm them back before scaffolding\n anything. Don't pick for them (§11); don't guess.\n - **Connection mode:** how Orca attaches to the environment — an **Orca server** (the VM runs\n `orca serve` and Orca pairs over its pairing URL; worked example §7f) or **SSH** (Orca connects to\n the host over SSH; §7g). This decides the recipe's connection shape, so settle it first.\n - **Checkout ownership:** do not ask by default. Only when the user requires the environment to\n create the exact final checkout, confirm `provisioned-root` and direct SSH; otherwise omit it.\n - **Provider:** Vercel Sandbox, Fly, Modal, an existing SSH host, … For non-obvious providers, also\n ask scope/project/region and plan limits (§2). Then **read that provider's CLI/SDK docs** (or\n ` --help`) before scaffolding — you need its exact create/exec/snapshot/remove verbs.\n If a provider advertises `ssh`, verify whether it exposes a real dialable SSH target\n (host/port/user/key or proxy command) or only a provider-mediated interactive shell; Orca SSH mode\n needs the former.\n - **Coding-agent CLI + account:** which agent runs in the VM (`codex`, `claude`, …) and that the user\n has an account for it — it gets logged in during the Phase-3 auth snapshot (§4).\n - **Git auth:** the token source for cloning a private repo (`GH_TOKEN`/`GITHUB_TOKEN` or `gh auth\ntoken`; §5).\n3. **Check prerequisites (§2)** — detect the provider CLI + auth and confirm the items above are in\n place before any paid step.\n4. **Scaffold scripts + state file** from §7 (worked Vercel example: §7f; SSH host: §7g; Docker SSH:\n §7h; Windows: §7i), filling in the provider's real commands. Make them executable.\n5. **[CHECKPOINT] Build the base snapshot (§3)** — paid, slow.\n6. **[CHECKPOINT] Authenticate the agent (§4)** — interactive; the user follows a URL/code. **You cannot\n drive this step** — you run commands non-interactively, so there's no TTY for `docker exec -it` /\n `ssh -t` to prompt against. The **user** runs the Phase-3 login in their own terminal (or via the\n Claude Code harness bang-prefix — `! `, with the required space after `!`); you scaffold and drive\n the non-interactive phases around it. After kicking it off, **ask the user to report back once the login\n finishes** — you can't observe it completing, and you need that confirmation before resuming the\n non-interactive steps (base/auth commit, doctor, provision).\n7. **Wire the recipe** so `orca.yaml` points create/suspend/resume/destroy at the scripts (§8). The\n workspace composer reads `environmentRecipes` from the project's primary checkout of `orca.yaml`, **not** from\n a feature branch or worktree. So a recipe added only on a branch won't appear as a \"Run on\" option\n until that `orca.yaml` change is committed and merged to the project's primary branch. Tell the user\n this up front: `doctor`/`--provision` validate the scripts from the working copy on any branch, but\n creating a workspace from the recipe in the picker needs it on primary.\n8. **Dry-run doctor** — `orca vm recipe doctor --repo-path --json` (free, static; §9).\n Fix every failure before going live.\n9. **[CHECKPOINT] Live self-test** — get the user's OK once, then run\n `orca vm recipe doctor --provision --json` as a loop: it runs create → validates →\n destroys, and on failure returns a full transcript. Read it, fix the scripts, and re-run yourself until\n it passes (§9). Spends cloud money; the one approval covers the loop.\n10. **[CHECKPOINT] Optional workspace test** — only if asked: create a workspace via the picker, then\n verify sleep/wake/delete.\n\n---\n\n## 2. Phase 1 — Prerequisites\n\nThe user's responsibility; verify what's verifiable, ask for the rest, invent nothing. State which\nitems you verified vs. which the user asserted.\n\n- **Connection mode** (Orca server vs SSH) confirmed with the user — see §1 step 2; it shapes the recipe.\n- **Cloud account + plan** that allows sandboxes/VMs. Ask.\n- **Provider CLI installed + authenticated** — detect (`command -v `), check auth (e.g.\n `vercel whoami`). If missing, point at the provider's docs; don't log them in.\n- **Scope / project / region** the sandboxes live under. Ask; flows into every script via state.\n- **Plan / timeout / RAM caps.** Record them — e.g. Vercel Hobby caps sandbox timeout at **45m**,\n which limits both the base build and per-workspace runtime (see §10).\n- **Git token for private repos** (`GH_TOKEN`/`GITHUB_TOKEN`, or the provider's git auth; can fall back\n to `gh auth token`). See §5.\n- **Coding-agent CLI choice** (`codex`, `claude`…) and that the user has an account — it gets\n authenticated into the VM in Phase 3.\n\n---\n\n## 3. Phase 2 — Base snapshot (the reusable image)\n\nBuild **once**, snapshot, and every workspace boots from it in seconds instead of rebuilding.\nProvisioning + building takes a while (often ~20–30 min), so it runs behind a checkpoint. The script\nshape is §7a; key points:\n\n- Build the **headless Electron main only** (not the renderer) so it fits in plan RAM.\n- Use the VM image's package manager (`apt`/`dnf`/`apk`, per the base distro — not the provider brand).\n- Clone with the git token via `GIT_ASKPASS` (§5).\n- **Trap errors and remove the half-built sandbox** so a crash doesn't leave a paid resource running.\n- **Never snapshot a machine on which the Orca runtime has already run.** The first `orca serve` creates\n the runtime's user-data dir, and everything in it gets baked into the image and shared by every VM\n booted from it: the pairing keypair and device-token registry (`orca-devices.json`,\n `orca-e2ee-keypair.json`), `agent-session-authority.key`, and the build box's logs, terminal history\n and orchestration db. Confirmed: two VMs from one such snapshot emitted **identical `deviceToken` and\n `pairedDeviceId`**. Snapshot **before** the runtime has ever run, or delete the resolved user-data\n directory first: `orca_user_data_path=\"${ORCA_USER_DATA_PATH:-${XDG_CONFIG_HOME:-$HOME/.config}/orca}\"; rm -rf -- \"$orca_user_data_path\"`.\n This matches Orca's Linux precedence for custom and default paths; deleting a named file list will\n drift as Orca adds state.\n- Snapshot the stopped sandbox, parse the snapshot id, and write it + scope/project/port/repo to state.\n\n---\n\n## 4. Phase 3 — Agent-auth snapshot (interactive)\n\nThe base snapshot has the agent CLI installed but **not logged in**, and per-workspace VMs are\nephemeral — so authenticate once and bake it into a second snapshot layer. Script shape is §7b:\n\n1. Boot a sandbox from the base `snapshotId` (from state).\n2. Run the agent's login **interactively** (`--interactive --tty`); the user completes the URL/code in\n their browser. On a **headless VM this must be the device-auth flow** (e.g. `codex login --device-auth`),\n **not** plain `codex login`: the default OAuth login starts a loopback callback server on a container\n port the host browser can't reach, so it hangs. Device-auth instead prints a URL + code the user opens\n on the **host**.\n3. Verify login; **refuse to snapshot an unauthenticated VM.** Prefer the status command's **exit code**\n (most agent CLIs exit non-zero when unauthenticated). If you grep instead, agent status often goes to\n **stderr** (e.g. `codex login status` prints \"Logged in using ChatGPT\" there), so **fold stderr first**\n (`... 2>&1 | grep …`) and match the agent's **exact success line** — never `grep -qi 'logged in'`, which\n also matches \"**not** logged in\" and would commit an unauthenticated image.\n4. Re-snapshot, parse the new id, and overwrite `snapshotId` in state to the authenticated image\n (recording `authSourceSnapshotId`). Remove the auth sandbox.\n\n**You can't drive step 2 yourself** (you run commands non-interactively — no TTY). The **user** runs it in\ntheir own terminal, or via the Claude Code harness bang-prefix (`! `, with the required space after\n`!`). You scaffold/boot the sandbox and run steps 3–4, but **you cannot observe the interactive login\nfinishing** — so **ask the user to tell you when it's done** before you verify and re-snapshot.\n\nThis layer inherits §3's rule: if you started `orca serve` on the base or auth sandbox to smoke-test it,\ndelete the runtime's user-data dir (`~/.config/orca` on Linux) before re-snapshotting, or every workspace\nbooted from this image shares one pairing identity and one `agent-session-authority.key`.\n\nIf the agent's credentials are short-lived, warn that the snapshot may need periodic re-auth (§10).\n\nFor disposable runtimes, do **not** treat a host agent config directory (for example `~/.codex`) as the\nauth snapshot by bind-mounting or copying it wholesale. Agent homes often contain sqlite state, hook\napproval state, caches, logs, and host-specific env/config. Instead, authenticate/configure the agent\ninside the disposable runtime and snapshot/commit that runtime layer.\n\n---\n\n## 5. Credentials\n\n- **Never** commit secrets or put them in `userData`, recipe JSON, comments, docs, or the state file.\n- **Git token:** read from env (`GH_TOKEN`/`GITHUB_TOKEN`), falling back to `gh auth token`. Pass to the\n VM only via the provider's ephemeral `--env`. Inside the VM, use a `GIT_ASKPASS` helper with\n `x-access-token` (not the token in the clone URL) and `GIT_TERMINAL_PROMPT=0` so a missing token fails\n fast instead of hanging. When you write the helper from inside `bash -lc` under `set -u`, escape the\n positional arg and the token (`\\$1`, `\\$GH_TOKEN`) so they land **literally** and resolve at git-runtime\n — an unescaped `$1` aborts with \"unbound variable\", and a literal `$GH_TOKEN` keeps the real token out of\n the written file. `rm -f` the helper after the clone/fetch.\n- **Provider auth:** rely on the provider CLI's logged-in session, not checked-in keys.\n- **Agent auth:** lives in the authenticated snapshot (Phase 3) — never a file you write or commit.\n- State holds only **non-secret** wiring (snapshot ids, scope, project, port, repo url/ref).\n\n---\n\n## 6. State file\n\nA repo-local JSON file (e.g. `scripts/orca-vm/-state.json`) threads non-secret values between\nphases. Each script resolves values as **env var → state → built-in fallback**, and merges its outputs\nback. Phase 2 writes the base `snapshotId`; Phase 3 overwrites it with the authenticated snapshot;\nper-workspace `create` boots from `snapshotId`.\n\n```json\n{\n \"baseName\": \"orca-base\",\n \"snapshotId\": \"snap_authenticated_image_id\",\n \"authSourceSnapshotId\": \"snap_base_image_id\",\n \"scope\": \"\",\n \"project\": \"\",\n \"port\": 7331,\n \"repoUrl\": \"https://host/org/repo.git\",\n \"repoRef\": \"main\",\n \"projectRoot\": \"/abs/path/on/remote/repo\"\n}\n```\n\n---\n\n## 7. Script templates (provider-agnostic shapes)\n\nScaffold under `scripts/orca-vm/`. These are **shapes** — fill in the provider's real commands. All\nreserve stdout for the final JSON and log progress to stderr. Include a shared `json_value ` /\n`env_value ` reader (env → state → fallback) in each.\n\n**Where each script runs:**\n\n- **Local-side** (`create`/`suspend`/`resume`/`destroy` + the base-snapshot/auth scripts the user\n invokes) runs **on the user's desktop**, so it must run on their OS. macOS/Linux: `#!/usr/bin/env\nbash`, `set -euo pipefail`, quoted paths. **Windows:** a bare `.sh` won't run — scaffold `.ps1`/`.cmd`\n or require WSL/Git-Bash and point `orca.yaml` at the right launcher.\n- **Remote-side** (commands you `exec` _inside_ the Linux VM) always runs in the VM's Linux shell, so\n bash is fine there regardless of the user's OS.\n\n### 7a. Base-snapshot (`-base-snapshot.sh`) — Phase 2\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve base_name/repo_url/repo_ref/project_root/port/scope/project/timeout (env→state→fallback)\n# resolve gh token: GH_TOKEN | GITHUB_TOKEN | `gh auth token`\n# 1. provision a sandbox (timeout/vcpus/published port/snapshot retention); trap: remove on error\n# 2. remote exec (long timeout): install pkgs + gh + corepack/pnpm + agent CLI;\n# clone with GIT_ASKPASS(token); write headless main-only build config;\n# dev setup; pnpm install; build CLI; build headless electron main; smoke-check tools\n# 3. snapshot stopped sandbox; parse snapshot id (fail if unparseable)\n# 4. merge { baseName, snapshotId, projectRoot, repoUrl, repoRef, port, scope, project } into state\n# print only the state JSON to stdout\n```\n\nWorked Vercel commands for this phase are in §7f. You run this script by hand (not via `orca.yaml`),\nafter exporting the first-run inputs the state file doesn't have yet — e.g. provider scope/project, the\nrepo URL/ref, and a git token (`GH_TOKEN`); later runs read them back from state.\n\n### 7b. Auth (`-base-auth.sh`) — Phase 3\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read source snapshot from state.snapshotId (fail if absent); auth_name=\"${base_name}-auth\"\n# 1. boot sandbox from source snapshot; trap: remove on error\n# 2. INTERACTIVE/TTY remote exec: agent login — user completes URL/code. Headless VM: MUST use the\n# device-auth flow (e.g. `codex login --device-auth`) — plain OAuth login binds a loopback callback\n# port the host can't reach and hangs. User runs this themselves (you have no interactive TTY); ask\n# them to report back when it's done before continuing.\n# 3. verify login, then refuse to snapshot if not logged in. Prefer the status command's EXIT CODE (most\n# agent CLIs exit non-zero when unauthenticated) over string-matching. If you must grep, fold stderr\n# first (`status 2>&1 | grep …` — many agents print the success line there) and match the agent's exact\n# success line; never `grep -qi 'logged in'`, which also matches \"not logged in\". Codex example: §7f.\n# 4. snapshot; parse new id\n# 5. merge { snapshotId:, authSourceSnapshotId: } into state; remove auth sandbox\n# print only the state JSON to stdout\n```\n\n### 7c. Create (`-create.sh`) — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read authenticated snapshotId/scope/project/port/repo*/project_root (env→state→fallback)\n# fail clearly if snapshotId is missing (point back to Phases 2–3)\n# name = orca-${ORCA_RECIPE_ID}-${ORCA_VM_INSTANCE_ID} (sanitized, length-capped)\n# 1. boot sandbox from snapshotId with a published port; capture the public URL → pairing address\n# (an externally reachable wss:// URL); trap: remove sandbox on error\n# 2. remote exec: ensure repo at desired commit; rebuild only if commit changed (cache marker)\n# 3. remote exec: start orca serve in the background and read the recipe JSON it writes (see below)\n# 4. print serve's JSON to stdout, optionally enriched with userData:\n# { schemaVersion:1, pairingCode, projectRoot, userData:{ provider, resourceId:name, snapshotId } }\n```\n\n**The exact `orca serve` invocation and its output (verified — do not improvise the flags).** Inside the\nVM, run:\n\n```bash\norca serve \\\n --port \"$PORT\" \\\n --project-root \"$ABS_REPO_PATH_ON_REMOTE\" \\\n --pairing-address \"$EXTERNAL_WSS_URL\" \\\n --recipe-json\n```\n\n**Binary name:** in a VM built from source (the Phase-2 flow), run it as `pnpm exec orca-dev serve …`\nfrom the repo root — `orca-dev` is the in-repo entrypoint and is what the §7f example uses. Plain\n`orca serve …` is the same command when the built CLI is installed on the VM's PATH. The flags/output\nare identical either way.\n\nThere is **no `--host` flag**. `--project-root` must be an absolute directory on the remote. With\n`--recipe-json` the server **stays running** and prints exactly this single object to **stdout**, then\nkeeps serving:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"\",\n \"projectRoot\": \"\"\n}\n```\n\n`pairingCode` is the pairing URL, already pointing at whatever you passed as `--pairing-address` — so set\n`--pairing-address` to the externally reachable address and **pass `pairingCode` through unchanged; never\nhand-rewrite it**. Because serve runs in the foreground and doesn't exit, redirect its stdout to a file\nand poll until that file parses as JSON (and bail if the process dies — dump its stderr log). Your\n`create` script then prints that JSON (optionally merging `userData`). Concrete pattern: §7f.\n\n### 7d. Suspend / resume / destroy — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\npayload=\"$(cat)\" # Orca passes lifecycle JSON on stdin\nresource_id=\"$(node -e 'const d=JSON.parse(process.argv[1]); process.stdout.write(d.recipeResult?.userData?.resourceId ?? \"\")' \"$payload\")\"\n[ -n \"$resource_id\" ] || { echo \"No resource id in lifecycle payload\" >&2; exit 1; }\n# suspend: provider suspend \"$resource_id\"\n# resume: provider resume \"$resource_id\"; then RE-EMIT fresh recipe JSON (pairing may change)\n# destroy: provider remove \"$resource_id\" (or set destroy: none in orca.yaml)\n```\n\n### 7e. State file — scaffold with scope/project/repo filled in and snapshot ids empty (§6).\n\n### 7f. Worked example — Vercel Sandbox (all three phases)\n\nA real, working shape (the Vercel surface is a CLI: `vercel sandbox create|exec|snapshot|remove`). Adapt\nnames; verify flags against `vercel sandbox --help` for the user's CLI version before relying on them.\nThese ground §7a (base snapshot) and §7b (auth), which are otherwise generic skeletons.\n\n**Phase 2 — base snapshot (§7a):** provision → install tools + clone + headless build → snapshot.\n\n```bash\n# provision a fresh build sandbox (retain a couple of snapshots); trap-remove on error\nvercel sandbox create --name \"$base\" --runtime node24 --timeout 30m --vcpus 4 --publish-port \"$port\" \\\n --snapshot-expiration 30d --keep-last-snapshots 2 \"${vercel_args[@]}\" >&2\n# remote build (long timeout): install pkgs+gh+pnpm+agent CLI, clone with GIT_ASKPASS (write the helper\n# with LITERAL \\$1/\\$GH_TOKEN so they resolve at git-runtime, not write-time — see §5/§7f create — then\n# `rm -f /tmp/askpass.sh`), write the headless main-only build config (drop the renderer), dev setup,\n# build CLI + headless main, smoke-check\nvercel sandbox exec \"$base\" \"${vercel_args[@]}\" --timeout 25m --env \"GH_TOKEN=$gh_token\" … -- bash -lc '…build…' >&2\n# snapshot the STOPPED sandbox and parse the id from CLI output (fail if unparseable)\nout=\"$(vercel sandbox snapshot \"$base\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nsnapshot_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# merge { baseName, snapshotId, scope, project, port, repoUrl, repoRef, projectRoot } into state; print state JSON\n```\n\n**Phase 3 — agent-auth snapshot (§7b):** boot the base, log the agent in interactively, re-snapshot.\n(`codex` below is an example — substitute the user's chosen agent's login/status verbs, e.g. `claude`.)\n\n```bash\nvercel sandbox create --name \"$auth\" --snapshot \"$snapshot_id\" --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" >&2\n# INTERACTIVE — the USER runs this in their own terminal (you have no interactive TTY) and completes the\n# URL/code on the HOST. --device-auth is MANDATORY on a headless VM: plain `codex login` binds a loopback\n# callback port the host browser can't reach and hangs. Ask the user to report back when login finishes.\nvercel sandbox exec --interactive --tty \"$auth\" \"${vercel_args[@]}\" -- bash -lc 'codex login --device-auth'\n# refuse to snapshot an unauthenticated VM — fold stderr, match codex's exact success line (§4)\nvercel sandbox exec \"$auth\" \"${vercel_args[@]}\" --timeout 30s -- bash -lc 'codex login status 2>&1' | grep -Eqi 'Logged in using ChatGPT|Logged in via device' \\\n || { echo \"agent not logged in; not snapshotting\" >&2; exit 1; }\nout=\"$(vercel sandbox snapshot \"$auth\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nnew_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# overwrite state.snapshotId = new_id, record authSourceSnapshotId = snapshot_id; remove the auth sandbox\n```\n\n**Per-workspace `create`** (the fast path):\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback: snapshot_id, scope, project, port, repo_url, repo_ref, project_root\nvercel_args=(); [ -n \"$scope\" ] && vercel_args+=(--scope \"$scope\"); [ -n \"$project\" ] && vercel_args+=(--project \"$project\")\n[ -n \"$snapshot_id\" ] || { echo \"snapshotId missing — run Phases 2–3 first\" >&2; exit 1; }\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nrecipe_id=\"${ORCA_RECIPE_ID:-vercel-sandbox}\"\nrecipe_id=\"${recipe_id//./-}\" # Vercel names forbid dots.\ninstance_id=\"${ORCA_VM_INSTANCE_ID:-$(date +%s)}\"\nmax_recipe_id_length=$((128 - ${#instance_id} - 6)) # Preserve the unique instance suffix.\n[ \"$max_recipe_id_length\" -gt 0 ] || { echo \"ORCA_VM_INSTANCE_ID is too long for a Vercel sandbox name\" >&2; exit 1; }\nname=\"orca-${recipe_id:0:max_recipe_id_length}-${instance_id}\"\n\n# Arm cleanup BEFORE create so a failing create can't leak a half-built paid sandbox.\ncleanup_on_error() { [ \"$?\" -ne 0 ] && vercel sandbox remove \"$name\" \"${vercel_args[@]}\" >/dev/null 2>&1 || true; }\ntrap cleanup_on_error EXIT\n\n# 1. boot from the authenticated snapshot, publish the serve port\ncreate_output=\"$(vercel sandbox create --name \"$name\" --snapshot \"$snapshot_id\" \\\n --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$create_output\" >&2\n# Vercel prints the published https URL; derive the external wss:// pairing address from it\npublic_url=\"$(printf '%s\\n' \"$create_output\" | sed -nE 's#.*(https://[^[:space:]]+\\.vercel\\.run).*#\\1#p' | head -1)\"\n[ -n \"$public_url\" ] || { echo \"no published URL in create output\" >&2; exit 1; }\npairing_ws=\"${public_url/https:\\/\\//wss://}\"\n\n# 2. (remote) ensure the repo is at the right commit; rebuild only if the commit changed (cache marker)\nvercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 20m \\\n --env \"GH_TOKEN=$gh_token\" --env \"ORCA_PROJECT_ROOT=$project_root\" \\\n --env \"ORCA_REPO_URL=$repo_url\" --env \"ORCA_REPO_REF=$repo_ref\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; \\\n # Re-establish git auth for the private-repo fetch (why + full rationale: §5); else it hangs on a prompt.\n # Load-bearing escaping: \\$1 and \\$GH_TOKEN must land LITERALLY and resolve at git-runtime. Test after\n # any edit here — reformatting the nested printf/node quoting silently breaks the fetch or leaks the token.\n if [ -n \"${GH_TOKEN:-}\" ]; then \\\n printf \"%s\\n\" \"#!/usr/bin/env bash\" \"case \\\"\\$1\\\" in *Username*) echo x-access-token;; *Password*) echo \\\"\\$GH_TOKEN\\\";; esac\" > /tmp/askpass.sh; \\\n chmod 700 /tmp/askpass.sh; export GIT_ASKPASS=/tmp/askpass.sh GIT_TERMINAL_PROMPT=0; fi; \\\n git fetch origin \"$ORCA_REPO_REF\"; \\\n git checkout -B \"$ORCA_REPO_REF\" FETCH_HEAD; \\\n rm -f /tmp/askpass.sh; \\\n c=\"$(git rev-parse HEAD)\"; [ -f .orca-built ] && [ \"$(cat .orca-built)\" = \"$c\" ] || { \\\n pnpm install --prefer-offline && pnpm run build:cli && \\\n node config/scripts/run-electron-vite-build.mjs --config config/electron-vite.vm-serve.config.ts && \\\n printf \"%s\" \"$c\" > .orca-built; }' >&2\n\n# 3. (remote) start orca serve in the background, writing recipe JSON to a file; poll until it parses\nrecipe_json=\"$(vercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 60s \\\n --env \"ORCA_PORT=$port\" --env \"ORCA_PROJECT_ROOT=$project_root\" --env \"ORCA_PAIRING_ADDRESS=$pairing_ws\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; rm -f /tmp/orca-recipe.json /tmp/orca-serve.log; \\\n nohup pnpm exec orca-dev serve --port \"$ORCA_PORT\" --project-root \"$ORCA_PROJECT_ROOT\" \\\n --pairing-address \"$ORCA_PAIRING_ADDRESS\" --recipe-json >/tmp/orca-recipe.json 2>/tmp/orca-serve.log /dev/null 2>&1 && { cat /tmp/orca-recipe.json; exit 0; }; \\\n kill -0 \"$pid\" 2>/dev/null || { cat /tmp/orca-serve.log >&2; exit 1; }; sleep 0.25; \\\n done; cat /tmp/orca-serve.log >&2; echo \"serve recipe JSON timed out\" >&2; exit 1')\"\n\n# 4. print serve's JSON enriched with userData (single object on stdout)\nnode -e 'const p=JSON.parse(process.argv[1]); console.log(JSON.stringify({...p, schemaVersion:1,\n userData:{...p.userData, provider:\"vercel-sandbox\", resourceId:process.argv[2], snapshotId:process.argv[3]}}))' \\\n \"$recipe_json\" \"$name\" \"$snapshot_id\"\ntrap - EXIT\n```\n\n`suspend`/`resume`/`destroy` use `vercel sandbox stop|...|remove \"$resource_id\"` reading\n`userData.resourceId` from stdin (§7d). This is the **Orca-server** connection mode (the recipe emits a\npairing URL). If the user chose **SSH** in the §1 interview, use §7g instead.\n\n### 7g. Worked example — existing SSH host (SSH connection mode)\n\nSSH mode is **fundamentally different from §7c/§7f**, not a relabeling of them:\n\n- **`create` does NOT run `orca serve` and does NOT emit a `pairingCode`.** Orca itself connects to the\n host over its SSH relay, brings up the git + filesystem providers, and imports the repo. The script's\n only job is to make the host ready and **print SSH connection details** Orca will dial.\n- The result uses a `connection` block with `type: \"ssh\"` and a `target`, **not** the flat\n `pairingCode`/`projectRoot` shape. Exact shape (Orca rejects anything else):\n\n```json\n{\n \"schemaVersion\": 1,\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/path/to/repo/on/host\",\n \"target\": {\n \"label\": \"my-box\",\n \"host\": \"192.0.2.10\",\n \"port\": 22,\n \"username\": \"ubuntu\",\n \"identityFile\": \"~/.ssh/id_ed25519\",\n \"jumpHost\": \"bastion.example.com\",\n \"proxyCommand\": \"cloudflared access ssh --hostname %h\",\n \"relayGracePeriodSeconds\": 0,\n \"portForwards\": []\n }\n }\n}\n```\n\n`label`, `host`, `port`, `username` are required; the rest are optional — omit any you don't need.\n\nFor an explicitly requested one-VM-per-workspace checkout, the create script must read\n`ORCA_RECIPE_RESULT_SCHEMA_VERSION`, `ORCA_REPO_URL`, `ORCA_REPO_REF`, `ORCA_REPO_REF_HEAD`, and\n`ORCA_REPO_BRANCH`. Use `ORCA_REPO_REF` to fetch the selected source, but create\n`ORCA_REPO_BRANCH` at the exact `ORCA_REPO_REF_HEAD` commit; resolving the symbolic ref again can race\nwith an upstream update. `ORCA_REPO_URL` and `ORCA_REPO_REF` are a matched fetch pair, including when\nthe desktop source uses multiple remotes. Return that primary checkout at `projectRoot` and emit the\nsame SSH result with:\n\n```bash\n[ -n \"${ORCA_REPO_REF_HEAD:-}\" ] || { echo \"missing pinned source commit\" >&2; exit 1; }\ngit fetch origin \"$ORCA_REPO_REF\"\ngit cat-file -e \"${ORCA_REPO_REF_HEAD}^{commit}\"\ngit checkout -B \"$ORCA_REPO_BRANCH\" \"$ORCA_REPO_REF_HEAD\"\n```\n\n```json\n{\n \"schemaVersion\": 2,\n \"checkoutMode\": \"provisioned-root\",\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/repo\",\n \"target\": { \"label\": \"my-box\", \"host\": \"192.0.2.10\", \"port\": 22, \"username\": \"ubuntu\" }\n }\n}\n```\n\nFail if the requested schema is not `2`; do not silently fall back to the ordinary recipe shape.\n\n**Networking → which `target` fields to set** (how _your desktop_ reaches the box — there is no\n`orca serve` URL in SSH mode):\n\n- Public IP / DNS, or a Tailscale/VPN address → `host`; SSH port → `port` (usually 22).\n- Key auth → `identityFile` (add `identitiesOnly: true` if the agent has many keys).\n- Through a bastion → `jumpHost` (a `user@host` ProxyJump) **or** a full `proxyCommand` (e.g. an access\n proxy). Use one, not both.\n- A service port the workspace needs → add entries to `portForwards`.\n- `relayGracePeriodSeconds` (optional): how long Orca keeps the SSH relay alive after the workspace\n detaches before tearing it down; `0` = tear down immediately. Leave it off unless the user wants a\n reconnect grace window.\n\n**Toolchain & agent auth on a persistent (no-snapshot) host — do this ONCE, by hand, before wiring the\nrecipe** (there's no base image to bake; the host _is_ the base). Run the §7f Phase-2 install steps and\nthe §7f Phase-3 ` login --device-auth` **directly over SSH on the host** (interactive, e.g.\n`ssh -t user@host ' login --device-auth'`). After that the host stays ready across workspaces.\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback (default unset optionals to \"\"): ssh_username, host,\n# ssh_port (default 22), identity_file, jump_host, proxy_command, project_root, repo_url, repo_ref\n: \"${identity_file:=}\"; : \"${jump_host:=}\"; : \"${proxy_command:=}\" # avoid set -u aborts on optionals\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nssh_target=\"${ssh_username}@${host}\"\nssh_opts=(-p \"$ssh_port\"); [ -n \"$identity_file\" ] && ssh_opts+=(-i \"$identity_file\")\n# Why: a fresh host's key isn't in known_hosts; a StrictHostKeyChecking prompt would HANG a\n# non-interactive create. Pre-add the key (or set the option) so it can't block.\nssh-keyscan -p \"$ssh_port\" \"$host\" >> \"$HOME/.ssh/known_hosts\" 2>/dev/null || true\n\n# 1. ensure the repo is present and at the right commit on the host (NO orca serve here)\nssh \"${ssh_opts[@]}\" \"$ssh_target\" \\\n \"GH_TOKEN='$gh_token' GIT_TERMINAL_PROMPT=0 bash -lc '\n set -euo pipefail\n [ -d \\\"$project_root/.git\\\" ] || git clone \\\"$repo_url\\\" \\\"$project_root\\\"\n cd \\\"$project_root\\\" && git fetch origin \\\"$repo_ref\\\" && git checkout -B \\\"$repo_ref\\\" FETCH_HEAD\n '\" >&2\n\n# 2. print the SSH connection block (NO pairingCode, NO orca serve). host/port/username tell Orca's\n# relay how to dial in; identityFile/jumpHost/proxyCommand/portForwards are emitted when set.\nnode -e 'const [host,port,user,idf,jh,pc,root]=process.argv.slice(1);\n const target={ label:\"per-workspace-host\", host, port:Number(port), username:user };\n if(idf) target.identityFile=idf; if(jh) target.jumpHost=jh; if(pc) target.proxyCommand=pc;\n // add target.portForwards=[...] here if the workspace needs forwarded service ports\n console.log(JSON.stringify({ schemaVersion:1, connection:{ type:\"ssh\", projectRoot:root, target } }))' \\\n \"$host\" \"$ssh_port\" \"$ssh_username\" \"$identity_file\" \"$jump_host\" \"$proxy_command\" \"$project_root\"\n```\n\n`suspend`/`resume`/`destroy`: on a persistent host there's usually nothing to tear down — set\n`destroy: none` and omit suspend/resume. (Orca still disconnects/reconnects its own SSH relay on\nsleep/wake/delete — that's separate from these scripts.)\n\nIf the SSH host is instead an **ephemeral/snapshot-capable VM** (your hypervisor, or a cloud VM with\nimage support), keep the §7f Phase-2/3 base-image model for provisioning, but still emit the\n`connection.type:\"ssh\"` block above instead of starting `orca serve`.\n\n### 7h. Worked example — local Docker SSH (SSH connection mode)\n\nLocal Docker can model an ephemeral SSH VM without cloud cost: build a base image with `sshd`, tools,\nrepo prerequisites, and the agent CLI; run an **interactive auth container** once; then `docker commit`\nthat container as the authenticated image used by per-workspace `create`.\n\nKey points:\n\n- Publish container SSH to a random localhost port (`-p 127.0.0.1::22`) and emit\n `connection.type:\"ssh\"` with `host:\"127.0.0.1\"`, that port, `username`, `identityFile`, and\n `identitiesOnly:true`.\n- Generate a repo-local SSH key if needed, but gitignore the private/public key files.\n- **Bake SSH host keys into the base image** (`ssh-keygen -A` at **build** time; at runtime only generate\n if absent). Ephemeral containers all present the **same** host key, so `known_hosts` on `127.0.0.1`\n doesn't churn as the published port rotates across workspaces (otherwise every container's freshly\n generated key collides on `localhost` and trips host-key-changed warnings).\n- The auth image is the Docker equivalent of Phase 3: the **user** runs the agent login **inside** the\n container (you can't drive it — you have no interactive TTY), configures proxy env/config, approves\n hooks, and you commit once they report it's done. On a headless container use the **device-auth** flow\n (§4). Verify login before committing — exit code, or fold stderr and match the exact success line (§4).\n- Do not bind-mount or copy the host's full agent home into the image. Let each container have writable\n agent state; only the committed auth image should carry reusable authenticated state.\n- If committing from an interactive shell, force the runtime entrypoint back to `sshd`:\n `docker commit --change='ENTRYPOINT [\"/usr/local/bin/orca-docker-ssh-entrypoint\"]' …`.\n- `destroy` should read `recipeResult.userData.resourceId` and run `docker rm -f \"$resource_id\"`.\n\nValidation before wiring/live use:\n\n```bash\ndocker image inspect \"$auth_image\" --format '{{json .Config.Entrypoint}}'\ndocker run -d --name \"$name\" -p 127.0.0.1::22 -e \"ORCA_SSH_PUBLIC_KEY=$pubkey\" \"$auth_image\"\ndocker ps -a --filter \"name=$name\"\ndocker logs \"$name\"\nssh -i \"$key\" -p \"$port\" -o IdentitiesOnly=yes user@127.0.0.1 'codex --version'\n```\n\nIf the container exits immediately, inspect logs before the cleanup trap removes it; a committed\ninteractive image with `ENTRYPOINT [\"bash\"]` is a common cause.\n\nAlso confirm the **host key is stable** across containers: the SSH `ssh -i … 127.0.0.1` dial should not\ntrigger a host-key-changed warning when a second container reuses the port. If it does, the host keys\nweren't baked into the base image (see the `ssh-keygen -A` point above).\n\n### 7i. Windows local-side scripts\n\nThe local-side scripts run on the user's desktop. On **Windows**, a bare `.sh` won't execute. Either\nrequire WSL/Git-Bash (and point `orca.yaml` at e.g. `bash ./scripts/orca-vm/.sh` via a `.cmd`\nlauncher), or scaffold PowerShell equivalents. Minimal PowerShell shape:\n\n```powershell\n#requires -Version 5\n$ErrorActionPreference = 'Stop'\n# resolve env→state→fallback; run the provider CLI / ssh the same way;\n# capture provider output; build the result object for the chosen mode and write ONE line of JSON to stdout.\n# Orca-server mode: @{ schemaVersion=1; pairingCode=$pairingCode; projectRoot=$projectRoot; userData=@{...} }\n# SSH mode: @{ schemaVersion=1; connection=@{ type=\"ssh\"; projectRoot=$projectRoot;\n# target=@{ label=$label; host=$host; port=$port; username=$user } } } (see §7g/§7h)\n($result | ConvertTo-Json -Compress -Depth 6)\n# progress/errors → Write-Error / the error stream, never stdout.\n```\n\nThe remote-side commands you run _inside_ the Linux VM stay bash regardless of the desktop OS.\n\n---\n\n## 8. Per-workspace recipe contract (the fast path)\n\nOnce the authenticated snapshot exists, this runs on every workspace create. Define recipes in\n`orca.yaml`:\n\n```yaml\nenvironmentRecipes:\n - id: cloud-sandbox\n name: Cloud Sandbox\n create: ./scripts/orca-vm/cloud-sandbox-create.sh\n suspend: ./scripts/orca-vm/cloud-sandbox-suspend.sh\n resume: ./scripts/orca-vm/cloud-sandbox-resume.sh\n destroy: ./scripts/orca-vm/cloud-sandbox-destroy.sh\n```\n\n`create` runs **locally from the repo root** and prints **one** JSON object to stdout. Its shape depends\non the connection mode chosen in §1:\n\n**Orca-server mode** — boot the env, start `orca serve` in it, and print serve's result:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"orca-pairing-code-or-url\",\n \"projectRoot\": \"/absolute/path/to/repo/on/remote\",\n \"userData\": { \"provider\": \"example\", \"resourceId\": \"provider-resource-id\" }\n}\n```\n\nHere `pairingCode` (from `orca serve --recipe-json`) and `projectRoot` are required; `schemaVersion` (`1`)\nand `userData` are optional.\n\n**SSH mode** — do **not** run `orca serve`; print the `connection.type:\"ssh\"` block instead (full shape +\nworked script in §7g). `pairingCode` is **not** used in SSH mode.\n\n**Optional provisioned root** — only for direct SSH and only when explicitly requested. Add\n`checkoutMode: provisioned-root` to the recipe, require `ORCA_RECIPE_RESULT_SCHEMA_VERSION=2`, create\nthe requested `ORCA_REPO_BRANCH` at the pinned `ORCA_REPO_REF_HEAD` commit (use `ORCA_REPO_REF` only\nto fetch that commit) at the returned `projectRoot`, and emit schema version 2 with\n`checkoutMode: \"provisioned-root\"`. All recipes without this field retain the schema-v1 behavior above.\n\nLifecycle hooks (all run locally):\n\n- `create`: required. Prints recipe result JSON.\n- `suspend`: optional. Sleep; reads lifecycle payload on stdin.\n- `resume`: optional. Wake; reads payload on stdin and **prints fresh recipe JSON** (pairing may change).\n- `destroy`: optional unless `destroy: none`. Delete/cleanup; reads payload on stdin.\n\nStart Orca remotely with `orca serve --port \"$PORT\" --project-root \"$ABS_ROOT\" --pairing-address\n\"$EXTERNAL_WSS_URL\" --recipe-json` (exact flags + output in §7c). Set `--pairing-address` to the\nexternally reachable address so the emitted `pairingCode` is reachable; tunneling/port mapping is the\nscript's job.\n\nBackward compatibility: `command`→`create`, `cleanup`→`destroy`, `cleanup: none`→`destroy: none`.\nPrefer the lifecycle names.\n\n---\n\n## 9. Doctor and validation\n\nValidate in two stages — the cheap dry run first, then the live self-test.\n\n### Dry run (free, non-destructive) — always do this first\n\n`orca vm recipe doctor --repo-path --json` validates **static wiring only** — it does\n**not** boot anything. It checks: local-host execution (v1), repo path, recipe id exists,\ncreate/destroy/suspend/resume command paths resolve, suspend/resume are paired, and each script is\nexecutable (POSIX exec bit; skipped on Windows). Fix every failure here before spending any cloud money.\n\n### Live self-test (`--provision`) — diagnose and iterate yourself\n\n`orca vm recipe doctor --repo-path --provision --json` actually runs the recipe end\nto end: it executes `create`, validates the returned recipe JSON, then runs `destroy` to **tear the\nenvironment back down** (so the test leaves nothing running, as long as `destroy` works). It spends real\ncloud money, so get the user's OK **once** before starting — that one approval covers the whole loop\nbelow; do not re-ask before each run.\n\nOn failure, the JSON result includes a `provisionTranscript` with the **complete** captured output of\neach stage so you can self-diagnose without asking the user to relay logs:\n\n```json\n{\n \"ok\": false,\n \"checks\": [{ \"id\": \"recipe.provision\", \"status\": \"fail\", \"message\": \"…\" }],\n \"provisionTranscript\": {\n \"provision\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\", \"parseError\": \"…\" },\n \"destroy\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\" }\n }\n}\n```\n\n**Run it as a loop:** read `provisionTranscript.provision.stderr` / `.stdout` / `.parseError` (and\n`destroy.*`), fix the script, and re-run `--provision` until `ok` is `true` — iterating on your own\nrather than waiting for the user to paste errors. Common reads: a non-empty `stderr` with `exitCode 0`\nplus a `parseError` means `create` ran but printed something other than the single recipe-result JSON on\nstdout (often a stray `echo` — route it to stderr, see §10); a non-zero `exitCode` is a provider/script\nfailure described in `stderr`. Each stream is redacted and capped (head+tail) — large logs keep both the\nsetup context and the failure.\n\nThe self-test cannot see provider-side truth beyond what the scripts print, so still confirm: state has a\npopulated **authenticated** `snapshotId` (Phases 2–3 done), and `destroy` is implemented/tested (or\nexplicitly `none` — in which case the self-test won't tear down, so clean up manually).\n\nFor SSH recipes, also smoke-test the exact emitted target before declaring success: dial the host/port\nwith the identity/proxy settings, run `pwd`, verify the repo path, check the agent binary, and confirm\n`destroy` removes the provider resource/container. For Docker, inspect the auth image entrypoint and do a\nstartup-only `docker run` before the full clone/install path.\n\n---\n\n## 10. Failure modes\n\n- **Build exceeds plan timeout (e.g. Hobby 45m).** Use enough vCPUs and a timeout covering the build;\n else split work or use a higher plan. The cap also limits per-workspace runtime — surface it.\n- **Build exceeds plan RAM.** Build the **headless main only** (drop the renderer) — the biggest fitter.\n- **Private-repo clone hangs/fails.** Wrong/missing token. Use `GIT_ASKPASS` + `GIT_TERMINAL_PROMPT=0`\n so it fails fast instead of prompting.\n- **`GIT_ASKPASS` helper aborts the clone with \"`$1: unbound variable`\".** The `printf`/heredoc that writes\n the helper inside `bash -lc` under `set -u` expanded `$1`/`$GH_TOKEN` at **write** time. Escape them\n (`\\$1`, `\\$GH_TOKEN`) so they land literally and resolve at git-runtime; this also keeps the real token\n out of the file. `rm -f` the helper afterward (§5, §7f).\n- **Agent verified as \"not logged in\" despite a good login.** `codex login status` (and similar) print\n \"Logged in …\" to **stderr**; an stdout-only `grep` misses it. Prefer the status **exit code**; if you\n grep, fold stderr first (`status 2>&1 | grep …`) and match the exact success line — not `grep -qi\n'logged in'`, which also matches \"not logged in\".\n- **Headless agent login hangs.** Plain OAuth `login` starts a loopback callback server on a VM/container\n port the host browser can't reach. Use the **device-auth** flow (`login --device-auth`) — it prints a\n URL + code the user opens on the host.\n- **`known_hosts` host-key churn on local Docker.** Each ephemeral container regenerating its SSH host key\n collides on `127.0.0.1` as the published port rotates. Bake host keys into the base image at build time\n (`ssh-keygen -A`; runtime generates only if absent) so all containers share one stable key (§7h).\n- **Snapshot expired/evicted.** If `create` hits an unknown snapshot id, rerun Phases 2–3 and update\n `snapshotId`.\n- **Agent auth didn't persist.** Confirm `snapshotId` points at the **authenticated** snapshot; re-run\n Phase 3. Warn that short-lived tokens may need periodic re-auth.\n- **Agent auth copied from the host breaks.** Do not bind-mount/copy a full host agent home; sqlite\n files can be unwritable or host-specific, hooks may need approval again, and config may reference\n local-only env vars. Authenticate inside the runtime and snapshot/commit that layer.\n- **Docker auth image exits immediately.** Inspect `docker image inspect … .Config.Entrypoint` and\n `docker logs`. If the image was committed from an interactive shell, reset the entrypoint to the SSH\n entrypoint during `docker commit`.\n- **Leaked paid resource.** Every long script must trap errors and remove the sandbox it created.\n- **`create` emits non-JSON on stdout.** A stray `echo` corrupts the result — stdout is for the final\n JSON only; everything else to stderr. The `--provision` self-test surfaces this as `exitCode 0` + a\n `parseError` with the offending stdout in `provisionTranscript` (§9).\n\n---\n\n## 11. Boundaries\n\n- Don't create accounts, choose plans/regions, or invent scope/project/org/image/billing ids.\n- Don't invent or store credentials; no secrets in `userData`, state, comments, docs, or commits.\n- Don't run paid/long phases (base snapshot, auth, live test) without an explicit OK.\n- Don't hide provider errors behind generic messages — preserve actionable stderr.\n- Don't make Orca own provider lifecycle beyond invoking the configured scripts.\n- Don't commit or create an Orca workspace unless asked.\n" // oxfmt-ignore -const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, coordinator loops, or decomposing work\n across agents. Use `orca-cli` instead for full ownership handoffs, including\n requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", or \"another worktree\" when the user did not explicitly ask to\n supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for\n terminal control, lightweight terminal prompts, shell commands, Orca\n worktree management, reading or waiting on terminals, and automation of the\n browser embedded inside Orca. Use Computer Use for external browser windows,\n webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when\n the task requires OS/window-level control such as focus, menus, dialogs,\n coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a\n page-automation tool such as Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" +const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli`\n instead for full ownership handoffs, including requests phrased as \"hand\n off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another\n worktree\" when the user did not explicitly ask to supervise, monitor, wait\n for results, or coordinate a DAG. Use `orca-cli` for terminal control,\n lightweight terminal prompts, shell commands, Orca worktree management,\n reading or waiting on terminals, and the Orca embedded browser. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI\n outside Orca's embedded browser only when the task requires OS/window-level\n control such as focus, menus, dialogs, coordinates, or screenshots. Use\n `orca-cli` for Orca's embedded pages and a page-automation tool such as\n Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" // Why: no current guide has bundled reference documents, so --full is byte-identical for now. // oxfmt-ignore @@ -86,7 +86,7 @@ export const BUNDLED_SKILL_GUIDES = [ }, { name: "orchestration", - description: "Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation waits, task DAGs, decision gates, coordinator loops, or decomposing work across agents. Use `orca-cli` instead for full ownership handoffs, including requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another worktree\" when the user did not explicitly ask to supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for terminal control, lightweight terminal prompts, shell commands, Orca worktree management, reading or waiting on terminals, and automation of the browser embedded inside Orca. Use Computer Use for external browser windows, webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when the task requires OS/window-level control such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages.", + description: "Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli` instead for full ownership handoffs, including requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another worktree\" when the user did not explicitly ask to supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for terminal control, lightweight terminal prompts, shell commands, Orca worktree management, reading or waiting on terminals, and the Orca embedded browser. Use Computer Use for external browser windows, webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when the task requires OS/window-level control such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages.", markdown: ORCHESTRATION_MARKDOWN, fullMarkdown: ORCHESTRATION_MARKDOWN, aliases: [] diff --git a/src/main/agent-awake-service-platform-assertions.test.ts b/src/main/agent-awake-service-platform-assertions.test.ts index cd1d7fb1adc..7b3566b322f 100644 --- a/src/main/agent-awake-service-platform-assertions.test.ts +++ b/src/main/agent-awake-service-platform-assertions.test.ts @@ -22,6 +22,20 @@ function workingStatus(): AgentAwakeStatus { } } +describe('AgentAwakeService status array ownership', () => { + it('does not observe rows appended to the caller array after setStatuses', () => { + const service = new AgentAwakeService() + service.setMode('auto') + const statuses: AgentAwakeStatus[] = [workingStatus()] + + service.setStatuses(statuses) + const before = service.getWorkingAgentCount() + statuses.push(workingStatus(), workingStatus()) + + expect(service.getWorkingAgentCount()).toBe(before) + }) +}) + function createBlocker() { const startedIds = new Set() let nextId = 1 diff --git a/src/main/agent-awake-service.ts b/src/main/agent-awake-service.ts index b79612e2b9c..6be27e9d0e6 100644 --- a/src/main/agent-awake-service.ts +++ b/src/main/agent-awake-service.ts @@ -105,7 +105,8 @@ export class AgentAwakeService { } setStatuses(statuses: AgentAwakeStatus[]): void { - this.statuses = statuses.map((status) => ({ ...status })) + // Copy the array, not every row: the hook server allocates each row fresh per event. + this.statuses = [...statuses] this.refresh('status-change') } @@ -171,7 +172,8 @@ export class AgentAwakeService { private getEligibleRunningStatusCount(): number { const now = this.now() - return this.statuses.filter((status) => this.isWakeEligible(status, now)).length + // Counted in place: the filtered array was only ever measured, and this runs per hook event. + return this.statuses.reduce((count, s) => count + (this.isWakeEligible(s, now) ? 1 : 0), 0) } private isWakeEligible(status: AgentAwakeStatus, now: number): boolean { diff --git a/src/main/claude-usage/transcript-record-parser-prefilter.test.ts b/src/main/claude-usage/transcript-record-parser-prefilter.test.ts new file mode 100644 index 00000000000..992b4ba8fd3 --- /dev/null +++ b/src/main/claude-usage/transcript-record-parser-prefilter.test.ts @@ -0,0 +1,40 @@ +import { describe, expect, it } from 'vitest' +import { parseClaudeUsageRecord } from './transcript-record-parser' + +function assistantLine(overrides: Record = {}): string { + return JSON.stringify({ + type: 'assistant', + sessionId: 'session-1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { usage: { input_tokens: 3, output_tokens: 5 } }, + ...overrides + }) +} + +describe('assistant-record prefilter', () => { + it('still parses an ordinary assistant record', () => { + expect(parseClaudeUsageRecord(assistantLine())?.inputTokens).toBe(3) + }) + + it('rejects a user record that never mentions assistant', () => { + const userLine = JSON.stringify({ + type: 'user', + sessionId: 'session-1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { content: 'x'.repeat(200) } + }) + + expect(parseClaudeUsageRecord(userLine)).toBeNull() + }) + + it('rejects a non-assistant record that happens to contain the word assistant', () => { + const userLine = JSON.stringify({ + type: 'user', + sessionId: 'session-1', + timestamp: '2026-01-01T00:00:00.000Z', + message: { content: 'ask the assistant about this' } + }) + + expect(parseClaudeUsageRecord(userLine)).toBeNull() + }) +}) diff --git a/src/main/claude-usage/transcript-record-parser.ts b/src/main/claude-usage/transcript-record-parser.ts index 59b0e75be39..2ad51f3ee90 100644 --- a/src/main/claude-usage/transcript-record-parser.ts +++ b/src/main/claude-usage/transcript-record-parser.ts @@ -84,10 +84,30 @@ function dedupeClaudeUsageTurns( return deduped } +/** + * Necessary condition for `JSON.parse(line).type === 'assistant'`, checked before the parse. + * + * Sound for any transcript written by a standard JSON serializer: `JSON.stringify` (which writes + * these files) escapes only quotes, backslashes and control characters, never ASCII letters, so + * the decoded value can only be `assistant` if the line spells it literally. The gate over-admits + * freely — the `parsed.type` check below stays authoritative. + * + * A `\u`-escape fallback was measured and rejected: it costs a second full-line scan and made + * transcripts whose tool results contain control characters 1.43x slower overall. + */ +function mayEncodeAssistantType(line: string): boolean { + return line.includes('assistant') +} + function parseClaudeUsageSourceRecord( line: string, fallbackSessionId: string | null = null ): ClaudeUsageParsedSourceTurn | null { + // Only assistant records carry usage, but transcripts interleave user/tool-result lines that + // routinely embed whole files. Reject those before paying for a full parse. + if (!mayEncodeAssistantType(line)) { + return null + } let parsed: ClaudeUsageSourceRecord try { parsed = JSON.parse(line) as ClaudeUsageSourceRecord diff --git a/src/main/codex/codex-app-server-process-teardown.test.ts b/src/main/codex/codex-app-server-process-teardown.test.ts index 1ddb672a531..cec8f91d081 100644 --- a/src/main/codex/codex-app-server-process-teardown.test.ts +++ b/src/main/codex/codex-app-server-process-teardown.test.ts @@ -1,7 +1,14 @@ import type { ChildProcess } from 'node:child_process' -import { describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + findSelfInitiatedTreeKills, + resetSelfInitiatedTreeKillLogForTest +} from '../crash-reporting/self-initiated-tree-kill-log' import { terminateCodexAppServerProcessTree } from './codex-app-server-process-teardown' +/** Above pid_max on every supported POSIX host, so the group signal is a real ESRCH. */ +const UNREACHABLE_PGID = 2_147_483_647 + function child() { return { pid: 1234, @@ -10,6 +17,10 @@ function child() { } describe('terminateCodexAppServerProcessTree', () => { + beforeEach(() => { + resetSelfInitiatedTreeKillLogForTest() + }) + it('waits for the Windows tree kill before releasing the wrapper', async () => { const target = child() const release = Promise.withResolvers() @@ -120,6 +131,55 @@ describe('terminateCodexAppServerProcessTree', () => { expect(target.kill).not.toHaveBeenCalled() }) + /** + * `selfInitiatedTreeKillCount` decides whether a `render-process-gone` was + * ours. A group that had already exited was killed by nobody, so crediting it + * puts a suspect in the five-second window that Orca never issued. Exercised + * through the real `process.kill(-pgid)` because the swallow being tested + * lives in the production default, not in an injectable seam. + */ + it('does not claim a snapshot group that was already gone', async () => { + const target = { pid: UNREACHABLE_PGID, kill: vi.fn(() => true) as ChildProcess['kill'] } + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ + rootPgid: UNREACHABLE_PGID, + descendants: [], + capturedAtMs: 1 + }), + terminateDescendants: async () => true + }) + ).resolves.toBe(true) + + expect(target.kill).toHaveBeenLastCalledWith('SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([]) + }) + + it('claims a snapshot group the signal actually reached', async () => { + const target = child() + const signalProcessGroup = vi.fn() + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ rootPgid: 1234, descendants: [], capturedAtMs: 1 }), + terminateDescendants: async () => true, + signalProcessGroup + }) + ).resolves.toBe(true) + + expect(signalProcessGroup).toHaveBeenCalledWith(1234, 'SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([ + expect.objectContaining({ + pid: 1234, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' + }) + ]) + }) + it('tears down 40 dedicated groups without process-table scans or cross-group fanout', async () => { const killMocks = Array.from({ length: 40 }, () => vi.fn(() => true)) const targets = killMocks.map((kill, index) => ({ diff --git a/src/main/codex/codex-app-server-process-teardown.ts b/src/main/codex/codex-app-server-process-teardown.ts index a35ad4d3164..5a9c6e3574b 100644 --- a/src/main/codex/codex-app-server-process-teardown.ts +++ b/src/main/codex/codex-app-server-process-teardown.ts @@ -128,19 +128,24 @@ async function terminatePosixTree( if (descendantsExited && snapshot.rootPgid === rootPid) { const signalGroup = deps.signalProcessGroup ?? - ((pgid: number, signal: NodeJS.Signals) => { - try { - process.kill(-pgid, signal) - } catch { - // Group already exited. - } + ((pgid: number, signal: NodeJS.Signals) => process.kill(-pgid, signal)) + let groupSignalled = false + try { + signalGroup(snapshot.rootPgid, 'SIGKILL') + groupSignalled = true + } catch { + // Already-gone is still the desired outcome, but nothing here killed it, + // and a crumb for a kill we never landed is a false render-process-gone suspect. + } + if (groupSignalled) { + // Outside the try, as in terminateDedicatedPosixGroup: that catch is the + // already-gone contract, not a breadcrumb handler. + recordSelfInitiatedTreeKill({ + pid: snapshot.rootPgid, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' }) - signalGroup(snapshot.rootPgid, 'SIGKILL') - recordSelfInitiatedTreeKill({ - pid: snapshot.rootPgid, - site: 'codex-app-server-teardown', - scope: 'posix-process-group' - }) + } } if (!descendantsExited) { child.kill('SIGCONT') diff --git a/src/main/codex/codex-prompt-registry-bounds.ts b/src/main/codex/codex-prompt-registry-bounds.ts index 84b3cda6151..e5f79fe3a31 100644 --- a/src/main/codex/codex-prompt-registry-bounds.ts +++ b/src/main/codex/codex-prompt-registry-bounds.ts @@ -20,10 +20,7 @@ export function codexJournalPromptIdPart(value: string): string { } const suffix = `#${digestPayload(value).slice(0, 32)}` const bounded = boundPayload(value, { - inlineHeadBytes: CODEX_JOURNAL_PROMPT_ID_COMPONENT_MAX_BYTES - suffix.length, - maxSessionBytes: Number.MAX_SAFE_INTEGER, - maxAppendsPerWindow: Number.MAX_SAFE_INTEGER, - appendWindowMs: Number.MAX_SAFE_INTEGER + inlineHeadBytes: CODEX_JOURNAL_PROMPT_ID_COMPONENT_MAX_BYTES - suffix.length }) return `${bounded.head}${suffix}` } diff --git a/src/main/codex/codex-structured-item-stream-bounds.ts b/src/main/codex/codex-structured-item-stream-bounds.ts index e84d8dd2efa..572c954b010 100644 --- a/src/main/codex/codex-structured-item-stream-bounds.ts +++ b/src/main/codex/codex-structured-item-stream-bounds.ts @@ -17,14 +17,8 @@ export function codexStructuredItemKey(threadId: string, itemId: string): string return `${key.slice(0, 960)}:${(hash >>> 0).toString(16)}` } -export function pendingPatchBytes(pending: { - body: unknown - blobs: readonly { payload: string }[] -}): number { - return ( - Buffer.byteLength(JSON.stringify(pending.body), 'utf8') + - pending.blobs.reduce((total, blob) => total + Buffer.byteLength(blob.payload, 'utf8'), 0) - ) +export function pendingPatchBytes(pending: { body: unknown }): number { + return Buffer.byteLength(JSON.stringify(pending.body), 'utf8') } export function boundStreamItem(item: Record): Record { diff --git a/src/main/codex/codex-structured-item-stream-contracts.ts b/src/main/codex/codex-structured-item-stream-contracts.ts index cf8aff5795d..f252d048210 100644 --- a/src/main/codex/codex-structured-item-stream-contracts.ts +++ b/src/main/codex/codex-structured-item-stream-contracts.ts @@ -24,7 +24,6 @@ export type CodexItemStreamState = { export type CodexPendingItemPatch = { identity: AgentJournalItemIdentity body: NonNullable['body']> - blobs: ReturnType['blobs'] } export type CodexStructuredItemStreamAdmission = diff --git a/src/main/codex/codex-structured-item-streams.ts b/src/main/codex/codex-structured-item-streams.ts index dda5e9688bd..a765f8339da 100644 --- a/src/main/codex/codex-structured-item-streams.ts +++ b/src/main/codex/codex-structured-item-streams.ts @@ -103,8 +103,8 @@ export function createCodexStructuredItemStreams( } const options = { coalescingKey: `checkpoint:${agentJournalItemKey(state.identity)}` } const admission = deps.sink.tryAppendItem - ? deps.sink.tryAppendItem(state.identity, translated.body, translated.blobs, options) - : (deps.sink.appendItem(state.identity, translated.body, translated.blobs, options), + ? deps.sink.tryAppendItem(state.identity, translated.body, options) + : (deps.sink.appendItem(state.identity, translated.body, options), { accepted: true as const }) if (!admission.accepted) { return false @@ -167,9 +167,8 @@ export function createCodexStructuredItemStreams( } for (const [key, pending] of pendingPatches) { const admission = deps.sink.tryAppendItem - ? deps.sink.tryAppendItem(pending.identity, pending.body, pending.blobs) - : (deps.sink.appendItem(pending.identity, pending.body, pending.blobs), - { accepted: true as const }) + ? deps.sink.tryAppendItem(pending.identity, pending.body) + : (deps.sink.appendItem(pending.identity, pending.body), { accepted: true as const }) if (!admission.accepted) { flushed = false continue @@ -193,9 +192,8 @@ export function createCodexStructuredItemStreams( return { accepted: true } } const admission = deps.sink.tryAppendItem - ? deps.sink.tryAppendItem(pending.identity, pending.body, pending.blobs) - : (deps.sink.appendItem(pending.identity, pending.body, pending.blobs), - { accepted: true as const }) + ? deps.sink.tryAppendItem(pending.identity, pending.body) + : (deps.sink.appendItem(pending.identity, pending.body), { accepted: true as const }) if (!admission.accepted) { return admission } @@ -237,8 +235,7 @@ export function createCodexStructuredItemStreams( if (translated.body) { const nextPending: CodexPendingItemPatch = { identity: state.identity, - body: translated.body, - blobs: translated.blobs + body: translated.body } const previous = pendingPatches.get(key) const previousBytes = previous ? pendingPatchBytes(previous) : 0 diff --git a/src/main/codex/codex-structured-item-translation.test.ts b/src/main/codex/codex-structured-item-translation.test.ts index 63c3d59b0b9..f0f843e4ed6 100644 --- a/src/main/codex/codex-structured-item-translation.test.ts +++ b/src/main/codex/codex-structured-item-translation.test.ts @@ -220,12 +220,6 @@ describe('codex item bodies', () => { throw new Error('expected bounded command output') } expect(body.output.head.length).toBeLessThan(20_000) - expect(translated.blobs).toEqual([ - { - digest: body.output.digest, - payload: output - } - ]) }) it('continues to accept camel-case command completion output', () => { diff --git a/src/main/codex/codex-structured-item-translation.ts b/src/main/codex/codex-structured-item-translation.ts index f640ad5fb3d..19052dca365 100644 --- a/src/main/codex/codex-structured-item-translation.ts +++ b/src/main/codex/codex-structured-item-translation.ts @@ -172,7 +172,6 @@ function commandState(item: CodexThreadItem): 'running' | 'completed' | 'failed' export type CodexJournalItem = { body: AgentJournalItemBody | null - blobs: { digest: string; payload: string }[] handled: boolean } @@ -190,10 +189,6 @@ function commandItem(item: CodexThreadItem): CodexJournalItem { state: commandState(item), ...(bounded === null ? {} : { output: bounded.bounded }) }, - blobs: - output !== null && bounded?.bounded.truncated - ? [{ digest: bounded.bounded.digest, payload: output }] - : [], handled: true } } @@ -215,7 +210,6 @@ function fileChangeItem(item: CodexThreadItem): CodexJournalItem { input: boundToolInput({ changes: item.changes ?? null }, DEFAULT_JOURNAL_PAYLOAD_LIMITS), state: commandState(item) }, - blobs: [], handled: true } } @@ -227,7 +221,6 @@ function fileChangeItem(item: CodexThreadItem): CodexJournalItem { path: changes.length === 1 ? changes[0]!.path : `${changes.length} files`, patch: bounded }, - blobs: bounded.truncated ? [{ digest: bounded.digest, payload: patch }] : [], handled: true } } @@ -246,7 +239,6 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem { blocks.length === 0 ? null : { kind: 'message', role: item.type === 'userMessage' ? 'user' : 'assistant', blocks }, - blobs: [], handled: true } } @@ -266,14 +258,11 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem { text === null ? null : { kind: 'status', text: boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text }, - blobs: [], handled: true } } const unhandled = unhandledProviderFrameJournalItem('codex', `item:${item.type}`, item) - return unhandled - ? { body: unhandled.body, blobs: unhandled.blobs, handled: false } - : { body: null, blobs: [], handled: true } + return unhandled ? { body: unhandled.body, handled: false } : { body: null, handled: true } } export function codexItemBody(item: CodexThreadItem): AgentJournalItemBody | null { @@ -292,7 +281,7 @@ export function codexStreamingMessageBody(text: string): AgentJournalItemBody { /** Snapshot body for any item-level stream, keyed onto its parent item. */ export function codexStreamingJournalItem(item: CodexThreadItem, text: string): CodexJournalItem { if (item.type === 'agentMessage') { - return { body: codexStreamingMessageBody(text), blobs: [], handled: true } + return { body: codexStreamingMessageBody(text), handled: true } } if (item.type === 'commandExecution') { return commandItem({ ...item, aggregatedOutput: text }) @@ -304,10 +293,9 @@ export function codexStreamingJournalItem(item: CodexThreadItem, text: string): const bounded = boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).bounded return { body: { kind: 'diff', path: path ?? 'pending patch', patch: bounded }, - blobs: bounded.truncated ? [{ digest: bounded.digest, payload: text }] : [], handled: true } } const bounded = boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS) - return { body: { kind: 'status', text: bounded.text }, blobs: [], handled: true } + return { body: { kind: 'status', text: bounded.text }, handled: true } } diff --git a/src/main/codex/codex-structured-journal-generic-frames.ts b/src/main/codex/codex-structured-journal-generic-frames.ts index f6b9ff375d3..6c211a8f5a4 100644 --- a/src/main/codex/codex-structured-journal-generic-frames.ts +++ b/src/main/codex/codex-structured-journal-generic-frames.ts @@ -91,13 +91,11 @@ export class CodexJournalGenericFrames { const admission = this.deps.sink.tryAppendItem ? this.deps.sink.tryAppendItem( { provider: 'orca', clientMessageId: `provider-frame:codex:${this.fallbackSequence}` }, - translated.body, - translated.blobs + translated.body ) : (this.deps.sink.appendItem( { provider: 'orca', clientMessageId: `provider-frame:codex:${this.fallbackSequence}` }, - translated.body, - translated.blobs + translated.body ), CODEX_JOURNAL_ADMITTED) if (!admission.accepted) { @@ -136,13 +134,11 @@ export class CodexJournalGenericFrames { kind: 'status', text }, - [], { coalescingKey: `provider-frame-suppressed:codex:${bucket}` } ) : (this.deps.sink.appendItem( { provider: 'orca', clientMessageId: `provider-frame-suppressed:codex:${bucket}` }, { kind: 'status', text }, - [], { coalescingKey: `provider-frame-suppressed:codex:${bucket}` } ), CODEX_JOURNAL_ADMITTED) diff --git a/src/main/codex/codex-structured-journal-items.ts b/src/main/codex/codex-structured-journal-items.ts index 2b5d8c04bba..0e4e3f5a900 100644 --- a/src/main/codex/codex-structured-journal-items.ts +++ b/src/main/codex/codex-structured-journal-items.ts @@ -2,7 +2,7 @@ import type { AgentJournalItemBody, AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' -import { requiresTerminalSettlement } from '../native-chat/agent-session-journal/journal-lifecycle-capacity' +import { requiresTerminalSettlement } from '../native-chat/agent-session-journal/journal-terminal-settlement' import { codexItemIdentity, codexJournalItem, @@ -126,19 +126,13 @@ export class CodexJournalItems { return CODEX_JOURNAL_ADMITTED } if (method === 'item/completed') { - const admission = appendCodexLifecycleItem( - this.deps.sink, - identity, - translated.body, - translated.blobs - ) + const admission = appendCodexLifecycleItem(this.deps.sink, identity, translated.body) return admission.accepted ? publishCodexLifecycle(this.deps.sink) : admission } const options = requiresTerminalSettlement(translated.body) ? { lifecycle: true } : {} const admission = this.deps.sink.tryAppendItem - ? this.deps.sink.tryAppendItem(identity, translated.body, translated.blobs, options) - : (this.deps.sink.appendItem(identity, translated.body, translated.blobs), - CODEX_JOURNAL_ADMITTED) + ? this.deps.sink.tryAppendItem(identity, translated.body, options) + : (this.deps.sink.appendItem(identity, translated.body), CODEX_JOURNAL_ADMITTED) if (!admission.accepted) { return admission } diff --git a/src/main/codex/codex-structured-journal-settlement.ts b/src/main/codex/codex-structured-journal-settlement.ts index f4378d1a16f..2b8eb627d4f 100644 --- a/src/main/codex/codex-structured-journal-settlement.ts +++ b/src/main/codex/codex-structured-journal-settlement.ts @@ -243,14 +243,12 @@ function appendLifecycleMutations( for (const mutation of chunk) { if (mutation.kind === 'item') { if (sink.tryAppendItem) { - admission = sink.tryAppendItem(mutation.identity, mutation.body, [], { - lifecycle: true - }) + admission = sink.tryAppendItem(mutation.identity, mutation.body, { lifecycle: true }) if (!admission.accepted) { return admission } } else { - sink.appendItem(mutation.identity, mutation.body, [], { lifecycle: true }) + sink.appendItem(mutation.identity, mutation.body, { lifecycle: true }) } } else { if (sink.tryAppendTombstone) { diff --git a/src/main/codex/codex-structured-journal-sink.ts b/src/main/codex/codex-structured-journal-sink.ts index b135c89ec0a..5c4ecec9658 100644 --- a/src/main/codex/codex-structured-journal-sink.ts +++ b/src/main/codex/codex-structured-journal-sink.ts @@ -4,7 +4,6 @@ import type { } from '../../shared/agent-session-journal-types' import type { StructuredAgentSessionEventSink, - StructuredAgentSessionJournalBlob, StructuredAgentSessionSinkAdmission } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' import type { CodexPendingJournalPrompt } from './codex-structured-journal-settlement' @@ -20,13 +19,12 @@ function criticalAdmission( export function appendCodexLifecycleItem( sink: StructuredAgentSessionEventSink, identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly StructuredAgentSessionJournalBlob[] = [] + body: AgentJournalItemBody ): CodexJournalTranslationAdmission { if (sink.tryAppendItem) { - return criticalAdmission(sink.tryAppendItem(identity, body, blobs, { lifecycle: true })) + return criticalAdmission(sink.tryAppendItem(identity, body, { lifecycle: true })) } - sink.appendItem(identity, body, blobs, { lifecycle: true }) + sink.appendItem(identity, body, { lifecycle: true }) return CODEX_JOURNAL_ADMITTED } diff --git a/src/main/codex/codex-structured-journal-translation-streams.test.ts b/src/main/codex/codex-structured-journal-translation-streams.test.ts index c06d468a1ba..a9bc79b71c4 100644 --- a/src/main/codex/codex-structured-journal-translation-streams.test.ts +++ b/src/main/codex/codex-structured-journal-translation-streams.test.ts @@ -89,11 +89,11 @@ describe('codex journal translation', () => { const { translator, tap } = translatorWith() let rejectTerminal = true const appendItem = tap.sink.appendItem - tap.sink.tryAppendItem = (identity, body, blobs, options) => { + tap.sink.tryAppendItem = (identity, body, options) => { if (rejectTerminal && body.kind === 'tool-call' && body.state === 'failed') { return { accepted: false as const, reason: 'backpressure' as const } } - appendItem(identity, body, blobs, options) + appendItem(identity, body, options) return { accepted: true as const } } for (let index = 0; index <= 256; index += 1) { diff --git a/src/main/codex/codex-structured-journal-translation-turns.ts b/src/main/codex/codex-structured-journal-translation-turns.ts index 3f295d2add1..06bb28f85f9 100644 --- a/src/main/codex/codex-structured-journal-translation-turns.ts +++ b/src/main/codex/codex-structured-journal-translation-turns.ts @@ -40,7 +40,6 @@ export function publishCodexTurnLifecycle(input: { text: 'Codex is working…', turnLifecycle: { turnId: input.turnId, state: input.state } }, - [], { lifecycle: true } ) : (input.sink.appendItem( @@ -50,7 +49,6 @@ export function publishCodexTurnLifecycle(input: { text: 'Codex is working…', turnLifecycle: { turnId: input.turnId, state: input.state } }, - [], { lifecycle: true } ), ADMITTED) diff --git a/src/main/codex/codex-turn-ordinals.ts b/src/main/codex/codex-turn-ordinals.ts index 7087c28e4ac..e2b4132d2b2 100644 --- a/src/main/codex/codex-turn-ordinals.ts +++ b/src/main/codex/codex-turn-ordinals.ts @@ -37,10 +37,7 @@ export class CodexTurnOrdinals { const suffix = `#${digestPayload(value).slice(0, 24)}` return `${ boundPayload(encoded, { - inlineHeadBytes: 256 - Buffer.byteLength(suffix, 'utf8'), - maxSessionBytes: Number.MAX_SAFE_INTEGER, - maxAppendsPerWindow: Number.MAX_SAFE_INTEGER, - appendWindowMs: Number.MAX_SAFE_INTEGER + inlineHeadBytes: 256 - Buffer.byteLength(suffix, 'utf8') }).head }${suffix}` } diff --git a/src/main/crash-reporting/gone-time-system-memory.ts b/src/main/crash-reporting/gone-time-system-memory.ts deleted file mode 100644 index 7cca89d2d4b..00000000000 --- a/src/main/crash-reporting/gone-time-system-memory.ts +++ /dev/null @@ -1,77 +0,0 @@ -import type { CrashReportDetailValue } from '../../shared/crash-reporting' - -// ─── System memory at gone time ───────────────────────────────────── -// Why: the system outlives the crashed process, so this IS sampleable at -// process-gone — it separates "renderer grew huge" from "machine out of -// memory/commit", which the per-process buckets alone cannot. -// Timing honesty: this reads AFTER the crashed process's memory returned to -// the OS, so free/swapFree can look healthier than they were at kill time. -// Platform honesty: swap* exist on Windows/Linux only. On Linux `free` is -// /proc/meminfo MemFree and is NOT the pressure signal — it excludes page cache -// and other reclaimable memory; `available` (MemAvailable, Linux-only) is. On -// macOS `free` is near-meaningless (file cache and compression keep it low on -// healthy machines); fileBacked/purgeable are the only reclaimability proxy this -// API gives there, and none of these fields answers "was the machine under -// pressure" on macOS — that needs a signal Electron does not expose. - -type CrashReportDetails = Record - -export function memoryKBFieldMB(value: unknown): number | undefined { - const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined - return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) -} - -type SystemMemoryInfoLike = { - total?: unknown - free?: unknown - available?: unknown - swapTotal?: unknown - swapFree?: unknown - fileBacked?: unknown - purgeable?: unknown -} - -type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null - -function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { - const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) - .getSystemMemoryInfo - if (typeof read !== 'function') { - return null - } - try { - return read.call(process) - } catch { - return null - } -} - -let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo - -export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { - systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo -} - -export function getSystemMemoryAtGoneDetails(): CrashReportDetails { - const info = systemMemoryInfoReader() - if (!info) { - return {} - } - const details: CrashReportDetails = {} - const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ - ['total', 'systemMemoryTotalMB'], - ['free', 'systemMemoryFreeMB'], - ['available', 'systemMemoryAvailableMB'], - ['swapTotal', 'systemMemorySwapTotalMB'], - ['swapFree', 'systemMemorySwapFreeMB'], - ['fileBacked', 'systemMemoryFileBackedMB'], - ['purgeable', 'systemMemoryPurgeableMB'] - ] - for (const [field, key] of fields) { - const mb = memoryKBFieldMB(info[field]) - if (mb !== undefined) { - details[key] = mb - } - } - return details -} diff --git a/src/main/crash-reporting/pre-gone-host-memory.test.ts b/src/main/crash-reporting/pre-gone-host-memory.test.ts new file mode 100644 index 00000000000..0df13d4fee5 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.test.ts @@ -0,0 +1,379 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + getSystemMemoryDetails, + setSystemMemoryInfoReaderForTest, + withSwapVolumeFreeSpace +} from './system-memory-details' +import { + readSwapVolumeFreeSpace, + setSwapVolumeFreeSpaceReaderForTest, + type SwapVolumeFreeSpace +} from './swap-volume-free-space' +import { samplePreGoneSystemMemory } from './pre-gone-host-memory' +import { + buildProcessGoneCrashDetails, + resetPreGoneCrashSamplingForTest, + samplePreGoneProcessMetrics, + startPreGoneCrashSampling +} from './process-gone-diagnostics' + +type MetricFixture = { + pid: number + creationTime: number + type: string + memory: { workingSetSize: number; peakWorkingSetSize?: number; privateBytes?: number } +} + +const { appMetricsMock } = vi.hoisted(() => ({ + appMetricsMock: vi.fn<() => MetricFixture[]>(() => []) +})) + +vi.mock('electron', () => ({ app: { getAppMetrics: appMetricsMock } })) + +const BROWSER_AND_RENDERER: MetricFixture[] = [ + { pid: 10, creationTime: 1, type: 'Browser', memory: { workingSetSize: 1024 * 250 } }, + { + pid: 11, + creationTime: 2, + type: 'Tab', + memory: { workingSetSize: 1024 * 400, peakWorkingSetSize: 1024 * 420, privateBytes: 1024 * 260 } + } +] + +const BROWSER_ONLY: MetricFixture[] = [BROWSER_AND_RENDERER[0]] + +const UNDER_COMMIT_PRESSURE = { + total: 16_000 * 1024, + free: 400 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 200 * 1024 +} + +const AFTER_THE_CORPSE_RELEASED = { + total: 16_000 * 1024, + free: 3_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 2_900 * 1024 +} + +// Commit limit ~= RAM: a disabled or fixed pagefile, which no amount of empty +// disk can grow into. `swapTotal > total` is all this API can say about that. +const FIXED_PAGEFILE_UNDER_PRESSURE = { + total: 16_000 * 1024, + free: 300 * 1024, + swapTotal: 16_100 * 1024, + swapFree: 180 * 1024 +} + +const NO_PAGEFILE_UNDER_PRESSURE = { + ...FIXED_PAGEFILE_UNDER_PRESSURE, + swapTotal: 15_900 * 1024 +} + +const BEFORE_THE_STORM = { + total: 16_000 * 1024, + free: 9_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 30_000 * 1024 +} + +describe('pre-gone host memory', () => { + beforeEach(() => { + resetPreGoneCrashSamplingForTest() + setSystemMemoryInfoReaderForTest(null) + setSwapVolumeFreeSpaceReaderForTest(null) + appMetricsMock.mockClear() + appMetricsMock.mockReturnValue(BROWSER_AND_RENDERER) + }) + + it('carries a pre-gone host reading, not only the post-mortem one', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + // The renderer dies; its ~400 MB returns to the OS, so the gone-time read + // now shows a much healthier machine than the one that refused the alloc. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + appMetricsMock.mockReturnValue(BROWSER_ONLY) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemorySwapFreeMB).toBe(2_900) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(200) + expect(details.systemMemoryPreGoneFreeMB).toBe(400) + expect(details.systemMemoryPreGoneTotalMB).toBe(16_000) + // Why: host memory keeps its own key family, so a `systemMemory` prefix scan sees both reads. + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) + }) + + // Why this decides the cluster: 200 MB available commit is only a REFUSAL when + // the pagefile cannot grow, which is what the volume's free space says. + it('reports swap-volume free space so low commit can be told from refused commit', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(120) + // Which volume was measured: Windows only names the DEFAULT pagefile drive. + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + }) + + it('omits swap-volume free space on Linux, where swap cannot grow into free disk', async () => { + // Linux swap is a fixed partition, a fixed-size swapfile, or zram; reporting + // root-fs free space next to SwapFreeMB 0 would read as headroom that is not there. + setSwapVolumeFreeSpaceReaderForTest(null) + + await expect(readSwapVolumeFreeSpace('linux')).resolves.toBeUndefined() + }) + + it('labels the reading with the pressure verdict the platform can actually give', () => { + // Windows available commit is only a REFUSAL when the pagefile cannot grow, + // which nothing here proves, so no label may read as that verdict. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + const windowsCommit = getSystemMemoryDetails('win32') + expect(windowsCommit.systemMemoryPressureSignal).toBe('available-commit-unqualified') + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32') + .systemMemoryPressureSignal + ).toBe('available-commit-volume-cotimed') + // A volume number from a different moment describes a different machine. + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32', false) + .systemMemoryPressureSignal + ).toBe('available-commit-unqualified') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, free: 400 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('none') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, available: 900 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('mem-available') + + // darwin free/fileBacked/purgeable answer reclaimability, never pressure. + setSystemMemoryInfoReaderForTest(() => ({ + total: 16_000 * 1024, + free: 272 * 1024, + fileBacked: 2_694 * 1024, + purgeable: 0 + })) + expect(getSystemMemoryDetails('darwin').systemMemoryPressureSignal).toBe('none') + }) + + // Why this and not the volume number: the branch's own repro needed a pagefile + // that CANNOT grow to kill anything, and neither the pagefile maximum nor its + // drive is readable here — `swapVolumeAnchor` measures SystemRoot's volume, + // which a relocated pagefile does not live on. + it('never reads free disk as proof the pagefile could have grown', () => { + setSystemMemoryInfoReaderForTest(() => FIXED_PAGEFILE_UNDER_PRESSURE) + const fixedPagefile = withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ) + // 180 MB of commit beside 812 GB of free disk: co-timed, and still not a + // verdict — reading it as "the pagefile had room" is the opposite conclusion. + expect(fixedPagefile.systemMemoryPressureSignal).toBe('available-commit-volume-cotimed') + + // The one decisive win32 case: commit limit at or below RAM means there is + // no pagefile behind it, so the floor cannot heal however empty the disk is. + setSystemMemoryInfoReaderForTest(() => NO_PAGEFILE_UNDER_PRESSURE) + expect( + withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ).systemMemoryPressureSignal + ).toBe('available-commit-hard-capped') + }) + + // Why the verdict and not just the field: a statfs issued on a healthy host at + // t=0 that resolves 20 s into a commit storm prints "200 MB commit, 40 GB of + // pagefile headroom" — which reads as NOT a commit refusal, the opposite + // conclusion, under the branch's most confident label. + it('will not let a statfs that outlived its tick qualify the win32 commit verdict', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // The storm arrives; the in-flight latch makes every tick skip the merge, + // so the pending statfs is as old as the tick that STARTED it. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(20_000) + const stale = buildProcessGoneCrashDetails({}, 'renderer') + expect(stale.systemMemoryPreGoneSwapFreeMB).toBe(200) + // The pre-storm volume number still ships — but carrying its own age, and + // without promoting the verdict the analyst reads. + expect(stale.systemMemoryPreGoneSwapVolumeFreeMB).toBe(40_000) + expect(stale.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(stale.systemMemoryPreGoneSwapVolumeAgeMs).toBe(20_000) + expect(stale.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + + // The next tick's statfs answers on its own tick, so it qualifies again. + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 900, volume: 'C:' })) + await samplePreGoneSystemMemory(30_000) + vi.setSystemTime(30_000) + const fresh = buildProcessGoneCrashDetails({}, 'renderer') + expect(fresh.systemMemoryPreGoneSwapVolumeFreeMB).toBe(900) + expect(fresh.systemMemoryPreGoneSwapVolumeAgeMs).toBe(0) + expect(fresh.systemMemoryPreGonePressureSignal).toBe('available-commit-volume-cotimed') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + // Round 5: sample identity alone could not see these ticks. A host read that + // returns nothing leaves the sample object in place, so `sample === issuedFor` + // still held 25 s and two ticks later and the statfs re-qualified the verdict. + it('will not let ticks with a failed host read pass a stale statfs off as co-timed', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // GlobalMemoryStatusEx starts failing: the sample is neither replaced nor erased. + setSystemMemoryInfoReaderForTest(() => null) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(25_000) + const details = buildProcessGoneCrashDetails({}, 'renderer') + // 25 s of lag: the label must not say co-timed beside that age. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(25_000) + expect(details.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + it("arms the host sampler on its own unref'd 10 s timer, not the metric sweep's", async () => { + vi.useFakeTimers() + vi.setSystemTime(0) + const readHostMemory = vi.fn(() => UNDER_COMMIT_PRESSURE) + setSystemMemoryInfoReaderForTest(readHostMemory) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') + try { + startPreGoneCrashSampling() + + // Literal millisecond values: asserting the constants against themselves + // would let a cadence regression through, and 37 s of staleness is the bug. + expect(setIntervalSpy.mock.calls.map(([, ms]) => ms)).toEqual([60_000, 10_000]) + for (const { value } of setIntervalSpy.mock.results) { + expect((value as NodeJS.Timeout).hasRef()).toBe(false) + } + expect(readHostMemory).toHaveBeenCalledTimes(1) + + readHostMemory.mockReturnValue(AFTER_THE_CORPSE_RELEASED) + await vi.advanceTimersByTimeAsync(10_000) + // One host tick, no extra metric sweep: the two samplers run independently. + expect(readHostMemory).toHaveBeenCalledTimes(2) + expect(appMetricsMock).toHaveBeenCalledTimes(1) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(2_900) + } finally { + setIntervalSpy.mockRestore() + vi.useRealTimers() + } + }) + + it('commits the host reading without waiting on the swap-volume statfs', async () => { + // Why: statfs is slowest during the paging storm this sampler targets, and + // a hung volume must not stall or silently skip host sampling. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + + void samplePreGoneSystemMemory(Date.now() - 5_000) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(200) + + // A second tick still refreshes the reading while that statfs hangs. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + void samplePreGoneSystemMemory(Date.now()) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(2_900) + }) + + it('publishes no pre-gone host keys when every memory field failed to read', async () => { + // Why not "no keys at all": the reading always carries its signal label, so a + // committed empty one would ship an age and a volume number with no memory + // numbers beside them — a disk-free figure standing in for a host reading. + setSystemMemoryInfoReaderForTest(() => ({ total: Number.NaN, free: undefined })) + await samplePreGoneSystemMemory(Date.now()) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) + + it('carries the last volume reading forward, aged, instead of dropping it', async () => { + vi.useFakeTimers() + try { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 42, volume: 'C:' })) + await samplePreGoneSystemMemory(0) + + // The next tick's statfs hangs — during the paging storm this targets, that + // is the normal case — so the tick has no volume reading of its own, and + // the sample that replaces the last one would otherwise drop the field. + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + void samplePreGoneSystemMemory(10_000) + vi.setSystemTime(10_000) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(42) + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + // Carried, not re-read: it ships at its real age, never as a fresh number. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(10_000) + } finally { + vi.useRealTimers() + } + }) + + it('keeps a failed host read from erasing the process-metric sample', async () => { + samplePreGoneProcessMetrics(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(() => { + throw new Error('getSystemMemoryInfo unavailable') + }) + await samplePreGoneSystemMemory(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(null) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.processMetricsPreGoneRendererWorkingSetMB).toBe(400) + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) +}) diff --git a/src/main/crash-reporting/pre-gone-host-memory.ts b/src/main/crash-reporting/pre-gone-host-memory.ts new file mode 100644 index 00000000000..0db56796750 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.ts @@ -0,0 +1,164 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import { readSwapVolumeFreeSpace } from './swap-volume-free-space' +import { + getSystemMemoryDetails, + SYSTEM_MEMORY_KEY_PREFIX, + withSwapVolumeFreeSpace +} from './system-memory-details' + +// ─── Pre-gone host memory sampling ────────────────────────────────── +// Why sample at all: the gone-time host read lands after the corpse released +// its pages, so it reports a healthier machine than the one that refused the +// allocation. +// Why 10 s and not the 60 s process-metrics cadence: at 60 s, four of five +// field OOMs carried a ~37 s old host reading — far too stale to see a +// transient commit refusal. A refusal shorter than the interval stays +// invisible; no cadence fixes that. + +export const PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS = 10_000 + +type CrashReportDetails = Record + +type PreGoneSystemMemorySample = { + details: CrashReportDetails + sampledAtMs: number + /** Tick that ISSUED the statfs now merged in — never the tick it resolved on. */ + swapVolumeSampledAtMs?: number +} + +let preGoneSample: PreGoneSystemMemorySample | null = null +let preGoneTimer: ReturnType | null = null +let swapVolumeReadInFlight = false +let samplingGeneration = 0 +let sampleTick = 0 + +const PRESSURE_SIGNAL_KEY = `${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal` + +/** + * Carries the last volume reading onto the sample that replaces its own. + * + * Why: a statfs slower than one tick would otherwise make the field vanish from + * the reports it exists for — the next tick replaces the sample wholesale, and + * the in-flight latch keeps intervening ticks from merging anything. It ships + * with its own (now larger) age and, not being co-timed, never names the label. + */ +function withCarriedSwapVolume(sample: PreGoneSystemMemorySample): PreGoneSystemMemorySample { + const previous = preGoneSample + if (!previous || previous.swapVolumeSampledAtMs === undefined) { + return sample + } + const freeMB = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`] + const volume = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`] + if (typeof freeMB !== 'number' || typeof volume !== 'string') { + return sample + } + return { + ...sample, + details: withSwapVolumeFreeSpace(sample.details, { freeMB, volume }, process.platform, false), + swapVolumeSampledAtMs: previous.swapVolumeSampledAtMs + } +} + +function commitHostMemorySample(nowMs: number): boolean { + try { + const details = getSystemMemoryDetails() + // Why not `length === 0`: the signal label is appended unconditionally, so a + // reading that resolved no memory field at all still arrives with one key. + if (!Object.keys(details).some((key) => key !== PRESSURE_SIGNAL_KEY)) { + return false + } + preGoneSample = withCarriedSwapVolume({ details, sampledAtMs: nowMs }) + return true + } catch { + // Why: a failed read must not erase the previous good sample. + return false + } +} + +async function mergeSwapVolumeFreeSpace(issuedOnTick: number): Promise { + if (swapVolumeReadInFlight) { + return + } + swapVolumeReadInFlight = true + const generation = samplingGeneration + const issuedFor = preGoneSample + try { + const volume = await readSwapVolumeFreeSpace() + if (volume && preGoneSample && generation === samplingGeneration) { + // Why only its own tick qualifies: a statfs that outlived its tick carries a + // pre-storm volume number, and the latch makes that lag unbounded. It still + // ships beside its age, but it may not decide the verdict. + // Why the tick counter and not sample identity: a tick whose host read fails + // leaves the sample object in place, so identity alone reads as co-timed. + const coTimed = issuedOnTick === sampleTick + preGoneSample = { + ...preGoneSample, + details: withSwapVolumeFreeSpace(preGoneSample.details, volume, process.platform, coTimed), + swapVolumeSampledAtMs: issuedFor?.sampledAtMs + } + } + } catch { + // Why: the memory reading is already committed and stands on its own. + } finally { + swapVolumeReadInFlight = false + } +} + +export async function samplePreGoneSystemMemory(nowMs: number = Date.now()): Promise { + // Why commit before awaiting: the volume read is a statfs, and under the very + // paging storm this targets it is slowest — it must never delay, or (via an + // in-flight latch) skip, the cheap synchronous host reading. + const tick = ++sampleTick + if (!commitHostMemorySample(nowMs)) { + return + } + await mergeSwapVolumeFreeSpace(tick) +} + +export function startPreGoneSystemMemorySampling( + intervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS +): void { + if (preGoneTimer) { + return + } + void samplePreGoneSystemMemory() + preGoneTimer = setInterval(() => void samplePreGoneSystemMemory(), intervalMs) + preGoneTimer.unref?.() +} + +export function resetPreGoneSystemMemorySamplingForTest(): void { + if (preGoneTimer) { + clearInterval(preGoneTimer) + } + preGoneTimer = null + preGoneSample = null + swapVolumeReadInFlight = false + // Why bump: an already-awaited volume read must not repopulate a reset sample. + samplingGeneration += 1 +} + +/** Keyed as `systemMemoryPreGone*` so a scan over the `systemMemory` family sees both reads. */ +export function preGoneSystemMemoryDetails(nowMs: number): CrashReportDetails { + if (!preGoneSample) { + return {} + } + const details: CrashReportDetails = { + [`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSampleAgeMs`]: Math.max( + 0, + nowMs - preGoneSample.sampledAtMs + ) + } + // Why its own age: the volume read resolves out of band, so it can be older + // than the memory reading printed beside it, and that gap must be readable. + if (preGoneSample.swapVolumeSampledAtMs !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSwapVolumeAgeMs`] = Math.max( + 0, + nowMs - preGoneSample.swapVolumeSampledAtMs + ) + } + for (const [key, value] of Object.entries(preGoneSample.details)) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGone${key.slice(SYSTEM_MEMORY_KEY_PREFIX.length)}`] = + value + } + return details +} diff --git a/src/main/crash-reporting/process-gone-diagnostics.test.ts b/src/main/crash-reporting/process-gone-diagnostics.test.ts index e31645865d8..6a6ec410733 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.test.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.test.ts @@ -2,11 +2,11 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { buildProcessGoneCrashDetails, collectProcessGoneMetricDetails, - resetPreGoneProcessMetricsSamplingForTest, + resetPreGoneCrashSamplingForTest, samplePreGoneProcessMetrics, - startPreGoneProcessMetricsSampling + startPreGoneCrashSampling } from './process-gone-diagnostics' -import { setSystemMemoryInfoReaderForTest } from './gone-time-system-memory' +import { setSystemMemoryInfoReaderForTest } from './system-memory-details' type MetricFixture = { pid?: number @@ -27,7 +27,7 @@ vi.mock('electron', () => ({ describe('process gone diagnostics', () => { beforeEach(() => { - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() setSystemMemoryInfoReaderForTest(null) }) @@ -141,8 +141,8 @@ describe('process gone diagnostics', () => { appMetricsMock.mockReturnValue([ { pid: 30, type: 'Tab', memory: { workingSetSize: 1024 * 100 } } ]) - startPreGoneProcessMetricsSampling(1_000) - startPreGoneProcessMetricsSampling(1_000) + startPreGoneCrashSampling(1_000) + startPreGoneCrashSampling(1_000) // A crash inside the first interval already has a sample to draw from. expect(buildProcessGoneCrashDetails({}, 'renderer')).toMatchObject({ @@ -582,12 +582,12 @@ describe('process gone diagnostics', () => { it("arms an unref'd interval so sampling never holds the event loop open", () => { const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') try { - startPreGoneProcessMetricsSampling(60_000) + startPreGoneCrashSampling(60_000) const timer = setIntervalSpy.mock.results[0]?.value as NodeJS.Timeout expect(timer.hasRef()).toBe(false) } finally { setIntervalSpy.mockRestore() - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() } }) @@ -641,7 +641,7 @@ describe('process gone diagnostics', () => { expect(details.systemMemoryTotalMB).toBe(16_384) }) - it('samples system memory at gone time but never into the pre-gone snapshot', () => { + it('samples system memory at gone time but never into the processMetrics family', () => { appMetricsMock.mockReturnValue([{ pid: 1, type: 'Browser', memory: { workingSetSize: 0 } }]) samplePreGoneProcessMetrics() setSystemMemoryInfoReaderForTest(() => ({ @@ -658,7 +658,9 @@ describe('process gone diagnostics', () => { systemMemorySwapTotalMB: 8_192, systemMemorySwapFreeMB: 40 }) - expect(details.processMetricsPreGoneSystemMemoryTotalMB).toBeUndefined() + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) }) it('leaves records unflagged when the crashed bucket is still populated', () => { diff --git a/src/main/crash-reporting/process-gone-diagnostics.ts b/src/main/crash-reporting/process-gone-diagnostics.ts index d0bb380a2b6..bf0d735a2c7 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.ts @@ -3,7 +3,13 @@ import { sanitizeCrashReportDetails, type CrashReportDetailValue } from '../../shared/crash-reporting' -import { getSystemMemoryAtGoneDetails, memoryKBFieldMB } from './gone-time-system-memory' +import { getSystemMemoryDetails, memoryKBFieldMB } from './system-memory-details' +import { + PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS, + preGoneSystemMemoryDetails, + resetPreGoneSystemMemorySamplingForTest, + startPreGoneSystemMemorySampling +} from './pre-gone-host-memory' type ProcessMetricLike = { pid?: unknown @@ -204,8 +210,9 @@ export function samplePreGoneProcessMetrics(nowMs: number = Date.now()): void { } } -export function startPreGoneProcessMetricsSampling( - intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS +export function startPreGoneCrashSampling( + intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS, + systemMemoryIntervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS ): void { if (preGoneSampleTimer) { return @@ -213,14 +220,16 @@ export function startPreGoneProcessMetricsSampling( samplePreGoneProcessMetrics() preGoneSampleTimer = setInterval(() => samplePreGoneProcessMetrics(), intervalMs) preGoneSampleTimer.unref?.() + startPreGoneSystemMemorySampling(systemMemoryIntervalMs) } -export function resetPreGoneProcessMetricsSamplingForTest(): void { +export function resetPreGoneCrashSamplingForTest(): void { if (preGoneSampleTimer) { clearInterval(preGoneSampleTimer) } preGoneSampleTimer = null preGoneSample = null + resetPreGoneSystemMemorySamplingForTest() } const PROCESS_METRICS_KEY_PREFIX = 'processMetrics' @@ -271,7 +280,7 @@ export function buildProcessGoneCrashDetails( const crashDetails: CrashReportDetails = { ...sanitizedDetails, ...liveMetricDetails, - ...getSystemMemoryAtGoneDetails() + ...getSystemMemoryDetails() } // Why: with the crasher gone, Largest names a survivor — flag that so the // live buckets are read as "everyone else", not as the crashed process. @@ -290,8 +299,10 @@ export function buildProcessGoneCrashDetails( if (liveMetricDetails[crashedBucketCountKey] === 0 || sampledSameBucketProcessVanished) { crashDetails.processMetricsCrashedProcessAbsent = true } + const nowMs = Date.now() if (preGoneSample) { - Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, Date.now())) + Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, nowMs)) } + Object.assign(crashDetails, preGoneSystemMemoryDetails(nowMs)) return crashDetails } diff --git a/src/main/crash-reporting/swap-volume-free-space.ts b/src/main/crash-reporting/swap-volume-free-space.ts new file mode 100644 index 00000000000..3ad40b7629b --- /dev/null +++ b/src/main/crash-reporting/swap-volume-free-space.ts @@ -0,0 +1,67 @@ +import { statfs } from 'node:fs/promises' +import path from 'node:path' + +// Why: a system-managed Windows pagefile — and a macOS swapfile — only grows +// into free space on its own volume, so low available commit is a REFUSED +// allocation only when that volume is full too. Linux is excluded on purpose: +// its swap is a fixed partition, a fixed-size swapfile, or zram, none of which +// grow into root-fs free space, so the number would read as headroom that +// cannot exist. The measured volume ships alongside because Windows only names +// the DEFAULT pagefile drive; a relocated pagefile lives elsewhere. + +const BYTES_PER_MB = 1024 * 1024 + +export type SwapVolumeFreeSpace = { + freeMB: number + /** Which volume was measured, separator-trimmed so redaction sees no path. */ + volume: string +} + +type SwapVolumeFreeSpaceReader = ( + platform: NodeJS.Platform +) => Promise + +function swapVolumeAnchor(platform: NodeJS.Platform): string | undefined { + if (platform === 'win32') { + const anchor = process.env.SystemRoot || process.env.SystemDrive + return anchor ? path.parse(anchor).root || anchor : undefined + } + return platform === 'darwin' ? path.sep : undefined +} + +function volumeLabel(root: string): string { + const trimmed = root.replace(/[\\/]+$/, '') + return trimmed.length > 0 ? trimmed : root +} + +async function statfsSwapVolumeFreeSpace( + platform: NodeJS.Platform +): Promise { + const root = swapVolumeAnchor(platform) + if (!root) { + return undefined + } + try { + const stats = await statfs(root) + const bytes = Number(stats.bsize) * Number(stats.bavail) + return Number.isFinite(bytes) + ? { freeMB: Math.round(Math.max(0, bytes) / BYTES_PER_MB), volume: volumeLabel(root) } + : undefined + } catch { + return undefined + } +} + +let swapVolumeFreeSpaceReader: SwapVolumeFreeSpaceReader = statfsSwapVolumeFreeSpace + +export function setSwapVolumeFreeSpaceReaderForTest( + reader: SwapVolumeFreeSpaceReader | null +): void { + swapVolumeFreeSpaceReader = reader ?? statfsSwapVolumeFreeSpace +} + +export function readSwapVolumeFreeSpace( + platform: NodeJS.Platform = process.platform +): Promise { + return swapVolumeFreeSpaceReader(platform) +} diff --git a/src/main/crash-reporting/system-memory-details.ts b/src/main/crash-reporting/system-memory-details.ts new file mode 100644 index 00000000000..1f2cf556faa --- /dev/null +++ b/src/main/crash-reporting/system-memory-details.ts @@ -0,0 +1,161 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import type { SwapVolumeFreeSpace } from './swap-volume-free-space' + +// ─── Host system memory for crash reports ─────────────────────────── +// Why: the system outlives the crashed process, so this IS sampleable at +// process-gone — it separates "renderer grew huge" from "machine out of +// memory/commit", which the per-process buckets alone cannot. The gone-time +// caller reads AFTER the corpse returned its pages, so free/swapFree read +// healthier than at kill time; the pre-gone sampler carries a live reading past +// that. +// Every reading is labelled `systemMemoryPressureSignal` so no report can be +// read as a pressure verdict the platform never gave: +// win32 — swapFree is MEMORYSTATUSEX.ullAvailPageFile, i.e. available +// COMMIT, which pagefile growth can heal (a 127 MB commit floor healed to +// 2029 MB mid-hold on the win-lowspec repro, killing nothing). Free space on +// the swap volume does NOT establish that it could: a fixed-size or disabled +// pagefile grows into no amount of empty disk, its maximum is unreadable +// here (needs a registry read), and the measured volume is only the DEFAULT +// pagefile drive. So a co-timed volume reading is context beside the commit +// number — `available-commit-volume-cotimed` — never a verdict. The one +// decisive win32 case is a commit limit at or below RAM: no pagefile exists +// to grow, so the floor cannot heal (`available-commit-hard-capped`). +// linux — MemAvailable is the real signal; MemFree is not (it excludes page +// cache and other reclaimable memory). +// darwin — none. `free` stays low on healthy machines and +// fileBacked/purgeable are only a reclaimability proxy. The real signal +// needs `memory_pressure -Q`; Orca's reader for it +// (src/main/memory/host-memory.ts) is on-demand, and spawning a subprocess +// on a 10 s app-lifetime timer costs more than the gap it closes. + +type CrashReportDetails = Record + +export const SYSTEM_MEMORY_KEY_PREFIX = 'systemMemory' + +export function memoryKBFieldMB(value: unknown): number | undefined { + const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined + return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) +} + +type SystemMemoryInfoLike = { + total?: unknown + free?: unknown + available?: unknown + swapTotal?: unknown + swapFree?: unknown + fileBacked?: unknown + purgeable?: unknown +} + +type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null + +/** How far this reading may be read as a "was the host under pressure" verdict. */ +export type SystemMemoryPressureSignal = + | 'available-commit-hard-capped' + | 'available-commit-volume-cotimed' + | 'available-commit-unqualified' + | 'mem-available' + | 'none' + +function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { + const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) + .getSystemMemoryInfo + if (typeof read !== 'function') { + return null + } + try { + return read.call(process) + } catch { + return null + } +} + +let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo + +export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { + systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo +} + +function numericDetail(details: CrashReportDetails, suffix: string): number | undefined { + const value = details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] + return typeof value === 'number' ? value : undefined +} + +/** Windows commit limit = RAM + pagefile, so a limit at or below RAM has no pagefile behind it. */ +function pagefileBacksCommit(details: CrashReportDetails): boolean | undefined { + const total = numericDetail(details, 'TotalMB') + const swapTotal = numericDetail(details, 'SwapTotalMB') + return total === undefined || swapTotal === undefined ? undefined : swapTotal > total +} + +function pressureSignal( + platform: NodeJS.Platform, + details: CrashReportDetails, + volumeCoTimed = true +): SystemMemoryPressureSignal { + if (platform === 'win32' && `${SYSTEM_MEMORY_KEY_PREFIX}SwapFreeMB` in details) { + if (pagefileBacksCommit(details) === false) { + return 'available-commit-hard-capped' + } + return volumeCoTimed && `${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB` in details + ? 'available-commit-volume-cotimed' + : 'available-commit-unqualified' + } + if (platform === 'linux' && `${SYSTEM_MEMORY_KEY_PREFIX}AvailableMB` in details) { + return 'mem-available' + } + return 'none' +} + +export function getSystemMemoryDetails( + platform: NodeJS.Platform = process.platform +): CrashReportDetails { + const info = systemMemoryInfoReader() + if (!info) { + return {} + } + const details: CrashReportDetails = {} + const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ + ['total', 'TotalMB'], + ['free', 'FreeMB'], + ['available', 'AvailableMB'], + ['swapTotal', 'SwapTotalMB'], + ['swapFree', 'SwapFreeMB'], + ['fileBacked', 'FileBackedMB'], + ['purgeable', 'PurgeableMB'] + ] + for (const [field, suffix] of fields) { + const mb = memoryKBFieldMB(info[field]) + if (mb !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] = mb + } + } + details[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, details) + return details +} + +/** + * Merges the statfs-derived volume datum, which needs an await and so is only + * reachable from the periodic sampler, and relabels the reading it sits beside. + * + * `coTimed` false means the statfs outlived the tick that issued it, so this + * volume number and the commit number beside it describe different moments — + * during a pagefile-growth storm that is exactly when they diverge, and a + * pre-storm 40 GB printed next to 200 MB of commit reads as "the pagefile had + * room", the opposite conclusion. The datum still ships (with its own age), but + * only a co-timed one is named in the label. + */ +export function withSwapVolumeFreeSpace( + details: CrashReportDetails, + volume: SwapVolumeFreeSpace, + platform: NodeJS.Platform = process.platform, + coTimed = true +): CrashReportDetails { + const merged: CrashReportDetails = { + ...details, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`]: volume.freeMB, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`]: volume.volume + } + merged[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, merged, coTimed) + return merged +} diff --git a/src/main/git/command-runner/git-exec-file.ts b/src/main/git/command-runner/git-exec-file.ts index dbd861d4897..acfeb9f8a3e 100644 --- a/src/main/git/command-runner/git-exec-file.ts +++ b/src/main/git/command-runner/git-exec-file.ts @@ -10,6 +10,7 @@ import { prepareWslLinkedWorktreeGitRouting } from '../wsl-linked-worktree-git-routing' import { resolveCommand, type ResolvedCommand } from './wsl-command-resolution' +import { annotateWslHostFailure } from './wsl-host-failure' import type { GitAdmissionTier, GitExecOptions } from './git-exec-options' import { execFileCapture, execFileCaptureToTermination } from './exec-file-capture' import { @@ -96,7 +97,7 @@ async function gitExecFileAsyncUnlocked( ? {} : { createTimeoutError: () => new GitCommandTimeoutError(timeoutMs) }) } - return options.terminationBarrier + const captured = options.terminationBarrier ? execFileCaptureToTermination( command.binary, command.args, @@ -104,6 +105,10 @@ async function gitExecFileAsyncUnlocked( command.termination ) : execFileCapture(command.binary, command.args, captureOptions) + // Why: a dead WSL distro fails with an empty stderr, so the span would carry no cause at all. + return captured.catch((error: unknown) => { + throw annotateWslHostFailure(error, command) + }) } const runCapturedCommand = async (): Promise<{ stdout: string; stderr: string }> => { let result: { stdout: string | Buffer; stderr: string | Buffer } diff --git a/src/main/git/command-runner/git-process-env.ts b/src/main/git/command-runner/git-process-env.ts index 6154364e5ab..9064f5ed900 100644 --- a/src/main/git/command-runner/git-process-env.ts +++ b/src/main/git/command-runner/git-process-env.ts @@ -63,6 +63,11 @@ export function nonInteractiveGitEnv( platform: NodeJS.Platform = process.platform ): NodeJS.ProcessEnv { const next = promptGuardGitEnv(env, platform) + if (platform === 'win32') { + // Why: without it wsl.exe writes its OWN failures ("no distribution with the supplied name") as + // UTF-16LE (#9010), which is how a dead distro reached telemetry as an error with no text. + next.WSL_UTF8 = '1' + } if (!next.GIT_SSH_COMMAND) { next.GIT_SSH_COMMAND = 'ssh -o BatchMode=yes' if (platform === 'win32') { diff --git a/src/main/git/command-runner/wsl-host-failure.test.ts b/src/main/git/command-runner/wsl-host-failure.test.ts new file mode 100644 index 00000000000..4b495817f6f --- /dev/null +++ b/src/main/git/command-runner/wsl-host-failure.test.ts @@ -0,0 +1,143 @@ +import { EventEmitter } from 'node:events' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { execFileMock } = vi.hoisted(() => ({ execFileMock: vi.fn() })) + +vi.mock('node:child_process', () => ({ + execFile: execFileMock, + execFileSync: vi.fn(), + spawn: vi.fn() +})) +vi.mock('../../observability/instrumentation', () => ({ + withGitSpan: (_attributes: unknown, run: (span: unknown) => unknown) => + run({ setAttribute: () => {} }) +})) +vi.mock('../../diagnostics/main-thread-churn-probe', () => ({ recordSubprocessSpawn: vi.fn() })) + +import { gitExecFileAsync } from '../runner' +import { _resetGitAdmissionForTests } from './git-subprocess-admission' +import { nonInteractiveGitEnv } from './git-process-env' +import { annotateWslHostFailure, readWslHostFailureDiagnostic } from './wsl-host-failure' +import type { ResolvedCommand } from './wsl-command-resolution' + +const WSL_COMMAND: ResolvedCommand = { + binary: 'wsl.exe', + args: ['-d', 'kali-linux', '--exec', 'sh', '-lc', 'git worktree list'], + cwd: 'C:\\Users\\paulius', + wsl: { distro: 'kali-linux', linuxPath: '/home/paulius/bugbounty' }, + wslMode: 'login-shell' +} + +const WSL_DIAGNOSTIC = + 'There is no distribution with the supplied name.\r\nError code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\r\n' + +/** wsl.exe without WSL_UTF8 writes its own diagnostic as UTF-16LE, which reaches Node as NUL-riddled text. */ +function asUtf16Mojibake(text: string): string { + return [...text].map((character) => `${character}\u0000`).join('') +} + +function hostFailure(stdout: string): Error { + return Object.assign(new Error('Command failed: wsl.exe -d kali-linux --exec sh -lc ...'), { + code: 4294967295, + stdout, + stderr: '' + }) +} + +async function withPlatform(platform: NodeJS.Platform, run: () => Promise): Promise { + const original = process.platform + Object.defineProperty(process, 'platform', { configurable: true, value: platform }) + try { + return await run() + } finally { + Object.defineProperty(process, 'platform', { configurable: true, value: original }) + } +} + +describe('wsl.exe host failure classification', () => { + it('reads the diagnostic wsl.exe left on stdout, including UTF-16 output', () => { + expect(readWslHostFailureDiagnostic(hostFailure(WSL_DIAGNOSTIC), WSL_COMMAND)).toContain( + 'Wsl/Service/WSL_E_DISTRO_NOT_FOUND' + ) + expect( + readWslHostFailureDiagnostic(hostFailure(asUtf16Mojibake(WSL_DIAGNOSTIC)), WSL_COMMAND) + ).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + }) + + it('leaves a guest failure and a non-WSL command alone', () => { + const guestFailure = Object.assign(new Error('Command failed'), { + code: 1, + stdout: '', + stderr: 'fatal: not a git repository\n' + }) + expect(readWslHostFailureDiagnostic(guestFailure, WSL_COMMAND)).toBeNull() + // Same exit code, but wsl.exe was never involved. + expect( + readWslHostFailureDiagnostic(hostFailure(WSL_DIAGNOSTIC), { + binary: 'git', + args: ['status'], + cwd: '/repo', + wsl: null, + wslMode: null + }) + ).toBeNull() + }) + + it('moves the diagnostic into the message the span records', () => { + const error = annotateWslHostFailure(hostFailure(WSL_DIAGNOSTIC), WSL_COMMAND) as Error & { + wslHostFailure?: boolean + wslDistro?: string + code?: number + } + expect(error.message).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + expect(error.message).toContain('kali-linux') + expect(error.wslHostFailure).toBe(true) + expect(error.wslDistro).toBe('kali-linux') + // The original failure detail must survive for callers that classify on it. + expect(error.code).toBe(4294967295) + expect(error.message).toContain('Command failed: wsl.exe') + }) +}) + +describe('WSL-routed git subprocess', () => { + beforeEach(() => { + execFileMock.mockReset() + }) + + afterEach(() => { + _resetGitAdmissionForTests() + }) + + it('sets WSL_UTF8 so wsl.exe explains itself in UTF-8', () => { + expect(nonInteractiveGitEnv({}, 'win32').WSL_UTF8).toBe('1') + expect(nonInteractiveGitEnv({}, 'darwin').WSL_UTF8).toBeUndefined() + }) + + it('reports a dead distro instead of an empty git error', async () => { + execFileMock.mockImplementation((_command, _args, _options, callback) => { + const child = new EventEmitter() as EventEmitter & { pid: number; kill: () => void } + child.pid = 4321 + child.kill = () => {} + queueMicrotask(() => + callback?.( + hostFailure(asUtf16Mojibake(WSL_DIAGNOSTIC)), + asUtf16Mojibake(WSL_DIAGNOSTIC), + '' + ) + ) + return child + }) + + const failure = await withPlatform('win32', () => + gitExecFileAsync(['worktree', 'list', '--porcelain', '-z'], { + cwd: '\\\\wsl.localhost\\kali-linux\\home\\paulius\\bugbounty' + }).then( + () => null, + (error: unknown) => error as Error + ) + ) + + expect(failure?.message).toContain('Wsl/Service/WSL_E_DISTRO_NOT_FOUND') + expect(execFileMock.mock.calls.at(-1)?.[2]?.env?.WSL_UTF8).toBe('1') + }) +}) diff --git a/src/main/git/command-runner/wsl-host-failure.ts b/src/main/git/command-runner/wsl-host-failure.ts new file mode 100644 index 00000000000..74f98509d0b --- /dev/null +++ b/src/main/git/command-runner/wsl-host-failure.ts @@ -0,0 +1,55 @@ +import type { ResolvedCommand } from './wsl-command-resolution' + +/** wsl.exe's own launch-failure exit, distinct from any status the guest process can return. */ +export const WSL_HOST_FAILURE_EXIT_CODE = 0xffffffff + +function outputText(value: unknown): string { + if (typeof value === 'string') { + return value + } + return Buffer.isBuffer(value) ? value.toString('utf8') : '' +} + +/** + * The message wsl.exe prints when it — not the guest — failed: a distro that was renamed or + * removed, or a VM that would not start. + * + * Why this needs decoding at all: wsl.exe exits 0xFFFFFFFF, leaves stderr EMPTY, and writes + * `Error code: Wsl/Service/WSL_E_*` to stdout, so every caller that reports stderr reports nothing. + * NULs are stripped because a wsl.exe that ignores WSL_UTF8 writes that line as UTF-16LE (#9010). + */ +export function readWslHostFailureDiagnostic( + error: unknown, + command: ResolvedCommand +): string | null { + if (!command.wsl || !error || typeof error !== 'object') { + return null + } + const { code, status, stdout, stderr } = error as { + code?: unknown + status?: unknown + stdout?: unknown + stderr?: unknown + } + const exitCode = typeof code === 'number' ? code : typeof status === 'number' ? status : null + // A guest failure always explains itself on stderr; an empty one plus this exit is the host. + if (exitCode !== WSL_HOST_FAILURE_EXIT_CODE || outputText(stderr).trim().length > 0) { + return null + } + const diagnostic = outputText(stdout).replaceAll('\u0000', '').trim() + return diagnostic.length > 0 ? diagnostic : 'wsl.exe reported no diagnostic.' +} + +/** + * Move a wsl.exe host failure into the error's message, which is what `git.exec` spans record ahead + * of the stack. Left untouched when the failure came from git itself. + */ +export function annotateWslHostFailure(error: unknown, command: ResolvedCommand): unknown { + const diagnostic = readWslHostFailureDiagnostic(error, command) + if (diagnostic === null || !(error instanceof Error) || !command.wsl) { + return error + } + const distro = command.wsl.distro + error.message = `wsl.exe host failure (distro "${distro}"): ${diagnostic}\n${error.message}` + return Object.assign(error, { wslHostFailure: true, wslDistro: distro }) +} diff --git a/src/main/git/source-control/submodule-status.ts b/src/main/git/source-control/submodule-status.ts index f3c808c9f2f..fb4ca8449e3 100644 --- a/src/main/git/source-control/submodule-status.ts +++ b/src/main/git/source-control/submodule-status.ts @@ -26,17 +26,22 @@ export async function getSubmoduleStatus( const submoduleWorktreePath = resolveSubmoduleWorktreePath(worktreePath, submodulePath) const limit = resolveGitStatusLimit(options.limit) // Why: staged expansion only represents HEAD→index; scanning the submodule worktree is wasted work. - const workingResult = options.staged - ? ({ entries: [], conflictOperation: 'unknown' } satisfies GitStatusResult) - : await getStatus(submoduleWorktreePath, options) - // Why: a moved gitlink (clean worktree) has no status rows; surface the parent-commit→checkout range as inner rows. - const fromOid = options.staged - ? await readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options) - : (await readGitlinkOidFromIndex(worktreePath, submodulePath, options)) || - (await readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options)) - const toOid = options.staged - ? await readGitlinkOidFromIndex(worktreePath, submodulePath, options) - : await readWorkingSubmoduleHead(submoduleWorktreePath, options) + // These three reads are independent, so they run concurrently — on SSH/WSL each one is a real round trip. + const [workingResult, fromOid, toOid] = await Promise.all([ + options.staged + ? Promise.resolve({ entries: [], conflictOperation: 'unknown' }) + : getStatus(submoduleWorktreePath, options), + // Why: a moved gitlink (clean worktree) has no status rows; surface the parent-commit→checkout range as inner rows. + options.staged + ? readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options) + : readGitlinkOidFromIndex(worktreePath, submodulePath, options).then( + (indexOid) => + indexOid || readGitlinkOidFromTree(worktreePath, 'HEAD', submodulePath, options) + ), + options.staged + ? readGitlinkOidFromIndex(worktreePath, submodulePath, options) + : readWorkingSubmoduleHead(submoduleWorktreePath, options) + ]) if (fromOid && toOid && fromOid !== toOid) { const rangeEntries = await computeSubmoduleRangeEntries( submoduleWorktreePath, diff --git a/src/main/git/status-submodule.test.ts b/src/main/git/status-submodule.test.ts index b67a4eca9a0..6528d373a78 100644 --- a/src/main/git/status-submodule.test.ts +++ b/src/main/git/status-submodule.test.ts @@ -395,4 +395,35 @@ describe('getSubmoduleStatus', () => { expect(result.didHitLimit).toBe(true) expect(result.statusLength).toBe(2) }) + + it('overlaps the inner status, index and HEAD reads instead of serializing them', async () => { + const OLD_OID = 'a'.repeat(40) + const NEW_OID = 'b'.repeat(40) + readFileMock.mockResolvedValue('gitdir: /repo/flutter_mine/.git\n') + existsSyncMock.mockReturnValue(false) + const events: string[] = [] + gitExecFileAsyncMock.mockReset() + gitExecFileAsyncMock.mockImplementation(async (args: string[]) => { + const name = args.includes('status') ? 'status' : args[0] + events.push(`start:${name}`) + if (name === 'status') { + // Hold the inner status open so a serialized caller could not have started the oid reads. + await new Promise((resolve) => setTimeout(resolve, 5)) + } + events.push(`end:${name}`) + if (name === 'ls-files') { + return { stdout: `160000 ${OLD_OID} 0\tflutter_mine\n` } + } + if (name === 'rev-parse') { + return { stdout: `${NEW_OID}\n` } + } + return { stdout: '' } + }) + + await getSubmoduleStatus('/repo', 'flutter_mine') + + // Serialized reads only start ls-files/rev-parse once the inner status has resolved. + expect(events.indexOf('start:ls-files')).toBeLessThan(events.indexOf('end:status')) + expect(events.indexOf('start:rev-parse')).toBeLessThan(events.indexOf('end:status')) + }) }) diff --git a/src/main/git/worktree-listing.ts b/src/main/git/worktree-listing.ts index a60819181d4..f027bac4bc1 100644 --- a/src/main/git/worktree-listing.ts +++ b/src/main/git/worktree-listing.ts @@ -35,17 +35,7 @@ export async function listWorktreeGraph( ? worktrees : worktrees.filter((worktree) => !isWorktreeCreatePreparation(worktree)) } catch (err) { - if (getErrorCode(err) === 'ENOENT') { - try { - await stat(repoPath) - } catch (statErr) { - if (getErrorCode(statErr) === 'ENOENT') { - console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) - return [] - } - } - } - if (isNotGitRepositoryError(err)) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { return [] } console.warn(`[git/worktree] listWorktreeGraph failed for ${repoPath}:`, err) @@ -64,17 +54,7 @@ export async function listWorktreesUnshared( : worktrees.filter((worktree) => !isWorktreeCreatePreparation(worktree)) return annotateSparseCheckoutStatus(repoPath, visibleWorktrees, options) } catch (err) { - if (getErrorCode(err) === 'ENOENT') { - try { - await stat(repoPath) - } catch (statErr) { - if (getErrorCode(statErr) === 'ENOENT') { - console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) - return [] - } - } - } - if (isNotGitRepositoryError(err)) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { return [] } // Why: don't swallow git-compat/repo-state failures — else they resurface as opaque "created but not found in listing" errors. @@ -83,6 +63,24 @@ export async function listWorktreesUnshared( } } +/** + * The two failures where an empty listing is the repo's true answer, not a broken scan: the repo + * path is gone, or it is not a Git repo. Every other failure means the scan could not read Git. + */ +async function isTrueEmptyWorktreeListing(repoPath: string, err: unknown): Promise { + if (getErrorCode(err) === 'ENOENT') { + try { + await stat(repoPath) + } catch (statErr) { + if (getErrorCode(statErr) === 'ENOENT') { + console.warn(`[git/worktree] repo path missing; skipping worktree list: ${repoPath}`) + return true + } + } + } + return isNotGitRepositoryError(err) +} + export async function listWorktreesStrict( repoPath: string, options: GitWorktreeExecOptions = {} @@ -97,6 +95,28 @@ export async function listWorktreesStrict( return annotateSparseCheckoutStatus(repoPath, visibleWorktrees, options) } +/** + * Strict except for the two true empties above. + * + * Why: a Git or host failure (dead WSL distro, hung mount) softened to `[]` reaches the detected + * listing as an *authoritative* empty scan, which then permanently prunes the repo's worktrees and + * the agent tabs attached to them. Rejecting keeps that listing non-authoritative, while a deleted + * repo still reports empty so real removals prune. + */ +export async function listWorktreesStrictAllowingTrueEmpty( + repoPath: string, + options: GitWorktreeExecOptions = {} +): Promise { + try { + return await listWorktreesStrict(repoPath, options) + } catch (err) { + if (await isTrueEmptyWorktreeListing(repoPath, err)) { + return [] + } + throw err + } +} + export async function annotateSparseCheckoutStatus( repoPath: string, worktrees: GitWorktreeInfo[], diff --git a/src/main/git/worktree-scan-cache.ts b/src/main/git/worktree-scan-cache.ts index 59a7691fe7f..a35f0a4f275 100644 --- a/src/main/git/worktree-scan-cache.ts +++ b/src/main/git/worktree-scan-cache.ts @@ -2,7 +2,8 @@ import type { GitWorktreeInfo } from '../../shared/worktree/types' import { annotateSparseCheckoutStatus, listWorktreeGraph as listWorktreeGraphUnshared, - listWorktreesStrict as listWorktreesStrictUnshared + listWorktreesStrict as listWorktreesStrictUnshared, + listWorktreesStrictAllowingTrueEmpty as listWorktreesStrictAllowingTrueEmptyUnshared } from './worktree-listing' import type { GitWorktreeExecOptions } from './worktree-operation-options' import { WORKTREE_LIST_TIMEOUT_MS } from './worktree-operation-options' @@ -10,7 +11,7 @@ import { WORKTREE_LIST_TIMEOUT_MS } from './worktree-operation-options' // Why: share concurrent `git worktree list` scans, which are expensive on Windows. const inFlightWorktreeScans = new Map>() -type WorktreeScanKind = 'graph' | 'lenient' | 'strict' +type WorktreeScanKind = 'graph' | 'lenient' | 'strict' | 'strict-true-empty' // Why: mutation generations prevent listings from joining stale scans. const worktreeScanGenerations = new Map() @@ -139,3 +140,20 @@ export function listWorktreesSharedStrict( ): Promise { return shareWorktreeScan(repoPath, options, 'strict', listWorktreesStrictUnshared) } + +/** + * The detected scan's discipline: reject a Git/host failure so it cannot publish as an + * authoritative empty listing, but still answer `[]` for a repo that is gone or not a repo. + * Its own kind because neither a strict nor a lenient joiner may inherit that middle contract. + */ +export function listWorktreesSharedStrictAllowingTrueEmpty( + repoPath: string, + options: GitWorktreeExecOptions = {} +): Promise { + return shareWorktreeScan( + repoPath, + options, + 'strict-true-empty', + listWorktreesStrictAllowingTrueEmptyUnshared + ) +} diff --git a/src/main/git/worktree.ts b/src/main/git/worktree.ts index 18e92e2873a..024485b6981 100644 --- a/src/main/git/worktree.ts +++ b/src/main/git/worktree.ts @@ -32,7 +32,8 @@ export { _resetWorktreeScanCacheForTests, listWorktreeGraph, listWorktrees, - listWorktreesSharedStrict + listWorktreesSharedStrict, + listWorktreesSharedStrictAllowingTrueEmpty } from './worktree-scan-cache' export { bumpWorktreeScanGeneration as notifyPreparedWorktreeMutation } from './worktree-scan-cache' export { addSparseWorktree } from './worktree-sparse-add' diff --git a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts index 97e0a8f9d65..8d126f557b5 100644 --- a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts +++ b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts @@ -49,6 +49,7 @@ function recordRendererBreadcrumbTrace( const DUPLICATE_TAB_OWNER_BREADCRUMB = 'terminal_tab_id_owned_by_multiple_worktrees' const PARK_VERDICT_CHURN_BREADCRUMB = 'terminal_park_verdict_churn' const REACT_COMMIT_CASCADE_BREADCRUMB = 'react_commit_cascade' +const REPLAY_GUARD_WEDGED_BREADCRUMB = 'terminal_replay_guard_wedged_release' const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ 'renderer_error', 'renderer_unhandled_rejection', @@ -56,6 +57,7 @@ const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ DUPLICATE_TAB_OWNER_BREADCRUMB, PARK_VERDICT_CHURN_BREADCRUMB, REACT_COMMIT_CASCADE_BREADCRUMB, + REPLAY_GUARD_WEDGED_BREADCRUMB, TERMINAL_WEBGL_DIAGNOSTIC_BREADCRUMB ]) const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 @@ -69,6 +71,11 @@ const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 // 30-entry ring to two such bursts. `suppressedSinceLast` keeps the pane count // — the only signal these carry — in one slot. const NAME_ONLY_COALESCED_BREADCRUMB_NAMES = new Set(['terminal_safe_fit_retry_exhausted']) +// Why: the 30-slot ring is the scarce sink; the durable span stream is not. For +// bounded-rate pane telemetry whose multiplicity is the whole signal, spans are the +// only place a burst survives the restart that clears the ring, so coalesce the ring +// but keep every event's span. +const PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES = new Set([REPLAY_GUARD_WEDGED_BREADCRUMB]) function rendererBreadcrumbCoalesceKey( name: string, @@ -77,6 +84,13 @@ function rendererBreadcrumbCoalesceKey( if (NAME_ONLY_COALESCED_BREADCRUMB_NAMES.has(name)) { return name } + // Why presence and not value: `ptyId`/`tabIdHash` are absent on the restore call + // site (layout-serialization restoreScrollbackBuffers) and present on reattach, so + // their presence is the call-site identity a mixed burst would otherwise lose. Four + // slots per storm at most, regardless of pane count. + if (name === REPLAY_GUARD_WEDGED_BREADCRUMB) { + return `${name}:${data?.ptyId ? 'pty' : ''}:${data?.tabIdHash ? 'tab' : ''}` + } // Why trigger and not name alone: `burst` means damping engaged a commit // short of React #185, `window` means slow benign churn. Collapsing them // would drop the near-crash signal into a slow-churn slot. Still bounded — @@ -191,9 +205,13 @@ export function recordRendererBreadcrumbFromRenderer( minIntervalMs: RENDERER_BREADCRUMB_COALESCE_MS, ...(origin ? { origin } : {}) }) - // Why: tracing every suppressed duplicate would preserve the same - // serialization and disk churn that breadcrumb coalescing removes. - if (coalesceResult) { + if (PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES.has(args.name)) { + // Why the raw data: every event already gets its own span, so folding the ring's + // running count in here would double-count in any span-stream total. + recordRendererBreadcrumbTrace(args.name, data) + } else if (coalesceResult) { + // Why gated: tracing every suppressed duplicate would preserve the same + // serialization and disk churn that breadcrumb coalescing removes. recordRendererBreadcrumbTrace( args.name, coalesceResult.suppressedSinceLast > 0 diff --git a/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts new file mode 100644 index 00000000000..e3823f0b313 --- /dev/null +++ b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts @@ -0,0 +1,128 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { + clearCrashBreadcrumbsForTest, + getCrashBreadcrumbSnapshot, + recordCrashBreadcrumb +} from '../crash-reporting/crash-breadcrumb-store' +import { recordRendererBreadcrumbFromRenderer } from './crash-reporting-renderer-breadcrumbs' + +type SpanOptions = { attributes: Record } +const startSpanMock = vi.fn((_name: string, _options: SpanOptions) => ({ end: () => {} })) +vi.mock('../observability/tracer', () => ({ + startSpan: (name: string, options: SpanOptions) => startSpanMock(name, options) +})) + +const WEDGE_BREADCRUMB = 'terminal_replay_guard_wedged_release' + +/** Reattach-path shape: identity-bearing (`tabIdHash`, optionally `ptyId`). */ +function emitReattachWedge(pane: number, withPtyId = false): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { + paneId: pane, + leafIdHash: `leaf${String(pane).padStart(5, '0')}`, + tabIdHash: `tab${String(pane).padStart(6, '0')}`, + worktreeIdHash: 'caa15fa9', + ...(withPtyId ? { ptyId: `…@@pty-${pane}` } : {}) + } + }) +} + +/** Restore-path shape (restoreScrollbackBuffers): no tabIdHash, no ptyId. */ +function emitRestoreWedge(pane: number): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { paneId: pane, leafIdHash: `leaf${String(pane).padStart(5, '0')}` } + }) +} + +function wedgeCrumbs(): ReturnType { + return getCrashBreadcrumbSnapshot().filter((entry) => entry.name === WEDGE_BREADCRUMB) +} + +function wedgeSpanCount(): number { + return startSpanMock.mock.calls.filter( + (call) => call[1].attributes['breadcrumb.name'] === WEDGE_BREADCRUMB + ).length +} + +beforeEach(() => { + startSpanMock.mockClear() +}) + +afterEach(() => { + clearCrashBreadcrumbsForTest() +}) + +// One mount/reveal/wake transition expires every in-flight replay write at once, so +// the burst reaches the 30-slot ring as N distinct entries. Field span streams measure +// bursts of 26 in 0.96s and 62 over 85s. No captured report in the 09-02 corpus shows +// a ring that actually drained — all nine bursts predate their report's ring window — +// so this bounds a demonstrated hazard, not an observed loss, and must not cost the +// durable span evidence that did carry those bursts. +describe('replay-guard wedge burst against the fixed-size breadcrumb ring', () => { + it('costs one ring slot per call site and preserves the pre-crash trail', () => { + for (let index = 0; index < 10; index += 1) { + recordCrashBreadcrumb(`pre_crash_evidence_${index}`, { index }) + } + + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + const snapshot = getCrashBreadcrumbSnapshot() + expect(snapshot.filter((entry) => entry.name.startsWith('pre_crash_evidence_'))).toHaveLength( + 10 + ) + expect(wedgeCrumbs()).toHaveLength(1) + }) + + it('carries the burst multiplicity into the ring as suppressedSinceLast', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + // 26 emissions: one owns the slot, 25 fold into it. + expect(wedgeCrumbs()[0]?.data?.suppressedSinceLast).toBe(25) + }) + + // The 121-event field corpus lives entirely in the renderer.breadcrumb span stream, + // and the ring is cleared by the restart that usually precedes the crash report, so + // ring coalescing must not suppress the per-event spans. + it('still emits one durable span per wedge event', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + expect(wedgeSpanCount()).toBe(26) + // Why no count on the span: one span per event already carries the multiplicity. + expect( + startSpanMock.mock.calls.some((call) => + JSON.stringify(call[1]).includes('suppressedSinceLast') + ) + ).toBe(false) + }) + + // Bundle 8907a508 mixes restore-path (identity-less) and reattach-path crumbs in one + // window; name-only keying would report only the last one's shape. + it('keeps restore-path and reattach-path call sites in separate slots', () => { + emitRestoreWedge(1) + emitRestoreWedge(2) + emitReattachWedge(3) + emitReattachWedge(4, true) + + const crumbs = wedgeCrumbs() + expect(crumbs).toHaveLength(3) + expect(crumbs.map((crumb) => Boolean(crumb.data?.tabIdHash))).toEqual([false, true, true]) + expect(crumbs.map((crumb) => Boolean(crumb.data?.ptyId))).toEqual([false, false, true]) + }) + + it('bounds a many-pane burst to one slot within a call site', () => { + for (let pane = 0; pane < 40; pane += 1) { + emitReattachWedge(pane, pane % 2 === 0) + } + + expect(wedgeCrumbs()).toHaveLength(2) + }) +}) diff --git a/src/main/ipc/filesystem-allowed-roots.test.ts b/src/main/ipc/filesystem-allowed-roots.test.ts new file mode 100644 index 00000000000..f94c99c5fdb --- /dev/null +++ b/src/main/ipc/filesystem-allowed-roots.test.ts @@ -0,0 +1,372 @@ +import { mkdir, mkdtemp, realpath, rm, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import type * as RepoWorktrees from '../repo-worktrees' +import { listRepoWorktreeGraph } from '../repo-worktrees' +import type * as ProjectGroupsModule from '../../shared/project-groups' +import { buildProjectGroupChildIndex, getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { isPathInsideOrEqual } from '../../shared/cross-platform-path' +import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import type { FolderWorkspace } from '../../shared/folder-workspace-types' +import type { ProjectGroup } from '../../shared/project-group-types' +import type { Project } from '../../shared/project-types' +import type { Repo } from '../../shared/repo-types' +import { getAllowedRoots } from './filesystem-allowed-roots' +import { authorizeExternalPath, resolveAuthorizedPath } from './filesystem-auth' +import { invalidateAuthorizedRootsCache } from './registered-worktree-roots-cache' +import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' + +vi.mock('../repo-worktrees', async () => { + const actual = await vi.importActual('../repo-worktrees') + return { ...actual, listRepoWorktreeGraph: vi.fn(async () => []) } +}) + +vi.mock('../../shared/project-groups', async () => { + const actual = await vi.importActual('../../shared/project-groups') + return { + ...actual, + buildProjectGroupChildIndex: vi.fn(actual.buildProjectGroupChildIndex), + getProjectGroupSubtreeIds: vi.fn(actual.getProjectGroupSubtreeIds) + } +}) + +type StoreFixture = { + repos: Repo[] + projects: Project[] + projectGroups: ProjectGroup[] + folderWorkspaces: FolderWorkspace[] + workspaceDir?: string +} + +type StoreCallCounts = { + getRepos: number + getProjects: number + getProjectGroups: number + getFolderWorkspaces: number +} + +function makeCountingStore(fixture: StoreFixture): { store: Store; counts: StoreCallCounts } { + const counts: StoreCallCounts = { + getRepos: 0, + getProjects: 0, + getProjectGroups: 0, + getFolderWorkspaces: 0 + } + const store = { + getRepos: () => { + counts.getRepos += 1 + // Match the real store, which rehydrates fresh repo objects on every read. + return fixture.repos.map((repo) => ({ ...repo })) + }, + getProjects: () => { + counts.getProjects += 1 + return fixture.projects.map((project) => ({ ...project })) + }, + getProjectGroups: () => { + counts.getProjectGroups += 1 + return fixture.projectGroups.map((group) => ({ ...group })) + }, + getFolderWorkspaces: () => { + counts.getFolderWorkspaces += 1 + return fixture.folderWorkspaces.map((workspace) => ({ ...workspace })) + }, + getSettings: () => ({ nestWorkspaces: false, workspaceDir: fixture.workspaceDir ?? '' }) + } as unknown as Store + return { store, counts } +} + +/** + * The pre-change `getAllowedRoots` algorithm, kept verbatim so the equivalence test compares the + * new root list against the old one rather than against a hand-written expectation. + */ +function referenceAllowedRoots(store: Store): string[] { + const scopeStore = store as unknown as { + getRepos: () => Repo[] + getProjectGroups?: () => ProjectGroup[] + getFolderWorkspaces?: () => FolderWorkspace[] + getSettings: () => { workspaceDir?: string; nestWorkspaces?: boolean } + } + const localRepos = scopeStore.getRepos().filter((repo) => !repo.connectionId) + const settings = scopeStore.getSettings() + + const scopeRepos = scopeStore.getRepos() + const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const isRemoteOnly = ( + folderPath: string, + projectGroupId: string, + connectionId: string | null | undefined + ): boolean => { + if (connectionId) { + return true + } + const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const candidates = scopeRepos.filter( + (repo) => + (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || + isPathInsideOrEqual(folderPath, repo.path) + ) + return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) + } + const folderScopeRoots: string[] = [] + for (const group of projectGroups) { + if (group.parentPath && !isRemoteOnly(group.parentPath, group.id, group.connectionId)) { + folderScopeRoots.push(resolve(group.parentPath)) + } + } + for (const workspace of scopeStore.getFolderWorkspaces?.() ?? []) { + const connectionId = + workspace.connectionId ?? + projectGroups.find((group) => group.id === workspace.projectGroupId)?.connectionId ?? + null + if (!isRemoteOnly(workspace.folderPath, workspace.projectGroupId, connectionId)) { + folderScopeRoots.push(resolve(workspace.folderPath)) + } + } + + const roots = [...localRepos.map((repo) => resolve(repo.path)), ...folderScopeRoots] + if (settings.workspaceDir) { + if (localRepos.length === 0) { + roots.push(resolve(settings.workspaceDir)) + } else { + for (const repo of localRepos) { + roots.push( + resolve( + computeWorkspaceRoot( + repo.path, + getWorktreePathSettings(repo, settings as never, getWorktreeMirrorDistro(store, repo)) + ) + ) + ) + } + } + } + return roots +} + +function makeRepo(overrides: Partial & Pick): Repo { + return { + displayName: overrides.id, + badgeColor: '#000000', + addedAt: 1, + kind: 'git', + ...overrides + } +} + +function makeGroup(overrides: Partial & Pick): ProjectGroup { + return { + name: overrides.id, + parentPath: null, + parentGroupId: null, + createdFrom: 'folder-scan', + tabOrder: 0, + isCollapsed: false, + color: null, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +function makeWorkspace( + overrides: Partial & Pick +): FolderWorkspace { + return { + projectGroupId: 'group-root', + name: overrides.id, + comment: '', + linkedTask: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 1, + lastActivityAt: 1, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +/** Repos, nested groups, folder workspaces (one not a git worktree), and an SSH repo. */ +function makeMixedFixture(): StoreFixture { + const repos = [ + makeRepo({ id: 'repo-local', path: '/repos/app', projectGroupId: 'group-root' }), + makeRepo({ id: 'repo-nested', path: '/repos/nested', projectGroupId: 'group-child' }), + makeRepo({ id: 'repo-folder', path: '/folders/plain', kind: 'folder' }), + makeRepo({ + id: 'repo-ssh', + path: '/remote/app', + connectionId: 'ssh-1', + projectGroupId: 'group-remote' + }) + ] + const projectGroups = [ + makeGroup({ id: 'group-root', parentPath: '/folders/root' }), + makeGroup({ id: 'group-child', parentGroupId: 'group-root', parentPath: '/folders/child' }), + makeGroup({ id: 'group-grandchild', parentGroupId: 'group-child' }), + makeGroup({ id: 'group-remote', parentPath: '/remote/scope' }), + makeGroup({ id: 'group-connection', parentPath: '/remote/via-group', connectionId: 'ssh-1' }) + ] + const folderWorkspaces = [ + makeWorkspace({ id: 'ws-git', folderPath: '/folders/root/feature' }), + // Not a git worktree: a plain folder workspace under a folder-kind repo. + makeWorkspace({ + id: 'ws-plain', + folderPath: '/folders/plain/scratch', + projectGroupId: 'group-child' + }), + makeWorkspace({ id: 'ws-remote', folderPath: '/remote/ws', projectGroupId: 'group-remote' }), + makeWorkspace({ + id: 'ws-connection', + folderPath: '/remote/direct', + projectGroupId: 'group-connection' + }), + makeWorkspace({ + id: 'ws-unlinked', + folderPath: '/folders/unlinked', + projectGroupId: 'group-orphan' + }) + ] + const projects: Project[] = [ + { + id: 'project-1', + displayName: 'App', + badgeColor: '#000000', + sourceRepoIds: ['repo-local', 'repo-nested'], + createdAt: 1, + updatedAt: 1 + }, + { + id: 'project-2', + displayName: 'Folder', + badgeColor: '#000000', + sourceRepoIds: ['repo-folder'], + createdAt: 1, + updatedAt: 1 + } + ] + return { repos, projects, projectGroups, folderWorkspaces, workspaceDir: '/workspaces' } +} + +beforeEach(() => { + invalidateAuthorizedRootsCache() + vi.mocked(buildProjectGroupChildIndex).mockClear() + vi.mocked(getProjectGroupSubtreeIds).mockClear() +}) + +describe('getAllowedRoots', () => { + it('produces the same roots as the pre-change implementation', () => { + const { store } = makeCountingStore(makeMixedFixture()) + + expect(getAllowedRoots(store)).toEqual(referenceAllowedRoots(store)) + }) + + it('reads the store once and indexes project groups once per build', () => { + const fixture = makeMixedFixture() + const { store, counts } = makeCountingStore(fixture) + + getAllowedRoots(store) + + expect.soft(counts.getRepos).toBe(1) + expect.soft(counts.getProjectGroups).toBe(1) + expect.soft(counts.getFolderWorkspaces).toBe(1) + // Batched runtime resolution scans the project list once, not once per local repo. + expect.soft(counts.getProjects).toBe(1) + // The per-scope subtree walk no longer rebuilds the parent->children index. + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(1) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) +}) + +describe('resolveAuthorizedPath allowed-root reuse', () => { + let repoRoot: string + let outsideRoot: string + let store: Store + let counts: StoreCallCounts + + beforeEach(async () => { + repoRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-allowed-roots-')) + outsideRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-outside-')) + const fixture = makeMixedFixture() + fixture.repos = [makeRepo({ id: 'repo-local', path: repoRoot }), ...fixture.repos] + fixture.projects[0]!.sourceRepoIds = ['repo-local'] + ;({ store, counts } = makeCountingStore(fixture)) + }) + + afterEach(async () => { + await rm(repoRoot, { recursive: true, force: true }) + await rm(outsideRoot, { recursive: true, force: true }) + }) + + it('builds the allowed-root list once per call across repeated reads', async () => { + const dirPath = join(repoRoot, 'src') + await mkdir(dirPath) + await writeFile(join(dirPath, 'index.ts'), 'export {}\n') + const callCount = 5 + + for (let index = 0; index < callCount; index += 1) { + await resolveAuthorizedPath(dirPath, store) + await resolveAuthorizedPath(join(dirPath, 'index.ts'), store) + } + + const buildCount = callCount * 2 + // One build per authorization, not one per raw-path check plus one per realpath check. + expect.soft(counts.getFolderWorkspaces).toBe(buildCount) + expect.soft(counts.getRepos).toBe(buildCount) + expect.soft(counts.getProjects).toBe(buildCount) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(buildCount) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) + + // Why (both symlink cases): creating a symlink on Windows needs elevation or + // Developer Mode, so these would fail EPERM in setup rather than exercise the + // escape check. Every non-symlink case still runs there. + it.skipIf(process.platform === 'win32')( + 'still refuses a symlink that escapes every allowed root', + async () => { + const secret = join(outsideRoot, 'secret.txt') + await writeFile(secret, 'secret\n') + const escape = join(repoRoot, 'escape.txt') + await symlink(secret, escape) + + await expect(resolveAuthorizedPath(escape, store)).rejects.toThrow('Access denied') + expect(vi.mocked(listRepoWorktreeGraph)).toHaveBeenCalled() + } + ) + + it('builds no allowed-root list at all for a granted external path', async () => { + const external = join(outsideRoot, 'external.md') + await writeFile(external, 'notes\n') + authorizeExternalPath(external) + counts.getRepos = 0 + counts.getProjects = 0 + counts.getFolderWorkspaces = 0 + + for (let index = 0; index < 5; index += 1) { + await expect(resolveAuthorizedPath(external, store)).resolves.toBe(external) + } + + // The grant answers on its own; hoisting the snapshot must not turn zero builds into one per read. + expect.soft(counts.getRepos).toBe(0) + expect.soft(counts.getProjects).toBe(0) + expect.soft(counts.getFolderWorkspaces).toBe(0) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).not.toHaveBeenCalled() + }) + + it.skipIf(process.platform === 'win32')( + 'still refuses a directory symlink that escapes every allowed root', + async () => { + const outsideDir = join(outsideRoot, 'nested') + await mkdir(outsideDir) + await writeFile(join(outsideDir, 'file.txt'), 'secret\n') + const escape = join(repoRoot, 'escape-dir') + await symlink(outsideDir, escape) + + await expect(resolveAuthorizedPath(join(escape, 'file.txt'), store)).rejects.toThrow( + 'Access denied' + ) + } + ) +}) diff --git a/src/main/ipc/filesystem-allowed-roots.ts b/src/main/ipc/filesystem-allowed-roots.ts index 3cb7fe4fa55..cef249430c6 100644 --- a/src/main/ipc/filesystem-allowed-roots.ts +++ b/src/main/ipc/filesystem-allowed-roots.ts @@ -1,9 +1,16 @@ import { resolve } from 'node:path' import type { Store } from '../persistence' import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' -import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import { + getWorktreeMirrorDistroForRuntime, + resolveLocalProjectRuntimesForRepos +} from '../project-runtime-git-options' import { isPathInsideOrEqual } from '../../shared/cross-platform-path' -import { getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { + buildProjectGroupChildIndex, + collectProjectGroupSubtreeIds, + type ProjectGroupChildIndex +} from '../../shared/project-groups' import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { ProjectGroup } from '../../shared/project-group-types' import type { Repo } from '../../shared/repo-types' @@ -11,18 +18,22 @@ import type { Repo } from '../../shared/repo-types' type FolderScopeStore = Pick & Partial> +// Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. +function filterLocalRepos(repos: readonly Repo[]): Repo[] { + return repos.filter((repo) => !repo.connectionId) +} + export function getLocalRepos(store: Store) { - // Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. - return store.getRepos().filter((repo) => !repo.connectionId) + return filterLocalRepos(store.getRepos()) } function getFolderScopeCandidateRepos( folderPath: string, projectGroupId: string, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): Repo[] { - const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const groupIds = collectProjectGroupSubtreeIds(childGroupIndex, projectGroupId) return repos.filter( (repo) => (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || @@ -34,13 +45,18 @@ function isRemoteOnlyFolderScope( folderPath: string, projectGroupId: string, connectionId: string | null | undefined, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): boolean { if (connectionId) { return true } - const candidates = getFolderScopeCandidateRepos(folderPath, projectGroupId, projectGroups, repos) + const candidates = getFolderScopeCandidateRepos( + folderPath, + projectGroupId, + childGroupIndex, + repos + ) return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) } @@ -55,16 +71,22 @@ function getFolderWorkspaceConnectionId( ) } -function getLocalFolderScopeRoots(store: Store): string[] { +function getLocalFolderScopeRoots(store: Store, repos: readonly Repo[]): string[] { const scopeStore = store as FolderScopeStore - const repos = scopeStore.getRepos() // Why: many filesystem tests use narrow Store doubles; folder scopes are additive. const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const childGroupIndex = buildProjectGroupChildIndex(projectGroups) const roots: string[] = [] for (const group of projectGroups) { if ( group.parentPath && - !isRemoteOnlyFolderScope(group.parentPath, group.id, group.connectionId, projectGroups, repos) + !isRemoteOnlyFolderScope( + group.parentPath, + group.id, + group.connectionId, + childGroupIndex, + repos + ) ) { roots.push(resolve(group.parentPath)) } @@ -75,7 +97,7 @@ function getLocalFolderScopeRoots(store: Store): string[] { workspace.folderPath, workspace.projectGroupId, getFolderWorkspaceConnectionId(workspace, projectGroups), - projectGroups, + childGroupIndex, repos ) ) { @@ -86,16 +108,19 @@ function getLocalFolderScopeRoots(store: Store): string[] { } export function getAllowedRoots(store: Store): string[] { - const localRepos = getLocalRepos(store) + // Why one read: `getRepos` rehydrates every repo, and this runs twice per filesystem IPC. + const repos = store.getRepos() + const localRepos = filterLocalRepos(repos) const settings = store.getSettings() const roots = [ ...localRepos.map((repo) => resolve(repo.path)), - ...getLocalFolderScopeRoots(store) + ...getLocalFolderScopeRoots(store, repos) ] if (settings.workspaceDir) { if (localRepos.length === 0) { roots.push(resolve(settings.workspaceDir)) } else { + const projectRuntimeByRepoId = resolveLocalProjectRuntimesForRepos(store, localRepos) for (const repo of localRepos) { roots.push( resolve( @@ -104,7 +129,11 @@ export function getAllowedRoots(store: Store): string[] { // Why enriched here too: placement has to agree with the create // flow, or renderer file access is denied for a worktree Orca // just put on the WSL side. - getWorktreePathSettings(repo, settings, getWorktreeMirrorDistro(store, repo)) + getWorktreePathSettings( + repo, + settings, + getWorktreeMirrorDistroForRuntime(projectRuntimeByRepoId.get(repo.id)) + ) ) ) ) diff --git a/src/main/ipc/filesystem-auth.ts b/src/main/ipc/filesystem-auth.ts index 122617845ed..894e39945c1 100644 --- a/src/main/ipc/filesystem-auth.ts +++ b/src/main/ipc/filesystem-auth.ts @@ -43,7 +43,24 @@ export function authorizeExternalPath(targetPath: string): void { } catch {} } -export function isPathAllowed(targetPath: string, store: Store): boolean { +/** + * One allowed-root list shared by every check in a single authorization. + * + * Lazy so a path already covered by an external grant still builds nothing at all, the way it did + * before the list was hoisted out of the individual checks. + */ +type AllowedRootsSnapshot = { get: () => readonly string[] } + +function createAllowedRootsSnapshot(store: Store): AllowedRootsSnapshot { + let roots: readonly string[] | undefined + return { get: () => (roots ??= getAllowedRoots(store)) } +} + +export function isPathAllowed( + targetPath: string, + store: Store, + allowedRoots?: AllowedRootsSnapshot +): boolean { const resolvedTarget = resolve(targetPath) if (authorizedExternalPaths.has(resolvedTarget)) { return true @@ -53,7 +70,9 @@ export function isPathAllowed(targetPath: string, store: Store): boolean { return true } } - return getAllowedRoots(store).some((root) => isDescendantOrEqual(resolvedTarget, root)) + return (allowedRoots?.get() ?? getAllowedRoots(store)).some((root) => + isDescendantOrEqual(resolvedTarget, root) + ) } export type ResolveAuthorizedPathOptions = { @@ -69,7 +88,10 @@ export async function resolveAuthorizedPath( options: ResolveAuthorizedPathOptions = {} ): Promise { const resolvedTarget = resolve(targetPath) - if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store))) { + // Why: the roots depend only on store state, not on the candidate path, so one snapshot serves + // every authorization below; each candidate is still checked against it in full. + const allowedRoots = createAllowedRootsSnapshot(store) + if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store, { allowedRoots }))) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) } @@ -80,14 +102,15 @@ export async function resolveAuthorizedPath( realParent = await realpath(dirname(resolvedTarget)) } catch (error) { if (isENOENT(error)) { - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } throw error } const candidateTarget = resolve(realParent, basename(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -100,7 +123,8 @@ export async function resolveAuthorizedPath( const realTarget = resolve(await realpath(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(realTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -110,11 +134,15 @@ export async function resolveAuthorizedPath( if (!isENOENT(error)) { throw error } - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } } -async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store): Promise { +async function resolveAuthorizedMissingPath( + resolvedTarget: string, + store: Store, + allowedRoots: AllowedRootsSnapshot +): Promise { let existingAncestor = resolvedTarget const missingSegments: string[] = [] @@ -124,7 +152,8 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store const candidateTarget = resolve(realAncestor, ...missingSegments) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -148,9 +177,9 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store async function isPathAllowedIncludingRegisteredWorktrees( targetPath: string, store: Store, - options: { canonicalSourcePath?: string } = {} + options: { canonicalSourcePath?: string; allowedRoots?: AllowedRootsSnapshot } = {} ): Promise { - if (isPathAllowed(targetPath, store)) { + if (isPathAllowed(targetPath, store, options.allowedRoots)) { return true } @@ -158,7 +187,14 @@ async function isPathAllowedIncludingRegisteredWorktrees( return true } - if (await isPathAllowedByCanonicalAllowedRoot(targetPath, options.canonicalSourcePath, store)) { + if ( + await isPathAllowedByCanonicalAllowedRoot( + targetPath, + options.canonicalSourcePath, + store, + options.allowedRoots + ) + ) { return true } @@ -178,12 +214,13 @@ async function isPathAllowedIncludingRegisteredWorktrees( async function isPathAllowedByCanonicalAllowedRoot( targetPath: string, sourcePath: string | undefined, - store: Store + store: Store, + allowedRoots?: AllowedRootsSnapshot ): Promise { if (!sourcePath) { return false } - for (const root of getAllowedRoots(store)) { + for (const root of allowedRoots?.get() ?? getAllowedRoots(store)) { const resolvedRoot = resolve(root) if (!isDescendantOrEqual(sourcePath, resolvedRoot)) { continue diff --git a/src/main/ipc/orca-profile-auth-status-broadcast.ts b/src/main/ipc/orca-profile-auth-status-broadcast.ts new file mode 100644 index 00000000000..b5ac8483943 --- /dev/null +++ b/src/main/ipc/orca-profile-auth-status-broadcast.ts @@ -0,0 +1,15 @@ +import { BrowserWindow } from 'electron' +import { ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL } from '../../shared/orca-profiles' + +export function broadcastOrcaProfileAuthStatusChanged(): void { + for (const window of BrowserWindow.getAllWindows()) { + if (window.isDestroyed()) { + continue + } + try { + window.webContents.send(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL) + } catch { + // A renderer can disappear between isDestroyed() and send(). + } + } +} diff --git a/src/main/ipc/orca-profiles.ts b/src/main/ipc/orca-profiles.ts index bf1bf666270..480c6f350f9 100644 --- a/src/main/ipc/orca-profiles.ts +++ b/src/main/ipc/orca-profiles.ts @@ -45,6 +45,8 @@ import { signOutCurrentOrcaProfile } from '../orca-profiles/profile-cloud-service' import { registerOrcaProfileOrgMemberHandlers } from './orca-profile-org-members-handlers' +import { onOrcaCloudSessionInvalidated } from '../orca-profiles/profile-cloud-session-invalidation' +import { broadcastOrcaProfileAuthStatusChanged } from './orca-profile-auth-status-broadcast' type RegisterOrcaProfileHandlersOptions = { onBeforeRelaunch?: () => void | Promise @@ -178,6 +180,12 @@ export function registerOrcaProfileHandlers( getCurrentOrcaProfileAuthStatus(getProfileUserDataPath()) ) + // Why: a background refresh can revoke the session with no renderer request in + // flight, so push the change instead of waiting for the next pane to ask. + // Why not options.onAuthMutation: that hook drives the relay coordinator, which + // is the caller that just failed the refresh — re-entering it here would be a loop. + onOrcaCloudSessionInvalidated(broadcastOrcaProfileAuthStatusChanged) + ipcMain.handle( 'orcaProfiles:createLocal', (_event, args?: CreateLocalOrcaProfileArgs): CreateLocalOrcaProfileResult => { diff --git a/src/main/ipc/pty/ipc/inspect.ts b/src/main/ipc/pty/ipc/inspect.ts index c13c4242fea..041abaa7854 100644 --- a/src/main/ipc/pty/ipc/inspect.ts +++ b/src/main/ipc/pty/ipc/inspect.ts @@ -170,7 +170,10 @@ export function installPtyInspectIpcHandlers(deps: { ipcMain.handle( 'pty:inspectProcess', - async (_event, args: { id: string; expectedIncarnationId?: string }) => { + async ( + _event, + args: { id: string; expectedIncarnationId?: string; scanChildProcesses?: boolean } + ) => { // Why: same routing hazard as pty:hasPty — an unroutable id must read as client-only unverifiable, not as a local-provider answer or a raised IPC error. if (typeof args?.id !== 'string' || !args.id || args.id.startsWith('remote:')) { return clientOnlyUnverifiableInspection('terminal_gone') @@ -182,10 +185,14 @@ export function installPtyInspectIpcHandlers(deps: { if (!hasPtyProviderForInspection(args.id)) { return clientOnlyUnverifiableInspection('terminal_gone') } - return args.expectedIncarnationId - ? inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id, { - expectedIncarnationId: args.expectedIncarnationId - }) + const options = { + ...(args.expectedIncarnationId + ? { expectedIncarnationId: args.expectedIncarnationId } + : {}), + ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}) + } + return Object.keys(options).length > 0 + ? inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id, options) : inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id) } ) diff --git a/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts b/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts new file mode 100644 index 00000000000..86e16121f5c --- /dev/null +++ b/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts @@ -0,0 +1,134 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { TERMINAL_INPUT_CHUNK_MAX_BYTES } from '../../../../shared/terminal-input' +import { agentSessionPtyWriteGate } from '../../../runtime/agent-session-pty-write-gate' +import { ptyOwnership } from '../provider/ownership-state' +import { createPtyWriteInput } from './write-input' + +const PTY_ID = 'pty-chunk-yield' + +const { provider } = vi.hoisted(() => ({ provider: { write: vi.fn() } })) + +vi.mock('../provider/registry', () => ({ + tryGetProviderForPty: (id: string) => (id === PTY_ID ? provider : undefined) +})) + +const realSetImmediate = globalThis.setImmediate +const realReadmit = agentSessionPtyWriteGate.readmit.bind(agentSessionPtyWriteGate) +const THREE_CHUNK_INPUT = 'x'.repeat(TERMINAL_INPUT_CHUNK_MAX_BYTES * 2 + 8) + +const mainWindow = { + isDestroyed: () => false, + webContents: { isDestroyed: () => false, send: vi.fn() } +} + +/** Resolves after `turns` real check-phase passes; never touches the (faked) timer queue. */ +function afterImmediateTurns(turns: number): Promise<'stalled'> { + return new Promise((resolve) => { + const step = (remaining: number): void => { + if (remaining === 0) { + resolve('stalled') + return + } + realSetImmediate(() => step(remaining - 1)) + } + step(turns) + }) +} + +function createWriteInput(): ReturnType['writePtyInput'] { + return createPtyWriteInput({ + mainWindow: mainWindow as never, + clearHiddenRendererResizeOutput: vi.fn() + }).writePtyInput +} + +beforeEach(() => { + ptyOwnership.set(PTY_ID, null) + provider.write.mockReset() + mainWindow.webContents.send.mockReset() + // Why: only setTimeout is faked. A setTimeout(0) yield would stall the write forever here, + // while a setImmediate yield still runs in Node's check phase — the race below is deterministic. + vi.useFakeTimers({ toFake: ['setTimeout'] }) +}) + +afterEach(() => { + ptyOwnership.delete(PTY_ID) + vi.restoreAllMocks() + vi.useRealTimers() +}) + +describe('chunked pty write yield', () => { + it('yields between chunks via setImmediate, not a timer, and readmits before each later chunk', async () => { + const events: string[] = [] + provider.write.mockImplementation((_id: string, data: string) => { + events.push(`write:${data.length}`) + }) + vi.spyOn(agentSessionPtyWriteGate, 'readmit').mockImplementation((...args) => { + events.push('readmit') + return realReadmit(...args) + }) + vi.spyOn(globalThis, 'setImmediate').mockImplementation(((callback: () => void) => { + events.push('yield') + return realSetImmediate(callback) + }) as typeof setImmediate) + + const outcome = await Promise.race([ + createWriteInput()({ id: PTY_ID, data: THREE_CHUNK_INPUT }), + afterImmediateTurns(50) + ]) + + expect(outcome).toBe(true) + expect(vi.getTimerCount()).toBe(0) + expect(events).toEqual([ + `write:${TERMINAL_INPUT_CHUNK_MAX_BYTES}`, + 'yield', + 'readmit', + `write:${TERMINAL_INPUT_CHUNK_MAX_BYTES}`, + 'yield', + 'readmit', + 'write:8' + ]) + expect(mainWindow.webContents.send).not.toHaveBeenCalled() + }) + + it('does not yield for input that fits in a single chunk', async () => { + const immediate = vi.spyOn(globalThis, 'setImmediate') + + const outcome = createWriteInput()({ + id: PTY_ID, + data: 'x'.repeat(TERMINAL_INPUT_CHUNK_MAX_BYTES) + }) + + expect(outcome).toBe(true) + expect(provider.write).toHaveBeenCalledTimes(1) + expect(immediate).not.toHaveBeenCalled() + }) + + it('stops after a yield once readmission refuses', async () => { + const readmit = vi.spyOn(agentSessionPtyWriteGate, 'readmit').mockReturnValue({ + admitted: false, + refusal: { + code: 'agent_session_checkpoint_stale', + sessionId: 'session-1', + ownerRuntimeKind: null, + handoffStage: null, + ownerPid: null, + runtimeFence: null + } + }) + + const outcome = await Promise.race([ + createWriteInput()({ id: PTY_ID, data: THREE_CHUNK_INPUT }), + afterImmediateTurns(50) + ]) + + expect(outcome).toBe(false) + expect(provider.write).toHaveBeenCalledTimes(1) + expect(readmit).toHaveBeenCalledTimes(1) + expect(mainWindow.webContents.send).toHaveBeenCalledWith( + 'pty:writeUnavailable', + expect.objectContaining({ id: PTY_ID }) + ) + }) +}) diff --git a/src/main/ipc/pty/ipc/write-input.ts b/src/main/ipc/pty/ipc/write-input.ts index f27604dc9f8..6713e49000b 100644 --- a/src/main/ipc/pty/ipc/write-input.ts +++ b/src/main/ipc/pty/ipc/write-input.ts @@ -152,7 +152,9 @@ export function createPtyWriteInput(deps: { first = false provider.write(id, chunk.value) if (!nextChunk.done) { - await new Promise((resolve) => setTimeout(resolve, 0)) + // setImmediate, not setTimeout(0): the yield exists to let abort/data callbacks run + // between chunks, and a clamped timer tick per 16 KiB is pure latency. + await new Promise((resolve) => setImmediate(resolve)) } chunk = nextChunk nextChunk = chunks.next() diff --git a/src/main/ipc/pty/runtime/queried-host-kinds.test.ts b/src/main/ipc/pty/runtime/queried-host-kinds.test.ts new file mode 100644 index 00000000000..1f7ce2459d8 --- /dev/null +++ b/src/main/ipc/pty/runtime/queried-host-kinds.test.ts @@ -0,0 +1,60 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { parseExecutionHostId } from '../../../../shared/execution-host' +import { sshProviders } from '../provider/registry' +import { listProcessesWithHostScopeFromRuntimeController } from './inventory-operations' +import type { PtyRuntimeControllerDeps } from './controller-deps' + +/** + * `hostScopeCensusIsComplete` discounts a `runtime:` host in `omittedHostIds` on the strength of + * one fact about this process: it has no paired-runtime PTY provider, so it never queried that + * host and never owed it coverage. This file pins the producer side of that fact. + * + * What it catches: a new branch here that spells a queried host `runtime:`. Every id this + * function emits is built by `toSshExecutionHostId` or is `LOCAL_EXECUTION_HOST_ID`, so a third + * shape is the observable form of "a runtime host can now answer an inventory" — at which point + * the client predicate would start calling a genuine gap complete. + * + * What it does NOT catch, so do not lean on it: a runtime-backed transport registered under an + * SSH connection id still reports as `ssh:` and passes, which is fine — the predicate only + * discounts the `runtime:` spelling. The consolidation moving the SSH path onto orcad is expected + * to look exactly like that. The other route into `queriedHostIds` is separately fenced to + * `kind === 'ssh'` in `orca-runtime-refresh-pty-worktree-records-with-controller-inventory.ts`. + */ +describe('the hosts a PTY inventory can report having queried', () => { + afterEach(() => { + sshProviders.clear() + }) + + it('emits only local and ssh spellings, never a paired-runtime one', async () => { + const listProcesses = vi.fn(async () => []) + sshProviders.set('box-1', { listProcesses } as never) + // A connection id shaped like an environment uuid still has to come back `ssh:`; the spelling + // is what the gate keys on, so a `runtime:` id appearing here is the breakage that matters. + sshProviders.set('a2478221-1d5c-4603-b8bf-b6b728eac9df', { listProcesses } as never) + + const { hostIds } = await listProcessesWithHostScopeFromRuntimeController({ + runtime: null + } as unknown as PtyRuntimeControllerDeps) + + expect(hostIds).toContain('ssh:a2478221-1d5c-4603-b8bf-b6b728eac9df') + expect(new Set(hostIds.map((hostId) => parseExecutionHostId(hostId)?.kind))).toEqual( + new Set(['local', 'ssh']) + ) + }) + + it('drops a provider that threw rather than reporting its host as queried', async () => { + sshProviders.set('box-live', { listProcesses: vi.fn(async () => []) } as never) + sshProviders.set('box-down', { + listProcesses: vi.fn(async () => { + throw new Error('relay unavailable') + }) + } as never) + + const { hostIds } = await listProcessesWithHostScopeFromRuntimeController({ + runtime: { markPtyLivenessUnverifiable: vi.fn() } + } as unknown as PtyRuntimeControllerDeps) + + expect(hostIds).toContain('ssh:box-live') + expect(hostIds).not.toContain('ssh:box-down') + }) +}) diff --git a/src/main/ipc/worktrees-test-module-mocks.ts b/src/main/ipc/worktrees-test-module-mocks.ts index f31f6ae1dd6..a1925d1ce72 100644 --- a/src/main/ipc/worktrees-test-module-mocks.ts +++ b/src/main/ipc/worktrees-test-module-mocks.ts @@ -115,6 +115,7 @@ export const gitWorktreeModuleMock = () => ({ listWorktrees: listWorktreesMock, listWorktreesStrict: listWorktreesMock, listWorktreesSharedStrict: listWorktreesMock, + listWorktreesSharedStrictAllowingTrueEmpty: listWorktreesMock, describeCreatedWorktree: describeCreatedWorktreeMock, parseWorktreeList: parseWorktreeListMock, assertWorktreeCleanForRemoval: assertWorktreeCleanForRemovalMock, diff --git a/src/main/ipc/worktrees-windows.test.ts b/src/main/ipc/worktrees-windows.test.ts index 4d4115264a3..7a54200d9f8 100644 --- a/src/main/ipc/worktrees-windows.test.ts +++ b/src/main/ipc/worktrees-windows.test.ts @@ -74,6 +74,7 @@ vi.mock('../git/worktree', () => ({ listWorktrees: listWorktreesMock, listWorktreesStrict: listWorktreesMock, listWorktreesSharedStrict: listWorktreesMock, + listWorktreesSharedStrictAllowingTrueEmpty: listWorktreesMock, describeCreatedWorktree: vi.fn().mockResolvedValue(undefined), assertWorktreeCleanForRemoval: assertWorktreeCleanForRemovalMock, addWorktree: addWorktreeMock, diff --git a/src/main/ipc/worktrees/listing/detected-provider-listing.ts b/src/main/ipc/worktrees/listing/detected-provider-listing.ts index 25c08d262fb..ec5e7606e45 100644 --- a/src/main/ipc/worktrees/listing/detected-provider-listing.ts +++ b/src/main/ipc/worktrees/listing/detected-provider-listing.ts @@ -25,9 +25,12 @@ import { type DetectedWorktreeMetadataPrune, type DetectedWorktreeSideEffectToken } from './detected-worktree-scan-cache' -import { loggedWorktreeListFailures, warnOnce } from './worktree-listing-diagnostics' -import { readAllWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' -import { getRepoExecutionHostId } from '../../../../shared/execution-host' +import { + describeWorktreeScanFailure, + loggedWorktreeListFailures, + warnOnce +} from './worktree-listing-diagnostics' +import { readAllWorktreeMetaForRepo } from '../../../persistence/host-qualified-worktree-meta' export async function listDetectedWorktreesForCapturedRepo( store: Store, @@ -40,9 +43,7 @@ export async function listDetectedWorktreesForCapturedRepo( providerAbort?.signal.aborted ? ({ providerAbortStatus: providerAbort.status() } as const) : undefined - const allMeta = isFolderRepo(repo) - ? undefined - : readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) + const allMeta = isFolderRepo(repo) ? undefined : readAllWorktreeMetaForRepo(store, repo) // Why: only the disconnected fallbacks read this, so keep parseWorktreeId over the whole host snapshot // off the connected path entirely. let cachedSshWorktreeMetaIndex: SshWorktreeMetaIndex | undefined @@ -160,16 +161,25 @@ export async function listDetectedWorktreesForCapturedRepo( `[worktrees] failed to list detected worktrees for repo "${repo.displayName}" (${repo.id}) at ${repo.path}`, err ) + // Why: retention alone leaves inert rows with no explanation; the cause rides with the listing. + const unavailableReason = describeWorktreeScanFailure(err) if (repo.connectionId) { const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex()) return { repoId: repo.id, authoritative: false, source: 'metadata-fallback', - worktrees: buildDisconnectedDetectedWorktrees(store, repo, worktrees) + worktrees: buildDisconnectedDetectedWorktrees(store, repo, worktrees), + unavailableReason } } - return { repoId: repo.id, authoritative: false, source: 'metadata-fallback', worktrees: [] } + return { + repoId: repo.id, + authoritative: false, + source: 'metadata-fallback', + worktrees: [], + unavailableReason + } } } diff --git a/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts b/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts new file mode 100644 index 00000000000..4e436b8dcf1 --- /dev/null +++ b/src/main/ipc/worktrees/listing/detected-scan-failure-authority.test.ts @@ -0,0 +1,166 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { DetectedWorktreeListResult } from '../../../../shared/worktree/types' + +const gitExecFileAsyncMock = vi.hoisted(() => vi.fn()) + +vi.mock('electron', () => ({ + ipcMain: { handle: vi.fn(), removeHandler: vi.fn() }, + app: { getPath: () => '/tmp/orca-test' } +})) +vi.mock('../../../git/runner', async (importOriginal) => ({ + ...(await importOriginal>()), + gitExecFileAsync: gitExecFileAsyncMock +})) + +const { listDetectedWorktreesForCapturedRepo } = await import('./detected-provider-listing') +const { __resetDetectedWorktreeScanCacheForTests } = await import('./detected-worktree-scan-cache') +const { _resetWorktreeScanCacheForTests } = await import('../../../git/worktree-scan-cache') +const { isRegisteredWorktreePath, invalidateAuthorizedRootsCache } = + await import('../../registered-worktree-roots-cache') + +const REPO_PATH = '/workspace/repo' +const repo = { + id: 'repo-1', + path: REPO_PATH, + displayName: 'repo', + badgeColor: '#000', + addedAt: 0 +} as Repo + +const removeWorktreeLineage = vi.fn() + +function createStore() { + return { + getRepo: () => repo, + getRepos: () => [repo], + getProjects: () => [], + getSettings: () => ({}), + getAllWorktreeMeta: () => ({}), + getProjectHostSetups: () => [], + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getAllWorkspaceLineage: () => ({}), + removeWorktreeLineage, + captureNativeLocalWorktreeMetadataScanExpectation: () => undefined + } as never +} + +/** The field failure: wsl.exe exits 0xFFFFFFFF, says nothing on stderr, and git never ran. */ +function wslHostFailure(): Error { + return Object.assign(new Error('Command failed: wsl.exe -d kali-linux --exec sh -lc ...'), { + code: 4294967295, + stdout: 'Error code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\r\n', + stderr: '' + }) +} + +async function listDetected(): Promise { + const result = await listDetectedWorktreesForCapturedRepo(createStore(), repo, () => true) + return result as DetectedWorktreeListResult +} + +describe('detected worktree listing authority', () => { + beforeEach(() => { + gitExecFileAsyncMock.mockReset() + removeWorktreeLineage.mockReset() + __resetDetectedWorktreeScanCacheForTests() + _resetWorktreeScanCacheForTests() + invalidateAuthorizedRootsCache() + }) + + it('reports a failed scan as non-authoritative and prunes nothing', async () => { + gitExecFileAsyncMock.mockRejectedValue(wslHostFailure()) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.source).toBe('metadata-fallback') + expect(result.worktrees).toEqual([]) + // Why: the retained rows must carry the cause, or the user sees inert worktrees with no explanation. + expect(result.unavailableReason).toContain('Command failed: wsl.exe') + // The destructive halves of a fresh scan must not run against a listing that failed. + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(false) + expect(removeWorktreeLineage).not.toHaveBeenCalled() + }) + + it('surfaces the annotated wsl.exe diagnostic as the unavailable reason', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign( + new Error( + 'wsl.exe host failure (distro "kali-linux"): There is no distribution with the supplied name.\r\nError code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND\nCommand failed: wsl.exe -d kali-linux --exec sh -lc ...' + ), + { code: 4294967295, stdout: '', stderr: '' } + ) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.unavailableReason).toBe( + 'wsl.exe host failure (distro "kali-linux"): There is no distribution with the supplied name. Error code: Wsl/Service/WSL_E_DISTRO_NOT_FOUND' + ) + }) + + // Why (measured on a real Windows host): under WSL the spawn cwd is the interop directory, so a + // deleted guest repo fails as `bash: cd` exit 1 — not ENOENT — and must stay retained, not pruned. + it('retains a WSL repo whose guest directory is gone, and says why', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('bash: line 1: cd: /home/neil/repo: No such file or directory'), { + code: 1, + stdout: '', + stderr: 'bash: line 1: cd: /home/neil/repo: No such file or directory\n' + }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(false) + expect(result.unavailableReason).toContain('No such file or directory') + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(false) + expect(removeWorktreeLineage).not.toHaveBeenCalled() + }) + + it('keeps an empty listing authoritative when the path is not a Git repo', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('Command failed: git worktree list'), { + code: 128, + stderr: 'fatal: not a git repository (or any of the parent directories): .git\n' + }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.source).toBe('git') + expect(result.worktrees).toEqual([]) + expect(result.unavailableReason).toBeUndefined() + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(true) + }) + + it('keeps an empty listing authoritative when the repo path is gone', async () => { + gitExecFileAsyncMock.mockRejectedValue( + Object.assign(new Error('spawn git ENOENT'), { code: 'ENOENT', stderr: '' }) + ) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.source).toBe('git') + expect(result.worktrees).toEqual([]) + }) + + it('stays authoritative for a healthy scan', async () => { + gitExecFileAsyncMock.mockResolvedValue({ + stdout: `worktree ${REPO_PATH}\u0000HEAD abc\u0000branch refs/heads/main\u0000\u0000`, + stderr: '' + }) + + const result = await listDetected() + + expect(result.authoritative).toBe(true) + expect(result.worktrees.map((worktree) => worktree.path)).toEqual([REPO_PATH]) + expect(isRegisteredWorktreePath(REPO_PATH)).toBe(true) + }) +}) diff --git a/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts b/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts index b694bfbf11d..09b0156cb54 100644 --- a/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts +++ b/src/main/ipc/worktrees/listing/detected-worktree-scan-cache.ts @@ -3,7 +3,7 @@ import type { Store } from '../../../persistence/loading-store/store' import type { Repo } from '../../../../shared/repo-types' import { getLocalProjectWorktreeGitOptions } from '../../../project-runtime-git-options' import { isFolderRepo } from '../../../../shared/repo-kind' -import { listRepoWorktrees } from '../../../repo-worktrees' +import { listRepoWorktreesForDetectedScan } from '../../../repo-worktrees' import { getRegisteredWorktreeRootsRevision, registerWorktreeRootsForRepo @@ -111,7 +111,7 @@ export async function listDetectedGitWorktrees( const localWorktreeGitOptions = getLocalProjectWorktreeGitOptions(store, repo) if (repo.connectionId || isFolderRepo(repo)) { return { - gitWorktrees: await listRepoWorktrees(repo, localWorktreeGitOptions), + gitWorktrees: await listRepoWorktreesForDetectedScan(repo, localWorktreeGitOptions), fresh: true } } @@ -144,7 +144,7 @@ export async function listDetectedGitWorktrees( : undefined const scan: DetectedWorktreeScan = { invalidated: false, - promise: listRepoWorktrees(repo, localWorktreeGitOptions), + promise: listRepoWorktreesForDetectedScan(repo, localWorktreeGitOptions), sideEffectToken: { generation, authorizedRootsRevision }, hygieneDue, ...(metadataPruneExpectation diff --git a/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts b/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts index cb8c48833ef..655cca4fa77 100644 --- a/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts +++ b/src/main/ipc/worktrees/listing/detected-worktree-scan-hygiene-gate.test.ts @@ -10,7 +10,9 @@ const { listRepoWorktreesMock, pruneLineageMock, pruneMetadataMock, registerWork registerWorktreeRootsMock: vi.fn() })) -vi.mock('../../../repo-worktrees', () => ({ listRepoWorktrees: listRepoWorktreesMock })) +vi.mock('../../../repo-worktrees', () => ({ + listRepoWorktreesForDetectedScan: listRepoWorktreesMock +})) vi.mock('../../../project-runtime-git-options', () => ({ getLocalProjectWorktreeGitOptions: () => ({}) })) diff --git a/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts b/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts index d684461a381..4f3055c63a2 100644 --- a/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts +++ b/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts @@ -23,7 +23,10 @@ import { warnOnce } from './worktree-listing-diagnostics' import type { WorktreeIpcContext } from '../worktree-ipc-context' -import { readAllWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' +import { + readAllWorktreeMetaForHost, + readAllWorktreeMetaForRepo +} from '../../../persistence/host-qualified-worktree-meta' import type { WorktreeMeta } from '../../../../shared/worktree/meta-types' const WORKTREE_LIST_ALL_CONCURRENCY = 8 @@ -174,9 +177,7 @@ export function registerWorktreeCatalogHandlers(context: WorktreeIpcContext): vo if (!repo) { return [] } - const allMeta = repo.connectionId - ? readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) - : undefined + const allMeta = repo.connectionId ? readAllWorktreeMetaForRepo(store, repo) : undefined const sshWorktreeMetaIndex = repo.connectionId ? createSshWorktreeMetaIndex(Object.entries(allMeta ?? {})) : new Map() @@ -226,7 +227,7 @@ export function registerWorktreeCatalogHandlers(context: WorktreeIpcContext): vo }) } loggedWorktreeListFailures.delete(`${repo.id}:${repo.path}`) - const metadata = allMeta ?? readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) + const metadata = allMeta ?? readAllWorktreeMetaForRepo(store, repo) return buildDetectedGitWorktrees(store, repo, gitWorktrees, metadata) .filter((worktree) => worktree.visible) .map((worktree) => stampAndMergeVisibleDetectedWorktree(store, repo, worktree, metadata)) diff --git a/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts b/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts index 5c734d8bcd9..ecf8ab3abc1 100644 --- a/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts +++ b/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts @@ -9,7 +9,7 @@ import type { GitWorktreeInfo, DetectedWorktree, Worktree } from '../../../../sh import type { Store } from '../../../persistence/loading-store/store' import { getRepoExecutionHostId } from '../../../../shared/execution-host' import { - readWorktreeMetaForHost, + readWorktreeMetaForRepo, writeWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from '../../../worktree-metadata-ownership' @@ -159,7 +159,7 @@ export function buildDetectedGitWorktrees( const legacyMeta = allMeta === undefined ? store.getWorktreeMeta?.(worktreeId) : undefined const metaById = allMeta ?? (legacyMeta ? { [worktreeId]: legacyMeta } : {}) const meta = - readWorktreeMetaForHost(store, worktreeId, getRepoExecutionHostId(repo)) ?? + readWorktreeMetaForRepo(store, worktreeId, repo) ?? getRepoOwnedWorktreeMeta(repo, worktreeId, metaById, repoOwnerCount) const worktree = mergeWorktree(repo.id, gitWorktree, meta, repo.displayName) const detected = toDetectedWorktree({ diff --git a/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts b/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts index b17677ddfcc..3edb8efc1d7 100644 --- a/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts +++ b/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts @@ -4,7 +4,7 @@ import type { WorktreeMeta } from '../../../../shared/worktree/meta-types' import { getProjectHostSetupWorktreeMeta } from '../../../../shared/project-host-setup-lookup' import { getRepoExecutionHostId } from '../../../../shared/execution-host' import { - readWorktreeMetaForHost, + readWorktreeMetaForRepo, writeWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from '../../../worktree-metadata-ownership' @@ -44,7 +44,7 @@ export function resolveWorktreeMetaWithDiscoveryBackfill( // Why: the locator-keyed row is only a stand-in for a missing snapshot, so don't read it when we have one. const legacyMeta = allMeta === undefined ? store.getWorktreeMeta?.(worktreeId) : undefined const existing = - readWorktreeMetaForHost(store, worktreeId, executionHostId) ?? + readWorktreeMetaForRepo(store, worktreeId, repo) ?? getRepoOwnedWorktreeMeta( repo, worktreeId, diff --git a/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts b/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts index 128bb46dd28..56bbe5d4ad2 100644 --- a/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts +++ b/src/main/ipc/worktrees/listing/worktree-listing-diagnostics.ts @@ -14,3 +14,23 @@ export function warnOnce(keySet: Set, key: string, message: string, erro console.warn(message) } } + +const SCAN_FAILURE_REASON_MAX_CHARS = 240 + +/** + * The cause a retained-but-unscannable repo shows the user. The first two lines carry the + * classifier's summary plus its `Wsl/Service/WSL_E_*` code; everything after is the raw command. + */ +export function describeWorktreeScanFailure(error: unknown): string { + const message = error instanceof Error ? error.message : String(error) + const summary = message + .split(/\r?\n/) + .map((line) => line.trim()) + .filter((line) => line.length > 0) + .slice(0, 2) + .join(' ') + const reason = summary.length > 0 ? summary : 'Worktree scan failed with no diagnostic.' + return reason.length > SCAN_FAILURE_REASON_MAX_CHARS + ? `${reason.slice(0, SCAN_FAILURE_REASON_MAX_CHARS - 1)}…` + : reason +} diff --git a/src/main/native-chat/agent-session-journal/journal-blob-store.ts b/src/main/native-chat/agent-session-journal/journal-blob-store.ts deleted file mode 100644 index 9b02877cec7..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-blob-store.ts +++ /dev/null @@ -1,110 +0,0 @@ -// Content-addressed store for the remainder of a bounded payload. -// -// Blobs are named by their sha256, so writing the same output twice costs one -// file and re-import is idempotent. They live beside the journal (host-side -// per-workspace state, never inside the user's working tree) and share the -// epoch's retention: compaction prunes every blob no retained row references. - -import { mkdir, readFile, readdir, rm, stat } from 'node:fs/promises' -import { join } from 'node:path' -import { durableWriteTempPath, writeFileDurable } from '../../durable-file-write' - -export const JOURNAL_BLOB_DIR = 'blobs' -const DIGEST_PATTERN = /^[0-9a-f]{64}$/ - -/** A digest arrives back from a row on disk, so it is untrusted by the time it - * reaches the filesystem: anything but a bare sha256 could escape the store. */ -function blobPath(journalDir: string, digest: string): string | null { - return DIGEST_PATTERN.test(digest) ? join(journalDir, JOURNAL_BLOB_DIR, digest) : null -} - -/** Persist `payload` under its digest. Returns the digest so the caller can - * stamp it on the row it is about to append. */ -export async function putJournalBlob( - journalDir: string, - digest: string, - payload: string -): Promise { - const target = blobPath(journalDir, digest) - if (!target) { - throw new Error('refusing to write a journal blob under a name that is not a sha256 digest') - } - // Content addressing makes a rewrite pointless: identical digest, identical bytes. - if (await pathExists(target)) { - return digest - } - await mkdir(join(journalDir, JOURNAL_BLOB_DIR), { recursive: true }) - await writeFileDurable(durableWriteTempPath(target), target, payload) - return digest -} - -export async function readJournalBlob(journalDir: string, digest: string): Promise { - const source = blobPath(journalDir, digest) - if (!source) { - return null - } - try { - return await readFile(source, 'utf-8') - } catch { - return null - } -} - -/** Remove a blob written speculatively for a row that was rejected. */ -export async function removeJournalBlob(journalDir: string, digest: string): Promise { - const target = blobPath(journalDir, digest) - if (target) { - await rm(target, { force: true }) - } -} - -/** Drop every blob outside `retained`. Called from compaction, under the - * current lease fence, after the snapshot is durable — so a crash mid-prune - * leaves extra blobs rather than dangling references. */ -export async function pruneJournalBlobs( - journalDir: string, - retained: ReadonlySet -): Promise { - let removed = 0 - let names: string[] - try { - names = await readdir(join(journalDir, JOURNAL_BLOB_DIR)) - } catch { - return 0 - } - for (const name of names) { - if (retained.has(name)) { - continue - } - await rm(join(journalDir, JOURNAL_BLOB_DIR, name), { force: true }).catch(() => {}) - removed += 1 - } - return removed -} - -async function pathExists(path: string): Promise { - try { - await stat(path) - return true - } catch { - return false - } -} - -export async function journalBlobFileSize( - journalDir: string, - digest: string -): Promise { - const target = blobPath(journalDir, digest) - if (!target) { - return null - } - try { - return (await stat(target)).size - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return null - } - throw error - } -} diff --git a/src/main/native-chat/agent-session-journal/journal-close-retry.ts b/src/main/native-chat/agent-session-journal/journal-close-retry.ts new file mode 100644 index 00000000000..e73be0a59b8 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-close-retry.ts @@ -0,0 +1,68 @@ +// Where a journal goes when its close REJECTS. +// +// `AgentSessionJournal.close()` is deliberately retryable: a rejection does not +// release the handle, and a second call is a real second attempt. Every caller +// that did `close().catch(() => undefined)` and then threw or overwrote its map +// entry defeated that contract — the store became unreachable with its SQLite +// connection still open. On POSIX that is a silent leak; on Windows the open +// handle blocks renaming or removing the journal directory outright. +// +// So a rejected close hands the journal here instead of dropping it, and host +// teardown retries everything this holds. Retention is bounded by construction: +// an entry leaves the set the moment its close fulfils, and a journal can be +// retained only once because the set is keyed by identity. + +/** Everything the registry needs; `AgentSessionJournal` satisfies it. */ +export type RetryableJournalClose = { + close: () => Promise + readonly directory: string +} + +export class JournalCloseRetryRegistry { + private readonly retained = new Set() + + /** Directories still held by a journal whose close has not fulfilled. */ + get pendingDirectories(): string[] { + return [...this.retained].map((journal) => journal.directory) + } + + /** + * Close it. Returns true when the handle is actually released; on a rejection + * the journal is RETAINED for `retryAll` and the rejection is returned rather + * than thrown, because every caller of this is already unwinding a different + * failure it must not lose. + */ + async closeOrRetain( + journal: RetryableJournalClose + ): Promise<{ closed: boolean; error?: unknown }> { + try { + await journal.close() + this.retained.delete(journal) + return { closed: true } + } catch (error) { + this.retained.add(journal) + return { closed: false, error } + } + } + + /** Retry every retained close. Ones that fulfil are dropped; ones that reject + * stay retained and their rejections are returned for the caller to report. */ + async retryAll(): Promise { + const entries = [...this.retained] + const failures: unknown[] = [] + for (const journal of entries) { + const result = await this.closeOrRetain(journal) + if (!result.closed) { + failures.push(result.error) + } + } + return failures + } +} + +/** + * Process-wide, because ownership of these handles is process-wide: the attach + * path, the recovery wrapper and runtime teardown are separate call trees that + * must all be able to reach the same orphan. + */ +export const agentSessionJournalCloseRetries = new JournalCloseRetryRegistry() diff --git a/src/main/native-chat/agent-session-journal/journal-compaction.ts b/src/main/native-chat/agent-session-journal/journal-compaction.ts deleted file mode 100644 index b9600a4f4e9..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-compaction.ts +++ /dev/null @@ -1,206 +0,0 @@ -// Retention and compaction. -// -// The snapshot carries the retained tail with it, so publishing both is ONE -// atomic write and there is no window where the folded state exists without the -// rows a reconnecting client still needs. Truncating the log afterwards is -// idempotent: a crash before it leaves the log a superset of the tail. -// -// The retained tail must cover the longest reconnect window Orca supports, or a -// client that was merely asleep gets a full snapshot reload instead of a resume. - -import { blobDigestsInBody, renderJournalState, type JournalReducerState } from './journal-reducer' -import { pruneJournalBlobs } from './journal-blob-store' -import { - rewriteJournalLog, - writeJournalSnapshotFile, - type JournalSnapshotFile -} from './journal-log-file' -import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' -import type { JournalRow } from './journal-row-schema' -import { AgentSessionJournalError } from './journal-write-guards' -import { assertJournalPhysicalCapacity, journalDirectoryBytes } from './journal-physical-quota' - -export type JournalCompactionPolicy = { - /** Always keep at least this many rows, however old they are. */ - minTailRows: number - /** Keep every row observed within this window. */ - retainTailMs: number - /** - * `window` honours `retainTailMs` outright. `budget-pressure` lets it yield: - * the alternative is refusing the user's writes until the window ages out, - * and the tail is only a resume optimization — compaction folds every shed - * row into the snapshot before truncating the log, so a client that loses - * its resume point reloads instead of losing conversation. Defaults to - * `window`. - */ - retention?: 'window' | 'budget-pressure' -} - -/** Two hours of tail comfortably covers a phone that slept through a commute, - * which is the longest reconnect Orca resumes rather than reloads. */ -export const DEFAULT_JOURNAL_COMPACTION_POLICY: JournalCompactionPolicy = { - minTailRows: 512, - retainTailMs: 2 * 60 * 60 * 1000 -} - -export type JournalCompactionResult = { - tailRows: JournalRow[] - compactedThrough: number - oldestSequence: number -} - -export async function compactJournal(input: { - journalDir: string - /** Parent quota root when compacting an in-directory staging journal. */ - physicalQuotaRoot?: string - state: JournalReducerState - tailRows: readonly JournalRow[] - policy?: JournalCompactionPolicy - now: number - maxSessionBytes: number - sessionId?: string -}): Promise { - const policy = input.policy ?? DEFAULT_JOURNAL_COMPACTION_POLICY - const retained = retainTail(input.tailRows, policy, input.now) - const rendered = renderJournalState(input.state) - const compactedThrough = input.state.lastSequence - - const snapshot: JournalSnapshotFile = { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: input.state.epoch, - compactedThrough, - highestFence: input.state.highestFence, - items: rendered.items, - submissions: rendered.submissions, - receipts: [...input.state.receipts.values()].map((receipt) => ({ - clientMessageId: receipt.clientMessageId, - providerItemId: receipt.providerItemId, - epoch: receipt.cursor.epoch, - sequence: receipt.cursor.sequence, - acceptedAt: receipt.acceptedAt - })), - aliases: [...input.state.aliases.entries()].map(([providerItemId, itemId]) => ({ - providerItemId, - itemId - })), - tombstones: [...input.state.tombstones.entries()].map(([itemId, revision]) => ({ - itemId, - revision - })), - appliedSettlementIds: [...input.state.appliedSettlementIds], - tail: retained - } - - const snapshotBytes = Buffer.byteLength(JSON.stringify(snapshot), 'utf8') - if (snapshotBytes > input.maxSessionBytes) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal snapshot reached its ${input.maxSessionBytes}-byte bound` - ) - } - - const sessionId = input.sessionId ?? input.state.sessionId - const quotaRoot = input.physicalQuotaRoot ?? input.journalDir - const retainedLogBytes = retained.reduce( - (total, row) => total + Buffer.byteLength(JSON.stringify(row), 'utf8') + 1, - 0 - ) - // Durable writes keep the old final alongside the new temp until rename. - // Reserve the complete compaction peak up front so a later copy cannot leave - // a half-published snapshot/log pair when the quota is tight. - await assertJournalPhysicalCapacity({ - journalDir: quotaRoot, - sessionId, - maxBytes: input.maxSessionBytes, - peakAdditionalBytes: snapshotBytes + retainedLogBytes - }) - await writeJournalSnapshotFile(input.journalDir, snapshot) - await rewriteJournalLog(input.journalDir, retained) - // Blobs are pruned last: a crash before this leaks bytes, whereas pruning - // first would strand a snapshot pointing at a payload that no longer exists. - // Recompute from exactly what the durable snapshot and retained log carry; - // this preserves reused/pre-existing blobs while allowing stale payloads to - // be pruned safely after both files are published. - const retainedDigests = new Set() - for (const item of snapshot.items) { - blobDigestsInBody(item.body, retainedDigests) - } - for (const row of retained) { - if (row.kind === 'item') { - blobDigestsInBody(row.body, retainedDigests) - } else if (row.kind === 'lifecycle-batch') { - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - blobDigestsInBody(mutation.body, retainedDigests) - } - } - } - } - await pruneJournalBlobs(input.journalDir, retainedDigests) - - if ((await journalDirectoryBytes(quotaRoot)) > input.maxSessionBytes) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${sessionId} exceeds its physical bound after compaction` - ) - } - - return { - tailRows: retained, - compactedThrough, - oldestSequence: retained[0]?.seq ?? compactedThrough + 1 - } -} - -function retainTail( - rows: readonly JournalRow[], - policy: JournalCompactionPolicy, - now: number -): JournalRow[] { - if (rows.length <= policy.minTailRows) { - return [...rows] - } - const floor = now - policy.retainTailMs - const byAge = rows.findIndex((row) => row.ts >= floor) - const byCount = rows.length - policy.minTailRows - const start = byAge === -1 ? byCount : Math.min(byAge, byCount) - if (policy.retention !== 'budget-pressure') { - return rows.slice(start) - } - // Halve rather than empty: the newer half keeps live clients resuming, and - // shedding at least one row guarantees the append that triggered this makes - // progress instead of latching the session read-only. - return rows.slice(Math.max(start, Math.ceil(rows.length / 2))) -} - -/** Only when the retention window would actually drop rows: inside it, - * compaction rewrites an identical log, and doing that per append is a full - * state serialization on the hot path. */ -export function journalTailIsReadyToCompact( - tailRows: readonly JournalRow[], - policy: JournalCompactionPolicy, - now: number -): boolean { - if (tailRows.length <= policy.minTailRows * 2) { - return false - } - return (tailRows[0]?.ts ?? now) < now - policy.retainTailMs -} - -/** The policy an append falls back to when the size bound would otherwise - * refuse it: both floors that normally protect the tail step aside. */ -export function budgetPressurePolicy(policy: JournalCompactionPolicy): JournalCompactionPolicy { - return { ...policy, minTailRows: 0, retention: 'budget-pressure' } -} - -/** Budget pressure may need to shed rows before the ordinary batching threshold. - * Pass a `budget-pressure` policy, or a tail wholly inside the retention - * window answers false and the size bound refuses every append until it ages - * out — two hours of a session the user cannot write to. */ -export function journalTailCanShedRows( - tailRows: readonly JournalRow[], - policy: JournalCompactionPolicy, - now: number -): boolean { - return retainTail(tailRows, policy, now).length < tailRows.length -} diff --git a/src/main/native-chat/agent-session-journal/journal-corruption-quarantine.ts b/src/main/native-chat/agent-session-journal/journal-corruption-quarantine.ts deleted file mode 100644 index e12d5b67a0a..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-corruption-quarantine.ts +++ /dev/null @@ -1,102 +0,0 @@ -// Corruption never deletes history. A journal that cannot be read end to end -// keeps its intact prefix live and moves the unreadable remainder aside, so the -// bytes stay on disk for inspection instead of being rebuilt into an empty epoch. - -import { readFile } from 'node:fs/promises' -import { join } from 'node:path' -import { - JOURNAL_SNAPSHOT_FILE, - quarantineJournalRemainder, - readJournalLog, - rewriteJournalLog -} from './journal-log-file' -import type { JournalRow } from './journal-row-schema' -import { assertJournalPhysicalCapacity } from './journal-physical-quota' - -/** Keep the readable prefix and set the unreadable suffix aside. */ -export async function quarantineCorruptSuffix( - journalDir: string, - retainedRows: readonly JournalRow[], - remainder: string | undefined, - quota?: { sessionId: string; maxBytes: number } -): Promise { - if (remainder) { - if (quota) { - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: Buffer.byteLength(remainder, 'utf8') - }) - } - await quarantineJournalRemainder(journalDir, remainder) - } - if (quota) { - const retainedBytes = retainedRows.reduce( - (total, row) => total + Buffer.byteLength(JSON.stringify(row), 'utf8') + 1, - 0 - ) - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: retainedBytes - }) - } - await rewriteJournalLog(journalDir, retainedRows) -} - -/** Copy everything aside before a read-only journal is rebuilt under a newer - * schema: those rows are unreadable to THIS build, not worthless. The - * snapshot is preserved as raw bytes — a future-version snapshot does not - * parse under this build's schema, and its bytes must survive verbatim. */ -export async function quarantineUnreadableSchema( - journalDir: string, - quota?: { sessionId: string; maxBytes: number } -): Promise { - const snapshot = await readSnapshotBytes(journalDir) - const log = await readJournalLog(journalDir) - const preserved = [ - snapshot ?? '', - log.rows.map((row) => JSON.stringify(row)).join('\n'), - log.remainder ?? '' - ] - .filter(Boolean) - .join('\n') - if (preserved) { - if (quota) { - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: Buffer.byteLength(preserved, 'utf8') - }) - } - await quarantineJournalRemainder(journalDir, preserved) - } -} - -async function readSnapshotBytes(journalDir: string): Promise { - try { - return await readFile(join(journalDir, JOURNAL_SNAPSHOT_FILE), 'utf-8') - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return null - } - throw error - } -} - -/** The disclosure row for lines that failed to parse. Skipped lines are lost - * rows; counting them silently is the drop this exists to prevent. */ -export function malformedRowsDisclosure(count: number): { - identity: { provider: 'orca'; clientMessageId: string } - body: { kind: 'status'; text: string } -} { - const plural = count === 1 ? '' : 's' - return { - // One stable identity, so a reopen upserts the same row instead of adding one. - identity: { provider: 'orca', clientMessageId: 'journal-malformed-lines' }, - body: { - kind: 'status', - text: `${count} journal line${plural} could not be read and ${count === 1 ? 'was' : 'were'} skipped` - } - } -} diff --git a/src/main/native-chat/agent-session-journal/journal-corruption-repair.test.ts b/src/main/native-chat/agent-session-journal/journal-corruption-repair.test.ts new file mode 100644 index 00000000000..978f076d6e9 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-corruption-repair.test.ts @@ -0,0 +1,265 @@ +// A repair drops what it cannot replay, and says so. +// +// Two things make a suffix unreplayable: a row this build cannot parse, and a +// sequence gap that makes every later row unanchored. The rejected suffix is +// DELETED; the load reports `corrupt`, and recovery rebuilds the epoch from +// provider history. Every case here asserts the same two halves: the live epoch +// holds only the replayable prefix, AND the epoch stays anchored so nothing +// replays a repaired journal as a clean timeline. + +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import type Database from '../../sqlite/sync-database' +import { openJournalDatabase } from './journal-database' +import { journalDatabaseFile } from './journal-paths' +import { parseJournalRow, type JournalRow } from './journal-row-schema' +import { loadJournal } from './journal-open' +import type { openAgentSessionJournal } from './journal-store-factory' +import { createTrackedJournalOpener } from './journal-store-test-open' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} + +let root: string +let clock = 1_000 +const journals = createTrackedJournalOpener() + +function tick(): number { + clock += 1 + return clock +} + +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } +} + +function body(value: string): AgentJournalItemBody { + return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } +} + +function open(overrides: Partial[0]> = {}) { + return journals.open({ + identity: IDENTITY, + journalDir: root, + now: tick, + mintEpoch: () => `epoch-${clock}`, + ...overrides + }) +} + +async function withJournalDatabase(run: (db: Database.Database) => void): Promise { + const opened = openJournalDatabase(journalDatabaseFile(root)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} + +/** The row replay anchors on, parsed exactly as replay parses it. */ +function firstLiveRow(): Promise { + let row: JournalRow | null = null + return withJournalDatabase((db) => { + const stored = db.prepare('SELECT row_json FROM journal_rows ORDER BY seq LIMIT 1').get() as + | { row_json: string } + | undefined + const parsed = stored ? parseJournalRow(stored.row_json) : null + row = parsed?.ok ? parsed.row : null + }).then(() => row) +} + +function liveSequences(): Promise { + let sequences: number[] = [] + return withJournalDatabase((db) => { + sequences = ( + db.prepare('SELECT seq FROM journal_rows ORDER BY seq').all() as { seq: number }[] + ).map((row) => row.seq) + }).then(() => sequences) +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-repair-')) + clock = 1_000 +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('a malformed row', () => { + it('keeps the readable prefix live and drops the rest of the epoch', async () => { + const journal = await open() + await journal.appendItem(item(0), body('readable'), { fence: 1 }) + await journal.appendItem(item(1), body('unreadable'), { fence: 1 }) + await journal.appendItem(item(2), body('after the fault'), { fence: 1 }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('UPDATE journal_rows SET row_json = ? WHERE seq = ?').run('{"not":"a row"}', 3) + }) + + const reopened = await open() + expect(reopened.repair.malformedRows).toBe(1) + // 1..2 is the surviving prefix; 3 is the disclosure the repair appends. + expect(await liveSequences()).toEqual([1, 2, 3]) + }) + + it('discloses the line it could not read', async () => { + const journal = await open() + await journal.appendItem(item(0), body('readable'), { fence: 1 }) + await journal.appendItem(item(1), body('later'), { fence: 1 }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('UPDATE journal_rows SET row_json = ? WHERE seq = ?').run('}{', 2) + }) + + const reopened = await open() + const disclosure = reopened + .snapshot() + .items.map((entry) => entry.body) + .find((entry) => entry.kind === 'status') + expect(disclosure).toMatchObject({ kind: 'status' }) + expect(disclosure && 'text' in disclosure ? disclosure.text : '').toContain( + '1 journal line could not be read' + ) + }) +}) + +describe('a sequence gap', () => { + it('drops every row after the hole and reports the epoch corrupt', async () => { + const journal = await open() + for (let ordinal = 0; ordinal < 5; ordinal += 1) { + await journal.appendItem(item(ordinal), body(`m${ordinal}`), { fence: 1 }) + } + await journal.close() + // Sequence 1 is the epoch row, so the items occupy 2..6. Removing 4 leaves + // 5 and 6 valid but unanchored. + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(4) + }) + + const reopened = await open() + expect(await liveSequences()).toEqual([1, 2, 3]) + expect(reopened.repair).toEqual({ malformedRows: 0 }) + expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([body('m0'), body('m1')]) + }) + + // The prefix survives, so there is no emptied epoch to re-anchor and — a gap + // costing no malformed row — no disclosure either. Without a durable marker + // the next probe reads a contiguous anchored prefix and calls it clean, and + // the rows the repair deleted are never asked for again. + it('still reports corrupt on the next probe, with the deleted suffix unrebuilt', async () => { + const journal = await open() + for (let ordinal = 0; ordinal < 5; ordinal += 1) { + await journal.appendItem(item(ordinal), body(`m${ordinal}`), { fence: 1 }) + } + await journal.close() + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(4) + }) + + const repaired = await open() + await repaired.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true }) + + // Same policy the emptied-epoch repair takes: a session that writes into the + // epoch owns it, and a later import must not replace rows the user has seen. + const writable = await open() + await writable.appendItem(item(9), body('typed after the repair'), { fence: 1 }) + await writable.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: false }) + }) + + // The disclosure is the repair talking about itself, not the session writing: + // counting it as content would retire the marker the instant it was raised. + it('is not settled by the repair disclosure it appends for a malformed row', async () => { + const journal = await open() + for (let ordinal = 0; ordinal < 3; ordinal += 1) { + await journal.appendItem(item(ordinal), body(`m${ordinal}`), { fence: 1 }) + } + await journal.close() + await withJournalDatabase((db) => { + db.prepare('UPDATE journal_rows SET row_json = ? WHERE seq = ?').run('}{', 3) + }) + + const repaired = await open() + expect(repaired.repair.malformedRows).toBe(1) + await repaired.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true }) + }) +}) + +describe('a missing epoch row', () => { + // Sequence 1 is the anchor for the whole epoch. Validating from the first row + // that HAPPENS to remain declares the leftovers contiguous, and replay then + // renders a repaired journal as a clean timeline. + it('rejects the whole surviving range rather than declaring it contiguous', async () => { + const journal = await open() + await journal.appendItem(item(0), body('anchor'), { fence: 1 }) + await journal.appendSubmission({ + clientMessageId: 'client-message-1', + payloadFingerprint: 'fingerprint-1', + body: { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'the user typed this' }] + }, + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'client-message-1', + state: 'accepted', + providerIdentity: item(1), + fence: 1 + }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(1) + }) + + const reopened = await open() + expect(reopened.repair).toEqual({ malformedRows: 0 }) + expect(reopened.snapshot().items).toEqual([]) + + // The epoch cannot be left row-less. An ordinary append would then take + // sequence 1, and replay would call that non-epoch row a clean timeline. + expect(await liveSequences()).toEqual([1]) + const anchor = await firstLiveRow() + expect(anchor).toMatchObject({ kind: 'epoch', reason: 'unreconcilable_prefix' }) + }) + + // The repair epoch is a placeholder for history it could not rebuild. Left + // clean it would end automatic recovery: the provider transcript is never + // consulted again and the dropped rows never come back. + it('keeps asking for provider history until the epoch has content of its own', async () => { + const journal = await open() + await journal.appendItem(item(0), body('anchor'), { fence: 1 }) + await journal.close() + await withJournalDatabase((db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(1) + }) + + const repaired = await open() + await repaired.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: true }) + + // A session that writes into the epoch owns it: its own rows are not a + // repair placeholder, and a later import must not replace them. + const writable = await open() + await writable.appendItem(item(1), body('typed after the repair'), { fence: 1 }) + await writable.close() + expect(loadJournal(root, IDENTITY.sessionId)).toMatchObject({ corrupt: false }) + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts b/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts index c58f1b57bc8..a7f54a14a4c 100644 --- a/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts @@ -21,7 +21,7 @@ import { type ProviderHistoryItem, type ProviderHistoryWindow } from './journal-submission-reconciler' -import { openAgentSessionJournal } from './journal-store-factory' +import { createTrackedJournalOpener } from './journal-store-test-open' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -52,8 +52,10 @@ function userMessage(text: string): AgentJournalMessageItem { return { kind: 'message', role: 'user', blocks: [{ type: 'text', text }] } } +const journals = createTrackedJournalOpener() + async function open() { - return openAgentSessionJournal({ + return journals.open({ identity: IDENTITY, journalDir: root, now: tick, @@ -89,6 +91,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -195,14 +198,8 @@ describe('crash between provider accept and journal commit', () => { ).toEqual([]) }) - it('keeps the receipt after the row that minted it was compacted away', async () => { - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - now: tick, - mintEpoch: () => `epoch-${clock}`, - compaction: { minTailRows: 1, retainTailMs: 0 } - }) + it('keeps the receipt across a reopen', async () => { + const journal = await open() await journal.appendSubmission({ clientMessageId: 'cm_1', payloadFingerprint: digestPayload('kept'), @@ -215,7 +212,7 @@ describe('crash between provider accept and journal commit', () => { providerIdentity: ACCEPTED_IDENTITY, fence: 1 }) - await journal.compact() + await journal.close() const reopened = await open() expect(reopened.receiptFor('cm_1')?.providerItemId).toBe(agentJournalItemKey(ACCEPTED_IDENTITY)) diff --git a/src/main/native-chat/agent-session-journal/journal-cursor.ts b/src/main/native-chat/agent-session-journal/journal-cursor.ts index bdec8adfd19..bd43623ae49 100644 --- a/src/main/native-chat/agent-session-journal/journal-cursor.ts +++ b/src/main/native-chat/agent-session-journal/journal-cursor.ts @@ -77,7 +77,7 @@ export function findSequenceGap( export function readJournalSince( source: { state: { epoch: string; lastSequence: number; oldestSequence: number } - tailRows: readonly JournalRow[] + rowsAfter: (afterSequence: number) => JournalRow[] readOnly: boolean }, cursor: AgentJournalCursor, @@ -92,7 +92,7 @@ export function readJournalSince( } return { ok: true, - rows: source.tailRows.filter((row) => row.seq > resume.afterSequence), + rows: source.rowsAfter(resume.afterSequence), cursor: currentCursor() } } diff --git a/src/main/native-chat/agent-session-journal/journal-database-schema.ts b/src/main/native-chat/agent-session-journal/journal-database-schema.ts new file mode 100644 index 00000000000..37675fbcd8e --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-database-schema.ts @@ -0,0 +1,37 @@ +// Table shape for one session's journal database. +// +// `journal_rows` is the append-only log; `journal_sessions` is the derived +// projection, upserted in the SAME transaction as every row insert so the live +// epoch and the rows that belong to it can never disagree. `journal_repairs` +// carries at most one row per session: the standing demand for a rebuild a +// partial repair leaves behind (see journal-repair-marker.ts). + +/** DB shape version, carried in `PRAGMA user_version`. Independent of the row + * body version (`JournalRow.v`): a newer build can change either alone. + * v2 added `journal_repairs`; a build without it would replay a partially + * repaired journal as clean, so it must latch read-only rather than write. */ +export const JOURNAL_DB_SCHEMA_VERSION = 2 + +export function createJournalTablesSql(): string { + return ` +CREATE TABLE IF NOT EXISTS journal_rows ( + session_id TEXT NOT NULL, + epoch TEXT NOT NULL, + seq INTEGER NOT NULL, + ts INTEGER NOT NULL, + row_json TEXT NOT NULL, + PRIMARY KEY (session_id, epoch, seq) +); +CREATE TABLE IF NOT EXISTS journal_sessions ( + session_id TEXT PRIMARY KEY, + epoch TEXT NOT NULL, + updated_at INTEGER NOT NULL +); +CREATE TABLE IF NOT EXISTS journal_repairs ( + session_id TEXT PRIMARY KEY, + epoch TEXT NOT NULL, + content_from INTEGER NOT NULL, + repaired_at INTEGER NOT NULL +); +` +} diff --git a/src/main/native-chat/agent-session-journal/journal-database.test.ts b/src/main/native-chat/agent-session-journal/journal-database.test.ts new file mode 100644 index 00000000000..358c42c8d55 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-database.test.ts @@ -0,0 +1,201 @@ +import { mkdtemp, rm, stat } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import Database from '../../sqlite/sync-database' +import { + JOURNAL_BUSY_TIMEOUT_MS, + journalPragmaNumber, + openJournalDatabase +} from './journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' +import { journalDatabaseFile } from './journal-paths' +import { + deleteAllJournalRows, + deleteJournalRowSuffix, + insertJournalRow, + readJournalEpochRows, + readJournalRowsAfter, + readJournalSessionEpoch, + upsertJournalSessionRow +} from './journal-row-table' +import type { JournalRow } from './journal-row-schema' +import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' + +let root: string +let dbPath: string + +function epochRow(seq: number, epoch = 'epoch-1'): JournalRow { + return { + kind: 'epoch', + reason: 'session_created', + providerHandle: { kind: 'codex', threadId: 'thread-1' }, + v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + epoch, + seq, + fence: 0, + ts: 1_700_000_000_000 + seq + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-db-')) + dbPath = journalDatabaseFile(root) +}) + +afterEach(async () => { + vi.restoreAllMocks() + await rm(root, { recursive: true, force: true }) +}) + +describe('journal database open', () => { + it('creates both tables and reads back every load-bearing pragma', () => { + const opened = openJournalDatabase(dbPath) + try { + const tables = opened.db + .prepare("SELECT name FROM sqlite_master WHERE type = 'table' ORDER BY name") + .all() + .map((entry) => (entry as { name: string }).name) + expect(tables).toContain('journal_rows') + expect(tables).toContain('journal_sessions') + expect(opened.db.pragma('journal_mode', { simple: true })).toBe('wal') + expect(journalPragmaNumber(opened.db, 'synchronous')).toBe(2) + expect(journalPragmaNumber(opened.db, 'busy_timeout')).toBe(JOURNAL_BUSY_TIMEOUT_MS) + expect(journalPragmaNumber(opened.db, 'foreign_keys')).toBe(1) + expect(journalPragmaNumber(opened.db, 'user_version')).toBe(JOURNAL_DB_SCHEMA_VERSION) + expect(opened.readOnly).toBe(false) + } finally { + opened.db.close() + } + }) + + it('latches read-only on a future user_version without touching the file', async () => { + const seeded = openJournalDatabase(dbPath) + upsertJournalSessionRow(seeded.db, 'session-1', 'epoch-1', 1) + insertJournalRow(seeded.db, 'session-1', epochRow(1)) + seeded.db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 5}`) + seeded.db.close() + const before = await stat(dbPath) + + const latched = openJournalDatabase(dbPath) + try { + expect(latched.readOnly).toBe(true) + expect(journalPragmaNumber(latched.db, 'user_version')).toBe(JOURNAL_DB_SCHEMA_VERSION + 5) + expect(readJournalEpochRows(latched.db, 'session-1', 'epoch-1')).toHaveLength(1) + expect(() => latched.db.exec("INSERT INTO journal_sessions VALUES ('x', 'y', 1)")).toThrow() + } finally { + latched.db.close() + } + expect((await stat(dbPath)).size).toBe(before.size) + expect(journalPragmaNumber(openJournalDatabase(dbPath).db, 'user_version')).toBe( + JOURNAL_DB_SCHEMA_VERSION + 5 + ) + }) + + // Site 1: the raw connection is owned by the open call until it returns. + it('closes the raw connection when schema setup throws', async () => { + const failing = join(root, 'nested', 'journal.db') + expect(() => openJournalDatabase(failing)).toThrow() + await expect(stat(`${failing}-wal`)).rejects.toThrow() + await expect(rm(root, { recursive: true, force: true })).resolves.toBeUndefined() + root = await mkdtemp(join(tmpdir(), 'orca-journal-db-')) + }) +}) + +describe('journal row statements', () => { + it('serves replay, resume, discard and suffix truncation from the primary key', () => { + const opened = openJournalDatabase(dbPath) + try { + const { db } = opened + db.exec('BEGIN IMMEDIATE') + for (let seq = 1; seq <= 5; seq += 1) { + insertJournalRow(db, 'session-1', epochRow(seq)) + } + insertJournalRow(db, 'session-1', epochRow(1, 'epoch-old')) + upsertJournalSessionRow(db, 'session-1', 'epoch-1', 42) + db.exec('COMMIT') + + expect(readJournalSessionEpoch(db, 'session-1')).toBe('epoch-1') + expect(readJournalSessionEpoch(db, 'absent')).toBeNull() + expect(readJournalEpochRows(db, 'session-1', 'epoch-1').map((row) => row.seq)).toEqual([ + 1, 2, 3, 4, 5 + ]) + expect(readJournalRowsAfter(db, 'session-1', 'epoch-1', 3).map((row) => row.seq)).toEqual([ + 4, 5 + ]) + + // The rejected suffix leaves `journal_rows`, scoped to its own epoch. + expect(deleteJournalRowSuffix(db, 'session-1', 'epoch-1', 4)).toBe(2) + expect(readJournalEpochRows(db, 'session-1', 'epoch-1').map((row) => row.seq)).toEqual([ + 1, 2, 3 + ]) + expect(readJournalEpochRows(db, 'session-1', 'epoch-old')).toHaveLength(1) + + deleteAllJournalRows(db) + expect(readJournalEpochRows(db, 'session-1', 'epoch-1')).toHaveLength(0) + expect(readJournalEpochRows(db, 'session-1', 'epoch-old')).toHaveLength(0) + expect(readJournalSessionEpoch(db, 'session-1')).toBe('epoch-1') + } finally { + opened.db.close() + } + }) + + it('refuses a duplicate sequence inside one epoch', () => { + const opened = openJournalDatabase(dbPath) + try { + insertJournalRow(opened.db, 'session-1', epochRow(1)) + expect(() => insertJournalRow(opened.db, 'session-1', epochRow(1))).toThrow() + insertJournalRow(opened.db, 'session-1', epochRow(1, 'epoch-2')) + } finally { + opened.db.close() + } + }) + + it('upserts the session projection in place', () => { + const opened = openJournalDatabase(dbPath) + try { + upsertJournalSessionRow(opened.db, 'session-1', 'epoch-1', 1) + upsertJournalSessionRow(opened.db, 'session-1', 'epoch-2', 2) + expect(readJournalSessionEpoch(opened.db, 'session-1')).toBe('epoch-2') + expect( + opened.db.prepare('SELECT count(*) AS total FROM journal_sessions').get() + ).toMatchObject({ total: 1 }) + } finally { + opened.db.close() + } + }) +}) + +describe('schema creation', () => { + // Creating the tables outside the migration transaction left a v2-shaped + // database still reporting version 0, which an older build does not latch + // read-only: it stamps its own version on and writes through v1 SQL. + it('publishes no table until the version bump commits with it', () => { + const original = Database.prototype.pragma + const pragma = vi.spyOn(Database.prototype, 'pragma').mockImplementation(function ( + this: Database.Database, + sql: string, + options?: { simple?: boolean } + ) { + if (sql.startsWith('user_version =')) { + throw new Error('crash before the version is published') + } + return original.call(this, sql, options) + }) + + expect(() => openJournalDatabase(dbPath)).toThrow('crash before the version is published') + pragma.mockRestore() + + const inspected = new Database(dbPath) + try { + expect(inspected.pragma('user_version', { simple: true })).toBe(0) + expect( + inspected + .prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'journal_rows'") + .get() + ).toBeUndefined() + } finally { + inspected.close() + } + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-database.ts b/src/main/native-chat/agent-session-journal/journal-database.ts new file mode 100644 index 00000000000..6f1cd60733b --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-database.ts @@ -0,0 +1,80 @@ +// Opening one session's journal database. +// +// `PRAGMA user_version` is read FIRST, on a connection that has set no +// persistent pragma and run no DDL: a future-schema database must be left +// byte-identical, and `journal_mode = WAL` writes the file header. + +import Database from '../../sqlite/sync-database' +import { hardenSqliteDatabaseFiles } from '../../sqlite/harden-database-files' +import { createJournalTablesSql, JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' + +export const JOURNAL_BUSY_TIMEOUT_MS = 5000 + +export type OpenJournalDatabase = { + db: Database.Database + /** A newer `user_version` was met: this build reads and never writes. */ + readOnly: boolean +} + +export function journalPragmaNumber(db: Database.Database, name: string): number { + return Number(db.pragma(name, { simple: true }) ?? 0) +} + +export function openJournalDatabase(dbPath: string): OpenJournalDatabase { + const probe = new Database(dbPath) + let stored: number + try { + stored = journalPragmaNumber(probe, 'user_version') + } catch (error) { + probe.close() + throw error + } + if (stored > JOURNAL_DB_SCHEMA_VERSION) { + probe.close() + return { db: new Database(dbPath, { readonly: true, fileMustExist: true }), readOnly: true } + } + let transferred = false + try { + configureJournalPragmas(probe) + createJournalSchema(probe, stored) + hardenSqliteDatabaseFiles(dbPath) + const opened = { db: probe, readOnly: false } + transferred = true + return opened + } finally { + if (!transferred) { + probe.close() + } + } +} + +function configureJournalPragmas(db: Database.Database): void { + db.pragma('journal_mode = WAL') + db.pragma(`busy_timeout = ${JOURNAL_BUSY_TIMEOUT_MS}`) + db.pragma('foreign_keys = ON') + // Why FULL rather than the house NORMAL: the write-ahead submission row must + // survive a power loss before the adapter dispatches anything, and NORMAL in + // WAL mode does not fsync at commit. + db.pragma('synchronous = FULL') +} + +/** + * Table creation and the `user_version` bump are ONE transaction. Creating the + * tables first left a shaped database still reporting version 0, which an older + * build does not latch read-only: it stamped its own version on and wrote + * through SQL for a schema it did not have. + */ +function createJournalSchema(db: Database.Database, stored: number): void { + if (stored >= JOURNAL_DB_SCHEMA_VERSION) { + return + } + db.exec('BEGIN IMMEDIATE') + try { + db.exec(createJournalTablesSql()) + db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION}`) + db.exec('COMMIT') + } catch (error) { + db.exec('ROLLBACK') + throw error + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts b/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts index 087ac9e6a26..f428675fb2c 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-controller.ts @@ -2,28 +2,21 @@ import type { AgentJournalCursor, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import type { JournalCompactionPolicy } from './journal-compaction' -import { quarantineUnreadableSchema } from './journal-corruption-quarantine' +import type Database from '../../sqlite/sync-database' import { replaceJournalEpoch, type JournalReplacementItem } from './journal-epoch-replacement' import { publishNewEpoch } from './journal-epoch-rollover' import type { JournalLoad } from './journal-open' import type { AgentJournalEpochReason } from './journal-row-schema' -import { - assertJournalFence, - assertJournalWritable, - type JournalAppendBudget -} from './journal-write-guards' +import { assertJournalFence, assertJournalWritable } from './journal-write-guards' export class JournalEpochController { constructor( private readonly deps: { identity: AgentSessionJournalIdentity - journalDir: string - budget: JournalAppendBudget - compaction: JournalCompactionPolicy now: () => number mintEpoch: () => string serialize: (run: () => Promise) => Promise + database: () => { db: Database.Database } readOnly: () => boolean setReadOnly: (readOnly: boolean) => void highestFence: () => number @@ -32,33 +25,33 @@ export class JournalEpochController { } ) {} - async start(reason: AgentJournalEpochReason, fence: number): Promise { - this.deps.adopt( - await publishNewEpoch({ - journalDir: this.deps.journalDir, - sessionId: this.deps.identity.sessionId, - providerHandle: this.deps.identity.providerHandle, - epoch: this.deps.mintEpoch(), - reason, - fence, - now: this.deps.now(), - maxSessionBytes: this.deps.budget.maxSessionBytes - }) - ) + start(reason: AgentJournalEpochReason, fence: number): void { + publishNewEpoch({ + db: this.deps.database().db, + sessionId: this.deps.identity.sessionId, + providerHandle: this.deps.identity.providerHandle, + epoch: this.deps.mintEpoch(), + reason, + fence, + now: this.deps.now(), + onPublished: this.deps.adopt + }) } - async roll(reason: AgentJournalEpochReason, fence: number): Promise { - if (reason !== 'schema_unreadable') { + /** + * Every reason takes the same writable guard. A latched store refuses a roll + * like any other write, and `schema_unreadable` has no production caller. + * + * Serialized like every other write, so the discard cannot land between an + * admitted append's sequence assignment and its commit. + */ + roll(reason: AgentJournalEpochReason, fence: number): Promise { + return this.deps.serialize(async () => { assertJournalWritable(this.deps.readOnly(), this.deps.identity.sessionId) - } else if (this.deps.readOnly()) { - await quarantineUnreadableSchema(this.deps.journalDir, { - sessionId: this.deps.identity.sessionId, - maxBytes: this.deps.budget.maxSessionBytes - }) - } - await this.start(reason, fence) - this.deps.setReadOnly(false) - return this.deps.cursor() + this.start(reason, fence) + this.deps.setReadOnly(false) + return this.deps.cursor() + }) } replace( @@ -69,17 +62,15 @@ export class JournalEpochController { return this.deps.serialize(async () => { assertJournalWritable(this.deps.readOnly(), this.deps.identity.sessionId) assertJournalFence(fence, this.deps.highestFence()) - await replaceJournalEpoch({ - journalDir: this.deps.journalDir, + replaceJournalEpoch({ + db: this.deps.database().db, identity: this.deps.identity, reason, fence, items, - budget: this.deps.budget.fork(), - compaction: this.deps.compaction, now: this.deps.now, mintEpoch: this.deps.mintEpoch, - onSnapshotPublished: this.deps.adopt + onPublished: this.deps.adopt }) return this.deps.cursor() }) diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts index d4036755fda..96273366d5f 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.test.ts @@ -1,18 +1,19 @@ -import { mkdtemp, readdir, rm } from 'node:fs/promises' +// Republishing an epoch is ONE transaction. + +import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' import type { - AgentJournalItemBody, + AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { DEFAULT_JOURNAL_COMPACTION_POLICY } from './journal-compaction' +import { openJournalDatabase, type OpenJournalDatabase } from './journal-database' import { replaceJournalEpoch } from './journal-epoch-replacement' -import { putJournalBlob, readJournalBlob } from './journal-blob-store' -import { boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalDirectoryBytes } from './journal-physical-quota' -import { JournalAppendBudget } from './journal-write-guards' -import { openAgentSessionJournal } from './journal-store-factory' +import type { JournalLoad } from './journal-open' +import { journalDatabaseFile } from './journal-paths' +import { readJournalEpochRows, readJournalSessionEpoch } from './journal-row-table' +import { createTrackedJournalOpener } from './journal-store-test-open' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -24,150 +25,79 @@ const IDENTITY: AgentSessionJournalIdentity = { let root: string let clock = 1_000 - -beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-replace-')) - clock = 1_000 -}) - -afterEach(async () => { - await rm(root, { recursive: true, force: true }) -}) +let database: OpenJournalDatabase +const journals = createTrackedJournalOpener() function now(): number { clock += 1 return clock } -function toolBody(output: ReturnType): AgentJournalItemBody { - return { - kind: 'tool-call', - name: 'shell', - input: {}, - state: 'completed', - output - } +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } } -describe('journal epoch replacement', () => { - it('publishes one observable replacement and prunes stale root blobs afterward', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 8 } - const stalePayload = 'stale'.repeat(1_000) - const retainedPayload = 'retained'.repeat(1_000) - const stale = boundPayload(stalePayload, limits) - const retained = boundPayload(retainedPayload, limits) - const published: unknown[] = [] - await putJournalBlob(root, stale.digest, stalePayload) +function replace(input: { + items: Parameters[0]['items'] + onPublished?: (loaded: JournalLoad) => void +}): void { + replaceJournalEpoch({ + db: database.db, + identity: IDENTITY, + reason: 'legacy_import', + fence: 1, + items: input.items, + now, + mintEpoch: () => `epoch-${clock}`, + onPublished: input.onPublished ?? (() => undefined) + }) +} - await replaceJournalEpoch({ - journalDir: root, - identity: IDENTITY, - reason: 'handle_forked', - fence: 2, - items: [ - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 }, - body: toolBody(retained), - blobs: [{ digest: retained.digest, payload: retainedPayload }] - } - ], - budget: new JournalAppendBudget(IDENTITY.sessionId, { - ...limits, - maxSessionBytes: 512 * 1024 - }), - compaction: DEFAULT_JOURNAL_COMPACTION_POLICY, - now, - mintEpoch: () => 'epoch-new', - onSnapshotPublished: (loaded) => published.push(loaded) +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-replace-')) + clock = 1_000 + database = openJournalDatabase(journalDatabaseFile(root)) +}) + +afterEach(async () => { + try { + database.db.close() + } catch { + // Already closed by the case. + } + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('journal epoch replacement', () => { + it('publishes one observable replacement', () => { + const published: JournalLoad[] = [] + + replace({ + items: [{ identity: item(1), body: { kind: 'status', text: 'republished' } }], + onPublished: (loaded) => published.push(loaded) }) expect(published).toHaveLength(1) - expect(await readJournalBlob(root, stale.digest)).toBeNull() - expect(await readJournalBlob(root, retained.digest)).toBe(retainedPayload) - expect((published[0] as { sizeBytes: number }).sizeBytes).toBe( - await journalDirectoryBytes(root) - ) + const epoch = readJournalSessionEpoch(database.db, IDENTITY.sessionId) + expect(epoch).toBe(published[0]?.state.epoch) + expect(readJournalEpochRows(database.db, IDENTITY.sessionId, epoch ?? '')).toHaveLength(2) }) - it('keeps root blobs and reports no publication when replacement never becomes authoritative', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 8, maxSessionBytes: 6_000 } - const stalePayload = 'stale'.repeat(500) - const stale = boundPayload(stalePayload, limits) - const published: unknown[] = [] - await putJournalBlob(root, stale.digest, stalePayload) + it('discards every superseded row in the same transaction', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.appendItem(item(1), { kind: 'status', text: 'old' }, { fence: 1 }) + await journal.appendItem(item(2), { kind: 'status', text: 'older' }, { fence: 1 }) + const before = journal.epoch - await expect( - replaceJournalEpoch({ - journalDir: root, - identity: IDENTITY, - reason: 'handle_forked', - fence: 2, - items: [ - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 }, - body: { - kind: 'message', - role: 'assistant', - blocks: [{ type: 'text', text: 'x'.repeat(10_000) }] - } - } - ], - budget: new JournalAppendBudget(IDENTITY.sessionId, limits), - compaction: DEFAULT_JOURNAL_COMPACTION_POLICY, - now, - mintEpoch: () => 'epoch-new', - onSnapshotPublished: (loaded) => published.push(loaded) - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) + await journal.replaceEpochItems('legacy_import', 1, [ + { identity: item(9), body: { kind: 'status', text: 'republished' } } + ]) - expect(published).toHaveLength(0) - expect(await readJournalBlob(root, stale.digest)).toBe(stalePayload) - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - }) - - it('charges replacement blobs cumulatively and rolls back staging on quota refusal', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 8, maxSessionBytes: 7_000 } - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits, - autoCompact: false, - now, - mintEpoch: () => `epoch-${clock}` - }) - const existingPayload = 'existing'.repeat(250) - const existing = boundPayload(existingPayload, limits) - await journal.appendItemWithBlobs( - { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 }, - toolBody(existing), - [{ digest: existing.digest, payload: existingPayload }], - { fence: 1 } - ) - - const replacementPayload = 'replacement'.repeat(200) - const replacement = boundPayload(replacementPayload, limits) - const secondPayload = 'second'.repeat(200) - const second = boundPayload(secondPayload, limits) - await expect( - journal.replaceEpochItems('handle_forked', 2, [ - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 1 }, - body: toolBody(replacement), - blobs: [{ digest: replacement.digest, payload: replacementPayload }] - }, - { - identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 2 }, - body: toolBody(second), - blobs: [{ digest: second.digest, payload: secondPayload }] - } - ]) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toMatch(/^epoch-/) - expect(await readJournalBlob(root, existing.digest)).toBe(existingPayload) - expect(await readJournalBlob(root, replacement.digest)).toBeNull() - expect(await readJournalBlob(root, second.digest)).toBeNull() - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) + expect(journal.epoch).not.toBe(before) + expect(readJournalEpochRows(database.db, IDENTITY.sessionId, before)).toHaveLength(0) + expect(journal.snapshot().items.map((entry) => entry.body)).toEqual([ + { kind: 'status', text: 'republished' } + ]) }) }) diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts index 342c60c4f1d..9512d5e9fa1 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-replacement.ts @@ -1,298 +1,85 @@ -import { mkdir, mkdtemp, rm, stat } from 'node:fs/promises' -import { join } from 'node:path' -import { copyFileDurable } from '../../durable-file-write' +// Republishing a live item set into a fresh epoch. +// +// One transaction: discard every row, insert the epoch row plus the replacement +// items, move the session projection, and retire any repair marker — this +// republished history is exactly what the marker was holding out for. + import type { AgentJournalItemBody, AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { compactJournal, type JournalCompactionPolicy } from './journal-compaction' -import { JOURNAL_LOG_FILE, JOURNAL_SNAPSHOT_FILE, appendJournalRows } from './journal-log-file' -import { - applyJournalRow, - blobDigestsInBody, - createJournalReducerState, - referencedBlobDigests, - type JournalReducerState -} from './journal-reducer' -import { buildJournalItemRow, journalRowBase } from './journal-row-builders' -import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -import { journalRowByteLength } from './journal-row-schema' -import { assertJournalFence, type JournalAppendBudget } from './journal-write-guards' +import type Database from '../../sqlite/sync-database' import type { JournalLoad } from './journal-open' -import { assertJournalPhysicalCapacity, journalDirectoryBytes } from './journal-physical-quota' +import { clearJournalRepairMarker } from './journal-repair-marker' +import { applyJournalRow, createJournalReducerState } from './journal-reducer' +import { buildJournalItemRow, journalRowBase } from './journal-row-builders' import { - JOURNAL_BLOB_DIR, - journalBlobFileSize, - putJournalBlob, - pruneJournalBlobs, - removeJournalBlob -} from './journal-blob-store' + deleteAllJournalRows, + insertJournalRow, + upsertJournalSessionRow +} from './journal-row-table' +import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' +import { assertJournalFence } from './journal-write-guards' export type JournalReplacementItem = { identity: AgentJournalItemIdentity body: AgentJournalItemBody - blobs?: readonly { digest: string; payload: string }[] observedAt?: number } -export async function replaceJournalEpoch(input: { - journalDir: string +export function replaceJournalEpoch(input: { + db: Database.Database identity: AgentSessionJournalIdentity reason: AgentJournalEpochReason fence: number items: readonly JournalReplacementItem[] - budget: JournalAppendBudget - compaction: JournalCompactionPolicy now: () => number mintEpoch: () => string - onSnapshotPublished: (loaded: JournalLoad) => void -}): Promise { - const stagingDir = await mkdtemp(join(input.journalDir, '.epoch-replacement-')) - const stagedBlobDigests = new Set() - const publishedBlobDigests: string[] = [] - let snapshotPublished = false - let adoptionReported = false - let publishedLoad: JournalLoad | null = null + /** Called the instant the transaction commits, before any fallible follow-up. */ + onPublished: (loaded: JournalLoad) => void +}): void { + const epoch = input.mintEpoch() + const state = createJournalReducerState(input.identity.sessionId, epoch) + const epochRow: JournalRow = { + kind: 'epoch', + reason: input.reason, + providerHandle: input.identity.providerHandle, + ...journalRowBase(epoch, 1, input.fence, input.now()) + } + const rows: JournalRow[] = [epochRow] + applyJournalRow(state, epochRow) + for (const item of input.items) { + const row = buildJournalItemRow({ + state, + identity: item.identity, + body: item.body, + seq: state.lastSequence + 1, + fence: input.fence, + ts: item.observedAt ?? input.now() + }) + assertJournalFence(row.fence, state.highestFence) + applyJournalRow(state, row) + rows.push(row) + } + + input.db.exec('BEGIN IMMEDIATE') try { - const epoch = input.mintEpoch() - const state = createJournalReducerState(input.identity.sessionId, epoch) - const epochRow: JournalRow = { - kind: 'epoch', - reason: input.reason, - providerHandle: input.identity.providerHandle, - ...journalRowBase(epoch, 1, input.fence, input.now()) - } - const rows: JournalRow[] = [epochRow] - applyJournalRow(state, epochRow) - let sizeBytes = journalRowByteLength(epochRow) - await assertStagingCapacity(input, sizeBytes) - await appendJournalRows(stagingDir, [epochRow]) - - for (const item of input.items) { - sizeBytes += await stageReplacementBlobs({ - journalDir: input.journalDir, - stagingDir, - identity: input.identity, - budget: input.budget, - stagedBlobDigests, - blobs: item.blobs ?? [] - }) - const appendTime = input.now() - const row = buildJournalItemRow({ - state, - identity: item.identity, - body: item.body, - seq: state.lastSequence + 1, - fence: input.fence, - ts: item.observedAt ?? appendTime - }) - assertJournalFence(row.fence, state.highestFence) - input.budget.assert(row, appendTime, sizeBytes) - await assertStagingCapacity(input, journalRowByteLength(row)) - await appendJournalRows(stagingDir, [row]) - applyJournalRow(state, row) - rows.push(row) - sizeBytes += journalRowByteLength(row) - } - - const compacted = await compactJournal({ - journalDir: stagingDir, - physicalQuotaRoot: input.journalDir, - state, - tailRows: rows, - policy: input.compaction, - now: input.now(), - maxSessionBytes: input.budget.maxSessionBytes - }) - // All destination publishes use durable temp files while the staging - // source and existing finals remain present. Reserve the whole publication - // peak before touching the live epoch so a later file cannot fail halfway - // through replacement. - const stagedSnapshotBytes = (await stat(join(stagingDir, JOURNAL_SNAPSHOT_FILE))).size - const stagedLogBytes = (await stat(join(stagingDir, JOURNAL_LOG_FILE))).size - let stagedPublishBytes = stagedSnapshotBytes + stagedLogBytes - for (const digest of stagedBlobDigests) { - stagedPublishBytes += (await stat(join(stagingDir, JOURNAL_BLOB_DIR, digest))).size - } - await assertJournalPhysicalCapacity({ - journalDir: input.journalDir, - sessionId: input.identity.sessionId, - maxBytes: input.budget.maxSessionBytes, - peakAdditionalBytes: stagedPublishBytes - }) - for (const digest of stagedBlobDigests) { - if ( - await publishPreparedBlob( - stagingDir, - input.journalDir, - digest, - input.identity.sessionId, - input.budget.maxSessionBytes - ) - ) { - publishedBlobDigests.push(digest) - } - } - await publishPreparedFile( - stagingDir, - input.journalDir, - JOURNAL_SNAPSHOT_FILE, - input.identity.sessionId, - input.budget.maxSessionBytes - ) - snapshotPublished = true - state.oldestSequence = compacted.oldestSequence - publishedLoad = { - state, - tailRows: compacted.tailRows, - compactedThrough: compacted.compactedThrough, - readOnly: false, - corrupt: false, - malformedRows: 0, - sizeBytes: 0 - } - await publishPreparedFile( - stagingDir, - input.journalDir, - JOURNAL_LOG_FILE, - input.identity.sessionId, - input.budget.maxSessionBytes - ) - await pruneJournalBlobs( - input.journalDir, - replacementRetainedBlobDigests(state, compacted.tailRows) - ) - } finally { - if (!snapshotPublished) { - for (const digest of publishedBlobDigests) { - await removeJournalBlob(input.journalDir, digest) - } - } - await rm(stagingDir, { recursive: true, force: true }) - if (snapshotPublished && publishedLoad && !adoptionReported) { - adoptionReported = true - input.onSnapshotPublished({ - ...publishedLoad, - sizeBytes: await journalDirectoryBytes(input.journalDir) - }) + deleteAllJournalRows(input.db) + clearJournalRepairMarker(input.db, input.identity.sessionId) + for (const row of rows) { + insertJournalRow(input.db, input.identity.sessionId, row) } + upsertJournalSessionRow(input.db, input.identity.sessionId, epoch, epochRow.ts) + input.db.exec('COMMIT') + } catch (error) { + input.db.exec('ROLLBACK') + throw error } -} -function replacementRetainedBlobDigests( - state: JournalReducerState, - tailRows: readonly JournalRow[] -): Set { - const retained = referencedBlobDigests(state) - for (const row of tailRows) { - if (row.kind === 'item') { - blobDigestsInBody(row.body, retained) - } else if (row.kind === 'lifecycle-batch') { - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - blobDigestsInBody(mutation.body, retained) - } - } - } - } - return retained -} - -async function stageReplacementBlobs(input: { - journalDir: string - stagingDir: string - identity: AgentSessionJournalIdentity - budget: JournalAppendBudget - stagedBlobDigests: Set - blobs: readonly { digest: string; payload: string }[] -}): Promise { - const toStage: { digest: string; payload: string; bytes: number }[] = [] - const unique = new Map(input.blobs.map((blob) => [blob.digest, blob])) - for (const blob of unique.values()) { - if (input.stagedBlobDigests.has(blob.digest)) { - continue - } - if ((await journalBlobFileSize(input.journalDir, blob.digest)) !== null) { - continue - } - const bytes = Buffer.byteLength(blob.payload, 'utf8') - toStage.push({ ...blob, bytes }) - } - // Reserve all new payloads together. The staging directory lives under the - // journal root, so the capacity check includes existing session bytes and - // every other .epoch-replacement-* directory already present. - const stagedBytes = toStage.reduce((total, blob) => total + blob.bytes, 0) - await assertStagingCapacity(input, stagedBytes) - for (const blob of toStage) { - await putJournalBlob(input.stagingDir, blob.digest, blob.payload) - input.stagedBlobDigests.add(blob.digest) - } - return stagedBytes -} - -function assertStagingCapacity( - input: { - journalDir: string - identity: AgentSessionJournalIdentity - budget: JournalAppendBudget - }, - additionalBytes: number -): Promise { - return assertJournalPhysicalCapacity({ - journalDir: input.journalDir, - sessionId: input.identity.sessionId, - maxBytes: input.budget.maxSessionBytes, - peakAdditionalBytes: additionalBytes - }) -} - -async function publishPreparedFile( - stagingDir: string, - journalDir: string, - fileName: string, - sessionId: string, - maxBytes: number -): Promise { - await assertJournalPhysicalCapacity({ - journalDir, - sessionId, - maxBytes, - peakAdditionalBytes: (await stat(join(stagingDir, fileName))).size - }) - const copied = await copyFileDurable(join(stagingDir, fileName), join(journalDir, fileName)) - if (!copied) { - throw new Error(`prepared journal file disappeared before publish: ${fileName}`) - } -} - -async function publishPreparedBlob( - stagingDir: string, - journalDir: string, - digest: string, - sessionId: string, - maxBytes: number -): Promise { - if ((await journalBlobFileSize(journalDir, digest)) !== null) { - return false - } - const size = await journalBlobFileSize(stagingDir, digest) - if (size === null) { - throw new Error(`prepared journal blob disappeared before publish: ${digest}`) - } - await assertJournalPhysicalCapacity({ - journalDir, - sessionId, - maxBytes, - peakAdditionalBytes: size - }) - await mkdir(join(journalDir, JOURNAL_BLOB_DIR), { recursive: true }) - const copied = await copyFileDurable( - join(stagingDir, JOURNAL_BLOB_DIR, digest), - join(journalDir, JOURNAL_BLOB_DIR, digest) - ) - if (!copied) { - throw new Error(`prepared journal blob disappeared before publish: ${digest}`) - } - return true + // COMMIT landed: on disk the superseded rows are gone and this epoch is the + // live one. The caller adopts that immediately, or a later failure leaves the + // live store writing into an epoch whose rows were just deleted. + state.oldestSequence = 1 + input.onPublished({ state, readOnly: false, corrupt: false, malformedRows: 0 }) } diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts b/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts index 40e1bc3a542..e95ff821058 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts @@ -1,29 +1,34 @@ // Opening a new epoch. // -// The snapshot is what names the live epoch, so it is published BEFORE the log -// is reset. A crash mid-rollover therefore leaves stale-epoch rows behind the -// new snapshot, which `loadJournal` drops — the reverse order would leave a -// journal whose log no longer matches any epoch anyone can name. +// One transaction: discard every row of the superseded epoch, insert the new +// epoch row at sequence 1, move the session projection onto it, and retire any +// repair marker the superseded epoch was carrying. Superseded rows are DELETED +// rather than retained — nothing would ever shed them. import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' import type { AgentSessionProviderHandle } from '../../../shared/agent-session-journal-types' -import { compactJournal } from './journal-compaction' -import { applyJournalRow, createJournalReducerState } from './journal-reducer' -import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -import { journalRowByteLength } from './journal-row-schema' +import type Database from '../../sqlite/sync-database' import type { JournalLoad } from './journal-open' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' +import { clearJournalRepairMarker } from './journal-repair-marker' +import { applyJournalRow, createJournalReducerState } from './journal-reducer' +import { + deleteAllJournalRows, + insertJournalRow, + upsertJournalSessionRow +} from './journal-row-table' +import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -export async function publishNewEpoch(input: { - journalDir: string +export function publishNewEpoch(input: { + db: Database.Database sessionId: string providerHandle: AgentSessionProviderHandle epoch: string reason: AgentJournalEpochReason fence: number now: number - maxSessionBytes?: number -}): Promise { + /** Called the instant the transaction commits, before any fallible follow-up. */ + onPublished: (loaded: JournalLoad) => void +}): void { const row: JournalRow = { kind: 'epoch', reason: input.reason, @@ -34,25 +39,24 @@ export async function publishNewEpoch(input: { fence: input.fence, ts: input.now } + + input.db.exec('BEGIN IMMEDIATE') + try { + deleteAllJournalRows(input.db) + clearJournalRepairMarker(input.db, input.sessionId) + insertJournalRow(input.db, input.sessionId, row) + upsertJournalSessionRow(input.db, input.sessionId, input.epoch, input.now) + input.db.exec('COMMIT') + } catch (error) { + input.db.exec('ROLLBACK') + throw error + } + + // COMMIT landed: on disk the superseded prefix is gone and this epoch is the + // live one. The caller adopts that immediately, or a later failure leaves the + // store writing into an epoch that no longer exists. const state = createJournalReducerState(input.sessionId, input.epoch) - await compactJournal({ - journalDir: input.journalDir, - state, - tailRows: [row], - policy: { minTailRows: 1, retainTailMs: Number.POSITIVE_INFINITY }, - now: input.now, - maxSessionBytes: input.maxSessionBytes ?? DEFAULT_JOURNAL_PAYLOAD_LIMITS.maxSessionBytes, - sessionId: input.sessionId - }) applyJournalRow(state, row) state.oldestSequence = 1 - return { - state, - tailRows: [row], - compactedThrough: 0, - readOnly: false, - corrupt: false, - malformedRows: 0, - sizeBytes: journalRowByteLength(row) - } + input.onPublished({ state, readOnly: false, corrupt: false, malformedRows: 0 }) } diff --git a/src/main/native-chat/agent-session-journal/journal-handle-ownership.test.ts b/src/main/native-chat/agent-session-journal/journal-handle-ownership.test.ts new file mode 100644 index 00000000000..592f7471c5c --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-handle-ownership.test.ts @@ -0,0 +1,161 @@ +// Every path that can open a SQLite connection releases it. +// +// Asserting that the happy path closes cleanly proves nothing: these sites are +// reached only when something has already gone wrong. On POSIX a leak is +// SILENT — the unlink succeeds — so each case asserts BOTH that the sidecars +// are gone and that the directory renames and removes, which is the half that +// actually fails on Windows. + +import { access, mkdtemp, rename, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import Database from '../../sqlite/sync-database' +import { openJournalDatabase } from './journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' +import { loadJournal } from './journal-open' +import { journalDatabaseFile } from './journal-paths' +import { createTrackedJournalOpener } from './journal-store-test-open' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} + +let base: string +let root: string +const journals = createTrackedJournalOpener() + +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } +} + +function runningTool(): AgentJournalItemBody { + return { kind: 'tool-call', name: 'command', input: {}, state: 'running' } +} + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +/** The platform-independent proof: an open handle blocks both of these on + * Windows, where every leak in this file actually shows up. */ +async function expectNothingHoldsTheDirectory(): Promise { + const dbPath = journalDatabaseFile(root) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + const moved = `${root}-recovered-vtest` + await rename(root, moved) + await rm(moved, { recursive: true }) + root = join(base, `journal-${Math.random().toString(36).slice(2)}`) +} + +beforeEach(async () => { + base = await mkdtemp(join(tmpdir(), 'orca-journal-handles-')) + root = join(base, 'journal') +}) + +afterEach(async () => { + await journals.closeAll() + await rm(base, { recursive: true, force: true }) +}) + +describe('the standalone probe owns its own connection', () => { + it('leaves no handle behind after fifty repeated loads', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.appendItem(item(1), { kind: 'message', role: 'user', blocks: [] }, { fence: 1 }) + await journal.close() + + for (let attempt = 0; attempt < 50; attempt += 1) { + const loaded = await loadJournal(root, IDENTITY.sessionId) + expect(loaded?.readOnly).toBe(false) + } + await expectNothingHoldsTheDirectory() + }) + + it('leaves no handle behind after fifty loads of a latched future schema', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.close() + const seeded = openJournalDatabase(journalDatabaseFile(root)) + seeded.db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 3}`) + seeded.db.close() + + for (let attempt = 0; attempt < 50; attempt += 1) { + // The latched open closes the probe connection and returns the read-only + // reopen, so probing repeatedly leaves nothing on a file we must not touch. + expect((await loadJournal(root, IDENTITY.sessionId))?.readOnly).toBe(true) + } + // A read-only connection cannot remove the sidecars it materialized, so only + // the rename/remove half is expected to hold here. + const moved = `${root}-recovered-vtest` + await rename(root, moved) + await rm(moved, { recursive: true }) + root = join(base, 'journal-after-latched') + }) + + it('returns null for a session with no journal, without creating one', async () => { + expect(await loadJournal(root, IDENTITY.sessionId)).toBeNull() + expect(await exists(journalDatabaseFile(root))).toBe(false) + }) +}) + +describe('failure paths inside the open call', () => { + // Site 1: the raw connection is owned by `openJournalDatabase` until it returns. + it('closes the raw connection when the version read cannot run', async () => { + await journals.open({ identity: IDENTITY, journalDir: root }).then((journal) => journal.close()) + await writeFile(journalDatabaseFile(root), 'this is not a database', 'utf8') + + expect(() => openJournalDatabase(journalDatabaseFile(root))).toThrow() + await expectNothingHoldsTheDirectory() + }) + + it('closes the raw connection when the migration cannot start', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.close() + const blocker = openJournalDatabase(journalDatabaseFile(root)) + // Roll the stored version back so the migration runs, then hold the write + // lock it needs: the throw lands after the connection already exists. + blocker.db.pragma('user_version = 0') + blocker.db.exec('BEGIN IMMEDIATE') + blocker.db.exec("INSERT INTO journal_sessions VALUES ('other', 'e', 1)") + try { + expect(() => openJournalDatabase(journalDatabaseFile(root))).toThrow() + } finally { + blocker.db.exec('ROLLBACK') + blocker.db.close() + } + await expectNothingHoldsTheDirectory() + }, 60_000) + + // Sites 2 and 3: `open()` closes its own connection on any throw after the + // connection exists, which is what lets the factory need no `finally`. + it('leaves nothing open when a post-connection step of open() throws', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + await journal.appendItem(item(1), runningTool(), { fence: 1 }) + await journal.close() + + // Replay runs after the connection is open, so a read it cannot serve + // throws with the handle already held. + const exec = vi.spyOn(Database.prototype, 'prepare').mockImplementation(() => { + throw new Error('replay cannot read this journal') + }) + try { + await expect(journals.open({ identity: IDENTITY, journalDir: root })).rejects.toThrow( + 'replay cannot read this journal' + ) + } finally { + exec.mockRestore() + } + await expectNothingHoldsTheDirectory() + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-item-appender.ts b/src/main/native-chat/agent-session-journal/journal-item-appender.ts index 3ed2b58b230..1bbc3d4fab3 100644 --- a/src/main/native-chat/agent-session-journal/journal-item-appender.ts +++ b/src/main/native-chat/agent-session-journal/journal-item-appender.ts @@ -5,23 +5,16 @@ import type { } from '../../../shared/agent-session-journal-types' import { journalItemRowBuilder } from './journal-row-builders' import type { JournalReducerState } from './journal-reducer' -import type { AgentSessionJournal } from './journal-store' import type { JournalAppendResult } from './journal-store-contracts' import type { JournalRow } from './journal-row-schema' -import { appendToolOutputFallback } from './journal-tool-output-fallback' type ItemAppendOptions = { fence: number; observedAt?: number; recovered?: true } -type JournalBlob = { digest: string; payload: string } export class JournalItemAppender { constructor( private readonly deps: { - journal: () => AgentSessionJournal state: () => JournalReducerState - enqueue: ( - build: (seq: number, ts: number) => JournalRow, - blobs?: readonly JournalBlob[] - ) => Promise + enqueue: (build: (seq: number, ts: number) => JournalRow) => Promise } ) {} @@ -33,37 +26,10 @@ export class JournalItemAppender { const itemId = agentJournalItemKey(identity) return this.deps .enqueue(journalItemRowBuilder(this.deps.state, identity, body, options)) - .then((row) => itemAppendResult(row, itemId)) - } - - appendWithBlobs( - identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly JournalBlob[], - options: ItemAppendOptions - ): Promise { - const itemId = agentJournalItemKey(identity) - return this.deps - .enqueue(journalItemRowBuilder(this.deps.state, identity, body, options), blobs) - .then((row) => itemAppendResult(row, itemId)) - .catch((error: unknown) => - appendToolOutputFallback({ - journal: this.deps.journal(), - error, - identity, - body, - blobs, - itemId, - fence: options.fence - }) - ) - } -} - -function itemAppendResult(row: JournalRow, itemId: string): JournalAppendResult { - return { - cursor: { epoch: row.epoch, sequence: row.seq }, - itemId, - revision: (row as Extract).revision + .then((row) => ({ + cursor: { epoch: row.epoch, sequence: row.seq }, + itemId, + revision: (row as Extract).revision + })) } } diff --git a/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts b/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts index a5a3e2be852..ed86d81861e 100644 --- a/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-legacy-import.test.ts @@ -2,19 +2,21 @@ // results by identity read off the same raw lines. Fixtures are shaped like the // files the providers actually write. -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' +import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' -import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { JOURNAL_BLOB_DIR, readJournalBlob } from './journal-blob-store' +import type { + AgentSessionJournalIdentity, + AgentSessionProviderHandle +} from '../../../shared/agent-session-journal-types' import { createLegacyIdentityTracker } from './journal-legacy-identity' import { appendLegacyTranscriptMessages, importLegacyTranscriptIntoJournal } from './journal-legacy-import' -import { boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' +import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' import { openAgentSessionJournal } from './journal-store-factory' import type { AgentSessionJournal } from './journal-store' @@ -29,21 +31,29 @@ function tick(): number { return clock } -function identity(agent: 'claude' | 'codex', sessionId: string): AgentSessionJournalIdentity { +type ImportAgent = 'claude' | 'codex' | 'grok' | 'omp' + +function providerHandle(agent: ImportAgent, sessionId: string): AgentSessionProviderHandle { + if (agent === 'claude') { + return { kind: 'claude', sessionId, leafUuid: null } + } + return agent === 'codex' + ? { kind: 'codex', threadId: sessionId } + : { kind: 'opaque', agent, value: sessionId } +} + +function identity(agent: ImportAgent, sessionId: string): AgentSessionJournalIdentity { return { sessionId, workspaceId: 'ws-1', hostId: 'host-1', agent, - providerHandle: - agent === 'claude' - ? { kind: 'claude', sessionId, leafUuid: null } - : { kind: 'codex', threadId: sessionId } + providerHandle: providerHandle(agent, sessionId) } } async function open( - agent: 'claude' | 'codex', + agent: ImportAgent, sessionId: string, overrides: Partial[0]> = {} ): Promise { @@ -387,7 +397,7 @@ describe('codex import', () => { }) describe('payload bounds on import', () => { - it('marks a clipped tool result and parks the remainder in the blob store', async () => { + it('marks a clipped tool result and discards the remainder', async () => { const output = 'y'.repeat(64 * 1024) const filePath = await writeFixture('claude-big.jsonl', [ { @@ -421,167 +431,13 @@ describe('payload bounds on import', () => { expect(body.output.truncated).toBe(true) expect(body.output.byteLength).toBe(64 * 1024) expect(body.output.head).toHaveLength(1_024) - expect(await readJournalBlob(root, body.output.digest)).toBe(output) - }) - - it('deduplicates staged blobs while importing a replacement epoch', async () => { - const journalDir = join(root, 'dedupe-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 512, - maxSessionBytes: 512 * 1024 - } - const output = 'd'.repeat(32 * 1024) - const bounded = boundPayload(output, limits) - const toolResultLine = (uuid: string) => ({ - parentUuid: null, - isSidechain: false, - type: 'user', - message: { - role: 'user', - content: [{ type: 'tool_result', tool_use_id: `toolu_${uuid}`, content: output }] - }, - uuid, - timestamp: '2026-08-05T10:00:09.000Z', - sessionId: CLAUDE_SESSION - }) - const filePath = await writeFixture('claude-duplicate-blobs.jsonl', [ - toolResultLine('aa11bb22-cc33-4d44-8e55-6f7788990011'), - toolResultLine('bb22cc33-dd44-4e55-8f66-778899001122') - ]) - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) - - await importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 1, - options: { filePath, limits } - }) - - expect(await readJournalBlob(journalDir, bounded.digest)).toBe(output) - expect(await readdir(join(journalDir, JOURNAL_BLOB_DIR))).toEqual([bounded.digest]) - expect( - journal - .snapshot() - .items.map((item) => (item.body.kind === 'tool-call' ? item.body.output?.digest : null)) - ).toEqual([bounded.digest, bounded.digest]) - }) - - it('prunes root-level blobs made stale by a later legacy import', async () => { - const journalDir = join(root, 'prune-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 512, - maxSessionBytes: 512 * 1024 - } - const output = 's'.repeat(32 * 1024) - const bounded = boundPayload(output, limits) - const first = await writeFixture('claude-stale-blob.jsonl', [ - { - parentUuid: null, - isSidechain: false, - type: 'user', - message: { - role: 'user', - content: [{ type: 'tool_result', tool_use_id: 'toolu_stale', content: output }] - }, - uuid: 'aa11bb22-cc33-4d44-8e55-6f7788990011', - timestamp: '2026-08-05T10:00:09.000Z', - sessionId: CLAUDE_SESSION - } - ]) - const second = await writeFixture('claude-without-blob.jsonl', [ - { - parentUuid: null, - isSidechain: false, - type: 'assistant', - message: { role: 'assistant', content: [{ type: 'text', text: 'replacement' }] }, - uuid: 'cc33dd44-ee55-4666-8777-889900112233', - timestamp: '2026-08-05T10:00:10.000Z', - sessionId: CLAUDE_SESSION - } - ]) - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) - - await importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 1, - options: { filePath: first, limits } - }) - expect(await readJournalBlob(journalDir, bounded.digest)).toBe(output) - - await importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 2, - options: { filePath: second, limits } - }) - - expect(await readJournalBlob(journalDir, bounded.digest)).toBeNull() - expect(journal.snapshot().items[0]?.body).toMatchObject({ - kind: 'message', - blocks: [{ type: 'text', text: 'replacement' }] - }) - }) - - it('uses managed catch-up appends when a tool-result blob exceeds quota', async () => { - const journalDir = join(root, 'catchup-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 128, - maxSessionBytes: 8_000 - } - const journal = await open('codex', CODEX_SESSION, { - journalDir, - limits, - autoCompact: false - }) - const output = 'z'.repeat(12_000) - const bounded = boundPayload(output, limits) - - await expect( - appendLegacyTranscriptMessages({ - journal, - agent: 'codex', - sessionId: CODEX_SESSION, - fence: 1, - messages: [ - { - id: 'catchup-tool-output', - role: 'tool', - blocks: [{ type: 'tool-result', output }], - timestamp: 1_800_000_000_000, - source: 'transcript' - } - ] - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(await readJournalBlob(journalDir, bounded.digest)).toBeNull() - expect(journal.snapshot().items).toEqual([]) }) }) describe('import failures', () => { it('rejects a legacy source above the fixed 16 MiB import cap before decoding', async () => { const journalDir = join(root, 'oversized-source-journal') - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 256 * 1024 * 1024 }, - autoCompact: false - }) + const journal = await open('claude', CLAUDE_SESSION, { journalDir }) const filePath = join(root, 'oversized-source.jsonl') await writeFile(filePath, 'x'.repeat(16 * 1024 * 1024 + 1), 'utf8') const epoch = journal.epoch @@ -602,115 +458,10 @@ describe('import failures', () => { expect(journal.snapshot().items).toEqual([]) }) - it('keeps the live epoch intact when a staged rebuild runs out of budget', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - const journal = await open('codex', CODEX_SESSION, { limits }) - await appendLegacyTranscriptMessages({ - journal, - agent: 'codex', - sessionId: CODEX_SESSION, - fence: 1, - messages: [ - { - id: 'durable-prefix', - role: 'assistant', - blocks: [{ type: 'text', text: 'keep me' }], - timestamp: 1_800_000_000_000, - source: 'transcript' - } - ] - }) - const filePath = await writeFixture('oversized-rollout.jsonl', [ - CODEX_LINES[0], - CODEX_LINES[1], - CODEX_LINES[2], - { - type: 'event_msg', - timestamp: '2026-08-05T10:00:03.000Z', - payload: { type: 'agent_message', message: 'x'.repeat(2_000) } - } - ]) - const epoch = journal.epoch - const snapshotPath = join(root, 'snapshot.json') - const logPath = join(root, 'log.jsonl') - const before = { - snapshot: await readFile(snapshotPath, 'utf-8'), - log: await readFile(logPath, 'utf-8') - } - - await expect( - importLegacyTranscriptIntoJournal({ - journal, - agent: 'codex', - sessionId: CODEX_SESSION, - fence: 1, - options: { filePath, limits } - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - expect(journal.epoch).toBe(epoch) - expect(await readFile(snapshotPath, 'utf-8')).toBe(before.snapshot) - expect(await readFile(logPath, 'utf-8')).toBe(before.log) - expect(journal.snapshot().items[0]?.body).toMatchObject({ - kind: 'message', - blocks: [{ type: 'text', text: 'keep me' }] - }) - }) - - it('cleans staged replacement blobs when legacy import exceeds physical quota', async () => { - const journalDir = join(root, 'replacement-journal') - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 128, - maxSessionBytes: 8_000 - } - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) - const output = 'q'.repeat(12_000) - const bounded = boundPayload(output, limits) - const filePath = await writeFixture('oversized-tool-result.jsonl', [ - { - parentUuid: null, - isSidechain: false, - type: 'user', - message: { - role: 'user', - content: [{ type: 'tool_result', tool_use_id: 'toolu_oversized', content: output }] - }, - uuid: 'ba11ad00-1111-4222-8333-444455556666', - timestamp: '2026-08-05T10:00:09.000Z', - sessionId: CLAUDE_SESSION - } - ]) - const epoch = journal.epoch - - await expect( - importLegacyTranscriptIntoJournal({ - journal, - agent: 'claude', - sessionId: CLAUDE_SESSION, - fence: 1, - options: { filePath, limits } - }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toBe(epoch) - expect(await readJournalBlob(journalDir, bounded.digest)).toBeNull() - expect((await readdir(journalDir)).some((name) => name.startsWith('.epoch-replacement-'))).toBe( - false - ) - }) - it('bounds oversized legacy tool-call input before journal publication', async () => { const journalDir = join(root, 'bounded-tool-input-journal') const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 64 } - const journal = await open('claude', CLAUDE_SESSION, { - journalDir, - limits, - autoCompact: false - }) + const journal = await open('claude', CLAUDE_SESSION, { journalDir }) const filePath = await writeFixture('oversized-tool-input.jsonl', [ { parentUuid: null, @@ -768,6 +519,39 @@ describe('import failures', () => { expect(journal.epoch).toBe(before) }) + // A transcript with no decodable messages recovers nothing. Publishing an + // empty replacement would roll the epoch and drop whatever the journal held — + // including a repair's own anchor and disclosure. + it('leaves the epoch untouched when the transcript decodes to no messages', async () => { + const journal = await open('codex', CODEX_SESSION) + await journal.appendItem( + { provider: 'codex', threadId: CODEX_SESSION, turnId: 'turn-1', ordinal: 1 }, + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'kept' }] }, + { fence: 1 } + ) + const before = journal.epoch + const metadataOnly = await writeFixture('metadata-only.jsonl', [ + { + type: 'session_meta', + timestamp: '2026-08-05T10:00:00.000Z', + payload: { id: CODEX_SESSION, session_id: CODEX_SESSION, cwd: '/Users/dev/project' } + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'codex', + sessionId: CODEX_SESSION, + fence: 1, + options: { filePath: metadataOnly } + }) + + expect(result).toMatchObject({ ok: true, imported: 0, replaced: false }) + expect(journal.epoch).toBe(before) + expect(journal.snapshot().items).toHaveLength(1) + await journal.close() + }) + it('rejects an agent with no transcript decoder', async () => { const journal = await open('claude', CLAUDE_SESSION) const result = await importLegacyTranscriptIntoJournal({ @@ -780,3 +564,132 @@ describe('import failures', () => { expect(result).toMatchObject({ ok: false }) }) }) + +// A tool call is only the SOLE block of its message when the provider wrote it +// that way. Claude interleaves it with narration, Grok hangs `tool_calls` off a +// row that also has text, and omp's execution cells always pair the invocation +// with its output — so the multi-block path carries untrusted tool input too. +describe('multi-block legacy messages', () => { + const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, inlineHeadBytes: 64 } + const oversized = 'x'.repeat(10_000) + + /** The tool-call block of the first imported multi-block message. */ + function importedToolCallBlock(journal: AgentSessionJournal): unknown { + for (const entry of journal.snapshot().items) { + if (entry.body.kind !== 'message') { + continue + } + const block = entry.body.blocks.find((candidate) => candidate.type === 'tool-call') + if (block) { + return block.input + } + } + return null + } + + it('bounds a Claude tool call that shares its message with narration', async () => { + const journal = await open('claude', CLAUDE_SESSION, { + journalDir: join(root, 'claude-mixed-journal') + }) + const filePath = await writeFixture('claude-mixed.jsonl', [ + { + parentUuid: null, + isSidechain: false, + type: 'assistant', + message: { + role: 'assistant', + content: [ + { type: 'text', text: 'Editing the file.' }, + { + type: 'tool_use', + id: 'toolu_mixed', + name: 'Edit', + input: { file_path: 'a.ts', patch: oversized } + } + ] + }, + uuid: 'dd22be00-1111-4222-8333-444455556666', + timestamp: '2026-08-05T10:00:09.000Z', + sessionId: CLAUDE_SESSION + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'claude', + sessionId: CLAUDE_SESSION, + fence: 1, + options: { filePath, limits } + }) + + expect(result.ok).toBe(true) + expect(importedToolCallBlock(journal)).toMatchObject({ + truncated: true, + byteLength: expect.any(Number), + digest: expect.stringMatching(/^[0-9a-f]{64}$/), + head: expect.any(String) + }) + expect(JSON.stringify(journal.snapshot().items)).not.toContain('x'.repeat(1_000)) + await journal.close() + }) + + it('bounds a Grok tool call that shares its row with assistant text', async () => { + const journal = await open('grok', CODEX_SESSION, { + journalDir: join(root, 'grok-mixed-journal') + }) + const filePath = await writeFixture('grok-mixed.jsonl', [ + { + type: 'assistant', + id: 'asst-mixed', + timestamp: '2026-08-05T10:00:09.000Z', + content: [{ type: 'text', text: 'Searching.' }], + tool_calls: [{ id: 'c1', name: 'grep', arguments: JSON.stringify({ pattern: oversized }) }] + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'grok', + sessionId: CODEX_SESSION, + fence: 1, + options: { filePath, limits } + }) + + expect(result.ok).toBe(true) + expect(importedToolCallBlock(journal)).toMatchObject({ truncated: true }) + expect(JSON.stringify(journal.snapshot().items)).not.toContain('x'.repeat(1_000)) + await journal.close() + }) + + it('bounds an omp execution cell, whose invocation always ships with its output', async () => { + const journal = await open('omp', CODEX_SESSION, { + journalDir: join(root, 'omp-mixed-journal') + }) + const filePath = await writeFixture('omp-mixed.jsonl', [ + { + type: 'message', + id: 'omp-mixed-1', + timestamp: '2026-08-05T10:00:09.000Z', + message: { + role: 'bashExecution', + command: `echo ${oversized}`, + output: 'done', + exitCode: 0 + } + } + ]) + + const result = await importLegacyTranscriptIntoJournal({ + journal, + agent: 'omp', + sessionId: CODEX_SESSION, + fence: 1, + options: { filePath, limits } + }) + + expect(result.ok).toBe(true) + expect(importedToolCallBlock(journal)).toMatchObject({ truncated: true }) + expect(JSON.stringify(journal.snapshot().items)).not.toContain('x'.repeat(1_000)) + await journal.close() + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-legacy-import.ts b/src/main/native-chat/agent-session-journal/journal-legacy-import.ts index 126196af38e..6a2ff099dbc 100644 --- a/src/main/native-chat/agent-session-journal/journal-legacy-import.ts +++ b/src/main/native-chat/agent-session-journal/journal-legacy-import.ts @@ -49,7 +49,14 @@ export type LegacyImportOptions = ResolveSessionFileOptions & { const MAX_LEGACY_IMPORT_SOURCE_BYTES = 16 * 1024 * 1024 export type LegacyImportResult = - | { ok: true; epoch: string; cursor: AgentJournalCursor; imported: number } + | { + ok: true + epoch: string + cursor: AgentJournalCursor + imported: number + /** False when the transcript held no messages and the epoch was left as it stood. */ + replaced: boolean + } | { ok: false; error: string } export async function appendLegacyTranscriptMessages(input: { @@ -61,16 +68,14 @@ export async function appendLegacyTranscriptMessages(input: { }): Promise { let appended = 0 for (const message of input.messages) { - const mapped = legacyItemBody(message, DEFAULT_JOURNAL_PAYLOAD_LIMITS) - await input.journal.appendItemWithBlobs( + await input.journal.appendItem( { provider: 'legacy', agent: input.agent, sessionId: input.sessionId, recordId: message.id }, - mapped.body, - mapped.blobs, + legacyItemBody(message, DEFAULT_JOURNAL_PAYLOAD_LIMITS), { fence: input.fence, observedAt: message.timestamp ?? undefined } ) appended += 1 @@ -134,16 +139,28 @@ export async function importLegacyTranscriptIntoJournal(input: { if (!identity) { continue } - const mapped = legacyItemBody(message, limits) replacement.push({ identity, - body: mapped.body, - blobs: mapped.blobs, + body: legacyItemBody(message, limits), observedAt: message.timestamp ?? undefined }) } + // A transcript that decodes to nothing reconstructs nothing, and an empty + // replacement is not a harmless no-op: it would delete the repair's anchor and + // its disclosure, leaving nothing to ask for the history again. The epoch + // stands so a later read can still rebuild it. + if (replacement.length === 0) { + const current = input.journal.cursor() + return { ok: true, epoch: current.epoch, cursor: current, imported: 0, replaced: false } + } const cursor = await input.journal.replaceEpochItems('legacy_import', input.fence, replacement) - return { ok: true, epoch: cursor.epoch, cursor, imported: decoded.messages.length } + return { + ok: true, + epoch: cursor.epoch, + cursor, + imported: decoded.messages.length, + replaced: true + } } const TRANSCRIPT_DECODERS = { @@ -199,11 +216,6 @@ async function decodeWithIdentities(input: { return { messages, identities } } -type MappedLegacyItem = { - body: AgentJournalItemBody - blobs: { digest: string; payload: string }[] -} - /** * A message whose only content is a tool invocation becomes a tool-call item so * the reducer renders it as one. Everything else stays a message item with its @@ -212,49 +224,40 @@ type MappedLegacyItem = { function legacyItemBody( message: NativeChatMessage, limits: JournalPayloadLimits -): MappedLegacyItem { +): AgentJournalItemBody { const only = message.blocks.length === 1 ? message.blocks[0] : undefined if (only?.type === 'tool-call') { + // Legacy transcripts are untrusted and can contain arbitrarily large tool + // arguments. Keep them on the same bounded path as live events before the + // replacement epoch is published. return { - // Legacy transcripts are untrusted and can contain arbitrarily large - // tool arguments. Keep them on the same bounded path as live events - // before the replacement epoch is staged or published. - body: { - kind: 'tool-call', - name: only.name, - input: boundToolInput(only.input, limits), - state: 'completed' - }, - blobs: [] + kind: 'tool-call', + name: only.name, + input: boundToolInput(only.input, limits), + state: 'completed' } } if (only?.type === 'tool-result') { - const output = boundPayload(only.output, limits) return { - body: { - kind: 'tool-call', - name: 'tool-result', - input: null, - state: only.isError ? 'failed' : 'completed', - output - }, - blobs: output.truncated ? [{ digest: output.digest, payload: only.output }] : [] + kind: 'tool-call', + name: 'tool-result', + input: null, + state: only.isError ? 'failed' : 'completed', + output: boundPayload(only.output, limits) } } return { - body: { - kind: 'message', - role: message.role, - blocks: message.blocks.map((block) => boundBlock(block, limits)) - }, - blobs: [] + kind: 'message', + role: message.role, + blocks: message.blocks.map((block) => boundBlock(block, limits)) } } -/** Inline block text keeps only a bounded head plus an explicit marker. No blob - * is written: the marker carries the digest and byte length, and the source - * transcript remains the full copy — a blob here would be unreferenced by the - * render model and pruned at the next compaction. */ +/** Every block that can carry untrusted bulk is bounded here, tool calls + * included: a provider decoder is free to put one alongside narration, and the + * sole-block path above never sees those. The remainder is discarded rather + * than stored elsewhere — the marker keeps its digest and byte length, and the + * source transcript remains the full copy. */ function boundBlock(block: NativeChatBlock, limits: JournalPayloadLimits): NativeChatBlock { if (block.type === 'text') { return { ...block, text: boundInlineText(block.text, limits).text } @@ -262,5 +265,8 @@ function boundBlock(block: NativeChatBlock, limits: JournalPayloadLimits): Nativ if (block.type === 'tool-result') { return { ...block, output: boundInlineText(block.output, limits).text } } + if (block.type === 'tool-call') { + return { ...block, input: boundToolInput(block.input, limits) } + } return block } diff --git a/src/main/native-chat/agent-session-journal/journal-lifecycle-admission.ts b/src/main/native-chat/agent-session-journal/journal-lifecycle-admission.ts deleted file mode 100644 index 063e807038e..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-lifecycle-admission.ts +++ /dev/null @@ -1,160 +0,0 @@ -import type { - AgentJournalItemBody, - AgentJournalSnapshot -} from '../../../shared/agent-session-journal-types' -import { - dispatchReservationId, - JournalLifecycleCapacity, - lifecycleReservationIdForItem, - requiresTerminalSettlement, - terminalReservationBytes, - type JournalLifecycleReservation -} from './journal-lifecycle-capacity' -import type { JournalRow } from './journal-row-schema' -import { journalRowByteLength } from './journal-row-schema' -import { AgentSessionJournalError } from './journal-write-guards' - -export type JournalLifecycleRowAdmission = { - releaseAfter: string[] - protectedBytes: number - lifecycleCovered: boolean - proposedCapacity: JournalLifecycleCapacity -} - -export class JournalLifecycleAdmission { - private readonly capacity = new JournalLifecycleCapacity() - - constructor( - private readonly sessionId: string, - private readonly maxBytes: number, - private readonly canonicalItemId: (itemId: string) => string, - private readonly maxAppendSlots = Number.MAX_SAFE_INTEGER - ) {} - - get state(): { reservedBytes: number; reservedAppendSlots: number } { - return { - reservedBytes: this.capacity.reservedBytes, - reservedAppendSlots: this.capacity.reservedAppendSlots - } - } - - rebuild(snapshot: AgentJournalSnapshot, currentPhysicalBytes: number): void { - if ( - !this.capacity.rebuild(snapshot, this.maxBytes, currentPhysicalBytes, this.maxAppendSlots) - ) { - throw this.capacityError('cannot rebuild lifecycle capacity') - } - } - - reserve(token: JournalLifecycleReservation, currentPhysicalBytes: number): boolean { - return this.capacity.reserve(token, currentPhysicalBytes, this.maxBytes, this.maxAppendSlots) - } - - transfer(fromId: string, toId: string): boolean { - return this.capacity.transfer(fromId, toId) - } - - release(id: string): void { - this.capacity.release(id) - } - - prepare(row: JournalRow, currentPhysicalBytes: number): JournalLifecycleRowAdmission { - const proposedCapacity = this.capacity.clone() - this.ensureActionable(row, currentPhysicalBytes, proposedCapacity) - const releaseAfter = this.reservationsSettledBy(row, proposedCapacity) - const releasedBytes = releaseAfter.reduce( - (total, id) => total + (proposedCapacity.token(id)?.bytes ?? 0), - 0 - ) - return { - releaseAfter, - protectedBytes: proposedCapacity.reservedBytes - releasedBytes, - lifecycleCovered: proposedCapacity.covers(releaseAfter, journalRowByteLength(row), 1), - proposedCapacity - } - } - - commit(admission: JournalLifecycleRowAdmission): void { - this.capacity.replaceFrom(admission.proposedCapacity) - for (const id of admission.releaseAfter) { - this.capacity.release(id) - } - } - - private ensureActionable( - row: JournalRow, - currentPhysicalBytes: number, - capacity: JournalLifecycleCapacity - ): void { - if (row.kind === 'item') { - this.ensureActionableItem(row.itemId, row.body, currentPhysicalBytes, capacity) - return - } - if (row.kind !== 'lifecycle-batch') { - return - } - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - this.ensureActionableItem(mutation.itemId, mutation.body, currentPhysicalBytes, capacity) - } - } - } - - private ensureActionableItem( - itemId: string, - body: AgentJournalItemBody, - currentPhysicalBytes: number, - capacity: JournalLifecycleCapacity - ): void { - if (!requiresTerminalSettlement(body)) { - return - } - const id = lifecycleReservationIdForItem(this.canonicalItemId(itemId)) - if (body.kind === 'status' && body.turnLifecycle?.state === 'running' && !capacity.has(id)) { - capacity.claimFirst('tentative-turn:', id) - } - if ( - !capacity.reserve( - { id, bytes: terminalReservationBytes(body), appendSlots: 1 }, - currentPhysicalBytes, - this.maxBytes, - this.maxAppendSlots - ) - ) { - throw this.capacityError('cannot reserve terminal capacity') - } - } - - private reservationsSettledBy(row: JournalRow, capacity: JournalLifecycleCapacity): string[] { - if (row.kind === 'dispatch') { - const id = dispatchReservationId(row.clientMessageId) - return capacity.has(id) ? [id] : [] - } - const itemIds = - row.kind === 'item' - ? requiresTerminalSettlement(row.body) - ? [] - : [row.itemId] - : row.kind === 'tombstone' - ? [row.itemId] - : row.kind === 'lifecycle-batch' - ? row.mutations.flatMap((mutation) => - mutation.kind === 'item' && requiresTerminalSettlement(mutation.body) - ? [] - : [mutation.itemId] - ) - : [] - return [ - ...new Set( - itemIds.map((itemId) => lifecycleReservationIdForItem(this.canonicalItemId(itemId))) - ) - ].filter((id) => capacity.has(id)) - } - - private capacityError(detail: string): AgentSessionJournalError { - return new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} ${detail}` - ) - } -} diff --git a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.test.ts b/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.test.ts deleted file mode 100644 index 331bc63c9dc..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.test.ts +++ /dev/null @@ -1,30 +0,0 @@ -import { describe, expect, it } from 'vitest' -import type { AgentJournalSnapshot } from '../../../shared/agent-session-journal-types' -import { JournalLifecycleCapacity } from './journal-lifecycle-capacity' - -describe('JournalLifecycleCapacity', () => { - it('enforces append-slot limits for both rebuilt submission reservations', () => { - const snapshot: AgentJournalSnapshot = { - sessionId: 'session-1', - cursor: { epoch: 'epoch-1', sequence: 1 }, - items: [], - submissions: [ - { - clientMessageId: 'message-1', - fence: 0, - payloadFingerprint: 'fingerprint', - dispatchState: 'pending', - providerItemId: null, - reason: null, - submittedAt: 1, - resolvedAt: null - } - ] - } - - const capacity = new JournalLifecycleCapacity() - - expect(capacity.rebuild(snapshot, Number.MAX_SAFE_INTEGER, 0, 1)).toBe(false) - expect(capacity.reservedAppendSlots).toBe(1) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.ts b/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.ts deleted file mode 100644 index 0c99070b1e4..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-lifecycle-capacity.ts +++ /dev/null @@ -1,193 +0,0 @@ -import type { - AgentJournalItemBody, - AgentJournalSnapshot -} from '../../../shared/agent-session-journal-types' - -export type JournalLifecycleReservation = { - id: string - bytes: number - appendSlots: number -} - -export const JOURNAL_TURN_TERMINAL_RESERVATION_BYTES = 128 * 1024 -export const JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES = 64 * 1024 -export const JOURNAL_DISPATCH_RESERVATION_BYTES = 32 * 1024 - -export class JournalLifecycleCapacity { - private readonly reservations = new Map() - - get reservedBytes(): number { - return [...this.reservations.values()].reduce((total, token) => total + token.bytes, 0) - } - - get reservedAppendSlots(): number { - return [...this.reservations.values()].reduce((total, token) => total + token.appendSlots, 0) - } - - has(id: string): boolean { - return this.reservations.has(id) - } - - token(id: string): JournalLifecycleReservation | null { - return this.reservations.get(id) ?? null - } - - clone(): JournalLifecycleCapacity { - const copy = new JournalLifecycleCapacity() - for (const token of this.reservations.values()) { - copy.reservations.set(token.id, { ...token }) - } - return copy - } - - replaceFrom(source: JournalLifecycleCapacity): void { - this.reservations.clear() - for (const token of source.reservations.values()) { - this.reservations.set(token.id, { ...token }) - } - } - - reserve( - token: JournalLifecycleReservation, - currentPhysicalBytes: number, - maxBytes: number, - maxAppendSlots = Number.MAX_SAFE_INTEGER - ): boolean { - if (this.reservations.has(token.id)) { - return true - } - if (currentPhysicalBytes + this.reservedBytes + token.bytes > maxBytes) { - return false - } - if (this.reservedAppendSlots + token.appendSlots > maxAppendSlots) { - return false - } - this.reservations.set(token.id, token) - return true - } - - transfer(fromId: string, toId: string): boolean { - const existing = this.reservations.get(fromId) - if (!existing) { - return false - } - this.reservations.delete(fromId) - this.reservations.set(toId, { ...existing, id: toId }) - return true - } - - claimFirst(prefix: string, toId: string): boolean { - const fromId = [...this.reservations.keys()].find((id) => id.startsWith(prefix)) - return fromId ? this.transfer(fromId, toId) : false - } - - release(id: string): void { - this.reservations.delete(id) - } - - covers(ids: readonly string[], bytes: number, appendSlots: number): boolean { - const tokens = ids.flatMap((id) => { - const token = this.reservations.get(id) - return token ? [token] : [] - }) - return ( - tokens.length > 0 && - tokens.reduce((total, token) => total + token.bytes, 0) >= bytes && - tokens.reduce((total, token) => total + token.appendSlots, 0) >= appendSlots - ) - } - - rebuild( - snapshot: AgentJournalSnapshot, - maxBytes: number, - currentPhysicalBytes: number, - maxAppendSlots = Number.MAX_SAFE_INTEGER - ): boolean { - this.reservations.clear() - for (const item of snapshot.items) { - if (!requiresTerminalSettlement(item.body)) { - continue - } - if ( - !this.reserve( - { - id: lifecycleReservationIdForItem(item.itemId), - bytes: terminalReservationBytes(item.body), - appendSlots: 1 - }, - currentPhysicalBytes, - maxBytes, - maxAppendSlots - ) - ) { - return false - } - } - for (const submission of snapshot.submissions) { - if (submission.dispatchState !== 'pending' && submission.dispatchState !== 'unknown') { - continue - } - // A write-ahead submission owns both its dispatch attempt and the - // terminal turn settlement. Rebuild both reservations after restart; - // restoring only the tentative turn token would let a new send consume - // the dispatch headroom still owed to this unresolved submission. - if ( - !this.reserve( - { - id: dispatchReservationId(submission.clientMessageId), - bytes: JOURNAL_DISPATCH_RESERVATION_BYTES, - appendSlots: 1 - }, - currentPhysicalBytes, - maxBytes, - maxAppendSlots - ) - ) { - return false - } - if ( - !this.reserve( - { - id: tentativeTurnReservationId(submission.clientMessageId), - bytes: JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - appendSlots: 1 - }, - currentPhysicalBytes, - maxBytes, - maxAppendSlots - ) - ) { - return false - } - } - return true - } -} - -export function lifecycleReservationIdForItem(itemId: string): string { - return `item:${itemId}` -} - -export function dispatchReservationId(clientMessageId: string): string { - return `dispatch:${clientMessageId}` -} - -export function tentativeTurnReservationId(clientMessageId: string): string { - return `tentative-turn:${clientMessageId}` -} - -export function requiresTerminalSettlement(body: AgentJournalItemBody): boolean { - if (body.kind === 'tool-call') { - return body.state === 'running' - } - if (body.kind === 'approval' || body.kind === 'question') { - return body.resolution.state === 'pending' - } - return body.kind === 'status' && body.turnLifecycle?.state === 'running' -} - -export function terminalReservationBytes(body: AgentJournalItemBody): number { - return body.kind === 'status' && body.turnLifecycle?.state === 'running' - ? JOURNAL_TURN_TERMINAL_RESERVATION_BYTES - : JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES -} diff --git a/src/main/native-chat/agent-session-journal/journal-log-file.test.ts b/src/main/native-chat/agent-session-journal/journal-log-file.test.ts deleted file mode 100644 index 98c0b5029d3..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-log-file.test.ts +++ /dev/null @@ -1,374 +0,0 @@ -import { mkdtemp, readdir, rm, writeFile } from 'node:fs/promises' -import type * as FsPromises from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { appendJournalRows, JOURNAL_SNAPSHOT_FILE, readJournalSnapshot } from './journal-log-file' -import type { JournalSnapshotFile } from './journal-log-file' -import type { JournalRow } from './journal-row-schema' -import { openAgentSessionJournal } from './journal-store-factory' -import { - projectStructuredAgentSessionStatus, - projectStructuredItemsToNativeChat -} from '../../../shared/structured-agent-session-projection' - -type FakeDirectoryHandle = { sync: ReturnType; close: ReturnType } - -let openDirectoryHook: ((path: unknown, flags: unknown) => FakeDirectoryHandle | undefined) | null = - null - -vi.mock('node:fs/promises', async (importOriginal) => { - const actual = await importOriginal() - return { - ...actual, - open: (async (...args: Parameters) => { - const fake = openDirectoryHook?.(args[0], args[1]) - return fake ?? actual.open(...args) - }) as typeof actual.open - } -}) - -let root: string - -beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-log-file-')) - openDirectoryHook = null -}) - -afterEach(async () => { - openDirectoryHook = null - await rm(root, { recursive: true, force: true }) -}) - -function validSnapshot(): JournalSnapshotFile { - return { - v: 1, - epoch: 'epoch-A', - compactedThrough: 2, - highestFence: 1, - items: [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'hi' }] }, - sequence: 2, - observedAt: 1_000 - } - ], - submissions: [], - receipts: [], - aliases: [], - tombstones: [{ itemId: 'codex:thread-1:turn-1:2', revision: 3 }], - tail: [] - } -} - -async function writeSnapshot(snapshot: unknown): Promise { - await writeFile(join(root, JOURNAL_SNAPSHOT_FILE), JSON.stringify(snapshot), 'utf-8') -} - -describe('readJournalSnapshot validation', () => { - it('accepts a well-formed snapshot, with and without the tombstones collection', async () => { - await writeSnapshot(validSnapshot()) - expect((await readJournalSnapshot(root)).status).toBe('valid') - - const { tombstones: _tombstones, ...withoutTombstones } = validSnapshot() - await writeSnapshot(withoutTombstones) - expect((await readJournalSnapshot(root)).status).toBe('valid') - }) - - it('accepts every canonical item kind and a fully-formed submission', async () => { - const snapshot = validSnapshot() - const payload = { head: 'x', byteLength: 4, digest: 'd'.repeat(64), truncated: true } - snapshot.items = [ - { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] }, - { - kind: 'tool-call', - name: 'Read', - input: { path: 'a' }, - state: 'completed', - output: payload - }, - { kind: 'diff', path: 'a.ts', patch: payload }, - { - kind: 'approval', - title: 'Run?', - detail: null, - options: [{ id: 'a', label: 'Yes' }], - resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } - }, - { - kind: 'question', - question: 'Deploy?', - options: [{ id: 'a', label: 'Yes' }], - freeTextQuestionId: 'q-free', - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 5 } - }, - { - kind: 'status', - text: 'working', - turnLifecycle: { turnId: 'turn-1', state: 'running' }, - providerFrame: { provider: 'codex', kind: 'raw', payload } - } - ].map((body, index) => ({ - itemId: `codex:thread-1:turn-1:${index + 1}`, - revision: 1, - body: body as JournalSnapshotFile['items'][number]['body'], - sequence: index + 1, - observedAt: 1_000, - ...(index === 0 ? { recovered: true as const } : {}) - })) - snapshot.compactedThrough = snapshot.items.length - snapshot.submissions = [ - { - clientMessageId: 'm-1', - fence: 1, - payloadFingerprint: 'a'.repeat(64), - dispatchState: 'accepted', - providerItemId: 'codex:thread-1:turn-1:1', - reason: null, - submittedAt: 1_000, - resolvedAt: 1_001 - } - ] - await writeSnapshot(snapshot) - expect((await readJournalSnapshot(root)).status).toBe('valid') - }) - - it('classifies a JSON-valid non-array tombstones collection as invalid instead of valid', async () => { - await writeSnapshot({ ...validSnapshot(), tombstones: {} }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects tombstone entries that would poison seeding', async () => { - for (const tombstones of [ - [{ itemId: 42, revision: 1 }], - [{ itemId: 'codex:thread-1:turn-1:1', revision: 'one' }], - [{ itemId: 'codex:thread-1:turn-1:1', revision: Number.NaN }], - ['codex:thread-1:turn-1:1'] - ]) { - await writeSnapshot({ ...validSnapshot(), tombstones }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - } - }) - - it('rejects JSON-valid nested item corruption instead of admitting it', async () => { - // A resolved question with `options: null` used to pass shallow admission and - // then throw `TypeError` in the shared projection's `options.map`. - const poisonedQuestion = validSnapshot() - poisonedQuestion.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { - kind: 'question', - question: 'Deploy?', - options: null, - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 1 } - }, - sequence: 2, - observedAt: 1_000 - } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(poisonedQuestion) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - - // Pending-prompt surfaces read `resolution.state` before anything else. - const nullResolution = validSnapshot() - nullResolution.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'question', question: 'Deploy?', options: [], resolution: null }, - sequence: 2, - observedAt: 1_000 - } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(nullResolution) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects a JSON-valid nested corruption in the retained tail', async () => { - const poisonedTail = validSnapshot() - poisonedTail.tail = [ - { - v: 1, - epoch: 'epoch-A', - seq: 3, - fence: 1, - ts: 1_000, - kind: 'item', - itemId: 'codex:thread-1:turn-1:3', - revision: 1, - body: { - kind: 'question', - question: 'Deploy?', - options: null, - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 1 } - } - } - ] as unknown as JournalSnapshotFile['tail'] - await writeSnapshot(poisonedTail) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects a submission that only carries a client message id', async () => { - const shallowSubmission = validSnapshot() - shallowSubmission.submissions = [ - { clientMessageId: 'm-1' } - ] as unknown as JournalSnapshotFile['submissions'] - await writeSnapshot(shallowSubmission) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) - - it('rejects items and counters that only look shallowly plausible', async () => { - const missingSequence = validSnapshot() - missingSequence.items = [ - { itemId: 'codex:thread-1:turn-1:1', revision: 1, body: { kind: 'status', text: 'x' } } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(missingSequence) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - - await writeSnapshot({ ...validSnapshot(), compactedThrough: Number.NaN }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - }) -}) - -describe('future-version snapshot classification', () => { - it('classifies a future version before shape validation so unknown bodies stay unreadable', async () => { - // The version can only advance because bodies changed, so a future snapshot - // legitimately carries kinds this build cannot parse. That is the - // schema-unreadable contract, not corruption. - const future = validSnapshot() as unknown as Record - future.v = 99 - future.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'future-render-kind', payload: { anything: true } }, - sequence: 2, - observedAt: 1_000 - } - ] - await writeSnapshot(future) - expect((await readJournalSnapshot(root)).status).toBe('unreadable') - }) - - it('classifies a future version as unreadable even when its shapes still parse today', async () => { - await writeSnapshot({ ...validSnapshot(), v: 99 }) - expect((await readJournalSnapshot(root)).status).toBe('unreadable') - }) - - it('treats a non-integer or sub-1 version as invalid, matching row admission', async () => { - for (const v of [0, 1.5]) { - await writeSnapshot({ ...validSnapshot(), v }) - expect((await readJournalSnapshot(root)).status).toBe('invalid') - } - }) -}) - -describe('journal startup isolation from a malformed snapshot', () => { - it('quarantines a JSON-valid malformed snapshot instead of throwing through open', async () => { - await writeSnapshot({ ...validSnapshot(), tombstones: {} }) - - const journal = await openAgentSessionJournal({ - identity: { - sessionId: 'session-1', - workspaceId: 'ws-1', - hostId: 'host-1', - agent: 'codex', - providerHandle: { kind: 'codex', threadId: 'thread-1' } - }, - journalDir: root - }) - - // Degraded exactly like other corrupt snapshots: quarantined on disk, never - // silently deleted, and the session does not adopt state it cannot trust. - const entries = await readdir(root) - expect(entries.some((entry) => entry.startsWith('quarantine-snapshot-'))).toBe(true) - expect(entries.includes(JOURNAL_SNAPSHOT_FILE)).toBe(false) - expect(journal.snapshot().items).toEqual([]) - }) -}) - -describe('reopen after a persisted JSON-valid poisoned question', () => { - it('quarantines the snapshot so reopen-to-render cannot throw in projection', async () => { - const poisoned = validSnapshot() - poisoned.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { - kind: 'question', - question: 'Deploy?', - options: null, - resolution: { state: 'resolved', selectedOptionId: 'a', resolvedBy: 'c', resolvedAt: 1 } - }, - sequence: 2, - observedAt: 1_000 - } - ] as unknown as JournalSnapshotFile['items'] - await writeSnapshot(poisoned) - - const journal = await openAgentSessionJournal({ - identity: { - sessionId: 'session-1', - workspaceId: 'ws-1', - hostId: 'host-1', - agent: 'codex', - providerHandle: { kind: 'codex', threadId: 'thread-1' } - }, - journalDir: root - }) - - // The poisoned item must land in quarantine, not in the reopened state: - // pre-fix it was admitted and the render path below threw - // `TypeError: Cannot read properties of null (reading 'map')`. - const entries = await readdir(root) - expect(entries.some((entry) => entry.startsWith('quarantine-snapshot-'))).toBe(true) - const items = journal.snapshot().items - expect(() => projectStructuredItemsToNativeChat(items)).not.toThrow() - expect(() => projectStructuredAgentSessionStatus(items)).not.toThrow() - expect(items).toEqual([]) - }) -}) - -describe('appendJournalRows directory fsync', () => { - const ROW: JournalRow = { - kind: 'epoch', - reason: 'session_created', - providerHandle: { kind: 'codex', threadId: 'thread-1' }, - v: 1, - epoch: 'epoch-A', - seq: 1, - fence: 0, - ts: 1_000 - } - - function hookDirectoryOpen(sync: ReturnType): FakeDirectoryHandle { - const fake: FakeDirectoryHandle = { sync, close: vi.fn(async () => undefined) } - openDirectoryHook = (path, flags) => (path === root && flags === 'r' ? fake : undefined) - return fake - } - - it('closes the directory handle when directory fsync fails', async () => { - const fake = hookDirectoryOpen( - vi.fn(async () => { - throw new Error('EINVAL: sync') - }) - ) - - // Tolerating unsupported directory fsync must not turn into a leak. - await expect(appendJournalRows(root, [ROW])).resolves.toBeUndefined() - expect(fake.sync).toHaveBeenCalledTimes(1) - expect(fake.close).toHaveBeenCalledTimes(1) - }) - - it('closes the directory handle when directory fsync succeeds', async () => { - const fake = hookDirectoryOpen(vi.fn(async () => undefined)) - - await expect(appendJournalRows(root, [ROW])).resolves.toBeUndefined() - expect(fake.sync).toHaveBeenCalledTimes(1) - expect(fake.close).toHaveBeenCalledTimes(1) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-log-file.ts b/src/main/native-chat/agent-session-journal/journal-log-file.ts deleted file mode 100644 index 44cbc26d0d5..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-log-file.ts +++ /dev/null @@ -1,365 +0,0 @@ -// On-disk layout for one session's journal. -// -// /log.jsonl append-only rows, fsynced before the caller is told the write landed -// /snapshot.json folded state at a compaction boundary PLUS the retained tail -// /blobs/ bounded-payload remainders -// -// The snapshot carries its own tail so compaction is one atomic write. A crash -// between publishing the snapshot and truncating the log leaves the log a -// superset of the tail, and recovery unions the two by sequence — never a hole. - -import { appendFile, mkdir, open, readFile, stat, type FileHandle } from 'node:fs/promises' -import { randomUUID } from 'node:crypto' -import { join } from 'node:path' -import { durableWriteTempPath, renameDurable, writeFileDurable } from '../../durable-file-write' -import { - AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - type AgentJournalRenderItem, - type AgentJournalSubmission -} from '../../../shared/agent-session-journal-types' -import { - isAdmissibleAgentJournalRenderItem, - isAdmissibleAgentJournalSubmission -} from '../../../shared/agent-session-journal-schemas' -import { parseJournalRow, serializeJournalRow, type JournalRow } from './journal-row-schema' -import { assertJournalPhysicalCapacity } from './journal-physical-quota' - -export const JOURNAL_LOG_FILE = 'log.jsonl' -export const JOURNAL_SNAPSHOT_FILE = 'snapshot.json' - -export type JournalSnapshotFile = { - v: number - epoch: string - /** Highest sequence folded into `items`; the tail starts after it. */ - compactedThrough: number - /** Fence monotonicity survives compaction and restart. */ - highestFence: number - items: AgentJournalRenderItem[] - submissions: AgentJournalSubmission[] - /** Receipts outlive the rows that minted them: a client reconnecting after - * compaction must still get the same answer instead of re-sending. */ - receipts: { - clientMessageId: string - providerItemId: string - epoch: string - sequence: number - acceptedAt: number - }[] - /** Provider item id → submission slot, preserved so a post-compaction echo - * still reconciles into the bubble it belongs to. */ - aliases: { providerItemId: string; itemId: string }[] - tombstones: { itemId: string; revision: number }[] - /** Bounded by compaction retention; used to deduplicate a replayed settlement. */ - appliedSettlementIds?: string[] - tail: JournalRow[] -} - -export type JournalReadResult = { - rows: JournalRow[] - /** True when a line used a schema version this build cannot read. Reading - * STOPS there — the row must not be skipped — and the host degrades to - * read-only: no writes, no compaction, no deletion. */ - unreadable: boolean - /** Lines that failed to parse for reasons other than schema version. */ - malformed: number - /** Raw suffix beginning at the first malformed line, if any. */ - remainder?: string - /** Distinguishes an absent/empty log from bytes that could not name an epoch. */ - hasBytes: boolean -} - -export type JournalSnapshotReadResult = - | { status: 'missing' } - | { status: 'valid'; snapshot: JournalSnapshotFile } - | { status: 'invalid' } - /** A future schema version: unreadable by this build, not corrupt. The file - * stays authoritative in place and the caller degrades to read-only. */ - | { status: 'unreadable' } - -const NEWLINE_BYTE = 0x0a - -export async function ensureJournalDir(journalDir: string): Promise { - await mkdir(journalDir, { recursive: true }) -} - -export async function readJournalSnapshot(journalDir: string): Promise { - try { - const raw = await readFile(join(journalDir, JOURNAL_SNAPSHOT_FILE), 'utf-8') - const parsed: unknown = JSON.parse(raw) - const version = snapshotSchemaVersion(parsed) - if (version === null) { - return { status: 'invalid' } - } - // Version is classified BEFORE shape validation, matching row admission: a - // version only advances because bodies changed, so a valid newer snapshot - // carries kinds this build cannot parse — unreadable, never corruption. - if (version > AGENT_SESSION_JOURNAL_SCHEMA_VERSION) { - return { status: 'unreadable' } - } - return isJournalSnapshotFile(parsed) - ? { status: 'valid', snapshot: parsed } - : { status: 'invalid' } - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return { status: 'missing' } - } - if (error instanceof SyntaxError) { - return { status: 'invalid' } - } - throw error - } -} - -export async function quarantineInvalidJournalSnapshot( - journalDir: string, - quota?: { sessionId: string; maxBytes: number } -): Promise { - // Rename is normally same-filesystem and size-neutral, but admission must - // happen before retaining evidence so a full journal never creates an - // unbounded quarantine artifact (or relies on a copy fallback). - if (quota) { - const source = join(journalDir, JOURNAL_SNAPSHOT_FILE) - // Account for the complete source bytes: rename is usually neutral, but a - // cross-device/filesystem fallback may briefly retain both inodes. - const sourceBytes = await stat(source) - .then((info) => info.size) - .catch((error) => { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return 0 - } - throw error - }) - await assertJournalPhysicalCapacity({ - journalDir, - ...quota, - peakAdditionalBytes: sourceBytes - }) - } - const source = join(journalDir, JOURNAL_SNAPSHOT_FILE) - const target = join(journalDir, `quarantine-snapshot-${Date.now()}-${randomUUID()}.json`) - await renameDurable(source, target) - return target -} - -export async function writeJournalSnapshotFile( - journalDir: string, - snapshot: JournalSnapshotFile -): Promise { - const target = join(journalDir, JOURNAL_SNAPSHOT_FILE) - await writeFileDurable(durableWriteTempPath(target), target, JSON.stringify(snapshot)) -} - -export async function readJournalLog(journalDir: string): Promise { - let raw: string - try { - raw = await readFile(join(journalDir, JOURNAL_LOG_FILE), 'utf-8') - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return { rows: [], unreadable: false, malformed: 0, hasBytes: false } - } - throw error - } - const rows: JournalRow[] = [] - let unreadable = false - let malformed = 0 - const lines = raw.split('\n') - let offset = 0 - for (const line of lines) { - if (!line.trim()) { - offset += line.length + 1 - continue - } - const parsed = parseJournalRow(line) - if (parsed.ok) { - rows.push(parsed.row) - offset += line.length + 1 - continue - } - if (parsed.unreadable) { - unreadable = true - return { rows, unreadable, malformed, remainder: raw.slice(offset), hasBytes: raw.length > 0 } - } - malformed += 1 - return { rows, unreadable, malformed, remainder: raw.slice(offset), hasBytes: raw.length > 0 } - } - return { rows, unreadable, malformed, hasBytes: raw.length > 0 } -} - -/** Row admission requires an integer version of at least 1; a snapshot whose - * version cannot even be read is malformed, not a schema statement. */ -function snapshotSchemaVersion(value: unknown): number | null { - const snapshot = recordOf(value) - const version = snapshot?.v - return typeof version === 'number' && Number.isInteger(version) && version >= 1 ? version : null -} - -function isJournalSnapshotFile(value: unknown): value is JournalSnapshotFile { - if (!value || typeof value !== 'object' || Array.isArray(value)) { - return false - } - const snapshot = value as Record - return ( - typeof snapshot.v === 'number' && - typeof snapshot.epoch === 'string' && - snapshot.epoch.length > 0 && - Number.isInteger(snapshot.compactedThrough) && - (snapshot.compactedThrough as number) >= 0 && - Number.isInteger(snapshot.highestFence) && - // Deep discriminated admission: a JSON-valid item with a corrupt nested - // shape (e.g. a question whose options are null) must land this snapshot - // in quarantine rather than throw later in projection or prompt render. - arrayOf(snapshot.items, isAdmissibleAgentJournalRenderItem) && - arrayOf(snapshot.submissions, isAdmissibleAgentJournalSubmission) && - arrayOf(snapshot.receipts, isReceipt) && - arrayOf(snapshot.aliases, isAlias) && - // Older snapshots predate tombstones; absence is fine, a non-array is not — - // seeding iterates this collection, so a JSON-valid wrong shape must land - // in quarantine rather than throw through startup restoration. - (snapshot.tombstones === undefined || arrayOf(snapshot.tombstones, isTombstone)) && - (snapshot.appliedSettlementIds === undefined || - arrayOf(snapshot.appliedSettlementIds, (entry) => typeof entry === 'string')) && - arrayOf(snapshot.tail, (row) => parseJournalRow(JSON.stringify(row)).ok) - ) -} - -function arrayOf(value: unknown, predicate: (entry: unknown) => boolean): value is unknown[] { - return Array.isArray(value) && value.every(predicate) -} - -function recordOf(value: unknown): Record | null { - return value && typeof value === 'object' && !Array.isArray(value) - ? (value as Record) - : null -} - -function isTombstone(value: unknown): boolean { - const tombstone = recordOf(value) - return Boolean( - tombstone && typeof tombstone.itemId === 'string' && Number.isInteger(tombstone.revision) - ) -} - -function isReceipt(value: unknown): boolean { - const receipt = recordOf(value) - return Boolean( - receipt && - typeof receipt.clientMessageId === 'string' && - typeof receipt.providerItemId === 'string' && - typeof receipt.epoch === 'string' && - typeof receipt.sequence === 'number' && - typeof receipt.acceptedAt === 'number' - ) -} - -function isAlias(value: unknown): boolean { - const alias = recordOf(value) - return Boolean( - alias && typeof alias.providerItemId === 'string' && typeof alias.itemId === 'string' - ) -} - -/** - * Append rows and fsync before returning. The caller treats a resolved promise - * as "this row survives a power loss" — the write-ahead submission row depends - * on exactly that, so this must never be relaxed to a buffered write. - */ -export async function appendJournalRows( - journalDir: string, - rows: readonly JournalRow[] -): Promise { - if (rows.length === 0) { - return - } - const path = join(journalDir, JOURNAL_LOG_FILE) - // A process death can leave a final JSON fragment without its newline. Never - // concatenate a new durable row onto that fragment: truncate the torn tail - // first, then fsync the repair before acknowledging this append. - try { - await repairJournalLogTail(path) - } catch (error) { - if ((error as NodeJS.ErrnoException).code !== 'ENOENT') { - throw error - } - // The append below creates a missing log. - } - const payload = `${rows.map(serializeJournalRow).join('\n')}\n` - await appendFile(path, payload, 'utf-8') - const handle = await open(path, 'r+') - try { - await handle.sync() - } finally { - await handle.close() - } - let directory: FileHandle | undefined - try { - directory = await open(journalDir, 'r') - await directory.sync() - } catch { - // Directory fsync is unavailable on some platforms (notably Windows). - } finally { - // The tolerance above must not leak the descriptor when open succeeded - // but sync failed — one leaked handle per append adds up fast. - await directory?.close().catch(() => undefined) - } -} - -/** Repair only a torn final row. The normal append path reads one byte; scanning - * backward is reserved for the crash-recovery case and never rereads the log. */ -async function repairJournalLogTail(path: string): Promise { - const handle = await open(path, 'r+') - try { - const { size } = await handle.stat() - if (size === 0) { - return - } - const lastByte = Buffer.alloc(1) - await handle.read(lastByte, 0, 1, size - 1) - if (lastByte[0] === NEWLINE_BYTE) { - return - } - - const scanChunkBytes = 64 * 1024 - let scanEnd = size - let boundary = -1 - while (scanEnd > 0 && boundary === -1) { - const scanStart = Math.max(0, scanEnd - scanChunkBytes) - const chunk = Buffer.alloc(scanEnd - scanStart) - await handle.read(chunk, 0, chunk.length, scanStart) - const newline = chunk.lastIndexOf(NEWLINE_BYTE) - if (newline !== -1) { - boundary = scanStart + newline - } - scanEnd = scanStart - } - - const lineStart = boundary + 1 - const finalLine = Buffer.alloc(size - lineStart) - await handle.read(finalLine, 0, finalLine.length, lineStart) - // A whole row that merely lost its newline is kept; a real fragment goes. - const complete = parseJournalRow(finalLine.toString('utf-8')).ok - await (complete ? handle.write('\n', size) : handle.truncate(lineStart)) - await handle.sync() - } finally { - await handle.close() - } -} - -export async function quarantineJournalRemainder( - journalDir: string, - remainder: string -): Promise { - const path = join(journalDir, `quarantine-${Date.now()}-${randomUUID()}.jsonl`) - await writeFileDurable(durableWriteTempPath(path), path, remainder) - return path -} - -/** Replace the log with exactly the retained tail. Runs only after the snapshot - * carrying that tail is durable, so a crash here loses nothing. */ -export async function rewriteJournalLog( - journalDir: string, - rows: readonly JournalRow[] -): Promise { - const target = join(journalDir, JOURNAL_LOG_FILE) - const payload = rows.length ? `${rows.map(serializeJournalRow).join('\n')}\n` : '' - await writeFileDurable(durableWriteTempPath(target), target, payload) -} diff --git a/src/main/native-chat/agent-session-journal/journal-open.ts b/src/main/native-chat/agent-session-journal/journal-open.ts index 94fb3137f26..350d2e9cfb9 100644 --- a/src/main/native-chat/agent-session-journal/journal-open.ts +++ b/src/main/native-chat/agent-session-journal/journal-open.ts @@ -1,194 +1,205 @@ -// Loading a journal from disk: snapshot + log → folded state. +// Loading a journal: the session projection names the live epoch, and that +// epoch's rows are folded through the reducer in sequence order. // -// The snapshot is authoritative for the current epoch. Log rows belonging to a -// superseded epoch are dropped rather than merged — a crash between publishing -// a rollover snapshot and rewriting the log is the ordinary way that happens. -// A gap in the surviving sequence is corruption, and the caller rolls the epoch +// There is no snapshot to anchor to and no superseded-epoch rows to drop — a +// roll deletes them in the same transaction that publishes the new epoch. A gap +// in the surviving sequence is corruption, and the caller rolls the epoch // rather than rendering a partial timeline. -import type { AgentJournalSubmission } from '../../../shared/agent-session-journal-types' +import { existsSync } from 'node:fs' +import type Database from '../../sqlite/sync-database' import { findSequenceGap } from './journal-cursor' -import { - quarantineInvalidJournalSnapshot, - readJournalLog, - readJournalSnapshot, - type JournalSnapshotFile -} from './journal-log-file' +import { openJournalDatabase } from './journal-database' +import { journalDatabaseFile } from './journal-paths' import { applyJournalRow, createJournalReducerState, - rememberAppliedSettlementId, type JournalReducerState } from './journal-reducer' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' +import { + readJournalEpochRows, + readJournalRowsAfter, + readJournalSessionEpoch +} from './journal-row-table' +import { JOURNAL_REPAIR_DISCLOSURE_ITEM_ID } from './journal-repair-disclosure' +import { pendingJournalRepairSequence } from './journal-repair-marker' +import { parseJournalRow, type JournalRow } from './journal-row-schema' + +/** Every epoch row is sequence 1, and no compaction moves that floor. */ +const FIRST_JOURNAL_SEQUENCE = 1 export type JournalLoad = { state: JournalReducerState - /** Rows still individually replayable, oldest first. */ - tailRows: JournalRow[] - /** Highest sequence folded into the snapshot; the tail starts after it. */ - compactedThrough: number - /** A future schema version was met: no writes, no compaction, no deletion. */ + /** A future schema version was met: no writes, no deletion. */ readOnly: boolean /** Set when the surviving prefix is unusable and the caller must roll the epoch. */ corrupt: boolean - /** Log lines skipped because they failed to parse (schema-version rows are + /** Rows skipped because their body failed to parse (future-version rows are * `readOnly`, never counted here). The store discloses these in the timeline. */ malformedRows: number - sizeBytes: number - /** Raw unreadable suffix retained for quarantine instead of deletion. */ - quarantineRemainder?: string + /** Directory-internal: the first sequence of an unusable suffix. The store + * deletes from here before it accepts a write; a probe leaves it alone. */ + truncateFrom?: number } -/** Returns null when no journal exists yet for this session. */ -export async function loadJournal( - journalDir: string, - sessionId: string, - quota?: { maxBytes: number } -): Promise { - const snapshotRead = await readJournalSnapshot(journalDir) - if (snapshotRead.status === 'unreadable') { - // Written by a newer schema: not corrupt, so never quarantined. The file - // stays authoritative in place, and this build must not write, compact, - // delete, or render a partial timeline from rows it cannot anchor to the - // snapshot it cannot read. +/** + * Replay on a connection this function does NOT own. Returns null when the + * session has no journal yet. + */ +export function replayJournal( + db: Database.Database, + readOnly: boolean, + sessionId: string +): JournalLoad | null { + if (readOnly) { return emptyReadOnlyLoad(sessionId) } - if (snapshotRead.status === 'invalid') { - await quarantineInvalidJournalSnapshot( - journalDir, - quota ? { sessionId, maxBytes: quota.maxBytes } : undefined - ) - } - const snapshot = snapshotRead.status === 'valid' ? snapshotRead.snapshot : null - const log = await readJournalLog(journalDir) - const epoch = resolveEpoch(snapshot, log.rows) + const epoch = readJournalSessionEpoch(db, sessionId) if (!epoch) { - return snapshotRead.status === 'invalid' || log.hasBytes ? emptyReadOnlyLoad(sessionId) : null + return null + } + const state = createJournalReducerState(sessionId, epoch) + const stored = readJournalEpochRows(db, sessionId, epoch) + // A partial repair keeps its prefix, so the surviving rows look contiguous and + // anchored however much of the timeline it deleted. Its marker is what still + // says otherwise, naming the sequence past which the epoch would be its own + // history again. + const repairedFrom = pendingJournalRepairSequence(db, sessionId, epoch) + const rows: JournalRow[] = [] + let malformedRows = 0 + let latched = false + let truncateFrom: number | undefined + for (const entry of stored) { + const parsed = parseJournalRow(entry.rowJson) + if (parsed.ok) { + rows.push(parsed.row) + continue + } + // Reading STOPS at the first row this build cannot represent. A future + // version latches read-only; anything else is one skipped row, disclosed. + truncateFrom = entry.seq + if (parsed.unreadable) { + latched = true + } else { + malformedRows = 1 + } + break } - const compactedThrough = snapshot?.epoch === epoch ? snapshot.compactedThrough : 0 - const state = seedState(sessionId, epoch, snapshot?.epoch === epoch ? snapshot : null) - const liveRows = log.rows.filter((row) => row.epoch === epoch) - let tailRows = unionBySequence(snapshot?.epoch === epoch ? snapshot.tail : [], liveRows, epoch) - - const oldest = tailRows[0]?.seq ?? compactedThrough + 1 + // Anchored at 1, never at the first row that HAPPENS to remain: nothing trims + // a prefix, so a missing epoch row is a hole like any other and everything + // behind it is unanchored. Validating from `rows[0].seq` would call the + // leftovers contiguous and leave them out of the repair that runs before + // provider history replaces the epoch. const gap = findSequenceGap( - tailRows.map((row) => row.seq), - oldest + rows.map((row) => row.seq), + FIRST_JOURNAL_SEQUENCE ) - // A hole below the snapshot boundary is unrecoverable too: the snapshot only - // covers `compactedThrough`, so a tail that starts above it lost rows. - let corrupt = Boolean(gap) || oldest > compactedThrough + 1 || log.malformed > 0 - let quarantineRemainder = log.remainder if (gap) { - const firstBad = tailRows.findIndex((row, index) => { - const expected = (tailRows[0]?.seq ?? compactedThrough + 1) + index - return row.seq !== expected - }) + const firstBad = rows.findIndex((row, index) => row.seq !== FIRST_JOURNAL_SEQUENCE + index) if (firstBad !== -1) { - const suffix = tailRows.slice(firstBad) - tailRows = tailRows.slice(0, firstBad) - quarantineRemainder ??= `${suffix.map((row) => JSON.stringify(row)).join('\n')}\n` + truncateFrom = rows[firstBad]?.seq ?? truncateFrom + rows.length = firstBad } } - - for (const row of tailRows) { - if (row.seq > compactedThrough) { - applyJournalRow(state, row) - } + // Contiguity from 1 is not the whole invariant: sequence 1 has to BE the epoch + // row. An ordinary row there is an epoch nothing anchors, and replaying it as + // clean is how a repaired journal silently adopts a timeline whose real + // history was never rebuilt. + if (rows.length > 0 && rows[0]?.kind !== 'epoch') { + truncateFrom = rows[0]?.seq ?? truncateFrom + rows.length = 0 } - state.oldestSequence = oldest - state.lastSequence = Math.max(state.lastSequence, compactedThrough) + for (const row of rows) { + applyJournalRow(state, row) + } + state.oldestSequence = FIRST_JOURNAL_SEQUENCE + // A latched journal reduces to nothing by design; only a writable one can be + // held to the anchor. + const unanchored = !latched && rows[0]?.kind !== 'epoch' return { state, - tailRows, - compactedThrough, - // A future-version snapshot never reaches here: it is classified - // unreadable above, so `valid` implies a version this build can write. - readOnly: log.unreadable, - corrupt, - malformedRows: log.malformed, - sizeBytes: tailRows.reduce((total, row) => total + journalRowByteLength(row), 0), - quarantineRemainder + readOnly: latched, + corrupt: + Boolean(gap) || + malformedRows > 0 || + unanchored || + (repairedFrom !== null && awaitsRebuild(rows, repairedFrom)) || + awaitsProviderHistory(rows), + malformedRows, + ...(truncateFrom !== undefined && !latched ? { truncateFrom } : {}) + } +} + +/** + * The epoch a total repair published, still holding nothing but its own anchor + * and disclosure. The rows it dropped were never reconstructed, so provider + * history has to be retried rather than this being called a clean timeline. + */ +function awaitsProviderHistory(rows: readonly JournalRow[]): boolean { + const anchor = rows[0] + if (anchor?.kind !== 'epoch' || anchor.reason !== 'unreconcilable_prefix') { + return false + } + // The anchor sits at sequence 1, so content of the epoch's own starts at 2. + return awaitsRebuild(rows, FIRST_JOURNAL_SEQUENCE + 1) +} + +/** + * True while everything at or above `contentFrom` is the repair's own + * bookkeeping: the deleted history was never rebuilt, so the provider has to be + * asked again. The moment the session writes content of its own past that + * sequence the epoch IS its own history, and the retry stops rather than a + * later import replacing rows the user has since seen. + */ +function awaitsRebuild(rows: readonly JournalRow[], contentFrom: number): boolean { + return rows.every( + (row) => + row.seq < contentFrom || + (row.kind === 'item' && row.itemId === JOURNAL_REPAIR_DISCLOSURE_ITEM_ID) + ) +} + +/** Rows after a cursor, in sequence order. Stops at the first row this build + * cannot parse, exactly as replay does. */ +export function readJournalRowsAfterCursor( + db: Database.Database, + sessionId: string, + epoch: string, + afterSequence: number +): JournalRow[] { + const rows: JournalRow[] = [] + for (const stored of readJournalRowsAfter(db, sessionId, epoch, afterSequence)) { + const parsed = parseJournalRow(stored.rowJson) + if (!parsed.ok) { + break + } + rows.push(parsed.row) + } + return rows +} + +/** Standalone probe. Opens its own connection and closes it before returning, + * so a caller holding only the returned value holds no handle. */ +export function loadJournal(journalDir: string, sessionId: string): JournalLoad | null { + const dbPath = journalDatabaseFile(journalDir) + if (!existsSync(dbPath)) { + return null + } + const opened = openJournalDatabase(dbPath) + try { + return replayJournal(opened.db, opened.readOnly, sessionId) + } finally { + opened.db.close() } } function emptyReadOnlyLoad(sessionId: string): JournalLoad { - const state = createJournalReducerState(sessionId, '') return { - state, - tailRows: [], - compactedThrough: 0, + state: createJournalReducerState(sessionId, ''), readOnly: true, corrupt: false, - malformedRows: 0, - sizeBytes: 0 + malformedRows: 0 } } - -/** The snapshot names the live epoch; without one, the newest valid row does. */ -function resolveEpoch(snapshot: JournalSnapshotFile | null, rows: JournalRow[]): string | null { - if (snapshot?.epoch) { - return snapshot.epoch - } - return rows.at(-1)?.epoch ?? null -} - -function seedState( - sessionId: string, - epoch: string, - snapshot: JournalSnapshotFile | null -): JournalReducerState { - const state = createJournalReducerState(sessionId, epoch) - if (!snapshot) { - return state - } - for (const item of snapshot.items) { - state.items.set(item.itemId, item) - } - for (const submission of snapshot.submissions) { - state.submissions.set(submission.clientMessageId, { ...submission } as AgentJournalSubmission) - } - for (const receipt of snapshot.receipts) { - state.receipts.set(receipt.clientMessageId, { - clientMessageId: receipt.clientMessageId, - providerItemId: receipt.providerItemId, - cursor: { epoch: receipt.epoch, sequence: receipt.sequence }, - acceptedAt: receipt.acceptedAt - }) - } - for (const alias of snapshot.aliases) { - state.aliases.set(alias.providerItemId, alias.itemId) - } - for (const tombstone of snapshot.tombstones ?? []) { - state.tombstones.set(tombstone.itemId, tombstone.revision) - } - for (const settlementId of snapshot.appliedSettlementIds ?? []) { - rememberAppliedSettlementId(state, settlementId) - } - state.highestFence = snapshot.highestFence ?? 0 - state.lastSequence = snapshot.compactedThrough - state.oldestSequence = snapshot.compactedThrough + 1 - return state -} - -/** Merge the snapshot's retained tail with the live log, preferring the log's - * copy of any sequence both hold, and dropping rows from a superseded epoch. */ -function unionBySequence( - retained: readonly JournalRow[], - live: readonly JournalRow[], - epoch: string -): JournalRow[] { - const bySequence = new Map() - for (const row of retained) { - if (row.epoch === epoch) { - bySequence.set(row.seq, row) - } - } - for (const row of live) { - bySequence.set(row.seq, row) - } - return [...bySequence.values()].sort((a, b) => a.seq - b.seq) -} diff --git a/src/main/native-chat/agent-session-journal/journal-paths.ts b/src/main/native-chat/agent-session-journal/journal-paths.ts index 87616b25b96..c21a7675f00 100644 --- a/src/main/native-chat/agent-session-journal/journal-paths.ts +++ b/src/main/native-chat/agent-session-journal/journal-paths.ts @@ -40,3 +40,10 @@ export function journalDirectoryFor( export function defaultJournalRoot(): Promise { return Promise.resolve(getAppEnvironment().getPath('userData')) } + +export const JOURNAL_DATABASE_FILE = 'journal.db' + +/** The session's SQLite database, inside the directory `journalDirectoryFor` names. */ +export function journalDatabaseFile(journalDir: string): string { + return join(journalDir, JOURNAL_DATABASE_FILE) +} diff --git a/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts b/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts index b47d1a1511c..ea09ecf6df3 100644 --- a/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts +++ b/src/main/native-chat/agent-session-journal/journal-payload-bounds.ts @@ -1,11 +1,9 @@ // Payload bounds for tool output and diffs. // -// A looping agent must not be able to fill the host disk, and a 40 MB tool -// result must not be inlined into a row that every reconnecting client -// replays. A bounded payload keeps a head plus the original byte length and -// digest; the remainder lives in the content-addressed blob store under the -// same retention as its epoch. Crossing a bound is always marked — never a -// silent drop. +// A 40 MB tool result must not be inlined into a row that every reconnecting +// client replays. A bounded payload keeps a head plus the original byte length +// and digest; the remainder is discarded. Crossing a bound is always marked — +// never a silent drop. import { createHash } from 'node:crypto' import type { AgentJournalBoundedPayload } from '../../../shared/agent-session-journal-types' @@ -13,18 +11,10 @@ import type { AgentJournalBoundedPayload } from '../../../shared/agent-session-j export type JournalPayloadLimits = { /** Bytes of the payload kept inline on the row. */ inlineHeadBytes: number - /** Total bytes of journal rows one session may hold before appends are refused. */ - maxSessionBytes: number - /** Appends allowed inside `appendWindowMs`, bounding a runaway agent's rate. */ - maxAppendsPerWindow: number - appendWindowMs: number } export const DEFAULT_JOURNAL_PAYLOAD_LIMITS: JournalPayloadLimits = { - inlineHeadBytes: 16 * 1024, - maxSessionBytes: 256 * 1024 * 1024, - maxAppendsPerWindow: 5000, - appendWindowMs: 60_000 + inlineHeadBytes: 16 * 1024 } /** Marker appended to a clipped inline string so the UI never presents a @@ -38,10 +28,8 @@ export function digestPayload(payload: string): string { return createHash('sha256').update(payload, 'utf8').digest('hex') } -/** - * Clip `payload` to the inline head. `truncated` means the remainder must be - * written to the blob store under `digest` before the row is appended. - */ +/** Clip `payload` to the inline head. `truncated` means the remainder was + * discarded; `digest` and `byteLength` describe the original. */ export function boundPayload( payload: string, limits: JournalPayloadLimits @@ -60,7 +48,7 @@ export function boundPayload( } /** Bound a plain string that must stay a string (a tool-result block's output), - * keeping the explicit marker inline. Returns the blob payload to persist. */ + * keeping the explicit marker inline. */ export function boundInlineText( payload: string, limits: JournalPayloadLimits @@ -75,7 +63,7 @@ export function boundInlineText( } } -/** Keep arbitrary tool input JSON bounded before lifecycle admission. */ +/** Keep arbitrary tool input JSON bounded before it reaches a row. */ export function boundToolInput(input: unknown, limits: JournalPayloadLimits): unknown { let encoded: string try { diff --git a/src/main/native-chat/agent-session-journal/journal-physical-quota.test.ts b/src/main/native-chat/agent-session-journal/journal-physical-quota.test.ts deleted file mode 100644 index b3ee4699fc0..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-physical-quota.test.ts +++ /dev/null @@ -1,125 +0,0 @@ -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' -import type { - AgentJournalItemBody, - AgentJournalItemIdentity, - AgentSessionJournalIdentity -} from '../../../shared/agent-session-journal-types' -import { JOURNAL_SNAPSHOT_FILE } from './journal-log-file' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalDirectoryBytes } from './journal-physical-quota' -import { openAgentSessionJournal } from './journal-store-factory' - -const IDENTITY: AgentSessionJournalIdentity = { - sessionId: 'session-1', - workspaceId: 'ws-1', - hostId: 'host-1', - agent: 'codex', - providerHandle: { kind: 'codex', threadId: 'thread-1' } -} - -let root: string - -function item(ordinal: number): AgentJournalItemIdentity { - return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } -} - -function body(value: string): AgentJournalItemBody { - return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } -} - -beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-quota-')) -}) - -afterEach(async () => { - await rm(root, { recursive: true, force: true }) -}) - -describe('journal physical quota peaks', () => { - it('refuses an epoch replacement whose staging peak exceeds the quota', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 8_000 } - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits, - autoCompact: false - }) - await journal.appendItem(item(1), body('old'.repeat(500)), { fence: 1 }) - const epoch = journal.epoch - - await expect( - journal.replaceEpochItems('handle_forked', 2, [ - { identity: item(2), body: body('replacement'.repeat(250)) } - ]) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toBe(epoch) - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) - }) - - it('refuses schema quarantine when its peak copy would exceed the quota', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 7_000 } - await openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - snapshot.items = [{ body: { kind: 'future', payload: 'x'.repeat(4_000) } }] - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - const reopened = await openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - - await expect(reopened.rollEpoch('schema_unreadable', 2)).rejects.toMatchObject({ - code: 'journal_bound_exceeded' - }) - - expect(reopened.isReadOnly).toBe(true) - expect((await readdir(root)).some((name) => name.startsWith('quarantine-'))).toBe(false) - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) - }) - - it('does not rename an invalid snapshot when the directory is already full', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - await openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - await writeFile(snapshotPath, '{"invalid":', 'utf8') - const current = await journalDirectoryBytes(root) - await writeFile( - join(root, 'quota-filler'), - 'x'.repeat(Math.max(0, limits.maxSessionBytes - current)), - 'utf8' - ) - - await expect( - openAgentSessionJournal({ identity: IDENTITY, journalDir: root, limits }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - expect(await readFile(snapshotPath, 'utf8')).toBe('{"invalid":') - expect((await readdir(root)).some((name) => name.startsWith('quarantine-snapshot-'))).toBe( - false - ) - }) - - it('counts pre-existing durable-write temps while staging an epoch replacement', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 8_000 } - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits, - autoCompact: false - }) - await journal.appendItem(item(1), body('old'), { fence: 1 }) - // Simulate a temp left by a crash. Replacement must refuse before writing - // its epoch row or creating a staging blob beside this file. - await writeFile(join(root, 'snapshot.json.crashed-write.tmp'), 'x'.repeat(7_500), 'utf8') - const epoch = journal.epoch - - await expect( - journal.replaceEpochItems('handle_forked', 2, [{ identity: item(2), body: body('new') }]) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - - expect(journal.epoch).toBe(epoch) - expect((await readdir(root)).some((name) => name.startsWith('.epoch-replacement-'))).toBe(false) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-physical-quota.ts b/src/main/native-chat/agent-session-journal/journal-physical-quota.ts deleted file mode 100644 index 478451b615c..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-physical-quota.ts +++ /dev/null @@ -1,41 +0,0 @@ -import { lstat, readdir } from 'node:fs/promises' -import type { Dirent } from 'node:fs' -import { join } from 'node:path' -import { AgentSessionJournalError } from './journal-write-guards' - -/** Counts every physical file owned by one session, including blobs, durable - * write temps, and retained quarantine evidence. Symlinks are charged as files - * but never followed outside the journal directory. */ -export async function journalDirectoryBytes(directory: string): Promise { - let entries: Dirent[] - try { - entries = await readdir(directory, { withFileTypes: true, encoding: 'utf8' }) - } catch (error) { - if ((error as NodeJS.ErrnoException).code === 'ENOENT') { - return 0 - } - throw error - } - let total = 0 - for (const entry of entries) { - const path = join(directory, entry.name) - total += entry.isDirectory() ? await journalDirectoryBytes(path) : (await lstat(path)).size - } - return total -} - -export async function assertJournalPhysicalCapacity(input: { - journalDir: string - sessionId: string - maxBytes: number - peakAdditionalBytes?: number -}): Promise { - const current = await journalDirectoryBytes(input.journalDir) - if (current + (input.peakAdditionalBytes ?? 0) > input.maxBytes) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${input.sessionId} reached its ${input.maxBytes}-byte physical bound` - ) - } - return current -} diff --git a/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts b/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts index fb8243b6926..58d12dfcf89 100644 --- a/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts +++ b/src/main/native-chat/agent-session-journal/journal-prompt-body-bounds.ts @@ -12,10 +12,7 @@ import { export const MAX_JOURNAL_PROMPT_OPTIONS = 64 -const JOURNAL_PROMPT_OPTION_LIMITS = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 1024 -} +const JOURNAL_PROMPT_OPTION_LIMITS = { inlineHeadBytes: 1024 } const JOURNAL_PROMPT_ID_MAX_BYTES = 1024 export function cancelledJournalPromptBody( @@ -78,11 +75,6 @@ function boundPromptIdentifier(value: string): string { if (Buffer.byteLength(value, 'utf8') <= JOURNAL_PROMPT_ID_MAX_BYTES) { return value } - const bounded = boundPayload(value, { - inlineHeadBytes: JOURNAL_PROMPT_ID_MAX_BYTES - 33, - maxSessionBytes: Number.MAX_SAFE_INTEGER, - maxAppendsPerWindow: Number.MAX_SAFE_INTEGER, - appendWindowMs: Number.MAX_SAFE_INTEGER - }) + const bounded = boundPayload(value, { inlineHeadBytes: JOURNAL_PROMPT_ID_MAX_BYTES - 33 }) return `${bounded.head}#${bounded.digest.slice(0, 32)}` } diff --git a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts index cbdc35a4698..676a37f63a7 100644 --- a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts @@ -11,7 +11,6 @@ import { applyJournalRow, createJournalReducerState, MAX_JOURNAL_APPLIED_SETTLEMENT_IDS, - referencedBlobDigests, renderJournalState, type JournalReducerState } from './journal-reducer' @@ -379,58 +378,6 @@ describe('lifecycle settlement deduplication', () => { }) }) -describe('blob retention', () => { - it('reports the digests live rows still reference', () => { - const state = fold([ - { - kind: 'item', - itemId: 'tool', - revision: 1, - body: { - kind: 'tool-call', - name: 'bash', - input: {}, - state: 'completed', - output: { head: 'x', byteLength: 999, digest: 'digest-a', truncated: true } - }, - ...base(1) - }, - { - kind: 'item', - itemId: 'inline', - revision: 1, - body: { - kind: 'tool-call', - name: 'bash', - input: {}, - state: 'completed', - output: { head: 'y', byteLength: 1, digest: 'digest-b', truncated: false } - }, - ...base(2) - } - ]) - expect([...referencedBlobDigests(state)]).toEqual(['digest-a']) - }) - - it('stops referencing a digest once its item is tombstoned', () => { - const state = fold([ - { - kind: 'item', - itemId: 'tool', - revision: 1, - body: { - kind: 'diff', - path: 'a.ts', - patch: { head: 'x', byteLength: 999, digest: 'digest-a', truncated: true } - }, - ...base(1) - }, - { kind: 'tombstone', itemId: 'tool', revision: 2, ...base(2) } - ]) - expect(referencedBlobDigests(state).size).toBe(0) - }) -}) - describe('malformed persisted item keys', () => { it('degrades a malformed-percent item id to an opaque key instead of throwing', () => { // A user-message body drives identity resolution through the key parser; diff --git a/src/main/native-chat/agent-session-journal/journal-reducer.ts b/src/main/native-chat/agent-session-journal/journal-reducer.ts index c660f80611f..3dc4385d797 100644 --- a/src/main/native-chat/agent-session-journal/journal-reducer.ts +++ b/src/main/native-chat/agent-session-journal/journal-reducer.ts @@ -9,7 +9,6 @@ import type { AgentJournalAcceptanceReceipt, - AgentJournalItemBody, AgentJournalRenderItem, AgentJournalSnapshot, AgentJournalSubmission @@ -272,26 +271,3 @@ export function renderJournalState(state: JournalReducerState): AgentJournalSnap submissions: [...state.submissions.values()].sort((a, b) => a.submittedAt - b.submittedAt) } } - -/** Blob digests one body points at. A retained row can outlive its render item - * (a tombstone drops the item), so compaction reads rows through this too. */ -export function blobDigestsInBody(body: AgentJournalItemBody, into: Set): void { - if (body.kind === 'tool-call' && body.output?.truncated) { - into.add(body.output.digest) - } - if (body.kind === 'diff' && body.patch.truncated) { - into.add(body.patch.digest) - } - if (body.kind === 'status' && body.providerFrame?.payload.truncated) { - into.add(body.providerFrame.payload.digest) - } -} - -/** Digests referenced by live rows, so compaction knows which blobs to keep. */ -export function referencedBlobDigests(state: JournalReducerState): Set { - const digests = new Set() - for (const item of state.items.values()) { - blobDigestsInBody(item.body, digests) - } - return digests -} diff --git a/src/main/native-chat/agent-session-journal/journal-repair-disclosure.ts b/src/main/native-chat/agent-session-journal/journal-repair-disclosure.ts new file mode 100644 index 00000000000..fee5edf142b --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-repair-disclosure.ts @@ -0,0 +1,32 @@ +// What a repair tells the user it did. +// +// The identity is a constant because replay reads it back: an epoch holding +// nothing but its anchor and this row is a repair that has not been +// reconstructed yet, not a timeline. + +import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import type { AgentJournalItemIdentity } from '../../../shared/agent-session-journal-types' + +/** One stable identity, so a reopen upserts the same row instead of adding one. */ +export const JOURNAL_REPAIR_DISCLOSURE_IDENTITY: AgentJournalItemIdentity = { + provider: 'orca', + clientMessageId: 'journal-malformed-lines' +} + +export const JOURNAL_REPAIR_DISCLOSURE_ITEM_ID = agentJournalItemKey( + JOURNAL_REPAIR_DISCLOSURE_IDENTITY +) + +export type JournalRepairDisclosure = { + identity: AgentJournalItemIdentity + body: { kind: 'status'; text: string } +} + +/** Disclosed when a repair skipped a row it could not read. */ +export function journalRepairDisclosure(input: { malformedRows: number }): JournalRepairDisclosure { + const lines = `${input.malformedRows} journal line${input.malformedRows === 1 ? '' : 's'}` + return { + identity: JOURNAL_REPAIR_DISCLOSURE_IDENTITY, + body: { kind: 'status', text: `${lines} could not be read` } + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-repair-marker.ts b/src/main/native-chat/agent-session-journal/journal-repair-marker.ts new file mode 100644 index 00000000000..f9ef09e688d --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-repair-marker.ts @@ -0,0 +1,69 @@ +// The standing demand for a rebuild that a repair leaves behind. +// +// A repair that empties the epoch republishes an `unreconcilable_prefix` anchor, +// and replay reads that back as history still owed. A repair that KEEPS a prefix +// has no such anchor to publish and — for a plain sequence gap — no malformed +// row to disclose either, so nothing on disk would record that the deleted +// suffix was never reconstructed. This marker is that record, written in the +// SAME transaction as the deletion: a crash between the two would otherwise +// leave the rows gone with nothing left asking for them back. +// +// It records the first sequence at which the epoch would hold content of its +// own again, because it retires under exactly the rule the emptied-epoch anchor +// takes: a fresh epoch carries the rebuild, and a session that writes past that +// sequence owns the epoch and stops the retry. + +import type Database from '../../sqlite/sync-database' +import { deleteJournalRowSuffix } from './journal-row-table' + +const SELECT_REPAIR = 'SELECT epoch, content_from FROM journal_repairs WHERE session_id = ?' +const UPSERT_REPAIR = `INSERT INTO journal_repairs (session_id, epoch, content_from, repaired_at) +VALUES (?, ?, ?, ?) +ON CONFLICT(session_id) DO UPDATE SET + epoch = excluded.epoch, content_from = excluded.content_from, repaired_at = excluded.repaired_at` +const DELETE_REPAIR = 'DELETE FROM journal_repairs WHERE session_id = ?' + +/** + * The sequence a pending repair on THIS epoch left free, or null when none is + * pending. Epoch-scoped: a marker raised on an epoch that has since been + * superseded says nothing about the live one. + */ +export function pendingJournalRepairSequence( + db: Database.Database, + sessionId: string, + epoch: string +): number | null { + const row = db.prepare(SELECT_REPAIR).get(sessionId) as + | { epoch?: string; content_from?: number } + | undefined + return row?.epoch === epoch ? (row.content_from ?? null) : null +} + +/** Retires the marker. Called from inside the epoch transactions, whose new + * epoch is the rebuilt history the marker was holding out for. */ +export function clearJournalRepairMarker(db: Database.Database, sessionId: string): void { + db.prepare(DELETE_REPAIR).run(sessionId) +} + +/** Drop the rejected suffix and record that it is owed, atomically. */ +export function deleteJournalRepairedSuffix(input: { + db: Database.Database + sessionId: string + epoch: string + /** First sequence of the rejected suffix. */ + fromSeq: number + /** First sequence left free once the suffix is gone. */ + contentFrom: number + now: number +}): number { + input.db.exec('BEGIN IMMEDIATE') + try { + const deleted = deleteJournalRowSuffix(input.db, input.sessionId, input.epoch, input.fromSeq) + input.db.prepare(UPSERT_REPAIR).run(input.sessionId, input.epoch, input.contentFrom, input.now) + input.db.exec('COMMIT') + return deleted + } catch (error) { + input.db.exec('ROLLBACK') + throw error + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-row-table.ts b/src/main/native-chat/agent-session-journal/journal-row-table.ts new file mode 100644 index 00000000000..306b3b300f2 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-row-table.ts @@ -0,0 +1,92 @@ +// Every statement the journal issues against `journal_rows` / `journal_sessions`. +// +// Each one is a prefix or range scan on the `(session_id, epoch, seq)` primary +// key; there is no secondary index, and no `max(seq)` tip query — replay folds +// the epoch to obtain `lastSequence`, so nothing needs the tip from SQL. +// Columns are always named: `SELECT *` is uncacheable and can drop a column. + +import type Database from '../../sqlite/sync-database' +import { serializeJournalRow, type JournalRow } from './journal-row-schema' + +export type JournalStoredRow = { epoch: string; seq: number; ts: number; rowJson: string } + +const SELECT_SESSION = 'SELECT epoch FROM journal_sessions WHERE session_id = ?' +const UPSERT_SESSION = `INSERT INTO journal_sessions (session_id, epoch, updated_at) +VALUES (?, ?, ?) +ON CONFLICT(session_id) DO UPDATE SET epoch = excluded.epoch, updated_at = excluded.updated_at` +const INSERT_ROW = + 'INSERT INTO journal_rows (session_id, epoch, seq, ts, row_json) VALUES (?, ?, ?, ?, ?)' +const SELECT_EPOCH_ROWS = `SELECT epoch, seq, ts, row_json FROM journal_rows +WHERE session_id = ? AND epoch = ? ORDER BY seq ASC` +const SELECT_ROWS_AFTER = `SELECT epoch, seq, ts, row_json FROM journal_rows +WHERE session_id = ? AND epoch = ? AND seq > ? ORDER BY seq ASC` +const DELETE_SUFFIX = 'DELETE FROM journal_rows WHERE session_id = ? AND epoch = ? AND seq >= ?' + +export function readJournalSessionEpoch(db: Database.Database, sessionId: string): string | null { + const row = db.prepare(SELECT_SESSION).get(sessionId) as { epoch?: string } | undefined + return row?.epoch ?? null +} + +export function upsertJournalSessionRow( + db: Database.Database, + sessionId: string, + epoch: string, + updatedAt: number +): void { + db.prepare(UPSERT_SESSION).run(sessionId, epoch, updatedAt) +} + +export function insertJournalRow( + db: Database.Database, + sessionId: string, + row: JournalRow +): number { + const rowJson = serializeJournalRow(row) + db.prepare(INSERT_ROW).run(sessionId, row.epoch, row.seq, row.ts, rowJson) + return Buffer.byteLength(rowJson, 'utf8') +} + +export function readJournalEpochRows( + db: Database.Database, + sessionId: string, + epoch: string +): JournalStoredRow[] { + return toStoredRows(db.prepare(SELECT_EPOCH_ROWS).all(sessionId, epoch)) +} + +export function readJournalRowsAfter( + db: Database.Database, + sessionId: string, + epoch: string, + afterSeq: number +): JournalStoredRow[] { + return toStoredRows(db.prepare(SELECT_ROWS_AFTER).all(sessionId, epoch, afterSeq)) +} + +/** + * Unqualified on purpose. One database per session means every row here belongs + * to this session, and the unqualified form takes SQLite's truncate + * optimization: measured at 0.26% of the database in WAL bytes where the + * `WHERE session_id = ?` form rewrote every emptied leaf at up to 99%. + */ +export function deleteAllJournalRows(db: Database.Database): void { + db.exec('DELETE FROM journal_rows') +} + +/** Drop the rejected suffix a repair found, from `fromSeq` to the tip. */ +export function deleteJournalRowSuffix( + db: Database.Database, + sessionId: string, + epoch: string, + fromSeq: number +): number { + const deleted = db.prepare(DELETE_SUFFIX).run(sessionId, epoch, fromSeq) + return Number(deleted.changes ?? 0) +} + +function toStoredRows(rows: readonly unknown[]): JournalStoredRow[] { + return rows.map((entry) => { + const record = entry as { epoch: string; seq: number; ts: number; row_json: string } + return { epoch: record.epoch, seq: record.seq, ts: record.ts, rowJson: record.row_json } + }) +} diff --git a/src/main/native-chat/agent-session-journal/journal-row-writer-read-only-latch.test.ts b/src/main/native-chat/agent-session-journal/journal-row-writer-read-only-latch.test.ts deleted file mode 100644 index 2b318bc5320..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-row-writer-read-only-latch.test.ts +++ /dev/null @@ -1,362 +0,0 @@ -import { mkdtemp, rm, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' -import type { AgentJournalItemBody } from '../../../shared/agent-session-journal-types' -import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' -import { readJournalBlob } from './journal-blob-store' -import { appendJournalRows } from './journal-log-file' -import { JournalLifecycleAdmission } from './journal-lifecycle-admission' -import { JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES } from './journal-lifecycle-capacity' -import { loadJournal } from './journal-open' -import { boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' -import { JournalRowWriter } from './journal-row-writer' -import { JournalAppendBudget } from './journal-write-guards' - -const SESSION_ID = 'session-1' - -function row(seq: number, ts: number): JournalRow { - return { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: 'epoch-1', - seq, - fence: 0, - ts, - kind: 'item', - itemId: 'item-1', - revision: 1, - body: { kind: 'status', text: 'ambiguous append' } - } -} - -function rowWithBlob(seq: number, ts: number, output: ReturnType): JournalRow { - return { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: 'epoch-1', - seq, - fence: 0, - ts, - kind: 'item', - itemId: 'item-with-blob', - revision: 1, - body: { - kind: 'tool-call', - name: 'shell', - input: {}, - state: 'completed', - output - } - } -} - -function runningToolRow(seq: number, ts: number, itemId = 'running-tool'): JournalRow { - return { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, - epoch: 'epoch-1', - seq, - fence: 0, - ts, - kind: 'item', - itemId, - revision: 1, - body: runningToolBody() - } -} - -function runningToolBody(): AgentJournalItemBody { - return { kind: 'tool-call', name: 'shell', input: {}, state: 'running' } -} - -describe('journal row writer read-only latch', () => { - let root: string - let readOnly = false - - beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-row-writer-')) - readOnly = false - }) - - afterEach(async () => { - await rm(root, { recursive: true, force: true }) - }) - - it('enforces the lifecycle append rate and allows a retry after the window', () => { - const appendWindowMs = 100 - const budget = new JournalAppendBudget(SESSION_ID, { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxAppendsPerWindow: 1, - appendWindowMs - }) - - budget.assertLifecycle(row(1, 1), 0) - expect(() => budget.assertLifecycle(row(2, 1), 0)).toThrow( - expect.objectContaining({ code: 'journal_rate_exceeded' }) - ) - expect(() => budget.assertLifecycle(row(2, appendWindowMs + 1), 0)).not.toThrow() - }) - - it('refuses lifecycle reservations once aggregate append capacity is saturated', () => { - const admission = new JournalLifecycleAdmission(SESSION_ID, 1_000_000, (itemId) => itemId, 2) - expect(admission.reserve({ id: 'first', bytes: 1, appendSlots: 1 }, 0)).toBe(true) - expect(admission.reserve({ id: 'second', bytes: 1, appendSlots: 1 }, 0)).toBe(true) - expect(admission.reserve({ id: 'third', bytes: 1, appendSlots: 1 }, 0)).toBe(false) - }) - - function writerHarness( - overrides: { - limits?: typeof DEFAULT_JOURNAL_PAYLOAD_LIMITS - physicalBytes?: number - appendRows?: (journalDir: string, rows: readonly JournalRow[]) => Promise - commit?: (row: JournalRow, physicalBytes: number) => void - } = {} - ) { - const limits = overrides.limits ?? DEFAULT_JOURNAL_PAYLOAD_LIMITS - const lifecycleAdmission = new JournalLifecycleAdmission( - SESSION_ID, - limits.maxSessionBytes, - (itemId) => itemId - ) - let physicalBytes = overrides.physicalBytes ?? 0 - let nextSequence = 1 - const committedRows: JournalRow[] = [] - const writer = new JournalRowWriter({ - journalDir: root, - sessionId: SESSION_ID, - budget: new JournalAppendBudget(SESSION_ID, limits), - lifecycleAdmission, - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 }, - now: () => 1, - serialize: (run) => run(), - readOnly: () => readOnly, - setReadOnly: (value) => { - readOnly = value - }, - physicalBytes: () => physicalBytes, - highestFence: () => 0, - nextSequence: () => nextSequence, - tailRows: () => committedRows, - referencedBlobDigests: () => new Set(), - compact: async () => undefined, - commit: (row, nextPhysicalBytes) => { - overrides.commit?.(row, nextPhysicalBytes) - committedRows.push(row) - physicalBytes = nextPhysicalBytes - nextSequence = row.seq + 1 - }, - ...(overrides.appendRows ? { appendRows: overrides.appendRows } : {}) - }) - return { writer, lifecycleAdmission, committedRows } - } - - it('latches read-only when a post-append failure makes durability ambiguous', async () => { - let committed = false - const writer = new JournalRowWriter({ - journalDir: root, - sessionId: 'session-1', - budget: new JournalAppendBudget('session-1', DEFAULT_JOURNAL_PAYLOAD_LIMITS), - lifecycleAdmission: new JournalLifecycleAdmission( - 'session-1', - DEFAULT_JOURNAL_PAYLOAD_LIMITS.maxSessionBytes, - (itemId) => itemId - ), - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 }, - now: () => 1, - serialize: (run) => run(), - readOnly: () => readOnly, - setReadOnly: (value) => { - readOnly = value - }, - physicalBytes: () => 0, - highestFence: () => 0, - nextSequence: () => 1, - tailRows: () => [], - referencedBlobDigests: () => new Set(), - compact: async () => undefined, - commit: () => { - committed = true - }, - appendRows: async (journalDir, rows) => { - await appendJournalRows(journalDir, rows) - throw new Error('fsync failed after append') - } - }) - - await expect(writer.enqueue(row)).rejects.toThrow('fsync failed after append') - - expect(readOnly).toBe(true) - expect(committed).toBe(false) - await expect(writer.enqueue(row)).rejects.toMatchObject({ code: 'journal_read_only' }) - }) - - it('keeps blobs for a durable row when a post-append crash is reported', async () => { - const payload = 'durable blob payload'.repeat(2_000) - const bounded = boundPayload(payload, { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 32 - }) - const writer = new JournalRowWriter({ - journalDir: root, - sessionId: 'session-1', - budget: new JournalAppendBudget('session-1', DEFAULT_JOURNAL_PAYLOAD_LIMITS), - lifecycleAdmission: new JournalLifecycleAdmission( - 'session-1', - DEFAULT_JOURNAL_PAYLOAD_LIMITS.maxSessionBytes, - (itemId) => itemId - ), - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 }, - now: () => 1, - serialize: (run) => run(), - readOnly: () => readOnly, - setReadOnly: (value) => { - readOnly = value - }, - physicalBytes: () => 0, - highestFence: () => 0, - nextSequence: () => 1, - tailRows: () => [], - referencedBlobDigests: () => new Set(), - compact: async () => undefined, - commit: () => undefined, - appendRows: async (journalDir, rows) => { - await appendJournalRows(journalDir, rows) - throw new Error('crash after row append') - } - }) - - await expect( - writer.enqueue( - (seq, ts) => rowWithBlob(seq, ts, bounded), - [{ digest: bounded.digest, payload }] - ) - ).rejects.toThrow('crash after row append') - - expect(readOnly).toBe(true) - await expect(writer.enqueue(row)).rejects.toMatchObject({ code: 'journal_read_only' }) - expect(await readJournalBlob(root, bounded.digest)).toBe(payload) - const reopened = await loadJournal(root, 'session-1') - const item = reopened?.state.items.get('item-with-blob') - expect(item?.body).toMatchObject({ - kind: 'tool-call', - output: { digest: bounded.digest, truncated: true } - }) - }) - - it('does not leak a lifecycle reservation after budget refusal', async () => { - const probe = runningToolRow(1, 1) - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxSessionBytes: JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES + journalRowByteLength(probe) - 1 - } - const { writer, lifecycleAdmission, committedRows } = writerHarness({ limits }) - - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).rejects.toMatchObject({ - code: 'journal_bound_exceeded' - }) - - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - await expect(writer.enqueue(row)).resolves.toMatchObject({ kind: 'item', itemId: 'item-1' }) - expect( - committedRows.map((entry) => (entry.kind === 'item' ? entry.itemId : 'non-item')) - ).toEqual(['item-1']) - }) - - it('preflights existing durable-write temps before creating a blob or row', async () => { - const tempBytes = 512 - const tempPath = join(root, 'log.jsonl.existing-write.tmp') - await writeFile(tempPath, 't'.repeat(tempBytes), 'utf8') - const probe = row(1, 1) - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxSessionBytes: tempBytes + journalRowByteLength(probe) - 1 - } - const { writer, committedRows } = writerHarness({ limits }) - - await expect(writer.enqueue((seq, ts) => row(seq, ts))).rejects.toMatchObject({ - code: 'journal_bound_exceeded' - }) - expect(committedRows).toHaveLength(0) - expect(await readJournalBlob(root, 'a'.repeat(64))).toBeNull() - }) - - it('does not leak a lifecycle reservation after blob lookup failure', async () => { - const { writer, lifecycleAdmission } = writerHarness() - const digest = 'a'.repeat(64) - await writeFile(join(root, 'blobs'), 'not a directory', 'utf8') - - await expect( - writer.enqueue((seq, ts) => runningToolRow(seq, ts), [{ digest, payload: 'payload' }]) - ).rejects.toThrow() - - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - await rm(join(root, 'blobs'), { force: true }) - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).resolves.toMatchObject({ - kind: 'item', - itemId: 'running-tool' - }) - expect(lifecycleAdmission.state).toEqual({ - reservedBytes: JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES, - reservedAppendSlots: 1 - }) - }) - - it('rolls back ordinary append-rate reservation after blob preflight failure', async () => { - const limits = { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - maxAppendsPerWindow: 1, - appendWindowMs: 100 - } - const { writer, committedRows } = writerHarness({ limits }) - const payload = 'retryable blob payload'.repeat(100) - const bounded = boundPayload(payload, limits) - await writeFile(join(root, 'blobs'), 'not a directory', 'utf8') - - await expect( - writer.enqueue( - (seq, ts) => rowWithBlob(seq, ts, bounded), - [{ digest: bounded.digest, payload }] - ) - ).rejects.toThrow() - - await rm(join(root, 'blobs'), { force: true }) - await expect( - writer.enqueue( - (seq, ts) => rowWithBlob(seq, ts, bounded), - [{ digest: bounded.digest, payload }] - ) - ).resolves.toMatchObject({ kind: 'item', itemId: 'item-with-blob' }) - expect(committedRows).toHaveLength(1) - }) - - it('does not leak a lifecycle reservation after durable append failure', async () => { - const { writer, lifecycleAdmission } = writerHarness({ - appendRows: async () => { - throw new Error('append failed before a durable row existed') - } - }) - - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).rejects.toThrow( - 'append failed before a durable row existed' - ) - - expect(readOnly).toBe(true) - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('does not leak a lifecycle reservation after reducer commit failure', async () => { - const { writer, lifecycleAdmission } = writerHarness({ - commit: () => { - throw new Error('commit failed after durable append') - } - }) - - await expect(writer.enqueue((seq, ts) => runningToolRow(seq, ts))).rejects.toThrow( - 'commit failed after durable append' - ) - - expect(lifecycleAdmission.state).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) -}) diff --git a/src/main/native-chat/agent-session-journal/journal-row-writer.test.ts b/src/main/native-chat/agent-session-journal/journal-row-writer.test.ts new file mode 100644 index 00000000000..dc08739a8bc --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-row-writer.test.ts @@ -0,0 +1,94 @@ +// The write path's transaction. +// +// A transaction either commits or does not, so the old "a post-append failure +// makes durability ambiguous" latch has nothing left to latch on: the case that +// used to assert the latch asserts the rollback instead. + +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' +import { openJournalDatabase, type OpenJournalDatabase } from './journal-database' +import { journalDatabaseFile } from './journal-paths' +import { + insertJournalRow, + readJournalEpochRows, + upsertJournalSessionRow +} from './journal-row-table' +import type { JournalRow } from './journal-row-schema' +import { JournalRowWriter } from './journal-row-writer' + +const SESSION_ID = 'session-1' +const EPOCH = 'epoch-1' + +function row(seq: number, ts: number): JournalRow { + return { + v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + epoch: EPOCH, + seq, + fence: 0, + ts, + kind: 'item', + itemId: 'item-1', + revision: 1, + body: { kind: 'status', text: 'plain append' } + } +} + +describe('journal row writer', () => { + let root: string + let database: OpenJournalDatabase + let readOnly = false + + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-row-writer-')) + database = openJournalDatabase(journalDatabaseFile(root)) + upsertJournalSessionRow(database.db, SESSION_ID, EPOCH, 1) + readOnly = false + }) + + afterEach(async () => { + try { + database.db.close() + } catch { + // Already closed by the case. + } + await rm(root, { recursive: true, force: true }) + }) + + function writerHarness() { + const committedRows: JournalRow[] = [] + let sequence = 1 + const writer = new JournalRowWriter({ + sessionId: SESSION_ID, + now: () => 1, + serialize: (run) => run(), + database: () => database, + readOnly: () => readOnly, + highestFence: () => 0, + nextSequence: () => sequence, + commit: (committed) => { + committedRows.push(committed) + sequence = committed.seq + 1 + } + }) + return { writer, committedRows } + } + + it('rolls the transaction back and sets no latch when the insert fails', async () => { + const { writer, committedRows } = writerHarness() + // A row already occupies sequence 1, so the insert violates the primary key. + insertJournalRow(database.db, SESSION_ID, row(1, 1)) + + await expect(writer.enqueue(row)).rejects.toThrow() + + expect(readOnly).toBe(false) + expect(committedRows).toHaveLength(0) + expect(readJournalEpochRows(database.db, SESSION_ID, EPOCH)).toHaveLength(1) + // Still writable: there is no ambiguity for a latch to protect against. + await expect(writer.enqueue((seq, ts) => row(seq + 1, ts))).resolves.toMatchObject({ + kind: 'item' + }) + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-row-writer.ts b/src/main/native-chat/agent-session-journal/journal-row-writer.ts index 980275d4c72..85ff7da7a3f 100644 --- a/src/main/native-chat/agent-session-journal/journal-row-writer.ts +++ b/src/main/native-chat/agent-session-journal/journal-row-writer.ts @@ -1,188 +1,42 @@ -import { - budgetPressurePolicy, - journalTailCanShedRows, - journalTailIsReadyToCompact, - type JournalCompactionPolicy -} from './journal-compaction' -import { journalBlobFileSize, putJournalBlob, removeJournalBlob } from './journal-blob-store' -import { appendJournalRows } from './journal-log-file' -import { blobDigestsInBody } from './journal-reducer' -import { journalDirectoryBytes } from './journal-physical-quota' -import type { JournalLifecycleAdmission } from './journal-lifecycle-admission' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' -import { - AgentSessionJournalError, - assertJournalFence, - assertJournalWritable, - type JournalAppendBudget -} from './journal-write-guards' - -type JournalBlob = { digest: string; payload: string } +import type Database from '../../sqlite/sync-database' +import { insertJournalRow, upsertJournalSessionRow } from './journal-row-table' +import type { JournalRow } from './journal-row-schema' +import { assertJournalFence, assertJournalWritable } from './journal-write-guards' export type JournalRowWriterDeps = { - journalDir: string sessionId: string - budget: JournalAppendBudget - lifecycleAdmission: JournalLifecycleAdmission - autoCompact: boolean - compaction: JournalCompactionPolicy now: () => number serialize: (run: () => Promise) => Promise + database: () => { db: Database.Database } readOnly: () => boolean - setReadOnly: (readOnly: boolean) => void - physicalBytes: () => number highestFence: () => number nextSequence: () => number - tailRows: () => readonly JournalRow[] - referencedBlobDigests: () => ReadonlySet - compact: (now: number, policy: JournalCompactionPolicy) => Promise - commit: (row: JournalRow, physicalBytes: number) => void - appendRows?: (journalDir: string, rows: readonly JournalRow[]) => Promise + commit: (row: JournalRow) => void } export class JournalRowWriter { constructor(private readonly deps: JournalRowWriterDeps) {} - enqueue( - build: (seq: number, ts: number) => JournalRow, - blobs: readonly JournalBlob[] = [] - ): Promise { + enqueue(build: (seq: number, ts: number) => JournalRow): Promise { return this.deps.serialize(async () => { assertJournalWritable(this.deps.readOnly(), this.deps.sessionId) - const ts = this.deps.now() - const row = build(this.deps.nextSequence(), ts) + const row = build(this.deps.nextSequence(), this.deps.now()) assertJournalFence(row.fence, this.deps.highestFence()) - // The in-memory counter is an optimization, not the quota source of - // truth: a prior crash may have left a durable-write temp beside the - // finals, and a concurrent/retried opener may have materialized files - // after the last commit callback. Recount before any speculative write - // so the peak check includes those bytes. - let physicalBytes = Math.max( - this.deps.physicalBytes(), - await journalDirectoryBytes(this.deps.journalDir) - ) - const admission = this.deps.lifecycleAdmission.prepare(row, physicalBytes) - const newBlobs = await uniqueNewBlobs(this.deps.journalDir, blobs) - const blobBytes = newBlobs.reduce( - (total, blob) => total + Buffer.byteLength(blob.payload, 'utf8'), - 0 - ) - const budgetCompaction = budgetPressurePolicy(this.deps.compaction) - let effectiveSize = physicalBytes + blobBytes + admission.protectedBytes - if ( - this.deps.autoCompact && - this.deps.budget.wouldExceedSize(row, effectiveSize) && - journalTailCanShedRows(this.deps.tailRows(), budgetCompaction, ts) - ) { - await this.deps.compact(ts, budgetCompaction) - physicalBytes = this.deps.physicalBytes() - effectiveSize = physicalBytes + blobBytes + admission.protectedBytes - } - const lifecycleRateCheckpoint = admission.lifecycleCovered - ? this.deps.budget.checkpoint() - : null - const appendRateCheckpoint = this.deps.budget.checkpoint() - let committed = false - let appendMayHaveLanded = false + const { db } = this.deps.database() + db.exec('BEGIN IMMEDIATE') try { - if (admission.lifecycleCovered) { - this.deps.budget.assertReservedLifecycle(row, effectiveSize) - } else { - this.deps.budget.assert(row, ts, effectiveSize) - } - const appendedBytes = blobBytes + journalRowByteLength(row) - if ( - physicalBytes + appendedBytes > - this.deps.budget.maxSessionBytes - admission.protectedBytes - ) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.deps.sessionId} reached its ${this.deps.budget.maxSessionBytes}-byte physical bound` - ) - } - await this.commitFiles(row, newBlobs, () => { - appendMayHaveLanded = true - }) - physicalBytes += appendedBytes - this.deps.commit(row, physicalBytes) - this.deps.lifecycleAdmission.commit(admission) - committed = true + insertJournalRow(db, this.deps.sessionId, row) + upsertJournalSessionRow(db, this.deps.sessionId, row.epoch, row.ts) + db.exec('COMMIT') } catch (error) { - if (!committed && lifecycleRateCheckpoint) { - this.deps.budget.restore(lifecycleRateCheckpoint) - } - if (!committed && !appendMayHaveLanded) { - this.deps.budget.restore(appendRateCheckpoint) - } + db.exec('ROLLBACK') throw error } - if ( - this.deps.autoCompact && - journalTailIsReadyToCompact(this.deps.tailRows(), this.deps.compaction, ts) - ) { - await this.deps.compact(ts, this.deps.compaction) - } + // COMMIT landed, so the row is durable: adopt it before anything that can + // fail. Rejecting here instead would leave the next append reusing a + // sequence the table already holds. + this.deps.commit(row) return row }) } - - private async commitFiles( - row: JournalRow, - blobs: readonly JournalBlob[], - markAppendLanded: () => void - ): Promise { - const persisted: string[] = [] - let appendMayHaveLanded = false - try { - for (const blob of blobs) { - await putJournalBlob(this.deps.journalDir, blob.digest, blob.payload) - persisted.push(blob.digest) - } - appendMayHaveLanded = true - markAppendLanded() - await (this.deps.appendRows ?? appendJournalRows)(this.deps.journalDir, [row]) - } catch (error) { - if (appendMayHaveLanded) { - this.deps.setReadOnly(true) - throw error - } - const retained = this.referencedBlobDigestsIncludingTail() - for (const digest of persisted) { - if (!retained.has(digest)) { - await removeJournalBlob(this.deps.journalDir, digest) - } - } - throw error - } - } - - private referencedBlobDigestsIncludingTail(): Set { - const retained = new Set(this.deps.referencedBlobDigests()) - for (const row of this.deps.tailRows()) { - if (row.kind === 'item') { - blobDigestsInBody(row.body, retained) - } else if (row.kind === 'lifecycle-batch') { - for (const mutation of row.mutations) { - if (mutation.kind === 'item') { - blobDigestsInBody(mutation.body, retained) - } - } - } - } - return retained - } -} - -async function uniqueNewBlobs( - journalDir: string, - blobs: readonly JournalBlob[] -): Promise { - const unique = new Map(blobs.map((blob) => [blob.digest, blob])) - const result: JournalBlob[] = [] - for (const blob of unique.values()) { - if ((await journalBlobFileSize(journalDir, blob.digest)) === null) { - result.push(blob) - } - } - return result } diff --git a/src/main/native-chat/agent-session-journal/journal-store-close.test.ts b/src/main/native-chat/agent-session-journal/journal-store-close.test.ts new file mode 100644 index 00000000000..960c65a638a --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-close.test.ts @@ -0,0 +1,259 @@ +// `close()`: enqueue-time admission, and the fulfilled/rejected split. +// +// The two failures this file exists to prevent: an append enqueued in the same +// turn as `close()` being rejected AFTER the queue accepted it, and a rejected +// close leaving a connection live but permanently unreachable through the API. + +import { access, mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import { journalDatabaseFile } from './journal-paths' +import type { AgentSessionJournal } from './journal-store' +import { createTrackedJournalOpener } from './journal-store-test-open' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } +} + +let root: string +const journals = createTrackedJournalOpener() + +function item(ordinal: number): AgentJournalItemIdentity { + return { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal } +} + +function body(value: string): AgentJournalItemBody { + return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } +} + +function openJournal(): Promise { + return journals.open({ identity: IDENTITY, journalDir: root }) +} + +/** Replaces the store's own release step, which is the only step that can + * reject the attempt in production. */ +function injectReleaseFailure(journal: AgentSessionJournal): { + calls: () => number + stopFailing: () => void +} { + const internals = journal as unknown as { database: { db: { close: () => void } } } + const release = internals.database.db.close.bind(internals.database.db) + let calls = 0 + let failing = true + internals.database.db.close = () => { + calls += 1 + if (failing) { + throw new Error('injected release failure') + } + release() + } + return { calls: () => calls, stopFailing: () => (failing = false) } +} + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-close-')) +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('closed-state admission happens at enqueue', () => { + it('completes a write enqueued in the same turn as the close', async () => { + const journal = await openJournal() + const append = journal.appendItem(item(1), body('before'), { fence: 1 }) + const closed = journal.close() + + await expect(append).resolves.toBeDefined() + await expect(closed).resolves.toBeUndefined() + const reopened = await openJournal() + expect(reopened.snapshot().items).toHaveLength(1) + }) + + it('refuses a write offered while the close is still in flight, without queueing it', async () => { + const journal = await openJournal() + const closing = journal.close() + const refused = journal.appendItem(item(1), body('during'), { fence: 1 }) + + // The rejection is available before the close step has run: it never joined + // the queue, so nothing is ever chained behind a close. + await expect(refused).rejects.toMatchObject({ code: 'journal_closed' }) + await expect(closing).resolves.toBeUndefined() + }) + + it('refuses every write entry point after the close has settled', async () => { + const journal = await openJournal() + await journal.close() + const settle = (attempt: Promise): Promise => + attempt.then( + () => new Error('resolved instead of refusing'), + (error: unknown) => error + ) + const refusals = [ + settle(journal.appendItem(item(1), body('after'), { fence: 1 })), + settle(journal.appendTombstone(item(3), { fence: 1 })), + settle( + journal.appendSubmission({ + clientMessageId: 'cm_1', + payloadFingerprint: 'f', + body: { kind: 'message', role: 'user', blocks: [] }, + fence: 1 + }) + ), + settle(journal.resolveDispatch({ clientMessageId: 'cm_1', state: 'rejected', fence: 1 })), + settle( + journal.appendLifecycleBatch({ + settlementId: 'settle', + fence: 1, + mutations: [{ kind: 'item', identity: item(4), body: body('x') }] + }) + ), + settle(journal.rollEpoch('handle_forked', 1)), + settle(journal.replaceEpochItems('handle_forked', 1, [])) + ] + for (const refusal of refusals) { + expect(await refusal).toMatchObject({ code: 'journal_closed' }) + } + }) + + // The property part 1 rests on: every entry point reaches the gate in the + // caller's own turn, so a refusal never advances the queue. + it('rejects without waiting for the queue to advance', async () => { + const journal = await openJournal() + const inFlight = journal.appendItem(item(1), body('admitted'), { fence: 1 }) + const closing = journal.close() + + // Settles while the admitted append is still running: it reached the gate in + // the caller's own turn and never joined the queue behind it. + await expect(journal.appendItem(item(2), body('later'), { fence: 1 })).rejects.toMatchObject({ + code: 'journal_closed' + }) + await expect(inFlight).resolves.toBeDefined() + await expect(closing).resolves.toBeUndefined() + }) + + it('resolves a second close as a no-op without running the routine again', async () => { + const journal = await openJournal() + const injected = injectReleaseFailure(journal) + injected.stopFailing() + await journal.close() + expect(injected.calls()).toBe(1) + + await expect(journal.close()).resolves.toBeUndefined() + expect(injected.calls()).toBe(1) + }) +}) + +describe('a rejected close is a real retry', () => { + it('retries the release, releases the handle, and then goes terminal', async () => { + const journal = await openJournal() + await journal.appendItem(item(1), body('durable'), { fence: 1 }) + const injected = injectReleaseFailure(journal) + + await expect(journal.close()).rejects.toThrow('injected release failure') + expect(injected.calls()).toBe(1) + + // Write-closed anyway: retry exists to release the OS handle, never to + // resurrect the store. + await expect(journal.appendItem(item(2), body('after'), { fence: 1 })).rejects.toMatchObject({ + code: 'journal_closed' + }) + + injected.stopFailing() + await expect(journal.close()).resolves.toBeUndefined() + // The retry RE-ENTERED the release. A completion flag on it would skip the + // one step that had not succeeded, and the handle would never be released. + expect(injected.calls()).toBe(2) + const dbPath = journalDatabaseFile(root) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + + // Fulfilment is terminal: a third call must not issue a second db.close(), + // which `node:sqlite` answers with ERR_INVALID_STATE. + await expect(journal.close()).resolves.toBeUndefined() + expect(injected.calls()).toBe(2) + }) + + it('hands concurrent callers the same outcome', async () => { + const journal = await openJournal() + const injected = injectReleaseFailure(journal) + const first = journal.close() + const second = journal.close() + await expect(first).rejects.toThrow('injected release failure') + await expect(second).rejects.toThrow('injected release failure') + expect(injected.calls()).toBe(1) + }) +}) + +// The two rejected readings of the contract, as models, because a green run on +// the real store proves nothing about what the alternatives would have done. +describe('negative controls', () => { + class UnconditionalNoOpClose { + calls = 0 + private called = false + constructor(private readonly release: () => void) {} + async close(): Promise { + if (this.called) { + return + } + this.called = true + this.calls += 1 + this.release() + } + } + + class AlwaysReentrantClose { + calls = 0 + async close(release: () => void): Promise { + this.calls += 1 + release() + } + } + + it('an unconditional second-call no-op leaves a failed close unreleasable', async () => { + let failing = true + let released = false + const model = new UnconditionalNoOpClose(() => { + if (failing) { + throw new Error('injected release failure') + } + released = true + }) + await expect(model.close()).rejects.toThrow('injected release failure') + failing = false + await expect(model.close()).resolves.toBeUndefined() + // Fulfilled without the routine running: the handle is still held. + expect(model.calls).toBe(1) + expect(released).toBe(false) + }) + + it('an always-reentrant close issues the second db.close() that throws', async () => { + let open = true + const release = (): void => { + if (!open) { + throw new Error('ERR_INVALID_STATE: database is not open') + } + open = false + } + const model = new AlwaysReentrantClose() + await model.close(release) + await expect(model.close(release)).rejects.toThrow('database is not open') + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-store-close.ts b/src/main/native-chat/agent-session-journal/journal-store-close.ts new file mode 100644 index 00000000000..f1b56d6a0db --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-close.ts @@ -0,0 +1,106 @@ +// Releasing the one SQLite connection a store holds for its lifetime. +// +// The contract, in five parts: +// +// 1. Closed-state admission happens at ENQUEUE, once, and is permanent. The +// flag lives on the store's write gate; it is never cleared, not by a +// rejection and not by a retry. A store that failed to close is still a +// store nobody may write to. +// 2. The close step shares the write queue and BYPASSES that gate — a close +// that consulted the flag it had just set would refuse itself. Same queue +// orders the release behind admitted work; separate gate lets it run. +// 3. One in-flight attempt, shared: concurrent callers get the same outcome. +// 4. Fulfilled is TERMINAL; rejected is not. A later call after fulfilment is a +// genuine no-op — `DatabaseSync.close()` throws `ERR_INVALID_STATE` on a +// second call, so re-entering after success is the bug, not the fix. +// 5. The release is deliberately unguarded. There is no way to ask whether a +// `db.close()` that threw released the handle first, and guarding the step +// would skip it on retry — guaranteeing a permanent leak in exactly the case +// where it did not release. + +import { AgentSessionJournalError } from './journal-write-guards' +import type Database from '../../sqlite/sync-database' + +/** + * The write queue and its closed gate. Admission is checked at ENQUEUE and is + * permanent; the close step reaches the same queue through `serializePastGate`, + * because a close that consulted the flag it had just set would refuse itself. + */ +export class JournalWriteQueue { + private writes: Promise = Promise.resolve() + private closed = false + + constructor(private readonly sessionId: string) {} + + markClosed(): void { + this.closed = true + } + + serialize(run: () => Promise): Promise { + if (this.closed) { + return Promise.reject( + new AgentSessionJournalError( + 'journal_closed', + `agent-session journal for ${this.sessionId} is closed` + ) + ) + } + return this.serializePastGate(run) + } + + serializePastGate(run: () => Promise): Promise { + const started = this.writes.then(run) + this.writes = started.catch(() => undefined) + return started + } +} + +export class JournalConnectionCloser { + private released = false + private inFlight: Promise | null = null + + constructor( + private readonly deps: { + connection: () => Database.Database | null + /** Chains onto the store's write queue past the closed gate. */ + enqueue: (run: () => Promise) => Promise + } + ) {} + + get isReleased(): boolean { + return this.released + } + + close(): Promise { + if (this.released) { + return Promise.resolve() + } + if (this.inFlight) { + return this.inFlight + } + const attempt = this.deps + .enqueue(() => this.release()) + .then( + () => { + this.released = true + this.inFlight = null + }, + (error: unknown) => { + this.inFlight = null + throw error + } + ) + this.inFlight = attempt + return attempt + } + + private async release(): Promise { + const db = this.deps.connection() + if (!db) { + return + } + // SQLite checkpoints and removes the WAL itself when the last connection to + // the database closes. + db.close() + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store-collaborators.ts b/src/main/native-chat/agent-session-journal/journal-store-collaborators.ts new file mode 100644 index 00000000000..a4ff455b31f --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-collaborators.ts @@ -0,0 +1,88 @@ +// Wiring for the store's collaborators. +// +// Split out of the store itself so the class stays a description of the public +// surface rather than sixty lines of constructor plumbing. + +import type { + AgentJournalCursor, + AgentSessionJournalIdentity +} from '../../../shared/agent-session-journal-types' +import type { OpenJournalDatabase } from './journal-database' +import { JournalEpochController } from './journal-epoch-controller' +import { JournalItemAppender } from './journal-item-appender' +import { JournalLifecycleBatchAppender } from './journal-lifecycle-batch-appender' +import type { JournalLoad } from './journal-open' +import type { JournalReducerState } from './journal-reducer' +import { JournalRowWriter } from './journal-row-writer' +import { restoreJournalStore } from './journal-store-restore' +import type { JournalRow } from './journal-row-schema' +import type { AgentSessionJournal } from './journal-store' + +export type JournalStoreHost = { + identity: AgentSessionJournalIdentity + journalDir: string + now: () => number + mintEpoch: () => string + serialize: (run: () => Promise) => Promise + database: () => OpenJournalDatabase + state: () => JournalReducerState + readOnly: () => boolean + setReadOnly: (readOnly: boolean) => void + cursor: () => AgentJournalCursor + adopt: (loaded: JournalLoad) => void + commit: (row: JournalRow) => void + /** A caller-supplied load, which suppresses replay entirely when present. */ + loaded: () => JournalLoad | null | undefined + malformedRows: () => number + setMalformedRows: (count: number) => void + journal: () => AgentSessionJournal + enqueue: (build: (seq: number, ts: number) => JournalRow) => Promise +} + +export type JournalStoreCollaborators = { + rowWriter: JournalRowWriter + epochController: JournalEpochController + itemAppender: JournalItemAppender + lifecycleBatchAppender: JournalLifecycleBatchAppender + /** Restores the store's state from disk. Owned here because it needs the same + * collaborators the constructor just built. */ + restore: () => Promise +} + +export function createJournalStoreCollaborators(host: JournalStoreHost): JournalStoreCollaborators { + const epochController = new JournalEpochController({ + identity: host.identity, + now: host.now, + mintEpoch: host.mintEpoch, + serialize: host.serialize, + database: host.database, + readOnly: host.readOnly, + setReadOnly: host.setReadOnly, + highestFence: () => host.state().highestFence, + cursor: host.cursor, + adopt: host.adopt + }) + return { + epochController, + restore: () => restoreJournalStore(host, { epochController }), + rowWriter: new JournalRowWriter({ + sessionId: host.identity.sessionId, + now: host.now, + serialize: host.serialize, + database: host.database, + readOnly: host.readOnly, + highestFence: () => host.state().highestFence, + nextSequence: () => host.state().lastSequence + 1, + commit: host.commit + }), + itemAppender: new JournalItemAppender({ + state: host.state, + enqueue: host.enqueue + }), + lifecycleBatchAppender: new JournalLifecycleBatchAppender({ + state: host.state, + cursor: host.cursor, + enqueue: host.enqueue + }) + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store-contracts.ts b/src/main/native-chat/agent-session-journal/journal-store-contracts.ts index 45a723a0f23..22e3a4c7cca 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-contracts.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-contracts.ts @@ -6,19 +6,13 @@ import type { AgentJournalResetReason, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import type { JournalCompactionPolicy } from './journal-compaction' import type { JournalLoad } from './journal-open' -import type { JournalPayloadLimits } from './journal-payload-bounds' import type { JournalLifecycleMutationInput } from './journal-row-builders' import type { JournalRow } from './journal-row-schema' export type AgentSessionJournalOptions = { identity: AgentSessionJournalIdentity journalDir: string - limits?: JournalPayloadLimits - compaction?: JournalCompactionPolicy - /** Compact as the tail grows. Defaults on: without it the log never sheds. */ - autoCompact?: boolean now?: () => number mintEpoch?: () => string /** A caller that already loaded the journal can avoid reading the same files again. */ @@ -45,7 +39,6 @@ export type JournalAppendResult = { } export type JournalItemAppendOptions = { fence: number; observedAt?: number; recovered?: true } -export type JournalBlobInput = { digest: string; payload: string } export type JournalTombstoneInput = { fence: number } export type JournalLifecycleBatchInput = { diff --git a/src/main/native-chat/agent-session-journal/journal-store-open.ts b/src/main/native-chat/agent-session-journal/journal-store-open.ts index 67d05c0b0c3..e002048a094 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-open.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-open.ts @@ -1,16 +1,14 @@ -import type { AgentJournalSnapshot } from '../../../shared/agent-session-journal-types' -import { malformedRowsDisclosure, quarantineCorruptSuffix } from './journal-corruption-quarantine' -import { ensureJournalDir } from './journal-log-file' -import { loadJournal, type JournalLoad } from './journal-open' -import { assertJournalPhysicalCapacity, journalDirectoryBytes } from './journal-physical-quota' -import type { JournalRow } from './journal-row-schema' +import { mkdir } from 'node:fs/promises' +import type { JournalLoad } from './journal-open' +import { journalRepairDisclosure, type JournalRepairDisclosure } from './journal-repair-disclosure' + +export async function ensureJournalDir(journalDir: string): Promise { + await mkdir(journalDir, { recursive: true }) +} export function journalStoreLoadedFields(loaded: JournalLoad) { return { state: loaded.state, - tailRows: loaded.tailRows, - compactedThrough: loaded.compactedThrough, - sizeBytes: loaded.sizeBytes, readOnly: loaded.readOnly, malformedRows: loaded.malformedRows } @@ -18,60 +16,47 @@ export function journalStoreLoadedFields(loaded: JournalLoad) { export async function openJournalStoreState(input: { journalDir: string - sessionId: string - maxBytes: number loaded: JournalLoad | null | undefined - start: () => Promise + replay: () => JournalLoad | null + /** Drops the rejected suffix and records the rebuild it owes, in ONE + * transaction. Corruption is not preserved; replay keeps reporting `corrupt` + * until provider history republishes the epoch or the session writes past + * `contentFrom`, the first sequence the repair left free. */ + deleteSuffix: (fromSeq: number, contentFrom: number) => number + start: () => void adopt: (loaded: JournalLoad) => void - tailRows: () => readonly JournalRow[] - snapshot: () => AgentJournalSnapshot - rebuildLifecycle: (snapshot: AgentJournalSnapshot, physicalBytes: number) => void + /** Republishes an anchor row for an epoch a repair emptied. */ + publishRepairEpoch: () => void appendDisclosure: ( - identity: ReturnType['identity'], - body: ReturnType['body'], + identity: JournalRepairDisclosure['identity'], + body: JournalRepairDisclosure['body'], fence: number ) => Promise highestFence: () => number malformedRows: () => number + setMalformedRows: (count: number) => void readOnly: () => boolean - setPhysicalBytes: (bytes: number) => void }): Promise { - await ensureJournalDir(input.journalDir) - input.setPhysicalBytes( - await assertJournalPhysicalCapacity({ - journalDir: input.journalDir, - sessionId: input.sessionId, - maxBytes: input.maxBytes - }) - ) - const loaded = - input.loaded !== undefined - ? input.loaded - : await loadJournal(input.journalDir, input.sessionId, { maxBytes: input.maxBytes }) + const loaded = input.loaded !== undefined ? input.loaded : input.replay() if (!loaded) { - await input.start() - input.setPhysicalBytes(await journalDirectoryBytes(input.journalDir)) + input.start() return } input.adopt(loaded) - if (loaded.corrupt && !loaded.readOnly) { - await quarantineCorruptSuffix(input.journalDir, input.tailRows(), loaded.quarantineRemainder, { - sessionId: input.sessionId, - maxBytes: input.maxBytes - }) + if (loaded.truncateFrom !== undefined && !loaded.readOnly) { + input.deleteSuffix(loaded.truncateFrom, loaded.state.lastSequence + 1) } - let physicalBytes = await journalDirectoryBytes(input.journalDir) - input.setPhysicalBytes(physicalBytes) - // A future-schema/read-only journal is inspection-only. Its reduced state is - // intentionally empty, and rebuilding reservations from it would mutate the - // in-memory quota model (and could influence later admission decisions). - if (!loaded.readOnly) { - input.rebuildLifecycle(input.snapshot(), physicalBytes) + // A repair that took every live row leaves the epoch with no anchor. Publish + // one before anything can append into it: an ordinary row at sequence 1 would + // replay as a clean timeline and hide that the history was never rebuilt. + if (!loaded.readOnly && loaded.state.lastSequence === 0) { + input.publishRepairEpoch() + // The replacement epoch adopts a clean load; what this open's repair did is + // still the answer `repair` and the disclosure below owe the caller. + input.setMalformedRows(loaded.malformedRows) } if (input.malformedRows() > 0 && !input.readOnly()) { - const disclosure = malformedRowsDisclosure(input.malformedRows()) + const disclosure = journalRepairDisclosure({ malformedRows: input.malformedRows() }) await input.appendDisclosure(disclosure.identity, disclosure.body, input.highestFence()) } - physicalBytes = await journalDirectoryBytes(input.journalDir) - input.setPhysicalBytes(physicalBytes) } diff --git a/src/main/native-chat/agent-session-journal/journal-store-restore.ts b/src/main/native-chat/agent-session-journal/journal-store-restore.ts new file mode 100644 index 00000000000..53683a797d1 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-restore.ts @@ -0,0 +1,49 @@ +// Bringing a store's in-memory state up from disk. +// +// Split out of the store for the same reason its collaborators were: this is the +// ORDERING between replay, suffix repair and disclosure, and none of it belongs +// to the store's public surface. Every step here reads or writes through the +// same host the collaborators use, so the store keeps the state and this owns +// the sequence. + +import type { JournalEpochController } from './journal-epoch-controller' +import { replayJournal } from './journal-open' +import type { JournalStoreHost } from './journal-store-collaborators' +import { openJournalStoreState } from './journal-store-open' +import { deleteJournalRepairedSuffix } from './journal-repair-marker' + +export function restoreJournalStore( + host: JournalStoreHost, + collaborators: { epochController: JournalEpochController } +): Promise { + return openJournalStoreState({ + journalDir: host.journalDir, + loaded: host.loaded(), + replay: () => { + const opened = host.database() + return replayJournal(opened.db, opened.readOnly, host.identity.sessionId) + }, + deleteSuffix: (fromSeq, contentFrom) => + deleteJournalRepairedSuffix({ + db: host.database().db, + sessionId: host.identity.sessionId, + epoch: host.state().epoch, + fromSeq, + contentFrom, + now: host.now() + }), + start: () => collaborators.epochController.start('session_created', 0), + // `unreconcilable_prefix` is the durable statement that this epoch exists + // because a repair emptied one: replay reads it back and keeps asking for + // provider history until the timeline is rebuilt or the session writes. + publishRepairEpoch: () => + collaborators.epochController.start('unreconcilable_prefix', host.state().highestFence), + adopt: host.adopt, + appendDisclosure: (identity, body, fence) => + host.journal().appendItem(identity, body, { fence }), + highestFence: () => host.state().highestFence, + malformedRows: host.malformedRows, + setMalformedRows: host.setMalformedRows, + readOnly: host.readOnly + }) +} diff --git a/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts b/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts index d78adeb30f5..df34ef73e73 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-schema.test.ts @@ -1,4 +1,11 @@ -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' +// Two independent version axes, both fail closed. +// +// `PRAGMA user_version` the DB SHAPE, known before the first read +// the row's `v` field the row BODY shape, met during replay +// +// A newer build can change either alone, so both are needed. + +import { mkdtemp, rm, stat } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -7,8 +14,12 @@ import type { AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { JOURNAL_LOG_FILE, JOURNAL_SNAPSHOT_FILE } from './journal-log-file' -import { openAgentSessionJournal } from './journal-store-factory' +import type Database from '../../sqlite/sync-database' +import { openJournalDatabase } from './journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from './journal-database-schema' +import { journalDatabaseFile } from './journal-paths' +import type { AgentSessionJournal } from './journal-store' +import { createTrackedJournalOpener } from './journal-store-test-open' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -20,6 +31,7 @@ const IDENTITY: AgentSessionJournalIdentity = { let root: string let clock = 1_000 +const journals = createTrackedJournalOpener() function tick(): number { clock += 1 @@ -34,87 +46,139 @@ function body(value: string): AgentJournalItemBody { return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } } -async function open(overrides: Partial[0]> = {}) { - return openAgentSessionJournal({ +function open(): Promise { + return journals.open({ identity: IDENTITY, journalDir: root, now: tick, - mintEpoch: () => `epoch-${clock}`, - ...overrides + mintEpoch: () => `epoch-${clock}` + }) +} + +async function withDatabase(run: (db: Database.Database) => void): Promise { + const opened = openJournalDatabase(journalDatabaseFile(root)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} + +/** Appends a raw `row_json` the way a newer build or a bad write would leave it. */ +async function appendRawRow(epoch: string, seq: number, rowJson: string): Promise { + await withDatabase((db) => { + db.prepare( + 'INSERT INTO journal_rows (session_id, epoch, seq, ts, row_json) VALUES (?, ?, ?, ?, ?)' + ).run(IDENTITY.sessionId, epoch, seq, 1, rowJson) }) } beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-journal-')) + root = await mkdtemp(join(tmpdir(), 'orca-journal-schema-')) clock = 1_000 }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) -describe('schema', () => { - it('quarantines an invalid compacted snapshot without replacing its tail', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - await journal.compact() - const epoch = journal.epoch - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const logPath = join(root, JOURNAL_LOG_FILE) - const invalidSnapshot = '{"folded history":' - await writeFile(snapshotPath, invalidSnapshot, 'utf-8') - const retainedTail = await readFile(logPath, 'utf-8') - expect(retainedTail).not.toContain('"kind":"epoch"') - - const reopened = await open() - expect(reopened.epoch).toBe(epoch) - expect(await readFile(logPath, 'utf-8')).toBe(retainedTail) - const quarantined = (await readdir(root)).find((name) => - name.startsWith('quarantine-snapshot-') - ) - expect(quarantined).toBeDefined() - expect(await readFile(join(root, quarantined!), 'utf-8')).toBe(invalidSnapshot) - }) - - it('degrades to read-only on a row from a newer build, without skipping or deleting it', async () => { +describe('axis 1: the database shape', () => { + it('latches read-only on a newer user_version and writes nothing', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - const future = JSON.stringify({ - v: 99, - kind: 'item', - epoch: journal.epoch, - seq: 99, - fence: 1, - ts: 1, - itemId: 'future', - revision: 1, - body: { kind: 'status', text: 'from a newer host' } - }) - const before = await readFile(logPath, 'utf-8') - await writeFile(logPath, `${before}${future}\n`, 'utf-8') + await journal.close() + await withDatabase((db) => db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`)) + const before = await stat(journalDatabaseFile(root)) const reopened = await open() expect(reopened.isReadOnly).toBe(true) await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ code: 'journal_read_only' }) - await expect(reopened.compact()).rejects.toMatchObject({ code: 'journal_read_only' }) - expect(reopened.readSince({ epoch: reopened.epoch, sequence: 0 })).toEqual({ - ok: false, - reset: 'schema_unreadable' + // The file this build must not touch is byte-identical afterwards. + await reopened.close() + expect((await stat(journalDatabaseFile(root))).size).toBe(before.size) + await withDatabase((db) => { + expect(db.pragma('user_version', { simple: true })).toBe(JOURNAL_DB_SCHEMA_VERSION + 1) }) - // The unreadable row is still on disk, and nothing was compacted past it. - expect(await readFile(logPath, 'utf-8')).toContain('"v":99') }) - it('skips a malformed line without giving up the journal, and discloses the skip', async () => { + it('refuses the schema escape hatch on a latched store', async () => { + const journal = await open() + await journal.close() + await withDatabase((db) => db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`)) + + const reopened = await open() + // With byte-copy quarantine gone there is nothing for `schema_unreadable` to + // do differently, so it takes the same writable guard as every other reason. + await expect(reopened.rollEpoch('schema_unreadable', 2)).rejects.toMatchObject({ + code: 'journal_read_only' + }) + expect(reopened.isReadOnly).toBe(true) + }) + + it('migrates an older user_version forward on reopen', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}{not json\n`, 'utf-8') + await journal.close() + await withDatabase((db) => db.pragma('user_version = 0')) + + const reopened = await open() + expect(reopened.isReadOnly).toBe(false) + expect(reopened.snapshot().items).toHaveLength(1) + await reopened.close() + await withDatabase((db) => { + expect(db.pragma('user_version', { simple: true })).toBe(JOURNAL_DB_SCHEMA_VERSION) + }) + }) +}) + +describe('axis 2: the row body shape', () => { + it('degrades to read-only on a row from a newer build, without skipping it', async () => { + const journal = await open() + await journal.appendItem(item(0), body('a'), { fence: 1 }) + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() + await appendRawRow( + epoch, + nextSeq, + JSON.stringify({ + v: 99, + kind: 'item', + epoch, + seq: nextSeq, + fence: 1, + ts: 1, + itemId: 'future', + revision: 1, + body: { kind: 'status', text: 'from a newer build' } + }) + ) + + const reopened = await open() + expect(reopened.isReadOnly).toBe(true) + await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ + code: 'journal_read_only' + }) + await reopened.close() + // Never skipped, never deleted: the row this build cannot read is still there. + await withDatabase((db) => { + const stored = db.prepare('SELECT row_json FROM journal_rows WHERE seq = ?').get(nextSeq) as { + row_json: string + } + expect(stored.row_json).toContain('"v":99') + }) + }) + + it('skips a malformed row without giving up the journal, and discloses the skip', async () => { + const journal = await open() + await journal.appendItem(item(0), body('a'), { fence: 1 }) + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() + await appendRawRow(epoch, nextSeq, '{not json') const reopened = await open() expect(reopened.isReadOnly).toBe(false) @@ -132,10 +196,12 @@ describe('schema', () => { it('keeps one disclosure row across reopens instead of stacking duplicates', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}{not json\n`, 'utf-8') + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() + await appendRawRow(epoch, nextSeq, '{not json') - await open() + await open().then((first) => first.close()) const reopened = await open() expect( reopened @@ -146,168 +212,32 @@ describe('schema', () => { ).toHaveLength(1) }) - it('repairs a torn tail before acknowledging the next append', async () => { + it('reopens a journal holding an admitted malformed-percent item id without throwing', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - const intact = await readFile(logPath, 'utf-8') - await writeFile(logPath, intact.slice(0, -1), 'utf-8') - - await journal.appendItem(item(1), body('b'), { fence: 1 }) - const reopened = await open() - expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([body('a'), body('b')]) - }) - - // Transcripts are full of emoji and CJK, so the repair's file offsets must be - // bytes: string indices would truncate mid-character and corrupt the prefix. - it('repairs a torn tail whose rows contain multi-byte characters', async () => { - const journal = await open() - await journal.appendItem(item(0), body('안녕하세요 🌊 café'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) - const intact = await readFile(logPath) - // Kill mid-row: keep the complete first row plus a fragment of the second. - const torn = Buffer.concat([intact, Buffer.from('{"seq":2,"kind":"it', 'utf-8')]) - await writeFile(logPath, torn) - - await journal.appendItem(item(1), body('b'), { fence: 1 }) - const reopened = await open() - expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([ - body('안녕하세요 🌊 café'), - body('b') - ]) - }) - - it('degrades to read-only when the snapshot comes from a newer schema', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - - const reopened = await open() - expect(reopened.isReadOnly).toBe(true) - await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ - code: 'journal_read_only' - }) - }) - - it('preserves a future-version snapshot with an unknown body kind in place instead of quarantining it', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - // The version advances because bodies changed: a valid newer snapshot - // carries kinds this build cannot parse and must stay unreadable in place. - snapshot.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'future-render-kind', payload: { anything: true } }, - sequence: 1, - observedAt: 1_000 - } - ] - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - - const reopened = await open() - const entries = await readdir(root) - expect(entries.some((name) => name.startsWith('quarantine-'))).toBe(false) - expect(entries.includes(JOURNAL_SNAPSHOT_FILE)).toBe(true) - expect(reopened.isReadOnly).toBe(true) - expect(reopened.snapshot().items).toHaveLength(0) - await expect(reopened.appendItem(item(1), body('b'), { fence: 1 })).rejects.toMatchObject({ - code: 'journal_read_only' - }) - }) - - it('keeps the future-version snapshot bytes when the schema escape hatch rolls the epoch', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - snapshot.items = [ - { - itemId: 'codex:thread-1:turn-1:1', - revision: 1, - body: { kind: 'future-render-kind', payload: { anything: true } }, - sequence: 1, - observedAt: 1_000 - } - ] - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - const reopened = await open() - // Still live in place before the explicit escape hatch runs. - expect((await readdir(root)).some((name) => name.startsWith('quarantine-'))).toBe(false) - - await reopened.rollEpoch('schema_unreadable', 2) - expect(reopened.isReadOnly).toBe(false) - const quarantine = (await readdir(root)).find((name) => name.startsWith('quarantine-')) - expect(quarantine).toBeDefined() - expect(await readFile(join(root, quarantine!), 'utf-8')).toContain('future-render-kind') - }) - - it('reopens a log holding an admitted malformed-percent item id without throwing', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const logPath = join(root, JOURNAL_LOG_FILE) + const epoch = journal.epoch + const nextSeq = journal.cursor().sequence + 1 + await journal.close() // `parseJournalRow` admits any string itemId, so replay must degrade a // malformed percent key to an opaque id instead of throwing URIError. - const malformedKeyRow = JSON.stringify({ - v: 1, - epoch: journal.epoch, - seq: journal.cursor().sequence + 1, - fence: 1, - ts: 1, - kind: 'item', - itemId: '%', - revision: 1, - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] } - }) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}${malformedKeyRow}\n`, 'utf-8') + await appendRawRow( + epoch, + nextSeq, + JSON.stringify({ + v: 1, + epoch, + seq: nextSeq, + fence: 1, + ts: 1, + kind: 'item', + itemId: '%', + revision: 1, + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] } + }) + ) const reopened = await open() expect(reopened.isReadOnly).toBe(false) expect(reopened.snapshot().items.some((entry) => entry.itemId === '%')).toBe(true) }) - - it('allows the explicit schema-unreadable epoch escape hatch while preserving the old files', async () => { - const journal = await open() - await journal.appendItem(item(0), body('a'), { fence: 1 }) - const snapshotPath = join(root, JOURNAL_SNAPSHOT_FILE) - const snapshot = JSON.parse(await readFile(snapshotPath, 'utf-8')) as Record - snapshot.v = 99 - await writeFile(snapshotPath, JSON.stringify(snapshot), 'utf-8') - const reopened = await open() - - await reopened.rollEpoch('schema_unreadable', 2) - expect(reopened.isReadOnly).toBe(false) - expect(reopened.snapshot().items).toHaveLength(0) - expect((await readdir(root)).some((name) => name.startsWith('quarantine-'))).toBe(true) - }) - - it('keeps the unreadable log suffix in the schema escape quarantine', async () => { - const journal = await open() - const logPath = join(root, JOURNAL_LOG_FILE) - const future = JSON.stringify({ - v: 99, - kind: 'item', - epoch: journal.epoch, - seq: 2, - fence: 1, - ts: 1, - itemId: 'future', - revision: 1, - body: { kind: 'status', text: 'preserve these bytes' } - }) - await writeFile(logPath, `${await readFile(logPath, 'utf-8')}${future}\n`, 'utf-8') - const reopened = await open() - - await reopened.rollEpoch('schema_unreadable', 2) - const quarantine = (await readdir(root)).find((name) => name.startsWith('quarantine-')) - expect(quarantine).toBeDefined() - expect(await readFile(join(root, quarantine!), 'utf-8')).toContain('preserve these bytes') - }) }) diff --git a/src/main/native-chat/agent-session-journal/journal-store-test-open.ts b/src/main/native-chat/agent-session-journal/journal-store-test-open.ts new file mode 100644 index 00000000000..fc8bce5af39 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-store-test-open.ts @@ -0,0 +1,34 @@ +// Shared tracked-open helper for journal tests. +// +// Tracking every opened INSTANCE rather than a variable is the point: `close()` +// is idempotent, a module-level `journal` binding can be reassigned to a second +// store mid-suite, and `allSettled` means one failing close cannot skip the rest +// or the directory removal behind it. + +import type { AgentSessionJournal } from './journal-store' +import type { AgentSessionJournalOptions } from './journal-store-contracts' +import { openAgentSessionJournal } from './journal-store-factory' + +export type TrackedJournalOpener = { + open: (options: AgentSessionJournalOptions) => Promise + track: (journal: T) => T + closeAll: () => Promise +} + +export function createTrackedJournalOpener(): TrackedJournalOpener { + const opened: AgentSessionJournal[] = [] + return { + open: async (options) => { + const journal = await openAgentSessionJournal(options) + opened.push(journal) + return journal + }, + track: (journal) => { + opened.push(journal) + return journal + }, + closeAll: async () => { + await Promise.allSettled(opened.splice(0).map((journal) => journal.close())) + } + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store.test.ts b/src/main/native-chat/agent-session-journal/journal-store.test.ts index 9ff37774d27..97eac02fe75 100644 --- a/src/main/native-chat/agent-session-journal/journal-store.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-store.test.ts @@ -1,4 +1,4 @@ -import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises' +import { mkdtemp, readdir, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -11,24 +11,17 @@ import { boundJournalKeyComponent, MAX_JOURNAL_KEY_COMPONENT_CHARS } from '../../../shared/agent-session-journal-item-key' -import { readJournalBlob } from './journal-blob-store' -import { JOURNAL_LOG_FILE, JOURNAL_SNAPSHOT_FILE } from './journal-log-file' import { loadJournal } from './journal-open' import { boundInlineText, boundPayload, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' -import { journalDirectoryFor, journalPathSegment } from './journal-paths' -import { journalDirectoryBytes } from './journal-physical-quota' -import type { JournalLifecycleMutationInput } from './journal-row-builders' +import { journalDatabaseFile, journalDirectoryFor, journalPathSegment } from './journal-paths' import { AgentSessionJournalError, type AgentSessionJournal } from './journal-store' -import { openAgentSessionJournal } from './journal-store-factory' -import { - JOURNAL_DISPATCH_RESERVATION_BYTES, - JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES, - JOURNAL_TURN_TERMINAL_RESERVATION_BYTES -} from './journal-lifecycle-capacity' +import type { openAgentSessionJournal } from './journal-store-factory' +import { createTrackedJournalOpener } from './journal-store-test-open' +import type Database from '../../sqlite/sync-database' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -54,8 +47,10 @@ function body(value: string): AgentJournalItemBody { return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: value }] } } +const journals = createTrackedJournalOpener() + async function open(overrides: Partial[0]> = {}) { - return openAgentSessionJournal({ + return journals.open({ identity: IDENTITY, journalDir: root, now: tick, @@ -70,6 +65,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -151,14 +147,12 @@ describe('fences', () => { }) describe('replay', () => { - it('adopts a caller-provided load without reading the journal files again', async () => { + it('adopts a caller-provided load without replaying the rows again', async () => { const journal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) const loaded = await loadJournal(root, IDENTITY.sessionId) expect(loaded).not.toBeNull() - - await rm(join(root, JOURNAL_LOG_FILE), { force: true }) - await rm(join(root, JOURNAL_SNAPSHOT_FILE), { force: true }) + await journal.close() const reopened = await open({ loaded }) expect(reopened.snapshot()).toEqual(journal.snapshot()) @@ -200,208 +194,27 @@ describe('replay', () => { expect(reopened.snapshot().items).toHaveLength(0) }) - it('preserves the intact prefix and quarantines a corrupt suffix', async () => { + it('keeps the intact prefix and drops the rejected suffix', async () => { const journal = await open() for (let index = 0; index < 4; index += 1) { await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) } const before = journal.epoch - const logPath = join(root, JOURNAL_LOG_FILE) - const lines = (await readFile(logPath, 'utf-8')).split('\n').filter(Boolean) - await writeFile(logPath, `${[...lines.slice(0, 2), ...lines.slice(3)].join('\n')}\n`, 'utf-8') + await journal.close() + await withJournalDatabase(root, (db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(3) + }) const reopened = await open() expect(reopened.epoch).toBe(before) expect(reopened.snapshot().items.map((entry) => entry.body)).toEqual([body('m0')]) - const files = await readdir(root) - expect(files.some((name) => name.startsWith('quarantine-'))).toBe(true) - }) -}) - -describe('automatic compaction', () => { - // Production passes no policy and never called compact(), so the log only - // ever grew — until the size bound refused every append for good. - it('compacts on append once the retention window has rows to shed', async () => { - const policy = { minTailRows: 2, retainTailMs: 0 } - const journal = await open({ compaction: policy }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - expect(journal.compactionBoundary).toBeGreaterThan(0) - // The log sheds instead of growing with every append (7 = epoch row + 6). - const log = await readFile(join(root, JOURNAL_LOG_FILE), 'utf-8') - expect(log.trim().split('\n').length).toBeLessThan(7) - // Nothing is lost: the folded prefix is served from the snapshot. - expect(journal.snapshot().items).toHaveLength(6) - }) - - it('does not rewrite the log while every row is inside the retention window', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 60_000 } }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - expect(journal.compactionBoundary).toBe(0) - }) - - it('can be turned off explicitly', async () => { - const journal = await open({ - autoCompact: false, - compaction: { minTailRows: 2, retainTailMs: 0 } + // Sequences 4 and 5 are VALID rows that the gap at 3 made unreplayable. + // Nothing preserves them; recovery rebuilds the epoch from provider history. + await withJournalDatabase(root, (db) => { + const rows = db.prepare('SELECT seq FROM journal_rows ORDER BY seq').all() + expect(rows.map((row) => (row as { seq: number }).seq)).toEqual([1, 2]) }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - expect(journal.compactionBoundary).toBe(0) - }) - - it('refuses an append when a tail shorter than the row floor cannot make room', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 900 }, - // The tail never reaches the floor, so honouring it would shed nothing. - compaction: { minTailRows: 512, retainTailMs: 10_000 } - }) - let rejected = 0 - for (let index = 0; index < 20; index += 1) { - try { - await journal.appendItem(item(index), body('x'.repeat(96)), { fence: 1 }) - } catch (error) { - expect(error).toMatchObject({ code: 'journal_bound_exceeded' }) - rejected += 1 - } - } - expect(rejected).toBeGreaterThan(0) - }) - - it('refuses once the retained snapshot itself reaches the session bound', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 10_000 }, - compaction: { minTailRows: 10, retainTailMs: 2 * 60 * 60 * 1000 } - }) - let rejected = 0 - for (let index = 0; index < 30; index += 1) { - try { - await journal.appendItem(item(index), body('x'.repeat(128)), { fence: 1 }) - } catch (error) { - expect(error).toMatchObject({ code: 'journal_bound_exceeded' }) - rejected += 1 - } - } - - expect(rejected).toBeGreaterThan(0) - expect(journal.snapshot().items.length).toBeLessThan(30) - }) - - it('keeps the newest rows resumable while shedding under budget pressure', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 10_000 }, - compaction: { minTailRows: 10, retainTailMs: 2 * 60 * 60 * 1000 } - }) - for (let index = 0; index < 30; index += 1) { - await journal.appendItem(item(index), body('x'.repeat(64)), { fence: 1 }) - } - - // The window yields oldest-first, never wholesale: the latest append is - // still in the log, so a client resuming from it does not reload. - const log = (await readFile(join(root, JOURNAL_LOG_FILE), 'utf-8')).trim().split('\n') - expect(log.length).toBeGreaterThan(0) - expect(log.at(-1)).toContain('"seq"') - }) -}) - -describe('compaction and retention', () => { - it('preserves the highest fence across compaction and reopen', async () => { - const journal = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await journal.appendItem(item(0), body('a'), { fence: 7 }) - await journal.compact() - const reopened = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await expect(reopened.appendItem(item(1), body('stale'), { fence: 6 })).rejects.toMatchObject({ - code: 'journal_stale_fence' - }) - }) - - it('preserves tombstones across compaction and reopen', async () => { - const journal = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await journal.appendItem(item(0), body('a'), { fence: 1 }) - await journal.appendTombstone(item(0), { fence: 1 }) - await journal.compact() - const reopened = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - await reopened.appendItem(item(0), body('stale'), { fence: 1 }) - expect(reopened.snapshot().items).toHaveLength(0) - }) - - it('folds the prefix into the snapshot and keeps serving the retained tail', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - for (let index = 0; index < 6; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - const rendered = journal.snapshot() - const tip = journal.cursor() - await journal.compact() - - expect(journal.snapshot()).toEqual(rendered) - expect(journal.readSince({ epoch: tip.epoch, sequence: 1 })).toEqual({ - ok: false, - reset: 'cursor_compacted' - }) - const nearTip = journal.readSince({ epoch: tip.epoch, sequence: tip.sequence - 1 }) - expect(nearTip.ok && nearTip.rows).toHaveLength(1) - - const reopened = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - expect(reopened.snapshot()).toEqual(rendered) - expect(reopened.compactionBoundary).toBe(tip.sequence) - }) - - it('publishes the snapshot and its tail as one write, so a crash before the log rewrite loses nothing', async () => { - const journal = await open({ compaction: { minTailRows: 2, retainTailMs: 0 } }) - for (let index = 0; index < 5; index += 1) { - await journal.appendItem(item(index), body(`m${index}`), { fence: 1 }) - } - const rendered = journal.snapshot() - const logBefore = await readFile(join(root, JOURNAL_LOG_FILE), 'utf-8') - await journal.compact() - const persistedSnapshot = JSON.parse( - await readFile(join(root, JOURNAL_SNAPSHOT_FILE), 'utf-8') - ) as { tail: unknown[] } - expect(persistedSnapshot.tail).toHaveLength(2) - // Simulate the crash: the snapshot landed, the truncation did not. - await writeFile(join(root, JOURNAL_LOG_FILE), logBefore, 'utf-8') - - const reopened = await open() - expect(reopened.snapshot()).toEqual(rendered) - expect(reopened.snapshot().items).toHaveLength(5) - }) - - it('prunes blobs no live row references and keeps the ones that survive', async () => { - const journal = await open({ compaction: { minTailRows: 1, retainTailMs: 0 } }) - const kept = boundPayload('k'.repeat(64), { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 8 - }) - const dropped = boundPayload('d'.repeat(64), { - ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, - inlineHeadBytes: 8 - }) - const { putJournalBlob } = await import('./journal-blob-store') - await putJournalBlob(root, kept.digest, 'k'.repeat(64)) - await putJournalBlob(root, dropped.digest, 'd'.repeat(64)) - await journal.appendItem( - item(0), - { kind: 'tool-call', name: 'bash', input: {}, state: 'completed', output: kept }, - { fence: 1 } - ) - await journal.compact() - - expect(await readJournalBlob(root, kept.digest)).toBe('k'.repeat(64)) - expect(await readJournalBlob(root, dropped.digest)).toBeNull() - }) - - it('refuses a blob name that is not a bare digest, on either slash', async () => { - const { putJournalBlob } = await import('./journal-blob-store') - // A corrupt or crafted row must not steer a read or a write out of the store. - for (const name of ['../../escape', '..\\..\\escape', 'nested/name', 'NOTHEX']) { - expect(await readJournalBlob(root, name)).toBeNull() - await expect(putJournalBlob(root, name, 'payload')).rejects.toThrow('sha256 digest') - } + expect(reopened.repair).toEqual({ malformedRows: 0 }) }) }) @@ -429,244 +242,9 @@ describe('bounds', () => { expect(bounded.head).toBe('small') expect(boundInlineText('small', DEFAULT_JOURNAL_PAYLOAD_LIMITS).text).toBe('small') }) - - it('refuses a single row larger than the per-session size bound', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - }) - // Shedding the whole tail still cannot make room, so the bound holds. - await expect( - journal.appendItem(item(0), body('x'.repeat(4_096)), { fence: 1 }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - }) - - it('refuses an append past the per-session size bound when compaction is off', async () => { - const journal = await open({ - autoCompact: false, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 2_000 } - }) - await expect( - (async () => { - for (let index = 0; index < 50; index += 1) { - await journal.appendItem(item(index), body('x'.repeat(64)), { fence: 1 }) - } - })() - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - }) - - it('refuses an append past the per-window rate bound', async () => { - const journal = await open({ - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxAppendsPerWindow: 3, appendWindowMs: 60_000 } - }) - await expect( - (async () => { - for (let index = 0; index < 10; index += 1) { - await journal.appendItem(item(index), body('x'), { fence: 1 }) - } - })() - ).rejects.toMatchObject({ code: 'journal_rate_exceeded' }) - }) - - it('charges unique blobs and abandoned staging files to one physical quota', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 8_000 } - const journal = await open({ limits, autoCompact: false }) - const payload = 'z'.repeat(1_200) - const bounded = boundPayload(payload, { ...limits, inlineHeadBytes: 8 }) - const toolBody: AgentJournalItemBody = { - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed', - output: bounded - } - - await journal.appendItemWithBlobs(item(1), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - const afterFirst = await journalDirectoryBytes(root) - await journal.appendItemWithBlobs(item(2), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - const afterDuplicate = await journalDirectoryBytes(root) - - expect(afterDuplicate - afterFirst).toBeLessThan(payload.length) - expect(await readdir(join(root, 'blobs'))).toEqual([bounded.digest]) - expect(afterDuplicate).toBeLessThanOrEqual(limits.maxSessionBytes) - - await writeFile(join(root, 'log.jsonl.abandoned.tmp'), 's'.repeat(2_000), 'utf8') - const physical = await journalDirectoryBytes(root) - await expect( - open({ limits: { ...limits, maxSessionBytes: physical - 1 } }) - ).rejects.toMatchObject({ code: 'journal_bound_exceeded' }) - }) - - it('uses a running tool reservation when its authoritative blob cannot fit', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 220 * 1024 } - const journal = await open({ limits, autoCompact: false }) - await journal.appendItem( - item(1), - { kind: 'tool-call', name: 'command', input: {}, state: 'running' }, - { fence: 1 } - ) - for (let ordinal = 10; ordinal < 100; ordinal += 1) { - try { - await journal.appendItem(item(ordinal), body('f'.repeat(4_000)), { fence: 1 }) - } catch (error) { - expect(error).toMatchObject({ code: 'journal_bound_exceeded' }) - break - } - } - const payload = 'o'.repeat(100 * 1024) - const bounded = boundPayload(payload, { ...limits, inlineHeadBytes: 16 * 1024 }) - - await journal.appendItemWithBlobs( - item(1), - { - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed', - output: bounded - }, - [{ digest: bounded.digest, payload }], - { fence: 1 } - ) - - const tool = journal.snapshot().items.find((entry) => entry.itemId.includes('turn-1:1')) - expect(tool?.body).toEqual({ - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed' - }) - expect( - journal - .snapshot() - .items.some( - (entry) => - entry.body.kind === 'status' && entry.body.text.includes('could not be retained') - ) - ).toBe(true) - expect(await readJournalBlob(root, bounded.digest)).toBeNull() - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('keeps cached physical bytes aligned after blob dedupe and compaction', async () => { - const limits = { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 45_000 } - const journal = await open({ - limits, - autoCompact: false, - compaction: { minTailRows: 0, retainTailMs: 0 } - }) - const payload = 'p'.repeat(20_000) - const bounded = boundPayload(payload, { ...limits, inlineHeadBytes: 8 }) - const toolBody: AgentJournalItemBody = { - kind: 'tool-call', - name: 'command', - input: {}, - state: 'completed', - output: bounded - } - - await journal.appendItemWithBlobs(item(1), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - await journal.appendItemWithBlobs(item(2), toolBody, [{ digest: bounded.digest, payload }], { - fence: 1 - }) - await journal.compact(tick() + 10, { minTailRows: 0, retainTailMs: 0 }) - const compactedBytes = await journalDirectoryBytes(root) - expect(compactedBytes).toBeLessThan(limits.maxSessionBytes) - - await journal.appendItem(item(3), body('after compaction'), { fence: 1 }) - - expect(await journalDirectoryBytes(root)).toBeLessThanOrEqual(limits.maxSessionBytes) - expect(await readdir(join(root, 'blobs'))).toEqual([bounded.digest]) - }) }) describe('lifecycle batches', () => { - it('uses a reserved append slot after ordinary rate pressure', async () => { - const journal = await open({ - autoCompact: false, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxAppendsPerWindow: 1, appendWindowMs: 60_000 } - }) - const identity: AgentJournalItemIdentity = { - provider: 'orca', - clientMessageId: 'reserved-prompt' - } - const pending: AgentJournalItemBody = { - kind: 'approval', - title: 'Run a command?', - detail: null, - options: [{ id: 'accept', label: 'Allow' }], - resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } - } - - // The pending row spends the only ordinary slot while reserving its - // terminal append slot for recovery. - await journal.appendLifecycleBatch({ - settlementId: 'reserved-start', - fence: 1, - mutations: [{ kind: 'item', identity, body: pending }] - }) - await expect( - journal.appendItem( - identity, - { - ...pending, - resolution: { - state: 'resolved', - selectedOptionId: 'accept', - resolvedBy: 'test', - resolvedAt: 1 - } - }, - { fence: 1 } - ) - ).resolves.toBeDefined() - }) - - it('rate-limits an unreserved lifecycle batch', async () => { - const journal = await open({ - autoCompact: false, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxAppendsPerWindow: 1, appendWindowMs: 60_000 } - }) - const mutation = (id: string): JournalLifecycleMutationInput => ({ - kind: 'item', - identity: { provider: 'orca', clientMessageId: id }, - body: { kind: 'status', text: 'provider diagnostic' } - }) - await journal.appendLifecycleBatch({ - settlementId: 'unreserved-1', - fence: 1, - mutations: [mutation('one')] - }) - await expect( - journal.appendLifecycleBatch({ - settlementId: 'unreserved-2', - fence: 1, - mutations: [mutation('two')] - }) - ).rejects.toMatchObject({ code: 'journal_rate_exceeded' }) - }) - - it('rebuilds dispatch and turn reservations for pending submissions after reopen', async () => { - const journal = await open({ autoCompact: false }) - await journal.appendSubmission({ - clientMessageId: 'pending-send', - payloadFingerprint: 'fingerprint', - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, - fence: 1 - }) - - const reopened = await open({ autoCompact: false }) - expect(reopened.lifecycleCapacityState()).toEqual({ - reservedBytes: JOURNAL_DISPATCH_RESERVATION_BYTES + JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - reservedAppendSlots: 2 - }) - }) - it('deduplicates concurrent submissions before appending a second row', async () => { const journal = await open() const input = { @@ -684,7 +262,7 @@ describe('lifecycle batches', () => { }) it('applies every mutation at one sequence and deduplicates a replay across reopen', async () => { - const journal = await open({ autoCompact: false }) + const journal = await open() const turn: AgentJournalItemIdentity = { provider: 'legacy', agent: 'codex', @@ -715,9 +293,9 @@ describe('lifecycle batches', () => { .snapshot() .items.some((entry) => entry.body.kind === 'status' && entry.body.text === 'working') ).toBe(false) - await journal.compact(tick() + 10, { minTailRows: 0, retainTailMs: 0 }) + await journal.close() - const reopened = await open({ autoCompact: false }) + const reopened = await open() const beforeReplay = reopened.cursor() const replay = await reopened.appendLifecycleBatch({ settlementId: 'exit:turn-1', @@ -737,53 +315,6 @@ describe('lifecycle batches', () => { ) ).toBe(false) }) - - it('reserves and releases terminal prompts created inside lifecycle batches', async () => { - const journal = await open() - const identity: AgentJournalItemIdentity = { provider: 'orca', clientMessageId: 'prompt-1' } - const pending: AgentJournalItemBody = { - kind: 'approval', - title: 'Run a command?', - detail: null, - options: [{ id: 'accept', label: 'Allow' }], - resolution: { - state: 'pending', - selectedOptionId: null, - resolvedBy: null, - resolvedAt: null - } - } - - await journal.appendLifecycleBatch({ - settlementId: 'prompt-start', - fence: 1, - mutations: [{ kind: 'item', identity, body: pending }] - }) - - expect(journal.lifecycleCapacityState()).toEqual({ - reservedBytes: JOURNAL_ITEM_TERMINAL_RESERVATION_BYTES, - reservedAppendSlots: 1 - }) - - await journal.appendItem( - identity, - { - ...pending, - resolution: { - state: 'resolved', - selectedOptionId: 'accept', - resolvedBy: 'test', - resolvedAt: tick() - } - }, - { fence: 1 } - ) - - expect(journal.lifecycleCapacityState()).toEqual({ - reservedBytes: 0, - reservedAppendSlots: 0 - }) - }) }) describe('journal location', () => { @@ -808,12 +339,32 @@ describe('journal location', () => { }) describe('on-disk layout', () => { - it('writes the log and snapshot beside each other', async () => { + it('keeps the session database and its projection in one directory', async () => { const journal: AgentSessionJournal = await open() await journal.appendItem(item(0), body('a'), { fence: 1 }) - await expect(readFile(join(root, JOURNAL_LOG_FILE), 'utf-8')).resolves.toContain( - '"kind":"item"' - ) - await expect(readFile(join(root, JOURNAL_SNAPSHOT_FILE), 'utf-8')).resolves.toContain('"epoch"') + expect(await readdir(root)).toContain('journal.db') + await journal.close() + await withJournalDatabase(root, (db) => { + const row = db.prepare('SELECT row_json FROM journal_rows WHERE seq = 2').get() + expect((row as { row_json: string }).row_json).toContain('"kind":"item"') + expect(db.prepare('SELECT epoch FROM journal_sessions').get()).toMatchObject({ + epoch: journal.epoch + }) + }) }) }) + +/** Opens the session database directly, so a case can stage a fault or read + * back what a commit actually stored. */ +async function withJournalDatabase( + journalDir: string, + run: (db: Database.Database) => void +): Promise { + const { openJournalDatabase } = await import('./journal-database') + const opened = openJournalDatabase(journalDatabaseFile(journalDir)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} diff --git a/src/main/native-chat/agent-session-journal/journal-store.ts b/src/main/native-chat/agent-session-journal/journal-store.ts index 33f9be0c2e4..ab2715d0d86 100644 --- a/src/main/native-chat/agent-session-journal/journal-store.ts +++ b/src/main/native-chat/agent-session-journal/journal-store.ts @@ -11,20 +11,16 @@ import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' -import { - compactJournal, - DEFAULT_JOURNAL_COMPACTION_POLICY, - type JournalCompactionPolicy -} from './journal-compaction' +import { agentSessionJournalCloseRetries } from './journal-close-retry' +import { openJournalDatabase, type OpenJournalDatabase } from './journal-database' import type { JournalReplacementItem } from './journal-epoch-replacement' import { readJournalSince } from './journal-cursor' -import type { JournalLoad } from './journal-open' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' +import { readJournalRowsAfterCursor, type JournalLoad } from './journal-open' +import { journalDatabaseFile } from './journal-paths' import { markJournalPendingSubmissionsUnknown } from './journal-pending-submission-recovery' import { applyJournalRow, createJournalReducerState, - referencedBlobDigests, renderJournalState, resolveJournalItemId, type JournalReducerState @@ -37,7 +33,6 @@ import { import type { AgentSessionJournalOptions, JournalAppendResult, - JournalBlobInput, JournalItemAppendOptions, JournalLifecycleBatchInput, JournalReadSince, @@ -46,112 +41,79 @@ import type { ResolveDispatchInput } from './journal-store-contracts' import type { AgentJournalEpochReason, JournalRow } from './journal-row-schema' -import { assertJournalWritable, JournalAppendBudget } from './journal-write-guards' -import { journalDirectoryBytes } from './journal-physical-quota' -import type { JournalLifecycleReservation } from './journal-lifecycle-capacity' -import { JournalLifecycleAdmission } from './journal-lifecycle-admission' -import { JournalRowWriter } from './journal-row-writer' -import { JournalEpochController } from './journal-epoch-controller' -import { journalStoreLoadedFields, openJournalStoreState } from './journal-store-open' -import { JournalItemAppender } from './journal-item-appender' -import { JournalLifecycleBatchAppender } from './journal-lifecycle-batch-appender' +import { AgentSessionJournalError } from './journal-write-guards' +import type { JournalRowWriter } from './journal-row-writer' +import type { JournalEpochController } from './journal-epoch-controller' +import { JournalConnectionCloser, JournalWriteQueue } from './journal-store-close' +import { createJournalStoreCollaborators } from './journal-store-collaborators' +import { ensureJournalDir, journalStoreLoadedFields } from './journal-store-open' +import type { JournalItemAppender } from './journal-item-appender' +import type { JournalLifecycleBatchAppender } from './journal-lifecycle-batch-appender' export { AgentSessionJournalError } from './journal-write-guards' export class AgentSessionJournal { private readonly identity: AgentSessionJournalIdentity private readonly journalDir: string - private readonly budget: JournalAppendBudget - private readonly compaction: JournalCompactionPolicy - private readonly autoCompact: boolean + private readonly dbPath: string private readonly now: () => number private readonly mintEpoch: () => string private readonly loaded: JournalLoad | null | undefined private state: JournalReducerState - private tailRows: JournalRow[] = [] - private compactedThrough = 0 - private sizeBytes = 0 private readOnly = false private malformedRows = 0 - private readonly lifecycleAdmission: JournalLifecycleAdmission + private database: OpenJournalDatabase | null = null + private readonly queue: JournalWriteQueue + private readonly closer: JournalConnectionCloser private readonly rowWriter: JournalRowWriter private readonly epochController: JournalEpochController private readonly itemAppender: JournalItemAppender private readonly lifecycleBatchAppender: JournalLifecycleBatchAppender - /** Serializes sequence assignment with the durable write behind it. */ - private writes: Promise = Promise.resolve() + private readonly restore: () => Promise constructor(options: AgentSessionJournalOptions) { this.identity = options.identity this.journalDir = options.journalDir - this.budget = new JournalAppendBudget( - options.identity.sessionId, - options.limits ?? DEFAULT_JOURNAL_PAYLOAD_LIMITS - ) - this.autoCompact = options.autoCompact ?? true - this.compaction = options.compaction ?? DEFAULT_JOURNAL_COMPACTION_POLICY + this.dbPath = journalDatabaseFile(options.journalDir) this.now = options.now ?? (() => Date.now()) this.mintEpoch = options.mintEpoch ?? randomUUID this.loaded = options.loaded this.state = createJournalReducerState(options.identity.sessionId, '') - this.lifecycleAdmission = new JournalLifecycleAdmission( - options.identity.sessionId, - this.budget.maxSessionBytes, - (itemId) => resolveJournalItemId(this.state, itemId), - this.budget.maxAppendsPerWindow - ) - this.rowWriter = new JournalRowWriter({ - journalDir: this.journalDir, - sessionId: options.identity.sessionId, - budget: this.budget, - lifecycleAdmission: this.lifecycleAdmission, - autoCompact: this.autoCompact, - compaction: this.compaction, - now: this.now, - serialize: (run) => this.serializeWrite(run), - readOnly: () => this.readOnly, - setReadOnly: (readOnly) => { - this.readOnly = readOnly - }, - physicalBytes: () => this.sizeBytes, - highestFence: () => this.state.highestFence, - nextSequence: () => this.state.lastSequence + 1, - tailRows: () => this.tailRows, - referencedBlobDigests: () => referencedBlobDigests(this.state), - compact: (now, policy) => this.compact(now, policy), - commit: (row, physicalBytes) => { - applyJournalRow(this.state, row) - this.tailRows.push(row) - this.sizeBytes = physicalBytes - } + // Serializes sequence assignment with the durable write behind it. + this.queue = new JournalWriteQueue(options.identity.sessionId) + this.closer = new JournalConnectionCloser({ + connection: () => this.database?.db ?? null, + enqueue: (run) => this.queue.serializePastGate(run) }) - this.epochController = new JournalEpochController({ + const collaborators = createJournalStoreCollaborators({ identity: this.identity, journalDir: this.journalDir, - budget: this.budget, - compaction: this.compaction, now: this.now, mintEpoch: this.mintEpoch, - serialize: (run) => this.serializeWrite(run), + serialize: (run) => this.queue.serialize(run), + database: () => this.requireDatabase(), + state: () => this.state, readOnly: () => this.readOnly, setReadOnly: (readOnly) => { this.readOnly = readOnly }, - highestFence: () => this.state.highestFence, cursor: this.cursor, - adopt: (loaded) => this.adoptLoadedJournal(loaded) - }) - this.itemAppender = new JournalItemAppender({ + adopt: (loaded) => this.adoptLoadedJournal(loaded), + commit: (row) => applyJournalRow(this.state, row), + loaded: () => this.loaded, + malformedRows: () => this.malformedRows, + setMalformedRows: (count) => { + this.malformedRows = count + }, journal: () => this, - state: () => this.state, - enqueue: (build, blobs) => this.enqueue(build, blobs) - }) - this.lifecycleBatchAppender = new JournalLifecycleBatchAppender({ - state: () => this.state, - cursor: this.cursor, enqueue: (build) => this.enqueue(build) }) + this.rowWriter = collaborators.rowWriter + this.epochController = collaborators.epochController + this.itemAppender = collaborators.itemAppender + this.lifecycleBatchAppender = collaborators.lifecycleBatchAppender + this.restore = collaborators.restore } get isReadOnly(): boolean { @@ -166,31 +128,31 @@ export class AgentSessionJournal { return this.journalDir } - /** Highest sequence folded into the snapshot; rows at or below it are no - * longer individually replayable. */ - get compactionBoundary(): number { - return this.compactedThrough + /** What the last open's repair did. */ + get repair(): { malformedRows: number } { + return { malformedRows: this.malformedRows } } async open(): Promise { - await openJournalStoreState({ - journalDir: this.journalDir, - sessionId: this.identity.sessionId, - maxBytes: this.budget.maxSessionBytes, - loaded: this.loaded, - start: () => this.epochController.start('session_created', 0), - adopt: (loaded) => this.adoptLoadedJournal(loaded), - tailRows: () => this.tailRows, - snapshot: this.snapshot, - rebuildLifecycle: (snapshot, bytes) => this.lifecycleAdmission.rebuild(snapshot, bytes), - appendDisclosure: (identity, body, fence) => this.appendItem(identity, body, { fence }), - highestFence: () => this.state.highestFence, - malformedRows: () => this.malformedRows, - readOnly: () => this.readOnly, - setPhysicalBytes: (bytes) => { - this.sizeBytes = bytes - } - }) + await ensureJournalDir(this.journalDir) + this.database = openJournalDatabase(this.dbPath) + try { + await this.restore() + } catch (error) { + // Nothing else holds a reference to this connection, so a throw here is + // the leak site unless the store releases it itself — and a close that + // REJECTS has not released it, so the store is retained for a later retry + // instead of being dropped with its handle open. + await agentSessionJournalCloseRetries.closeOrRetain(this) + throw error + } + } + + /** Releases the session's SQLite handle. Idempotent on success, a real retry + * after a failure, and permanently closed to writes either way (§ close). */ + close(): Promise { + this.queue.markClosed() + return this.closer.close() } cursor = (): AgentJournalCursor => ({ @@ -212,27 +174,19 @@ export class AgentSessionJournal { canonicalItemId = (itemId: string): string => resolveJournalItemId(this.state, itemId) - reserveLifecycleCapacity(token: JournalLifecycleReservation): Promise { - return this.serializeCapacityMutation(async () => { - this.sizeBytes = await journalDirectoryBytes(this.journalDir) - return this.lifecycleAdmission.reserve(token, this.sizeBytes) - }) - } - - transferLifecycleCapacity(fromId: string, toId: string): Promise { - return this.serializeCapacityMutation(() => this.lifecycleAdmission.transfer(fromId, toId)) - } - - releaseLifecycleCapacity(id: string): Promise { - return this.serializeCapacityMutation(() => this.lifecycleAdmission.release(id)) - } - - lifecycleCapacityState = (): { reservedBytes: number; reservedAppendSlots: number } => - this.lifecycleAdmission.state - readSince(cursor: AgentJournalCursor): JournalReadSince { return readJournalSince( - { state: this.state, tailRows: this.tailRows, readOnly: this.readOnly }, + { + state: this.state, + rowsAfter: (afterSequence) => + readJournalRowsAfterCursor( + this.requireDatabase().db, + this.identity.sessionId, + this.state.epoch, + afterSequence + ), + readOnly: this.readOnly + }, cursor, () => this.cursor() ) @@ -248,16 +202,6 @@ export class AgentSessionJournal { return this.itemAppender.append(identity, body, options) } - /** Blob-before-row admission on the same serialized path as sequence assignment. */ - appendItemWithBlobs( - identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly JournalBlobInput[], - options: JournalItemAppendOptions = { fence: 0 } - ): Promise { - return this.itemAppender.appendWithBlobs(identity, body, blobs, options) - } - appendTombstone( identity: AgentJournalItemIdentity, options: JournalTombstoneInput @@ -303,26 +247,6 @@ export class AgentSessionJournal { return markJournalPendingSubmissionsUnknown(this, fence) } - async compact( - now = this.now(), - policy: JournalCompactionPolicy = this.compaction - ): Promise { - assertJournalWritable(this.readOnly, this.identity.sessionId) - const result = await compactJournal({ - journalDir: this.journalDir, - state: this.state, - tailRows: this.tailRows, - policy, - now, - maxSessionBytes: this.budget.maxSessionBytes, - sessionId: this.identity.sessionId - }) - this.tailRows = result.tailRows - this.compactedThrough = result.compactedThrough - this.state.oldestSequence = result.oldestSequence - this.sizeBytes = await journalDirectoryBytes(this.journalDir) - } - /** The escape hatch for corruption, an unreconcilable prefix, a forked handle, * and an unreadable schema. It invalidates every cursor; clients reload. */ async rollEpoch(reason: AgentJournalEpochReason, fence: number): Promise { @@ -341,24 +265,22 @@ export class AgentSessionJournal { Object.assign(this, journalStoreLoadedFields(loaded)) } + private requireDatabase(): OpenJournalDatabase { + if (!this.database) { + throw new AgentSessionJournalError( + 'journal_closed', + `agent-session journal for ${this.identity.sessionId} is not open` + ) + } + return this.database + } + /** * Assign the next sequence, make the row durable, and fold it through the * SAME reducer replay uses — all inside one serialized step, so concurrent * callers cannot interleave and mint the same sequence. */ - private enqueue( - build: (seq: number, ts: number) => JournalRow, - blobs: readonly JournalBlobInput[] = [] - ): Promise { - return this.rowWriter.enqueue(build, blobs) - } - - private serializeCapacityMutation = (runMutation: () => Promise | T): Promise => - this.serializeWrite(async () => runMutation()) - - private serializeWrite(runWrite: () => Promise): Promise { - const run = this.writes.then(runWrite) - this.writes = run.catch(() => undefined) - return run + private enqueue(build: (seq: number, ts: number) => JournalRow): Promise { + return this.rowWriter.enqueue(build) } } diff --git a/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts new file mode 100644 index 00000000000..ade04667174 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts @@ -0,0 +1,13 @@ +import type { AgentJournalItemBody } from '../../../shared/agent-session-journal-types' + +/** True while an item is still awaiting the row that settles it, so a sink can + * treat that row as lifecycle-critical rather than sheddable under pressure. */ +export function requiresTerminalSettlement(body: AgentJournalItemBody): boolean { + if (body.kind === 'tool-call') { + return body.state === 'running' + } + if (body.kind === 'approval' || body.kind === 'question') { + return body.resolution.state === 'pending' + } + return body.kind === 'status' && body.turnLifecycle?.state === 'running' +} diff --git a/src/main/native-chat/agent-session-journal/journal-tool-output-fallback.ts b/src/main/native-chat/agent-session-journal/journal-tool-output-fallback.ts deleted file mode 100644 index 5b3209ea594..00000000000 --- a/src/main/native-chat/agent-session-journal/journal-tool-output-fallback.ts +++ /dev/null @@ -1,58 +0,0 @@ -import type { - AgentJournalItemBody, - AgentJournalItemIdentity -} from '../../../shared/agent-session-journal-types' -import type { AgentSessionJournal } from './journal-store' -import type { JournalAppendResult } from './journal-store-contracts' -import { AgentSessionJournalError } from './journal-write-guards' -import { boundToolInput, DEFAULT_JOURNAL_PAYLOAD_LIMITS } from './journal-payload-bounds' - -export async function appendToolOutputFallback(input: { - journal: AgentSessionJournal - error: unknown - identity: AgentJournalItemIdentity - body: AgentJournalItemBody - blobs: readonly { digest: string; payload: string }[] - itemId: string - fence: number -}): Promise { - if ( - !(input.error instanceof AgentSessionJournalError) || - input.error.code !== 'journal_bound_exceeded' || - input.body.kind !== 'tool-call' || - input.body.state === 'running' || - input.blobs.length === 0 - ) { - throw input.error - } - const digest = input.blobs[0]?.digest ?? 'unknown' - const cursor = await input.journal.appendLifecycleBatch({ - settlementId: `tool-output-unavailable:${input.itemId}:${digest}`, - fence: input.fence, - mutations: [ - { - kind: 'item', - identity: input.identity, - body: { - kind: 'tool-call', - name: input.body.name, - input: boundToolInput(input.body.input, DEFAULT_JOURNAL_PAYLOAD_LIMITS), - state: input.body.state - } - }, - { - kind: 'item', - identity: { provider: 'orca', clientMessageId: `output-unavailable:${input.itemId}` }, - body: { - kind: 'status', - text: 'The tool completed, but its output could not be retained within the session storage limit.' - } - } - ] - }) - const item = input.journal.snapshot().items.find((entry) => entry.itemId === input.itemId) - if (!item) { - throw new Error('journal_tool_output_fallback_lost') - } - return { cursor, itemId: input.itemId, revision: item.revision } -} diff --git a/src/main/native-chat/agent-session-journal/journal-write-guards.ts b/src/main/native-chat/agent-session-journal/journal-write-guards.ts index 9f76de55745..e1869ae0c73 100644 --- a/src/main/native-chat/agent-session-journal/journal-write-guards.ts +++ b/src/main/native-chat/agent-session-journal/journal-write-guards.ts @@ -1,18 +1,11 @@ // Guards an append clears before it becomes durable. // -// All four refuse loudly rather than degrade: a silent drop here is a message +// Both refuse loudly rather than degrade: a silent drop here is a message // missing from the transcript with nothing to explain it. -import type { JournalPayloadLimits } from './journal-payload-bounds' -import { journalRowByteLength, type JournalRow } from './journal-row-schema' - export class AgentSessionJournalError extends Error { constructor( - readonly code: - | 'journal_read_only' - | 'journal_stale_fence' - | 'journal_bound_exceeded' - | 'journal_rate_exceeded', + readonly code: 'journal_read_only' | 'journal_stale_fence' | 'journal_closed', message: string ) { super(message) @@ -41,94 +34,3 @@ export function assertJournalFence(fence: number, highestFence: number): void { ) } } - -/** Total size and append rate for one session, bounding a runaway agent. */ -export class JournalAppendBudget { - private windowStart = 0 - private appendsInWindow = 0 - - constructor( - private readonly sessionId: string, - private readonly limits: JournalPayloadLimits - ) {} - - fork(): JournalAppendBudget { - return new JournalAppendBudget(this.sessionId, this.limits) - } - - get maxSessionBytes(): number { - return this.limits.maxSessionBytes - } - - get maxAppendsPerWindow(): number { - return this.limits.maxAppendsPerWindow - } - - /** Capture rate state so a speculative append can be rolled back safely. */ - checkpoint(): { windowStart: number; appendsInWindow: number } { - return { windowStart: this.windowStart, appendsInWindow: this.appendsInWindow } - } - - restore(checkpoint: { windowStart: number; appendsInWindow: number }): void { - this.windowStart = checkpoint.windowStart - this.appendsInWindow = checkpoint.appendsInWindow - } - - wouldExceedSize(row: JournalRow, sizeBytes: number): boolean { - return sizeBytes + journalRowByteLength(row) > this.limits.maxSessionBytes - } - - assert(row: JournalRow, ts: number, sizeBytes: number): void { - if (this.wouldExceedSize(row, sizeBytes)) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} reached its ${this.limits.maxSessionBytes}-byte bound` - ) - } - this.assertRate(ts) - } - - /** Lifecycle capacity cannot bypass the session-wide append rate. */ - assertLifecycle(row: JournalRow, sizeBytes: number): void { - if (this.wouldExceedSize(row, sizeBytes)) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} reached its ${this.limits.maxSessionBytes}-byte bound` - ) - } - this.assertRate(row.ts) - } - - /** - * Consume a lifecycle row covered by a pre-reserved append slot. Reserved - * rows still observe the physical quota, but do not spend ordinary window - * rate headroom that may be needed by unrelated traffic. - */ - assertReservedLifecycle(row: JournalRow, sizeBytes: number): void { - if (this.wouldExceedSize(row, sizeBytes)) { - throw new AgentSessionJournalError( - 'journal_bound_exceeded', - `agent-session journal for ${this.sessionId} reached its ${this.limits.maxSessionBytes}-byte bound` - ) - } - } - - private assertRate(ts: number): void { - let windowStart = this.windowStart - let appendsInWindow = this.appendsInWindow - if (ts - windowStart >= this.limits.appendWindowMs) { - windowStart = ts - appendsInWindow = 0 - } - appendsInWindow += 1 - if (appendsInWindow > this.limits.maxAppendsPerWindow) { - // A refusal must not consume a slot, so a later retry can succeed. - throw new AgentSessionJournalError( - 'journal_rate_exceeded', - `agent-session journal for ${this.sessionId} exceeded ${this.limits.maxAppendsPerWindow} appends per ${this.limits.appendWindowMs}ms` - ) - } - this.windowStart = windowStart - this.appendsInWindow = appendsInWindow - } -} diff --git a/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts b/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts index 3930abb6fe8..cda878f5a01 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-history-page.test.ts @@ -1,4 +1,4 @@ -import { appendFile, mkdtemp, rm } from 'node:fs/promises' +import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -19,15 +19,16 @@ import { serializeRemoteRuntimePayload } from '../../../shared/remote-runtime-memory-limits' import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' -import { JOURNAL_LOG_FILE } from '../agent-session-journal/journal-log-file' -import { - serializeJournalRow, - type JournalItemRow, - type JournalRow, - type JournalTombstoneRow +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { insertJournalRow } from '../agent-session-journal/journal-row-table' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import type { + JournalItemRow, + JournalRow, + JournalTombstoneRow } from '../agent-session-journal/journal-row-schema' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' import { projectJournalBatch } from './agent-session-journal-batch' import { readAgentSessionHistory, resolveHistoryLimit } from './agent-session-history-page' @@ -39,6 +40,7 @@ const IDENTITY: AgentSessionJournalIdentity = { providerHandle: { kind: 'codex', threadId: 'thread-1' } } +const journals = createTrackedJournalOpener() let root: string let clock = 1_000 let epochs = 0 @@ -67,7 +69,7 @@ beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-wire-history-')) clock = 1_000 epochs = 0 - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: IDENTITY, journalDir: root, now: tick, @@ -79,6 +81,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -477,12 +480,20 @@ async function reopenWithRawRows(rows: readonly RawSeedRow[]): Promise { diff --git a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts index 2efcc08cde9..214c889b5c9 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.test.ts @@ -1,7 +1,9 @@ // Recovery drives the real journal loader against real on-disk damage: a hole -// punched in the log, and a row stamped with a schema this host cannot read. +// punched in the row sequence, and a row stamped with a schema this host cannot +// read — on both version axes, because only one of them is detectable before a +// read. -import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -9,7 +11,13 @@ import type { AgentJournalItemIdentity, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { JOURNAL_DB_SCHEMA_VERSION } from '../agent-session-journal/journal-database-schema' +import { loadJournal } from '../agent-session-journal/journal-open' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { readJournalEpochRows } from '../agent-session-journal/journal-row-table' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import type Database from '../../sqlite/sync-database' import { openAgentSessionJournalWithRecovery, providerHistoryId, @@ -53,14 +61,15 @@ const CODEX_LINES = [ let root: string let journalDir: string let historyFilePath: string +const journals = createTrackedJournalOpener() function item(ordinal: number): AgentJournalItemIdentity { return { provider: 'codex', threadId: CODEX_SESSION, turnId: 'turn-1', ordinal } } -/** Fills a journal with `count` items and hands back the raw log lines. */ -async function seedJournal(count: number): Promise { - const journal = await openAgentSessionJournal({ identity: IDENTITY, journalDir }) +/** Fills a journal with `count` items and hands back its epoch. */ +async function seedJournal(count: number): Promise { + const journal = await journals.open({ identity: IDENTITY, journalDir }) for (let ordinal = 1; ordinal <= count; ordinal += 1) { await journal.appendItem( item(ordinal), @@ -68,8 +77,48 @@ async function seedJournal(count: number): Promise { { fence: 1 } ) } - const raw = await readFile(join(journalDir, 'log.jsonl'), 'utf-8') - return raw.split('\n').filter((line) => line.trim().length > 0) + const epoch = journal.epoch + await journal.close() + return epoch +} + +/** A journal whose epoch row is gone: every surviving row is unanchored, so a + * repair has to set aside the whole range. */ +async function seedRepairableSession(): Promise { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + await journal.appendSubmission({ + clientMessageId: 'client-message-1', + payloadFingerprint: 'fingerprint-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'add a retry' }] }, + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'client-message-1', + state: 'accepted', + providerIdentity: item(1), + fence: 1 + }) + await journal.close() + await deleteRow(1) +} + +async function withJournalDatabase( + directory: string, + run: (db: Database.Database) => void +): Promise { + const opened = openJournalDatabase(journalDatabaseFile(directory)) + try { + run(opened.db) + } finally { + opened.db.close() + } +} + +/** The same logical hole `findSequenceGap` detects at replay. */ +async function deleteRow(seq: number): Promise { + await withJournalDatabase(journalDir, (db) => { + db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(seq) + }) } beforeEach(async () => { @@ -84,6 +133,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -99,20 +149,20 @@ describe('providerHistoryId', () => { describe('openAgentSessionJournalWithRecovery', () => { it('opens a healthy journal untouched', async () => { await seedJournal(2) - const opened = await openAgentSessionJournalWithRecovery({ - identity: IDENTITY, - journalDir, - fence: 1, - historyFilePath - }) - expect(opened.recovery).toBeNull() - expect(opened.journal.snapshot().items).toHaveLength(2) + const opened = journals.track( + await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }).then((result) => result.journal) + ) + expect(opened.snapshot().items).toHaveLength(2) }) it('rebuilds a holed journal in place on a fresh epoch', async () => { - const lines = await seedJournal(3) - const holed = lines.filter((_line, index) => index !== 1) - await writeFile(join(journalDir, 'log.jsonl'), `${holed.join('\n')}\n`, 'utf-8') + await seedJournal(3) + await deleteRow(3) const opened = await openAgentSessionJournalWithRecovery({ identity: IDENTITY, @@ -120,6 +170,7 @@ describe('openAgentSessionJournalWithRecovery', () => { fence: 1, historyFilePath }) + journals.track(opened.journal) expect(opened.recovery).toMatchObject({ trigger: 'journal_corrupt', reset: 'epoch_changed' }) expect(opened.recovery?.imported).toBeGreaterThan(0) expect(opened.journal.isReadOnly).toBe(false) @@ -130,12 +181,46 @@ describe('openAgentSessionJournalWithRecovery', () => { expect(texts.some((text) => text.includes('add a retry'))).toBe(true) }) - it('reconstructs a future-schema journal into a schema-scoped sibling, never in place', async () => { - const lines = await seedJournal(1) - await writeFile( - join(journalDir, 'log.jsonl'), - `${lines.join('\n')}\n${JSON.stringify({ v: 99, seq: 2, epoch: 'e', kind: 'item' })}\n`, - 'utf-8' + it('reconstructs a future row-body version into a sibling, never in place', async () => { + const epoch = await seedJournal(1) + await withJournalDatabase(journalDir, (db) => { + db.prepare( + 'INSERT INTO journal_rows (session_id, epoch, seq, ts, row_json) VALUES (?, ?, ?, ?, ?)' + ).run( + CODEX_SESSION, + epoch, + 3, + 1, + JSON.stringify({ v: 99, seq: 3, epoch, kind: 'item', fence: 1, ts: 1 }) + ) + }) + + const opened = journals.track( + await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }).then((result) => result.journal) + ) + + // The unreadable journal is left exactly as found; a newer host still owns it. + await withJournalDatabase(journalDir, (db) => { + const rows = readJournalEpochRows(db, CODEX_SESSION, epoch) + expect(rows.some((entry) => entry.rowJson.includes('"v":99'))).toBe(true) + expect(rows).toHaveLength(3) + }) + await opened.close() + await withJournalDatabase(recoveryJournalDir(journalDir), (db) => { + const sibling = db.prepare('SELECT row_json FROM journal_rows').all() + expect(JSON.stringify(sibling)).toContain('add a retry') + }) + }) + + it('reconstructs a future database version into a sibling, never in place', async () => { + await seedJournal(1) + await withJournalDatabase(journalDir, (db) => + db.pragma(`user_version = ${JOURNAL_DB_SCHEMA_VERSION + 1}`) ) const opened = await openAgentSessionJournalWithRecovery({ @@ -144,23 +229,71 @@ describe('openAgentSessionJournalWithRecovery', () => { fence: 1, historyFilePath }) + journals.track(opened.journal) expect(opened.recovery).toMatchObject({ trigger: 'schema_unreadable', reset: 'schema_unreadable' }) expect(opened.recovery?.imported).toBeGreaterThan(0) + // No schema change, no row written, no row deleted. + await withJournalDatabase(journalDir, (db) => { + expect(db.pragma('user_version', { simple: true })).toBe(JOURNAL_DB_SCHEMA_VERSION + 1) + expect(db.prepare('SELECT count(*) AS total FROM journal_rows').get()).toMatchObject({ + total: 2 + }) + }) + }) - // The unreadable journal is left exactly as found; a newer host still owns it. - const untouched = await readFile(join(journalDir, 'log.jsonl'), 'utf-8') - expect(untouched).toContain('"v":99') - const sibling = await readFile(join(recoveryJournalDir(journalDir), 'log.jsonl'), 'utf-8') - expect(sibling).toContain('add a retry') + // The rehydrate deletes every live row to publish its replacement epoch, so + // everything replay rejected is gone for good by the time the import runs. + // Orca minted the submission, receipt and lifecycle identities; no provider + // transcript can hand them back. + it('rebuilds from provider history when the epoch row itself is gone', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + await journal.appendSubmission({ + clientMessageId: 'client-message-1', + payloadFingerprint: 'fingerprint-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'add a retry' }] }, + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'client-message-1', + state: 'accepted', + providerIdentity: item(1), + fence: 1 + }) + await journal.appendLifecycleBatch({ + settlementId: 'settlement-1', + fence: 1, + mutations: [ + { + kind: 'item', + identity: { provider: 'orca', clientMessageId: 'approval-1' }, + body: { kind: 'status', text: 'approved' } + } + ] + }) + await journal.close() + // Sequence 1 is the epoch row: everything behind it is valid but unanchored. + await deleteRow(1) + + const opened = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(opened.journal) + expect(opened.recovery).toMatchObject({ trigger: 'journal_corrupt' }) + expect(opened.recovery?.imported).toBeGreaterThan(0) + expect(JSON.stringify(opened.journal.snapshot().items.map((entry) => entry.body))).toContain( + 'add a retry' + ) }) it('still opens the session when provider history cannot be read', async () => { - const lines = await seedJournal(3) - const holed = lines.filter((_line, index) => index !== 2) - await writeFile(join(journalDir, 'log.jsonl'), `${holed.join('\n')}\n`, 'utf-8') + await seedJournal(3) + await deleteRow(3) const opened = await openAgentSessionJournalWithRecovery({ identity: IDENTITY, @@ -168,9 +301,170 @@ describe('openAgentSessionJournalWithRecovery', () => { fence: 1, historyFilePath: join(root, 'missing.jsonl') }) + journals.track(opened.journal) expect(opened.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) expect(opened.recovery?.error).toBeTruthy() // A missing provider transcript must not clear the intact journal prefix. - expect(opened.journal.snapshot().items).toHaveLength(1) + expect(opened.journal.snapshot().items.map((entry) => entry.body.kind)).toEqual(['message']) + }) + + // A repair that KEEPS a prefix has no emptied epoch to anchor, so nothing + // about the surviving rows records that the deleted suffix was never rebuilt. + // Unmarked, the next probe reads a contiguous anchored prefix, calls it clean, + // and the dropped stretch of timeline is gone for good. + it('keeps a partially repaired journal corrupt until provider history replaces it', async () => { + await seedJournal(3) + await deleteRow(3) + const empty = join(root, 'empty.jsonl') + await writeFile(empty, '', 'utf-8') + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: empty + }) + journals.track(first.journal) + expect(first.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) + expect(first.recovery?.error).toBeTruthy() + // Only the unanchored suffix went; the prefix the repair kept is still live. + expect(first.journal.snapshot().items.map((entry) => entry.body.kind)).toEqual(['message']) + await first.journal.close() + + // The deletion is durable, so the demand for a rebuild has to be too. + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: true }) + + // A readable transcript rebuilds the epoch, and THAT is what retires it. + const retried = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(retried.journal) + expect(retried.recovery?.imported).toBeGreaterThan(0) + await retried.journal.close() + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: false }) + }) + + // The reproduced path. Deleting sequence 1 leaves every surviving row + // unanchored, so the repair drops ALL of them — and provider history is not + // there to publish a replacement. A journal in that state used to reopen as + // clean: an append took sequence 1 as an ordinary row, replay accepted it, + // and recovery never asked the provider for the timeline again. + it('does not normalize an epoch a repair emptied while provider history was unavailable', async () => { + await seedRepairableSession() + const missing = join(root, 'missing.jsonl') + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: missing + }) + journals.track(first.journal) + expect(first.recovery?.error).toBeTruthy() + expect(first.recovery?.imported).toBe(0) + await first.journal.close() + + // Reopen: the epoch still holds nothing but the repair, so recovery runs again. + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: true }) + const reopened = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: missing + }) + journals.track(reopened.journal) + expect(reopened.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) + // Nothing but the anchor: the repair rebuilt no history of its own. + expect(reopened.journal.snapshot().items).toEqual([]) + + // The append lands ABOVE the epoch anchor, never on top of it. + await reopened.journal.appendItem( + item(2), + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'typed later' }] }, + { fence: 1 } + ) + const epoch = reopened.journal.epoch + await reopened.journal.close() + await withJournalDatabase(journalDir, (db) => { + const rows = readJournalEpochRows(db, CODEX_SESSION, epoch) + expect(JSON.parse(rows[0]?.rowJson ?? '{}')).toMatchObject({ kind: 'epoch', seq: 1 }) + }) + }) + + // An empty transcript is a plausible transient provider state, and it used to + // end recovery for good: the import published an empty replacement epoch that + // deleted the repair's anchor, the next probe called that clean, and the user's + // timeline was never rebuilt. + it('does not retire the repair marker when provider history exists but holds no messages', async () => { + await seedRepairableSession() + const empty = join(root, 'empty.jsonl') + await writeFile(empty, '', 'utf-8') + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: empty + }) + journals.track(first.journal) + expect(first.recovery).toMatchObject({ trigger: 'journal_corrupt', imported: 0 }) + expect(first.recovery?.error).toBeTruthy() + // The anchor the repair published is still the epoch; the empty import did + // not replace it with a clean one. + expect(first.journal.snapshot().items).toEqual([]) + const epoch = first.journal.epoch + await first.journal.close() + await withJournalDatabase(journalDir, (db) => { + const rows = readJournalEpochRows(db, CODEX_SESSION, epoch) + expect(JSON.parse(rows[0]?.rowJson ?? '{}')).toMatchObject({ + kind: 'epoch', + seq: 1, + reason: 'unreconcilable_prefix' + }) + }) + + // The session still reports corrupt, so the next attach retries. + expect(await loadJournal(journalDir, CODEX_SESSION)).toMatchObject({ corrupt: true }) + + // And a transcript that DOES have content still rebuilds the timeline. + const retried = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(retried.journal) + expect(retried.recovery?.imported).toBeGreaterThan(0) + expect(JSON.stringify(retried.journal.snapshot().items.map((entry) => entry.body))).toContain( + 'add a retry' + ) + }) + + it('rebuilds the emptied epoch once provider history is readable again', async () => { + await seedRepairableSession() + + const first = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: join(root, 'missing.jsonl') + }) + journals.track(first.journal) + await first.journal.close() + + const retried = await openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath + }) + journals.track(retried.journal) + expect(retried.recovery?.imported).toBeGreaterThan(0) + expect(JSON.stringify(retried.journal.snapshot().items.map((entry) => entry.body))).toContain( + 'add a retry' + ) }) }) diff --git a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts index 05f58a25afe..6e771a31809 100644 --- a/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts +++ b/src/main/native-chat/agent-session-wire/agent-session-journal-recovery.ts @@ -14,6 +14,7 @@ import { type AgentSessionJournalIdentity, type AgentSessionProviderHandle } from '../../../shared/agent-session-journal-types' +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' import { importLegacyTranscriptIntoJournal } from '../agent-session-journal/journal-legacy-import' import { loadJournal } from '../agent-session-journal/journal-open' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' @@ -25,7 +26,8 @@ export type AgentSessionJournalRecovery = { reset: AgentJournalResetReason epoch: string imported: number - /** Set when provider history could not be read; the intact journal prefix remains live. */ + /** Set when provider history could not be read, or held nothing to restore; the + * intact journal prefix remains live. */ error?: string } @@ -55,16 +57,13 @@ export async function openAgentSessionJournalWithRecovery(input: { /** Resolve directly to a transcript instead of discovering it by session id. */ historyFilePath?: string | null }): Promise { - const probe = await loadJournal(input.journalDir, input.identity.sessionId) + const probe = loadJournal(input.journalDir, input.identity.sessionId) if (probe?.readOnly) { const journal = await openAgentSessionJournal({ identity: input.identity, journalDir: recoveryJournalDir(input.journalDir) }) - return { - journal, - recovery: await rehydrate({ ...input, journal, trigger: 'schema_unreadable' }) - } + return { journal, recovery: await rehydrateOrClose(input, journal, 'schema_unreadable') } } const journal = await openAgentSessionJournal({ identity: input.identity, @@ -73,9 +72,30 @@ export async function openAgentSessionJournalWithRecovery(input: { if (!probe?.corrupt) { return { journal, recovery: null } } - // `open()` quarantines the unusable suffix; a successful import rolls once - // more so the rebuilt timeline is the only content of its epoch. - return { journal, recovery: await rehydrate({ ...input, journal, trigger: 'journal_corrupt' }) } + // `open()` drops the unusable suffix; a successful import rolls once more so + // the rebuilt timeline is the only content of its epoch. + return { journal, recovery: await rehydrateOrClose(input, journal, 'journal_corrupt') } +} + +/** `importLegacyTranscriptIntoJournal` can THROW rather than report `ok: false` + * — a journal write failure, for instance — and nothing else holds a reference + * to the journal this function just opened. A close that rejects is retryable, + * so the journal is retained rather than dropped with its handle still open. */ +async function rehydrateOrClose( + input: { + identity: AgentSessionJournalIdentity + fence: number + historyFilePath?: string | null + }, + journal: AgentSessionJournal, + trigger: AgentSessionJournalRecovery['trigger'] +): Promise { + try { + return await rehydrate({ ...input, journal, trigger }) + } catch (error) { + await agentSessionJournalCloseRetries.closeOrRetain(journal) + throw error + } } async function rehydrate(input: { @@ -94,13 +114,16 @@ async function rehydrate(input: { fence: input.fence, ...(input.historyFilePath ? { options: { filePath: input.historyFilePath } } : {}) }) - if (!result.ok) { + // A transcript that held nothing is the same outcome as one that could not be + // read: nothing was restored, so the repair's marker has to stand and be + // retried on a later attach rather than being retired as a completed recovery. + if (!result.ok || !result.replaced) { return { trigger: input.trigger, reset, epoch: input.journal.epoch, imported: 0, - error: result.error + error: result.ok ? 'Provider history held no messages to restore' : result.error } } return { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts index 5be2d8ed09e..a21edcee35c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts @@ -58,8 +58,9 @@ export type AttachFlowInput = { onAcquiring?: () => Promise | void /** Settles writes already captured by the superseded journal before opening another. */ beforeJournalOpen?: () => Promise | void - /** Removes any partial host publication after journal attachment fails. */ - onAttachFailed?: () => void + /** Removes any partial host publication after journal attachment fails, and + * closes the journal handle of the map entry it drops. Awaited: see eviction. */ + onAttachFailed?: () => Promise } export async function performAttach( @@ -216,7 +217,9 @@ async function settlePostAcquisitionAttachFailure( exitProof = error instanceof AgentSessionAcquisitionExitUnprovenError ? 'unproven' : 'exit-proven' } - input.onAttachFailed?.() + // Why: the close is awaited so the map entry is gone only once its handle is + // released, but a failed close must not also cost the store settlement below. + await Promise.resolve(input.onAttachFailed?.()).catch(() => undefined) try { await input.store.settleFailedPostAcquisitionAttachment({ sessionId: record.sessionId, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts index bd79bb1bb45..f4551ef9313 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts @@ -18,6 +18,9 @@ import { } from './structured-agent-session-launch-env' import { refuseAgentSessionMutation } from './structured-agent-session-mutation-admission' import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' +import type { DeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' export function attachStructuredAgentSession( context: StructuredAgentSessionAttachContext, @@ -71,7 +74,10 @@ export function attachStructuredAgentSession( callerKey, params, now: () => context.now(), - onAttachFailed: () => { + // Site 9: this closes the PRIOR map entry it drops, never the provisional + // journal — it has no reference to that one. `onAttached` owns that. + onAttachFailed: async () => { + await context.sessions.get(sessionId)?.journal.close() context.sessions.delete(sessionId) eventSink.close() context.runtimeState.discardEventSink(sessionId) @@ -80,14 +86,27 @@ export function attachStructuredAgentSession( const fence = context.deps.store.getRecord(sessionId)?.lease.runtimeFence ?? 0 const previous = context.sessions.get(sessionId) const previousFence = previous?.fence - eventSink.bind({ - journal: attached.journal, - fence, - publish: () => context.subscribers.publish(sessionId, attached.journal) - }) - const barrier = await eventSink.drained() - if (!barrier.ok) { - throw barrier.error + // Site 8: the provisional journal has no owner until the map takes it, + // and the barrier below throws by design. + try { + await bindAndDrain(eventSink, attached.journal, fence, () => + context.subscribers.publish(sessionId, attached.journal) + ) + } catch (error) { + await agentSessionJournalCloseRetries.closeOrRetain(attached.journal) + throw error + } + // Site 10: a `set` over a live entry would orphan its handle — and a + // close that REJECTED did not release it. The replacement is therefore + // ABORTED rather than completed over a handle nothing can reach again: + // `previous` stays indexed, so teardown still owns it and can retry. + if (previous && previous.journal !== attached.journal) { + try { + await previous.journal.close() + } catch (error) { + await agentSessionJournalCloseRetries.closeOrRetain(attached.journal) + throw error + } } context.sessions.set(sessionId, { journal: attached.journal, @@ -115,3 +134,18 @@ export function attachStructuredAgentSession( }) return context.tasks.trackAttach(attaching) } + +/** Binds the sink to the journal and waits for the barrier the host publishes + * behind. It throws by design when a sink barrier fails. */ +async function bindAndDrain( + eventSink: DeferredStructuredAgentSessionEventSink, + journal: AgentSessionJournal, + fence: number, + publish: () => void +): Promise { + eventSink.bind({ journal, fence, publish }) + const barrier = await eventSink.drained() + if (!barrier.ok) { + throw barrier.error + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts index cfdbf786e14..83766f25fc5 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts @@ -32,6 +32,7 @@ import { } from '../../../shared/agent-session-mutation-envelope' import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { agentSessionProviderHandleChainHead } from '../../../shared/agent-session-provider-handle' +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' import { journalDirectoryFor } from '../agent-session-journal/journal-paths' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import { @@ -162,9 +163,18 @@ export async function attachJournal(input: { fence, historyFilePath }) - return { - ...opened, - unconfirmedClientMessageIds: await opened.journal.markPendingSubmissionsUnknown(fence) + try { + // That await is a WRITE. A failure in it leaves the journal with no caller + // holding a reference to close it. + return { + ...opened, + unconfirmedClientMessageIds: await opened.journal.markPendingSubmissionsUnknown(fence) + } + } catch (error) { + // A rejected close leaves the handle open, so the journal is retained for a + // later retry rather than dropped along with the only reference to it. + await agentSessionJournalCloseRetries.closeOrRetain(opened.journal) + throw error } } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts new file mode 100644 index 00000000000..a4666b4f045 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts @@ -0,0 +1,257 @@ +// A close that REJECTED did not release the handle. +// +// `AgentSessionJournal.close()` is retryable by design: the release step is +// unguarded precisely so a second call is a second attempt. Callers that did +// `close().catch(() => undefined)` and then threw or overwrote their map entry +// turned that retryable failure into a permanent orphan — on POSIX a silent +// leak, on Windows a handle that blocks renaming or removing the directory. +// +// These drive the REAL callers: the attach orchestration's `onAttached`, and +// host teardown, which is what runtime stop calls. Only the lease/record +// machinery around them is stubbed. + +import { access, mkdtemp, rename, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import { + agentSessionJournalCloseRetries, + JournalCloseRetryRegistry +} from '../agent-session-journal/journal-close-retry' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { attachStructuredAgentSession } from './structured-agent-session-attach-orchestration' +import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' +import { tearDownStructuredAgentSessionHost } from './structured-agent-session-host-teardown' +import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' + +const attachFlow = vi.hoisted(() => ({ + journal: null as AgentSessionJournal | null +})) + +// The lease reservation, the record store and the provider child are not what +// these cases are about; `onAttached` is, and it is the real one. +vi.mock('./structured-agent-session-attach-flow', () => ({ + performAttach: async (input: { + onAttached: ( + attached: { journal: AgentSessionJournal; recovery: null }, + generation: string | null + ) => Promise + }) => { + await input.onAttached({ journal: attachFlow.journal!, recovery: null }, null) + return { ok: true, value: {} } + } +})) + +const SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: SESSION, + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: SESSION } +} + +let root: string +const journals = createTrackedJournalOpener() + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +async function expectNothingHoldsTheDirectory(directory: string): Promise { + const dbPath = journalDatabaseFile(directory) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + // The half that actually fails on Windows when a handle is still open. + const moved = `${directory}-moved` + await rename(directory, moved) + await rm(moved, { recursive: true }) +} + +function hostSession(journal: AgentSessionJournal): StructuredAgentSessionHostSession { + return { + journal, + params: {} as StructuredAgentSessionHostSession['params'], + fence: 1, + hasProviderChild: false, + acquisitionGeneration: null + } +} + +/** A journal whose close rejects until `failures` is exhausted, wrapping a real + * store so the handle it holds is a real one. */ +function flakyClose(journal: AgentSessionJournal, failures: number): AgentSessionJournal { + let remaining = failures + return new Proxy(journal, { + get(target, property, receiver) { + if (property !== 'close') { + return Reflect.get(target, property, receiver) + } + return async () => { + if (remaining > 0) { + remaining -= 1 + throw new Error('close rejected') + } + await target.close() + } + } + }) +} + +function attachContext( + sessions: Map +): StructuredAgentSessionAttachContext { + const eventSink = { + sink: {}, + drained: async () => ({ ok: true }) as const, + unbind: () => undefined, + bind: () => undefined, + close: () => undefined + } + return { + deps: { store: { getRecord: () => null }, claimKeyId: 'key-1', journalRoot: root }, + runtimeState: { + resolveRecovery: async () => undefined, + eventSinkFor: () => eventSink, + probeOwner: async () => ({ outcome: 'pid-absent' }), + discardEventSink: () => undefined + }, + sessions, + subscribers: { + reset: () => undefined, + snapshot: () => undefined, + publish: () => undefined + }, + tasks: { trackAttach: (task: Promise) => task }, + reconcileLeases: async () => null, + serialize: (_sessionId: string, task: () => Promise) => task(), + now: () => 1 + } as unknown as StructuredAgentSessionAttachContext +} + +const attachParams = { + envelope: { sessionId: SESSION, clientOperationId: 'op-1' } +} as unknown as Parameters[2] + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-close-retry-')) + // The registry is process-wide; drain it so one case cannot see another's. + await agentSessionJournalCloseRetries.retryAll() +}) + +afterEach(async () => { + await agentSessionJournalCloseRetries.retryAll() + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('the registry', () => { + it('retains a journal whose close rejected and releases it on the retry', async () => { + const directory = join(root, 'retained') + const registry = new JournalCloseRetryRegistry() + const journal = flakyClose( + await journals.open({ identity: IDENTITY, journalDir: directory }), + 1 + ) + + const first = await registry.closeOrRetain(journal) + expect(first.closed).toBe(false) + expect(registry.pendingDirectories).toEqual([directory]) + + expect(await registry.retryAll()).toEqual([]) + expect(registry.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(directory) + }) +}) + +describe('the attach orchestration', () => { + it('ABORTS the map replacement when the previous journal will not close', async () => { + const previousDir = join(root, 'previous') + const provisionalDir = join(root, 'provisional') + const previous = flakyClose( + await journals.open({ identity: IDENTITY, journalDir: previousDir }), + 1 + ) + const provisional = await journals.open({ + identity: IDENTITY, + journalDir: provisionalDir + }) + attachFlow.journal = provisional + const sessions = new Map([[SESSION, hostSession(previous)]]) + + await expect( + attachStructuredAgentSession(attachContext(sessions), 'caller-1', attachParams) + ).rejects.toThrow('close rejected') + + // The live entry is UNTOUCHED: overwriting it would have left its handle + // open with nothing able to reach it again. + expect(sessions.get(SESSION)?.journal).toBe(previous) + // And the provisional journal is owned by the registry, not orphaned. + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(provisionalDir) + }) + + it('retains the provisional journal when its own close rejects on the barrier path', async () => { + const provisionalDir = join(root, 'provisional-barrier') + const provisional = flakyClose( + await journals.open({ identity: IDENTITY, journalDir: provisionalDir }), + 1 + ) + attachFlow.journal = provisional + const sessions = new Map() + const context = attachContext(sessions) + const failing = { + sink: {}, + drained: async () => ({ ok: false, error: new Error('sink barrier failed') }) as const, + unbind: () => undefined, + bind: () => undefined, + close: () => undefined + } + context.runtimeState.eventSinkFor = (() => + failing) as unknown as typeof context.runtimeState.eventSinkFor + + await expect(attachStructuredAgentSession(context, 'caller-1', attachParams)).rejects.toThrow( + 'sink barrier failed' + ) + + expect(sessions.size).toBe(0) + // Retained rather than dropped, so teardown can still release the handle. + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([provisionalDir]) + }) +}) + +describe('teardown, which is what runtime stop calls', () => { + it('retries the journals earlier failure paths could not close', async () => { + const orphanDir = join(root, 'orphan') + const orphan = flakyClose(await journals.open({ identity: IDENTITY, journalDir: orphanDir }), 1) + expect((await agentSessionJournalCloseRetries.closeOrRetain(orphan)).closed).toBe(false) + + // The first teardown reports the still-failing close instead of hiding it. + await tearDownStructuredAgentSessionHost({ phases: [], sessions: new Map() }) + + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(orphanDir) + }) + + it('surfaces a retained close that still rejects, and keeps it for the next stop', async () => { + const orphanDir = join(root, 'stubborn') + const orphan = flakyClose(await journals.open({ identity: IDENTITY, journalDir: orphanDir }), 2) + await agentSessionJournalCloseRetries.closeOrRetain(orphan) + + await expect( + tearDownStructuredAgentSessionHost({ phases: [], sessions: new Map() }) + ).rejects.toMatchObject({ errors: [expect.objectContaining({ message: 'close rejected' })] }) + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([orphanDir]) + + // A later stop is a real retry, not a no-op. + await tearDownStructuredAgentSessionHost({ phases: [], sessions: new Map() }) + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + await expectNothingHoldsTheDirectory(orphanDir) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts index 4aed742f51d..e16f6a63c9d 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink-estimate.ts @@ -2,16 +2,10 @@ import type { AgentJournalItemBody, AgentJournalItemIdentity } from '../../../shared/agent-session-journal-types' -import type { StructuredAgentSessionJournalBlob } from './structured-agent-session-event-sink' export function estimateStructuredAgentSessionItemBytes( identity: AgentJournalItemIdentity, - body: AgentJournalItemBody, - blobs: readonly StructuredAgentSessionJournalBlob[] + body: AgentJournalItemBody ): number { - return ( - Buffer.byteLength(JSON.stringify({ identity, body }), 'utf8') + - blobs.reduce((total, blob) => total + Buffer.byteLength(blob.payload, 'utf8'), 0) + - 512 - ) + return Buffer.byteLength(JSON.stringify({ identity, body }), 'utf8') + 512 } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts index 6befa3b0b62..c97161ce3dd 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.test.ts @@ -287,13 +287,13 @@ describe('deferred structured agent-session event sink', () => { releaseSecond?.() }) - it('replaces a queued same-item checkpoint before any blob is created', async () => { + it('replaces a queued same-item checkpoint before it runs', async () => { const log: Recorded[] = [] const deferred = createDeferredStructuredAgentSessionEventSink() const options = { coalescingKey: 'checkpoint:item-1' } - deferred.sink.appendItem(identity(0), BODY, [], options) - deferred.sink.appendItem(identity(1), BODY, [], options) + deferred.sink.appendItem(identity(0), BODY, options) + deferred.sink.appendItem(identity(1), BODY, options) expect(deferred.state().queuedOperations).toBe(1) deferred.bind(target(6, log)) @@ -306,9 +306,9 @@ describe('deferred structured agent-session event sink', () => { const deferred = createDeferredStructuredAgentSessionEventSink() const options = { coalescingKey: 'checkpoint:item-1' } - deferred.sink.appendItem(identity(0), BODY, [], options) + deferred.sink.appendItem(identity(0), BODY, options) deferred.sink.appendItem(identity(1), BODY) - deferred.sink.appendItem(identity(2), BODY, [], options) + deferred.sink.appendItem(identity(2), BODY, options) deferred.bind(target(6, log)) await deferred.drained() diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts index e0d91f93b71..7e0192f179c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts @@ -8,8 +8,6 @@ import type { JournalLifecycleMutationInput } from '../agent-session-journal/jou import { estimateStructuredAgentSessionItemBytes } from './structured-agent-session-event-sink-estimate' import { StructuredAgentSessionSinkQueue } from './structured-agent-session-event-sink-queue' -export type StructuredAgentSessionJournalBlob = { digest: string; payload: string } - export type StructuredAgentSessionSinkAdmission = | { accepted: true } | { accepted: false; reason: 'backpressure' | 'failed' | 'closed' } @@ -24,7 +22,7 @@ export type StructuredAgentSessionSinkState = { export type StructuredAgentSessionSinkBarrier = { ok: true } | { ok: false; error: unknown } export type StructuredAgentSessionAppendOptions = { - /** Pending checkpoints with this key replace one another before blob writes. */ + /** Pending checkpoints with this key replace one another before they run. */ coalescingKey?: string /** Marks a critical lifecycle operation for lifecycle barriers and diagnostics. */ lifecycle?: boolean @@ -34,7 +32,6 @@ export type StructuredAgentSessionEventSink = { appendItem( identity: AgentJournalItemIdentity, body: AgentJournalItemBody, - blobs?: readonly StructuredAgentSessionJournalBlob[], options?: StructuredAgentSessionAppendOptions ): void appendTombstone( @@ -49,7 +46,6 @@ export type StructuredAgentSessionEventSink = { tryAppendItem?( identity: AgentJournalItemIdentity, body: AgentJournalItemBody, - blobs?: readonly StructuredAgentSessionJournalBlob[], options?: StructuredAgentSessionAppendOptions ): StructuredAgentSessionSinkAdmission appendLifecycleBatch?( @@ -159,32 +155,22 @@ export function createDeferredStructuredAgentSessionEventSink( return { sink: { - appendItem: (identity, body, blobs = [], options = {}) => { + appendItem: (identity, body, options = {}) => { queue.submit( { - bytes: estimateStructuredAgentSessionItemBytes(identity, body, blobs), + bytes: estimateStructuredAgentSessionItemBytes(identity, body), coalescingKey: options.coalescingKey, - run: (bound) => - blobs.length > 0 && typeof bound.journal.appendItemWithBlobs === 'function' - ? bound.journal.appendItemWithBlobs(identity, body, blobs, { - fence: bound.fence - }) - : bound.journal.appendItem(identity, body, { fence: bound.fence }) + run: (bound) => bound.journal.appendItem(identity, body, { fence: bound.fence }) }, options ) }, - tryAppendItem: (identity, body, blobs = [], options = {}) => + tryAppendItem: (identity, body, options = {}) => queue.submit( { - bytes: estimateStructuredAgentSessionItemBytes(identity, body, blobs), + bytes: estimateStructuredAgentSessionItemBytes(identity, body), coalescingKey: options.coalescingKey, - run: (bound) => - blobs.length > 0 && typeof bound.journal.appendItemWithBlobs === 'function' - ? bound.journal.appendItemWithBlobs(identity, body, blobs, { - fence: bound.fence - }) - : bound.journal.appendItem(identity, body, { fence: bound.fence }) + run: (bound) => bound.journal.appendItem(identity, body, { fence: bound.fence }) }, options ), diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts index d1430ac237c..0d2c693c75f 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.test.ts @@ -26,7 +26,9 @@ function context(): StructuredAgentSessionEvictionContext & { order: string[] } return true }) } as unknown as StructuredAgentSessionEvictionContext['adapter'], - forget: vi.fn(() => order.push('forget')), + forget: vi.fn(async () => { + order.push('forget') + }), discardSink: vi.fn(() => order.push('discardSink')), releaseLease: vi.fn(async () => { order.push('releaseLease') @@ -122,7 +124,7 @@ describe('rows the provider emits while closing', () => { return true } } as never, - forget: () => {}, + forget: async () => {}, discardSink: () => state.discardEventSink(sessionId), releaseLease: async () => {} }) @@ -171,7 +173,7 @@ describe('eviction against the real sink cache', () => { sessionId, eventSink: state.eventSinkFor(sessionId), adapter: { closeSession: async () => true } as never, - forget: () => {}, + forget: async () => {}, discardSink: () => state.discardEventSink(sessionId), releaseLease: async () => {} }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts index 5d1bcaf180c..c2591bba567 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-eviction.ts @@ -25,7 +25,10 @@ export type StructuredAgentSessionEvictionContext = { hasProviderChild?: boolean eventSink: DeferredStructuredAgentSessionEventSink adapter: StructuredAgentSessionAdapter - forget: () => void + /** Closes the session's journal handle and drops the map entry. Async and + * awaited: `close()` is ordered behind queued writes, and a delete that + * returns while the close is still queued leaves nothing to retry. */ + forget: () => Promise /** Drops the cached sink so a later attach mints a fresh one. */ discardSink: () => void /** Hands the lease back now that this host's child is proven gone. No-ops when the record is diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts index e125d3e49ae..f0f410b66aa 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.test.ts @@ -12,7 +12,8 @@ import { setStoredAgentSessionHandoffStage, stopStoredAgentSessionOwnerForHandoff } from '../../runtime/agent-session-handoff-record-transitions' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import type { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' import { createStructuredHandoffFlowContext } from './structured-agent-session-handoff-flow-context' import { handoffStructuredSessionToTui } from './structured-agent-session-handoff-forward' @@ -21,6 +22,8 @@ import type { StructuredTuiOwner } from './structured-agent-session-handoff-types' +const journals = createTrackedJournalOpener() + const NOW = 1_800_000_000_000 const SESSION = 'session-handoff' const PLAIN_RESIDUE = 'session-plain-residue' @@ -211,7 +214,7 @@ beforeEach(async () => { stopRecoveredOwner = vi.fn(async () => undefined) store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) await establishNativeOwner() - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -232,6 +235,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -426,7 +430,7 @@ describe('structured session ownership recovery on restore', () => { : { outcome: 'pid-absent' }, now: NOW + 1_000 }) - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts index 1828807f887..4c807d879b6 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.test.ts @@ -8,13 +8,15 @@ import type { } from '../../../shared/agent-session-record' import { LOCAL_EXECUTION_HOST_ID } from '../../../shared/execution-host' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { createDeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' import { acquireNativeHandoffOwner, structuredTuiTranscriptImportOptions } from './structured-agent-session-host-handoff' +const journals = createTrackedJournalOpener() + function importRecord(provider: 'claude' | 'codex', accountHome: string): AgentSessionRecord { return { provider, @@ -54,6 +56,7 @@ describe('native handoff acquisition', () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -82,7 +85,7 @@ describe('native handoff acquisition', () => { }, now }) - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId, workspaceId: location.workspaceId, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts index 5f3bbc5c731..2afd94ba128 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts @@ -48,7 +48,10 @@ export async function evictHeldStructuredAgentSession( hasProviderChild: hasProviderChild(context, sessionId), eventSink: context.runtimeState.eventSinkFor(sessionId), adapter: context.deps.adapter, - forget: () => context.sessions.delete(sessionId), + forget: async () => { + await context.sessions.get(sessionId)?.journal.close() + context.sessions.delete(sessionId) + }, discardSink: () => context.runtimeState.discardEventSink(sessionId), releaseLease: () => releaseStoredStructuredAgentSessionOwner({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-teardown.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-teardown.ts new file mode 100644 index 00000000000..2b230db1357 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-teardown.ts @@ -0,0 +1,53 @@ +// Host teardown, made failure-complete. +// +// A trailing "close every journal" statement is skipped on exactly the path +// that leaks: `flushAllEventSinks` throws BY DESIGN when a sink barrier fails, +// and the attach drain can reject too. Every connection would then be left open +// with the global runtime reference already cleared — the one state from which +// nothing can ever close them. + +import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' +import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' + +export type StructuredAgentSessionTeardownPhase = { + name: string + run: () => Promise | void +} + +export async function tearDownStructuredAgentSessionHost(input: { + phases: readonly StructuredAgentSessionTeardownPhase[] + sessions: Map +}): Promise { + const failures: unknown[] = [] + for (const phase of input.phases) { + try { + await phase.run() + } catch (error) { + failures.push(error) + } + } + + const entries = [...input.sessions.entries()] + // `allSettled`, so one rejected close cannot skip the others. + const closed = await Promise.allSettled(entries.map(([, session]) => session.journal.close())) + closed.forEach((result, index) => { + const sessionId = entries[index]?.[0] + if (result.status === 'fulfilled') { + // Only a FULFILLED close drops the entry. One that rejected stays indexed, + // which is what makes a later close a real retry rather than a no-op. + if (sessionId !== undefined) { + input.sessions.delete(sessionId) + } + return + } + failures.push(result.reason) + }) + + // Journals an earlier failure path could not close are retried HERE, which is + // the only place that owns them once their caller has unwound. + failures.push(...(await agentSessionJournalCloseRetries.retryAll())) + + if (failures.length > 0) { + throw new AggregateError(failures, 'agent session host teardown failed') + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts index 8de64e3f419..5354670fa0b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts @@ -13,7 +13,7 @@ import type { import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { journalDirectoryFor } from '../agent-session-journal/journal-paths' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import type { AgentSessionDispatchOutcome, StructuredAgentSessionAdapter @@ -30,6 +30,8 @@ import { resetHostTestOperationIds } from './structured-agent-session-host-test-data' +const journals = createTrackedJournalOpener() + const CALLER = { callerKey: 'client-1' } function envelope( @@ -97,7 +99,7 @@ async function attach(): Promise { async function seedApproval(optionId = 'allow'): Promise<{ itemId: string; revision: number }> { const identity = { provider: 'codex' as const, threadId: THREAD, turnId: 'turn-1', ordinal: 99 } const journalDir = journalDirectoryFor(root, { workspaceId: 'workspace-1', sessionId: SESSION }) - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -157,6 +159,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await host.flushAllStreamedEvents() await rm(root, { recursive: true, force: true }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index 3c5256cece0..99d6c212c10 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -49,6 +49,7 @@ import { type StructuredAgentSessionMutationContext } from './structured-agent-session-host-mutations' import { StructuredAgentSessionReadableRestorer } from './structured-agent-session-readable-restorer' +import { tearDownStructuredAgentSessionHost } from './structured-agent-session-host-teardown' import type { StructuredAgentSessionCaller, StructuredAgentSessionHostDeps, @@ -108,6 +109,8 @@ export class StructuredAgentSessionHost { resolveRecovery: (sessionId) => this.runtimeState.resolveRecovery(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), hasSession: (sessionId) => this.sessions.has(sessionId), + // Site 10: cannot overwrite a live entry — the restorer returns early on + // `hasSession` inside the same serialized step as this `set`. onReadable: (sessionId, restored) => this.sessions.set(sessionId, restored), restoreHandoff: (sessionId) => this.handoffs.restore(sessionId) }) @@ -249,11 +252,16 @@ export class StructuredAgentSessionHost { this.runtimeState.flushEventSink(sessionId) async flushAllStreamedEvents(): Promise { - this.holds.dispose() - this.runtimeState.stopLeaseRenewal() - this.handoffs.stopTuiHistoryCatchup() - await this.tasks.drainAttaches() - await this.runtimeState.flushAllEventSinks() + await tearDownStructuredAgentSessionHost({ + phases: [ + { name: 'dispose-holds', run: () => this.holds.dispose() }, + { name: 'stop-lease-renewal', run: () => this.runtimeState.stopLeaseRenewal() }, + { name: 'stop-tui-catchup', run: () => this.handoffs.stopTuiHistoryCatchup() }, + { name: 'drain-attaches', run: () => this.tasks.drainAttaches() }, + { name: 'flush-event-sinks', run: () => this.runtimeState.flushAllEventSinks() } + ], + sessions: this.sessions + }) } private mutationContext(): StructuredAgentSessionMutationContext { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-journal-handles.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-journal-handles.test.ts new file mode 100644 index 00000000000..b1cfd81b2a5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-journal-handles.test.ts @@ -0,0 +1,250 @@ +// Journal handle ownership across the wire layer. +// +// Every one of these sites is reached only when something has already gone +// wrong, so a happy-path assertion proves nothing about them. On POSIX a leak +// is silent; the rename/remove pair below is the half that actually fails on +// Windows. + +import { access, mkdtemp, rename, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import type * as JournalLegacyImport from '../agent-session-journal/journal-legacy-import' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { openAgentSessionJournalWithRecovery } from './agent-session-journal-recovery' +import { + evictStructuredAgentSession, + STRUCTURED_AGENT_SESSION_EVICTION_STEPS, + type StructuredAgentSessionEvictionContext +} from './structured-agent-session-eviction' +import { tearDownStructuredAgentSessionHost } from './structured-agent-session-host-teardown' +import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' + +const legacyImport = vi.hoisted(() => ({ throws: false })) + +vi.mock('../agent-session-journal/journal-legacy-import', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + importLegacyTranscriptIntoJournal: async ( + input: Parameters[0] + ) => { + if (legacyImport.throws) { + throw new Error('legacy import threw instead of reporting a failure') + } + return actual.importLegacyTranscriptIntoJournal(input) + } + } +}) + +const SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: SESSION, + workspaceId: 'ws-1', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: SESSION } +} + +let root: string +let journalDir: string +const journals = createTrackedJournalOpener() + +async function exists(path: string): Promise { + return access(path) + .then(() => true) + .catch(() => false) +} + +async function expectNothingHoldsTheDirectory(directory: string): Promise { + const dbPath = journalDatabaseFile(directory) + expect(await exists(`${dbPath}-wal`)).toBe(false) + expect(await exists(`${dbPath}-shm`)).toBe(false) + const moved = `${directory}-moved` + await rename(directory, moved) + await rm(moved, { recursive: true }) +} + +function hostSession(journal: AgentSessionJournal): StructuredAgentSessionHostSession { + return { + journal, + params: {} as StructuredAgentSessionHostSession['params'], + fence: 1, + hasProviderChild: false, + acquisitionGeneration: null + } +} + +function evictionContext( + overrides: Partial +): StructuredAgentSessionEvictionContext { + return { + sessionId: SESSION, + hasProviderChild: false, + eventSink: { + drained: async () => ({ ok: true }) as const, + unbind: () => undefined, + close: () => undefined + } as unknown as StructuredAgentSessionEvictionContext['eventSink'], + adapter: {} as StructuredAgentSessionEvictionContext['adapter'], + forget: async () => undefined, + discardSink: () => undefined, + releaseLease: async () => undefined, + ...overrides + } +} + +beforeEach(async () => { + legacyImport.throws = false + root = await mkdtemp(join(tmpdir(), 'orca-wire-handles-')) + journalDir = join(root, 'journal') +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +describe('site 6: recovery rehydration', () => { + it('closes the journal it opened when the legacy import throws', async () => { + const seeded = await journals.open({ identity: IDENTITY, journalDir }) + for (let ordinal = 1; ordinal <= 3; ordinal += 1) { + await seeded.appendItem( + { provider: 'codex', threadId: SESSION, turnId: 'turn-1', ordinal }, + { kind: 'status', text: `seed-${ordinal}` }, + { fence: 1 } + ) + } + await seeded.close() + // Punch a hole in the middle so recovery takes the `journal_corrupt` branch. + const { openJournalDatabase } = await import('../agent-session-journal/journal-database') + const opened = openJournalDatabase(journalDatabaseFile(journalDir)) + opened.db.prepare('DELETE FROM journal_rows WHERE seq = ?').run(3) + opened.db.close() + legacyImport.throws = true + + await expect( + openAgentSessionJournalWithRecovery({ + identity: IDENTITY, + journalDir, + fence: 1, + historyFilePath: join(root, 'missing.jsonl') + }) + ).rejects.toThrow('legacy import threw') + await expectNothingHoldsTheDirectory(journalDir) + }) +}) + +describe('sites 9 and 10: the delete and overwrite callbacks', () => { + it('awaits the journal close before dropping the map entry', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + const sessions = new Map([[SESSION, hostSession(journal)]]) + const order: string[] = [] + + await evictStructuredAgentSession( + evictionContext({ + forget: async () => { + order.push('close-started') + await sessions.get(SESSION)?.journal.close() + order.push('closed') + sessions.delete(SESSION) + order.push('forgotten') + } + }), + STRUCTURED_AGENT_SESSION_EVICTION_STEPS + ) + + expect(order).toEqual(['close-started', 'closed', 'forgotten']) + expect(sessions.size).toBe(0) + await expectNothingHoldsTheDirectory(journalDir) + }) + + it('aborts the eviction with the session still indexed when the close rejects', async () => { + const journal = await journals.open({ identity: IDENTITY, journalDir }) + const sessions = new Map([[SESSION, hostSession(journal)]]) + + await expect( + evictStructuredAgentSession( + evictionContext({ + forget: async () => { + await Promise.reject(new Error('close rejected')) + } + }), + STRUCTURED_AGENT_SESSION_EVICTION_STEPS + ) + ).rejects.toMatchObject({ step: 'forget-session' }) + // Still indexed, so the next close is a real retry. + expect(sessions.has(SESSION)).toBe(true) + }) +}) + +describe('site 11: host teardown is failure-complete', () => { + async function twoSessions(): Promise> { + const first = await journals.open({ identity: IDENTITY, journalDir }) + const second = await journals.open({ + identity: { ...IDENTITY, sessionId: `${SESSION}-b` }, + journalDir: join(root, 'journal-b') + }) + return new Map([ + [SESSION, hostSession(first)], + [`${SESSION}-b`, hostSession(second)] + ]) + } + + it('closes every journal and clears the map on the happy path', async () => { + const sessions = await twoSessions() + await tearDownStructuredAgentSessionHost({ phases: [], sessions }) + + expect(sessions.size).toBe(0) + await expectNothingHoldsTheDirectory(journalDir) + await expectNothingHoldsTheDirectory(join(root, 'journal-b')) + }) + + // Against a trailing-statement design this case fails: `flushAllEventSinks` + // throws by design, so the close would be skipped on exactly the leaking path. + it('still closes every journal when a teardown phase throws', async () => { + const sessions = await twoSessions() + const barrierError = new Error('sink barrier failed') + + await expect( + tearDownStructuredAgentSessionHost({ + phases: [ + { + name: 'flush-event-sinks', + run: () => { + throw barrierError + } + } + ], + sessions + }) + ).rejects.toMatchObject({ errors: [barrierError] }) + + expect(sessions.size).toBe(0) + await expectNothingHoldsTheDirectory(journalDir) + await expectNothingHoldsTheDirectory(join(root, 'journal-b')) + }) + + it('keeps the entry whose close rejected, and surfaces the rejection', async () => { + const sessions = await twoSessions() + const failing = sessions.get(SESSION) + const closeError = new Error('close rejected') + if (failing) { + failing.journal = { + close: () => Promise.reject(closeError) + } as unknown as AgentSessionJournal + } + + await expect( + tearDownStructuredAgentSessionHost({ phases: [], sessions }) + ).rejects.toMatchObject({ errors: [closeError] }) + + // Only the failure stays indexed — `status === 'fulfilled'`, not "settled". + expect([...sessions.keys()]).toEqual([SESSION]) + await expectNothingHoldsTheDirectory(join(root, 'journal-b')) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts index 5f2589de4f2..5bed8f2920e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-read-restore.ts @@ -39,7 +39,7 @@ export async function restoreStructuredAgentSessionRead( workspaceId: record.location.workspaceId, sessionId }) - const loaded = await loadJournal(journalDir, sessionId) + const loaded = loadJournal(journalDir, sessionId) if (!loaded || loaded.corrupt) { return null } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts index b06ec51018f..581743633c0 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts @@ -4,17 +4,19 @@ import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { AgentJournalMessageItem } from '../../../shared/agent-session-journal-types' import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { performSend, type AgentSessionTurnContext } from './structured-agent-session-turns' +const journals = createTrackedJournalOpener() + let root: string let journal: AgentSessionJournal beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-send-idempotency-')) - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: 'session-1', workspaceId: 'workspace-1', @@ -27,6 +29,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts index bd4dd1958ef..6b627726f98 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts @@ -1,4 +1,4 @@ -import { appendFile, mkdtemp, rm } from 'node:fs/promises' +import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it } from 'vitest' @@ -11,26 +11,30 @@ import { REMOTE_RUNTIME_MAX_OUTBOUND_JSON_BYTES, serializeRemoteRuntimePayload } from '../../../shared/remote-runtime-memory-limits' -import { JOURNAL_LOG_FILE } from '../agent-session-journal/journal-log-file' -import { serializeJournalRow, type JournalRow } from '../agent-session-journal/journal-row-schema' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { insertJournalRow } from '../agent-session-journal/journal-row-table' +import type { JournalRow } from '../agent-session-journal/journal-row-schema' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { AgentSessionSubscribers } from './structured-agent-session-subscribers' const SESSION = 'subscriber-session' let root: string +const journals = createTrackedJournalOpener() beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-agent-subscribers-')) }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) describe('AgentSessionSubscribers', () => { it('publishes the current fence when a resumed cursor is already caught up', async () => { - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -67,7 +71,7 @@ describe('AgentSessionSubscribers', () => { }) it('publishes handoff-only changes without serializing a transcript snapshot', async () => { - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -112,7 +116,7 @@ describe('AgentSessionSubscribers', () => { it('catches a subscriber up past a pre-existing unsendable removal with a bounded reset', async () => { const journalDir = join(root, 'oversized-removal-journal') - const seeded = await openAgentSessionJournal({ + const seeded = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -150,12 +154,20 @@ describe('AgentSessionSubscribers', () => { ts: 2_001 } ] - await appendFile( - join(journalDir, JOURNAL_LOG_FILE), - `${rows.map(serializeJournalRow).join('\n')}\n`, - 'utf-8' - ) - const journal = await openAgentSessionJournal({ + // Staged straight into the session database, exactly as a previous writer + // would have committed them. + await seeded.close() + const opened = openJournalDatabase(journalDatabaseFile(journalDir)) + try { + opened.db.exec('BEGIN IMMEDIATE') + for (const row of rows) { + insertJournalRow(opened.db, SESSION, row) + } + opened.db.exec('COMMIT') + } finally { + opened.db.close() + } + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts index 47edf3470e6..5222f9557f9 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts @@ -3,14 +3,9 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' -import { - performCancel, - performSend, - type AgentSessionTurnContext -} from './structured-agent-session-turns' -import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from '../agent-session-journal/journal-payload-bounds' +import { performCancel, type AgentSessionTurnContext } from './structured-agent-session-turns' const IDENTITY: AgentSessionJournalIdentity = { sessionId: 'session-1', @@ -21,8 +16,10 @@ const IDENTITY: AgentSessionJournalIdentity = { } let root: string | null = null +const journals = createTrackedJournalOpener() afterEach(async () => { + await journals.closeAll() if (root) { await rm(root, { recursive: true, force: true }) root = null @@ -32,7 +29,7 @@ afterEach(async () => { describe('performCancel', () => { it('acknowledges only the request and leaves the running lifecycle row intact', async () => { root = await mkdtemp(join(tmpdir(), 'orca-turn-cancel-')) - const journal = await openAgentSessionJournal({ identity: IDENTITY, journalDir: root }) + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) const lifecycleIdentity = { provider: 'legacy' as const, agent: 'codex' as const, @@ -77,190 +74,3 @@ describe('performCancel', () => { ]) }) }) - -describe('performSend lifecycle capacity', () => { - it('refuses before provider contact when dispatch plus terminal capacity cannot fit', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 100 * 1024 } - }) - const dispatch = vi.fn() - const ctx = turnContext(journal, { dispatch } as unknown as StructuredAgentSessionAdapter) - - const result = await performSend(ctx, { - clientMessageId: 'message-1', - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - - expect(result).toMatchObject({ ok: false }) - expect(dispatch).not.toHaveBeenCalled() - expect(journal.submissions()).toEqual([]) - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('binds a synchronous turn start to tentative capacity and releases only on terminality', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 400 * 1024 } - }) - const turnIdentity = { - provider: 'legacy' as const, - agent: 'codex' as const, - sessionId: 'session-1', - recordId: 'turn-lifecycle:turn-1' - } - const dispatch = vi.fn(async () => { - await journal.appendItem( - turnIdentity, - { - kind: 'status', - text: 'Agent is working…', - turnLifecycle: { turnId: 'turn-1', state: 'running' } - }, - { fence: 1 } - ) - return { - state: 'accepted' as const, - providerIdentity: { - provider: 'codex' as const, - threadId: 'thread-1', - turnId: 'turn-1', - ordinal: 0 - } - } - }) - const ctx = turnContext(journal, { dispatch } as unknown as StructuredAgentSessionAdapter) - - const result = await performSend(ctx, { - clientMessageId: 'message-1', - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - - expect(result).toMatchObject({ ok: true }) - expect(journal.lifecycleCapacityState()).toEqual({ - reservedBytes: 128 * 1024, - reservedAppendSlots: 1 - }) - await journal.appendLifecycleBatch({ - settlementId: 'turn-completed:turn-1', - fence: 1, - mutations: [{ kind: 'tombstone', identity: turnIdentity }] - }) - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - }) - - it('keeps response-before-start capacity on the Codex turn lifecycle identity', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 220 * 1024 } - }) - const turnIdentity = { - provider: 'legacy' as const, - agent: 'codex' as const, - sessionId: 'session-1', - recordId: 'turn-lifecycle:turn-1' - } - const ctx = turnContext(journal, { - dispatch: vi.fn(async () => ({ - state: 'accepted' as const, - providerIdentity: { - provider: 'codex' as const, - threadId: 'thread-1', - turnId: 'turn-1', - ordinal: 0 - } - })) - } as unknown as StructuredAgentSessionAdapter) - - await expect( - performSend(ctx, { - clientMessageId: 'message-1', - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - ).resolves.toMatchObject({ ok: true }) - await expect( - journal.appendItem( - turnIdentity, - { - kind: 'status', - text: 'Agent is working…', - turnLifecycle: { turnId: 'turn-1', state: 'running' } - }, - { fence: 1 } - ) - ).resolves.toBeDefined() - expect( - journal - .snapshot() - .items.some( - (item) => - item.body.kind === 'status' && - item.body.turnLifecycle?.turnId === 'turn-1' && - item.body.turnLifecycle.state === 'running' - ) - ).toBe(true) - }) - - it('transfers non-Codex reservations so repeated sends can settle without leaking capacity', async () => { - root = await mkdtemp(join(tmpdir(), 'orca-turn-capacity-')) - const journal = await openAgentSessionJournal({ - identity: IDENTITY, - journalDir: root, - limits: { ...DEFAULT_JOURNAL_PAYLOAD_LIMITS, maxSessionBytes: 220 * 1024 } - }) - const dispatch = vi.fn(async ({ clientMessageId }: { clientMessageId: string }) => ({ - state: 'accepted' as const, - providerIdentity: { - provider: 'claude' as const, - sessionId: 'claude-session', - uuid: `turn-${clientMessageId}` - } - })) - const ctx = turnContext(journal, { dispatch } as unknown as StructuredAgentSessionAdapter) - - for (let index = 0; index < 6; index += 1) { - const clientMessageId = `message-${index}` - const result = await performSend(ctx, { - clientMessageId, - payloadFingerprint: 'a'.repeat(64), - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'run' }] } - }) - expect(result).toMatchObject({ ok: true }) - await journal.appendItem( - { - provider: 'claude', - sessionId: 'claude-session', - uuid: `turn-${clientMessageId}` - }, - { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'done' }] }, - { fence: 1 } - ) - expect(journal.lifecycleCapacityState()).toEqual({ reservedBytes: 0, reservedAppendSlots: 0 }) - } - }) -}) - -function turnContext( - journal: Awaited>, - adapter: StructuredAgentSessionAdapter -): AgentSessionTurnContext { - return { - sessionId: 'session-1', - journal, - fence: 1, - adapter, - persistOptions: async () => undefined, - resolvedBy: 'client-1', - publish: vi.fn(), - now: () => 1 - } -} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts index 0c8f43efe1b..f1717027b8b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts @@ -7,7 +7,6 @@ // turn the provider already accepted. import type { AgentJournalMessageItem } from '../../../shared/agent-session-journal-types' -import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' import type { AgentSessionCancelResult, AgentSessionSendResult, @@ -18,14 +17,6 @@ import type { AgentSessionDispatchOutcome, StructuredAgentSessionAdapter } from './structured-agent-session-adapter' -import { - dispatchReservationId, - JOURNAL_DISPATCH_RESERVATION_BYTES, - JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - lifecycleReservationIdForItem, - tentativeTurnReservationId -} from '../agent-session-journal/journal-lifecycle-capacity' - export { performSetOption } from './structured-agent-session-turns-options' export { performPrompt } from './structured-agent-session-turns-prompt' @@ -104,42 +95,8 @@ export async function performSend( } } if (!(input.retryUnknown && existing?.dispatchState === 'unknown')) { - const dispatchReservation = dispatchReservationId(input.clientMessageId) - const tentativeReservation = tentativeTurnReservationId(input.clientMessageId) - const dispatchReserved = await ctx.journal.reserveLifecycleCapacity({ - id: dispatchReservation, - bytes: JOURNAL_DISPATCH_RESERVATION_BYTES, - appendSlots: 1 - }) - const turnReserved = - dispatchReserved && - (await ctx.journal.reserveLifecycleCapacity({ - id: tentativeReservation, - bytes: JOURNAL_TURN_TERMINAL_RESERVATION_BYTES, - appendSlots: 1 - })) - if (!dispatchReserved || !turnReserved) { - await ctx.journal.releaseLifecycleCapacity(dispatchReservation) - await ctx.journal.releaseLifecycleCapacity(tentativeReservation) - return invalid('The session does not have enough durable capacity to start another turn.') - } - try { - await ctx.journal.appendSubmission({ ...input, fence: ctx.fence }) - } catch (error) { - await ctx.journal.releaseLifecycleCapacity(dispatchReservation) - await ctx.journal.releaseLifecycleCapacity(tentativeReservation) - throw error - } + await ctx.journal.appendSubmission({ ...input, fence: ctx.fence }) ctx.publish() - } else { - const retryReserved = await ctx.journal.reserveLifecycleCapacity({ - id: dispatchReservationId(input.clientMessageId), - bytes: JOURNAL_DISPATCH_RESERVATION_BYTES, - appendSlots: 1 - }) - if (!retryReserved) { - return invalid('The session does not have enough durable capacity to retry this turn.') - } } const outcome = await dispatchSafely(ctx, input.clientMessageId, input.body) @@ -161,7 +118,7 @@ export async function performSend( ) } catch (error) { // A failed resolution must not strand a pending row; an unknown result is - // explicitly replayable and keeps tentative capacity for that retry. + // explicitly replayable. try { await ctx.journal.resolveDispatch({ clientMessageId: input.clientMessageId, @@ -171,34 +128,11 @@ export async function performSend( recovered: true }) } catch { - await ctx.journal.releaseLifecycleCapacity(dispatchReservationId(input.clientMessageId)) + // Nothing further to record; the pending row is settled on the next attach. } ctx.publish() throw error } - if (outcome.state === 'accepted') { - // Codex publishes its running lifecycle row under the legacy turn identity, - // while the dispatch response identifies the user's message item. Bind the - // tentative turn reservation to the lifecycle identity so a response that - // wins the race with turn/started cannot strand that row at the quota edge. - const reservationTarget = - outcome.providerIdentity.provider === 'codex' - ? { - provider: 'legacy' as const, - agent: 'codex' as const, - sessionId: ctx.sessionId, - recordId: `turn-lifecycle:${outcome.providerIdentity.turnId}` - } - : outcome.providerIdentity - await ctx.journal.transferLifecycleCapacity( - tentativeTurnReservationId(input.clientMessageId), - lifecycleReservationIdForItem( - ctx.journal.canonicalItemId(agentJournalItemKey(reservationTarget)) - ) - ) - } else if (outcome.state === 'rejected') { - await ctx.journal.releaseLifecycleCapacity(tentativeTurnReservationId(input.clientMessageId)) - } ctx.publish() const submission = ctx.journal diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts index f1b127c3fd4..fcd5625a48a 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-wire-admission.test.ts @@ -9,8 +9,13 @@ import type { import type { AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' import { REMOTE_RUNTIME_MAX_OUTBOUND_JSON_BYTES } from '../../../shared/remote-runtime-memory-limits' import { mobileE2EETextPayloadAdmissionBytes } from '../../runtime/rpc/mobile-e2ee-outbound-admission' +import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' +import { openJournalDatabase } from '../agent-session-journal/journal-database' +import { journalDatabaseFile } from '../agent-session-journal/journal-paths' +import { insertJournalRow } from '../agent-session-journal/journal-row-table' +import type { JournalRow } from '../agent-session-journal/journal-row-schema' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { readAgentSessionHistory } from './agent-session-history-page' import { AgentSessionSubscribers } from './structured-agent-session-subscribers' @@ -19,10 +24,11 @@ const LARGE_TEXT = 'x'.repeat(250 * 1024) let root: string let journal: AgentSessionJournal +const journals = createTrackedJournalOpener() beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-wire-admission-')) - journal = await openAgentSessionJournal({ + journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', @@ -30,8 +36,7 @@ beforeEach(async () => { agent: 'codex', providerHandle: { kind: 'codex', threadId: 'thread-1' } }, - journalDir: root, - autoCompact: false + journalDir: root }) for (let ordinal = 1; ordinal <= 20; ordinal += 1) { await journal.appendItem(item(ordinal), body(`${ordinal}:${LARGE_TEXT}`), { fence: 1 }) @@ -39,6 +44,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -91,23 +97,16 @@ describe('structured agent-session outbound admission', () => { expect(epochHistory).toMatchObject({ ok: false, reset: 'epoch_changed' }) expectAdmitted(epochHistory) - await journal.compact(Date.now() + 1, { minTailRows: 0, retainTailMs: 0 }) - const compactedReset: AgentSessionSubscribeEvent[] = [] - subscribers.open({ - id: 'compacted', - sessionId: SESSION, - journal, - fence: 2, - cursor: { epoch: journal.epoch, sequence: 0 }, - emit: (event) => compactedReset.push(event) - }) - expect(compactedReset[0]).toMatchObject({ type: 'reset', reset: 'cursor_compacted' }) - expectAdmitted(compactedReset[0]) - - const history = readAgentSessionHistory(journal, { + // The store can no longer produce a `cursor_compacted` reset — with no row + // shedding inside an epoch, `oldestSequence` is always 1. The reset reason + // stays in the wire vocabulary through the over-budget page path, which is + // where this file's subject — is such a frame admitted outbound? — now lives. + const cursorBefore = journal.cursor() + const overBudget = await reopenWithOversizedRemoval(cursorBefore.sequence) + const history = readAgentSessionHistory(overBudget, { sessionId: SESSION, direction: 'after', - cursor: { epoch: journal.epoch, sequence: 0 } + cursor: cursorBefore }) expect(history).toMatchObject({ ok: false, reset: 'cursor_compacted' }) expectAdmitted(history) @@ -156,3 +155,42 @@ function item(ordinal: number): AgentJournalItemIdentity { function body(text: string): AgentJournalItemBody { return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text }] } } + +/** Stages a pre-bounding oversized removal id — the one remaining producer of a + * `cursor_compacted` reset — straight into the session database. */ +async function reopenWithOversizedRemoval(afterSequence: number): Promise { + const hugeItemId = `codex:thread-1:${'h'.repeat(5 * 1024 * 1024)}:1` + const base = { v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, epoch: journal.epoch, fence: 1, ts: 1 } + const rows: JournalRow[] = [ + { + ...base, + kind: 'item', + itemId: hugeItemId, + revision: 1, + seq: afterSequence + 1, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'big' }] } + }, + { ...base, kind: 'tombstone', itemId: hugeItemId, revision: 2, seq: afterSequence + 2 } + ] + await journal.close() + const opened = openJournalDatabase(journalDatabaseFile(root)) + try { + opened.db.exec('BEGIN IMMEDIATE') + for (const row of rows) { + insertJournalRow(opened.db, SESSION, row) + } + opened.db.exec('COMMIT') + } finally { + opened.db.close() + } + return journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + journalDir: root + }) +} diff --git a/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts b/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts index 956e8a0eaa8..6e4cf64b5bf 100644 --- a/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-tui-transcript-catchup.test.ts @@ -3,9 +3,11 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' -import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { StructuredTuiTranscriptCatchup } from './structured-tui-transcript-catchup' +const journals = createTrackedJournalOpener() + const NOW = 1_800_000_000_000 const SESSION = 'session-catchup' const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' @@ -27,6 +29,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await rm(root, { recursive: true, force: true }) }) @@ -83,7 +86,7 @@ async function createCatchupFixture() { }, now: NOW }) - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity: { sessionId: SESSION, workspaceId: 'workspace-1', diff --git a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts index d287838a13c..e389b30aba2 100644 --- a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts +++ b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.test.ts @@ -8,12 +8,7 @@ describe('unhandled provider frame journal fallback', () => { 'future-provider', 'notification:new/event', { body: 'abcdefghij' }, - { - inlineHeadBytes: 8, - maxSessionBytes: 1024, - maxAppendsPerWindow: 10, - appendWindowMs: 1000 - } + { inlineHeadBytes: 8 } ) expect(item).not.toBeNull() @@ -32,12 +27,6 @@ describe('unhandled provider frame journal fallback', () => { expect( Buffer.byteLength(item.body.providerFrame?.payload.head ?? '', 'utf8') ).toBeLessThanOrEqual(8) - expect(item.blobs).toEqual([ - { - digest: item.body.providerFrame?.payload.digest, - payload: '{"body":"abcdefghij"}' - } - ]) }) it('turns an unserializable message-shaped payload into an explicit visible value', () => { @@ -198,12 +187,7 @@ describe('unhandled provider frame journal fallback', () => { 'codex', 'notification:warning', { message }, - { - inlineHeadBytes: 8, - maxSessionBytes: 1024, - maxAppendsPerWindow: 10, - appendWindowMs: 1000 - } + { inlineHeadBytes: 8 } ) expect(row?.body.text).toContain('abcdefgh') diff --git a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts index b4651cfc952..804d2900a22 100644 --- a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts +++ b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts @@ -9,7 +9,6 @@ import { classifyProviderFrame } from './provider-frame-disposition' export type UnhandledProviderFrameJournalItem = { body: AgentJournalStatusItem - blobs: { digest: string; payload: string }[] /** Why the frame surfaced. Error frames are exempt from generic-row caps. */ classification: 'timeline-substantive' | 'error-surface' } @@ -98,7 +97,6 @@ export function unhandledProviderFrameJournalItem( text: display?.text ?? `${provider} · ${kind}`, providerFrame: { provider, kind, payload: bounded } }, - blobs: bounded.truncated ? [{ digest: bounded.digest, payload: serialized }] : [], classification: classification === 'error-surface' ? 'error-surface' : 'timeline-substantive' } } diff --git a/src/main/orca-chromium-process-pids.ts b/src/main/orca-chromium-process-pids.ts index f22babc6921..b2613e42b79 100644 --- a/src/main/orca-chromium-process-pids.ts +++ b/src/main/orca-chromium-process-pids.ts @@ -1,4 +1,5 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' +import { recordCoalescedDurableCrashBreadcrumb } from './crash-reporting/durable-crash-breadcrumb' /** * PIDs of Orca's own Chromium processes — browser, renderers, GPU, utilities. @@ -11,6 +12,14 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' * Empty on a Node host and empty on failure: that is "no refusal proven", never * "safe to kill" — callers must keep every other guard they already have. * + * Why failure stays open rather than refusing everything: a refusal is not free. + * `terminateWindowsProcessTree` resolves without killing, and + * `killSourceControlAgentProcess` returns that straight to a caller that then + * releases the managed-home lock, so failing closed would trade one unreadable + * metrics table for every PTY, git, codex and notebook tree in main leaking at + * once. The `own_chromium_pids_unreadable` crumb is the price of that choice: + * without it a throw is byte-identical to "no Chromium on this host". + * * Host coverage: only Electron main installs a Chromium-backed AppEnvironment * (main-process-preflight). The standalone daemon installs none and `orcad` * installs a Node one whose `getAppMetrics()` is `[]`, so this set is empty in @@ -30,7 +39,26 @@ export function readOrcaChromiumProcessPids(): ReadonlySet { .map((metric) => metric.pid) .filter((pid) => Number.isInteger(pid) && pid > 0) return new Set(pids) - } catch { + } catch (error) { + recordUnreadableOwnChromiumMetrics(error) return new Set() } } + +// Why coalesced: the gate reads this set on every tree kill, so a persistently +// broken metrics table would otherwise flood the 30-slot ring it shares. +const UNREADABLE_METRICS_COALESCE_MS = 60_000 + +function recordUnreadableOwnChromiumMetrics(error: unknown): void { + try { + recordCoalescedDurableCrashBreadcrumb({ + name: 'own_chromium_pids_unreadable', + data: { cause: error instanceof Error ? error.message : String(error) }, + coalesceKey: 'own-chromium-pids-unreadable', + minIntervalMs: UNREADABLE_METRICS_COALESCE_MS + }) + } catch { + // Diagnostics must never turn an admitted kill into a thrown one: callers + // read this set outside their own try. + } +} diff --git a/src/main/orca-profiles/profile-cloud-client.ts b/src/main/orca-profiles/profile-cloud-client.ts index e7657bbdb87..5893c8d109a 100644 --- a/src/main/orca-profiles/profile-cloud-client.ts +++ b/src/main/orca-profiles/profile-cloud-client.ts @@ -159,19 +159,37 @@ function normalizeSessionResponse(value: unknown): OrcaCloudSessionExchangeRespo const CLOUD_REQUEST_TIMEOUT_MS = 30_000 -async function postJson(url: string, body: unknown, accessToken?: string): Promise { +// Why: refresh tokens rotate, so an aborted refresh is ambiguous — the server +// may have rotated ours before the reply was lost, and the only recovery is a +// replay the server reads as reuse. One long attempt beats a short attempt plus +// a replayed retry. +const CLOUD_REFRESH_TIMEOUT_MS = 60_000 + +type PostJsonOptions = { + accessToken?: string + timeoutMs?: number +} + +// Only a status line proves the server rejected the request without consuming +// what was in it. Everything else — an abort, a dropped socket, a 200 we could +// not parse — leaves a rotating credential possibly already spent. +export function isAmbiguousCloudRequestFailure(error: unknown): boolean { + return !(error instanceof OrcaCloudRequestError) +} + +async function postJson(url: string, body: unknown, options?: PostJsonOptions): Promise { const response = await fetch(url, { method: 'POST', headers: { 'content-type': 'application/json', - ...(accessToken ? { authorization: `Bearer ${accessToken}` } : {}) + ...(options?.accessToken ? { authorization: `Bearer ${options.accessToken}` } : {}) }, body: JSON.stringify(body), // Why: these are fixed first-party token endpoints; following a redirect // would re-send refresh tokens/code verifiers to another origin, and a // stalled server must not hang the renderer's awaited IPC call forever. redirect: 'error', - signal: AbortSignal.timeout(CLOUD_REQUEST_TIMEOUT_MS) + signal: AbortSignal.timeout(options?.timeoutMs ?? CLOUD_REQUEST_TIMEOUT_MS) }) if (!response.ok) { await cancelUnreadResponseBody(response) @@ -204,7 +222,7 @@ export async function refreshOrcaCloudCapabilities( cloud?: unknown organizations?: unknown capabilities: unknown - }>(config.capabilitiesEndpoint, {}, session.accessToken) + }>(config.capabilitiesEndpoint, {}, { accessToken: session.accessToken }) return { cloud: response.cloud === undefined ? undefined : normalizeCloudSummary(response.cloud), organizations: normalizeOrganizations(response.organizations), @@ -217,9 +235,11 @@ export async function refreshOrcaCloudSession( session: OrcaCloudSession ): Promise { return normalizeSessionResponse( - await postJson(config.refreshEndpoint, { - refreshToken: session.refreshToken - }) + await postJson( + config.refreshEndpoint, + { refreshToken: session.refreshToken }, + { timeoutMs: CLOUD_REFRESH_TIMEOUT_MS } + ) ) } @@ -235,7 +255,7 @@ export async function createOrcaCloudProfile( orgId: args.orgId, name: args.name }, - session.accessToken + { accessToken: session.accessToken } ) ) } @@ -249,7 +269,7 @@ export async function selectOrcaCloudOrg( cloud: unknown organizations?: unknown capabilities: unknown - }>(config.orgEndpoint, { orgId }, session.accessToken) + }>(config.orgEndpoint, { orgId }, { accessToken: session.accessToken }) return { cloud: normalizeCloudSummary(response.cloud), organizations: normalizeOrganizations(response.organizations), @@ -261,5 +281,9 @@ export async function revokeOrcaCloudSession( config: OrcaCloudAuthConfig, session: OrcaCloudSession ): Promise { - await postJson(config.logoutEndpoint, { refreshToken: session.refreshToken }, session.accessToken) + await postJson( + config.logoutEndpoint, + { refreshToken: session.refreshToken }, + { accessToken: session.accessToken } + ) } diff --git a/src/main/orca-profiles/profile-cloud-refresh-replay-guard.ts b/src/main/orca-profiles/profile-cloud-refresh-replay-guard.ts new file mode 100644 index 00000000000..d5c22f6b527 --- /dev/null +++ b/src/main/orca-profiles/profile-cloud-refresh-replay-guard.ts @@ -0,0 +1,55 @@ +// Refresh tokens whose server-side fate is unknown: the POST left the client but +// no status came back, so the server may already have rotated the token before +// the reply was lost. Sending it again reads as reuse and revokes the whole +// token family — on 2026-09-04 that turned one slow refresh endpoint into 21,605 +// sign-outs, because every caller's retry loop replayed the same stored token. + +// A replay this soon after the ambiguous attempt is a retry loop, not a person +// asking again; holding it back keeps one lost reply from becoming a storm. +export const AMBIGUOUS_REFRESH_REPLAY_DELAY_MS = 30_000 + +type AmbiguousRefreshAttempt = { + refreshToken: string + attemptedAt: number +} + +const ambiguousRefreshAttempts = new Map() + +export class AmbiguousRefreshReplayBlockedError extends Error { + constructor() { + super('orca_cloud_refresh_replay_blocked') + this.name = 'AmbiguousRefreshReplayBlockedError' + } +} + +export function recordAmbiguousRefreshAttempt( + key: string, + refreshToken: string, + now = Date.now() +): void { + ambiguousRefreshAttempts.set(key, { refreshToken, attemptedAt: now }) +} + +// Call once the token's fate is known: it rotated, or the session it belonged to +// is gone. Leaving the record would mislabel a later, unrelated 401. +export function forgetAmbiguousRefreshAttempt(key: string): void { + ambiguousRefreshAttempts.delete(key) +} + +export function wasRefreshTokenAmbiguouslyAttempted(key: string, refreshToken: string): boolean { + return ambiguousRefreshAttempts.get(key)?.refreshToken === refreshToken +} + +export function blocksAmbiguousRefreshReplay( + key: string, + refreshToken: string, + now = Date.now() +): boolean { + const attempt = ambiguousRefreshAttempts.get(key) + if (!attempt || attempt.refreshToken !== refreshToken) { + return false + } + // Why bounded rather than permanent: the token is only *possibly* spent. A + // permanent block would sign out every desktop whose refresh merely timed out. + return now - attempt.attemptedAt < AMBIGUOUS_REFRESH_REPLAY_DELAY_MS +} diff --git a/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts b/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts index 9db1d830986..53b26a75595 100644 --- a/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts +++ b/src/main/orca-profiles/profile-cloud-service-auth-retry.test.ts @@ -53,6 +53,7 @@ vi.mock('./profile-cloud-pkce', () => ({ vi.mock('./profile-cloud-client', () => ({ OrcaCloudRequestError: OrcaCloudRequestErrorMock, + isAmbiguousCloudRequestFailure: (error: unknown) => !(error instanceof OrcaCloudRequestErrorMock), createOrcaCloudProfile: createOrcaCloudProfileMock, exchangeOrcaCloudAuthCode: exchangeOrcaCloudAuthCodeMock, refreshOrcaCloudCapabilities: refreshOrcaCloudCapabilitiesMock, diff --git a/src/main/orca-profiles/profile-cloud-service-refresh.test.ts b/src/main/orca-profiles/profile-cloud-service-refresh.test.ts index 075573eea3a..877c74c3c46 100644 --- a/src/main/orca-profiles/profile-cloud-service-refresh.test.ts +++ b/src/main/orca-profiles/profile-cloud-service-refresh.test.ts @@ -51,6 +51,7 @@ vi.mock('./profile-cloud-pkce', () => ({ vi.mock('./profile-cloud-client', () => ({ OrcaCloudRequestError: OrcaCloudRequestErrorMock, + isAmbiguousCloudRequestFailure: (error: unknown) => !(error instanceof OrcaCloudRequestErrorMock), createOrcaCloudProfile: createOrcaCloudProfileMock, exchangeOrcaCloudAuthCode: exchangeOrcaCloudAuthCodeMock, refreshOrcaCloudCapabilities: refreshOrcaCloudCapabilitiesMock, diff --git a/src/main/orca-profiles/profile-cloud-session-invalidation.ts b/src/main/orca-profiles/profile-cloud-session-invalidation.ts new file mode 100644 index 00000000000..a9415e28ac6 --- /dev/null +++ b/src/main/orca-profiles/profile-cloud-session-invalidation.ts @@ -0,0 +1,30 @@ +type OrcaCloudSessionInvalidationListener = () => void + +const listeners = new Set() + +/** + * Fires when an auth failure (revoked or rotated-away refresh token) clears a + * stored cloud session. Never fires for an explicit user sign-out, which already + * hands the fresh auth status back to its caller. + */ +export function onOrcaCloudSessionInvalidated( + listener: OrcaCloudSessionInvalidationListener +): () => void { + listeners.add(listener) + return () => { + listeners.delete(listener) + } +} + +export function emitOrcaCloudSessionInvalidated(): void { + for (const listener of listeners) { + try { + listener() + } catch (error) { + console.warn( + '[orca-profiles] Cloud session invalidation listener failed:', + error instanceof Error ? error.message : String(error) + ) + } + } +} diff --git a/src/main/orca-profiles/profile-cloud-session-refresh.test.ts b/src/main/orca-profiles/profile-cloud-session-refresh.test.ts index 42b665d4e35..2aa3dc0266e 100644 --- a/src/main/orca-profiles/profile-cloud-session-refresh.test.ts +++ b/src/main/orca-profiles/profile-cloud-session-refresh.test.ts @@ -36,6 +36,9 @@ vi.mock('./profile-cloud-client', async (importOriginal) => { vi.mock('./profile-cloud-index', () => ({ linkOrcaProfileToCloud: linkMock })) import { readFreshOrcaCloudSession } from './profile-cloud-session-refresh' +import { OrcaCloudRequestError } from './profile-cloud-client' +import { onOrcaCloudSessionInvalidated } from './profile-cloud-session-invalidation' +import { forgetAmbiguousRefreshAttempt } from './profile-cloud-refresh-replay-guard' const config = {} as OrcaCloudAuthConfig const active = { @@ -59,6 +62,8 @@ const staleSession = { describe('profile cloud session refresh', () => { beforeEach(() => { vi.clearAllMocks() + vi.restoreAllMocks() + forgetAmbiguousRefreshAttempt('/data\0profile-1') saveIfCurrentMock.mockReturnValue('memory-only') readMock.mockReturnValue({ status: 'found', session: staleSession, persistence: 'memory-only' }) }) @@ -110,4 +115,166 @@ describe('profile cloud session refresh', () => { expect(saveIfCurrentMock).toHaveBeenCalledTimes(1) expect(linkMock).toHaveBeenCalledTimes(1) }) + + it('notifies subscribers when an auth failure clears the stored session', async () => { + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(clearMock).toHaveBeenCalledTimes(1) + expect(invalidated).toHaveBeenCalledTimes(1) + unsubscribe() + }) + + it('stays silent when a concurrent rotation already replaced the failed session', async () => { + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + readMock.mockReturnValueOnce({ + status: 'found', + session: staleSession, + persistence: 'memory-only' + }) + readMock.mockReturnValueOnce({ + status: 'found', + session: staleSession, + persistence: 'memory-only' + }) + readMock.mockReturnValue({ + status: 'found', + session: { ...staleSession, refreshToken: 'rotated-refresh' }, + persistence: 'memory-only' + }) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(clearMock).not.toHaveBeenCalled() + expect(invalidated).not.toHaveBeenCalled() + unsubscribe() + }) +}) + +describe('refresh-token replay after an ambiguous attempt', () => { + const timeout = (): Error => + Object.assign(new Error('The operation timed out.'), { name: 'TimeoutError' }) + const rotatedResponse = { + accessToken: 'new-access', + refreshToken: 'new-refresh', + expiresAt: 4_000_000, + organizations: [], + capabilities: { flags: { 'relay.use': true }, refreshedAt: 2 }, + cloud: { userId: 'user-1', cloudProfileId: 'cloud-profile-1', activeOrgId: 'org-1' } + } + + beforeEach(() => { + vi.clearAllMocks() + vi.restoreAllMocks() + forgetAmbiguousRefreshAttempt('/data\0profile-1') + saveIfCurrentMock.mockReturnValue('memory-only') + readMock.mockReturnValue({ status: 'found', session: staleSession, persistence: 'memory-only' }) + }) + + it('never resends a refresh token whose attempt timed out', async () => { + refreshMock.mockRejectedValue(timeout()) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'The operation timed out.' + ) + expect(refreshMock).toHaveBeenCalledTimes(1) + + // The retry loop above this module re-enters immediately; it must not turn + // one lost reply into a second POST of the same token. + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'orca_cloud_refresh_replay_blocked' + ) + expect(refreshMock).toHaveBeenCalledTimes(1) + expect(clearMock).not.toHaveBeenCalled() + }) + + it('adopts the stored session when a timed-out attempt was rotated elsewhere', async () => { + const rotated = { ...staleSession, refreshToken: 'rotated-refresh', expiresAt: 4_000_000 } + refreshMock.mockRejectedValue(timeout()) + readMock + .mockReturnValueOnce({ status: 'found', session: staleSession, persistence: 'memory-only' }) + .mockReturnValueOnce({ status: 'found', session: staleSession, persistence: 'memory-only' }) + .mockReturnValue({ status: 'found', session: rotated, persistence: 'memory-only' }) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'found', + session: rotated + }) + expect(refreshMock).toHaveBeenCalledTimes(1) + expect(saveIfCurrentMock).not.toHaveBeenCalled() + }) + + it('retries once after a definitive 5xx, which cannot have rotated the token', async () => { + refreshMock + .mockRejectedValueOnce(new OrcaCloudRequestError(503)) + .mockResolvedValueOnce(rotatedResponse) + + const result = await readFreshOrcaCloudSession(config, active, '/data') + + expect(refreshMock).toHaveBeenCalledTimes(2) + expect(refreshMock).toHaveBeenNthCalledWith(2, config, staleSession) + expect(result).toEqual({ + status: 'found', + session: expect.objectContaining({ + accessToken: 'new-access', + refreshToken: 'new-refresh' + }) + }) + }) + + it('gives a definitive 5xx exactly one retry', async () => { + refreshMock.mockRejectedValue(new OrcaCloudRequestError(503)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'orca_cloud_request_failed_503' + ) + expect(refreshMock).toHaveBeenCalledTimes(2) + expect(clearMock).not.toHaveBeenCalled() + }) + + it('marks a 401 that follows an ambiguous attempt as a possible self-replay', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const now = vi.spyOn(Date, 'now').mockReturnValue(1_000_000) + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValueOnce(timeout()) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).rejects.toThrow( + 'The operation timed out.' + ) + + now.mockReturnValue(1_000_000 + 31_000) + refreshMock.mockRejectedValueOnce(new OrcaCloudRequestError(401)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(refreshMock).toHaveBeenCalledTimes(2) + expect(warn.mock.calls.flat().join(' ')).toContain('orca_cloud_refresh_possible_replay') + expect(clearMock).toHaveBeenCalledTimes(1) + expect(invalidated).toHaveBeenCalledTimes(1) + unsubscribe() + }) + + it('does not mark a 401 that follows no ambiguous attempt', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(warn.mock.calls.flat().join(' ')).not.toContain('orca_cloud_refresh_possible_replay') + expect(clearMock).toHaveBeenCalledTimes(1) + }) }) diff --git a/src/main/orca-profiles/profile-cloud-session-refresh.ts b/src/main/orca-profiles/profile-cloud-session-refresh.ts index ced48a16e7b..221b4908bee 100644 --- a/src/main/orca-profiles/profile-cloud-session-refresh.ts +++ b/src/main/orca-profiles/profile-cloud-session-refresh.ts @@ -6,13 +6,26 @@ import { readOrcaCloudSession, saveOrcaCloudSessionIfCurrent } from './profile-cloud-session-store' -import { OrcaCloudRequestError, refreshOrcaCloudSession } from './profile-cloud-client' +import { + isAmbiguousCloudRequestFailure, + OrcaCloudRequestError, + refreshOrcaCloudSession +} from './profile-cloud-client' import { linkOrcaProfileToCloud } from './profile-cloud-index' +import type { OrcaCloudSessionExchangeResponse } from './profile-cloud-session-exchange' +import { + AmbiguousRefreshReplayBlockedError, + blocksAmbiguousRefreshReplay, + forgetAmbiguousRefreshAttempt, + recordAmbiguousRefreshAttempt, + wasRefreshTokenAmbiguouslyAttempted +} from './profile-cloud-refresh-replay-guard' import { captureCloudSessionMutation, cloudSessionIdentity, tombstoneCloudSession } from './profile-cloud-session-mutation' +import { emitOrcaCloudSessionInvalidated } from './profile-cloud-session-invalidation' const CLOUD_SESSION_REFRESH_SKEW_MS = 60_000 @@ -66,6 +79,76 @@ function clearCloudSessionIfUnchanged( ) } clearOrcaCloudSession(profileId, userDataPath) + forgetAmbiguousRefreshAttempt(cloudSessionRefreshKey(profileId, userDataPath)) + // Why: the renderer cached auth status at startup; without this it keeps + // showing "Connected" until the app restarts. + emitOrcaCloudSessionInvalidated() +} + +// Why: support cannot otherwise tell a genuine revocation from a sign-out we +// caused ourselves by resending a refresh token whose first attempt never +// answered. Never log the token itself. +function warnIfPossibleRefreshReplay( + profileId: string, + userDataPath: string, + failed: OrcaCloudSession, + error: unknown +): void { + if (!(error instanceof OrcaCloudRequestError) || error.statusCode !== 401) { + return + } + const key = cloudSessionRefreshKey(profileId, userDataPath) + if (!wasRefreshTokenAmbiguouslyAttempted(key, failed.refreshToken)) { + return + } + console.warn( + '[orca-cloud] orca_cloud_refresh_possible_replay: refresh rejected 401 for a token whose earlier attempt never answered' + ) +} + +type CloudSessionRefreshAttempt = + | { status: 'refreshed'; response: OrcaCloudSessionExchangeResponse } + | { status: 'rotated-elsewhere'; session: OrcaCloudSession } + +function isRetryableCloudRefreshRejection(error: unknown): boolean { + return error instanceof OrcaCloudRequestError && error.statusCode >= 500 +} + +async function attemptCloudSessionRefresh( + key: string, + config: OrcaCloudAuthConfig, + active: ActiveOrcaProfileState, + userDataPath: string, + session: OrcaCloudSession +): Promise { + for (let attempt = 0; ; attempt++) { + try { + const response = await refreshOrcaCloudSession(config, session) + forgetAmbiguousRefreshAttempt(key) + return { status: 'refreshed', response } + } catch (error) { + const ambiguous = isAmbiguousCloudRequestFailure(error) + if (ambiguous) { + recordAmbiguousRefreshAttempt(key, session.refreshToken) + } + // Only a status line proves the server rejected this token without + // rotating it, so a definitive 5xx is the only failure worth retrying. + const retryable = !ambiguous && attempt === 0 && isRetryableCloudRefreshRejection(error) + if (!ambiguous && !retryable) { + throw error + } + // Another caller may have rotated the stored session while this attempt + // was in flight; that result is the one to use, and the token this attempt + // held is no longer ours to send again. + const current = readOrcaCloudSession(active.profile.id, userDataPath) + if (current.status === 'found' && current.session.refreshToken !== session.refreshToken) { + return { status: 'rotated-elsewhere', session: current.session } + } + if (ambiguous) { + throw error + } + } + } } async function refreshStoredCloudSession( @@ -91,9 +174,16 @@ async function refreshStoredCloudSession( if (!active.profile.cloud) { throw new StaleCloudSessionMutationError() } + if (blocksAmbiguousRefreshReplay(key, session.refreshToken)) { + throw new AmbiguousRefreshReplayBlockedError() + } const expectedIdentity = cloudSessionIdentity(active.profile.id, active.profile.cloud) const snapshot = captureCloudSessionMutation(expectedIdentity, userDataPath) - const refreshed = await refreshOrcaCloudSession(config, session) + const attempt = await attemptCloudSessionRefresh(key, config, active, userDataPath, session) + if (attempt.status === 'rotated-elsewhere') { + return attempt.session + } + const refreshed = attempt.response const refreshedIdentity = cloudSessionIdentity(active.profile.id, refreshed.cloud) if ( refreshedIdentity.cloudUserId !== expectedIdentity.cloudUserId || @@ -144,6 +234,7 @@ export async function readFreshOrcaCloudSession( } } catch (error) { if (isOrcaCloudAuthFailure(error)) { + warnIfPossibleRefreshReplay(active.profile.id, userDataPath, session.session, error) clearCloudSessionIfUnchanged(active.profile.id, userDataPath, session.session, active) return { status: 'reconnect-required' } } @@ -164,6 +255,7 @@ export async function forceRefreshOrcaCloudSession( } } catch (error) { if (isOrcaCloudAuthFailure(error)) { + warnIfPossibleRefreshReplay(active.profile.id, userDataPath, session, error) clearCloudSessionIfUnchanged(active.profile.id, userDataPath, session, active) return { status: 'reconnect-required' } } diff --git a/src/main/own-chromium-tree-kill-guard.test.ts b/src/main/own-chromium-tree-kill-guard.test.ts index 7e98661aca6..bd3b1674e18 100644 --- a/src/main/own-chromium-tree-kill-guard.test.ts +++ b/src/main/own-chromium-tree-kill-guard.test.ts @@ -143,6 +143,40 @@ describe('refusing to tree-kill our own Chromium processes', () => { ) }) + /** + * Fail-open is the deliberate choice — see `orca-chromium-process-pids.ts` for + * why refusing everything is worse — so the crumb is the only thing that keeps + * an unreadable metrics table distinguishable from a host that has no Chromium. + */ + it('leaves proof, and still admits the kill, when the Chromium metrics cannot be read', () => { + appMetricsMock.mockImplementation(() => { + throw new Error('getAppMetrics unavailable') + }) + + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + // Coalesced: the gate reads this set on every kill, so a broken table must + // not evict the ring it shares with the refusal crumb. + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + expect( + admitSelfInitiatedTreeKill({ + pid: RENDERER_PID, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree' + }) + ).toBe(true) + + expect( + getCrashBreadcrumbSnapshot().filter( + (breadcrumb) => breadcrumb.name === 'own_chromium_pids_unreadable' + ) + ).toEqual([ + expect.objectContaining({ + name: 'own_chromium_pids_unreadable', + data: expect.objectContaining({ cause: 'getAppMetrics unavailable' }) + }) + ]) + }) + it('refuses an own-Chromium pid at the gate the account teardowns share', () => { expect( admitSelfInitiatedTreeKill({ diff --git a/src/main/persistence/host-qualified-worktree-meta.ts b/src/main/persistence/host-qualified-worktree-meta.ts index a9d1c0e8fb1..6267311f471 100644 --- a/src/main/persistence/host-qualified-worktree-meta.ts +++ b/src/main/persistence/host-qualified-worktree-meta.ts @@ -1,4 +1,5 @@ -import type { ExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' +import type { Repo } from '../../shared/repo-types' import type { WorktreeMeta } from '../../shared/worktree/meta-types' /** @@ -52,6 +53,26 @@ export function readWorktreeMetaForHost( return store.getWorktreeMetaForHost?.(worktreeId, executionHostId) } +/** + * The same two reads keyed off a repo row, so the resolve-then-read pair lives in one place. Four + * call sites had open-coded it identically, which is the shape that lets one copy drift from the + * rest (F7/F8). + */ +export function readAllWorktreeMetaForRepo( + store: Pick, + repo: Pick +): Record { + return readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) +} + +export function readWorktreeMetaForRepo( + store: Pick, + worktreeId: string, + repo: Pick +): WorktreeMeta | undefined { + return readWorktreeMetaForHost(store, worktreeId, getRepoExecutionHostId(repo)) +} + export function writeWorktreeMetaForHost( store: Pick, worktreeId: string, diff --git a/src/main/ports/local-workspace-platform-port-scanner.ts b/src/main/ports/local-workspace-platform-port-scanner.ts index 7e3b9941617..9a760870540 100644 --- a/src/main/ports/local-workspace-platform-port-scanner.ts +++ b/src/main/ports/local-workspace-platform-port-scanner.ts @@ -3,6 +3,7 @@ import { getProcessOutputFields } from '../../shared/process-output-field-scanne import { readWindowsProcessTable } from '../windows/windows-process-table' import { runPortScanCommand } from './port-scan-command-client' import { + partitionListenersNeedingMetadata, recallListenerMetadata, rememberListenerMetadata, shouldSkipMetadataCommands, @@ -19,25 +20,27 @@ import { export function parseLsofListeningOutput(output: string): RawListeningPort[] { const ports: RawListeningPort[] = [] - let currentPid: number | undefined - let currentProcessName: string | undefined + let pid: number | undefined + let processName: string | undefined + let socketId: string | undefined for (const line of output.split('\n')) { - if (!line) { - continue - } const tag = line[0] const value = line.slice(1) if (tag === 'p') { - const pid = Number.parseInt(value, 10) - currentPid = Number.isFinite(pid) ? pid : undefined - currentProcessName = undefined + const parsedPid = Number.parseInt(value, 10) + pid = Number.isFinite(parsedPid) ? parsedPid : undefined + processName = socketId = undefined } else if (tag === 'c') { - currentProcessName = value + processName = value + } else if (tag === 'f') { + socketId = undefined // each file record restarts; a socket without `d` must not inherit one + } else if (tag === 'd') { + socketId = value || undefined } else if (tag === 'n') { const parsed = parseAddressWithPort(value) if (parsed) { - ports.push({ pid: currentPid, processName: currentProcessName, ...parsed }) + ports.push({ pid, processName, ...(socketId ? { socketId } : {}), ...parsed }) } } } @@ -118,17 +121,23 @@ async function scanDarwinLsofPorts( '-iTCP', '-sTCP:LISTEN', '-F', - 'pcn' + // Why `d`: the socket's kernel identity is free on this command and lets the metadata cache + // tell a recycled pid on the same port apart from the process it remembered. + 'pcnd' ]) const ports = parseLsofListeningOutput(stdout) if (shouldSkipMetadataCommands(spawnMs, options)) { return { ports, metadataAvailable: false } } - const metadata = await loadDarwinProcessMetadata( - new Set(ports.flatMap((p) => (p.pid ? [p.pid] : []))) - ) + // Why: on a quiet machine the same servers keep listening, so the two metadata commands — the + // expensive half of the scan — would re-derive answers the last scan already has. + const { hydrated, pidsNeedingMetadata } = partitionListenersNeedingMetadata(ports, options) + if (pidsNeedingMetadata.size === 0) { + return { ports: hydrated, metadataAvailable: true } + } + const metadata = await loadDarwinProcessMetadata(pidsNeedingMetadata) return { - ports: ports.map((port) => ({ ...metadata.get(port.pid ?? -1), ...port })), + ports: hydrated.map((port) => ({ ...metadata.get(port.pid ?? -1), ...port })), metadataAvailable: true } } diff --git a/src/main/ports/local-workspace-port-scan-state.ts b/src/main/ports/local-workspace-port-scan-state.ts index 582e8b9731e..e4025944c79 100644 --- a/src/main/ports/local-workspace-port-scan-state.ts +++ b/src/main/ports/local-workspace-port-scan-state.ts @@ -7,10 +7,14 @@ import { } from './workspace-port-scan-timeout-backoff' const SLOW_SPAWN_SKIP_METADATA_MS = 2_000 +/** Re-probe a remembered listener every Nth scan (~5 min at 30s) so a cwd change cannot go stale forever. */ +const METADATA_REPROBE_INTERVAL_SCANS = 10 const commandTimeoutBackoff = new WorkspacePortScanTimeoutBackoff() let loggedWorkerUnavailable = false let skippedMetadataOnLastScan = false -let lastListenerMetadata = new Map() +let lastListenerMetadata = new Map() +let metadataScanSequence = 0 +let reusedListenerKeys = new Set() export type WorkspacePortScanOptions = { requireMetadata?: boolean @@ -21,6 +25,8 @@ export type RawListeningPort = { port: number pid?: number processName?: string + /** Kernel socket identity from `lsof -F d`; a new socket on the same pid:port gets a new one. */ + socketId?: string commandLine?: string cwd?: string } @@ -31,6 +37,11 @@ export type ProcessMetadata = { cwd?: string } +type RememberedListenerMetadata = ProcessMetadata & { + socketId?: string + probedAtScan: number +} + export type NormalizedWorkspacePortProbe = { worktree: WorkspacePortProbe normalizedPath: string @@ -58,6 +69,8 @@ export function resetWorkspacePortScanTimeoutBackoffForTests(): void { loggedWorkerUnavailable = false skippedMetadataOnLastScan = false lastListenerMetadata = new Map() + metadataScanSequence = 0 + reusedListenerKeys = new Set() } export function shouldSkipMetadataCommands( @@ -72,17 +85,98 @@ export function shouldSkipMetadataCommands( return skip } +// dedupeRawPorts already collapses rows by connectHost:port:pid, so this key is unique per row. function listenerMetadataKey(port: RawListeningPort): string { return `${port.pid ?? 'unknown'}:${port.host}:${port.port}` } +/** Call after partitionListenersNeedingMetadata: it consumes the reuse set that call recorded. */ export function rememberListenerMetadata(ports: readonly RawListeningPort[]): void { - lastListenerMetadata = new Map( - ports.map((port) => [ - listenerMetadataKey(port), - { processName: port.processName, commandLine: port.commandLine, cwd: port.cwd } - ]) - ) + const previous = lastListenerMetadata + lastListenerMetadata = new Map() + for (const port of ports) { + const key = listenerMetadataKey(port) + // Why: a reused entry keeps its original probe time so the staleness ceiling still expires it. + const probedAtScan = reusedListenerKeys.has(key) + ? (previous.get(key)?.probedAtScan ?? metadataScanSequence) + : metadataScanSequence + lastListenerMetadata.set(key, { + processName: port.processName, + socketId: port.socketId, + commandLine: port.commandLine, + cwd: port.cwd, + probedAtScan + }) + } + reusedListenerKeys = new Set() +} + +/** + * Split listeners into those a previous scan already resolved and the pids still needing a probe. + * + * Records which keys were reused; rememberListenerMetadata reads and clears that on the same scan. + * + * Why: the metadata commands are the expensive half of a macOS scan, and a listener that is still + * the same process on the same address has the same command line it had 30s ago. A remembered + * entry is only trusted when the free `lsof -F c` process name and `-F d` socket identity from + * this scan still match, so a recycled pid re-probes instead of inheriting the dead process's + * metadata; and every entry is re-probed after METADATA_REPROBE_INTERVAL_SCANS so a process that + * chdir'd while listening cannot keep a stale cwd forever. + */ +export function partitionListenersNeedingMetadata( + ports: readonly RawListeningPort[], + options: WorkspacePortScanOptions = {} +): { hydrated: RawListeningPort[]; pidsNeedingMetadata: Set } { + metadataScanSequence += 1 + reusedListenerKeys = new Set() + // Why requireMetadata opts out: that caller is the SIGTERM authorization re-scan, so it must + // attribute the owner from this cycle's probe and never from a remembered cwd. + if (options.requireMetadata) { + return { + hydrated: [...ports], + pidsNeedingMetadata: new Set(ports.flatMap((port) => (port.pid ? [port.pid] : []))) + } + } + const hydrated: RawListeningPort[] = [] + const pidsNeedingMetadata = new Set() + const reusableByPort = new Map() + for (const port of ports) { + const remembered = lastListenerMetadata.get(listenerMetadataKey(port)) + // Why require commandLine: a probe that returned nothing must not be cached as an answer. + if ( + remembered?.commandLine !== undefined && + remembered.processName === port.processName && + remembered.socketId === port.socketId && + metadataScanSequence - remembered.probedAtScan < METADATA_REPROBE_INTERVAL_SCANS && + port.pid !== undefined + ) { + reusableByPort.set(port, remembered) + continue + } + if (port.pid !== undefined) { + pidsNeedingMetadata.add(port.pid) + } + } + // Why the second pass: if any of a pid's sockets needs a probe, none of its sockets may be + // served from cache — otherwise one process reports a fresh cwd on one row and a remembered + // cwd on another, i.e. two different workspace attributions. + for (const port of ports) { + const remembered = + port.pid !== undefined && !pidsNeedingMetadata.has(port.pid) + ? reusableByPort.get(port) + : undefined + if (remembered) { + reusedListenerKeys.add(listenerMetadataKey(port)) + hydrated.push({ + ...port, + commandLine: port.commandLine ?? remembered.commandLine, + cwd: port.cwd ?? remembered.cwd + }) + continue + } + hydrated.push(port) + } + return { hydrated, pidsNeedingMetadata } } export function recallListenerMetadata(port: RawListeningPort): RawListeningPort { diff --git a/src/main/ports/local-workspace-port-scanner.test.ts b/src/main/ports/local-workspace-port-scanner.test.ts index be37ad60eb3..4f6820be0bd 100644 --- a/src/main/ports/local-workspace-port-scanner.test.ts +++ b/src/main/ports/local-workspace-port-scanner.test.ts @@ -50,6 +50,35 @@ describe('local workspace port scanner parsing', () => { ]) }) + it('keeps the socket identity from lsof -F d and tolerates its absence', () => { + const ports = parseLsofListeningOutput( + [ + 'p123', + 'cnode', + 'f18', + 'd0x469ca588d83e7924', + 'n127.0.0.1:5173', + 'f19', + 'n127.0.0.1:5174', + 'p456', + 'cnginx', + 'n*:8080' + ].join('\n') + ) + + expect(ports).toEqual([ + { + pid: 123, + processName: 'node', + socketId: '0x469ca588d83e7924', + host: '127.0.0.1', + port: 5173 + }, + { pid: 123, processName: 'node', host: '127.0.0.1', port: 5174 }, + { pid: 456, processName: 'nginx', host: '*', port: 8080 } + ]) + }) + it('parses multiple lsof listening ports for the same process', () => { const ports = parseLsofListeningOutput( ['p123', 'cnode', 'n127.0.0.1:5173', 'n127.0.0.1:55173'].join('\n') @@ -387,7 +416,19 @@ describe('scanWorkspacePorts with delayed process creation', () => { it('does not let a required-metadata scan reset the background skip parity', async () => { vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') - mockStalledDarwinScan() + // Why a fresh pid each cycle: a listener the previous scan already resolved is served from the + // remembered metadata, so a stable pid would hide whether this scan skipped the probe or not. + let listenerPid = 123 + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + listenerPid += 1 + return { stdout: `p${listenerPid}\ncnode\nn127.0.0.1:5173`, spawnMs: 4_200 } + } + if (command === 'lsof') { + return { stdout: [`p${listenerPid}`, 'n/repo'].join('\n'), spawnMs: 4_200 } + } + return { stdout: `${listenerPid} node /repo/server.js`, spawnMs: 4_200 } + }) await scanWorkspacePorts(worktrees, urlWatcherStub()) await scanWorkspacePorts(worktrees, urlWatcherStub(), { requireMetadata: true }) @@ -397,6 +438,168 @@ describe('scanWorkspacePorts with delayed process creation', () => { expect(runPortScanCommandMock).toHaveBeenCalledTimes(7) }) + it('serves an unchanged listener from remembered metadata instead of re-probing', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: LSOF_LISTEN_OUTPUT, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node /repo/server.js', spawnMs: 5 } + }) + + const first = await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + + const second = await scanWorkspacePorts(worktrees, urlWatcherStub()) + + // Only the listening scan itself runs; the two metadata commands are served from the cache. + expect(runPortScanCommandMock).toHaveBeenCalledTimes(4) + expect(second.ports).toEqual(first.ports) + }) + + it('re-probes every port of a pid when any one of them needs metadata', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let secondSocket = 'd0xbbbb' + let cwd = '/repo' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { + stdout: [ + 'p123', + 'cnode', + 'f10', + 'd0xaaaa', + 'n127.0.0.1:5173', + 'f11', + secondSocket, + 'n127.0.0.1:5174' + ].join('\n'), + spawnMs: 5 + } + } + if (command === 'lsof') { + return { stdout: ['p123', `n${cwd}`].join('\n'), spawnMs: 5 } + } + return { stdout: `123 node ${cwd}/server.js`, spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + + // One socket is replaced and the process has since moved, so this pid must be re-derived for + // BOTH its rows — serving one from cache would report two different workspaces for one process. + secondSocket = 'd0xcccc' + cwd = '/elsewhere' + const scan = await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(6) + expect(new Set(scan.ports.map((entry) => entry.kind)).size).toBe(1) + }) + + it('always probes metadata for a requireMetadata scan, even when the cache is warm', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: LSOF_LISTEN_OUTPUT, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node /repo/server.js', spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + await scanWorkspacePorts(worktrees, urlWatcherStub()) + // Warm cache: the background poll is served without the metadata commands. + expect(runPortScanCommandMock).toHaveBeenCalledTimes(4) + + // Why: this is the SIGTERM authorization re-scan; a remembered cwd must never authorize a kill. + await scanWorkspacePorts(worktrees, urlWatcherStub(), { requireMetadata: true }) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(7) + }) + + it('re-probes when a recycled pid is running a different process', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let processName = 'node' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: `p123\nc${processName}\nn127.0.0.1:5173`, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: `123 ${processName} /repo/server.js`, spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(3) + + // Same pid and address, different process: the remembered metadata must not be reused. + processName = 'python3' + await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(6) + }) + + it('re-probes when the same pid and name listen through a different socket', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let socketId = '0x1' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: `p123\ncnode\nf18\nd${socketId}\nn127.0.0.1:5173`, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', 'n/repo'].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node /repo/server.js', spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(runPortScanCommandMock).toHaveBeenCalledTimes(4) + + // Same pid, name and address but a new kernel socket: a restarted process, not the cached one. + socketId = '0x2' + await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(7) + }) + + it('re-probes a remembered listener every tenth scan so a changed cwd cannot stay stale', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + let cwd = '/repo' + runPortScanCommandMock.mockImplementation(async (command: string, args: string[]) => { + if (command === 'lsof' && args.includes('-iTCP')) { + return { stdout: LSOF_LISTEN_OUTPUT, spawnMs: 5 } + } + if (command === 'lsof') { + return { stdout: ['p123', `n${cwd}`].join('\n'), spawnMs: 5 } + } + return { stdout: '123 node server.js', spawnMs: 5 } + }) + + await scanWorkspacePorts(worktrees, urlWatcherStub()) + cwd = '/repo/worktrees/feature' + for (let scan = 2; scan <= 10; scan += 1) { + const cached = await scanWorkspacePorts(worktrees, urlWatcherStub()) + expect(cached.ports[0]).toMatchObject({ owner: { worktreeId: 'repo::/repo' } }) + } + // 3 for the first scan, then one listening command per cached scan. + expect(runPortScanCommandMock).toHaveBeenCalledTimes(12) + + const reprobed = await scanWorkspacePorts(worktrees, urlWatcherStub()) + + expect(runPortScanCommandMock).toHaveBeenCalledTimes(15) + expect(reprobed.ports[0]).toMatchObject({ + owner: { worktreeId: 'repo::/repo/worktrees/feature' } + }) + }) + // Regression for #11161 review: without carry-forward the panel moves every // workspace port into External on each skipped cycle. it('carries the previous cycle attribution through a skipped scan', async () => { diff --git a/src/main/project-runtime-git-options.ts b/src/main/project-runtime-git-options.ts index 808d31d5fcf..20aa0e9659a 100644 --- a/src/main/project-runtime-git-options.ts +++ b/src/main/project-runtime-git-options.ts @@ -102,7 +102,12 @@ export function getWorktreeMirrorDistro( store: ProjectRuntimeResolutionStore, repo: Repo ): string | undefined { - const projectRuntime = resolveLocalProjectRuntimeForRepo(store, repo) + return getWorktreeMirrorDistroForRuntime(resolveLocalProjectRuntimeForRepo(store, repo)) +} + +export function getWorktreeMirrorDistroForRuntime( + projectRuntime: ProjectExecutionRuntimeResolution | undefined +): string | undefined { if (!projectRuntime || projectRuntime.status !== 'resolved') { return undefined } diff --git a/src/main/providers/pty-process-inspection.ts b/src/main/providers/pty-process-inspection.ts index 6869899de8b..59d910b2238 100644 --- a/src/main/providers/pty-process-inspection.ts +++ b/src/main/providers/pty-process-inspection.ts @@ -11,14 +11,24 @@ export type PtyProcessInspection = TerminalProcessInspection type CompletionSensitivePtyProvider = IPtyProvider & { inspectProcess?: ( id: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: PtyProcessInspectionOptions ) => Promise } +/** + * `scanChildProcesses` marks a read whose answer decides something once, rather than a poll that + * self-corrects on its next tick. Only hosts where the child answer costs a process-table read + * act on it; everywhere else the answer was already captured. + */ +export type PtyProcessInspectionOptions = { + expectedIncarnationId?: PtyIncarnationId + scanChildProcesses?: boolean +} + export async function inspectPtyProviderProcess( provider: IPtyProvider, ptyId: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: PtyProcessInspectionOptions ): Promise { if (provider.hasPty?.(ptyId) === false) { throw new Error('terminal_gone') @@ -37,7 +47,7 @@ export async function inspectPtyProviderProcess( export async function inspectPtyProviderProcessForRenderer( provider: IPtyProvider, ptyId: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: PtyProcessInspectionOptions ): Promise { try { return await inspectPtyProviderProcess(provider, ptyId, options) diff --git a/src/main/providers/ssh-pty-provider-rpc-operations.ts b/src/main/providers/ssh-pty-provider-rpc-operations.ts index 71ddbefce97..2e273239cd4 100644 --- a/src/main/providers/ssh-pty-provider-rpc-operations.ts +++ b/src/main/providers/ssh-pty-provider-rpc-operations.ts @@ -56,13 +56,15 @@ export function createSshPtyProviderRpcOperations({ mux, toRelayPtyId }: SshPtyP // Guarded by ssh-pty-inspect-observation-identity.test.ts; #17525 removes the poll. inspectProcess: async ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise => { return (await mux.request('pty.inspectProcess', { id: toRelayPtyId(id), ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } - : {}) + : {}), + // Additive request member: an older relay ignores it and answers as it always did. + ...(options?.scanChildProcesses === true ? { scanChildProcesses: true } : {}) })) as PtyProcessInspection }, serialize: async (ids: string[]): Promise => { diff --git a/src/main/providers/ssh-pty-provider.ts b/src/main/providers/ssh-pty-provider.ts index 70c01b5a1c7..3d0c46d7a03 100644 --- a/src/main/providers/ssh-pty-provider.ts +++ b/src/main/providers/ssh-pty-provider.ts @@ -63,7 +63,7 @@ export class SshPtyProvider implements IPtyProvider { this.rpcOperations.getForegroundProcess(id) inspectProcess = ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise => this.rpcOperations.inspectProcess(id, options) serialize = (ids: string[]): Promise => this.rpcOperations.serialize(ids) revive = (state: string): Promise => this.rpcOperations.revive(state) diff --git a/src/main/pty/node-pty-self-exit-pseudoconsole-close.test.ts b/src/main/pty/node-pty-self-exit-pseudoconsole-close.test.ts new file mode 100644 index 00000000000..1f9cae275c3 --- /dev/null +++ b/src/main/pty/node-pty-self-exit-pseudoconsole-close.test.ts @@ -0,0 +1,180 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * A shell that exits by itself must still close its pseudoconsole. + * + * `ClosePseudoConsole` is the only thing that reaps a ConPTY's console host — + * Orca's own job-ownership patch says so, because `CreatePseudoConsole` spawns + * that host before the per-pty job exists and it is therefore not a job member. + * Upstream node-pty calls it from exactly one place, `PtyKill`, which begins by + * looking the baton up by id — and the exit watcher in `SetupExitCallback` + * erased the baton the moment the shell died. So on the self-exit path (typing + * `exit`, which is how panes usually close) that lookup missed, `PtyKill` did + * nothing at all, and the pseudoconsole was never closed. + * + * There is a SECOND, independent defect on the same path: the `useConptyDll` + * branch of `WindowsPtyAgent.kill()` disposed the conout worker only from an + * `_outSocket.on('data')` handler, and no more data arrives once the shell has + * gone — so that worker leaked too. The non-DLL branch beside it already + * disposed unconditionally. The desktop always sets `useConptyDll`, so it hit + * both; the relay sets neither and hit only the first. + * + * Measured on Windows 11 / awin, 20 cycles, handles bucketed by NT object type, + * totals before -> after: + * + * self-exit, relay spawn 225 -> 285 becomes 219 -> 219 FLAT + * self-exit, desktop spawn 239 -> 439 becomes 222 -> 222 FLAT + * explicit kill, relay spawn 225 -> 285 becomes 219 -> 219 FLAT + * explicit kill, desktop spawn 235 -> 395 becomes 219 -> 219 FLAT + * + * Neither fix alone is enough on the desktop: the pseudoconsole close is worth + * +1 Process +1 File per terminal, the dispose +2 Thread +4 File. + * + * WHY THIS IS A PATCH-CONTENT PIN AND NOT A BEHAVIOURAL TEST: the defect is + * only observable as a per-NT-type handle count, which needs + * `NtQuerySystemInformation(SystemExtendedHandleInformation)`. Nothing in the + * repo can read that, and the cheaper Windows-observable proxies do not + * discriminate — the console host process is reaped either way (the leak is a + * handle to an already-exited object, not an orphaned process), and the + * `\\.\pipe\conpty-*` entries disappear either way. Both were measured and + * rejected as assertions rather than assumed. So this pins the mechanism + * instead, which is the real risk: a future resync of the vendored patch + * silently dropping the hunk. + */ + +const PATCH = readFileSync(join(__dirname, '../../../config/patches/node-pty@1.1.0.patch'), 'utf8') + +/** + * Just the `PtyKill` hunk. Several markers below also occur in the `PtyConnect` + * hunk above it, and a bare `indexOf` on the whole patch silently matched the + * wrong one — an assertion that then held regardless of what `PtyKill` did. + */ +const ptyKillHunk = (() => { + // Anchored on the hunk header's function context rather than its line + // numbers, which shift whenever anything above it in the patch changes. + const header = /^@@ .* @@ static Napi::Value PtyKill\(.*$/m.exec(PATCH) + if (!header) { + throw new Error('no PtyKill hunk in config/patches/node-pty@1.1.0.patch') + } + const from = header.index + const next = PATCH.indexOf('\n@@ ', from + 1) + return PATCH.slice(from, next === -1 ? undefined : next) +})() + +/** + * `indexOf` that throws instead of returning -1. A missing marker must fail the + * assertion that depends on it, not quietly make a slice or comparison vacuous. + */ +function indexIn(haystack: string, marker: string): number { + const at = haystack.indexOf(marker) + if (at === -1) { + throw new Error(`marker not found in the PtyKill hunk: ${marker}`) + } + return at +} + +describe('node-pty patch: pseudoconsole close on the self-exit path', () => { + it('does not let the exit watcher free the baton while the close is still owed', () => { + // Pinned as one block: the erase must stay INSIDE the consoleClosed guard. + // Upstream ran it unconditionally, which is the line that caused the leak, + // and a resync that re-flattens this is the failure mode worth catching. + expect(PATCH).toContain( + [ + '+ baton->shellExited = true;', + '+ if (baton->consoleClosed) {', + '+ const bool removed = remove_pty_baton(baton->id);', + '+ assert(removed);', + '+ (void)removed;', + '+ }' + ].join('\n') + ) + }) + + it('closes the pseudoconsole from PtyKill even after the shell has exited', () => { + // hpc is copied out under the lock, so the close survives the baton's removal. + expect(PATCH).toContain('+ hpc = handle->hpc;') + expect(PATCH).toContain('+ pfnClosePseudoConsole(hpc);') + }) + + it('resolves the ConPTY DLL before it claims the close', () => { + // LoadConptyDll throws when conpty.dll is missing. Throwing after + // consoleClosed was set would strand the pseudoconsole for good: the retry + // finds the work claimed and does nothing. + // + // Anchored inside PtyKill, not by a bare indexOf: the identical line also + // appears in the PtyConnect hunk, earlier in the file, and matching that one + // made this assertion pass no matter where PtyKill resolved the DLL. + const dllResolve = indexIn( + ptyKillHunk, + '+ HANDLE hLibrary = LoadConptyDll(info, useConptyDll);' + ) + const claim = indexIn(ptyKillHunk, '+ handle->consoleClosed = true;') + expect(dllResolve).toBeLessThan(claim) + }) + + it('reaches hShell only under the null check the watcher can trip', () => { + // Pinned as one block. The watcher nulls hShell on exit, and upstream + // dereferenced it unconditionally; every remaining use — the duplication and + // the failure fallback below it — must stay inside this guard. + const start = indexIn(ptyKillHunk, '+ if (useConptyDll && handle->hShell != nullptr) {') + const end = indexIn(ptyKillHunk, '+ if (handle->shellExited) {') + const guarded = ptyKillHunk.slice(start, end) + expect(guarded).toContain('DuplicateHandle(GetCurrentProcess(), handle->hShell') + expect(guarded).toContain('TerminateProcess(handle->hShell, 1);') + // No ADDED line outside that guard may terminate through hShell. Removed + // (`-`) lines still carry upstream's unguarded call, which is the point. + const strayAdds = PATCH.replace(guarded, '') + .split('\n') + .filter((line) => line.startsWith('+') && line.includes('TerminateProcess(handle->hShell')) + expect(strayAdds).toEqual([]) + }) + + it('frees the baton from PtyKill when the shell has already exited', () => { + // The other half of the two-sided handshake. Without it a self-exit followed + // by kill() — the ordinary pane close — leaks one baton and one entry in the + // vector get_pty_baton scans linearly, forever. + expect(ptyKillHunk).toContain( + [ + '+ if (handle->shellExited) {', + '+ const bool removed = remove_pty_baton(id);', + '+ assert(removed);', + '+ (void)removed;' + ].join('\n') + ) + }) + + it('still kills the shell when DuplicateHandle fails', () => { + // A null hShellDup is indistinguishable from the self-exit case, so a + // swallowed failure would leave the shell running after its pane closed — + // a worse outcome than the leak this patch exists to fix. + expect(PATCH).toContain( + [ + '+ hShellDup = nullptr;', + '+ TerminateProcess(handle->hShell, 1);', + '+ }' + ].join('\n') + ) + }) + + it('keeps the close idempotent so a second kill cannot double-close', () => { + expect(PATCH).toContain('+ if (handle != nullptr && !handle->consoleClosed) {') + expect(PATCH).toContain('+ handle->consoleClosed = true;') + }) +}) + +describe('node-pty patch: conout worker disposal on the self-exit path', () => { + // The desktop's larger half: 8 of its 10 leaked handles per terminal. + it('disposes the conout worker unconditionally in the useConptyDll branch', () => { + expect(PATCH).toContain('+ this._conoutSocketWorker.dispose();') + // The data handler is what never fired once the shell had gone. + expect(PATCH).toContain("- this._outSocket.on('data', function () {") + expect(PATCH).toContain('- _this._conoutSocketWorker.dispose();') + }) + + it('applies the same change to the TypeScript source the patch also carries', () => { + expect(PATCH).toContain('+ this._conoutSocketWorker.dispose();') + expect(PATCH).toContain("- this._outSocket.on('data', () => {") + }) +}) diff --git a/src/main/repo-worktrees.ts b/src/main/repo-worktrees.ts index 228f303bfa2..f5d67523286 100644 --- a/src/main/repo-worktrees.ts +++ b/src/main/repo-worktrees.ts @@ -1,6 +1,11 @@ import type { Repo } from '../shared/repo-types' import type { GitWorktreeInfo } from '../shared/worktree/types' -import { listWorktreeGraph, listWorktrees, listWorktreesStrict } from './git/worktree' +import { + listWorktreeGraph, + listWorktrees, + listWorktreesSharedStrictAllowingTrueEmpty, + listWorktreesStrict +} from './git/worktree' import { isFolderRepo } from '../shared/repo-kind' import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' import { resolveGitRouteForHost } from './providers/execution-host-provider-dispatch' @@ -42,6 +47,29 @@ export function createFolderWorktree(repo: Repo): GitWorktreeInfo { export async function listRepoWorktrees( repo: Repo, options?: LocalRepoWorktreeListOptions +): Promise { + return listRoutedRepoWorktrees(repo, options, listWorktrees) +} + +/** + * The detected scan's listing: a Git or host failure rejects instead of softening to `[]`, so a + * failed scan cannot be published as an authoritative empty listing and prune the repo's worktrees + * (#1158's retention guard only fires when the listing admits it failed). + */ +export async function listRepoWorktreesForDetectedScan( + repo: Repo, + options?: LocalRepoWorktreeListOptions +): Promise { + return listRoutedRepoWorktrees(repo, options, listWorktreesSharedStrictAllowingTrueEmpty) +} + +async function listRoutedRepoWorktrees( + repo: Repo, + options: LocalRepoWorktreeListOptions | undefined, + listLocal: ( + repoPath: string, + options?: LocalRepoWorktreeListOptions + ) => Promise ): Promise { if (isFolderRepo(repo)) { return [createFolderWorktree(repo)] @@ -66,8 +94,8 @@ export async function listRepoWorktrees( return await route.provider.listWorktrees(repo.path) } return hasLocalRepoWorktreeListOptions(options) - ? await listWorktrees(repo.path, options) - : await listWorktrees(repo.path) + ? await listLocal(repo.path, options) + : await listLocal(repo.path) } /** diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index 0ed3700ba99..4466594b0dc 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -153,7 +153,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu async inspectTerminalProcess( terminalSelector: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise { const leaf = this.resolveLiveLeafForHandle(terminalSelector) if (!leaf?.ptyId || !this.ptyController) { diff --git a/src/main/runtime/orchestration/db/database-file-permissions.ts b/src/main/runtime/orchestration/db/database-file-permissions.ts index 1bbd640ba77..dd2e49e77a5 100644 --- a/src/main/runtime/orchestration/db/database-file-permissions.ts +++ b/src/main/runtime/orchestration/db/database-file-permissions.ts @@ -1,17 +1,5 @@ -import { chmodSync, existsSync } from 'node:fs' +import { hardenSqliteDatabaseFiles } from '../../../sqlite/harden-database-files' export function hardenOrchestrationDatabaseFiles(dbPath: (string & {}) | ':memory:'): void { - if (dbPath === ':memory:' || process.platform === 'win32') { - // Why: Windows protects these files through Orca's current-user-only userData DACL; POSIX mode bits are inert there. - return - } - for (const path of [dbPath, `${dbPath}-wal`, `${dbPath}-shm`]) { - try { - if (existsSync(path)) { - chmodSync(path, 0o600) - } - } catch { - // Why: best-effort — a mount that rejects chmod (SSHFS, some network shares) must not fail DB startup. - } - } + hardenSqliteDatabaseFiles(dbPath) } diff --git a/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts b/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts index 26f82410cfe..8e36605aeb5 100644 --- a/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts +++ b/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts @@ -131,6 +131,32 @@ describe('orchestration hot-path statement compilation', () => { } }) + // Why: getTask sits on the dispatch/lifecycle path and listTasks runs several times per + // coordinator tick, so a wildcard there recompiles on every call the same way the publish did. + it('compiles the task lookup and listing SQL exactly once across repeated calls', () => { + const db = openDatabase(':memory:') + seedDispatchedWorker(db) + const seeded = db.listTasks() + expect(seeded.length).toBeGreaterThan(0) + const taskId = seeded[0].id + + const compiled = trackCompiledSql(db) + for (let call = 0; call < 5; call += 1) { + db.getTask(taskId) + db.listTasks() + db.listTasks({ ready: true }) + db.listTasks({ status: 'pending' }) + db.listTasks({ runId: seeded[0].run_id }) + } + + const compilationsPerSql = new Map() + for (const sql of compiled) { + compilationsPerSql.set(sql, (compilationsPerSql.get(sql) ?? 0) + 1) + } + expect([...compilationsPerSql].filter(([, count]) => count > 1)).toEqual([]) + expect(compiled.filter((sql) => WILDCARD_PROJECTION.test(sql))).toEqual([]) + }) + // Why: `SELECT *` is what made these statements uncacheable, and a retained wildcard is the only // way node:sqlite could build a row from stale column names after another connection's ALTER. // Seeds on one connection and publishes on a second so every compilation here is hot-path SQL. diff --git a/src/main/runtime/orchestration/db/tasks/task-store.ts b/src/main/runtime/orchestration/db/tasks/task-store.ts index 69316765a74..8bcc74ca3e5 100644 --- a/src/main/runtime/orchestration/db/tasks/task-store.ts +++ b/src/main/runtime/orchestration/db/tasks/task-store.ts @@ -79,9 +79,13 @@ export function createTask( runId, depsJson ) - return this.db.prepare('SELECT * FROM tasks WHERE id = ?').get(id) as TaskRow + return this.db.prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE id = ?`).get(id) as TaskRow } +// Why wildcard-free: SyncDatabase refuses to cache any statement containing `*`, so a `SELECT *` +// here recompiles on every call — including the hot dispatch lookups and the coordinator poll. +const TASK_COLUMN_LIST = selectColumns(TASK_COLUMNS) + // Why: hoisted and wildcard-free so the per-publish lineage lookup hits the SyncDatabase statement cache. const TASK_RUNTIME_LINEAGE_SQL = `SELECT ${selectColumns(TASK_COLUMNS, 't')}, creator.id AS creator_dispatch_id, @@ -113,7 +117,9 @@ export function getTask( dispatchRunId?: string ): TaskRow | TaskRuntimeLineageRow | undefined { if (dispatchRunId === undefined) { - return this.db.prepare('SELECT * FROM tasks WHERE id = ?').get(id) as TaskRow | undefined + return this.db.prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE id = ?`).get(id) as + | TaskRow + | undefined } return this.db.prepare(TASK_RUNTIME_LINEAGE_SQL).get(dispatchRunId, id) as | TaskRuntimeLineageRow @@ -128,20 +134,26 @@ export function listTasks( const runParams: Database.BindValue[] = filter?.runId ? [filter.runId] : [] if (filter?.ready) { return this.db - .prepare(`SELECT * FROM tasks WHERE ${runWhere}status = 'ready' ORDER BY created_at`) + .prepare( + `SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE ${runWhere}status = 'ready' ORDER BY created_at` + ) .all(...runParams) as TaskRow[] } if (filter?.status) { return this.db - .prepare(`SELECT * FROM tasks WHERE ${runWhere}status = ? ORDER BY created_at`) + .prepare( + `SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE ${runWhere}status = ? ORDER BY created_at` + ) .all(...runParams, filter.status) as TaskRow[] } if (filter?.runId) { return this.db - .prepare('SELECT * FROM tasks WHERE run_id = ? ORDER BY created_at') + .prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE run_id = ? ORDER BY created_at`) .all(filter.runId) as TaskRow[] } - return this.db.prepare('SELECT * FROM tasks ORDER BY created_at').all() as TaskRow[] + return this.db + .prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks ORDER BY created_at`) + .all() as TaskRow[] } // Why: the correlated indexed lookup avoids materializing every retained Dispatch before filtering Tasks. @@ -195,7 +207,7 @@ export function listTasksWithDispatch( // Why: runs in the status-update transaction, so a completed task never leaves its ready children unpromoted. export function promoteReadyTasks(this: OrchestrationDb, completedTaskId: string): void { const candidates = this.db - .prepare("SELECT * FROM tasks WHERE status = 'pending'") + .prepare(`SELECT ${TASK_COLUMN_LIST} FROM tasks WHERE status = 'pending'`) .all() as TaskRow[] for (const task of candidates) { diff --git a/src/main/runtime/orchestration/worker-output-archive-bounding.test.ts b/src/main/runtime/orchestration/worker-output-archive-bounding.test.ts new file mode 100644 index 00000000000..a45cc6c0531 --- /dev/null +++ b/src/main/runtime/orchestration/worker-output-archive-bounding.test.ts @@ -0,0 +1,41 @@ +import { describe, expect, it } from 'vitest' +import { boundArchiveLines } from './worker-output-archive' + +const TERMINAL_ARCHIVE_MAX_CHARS = 262_144 + +function totalCost(lines: string[]): number { + return lines.reduce((sum, line) => sum + line.length + 1, 0) +} + +describe('boundArchiveLines', () => { + it('returns the original array untouched when the tail already fits', () => { + const lines = ['one', 'two', 'three'] + const bounded = boundArchiveLines(lines) + expect(bounded.truncated).toBe(false) + expect(bounded.lines).toBe(lines) + }) + + it('keeps the newest lines in order and reports truncation', () => { + const lines = Array.from({ length: 40_000 }, (_, index) => `line ${index}`) + const bounded = boundArchiveLines(lines) + expect(bounded.truncated).toBe(true) + expect(totalCost(bounded.lines)).toBeLessThanOrEqual(TERMINAL_ARCHIVE_MAX_CHARS) + expect(bounded.lines.at(-1)).toBe(lines.at(-1)) + expect(bounded.lines).toEqual(lines.slice(lines.length - bounded.lines.length)) + }) + + it('truncates a single oversized line from its tail', () => { + const bounded = boundArchiveLines(['x'.repeat(TERMINAL_ARCHIVE_MAX_CHARS * 2)]) + expect(bounded.truncated).toBe(true) + expect(bounded.lines).toHaveLength(1) + expect(bounded.lines[0]).toHaveLength(TERMINAL_ARCHIVE_MAX_CHARS - 1) + }) + + it('bounds a blank-line flood in linear time', () => { + // The char budget admits ~262k blank lines; an unshift-per-line build was ~4.3s here. + const startedAt = performance.now() + const bounded = boundArchiveLines(Array.from({ length: 300_000 }, () => '')) + expect(bounded.lines).toHaveLength(TERMINAL_ARCHIVE_MAX_CHARS) + expect(performance.now() - startedAt).toBeLessThan(500) + }) +}) diff --git a/src/main/runtime/orchestration/worker-output-archive.ts b/src/main/runtime/orchestration/worker-output-archive.ts index f6f2b52ecb1..55d16467269 100644 --- a/src/main/runtime/orchestration/worker-output-archive.ts +++ b/src/main/runtime/orchestration/worker-output-archive.ts @@ -114,7 +114,7 @@ export async function captureWorkerOutputArchive(args: { } } -function boundArchiveLines(lines: string[]): { lines: string[]; truncated: boolean } { +export function boundArchiveLines(lines: string[]): { lines: string[]; truncated: boolean } { let total = 0 for (const line of lines) { total += line.length + 1 @@ -122,18 +122,21 @@ function boundArchiveLines(lines: string[]): { lines: string[]; truncated: boole if (total <= TERMINAL_ARCHIVE_MAX_CHARS) { return { lines, truncated: false } } - const kept: string[] = [] + // Collected newest-first and reversed once: unshift per line is O(n^2) and the + // char budget admits ~260k blank lines. + const keptReversed: string[] = [] let budget = TERMINAL_ARCHIVE_MAX_CHARS for (let index = lines.length - 1; index >= 0; index -= 1) { const cost = lines[index].length + 1 if (cost > budget) { - if (kept.length === 0 && budget > 1) { - kept.unshift(lines[index].slice(-(budget - 1))) + if (keptReversed.length === 0 && budget > 1) { + keptReversed.push(lines[index].slice(-(budget - 1))) } break } - kept.unshift(lines[index]) + keptReversed.push(lines[index]) budget -= cost } - return { lines: kept, truncated: true } + keptReversed.reverse() + return { lines: keptReversed, truncated: true } } diff --git a/src/main/runtime/relay/desktop-relay-service.ts b/src/main/runtime/relay/desktop-relay-service.ts index 55b993bd5a6..def786e7758 100644 --- a/src/main/runtime/relay/desktop-relay-service.ts +++ b/src/main/runtime/relay/desktop-relay-service.ts @@ -6,6 +6,7 @@ import type { PairingGetEndpointsResult, PairingProvisionRelayParams } from '../../../shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import { readRelayAuthContext } from './relay-auth-context' import { RelayAuthCoordinator } from './relay-auth-coordinator' import { RelaySessionBroker, type RelayBrokerStatus } from './relay-session-broker' @@ -112,7 +113,7 @@ export class DesktopRelayService { this.refreshDemand() } - fenceAndCloseNow(): void { + fenceAndCloseNow(hostCloseReason?: RelayHostCloseReason): void { // Why: a fence must be hard — a surviving liveness tick could catch the // window between the pre-sign-out fence and the profile wipe and briefly // resurrect a broker. The next auth mutation re-arms via refreshDemand. @@ -120,7 +121,7 @@ export class DesktopRelayService { clearInterval(this.livenessTimer) this.livenessTimer = null } - this.coordinator.fenceAndCloseNow() + this.coordinator.fenceAndCloseNow(hostCloseReason) } async createPairingRelay( diff --git a/src/main/runtime/relay/relay-auth-coordinator.ts b/src/main/runtime/relay/relay-auth-coordinator.ts index ae439a320d9..a7db0e3a81e 100644 --- a/src/main/runtime/relay/relay-auth-coordinator.ts +++ b/src/main/runtime/relay/relay-auth-coordinator.ts @@ -1,3 +1,7 @@ +import { + RELAY_HOST_CLOSE_REASON, + type RelayHostCloseReason +} from '../../../shared/relay-host-close-reason' import type { RelayBrokerStatus } from './relay-session-broker' import { RelayHttpError, shouldRetryRelayConnectionError } from './relay-http-client' @@ -14,7 +18,7 @@ export type RelayAuthContext = { } export type CoordinatedRelayBroker = { - closeNow(): void + closeNow(hostCloseReason?: RelayHostCloseReason): void isLive?(): boolean } @@ -78,13 +82,16 @@ export class RelayAuthCoordinator { void reconcile } - fenceAndCloseNow(): void { + // hostCloseReason names an auth loss the phone should be told about. Quit, + // relaunch and every other fence pass nothing, so the control socket dies + // abruptly exactly as before and the cell records no cause. + fenceAndCloseNow(hostCloseReason?: RelayHostCloseReason): void { ++this.authEpoch this.cancelLinger() this.cancelRetry() this.retryAttempt = 0 this.invalidatePendingOwnerships() - this.invalidateOwnership() + this.invalidateOwnership(hostCloseReason) this.options.onStatus('offline') } @@ -146,7 +153,11 @@ export class RelayAuthCoordinator { if (!context || !context.relayEntitled) { this.cancelLinger() this.retryAttempt = 0 - this.invalidateOwnership() + // Why only the null case: readContext throws on transient failures and + // returns null solely when the cloud session is gone (absent, or cleared + // by a 401). A present-but-unentitled context is still a signed-in + // desktop, and "sign in to reconnect" would be wrong advice for it. + this.invalidateOwnership(context ? undefined : RELAY_HOST_CLOSE_REASON.SIGNED_OUT) this.options.onStatus('offline') return } @@ -271,12 +282,12 @@ export class RelayAuthCoordinator { return context.accessToken } - private invalidateOwnership(): void { + private invalidateOwnership(hostCloseReason?: RelayHostCloseReason): void { const ownership = this.ownership this.ownership = null if (ownership) { ownership.valid = false - ownership.broker?.closeNow() + ownership.broker?.closeNow(hostCloseReason) } } diff --git a/src/main/runtime/relay/relay-auth-host-close-reason.test.ts b/src/main/runtime/relay/relay-auth-host-close-reason.test.ts new file mode 100644 index 00000000000..7d10e207a33 --- /dev/null +++ b/src/main/runtime/relay/relay-auth-host-close-reason.test.ts @@ -0,0 +1,122 @@ +import { describe, expect, it, vi } from 'vitest' +import { RELAY_HOST_CLOSE_REASON } from '../../../shared/relay-host-close-reason' +import { RelayAuthCoordinator, type RelayAuthContext } from './relay-auth-coordinator' + +const context: RelayAuthContext = { + identity: { userId: 'user-1', profileId: 'profile-1', organizationId: 'org-1' }, + accessToken: 'access-1', + relayEntitled: true +} + +function coordinatorOver(readContext: () => Promise) { + const broker = { closeNow: vi.fn() } + const coordinator = new RelayAuthCoordinator({ + readContext, + openBroker: async () => broker, + onStatus: vi.fn() + }) + return { broker, coordinator } +} + +describe('relay control close reason', () => { + it('names the sign-out when the cloud session is gone', async () => { + let current: RelayAuthContext | null = context + const { broker, coordinator } = coordinatorOver(async () => current) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + current = null + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(broker.closeNow).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('names the sign-out on the explicit pre-sign-out fence', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.fenceAndCloseNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + expect(broker.closeNow).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('stays silent on quit, which fences without a reason', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.fenceAndCloseNow() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + it('stays silent on stop, which is teardown rather than auth loss', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.stop() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + // A signed-in desktop that merely lost the entitlement must not tell the + // phone to sign in — the copy would be wrong and the user has nothing to do. + it('stays silent when the session survives but the entitlement is gone', async () => { + let current: RelayAuthContext = context + const { broker, coordinator } = coordinatorOver(async () => current) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + current = { ...context, relayEntitled: false } + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + it('stays silent when demand drops and the broker lingers out', async () => { + let demanded = true + const broker = { closeNow: vi.fn() } + const coordinator = new RelayAuthCoordinator({ + readContext: async () => context, + hasDemand: () => demanded, + openBroker: async () => broker, + onStatus: vi.fn(), + lingerMs: 5 + }) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + demanded = false + coordinator.reconcile() + await vi.waitFor(() => expect(broker.closeNow).toHaveBeenCalled()) + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + // Replacing a stale broker is a reconnect, not a sign-out. + it('stays silent when an identity switch replaces the broker', async () => { + let current = context + const brokers: { closeNow: ReturnType }[] = [] + const coordinator = new RelayAuthCoordinator({ + readContext: async () => current, + openBroker: async () => { + const broker = { closeNow: vi.fn() } + brokers.push(broker) + return broker + }, + onStatus: vi.fn() + }) + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + current = { ...context, identity: { ...context.identity, organizationId: 'org-2' } } + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(brokers[0]?.closeNow).toHaveBeenCalledWith(undefined) + }) +}) diff --git a/src/main/runtime/relay/relay-control-client.ts b/src/main/runtime/relay/relay-control-client.ts index 7e742173f72..968f05795b2 100644 --- a/src/main/runtime/relay/relay-control-client.ts +++ b/src/main/runtime/relay/relay-control-client.ts @@ -1,6 +1,7 @@ import { randomUUID } from 'node:crypto' import WebSocket, { type RawData } from 'ws' import { MOBILE_RELAY_CLOSE_CODE } from '../../../shared/mobile-relay-close-codes' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { E2EEKeypair } from '../e2ee-keypair' import { RelayConnectionOpenMessageSchema, @@ -8,6 +9,7 @@ import { RelayHostChallengeMessageSchema, RelayHostHelloAckMessageSchema, RelayPingMessageSchema, + encodeRelayHostHello, parseRelayControlMessage, type RelayConnectionOpenMessage, type RelayDrainMessage, @@ -21,6 +23,7 @@ import { RELAY_CONTROL_SILENCE_LIMIT_MS, RelayControlSilenceWatchdog } from './relay-control-silence-watchdog' +import { closeRelayControlSocket } from './relay-control-socket-close' import { controlWebSocketUrl } from './relay-control-url' type RelayControlState = 'idle' | 'opening' | 'proving' | 'active' | 'draining' | 'closed' @@ -166,7 +169,7 @@ export class RelayControlClient { return this.requests.confirmResume(reqId, basisConnId, (payload) => this.sendActive(payload)) } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { const wasConnecting = this.state === 'opening' || this.state === 'proving' this.state = 'closed' this.silenceWatchdog.stop() @@ -175,8 +178,9 @@ export class RelayControlClient { this.clearConnectPromise() } this.requests.rejectAll(new Error('relay_control_closed')) - this.socket?.terminate() + const socket = this.socket this.socket = null + closeRelayControlSocket(socket, hostCloseReason) } private sendHostHello(): void { @@ -185,19 +189,9 @@ export class RelayControlClient { } this.state = 'proving' this.socket.send( - JSON.stringify({ - type: 'host-hello', - v: 1, - relayHostId: this.options.relayHostId, - assignmentEpoch: this.options.assignmentEpoch, - hostPublicKeyB64: this.options.keypair.publicKeyB64, - appVersion: this.options.appVersion, - ...(this.options.previousGeneration === undefined - ? {} - : { previousGeneration: this.options.previousGeneration }), - ...(this.options.controlResumeSecret - ? { controlResumeSecret: this.options.controlResumeSecret } - : {}) + encodeRelayHostHello({ + ...this.options, + hostPublicKeyB64: this.options.keypair.publicKeyB64 }) ) } diff --git a/src/main/runtime/relay/relay-control-close-reason.test.ts b/src/main/runtime/relay/relay-control-close-reason.test.ts new file mode 100644 index 00000000000..0d9aba54f88 --- /dev/null +++ b/src/main/runtime/relay/relay-control-close-reason.test.ts @@ -0,0 +1,92 @@ +import { createHash } from 'node:crypto' +import { afterEach, describe, expect, it, vi } from 'vitest' +import nacl from 'tweetnacl' +import { WebSocketServer, type WebSocket } from 'ws' +import { RELAY_HOST_CLOSE_REASON } from '../../../shared/relay-host-close-reason' +import { RelayControlClient } from './relay-control-client' + +type ObservedClose = { code: number; reason: string } + +describe('RelayControlClient close reason', () => { + const servers: WebSocketServer[] = [] + const clients: RelayControlClient[] = [] + + afterEach(async () => { + for (const client of clients.splice(0)) { + client.closeNow() + } + await Promise.all( + servers.splice(0).map( + (server) => + new Promise((resolve) => { + for (const socket of server.clients) { + socket.terminate() + } + server.close(() => resolve()) + }) + ) + ) + }) + + async function connectedClient(): Promise<{ + client: RelayControlClient + closed: Promise + }> { + const server = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) + servers.push(server) + await new Promise((resolve) => server.once('listening', resolve)) + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('expected TCP relay test server') + } + const accepted = new Promise((resolve) => server.once('connection', resolve)) + const keypair = nacl.box.keyPair() + const client = new RelayControlClient({ + cellUrl: `http://127.0.0.1:${address.port}`, + relayJwt: 'scoped-token', + relayHostId: createHash('sha256').update(keypair.publicKey).digest('base64url').slice(0, 16), + assignmentEpoch: 1, + identity: { userId: 'user-1', profileId: 'profile-1', organizationId: 'org-1' }, + keypair: { ...keypair, publicKeyB64: Buffer.from(keypair.publicKey).toString('base64') }, + appVersion: '1.2.3', + onConnectionOpen: vi.fn(), + onDrain: vi.fn(), + onClose: vi.fn() + }) + clients.push(client) + // The handshake never completes here; only the transport close matters. + void client.connect().catch(() => {}) + const socket = await accepted + // A pong proves the client socket left CONNECTING; closeNow can only send a + // close frame from OPEN, and that is the state a real sign-out fences from. + await new Promise((resolve) => { + socket.once('pong', () => resolve()) + socket.ping() + }) + const closed = new Promise((resolve) => { + socket.once('close', (code, reason) => resolve({ code, reason: reason.toString() })) + }) + return { client, closed } + } + + it('delivers the sign-out reason to the cell', async () => { + const { client, closed } = await connectedClient() + + client.closeNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + await expect(closed).resolves.toEqual({ + code: 1000, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + }) + + // Every non-auth close keeps today's abrupt terminate, so a cell can never + // read a quit, a rotation or a sleep as a sign-out. + it('closes abruptly with no reason when none is given', async () => { + const { client, closed } = await connectedClient() + + client.closeNow() + + await expect(closed).resolves.toEqual({ code: 1006, reason: '' }) + }) +}) diff --git a/src/main/runtime/relay/relay-control-origin.ts b/src/main/runtime/relay/relay-control-origin.ts index 4d42da47977..3a33e1617e6 100644 --- a/src/main/runtime/relay/relay-control-origin.ts +++ b/src/main/runtime/relay/relay-control-origin.ts @@ -8,6 +8,7 @@ import type { RelayDrainMessage, RelayHostHelloAckMessage } from './relay-control-protocol' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { RelayIdentity } from './relay-session-broker-contract' import type { RelayAssignment } from './relay-http-client' @@ -144,7 +145,7 @@ export class RelayControlOrigin { } } - async close(): Promise { + async close(hostCloseReason?: RelayHostCloseReason): Promise { if (this.closed) { return } @@ -154,7 +155,7 @@ export class RelayControlOrigin { } this.retiredControlTimers.clear() for (const control of this.controls) { - control.closeNow() + control.closeNow(hostCloseReason) } this.controls.clear() this.activeControl = null @@ -166,8 +167,8 @@ export class RelayControlOrigin { } } - closeNow(): void { - void this.close() + closeNow(hostCloseReason?: RelayHostCloseReason): void { + void this.close(hostCloseReason) } private async openControl(overrides?: { diff --git a/src/main/runtime/relay/relay-control-protocol.ts b/src/main/runtime/relay/relay-control-protocol.ts index c94797f4815..75bf3a390e2 100644 --- a/src/main/runtime/relay/relay-control-protocol.ts +++ b/src/main/runtime/relay/relay-control-protocol.ts @@ -143,3 +143,29 @@ export function parseRelayControlMessage(raw: RawData): Record return null } } + +export type RelayHostHello = { + relayHostId: string + assignmentEpoch: number + hostPublicKeyB64: string + appVersion: string + previousGeneration?: number + controlResumeSecret?: string +} + +// Optional members are omitted rather than sent as undefined: the cell parses +// host-hello strictly and an explicit null is not the same as absent. +export function encodeRelayHostHello(hello: RelayHostHello): string { + return JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hello.relayHostId, + assignmentEpoch: hello.assignmentEpoch, + hostPublicKeyB64: hello.hostPublicKeyB64, + appVersion: hello.appVersion, + ...(hello.previousGeneration === undefined + ? {} + : { previousGeneration: hello.previousGeneration }), + ...(hello.controlResumeSecret ? { controlResumeSecret: hello.controlResumeSecret } : {}) + }) +} diff --git a/src/main/runtime/relay/relay-control-socket-close.ts b/src/main/runtime/relay/relay-control-socket-close.ts new file mode 100644 index 00000000000..2984b075425 --- /dev/null +++ b/src/main/runtime/relay/relay-control-socket-close.ts @@ -0,0 +1,26 @@ +import type WebSocket from 'ws' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' + +const NORMAL_CLOSE_CODE = 1000 +const REASONED_CLOSE_FLUSH_MS = 1_000 + +// hostCloseReason: only auth loss names itself. Every other control close +// (rotation, drain, quit, sleep) stays an abrupt terminate, so the cell learns +// nothing and can never read a restart as a sign-out. A named close has to +// reach the cell as a real close frame, but the fence must still be hard — +// bound the handshake and then terminate. +export function closeRelayControlSocket( + socket: WebSocket | null, + hostCloseReason?: RelayHostCloseReason +): void { + if (!socket) { + return + } + if (hostCloseReason && socket.readyState === socket.OPEN) { + socket.close(NORMAL_CLOSE_CODE, hostCloseReason) + const timer = setTimeout(() => socket.terminate(), REASONED_CLOSE_FLUSH_MS) + timer.unref?.() + return + } + socket.terminate() +} diff --git a/src/main/runtime/relay/relay-origin-pool.ts b/src/main/runtime/relay/relay-origin-pool.ts index a95d8f21340..acd2f90292e 100644 --- a/src/main/runtime/relay/relay-origin-pool.ts +++ b/src/main/runtime/relay/relay-origin-pool.ts @@ -4,8 +4,10 @@ import type { MobileSocketWiring } from '../rpc/mobile-socket-wiring' import { RelayControlOrigin } from './relay-control-origin' import type { RelayControlClient } from './relay-control-client' import type { RelayDrainMessage } from './relay-control-protocol' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import { RelayDrainRetrySchedule } from './relay-drain-retry-schedule' import { RelayHttpError, requestRelayAssignment, type RelayAssignment } from './relay-http-client' +import { relayRenewalDelayMs } from './relay-renewal-jitter' import type { RelayBrokerStatus, RelayIdentity } from './relay-session-broker-contract' import type { RelayRegion } from './relay-region-preference' @@ -79,7 +81,7 @@ export class RelayOriginPool { } } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { if (this.closed) { return } @@ -94,7 +96,7 @@ export class RelayOriginPool { } this.drainTimers.clear() for (const origin of this.origins) { - origin.closeNow() + origin.closeNow(hostCloseReason) } this.origins.clear() this.drainingOrigins.clear() @@ -244,8 +246,7 @@ export class RelayOriginPool { } const now = (this.options.now ?? Date.now)() const random = this.options.random ?? Math.random - const earlyMs = 60_000 + Math.floor(random() * 60_001) - const delay = Math.max(0, origin.controlLeaseExpiresAt - earlyMs - now) + const delay = relayRenewalDelayMs(origin.controlLeaseExpiresAt, now, random) this.rotationTimer = setTimeout(() => void this.rebindActiveControl(origin), delay) } diff --git a/src/main/runtime/relay/relay-renewal-jitter.test.ts b/src/main/runtime/relay/relay-renewal-jitter.test.ts new file mode 100644 index 00000000000..5bbc0994836 --- /dev/null +++ b/src/main/runtime/relay/relay-renewal-jitter.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from 'vitest' +import { + RELAY_RENEWAL_JITTER_RATIO, + RELAY_RENEWAL_SAFETY_MARGIN_MS, + relayRenewalDelayMs +} from './relay-renewal-jitter' + +const LEASE_MS = 55 * 60_000 +const latest = LEASE_MS - RELAY_RENEWAL_SAFETY_MARGIN_MS +const base = latest / (1 + RELAY_RENEWAL_JITTER_RATIO) + +describe('relay renewal jitter', () => { + it('keeps every sample inside the jitter band and before the safety margin', () => { + let seed = 1 + const random = (): number => { + seed = (seed * 1103515245 + 12345) % 2147483648 + return seed / 2147483648 + } + const samples: number[] = [] + for (let i = 0; i < 20_000; i++) { + samples.push(relayRenewalDelayMs(LEASE_MS, 0, random)) + } + for (const sample of samples) { + expect(sample).toBeGreaterThanOrEqual(Math.floor(base * (1 - RELAY_RENEWAL_JITTER_RATIO))) + expect(sample).toBeLessThanOrEqual(latest) + // The renewal never lands inside the margin, so it never races expiry. + expect(LEASE_MS - sample).toBeGreaterThanOrEqual(RELAY_RENEWAL_SAFETY_MARGIN_MS) + } + const mean = samples.reduce((total, sample) => total + sample, 0) / samples.length + expect(Math.abs(mean - base) / base).toBeLessThan(0.005) + }) + + it('spreads a same-second cohort over minutes instead of one second', () => { + const delays = Array.from({ length: 1000 }, (_, index) => + relayRenewalDelayMs(LEASE_MS, 0, () => index / 999) + ) + const spread = Math.max(...delays) - Math.min(...delays) + expect(spread).toBeGreaterThan(9 * 60_000) + }) + + it('pins the band ends to the base interval', () => { + expect(relayRenewalDelayMs(LEASE_MS, 0, () => 0)).toBe( + Math.floor(base * (1 - RELAY_RENEWAL_JITTER_RATIO)) + ) + expect(relayRenewalDelayMs(LEASE_MS, 0, () => 0.5)).toBe(Math.floor(base)) + expect(relayRenewalDelayMs(LEASE_MS, 0, () => 1)).toBe(latest) + }) + + it('renews immediately once the lease is inside the safety margin', () => { + expect(relayRenewalDelayMs(RELAY_RENEWAL_SAFETY_MARGIN_MS, 0, () => 1)).toBe(0) + expect(relayRenewalDelayMs(0, 60_000, () => 1)).toBe(0) + }) + + it('measures the delay from now, not from the epoch', () => { + expect(relayRenewalDelayMs(LEASE_MS + 1_000_000, 1_000_000, () => 0.5)).toBe(Math.floor(base)) + }) +}) diff --git a/src/main/runtime/relay/relay-renewal-jitter.ts b/src/main/runtime/relay/relay-renewal-jitter.ts new file mode 100644 index 00000000000..8b1f1ec6412 --- /dev/null +++ b/src/main/runtime/relay/relay-renewal-jitter.ts @@ -0,0 +1,25 @@ +// Why: a cell recreate reconnects a whole cohort inside one second. Every host +// in it then took its lease from the same second and, with only a 60s-wide +// spread, renewed inside the same second ~54 minutes later — a self-sustaining +// fleet-wide reconnect burst. Full +/-10% jitter spreads that cohort over +// minutes instead. +export const RELAY_RENEWAL_JITTER_RATIO = 0.1 + +// The latest jittered renewal still lands this far before expiry. +export const RELAY_RENEWAL_SAFETY_MARGIN_MS = 90_000 + +// Why: the relay accepts a rebind at any point in the lease and resets the full +// TTL from it (cloud/apps/relay/src/host-session-registry.ts:736-743), so +// renewing early is free; only renewing late is fatal (:997 drains an expired +// lease). That asymmetry is why the base is shrunk to fit the upward jitter +// rather than the jittered value being clipped at the margin. +export function relayRenewalDelayMs(expiresAt: number, now: number, random: () => number): number { + const remaining = expiresAt - now + const latest = remaining - RELAY_RENEWAL_SAFETY_MARGIN_MS + if (latest <= 0) { + return 0 + } + const base = latest / (1 + RELAY_RENEWAL_JITTER_RATIO) + const jittered = base * (1 + (random() * 2 - 1) * RELAY_RENEWAL_JITTER_RATIO) + return Math.max(0, Math.min(Math.floor(jittered), latest)) +} diff --git a/src/main/runtime/relay/relay-session-broker.ts b/src/main/runtime/relay/relay-session-broker.ts index e12e09119a0..cd83545e9ca 100644 --- a/src/main/runtime/relay/relay-session-broker.ts +++ b/src/main/runtime/relay/relay-session-broker.ts @@ -6,6 +6,7 @@ import type { MobileRelayEndpoint, PairingProvisionRelayParams } from '../../../shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { DeviceCredentialInstallAuthorization } from './relay-control-requests' import { deriveRelayHostId, @@ -15,6 +16,7 @@ import { type RelayAssignment } from './relay-http-client' import { RelayOriginPool } from './relay-origin-pool' +import { relayRenewalDelayMs } from './relay-renewal-jitter' import type { RelayBrokerStatus, RelaySessionBrokerOptions } from './relay-session-broker-contract' export type { RelayBrokerStatus } from './relay-session-broker-contract' @@ -180,7 +182,7 @@ export class RelaySessionBroker { return result } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { if (this.closed) { return } @@ -190,7 +192,7 @@ export class RelaySessionBroker { clearTimeout(this.refreshTimer) this.refreshTimer = null } - this.originPool.closeNow() + this.originPool.closeNow(hostCloseReason) if (publishOffline) { this.options.onStatus('offline') } @@ -242,8 +244,7 @@ export class RelaySessionBroker { } const now = (this.options.now ?? Date.now)() const random = this.options.random ?? Math.random - const earlyMs = 60_000 + Math.floor(random() * 60_001) - const delay = Math.max(0, authorization.expiresAt - earlyMs - now) + const delay = relayRenewalDelayMs(authorization.expiresAt, now, random) this.refreshTimer = setTimeout(() => void this.refreshAuthorization(), delay) } diff --git a/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts new file mode 100644 index 00000000000..48649bc372c --- /dev/null +++ b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts @@ -0,0 +1,92 @@ +// The host half of the same contract: an RPC schema silently strips keys it does not declare, which +// is exactly what forward compatibility needs and exactly how a caller's option can vanish inside +// one version. `scanChildProcesses` has to be declared here, and only here -- the sibling handle +// methods have no use for it and must keep refusing it. +import { describe, expect, it, vi } from 'vitest' +import type { ZodType } from 'zod' +import { TERMINAL_QUERY_METHODS } from './terminal-query-methods' +import { TerminalHandle, TerminalInspectProcess } from './unary-schemas' + +/** The method as registered, so a schema swap on the definition cannot pass unseen. */ +function inspectProcessMethod() { + const method = TERMINAL_QUERY_METHODS.find((entry) => entry.name === 'terminal.inspectProcess') + if (!method) { + throw new Error('terminal.inspectProcess is not registered') + } + return method +} + +async function callRegisteredHandler( + params: Record +): Promise<{ terminal: string; options: unknown }> { + const method = inspectProcessMethod() + const parsed = (method.params as ZodType).parse(params) + const inspectTerminalProcess = vi.fn(async () => ({ + foregroundProcess: null, + hasChildProcesses: false + })) + await method.handler(parsed, { runtime: { inspectTerminalProcess } } as never, undefined as never) + const [terminal, options] = inspectTerminalProcess.mock.calls[0] as unknown as [string, unknown] + return { terminal, options } +} + +describe('terminal.inspectProcess registration', () => { + // The half the schema test alone cannot see: pointing the method back at the shared handle schema + // compiles, parses, and silently drops the option. This exercises the registered definition. + it('carries scanChildProcesses from the wire into the runtime call', async () => { + await expect( + callRegisteredHandler({ terminal: 'term_1', scanChildProcesses: true }) + ).resolves.toEqual({ terminal: 'term_1', options: { scanChildProcesses: true } }) + }) + + it('carries it alongside the incarnation fence', async () => { + await expect( + callRegisteredHandler({ + terminal: 'term_1', + expectedIncarnationId: 'inc-1', + scanChildProcesses: true + }) + ).resolves.toEqual({ + terminal: 'term_1', + options: { expectedIncarnationId: 'inc-1', scanChildProcesses: true } + }) + }) + + it('keeps the legacy one-argument shape for a bare poll', async () => { + await expect(callRegisteredHandler({ terminal: 'term_1' })).resolves.toEqual({ + terminal: 'term_1', + options: undefined + }) + }) +}) + +describe('terminal.inspectProcess params', () => { + it('preserves scanChildProcesses', () => { + expect(TerminalInspectProcess.parse({ terminal: 'term_1', scanChildProcesses: true })).toEqual({ + terminal: 'term_1', + scanChildProcesses: true + }) + }) + + it('preserves it alongside the incarnation fence', () => { + expect( + TerminalInspectProcess.parse({ + terminal: 'term_1', + expectedIncarnationId: 'inc-1', + scanChildProcesses: true + }) + ).toEqual({ terminal: 'term_1', expectedIncarnationId: 'inc-1', scanChildProcesses: true }) + }) + + it('leaves it absent for a polling caller', () => { + expect(TerminalInspectProcess.parse({ terminal: 'term_1' })).toEqual({ terminal: 'term_1' }) + }) + + // The shape that produced the bug, pinned so nobody "simplifies" the method back onto the shared + // handle schema: TerminalHandle drops the option on the floor without complaining. + it('shows why the shared handle schema could not carry it', () => { + expect(TerminalHandle.parse({ terminal: 'term_1', scanChildProcesses: true })).toEqual({ + terminal: 'term_1' + }) + }) +}) diff --git a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts index a8aec9ff573..bf7b4a5bd87 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts @@ -1,6 +1,7 @@ import { defineMethod, type RpcAnyMethod } from '../../core' import { TerminalHandle, + TerminalInspectProcess, TerminalListParams, TerminalRead, TerminalRecoverPane, @@ -65,15 +66,21 @@ export const TERMINAL_QUERY_METHODS: RpcAnyMethod[] = [ }), defineMethod({ name: 'terminal.inspectProcess', - params: TerminalHandle, - handler: async (params, { runtime }) => ({ - process: await runtime.inspectTerminalProcess( - params.terminal, - params.expectedIncarnationId + params: TerminalInspectProcess, + handler: async (params, { runtime }) => { + const options = { + ...(params.expectedIncarnationId ? { expectedIncarnationId: params.expectedIncarnationId } - : undefined - ) - }) + : {}), + ...(params.scanChildProcesses === true ? { scanChildProcesses: true } : {}) + } + return { + process: await runtime.inspectTerminalProcess( + params.terminal, + Object.keys(options).length > 0 ? options : undefined + ) + } + } }), defineMethod({ name: 'terminal.isRunningAgent', diff --git a/src/main/runtime/rpc/methods/terminal/unary-schemas.ts b/src/main/runtime/rpc/methods/terminal/unary-schemas.ts index 323b34a7544..afe0a3bb486 100644 --- a/src/main/runtime/rpc/methods/terminal/unary-schemas.ts +++ b/src/main/runtime/rpc/methods/terminal/unary-schemas.ts @@ -13,6 +13,17 @@ export const TerminalFocus = TerminalHandle.extend({ navigation: z.enum(['caller', 'host']).optional() }) +/** + * `terminal.inspectProcess` carries one member the sibling handle methods must not: whether the + * caller's answer decides something once, which is what licenses the host to pay for a process-table + * read. Extended rather than added to `TerminalHandle` so `clearBuffer`/`agentStatus`/`isRunningAgent` + * keep refusing an option they have no use for. + */ +export const TerminalInspectProcess = TerminalHandle.extend({ + // Additive request member understood by newer hosts; legacy hosts safely ignore it. + scanChildProcesses: z.boolean().optional() +}) + export const TerminalListParams = z.object({ worktree: OptionalString, limit: OptionalFiniteNumber, diff --git a/src/main/runtime/runtime-managed-worktree-queries.test.ts b/src/main/runtime/runtime-managed-worktree-queries.test.ts index 01df53cbd1d..3fb792fcc7c 100644 --- a/src/main/runtime/runtime-managed-worktree-queries.test.ts +++ b/src/main/runtime/runtime-managed-worktree-queries.test.ts @@ -40,14 +40,18 @@ function metadata(overrides: Partial = {}): WorktreeMeta { } } -function queries(store: RuntimeStore): RuntimeManagedWorktreeQueries { +function queries( + store: RuntimeStore, + overrides: Partial[0]> = {} +): RuntimeManagedWorktreeQueries { return new RuntimeManagedWorktreeQueries({ getStore: () => store, listResolved: async () => [], resolveRepo: async () => store.getRepos()[0]!, selectRepos: () => store.getRepos(), scanRepo: async () => ({ ok: true, worktrees: [] }), - listKnownHostIds: () => [] + listKnownHostIds: () => [], + ...overrides }) } @@ -105,3 +109,115 @@ describe('RuntimeManagedWorktreeQueries.listDetected', () => { expect(legacy.worktrees[0]).not.toHaveProperty('visibilitySource') }) }) + +describe('RuntimeManagedWorktreeQueries.list host scope', () => { + // Measured on hardware before this fix, same runtime and same refusing SSH host in the same + // second: the UNSCOPED listing reported `omittedHostIds: ["local","ssh:ssh-scope-refused"]` with + // `--host` selectors, while the SCOPED listing reported `{"hostIds":[],"omittedHostIds":[]}`. + // A listing that covered nothing, reporting no gaps, is indistinguishable from a repo that + // genuinely has no worktrees -- the thing docs/reference/ssh-execution-boundary.md forbids. + function sshStore(): RuntimeStore { + const repo = folderRepo({ + id: 'repo-ssh', + kind: 'git', + connectionId: 'conn-1', + path: '/home/dev/app' + }) + return { + getRepos: () => [repo], + getRepo: () => repo, + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getSettings: () => settings + } as unknown as RuntimeStore + } + + it('names the scoped repo host as omitted when the listing covered nothing', async () => { + const result = await queries(sshStore()).list('repo-ssh', 50) + + expect(result.totalCount).toBe(0) + expect(result.hostScope).toEqual({ + hostIds: [], + omittedHostIds: ['ssh:conn-1'] + }) + }) + + it('does not report the scoped host as omitted once it contributes rows', async () => { + const store = sshStore() + const result = await queries(store, { + listResolved: async () => + [ + { + id: 'repo-ssh::/home/dev/app', + repoId: 'repo-ssh', + path: '/home/dev/app', + hostId: 'ssh:conn-1' + } + ] as never + }).list('repo-ssh', 50) + + expect(result.hostScope?.hostIds).toEqual(['ssh:conn-1']) + expect(result.hostScope?.omittedHostIds).toEqual([]) + }) + + // The caller scoped the listing, so the hosts they excluded must not come back as gaps. + it('never names a host the caller scoped out', async () => { + const scoped = await queries(sshStore(), { + listKnownHostIds: () => ['local', 'ssh:other', 'runtime:elsewhere'] as never + }).list('repo-ssh', 50) + + expect(scoped.hostScope?.omittedHostIds).toEqual(['ssh:conn-1']) + }) + + // `getRepoExecutionHostId` derives the host from two spellings, and a scoped listing that named + // the wrong one would be worse than naming none. These pin both. + it('names the local host for a scoped local repo', async () => { + const repo = folderRepo({ id: 'repo-local', kind: 'git', path: '/workspace/local' }) + const store = { + getRepos: () => [repo], + getRepo: () => repo, + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getSettings: () => settings + } as unknown as RuntimeStore + + const result = await queries(store).list('repo-local', 50) + + expect(result.hostScope?.omittedHostIds).toEqual(['local']) + }) + + it('prefers executionHostId over connectionId for the scoped host', async () => { + const repo = folderRepo({ + id: 'repo-runtime', + kind: 'git', + connectionId: 'conn-legacy', + executionHostId: 'runtime:env-1', + path: '/workspace/runtime' + }) + const store = { + getRepos: () => [repo], + getRepo: () => repo, + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getSettings: () => settings + } as unknown as RuntimeStore + + const result = await queries(store).list('repo-runtime', 50) + + expect(result.hostScope?.omittedHostIds).toEqual(['runtime:env-1']) + }) + + it('still reports every configured host when the listing is unscoped', async () => { + const unscoped = await queries(sshStore(), { + listKnownHostIds: () => ['local', 'ssh:conn-1'] as never + }).list(undefined, 50) + + expect(unscoped.hostScope?.omittedHostIds).toEqual(['local', 'ssh:conn-1']) + }) +}) diff --git a/src/main/runtime/runtime-managed-worktree-queries.ts b/src/main/runtime/runtime-managed-worktree-queries.ts index 5a5812236af..158ab77cdd0 100644 --- a/src/main/runtime/runtime-managed-worktree-queries.ts +++ b/src/main/runtime/runtime-managed-worktree-queries.ts @@ -2,7 +2,7 @@ import type { DetectedWorktreeListResult, Worktree } from '../../shared/worktree import type { Repo } from '../../shared/repo-types' import type { RuntimeWorktreeListResult } from '../../shared/runtime-types' import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' -import { buildWorktreeListingPage } from './worktree-listing-host-scope' +import { buildWorktreeListingPage, listingKnownHostIds } from './worktree-listing-host-scope' import { readWorktreeMetaForHost } from '../persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from '../worktree-metadata-ownership' import type { WorktreeMeta } from '../../shared/worktree/meta-types' @@ -80,7 +80,7 @@ export class RuntimeManagedWorktreeQueries { throw new Error('invalid_limit') } const resolved = await this.deps.listResolved() - const repoId = repoSelector ? (await this.deps.resolveRepo(repoSelector)).id : null + const scopedRepo = repoSelector ? await this.deps.resolveRepo(repoSelector) : null const pathsByRepo = new Map() for (const worktree of resolved) { const paths = pathsByRepo.get(worktree.repoId) ?? [] @@ -100,12 +100,12 @@ export class RuntimeManagedWorktreeQueries { ) const worktrees = resolved.filter( (worktree) => - (!repoId || worktree.repoId === repoId) && + (!scopedRepo || worktree.repoId === scopedRepo.id) && this.isVisible(worktree, matchers.get(worktree.repoId), sourceDefaultsSupported) ) - // Why: a `--repo` listing was scoped by the caller, so naming every configured host as - // omitted would report a gap the caller deliberately excluded. - return buildWorktreeListingPage(worktrees, limit, repoId ? [] : this.deps.listKnownHostIds()) + // See `listingKnownHostIds`: a scoped listing must still name the host it was asked about. + const knownHostIds = listingKnownHostIds(scopedRepo, () => this.deps.listKnownHostIds()) + return buildWorktreeListingPage(worktrees, limit, knownHostIds) } resolveRepoForConnection(selector: string, connectionId?: string | null): Promise { diff --git a/src/main/runtime/runtime-pty-controller-contract.ts b/src/main/runtime/runtime-pty-controller-contract.ts index 194d0bb665c..665c6fdb609 100644 --- a/src/main/runtime/runtime-pty-controller-contract.ts +++ b/src/main/runtime/runtime-pty-controller-contract.ts @@ -109,7 +109,7 @@ export type RuntimePtyController = { getForegroundProcess(ptyId: string): Promise inspectProcess?( ptyId: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: { expectedIncarnationId?: PtyIncarnationId; scanChildProcesses?: boolean } ): Promise confirmForegroundProcess?(ptyId: string): Promise confirmShellForeground?(ptyId: string): Promise diff --git a/src/main/runtime/structured-agent-session-integration-replay.test.ts b/src/main/runtime/structured-agent-session-integration-replay.test.ts index 7bc5e66dc45..4edec30499b 100644 --- a/src/main/runtime/structured-agent-session-integration-replay.test.ts +++ b/src/main/runtime/structured-agent-session-integration-replay.test.ts @@ -21,7 +21,7 @@ import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protoc import type { AgentJournalRenderItem } from '../../shared/agent-session-journal-types' import { attachFingerprintFields } from '../native-chat/agent-session-wire/structured-agent-session-attach' import { journalDirectoryFor } from '../native-chat/agent-session-journal/journal-paths' -import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' import type { OrcaRuntimeService } from './orca-runtime' import type { RpcRequest, RpcResponse } from './rpc/core' import { RpcDispatcher } from './rpc/dispatcher' @@ -31,6 +31,8 @@ import { stopStructuredAgentSessionRuntime } from './structured-agent-session-runtime' +const journals = createTrackedJournalOpener() + const SESSION = 'session-integration-1' const THREAD = 'thread-integration' const TURN = 'turn-1' @@ -285,6 +287,7 @@ beforeEach(async () => { }) afterEach(async () => { + await journals.closeAll() await stopStructuredAgentSessionRuntime() await rm(root, { recursive: true, force: true }) }) @@ -364,7 +367,7 @@ describe('a structured codex session over agentSession.*', () => { agent: 'codex' as const, providerHandle: { kind: 'codex' as const, threadId: THREAD } } - const reopened = await openAgentSessionJournal({ + const reopened = await journals.open({ identity, journalDir: journalDirectoryFor(root, identity) }) diff --git a/src/main/runtime/structured-agent-session-integration.test.ts b/src/main/runtime/structured-agent-session-integration.test.ts index 582bd225b40..aa5f819e1fb 100644 --- a/src/main/runtime/structured-agent-session-integration.test.ts +++ b/src/main/runtime/structured-agent-session-integration.test.ts @@ -25,10 +25,9 @@ import type { import { attachFingerprintFields } from '../native-chat/agent-session-wire/structured-agent-session-attach' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' import { journalDirectoryFor } from '../native-chat/agent-session-journal/journal-paths' -import { readJournalBlob } from '../native-chat/agent-session-journal/journal-blob-store' import { appendLegacyTranscriptMessages } from '../native-chat/agent-session-journal/journal-legacy-import' import type { AgentSessionJournal } from '../native-chat/agent-session-journal/journal-store' -import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' import type { OrcaRuntimeService } from './orca-runtime' import type { RpcRequest, RpcResponse } from './rpc/core' import { RpcDispatcher } from './rpc/dispatcher' @@ -38,6 +37,8 @@ import { stopStructuredAgentSessionRuntime } from './structured-agent-session-runtime' +const journals = createTrackedJournalOpener() + const SESSION = 'session-integration-1' const THREAD = 'thread-integration' const TURN = 'turn-1' @@ -363,6 +364,7 @@ function cursorOf(frames: AgentSessionSubscribeEvent[]): { epoch: string; sequen } afterEach(async () => { + await journals.closeAll() await stopStructuredAgentSessionRuntime() await rm(root, { recursive: true, force: true }) }) @@ -376,7 +378,7 @@ describe('a structured codex session over agentSession.*', () => { agent: 'codex' as const, providerHandle: { kind: 'codex' as const, threadId: THREAD } } - const journal = await openAgentSessionJournal({ + const journal = await journals.open({ identity, journalDir: journalDirectoryFor(root, identity) }) @@ -761,7 +763,7 @@ describe('a structured codex session over agentSession.*', () => { agent: 'codex' as const, providerHandle: { kind: 'codex' as const, threadId: THREAD } } - const reopened = await openAgentSessionJournal({ + const reopened = await journals.open({ identity, journalDir: journalDirectoryFor(root, identity) }) @@ -800,7 +802,6 @@ describe('a structured codex session over agentSession.*', () => { const item = journal.snapshot().items.find((candidate) => candidate.body?.kind === 'tool-call') const bounded = item?.body?.kind === 'tool-call' ? item.body.output : undefined expect(bounded).toMatchObject({ truncated: true, byteLength: Buffer.byteLength(output) }) - expect(await readJournalBlob(journal.directory, bounded?.digest ?? '')).toBe(output) }) it('keeps an answered prompt resolved after the provider exits', async () => { diff --git a/src/main/runtime/structured-agent-session-runtime.test.ts b/src/main/runtime/structured-agent-session-runtime.test.ts index 6adf5d368fd..b03d17cdb3f 100644 --- a/src/main/runtime/structured-agent-session-runtime.test.ts +++ b/src/main/runtime/structured-agent-session-runtime.test.ts @@ -2,6 +2,10 @@ import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { agentSessionJournalCloseRetries } from '../native-chat/agent-session-journal/journal-close-retry' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' +import type { AgentSessionJournal } from '../native-chat/agent-session-journal/journal-store' import type { AgentSessionClaimStatus, AgentSessionProcessIdentity, @@ -263,3 +267,68 @@ describe('structured agent-session runtime install', () => { ) }) }) + +// A stop whose teardown fails must not forget the runtime it was tearing down. +// `installing` is cleared either way so nothing new attaches, but the host keeps +// every journal whose close rejected, and this module slot is the only handle +// onto that host once it is gone. +describe('a teardown that fails is retried by the next stop', () => { + const JOURNAL_IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-teardown-retry', + workspaceId: 'ws-1', + hostId: HOST_ID, + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + } + const journals = createTrackedJournalOpener() + let directory: string | null = null + + afterEach(async () => { + await agentSessionJournalCloseRetries.retryAll() + await journals.closeAll() + await stopStructuredAgentSessionRuntime().catch(() => undefined) + if (directory) { + await rm(directory, { recursive: true, force: true }) + directory = null + } + }) + + it('reports the failure, then releases the handle on the following stop', async () => { + directory = await mkdtemp(join(tmpdir(), 'orca-structured-runtime-')) + await ensureStructuredAgentSessionHost({ + stateDirectory: directory, + hostId: HOST_ID, + claimKeyId: 'key-1', + resolveWorkspacePath: async () => directory!, + resolveEnvironment: async () => ({}), + reapOrphanChildren: async () => [] + }) + + const journalDir = join(directory, 'stubborn-journal') + const real = await journals.open({ identity: JOURNAL_IDENTITY, journalDir }) + let closeFailures = 2 + const flaky = new Proxy(real, { + get(target, property, receiver) { + if (property !== 'close') { + return Reflect.get(target, property, receiver) + } + return async () => { + if (closeFailures > 0) { + closeFailures -= 1 + throw new Error('close rejected') + } + await target.close() + } + } + }) as AgentSessionJournal + await agentSessionJournalCloseRetries.closeOrRetain(flaky) + + // The host's teardown runs the registry retry, so this stop surfaces it. + await expect(stopStructuredAgentSessionRuntime()).rejects.toThrow() + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([journalDir]) + + // The retained runtime is what makes this a retry rather than a no-op. + await stopStructuredAgentSessionRuntime() + expect(agentSessionJournalCloseRetries.pendingDirectories).toEqual([]) + }) +}) diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index ce916bc6769..52a3bd81818 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -78,6 +78,15 @@ type InstalledRuntime = { let installing: Promise | null = null +/** + * Runtimes whose teardown did not finish. `installing` is cleared regardless so + * nothing new attaches, but dropping the runtime as well would strand every + * journal the host retained for a retry: `tearDownStructuredAgentSessionHost` + * deliberately keeps a failed close indexed, and only a later stop through this + * same runtime can reach those entries again. + */ +const pendingTeardown = new Set() + export function ensureStructuredAgentSessionHost( deps: StructuredAgentSessionRuntimeDeps ): Promise { @@ -90,19 +99,40 @@ export function ensureStructuredAgentSessionHost( } /** Drops the host and reaps every Codex child under it. Runtime teardown and - * test isolation take the same path, so neither can leave a live app-server. */ + * test isolation take the same path, so neither can leave a live app-server. + * + * A teardown that fails is RETRIED by the next stop rather than forgotten: the + * host keeps every journal whose close rejected, and this is the only handle + * onto that host once the module slot is cleared. */ export async function stopStructuredAgentSessionRuntime(): Promise { const pending = installing installing = null setStructuredAgentSessionHost(null) agentSessionPtyWriteGate.detachRecordLookup() - if (!pending) { - return + const outstanding = [...pendingTeardown] + pendingTeardown.clear() + const installed = pending ? await pending.catch(() => null) : null + if (installed) { + outstanding.push(installed) } - const installed = await pending.catch(() => null) - if (!installed) { - return + const failures: unknown[] = [] + for (const runtime of outstanding) { + try { + await tearDownRuntime(runtime) + } catch (error) { + pendingTeardown.add(runtime) + failures.push(error) + } } + if (failures.length === 1) { + throw failures[0] + } + if (failures.length > 1) { + throw new AggregateError(failures, 'structured agent-session runtime teardown failed') + } +} + +async function tearDownRuntime(installed: InstalledRuntime): Promise { // Drain an in-flight recovery before stopping children; recovery may still // be writing lifecycle rows or acquiring a replacement child. await installed.waitForRecovery() diff --git a/src/main/runtime/worktree-launch-host-repo.ts b/src/main/runtime/worktree-launch-host-repo.ts index 4decb7acb56..7db9f4dae18 100644 --- a/src/main/runtime/worktree-launch-host-repo.ts +++ b/src/main/runtime/worktree-launch-host-repo.ts @@ -17,7 +17,10 @@ export type WorktreeHostRouting = | { kind: 'resolved'; hostId: ExecutionHostId; repo: T | null } /** No row carries this repo id and the worktree names no host — nothing ever named a host. */ | { kind: 'unowned' } - /** Rival rows disagree about the host; guessing one is the cross-host leak. */ + /** + * No single trustworthy host: rival rows disagree, or the resolved row named one that cannot be + * parsed. Guessing is the cross-host leak in both cases. + */ | { kind: 'ambiguous' } /** @@ -33,7 +36,10 @@ export function resolveWorktreeHostRouting { const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) if (resolution.kind === 'unresolved') { - return resolution.reason === 'ambiguous' ? { kind: 'ambiguous' } : { kind: 'unowned' } + // Only `unknown` — nothing anywhere carries the id — becomes `unowned`, which callers dispose of + // as a plain local folder. `malformed` is a row that declared a host and named an unparseable + // one, so it joins `ambiguous`: guessing is the cross-host leak either way. + return resolution.reason === 'unknown' ? { kind: 'unowned' } : { kind: 'ambiguous' } } return { kind: 'resolved', hostId: resolution.hostId, repo: resolution.owner } } diff --git a/src/main/runtime/worktree-listing-host-scope.ts b/src/main/runtime/worktree-listing-host-scope.ts index 99650882670..e499d7faaf3 100644 --- a/src/main/runtime/worktree-listing-host-scope.ts +++ b/src/main/runtime/worktree-listing-host-scope.ts @@ -1,4 +1,5 @@ -import type { ExecutionHostId } from '../../shared/execution-host' +import type { Repo } from '../../shared/repo-types' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' import { selectHostBalancedPage } from '../../shared/host-balanced-listing-page' import type { RuntimeListingHostScope } from '../../shared/runtime-listing-host-scope' @@ -62,3 +63,28 @@ export function buildWorktreeListingHostScope(args: { } return { hostIds: [...covered].sort(), omittedHostIds: [...omitted].sort() } } + +/** + * Which hosts a listing claims to have been looking at. + * + * A `--repo` listing was scoped by the caller, so naming every configured host would report gaps + * the caller deliberately excluded. Naming NONE — which is what a scoped listing did before — means + * the scope can never report a gap at all, for any host kind, because `covered` and `omitted` are + * both derived from the returned rows plus this list. A scoped listing whose scan did not succeed + * then answers `{hostIds: [], omittedHostIds: []}`: byte-identical to a repo that genuinely has no + * worktrees, which is the one thing docs/reference/ssh-execution-boundary.md forbids a listing from + * implying. + * + * Measured before the fix, on one runtime with one refusing SSH host, in the same second: the + * unscoped listing reported `omittedHostIds: ["local", "ssh:"]` while the scoped listing + * reported `[]`. + * + * Naming the single host the caller asked about costs nothing when rows come back — it lands in + * `covered`, so it is never reported omitted — and is the whole answer when they do not. + */ +export function listingKnownHostIds( + scopedRepo: Repo | null, + listKnownHostIds: () => Iterable +): Iterable { + return scopedRepo ? [getRepoExecutionHostId(scopedRepo)] : listKnownHostIds() +} diff --git a/src/main/sqlite/harden-database-files.ts b/src/main/sqlite/harden-database-files.ts new file mode 100644 index 00000000000..2183f87700e --- /dev/null +++ b/src/main/sqlite/harden-database-files.ts @@ -0,0 +1,18 @@ +import { chmodSync, existsSync } from 'node:fs' + +/** Restrict a SQLite database and its sidecars to the owning user. */ +export function hardenSqliteDatabaseFiles(dbPath: (string & {}) | ':memory:'): void { + if (dbPath === ':memory:' || process.platform === 'win32') { + // Why: Windows protects these files through Orca's current-user-only userData DACL; POSIX mode bits are inert there. + return + } + for (const path of [dbPath, `${dbPath}-wal`, `${dbPath}-shm`]) { + try { + if (existsSync(path)) { + chmodSync(path, 0o600) + } + } catch { + // Why: best-effort — a mount that rejects chmod (SSHFS, some network shares) must not fail DB startup. + } + } +} diff --git a/src/main/ssh/relay-daemon-service-children.ts b/src/main/ssh/relay-daemon-service-children.ts new file mode 100644 index 00000000000..35e755ea518 --- /dev/null +++ b/src/main/ssh/relay-daemon-service-children.ts @@ -0,0 +1,62 @@ +/** + * Telling a relay daemon's own service processes apart from the work it holds. + * + * The reap gate used to ask `pgrep -P | grep -c .` and demand zero. But the daemon + * forks service children of its own — `relay-ai-vault-service.js` is spawned lazily and then + * never exits — so that count is permanently non-zero on any relay that has touched the AI + * Vault, whether or not it holds a single PTY. A superseded, disconnected relay holding + * nothing therefore reported `retained-live-work` forever, its version directory stayed + * pinned against GC by its own live socket, and the population grew without bound (#13614). + * + * The asymmetry below is the whole safety argument, and it follows + * docs/reference/ssh-execution-boundary.md: *subtracting a child we can positively identify + * as relay infrastructure is sound; assuming anything about a child we cannot identify is + * not.* An argv that does not match, an argv `ps` would not print, and a host without + * `pgrep` all count against the relay and keep it unreapable. Losing sight of a child is + * never evidence that it holds nothing. + */ +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' +import { shellEscape } from './ssh-connection-utils' + +/** Shell variable set to the daemon's direct-child count, or `unknown`. */ +export const RELAY_CHILD_COUNT_VAR = 'kids' + +/** Shell variable set to the count of children not identified as relay services, or `unknown`. */ +export const RELAY_UNRECOGNIZED_CHILD_COUNT_VAR = 'unrecognized_kids' + +/** + * `case` patterns matching a service child's argv. Suffix-anchored on purpose: both entries + * are forked with no script arguments, so the argv ends at the filename, and the leading `/` + * requires the absolute path the daemon forks rather than a bare mention of the name. A + * future arg would stop matching and the relay would go back to being retained — the safe + * direction to fail in. + */ +function serviceChildArgvPatterns(): string { + return RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.map( + (filename) => `*${shellEscape(`/${filename}`)}` + ).join('|') +} + +/** + * POSIX shell that censuses the direct children of `$pid`, setting `kids` and + * `unrecognized_kids`. Both stay `unknown` when the host cannot enumerate children at all. + */ +export function relayDaemonChildCensusShell(): string[] { + return [ + `${RELAY_CHILD_COUNT_VAR}=unknown`, + `${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=unknown`, + 'if command -v pgrep >/dev/null 2>&1; then', + ` ${RELAY_CHILD_COUNT_VAR}=0`, + ` ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=0`, + ' for kid in $(pgrep -P "$pid" 2>/dev/null); do', + ` ${RELAY_CHILD_COUNT_VAR}=$((${RELAY_CHILD_COUNT_VAR}+1))`, + ' kid_args=$(ps -o args= -p "$kid" 2>/dev/null | tr -d "\\n")', + ' case "$kid_args" in', + ` ${serviceChildArgvPatterns()}) ;;`, + // An unreadable or unrecognised argv lands here, which is what keeps the relay retained. + ` *) ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=$((${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}+1)) ;;`, + ' esac', + ' done', + 'fi' + ] +} diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index 257eb984caa..e7450478d92 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -743,6 +743,7 @@ function uploadStageNamespaceIfSupported( const NODE_PTY_VERSION = '1.1.0' const NODE_PTY_CONSOLE_LIST_PATCH_FILENAME = 'node-pty-1.1.0-console-list-agent-patch.cjs' +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME = 'node-pty-1.1.0-windows-pty-teardown-patch.cjs' const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' const NODE_PTY_CLOEXEC_STATUS_PREFIX = 'ORCA-NPTY-CLOEXEC:' /** @@ -791,7 +792,8 @@ function nativeDepsProbeJs(successToken: string): string { // Why: node-pty's Windows wrapper defers conpty.node until first spawn, so require("node-pty") alone can't prove the binding is healthy. const loadNodePty = 'require("node-pty"); require("node-pty/lib/utils").loadNativeModule(process.platform==="win32"&&Number(require("os").release().split(".")[2])>=18309?"conpty":"pty");' + - `if(process.platform==="win32"){require("./${NODE_PTY_CONSOLE_LIST_PATCH_FILENAME}").assertPatchedNodePtyConsoleListAgent(process.cwd())}` + `if(process.platform==="win32"){require("./${NODE_PTY_CONSOLE_LIST_PATCH_FILENAME}").assertPatchedNodePtyConsoleListAgent(process.cwd());` + + `require("./${NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME}").assertPatchedNodePtyWindowsTeardown(process.cwd())}` return `(()=>{const missing=[];try{${loadNodePty}}catch{missing.push("node-pty")}try{require("@parcel/watcher")}catch{missing.push("@parcel/watcher")}if(missing.length){console.log("${NATIVE_DEPS_MISSING_PREFIX}"+missing.join(","));process.exitCode=1}else{console.log(${JSON.stringify(successToken)})}})()` } @@ -1327,10 +1329,16 @@ async function applyNodePtyMasterCloexecPatch( nodePath: string, signal?: AbortSignal ): Promise { - // Both Unix relay platforms leak, by different bugs: Linux inherits the master through forkpty()'s - // no-O_CLOEXEC path, macOS orphans one throwaway /dev/ptmx fd per spawn in pty_posix_spawn. Only - // Windows, which has no fds, is short-circuited -- and answering 'fixed' from a gate that ran - // nothing is exactly how a leaking darwin tree got published to the shared cache. + // Both Unix relay platforms leak the pty master, by different bugs: Linux inherits it through + // forkpty()'s no-O_CLOEXEC path, macOS orphans one throwaway /dev/ptmx fd per spawn in + // pty_posix_spawn. Windows is short-circuited because it has no fds for a master to leak into -- + // and answering 'fixed' from a gate that ran nothing is exactly how a leaking darwin tree got + // published to the shared cache. + // + // What 'fixed' means here is exactly "this tree does not leak the pty MASTER", which is the only + // thing the shared native-deps cache keys on. It is NOT a statement that a Windows relay leaks + // nothing: it leaked one Windows File handle per terminal until the ConPTY teardown patch above, + // by a mechanism that has nothing to do with fds. Read this gate as scoped to its own question. if (isWindowsRemoteHost(hostPlatform) || isWindowsRelayPlatform(platform)) { return 'fixed' } @@ -1530,8 +1538,11 @@ async function rebuildNativeDeps( } function windowsNodePtyPatchCommand(nodePath: string): string { - // Why: pnpm patches do not cross the SSH boundary; apply the version-checked fallback to the remote npm package. - return `& ${powerShellLiteral(nodePath)} ${powerShellLiteral(NODE_PTY_CONSOLE_LIST_PATCH_FILENAME)}; if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }` + // Why: pnpm patches do not cross the SSH boundary; apply the version-checked fallbacks to the remote npm package. + return [ + `& ${powerShellLiteral(nodePath)} ${powerShellLiteral(NODE_PTY_CONSOLE_LIST_PATCH_FILENAME)}; if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }`, + `& ${powerShellLiteral(nodePath)} ${powerShellLiteral(NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME)}; if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }` + ].join('; ') } async function makeNodePtySpawnHelperExecutable( diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts index 7ece0b6532e..a8975d0520b 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts @@ -4,7 +4,7 @@ * generated scripts through /bin/sh against real unix sockets and real processes. */ import { execFile, spawn, type ChildProcess } from 'node:child_process' -import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterAll, afterEach, beforeAll, describe, expect, it } from 'vitest' @@ -15,21 +15,32 @@ import { type RelayEndpointIncumbent } from './ssh-relay-endpoint-incumbent' import { reapEmptyRelayHuskCommand } from './ssh-relay-endpoint-takeover' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' const posixOnly = process.platform === 'win32' ? describe.skip : describe const FAKE_RELAY_SOURCE = ` const net = require('net') +const path = require('path') const sock = process.argv[process.argv.indexOf('--sock-path') + 1] +function spawnChild(args) { + require('child_process').spawn(process.execPath, args, { stdio: 'ignore' }) +} if (process.argv.includes('--with-child')) { - require('child_process').spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000)'], { - stdio: 'ignore' - }) + spawnChild(['-e', 'setTimeout(() => {}, 60000)']) +} +// Why forked the same way production does: the exclusion is argv-shaped, so a hand-written +// stand-in would test the test rather than the shell that runs on someone's host. +for (const name of process.argv.filter((arg) => arg.startsWith('--service-child='))) { + spawnChild([path.join(__dirname, name.slice('--service-child='.length))]) } net.createServer(() => {}).listen(sock, () => process.stdout.write('READY\\n')) process.on('SIGTERM', () => process.exit(0)) ` +// Self-limiting: these are orphaned when the relay under test is reaped. +const IDLE_SERVICE_SOURCE = 'setTimeout(() => {}, 60000)\n' + function sh(script: string): Promise { return new Promise((resolve, reject) => { execFile('/bin/sh', ['-c', script], { timeout: 20_000 }, (error, stdout) => { @@ -43,14 +54,21 @@ function sh(script: string): Promise { } let workDir: string +let pgreplessBinDir: string let hasLsof = false const running: ChildProcess[] = [] -function startFakeRelay(sockPath: string, withChild = false): Promise { +function startFakeRelay( + sockPath: string, + options: { withChild?: boolean; serviceChildren?: readonly string[] } = {} +): Promise { const args = [join(workDir, 'relay.js'), '--sock-path', sockPath] - if (withChild) { + if (options.withChild) { args.push('--with-child') } + for (const name of options.serviceChildren ?? []) { + args.push(`--service-child=${name}`) + } const child = spawn(process.execPath, args, { stdio: ['ignore', 'pipe', 'ignore'] }) running.push(child) return new Promise((resolve, reject) => { @@ -68,9 +86,31 @@ async function probe(sockPath: string): Promise { return parseRelayEndpointIncumbentProbe(sockPath, output) } +/** The relay forks its children after it starts listening, so the probe can race them. */ +async function waitForChildCount( + sockPath: string, + expected: number +): Promise { + let incumbent = await probe(sockPath) + for (let attempt = 0; attempt < 50 && incumbent.holders[0]?.childCount !== expected; attempt++) { + await new Promise((resolve) => setTimeout(resolve, 100)) + incumbent = await probe(sockPath) + } + return incumbent +} + beforeAll(async () => { workDir = mkdtempSync(join(tmpdir(), 'orca-relay-incumbent-')) writeFileSync(join(workDir, 'relay.js'), FAKE_RELAY_SOURCE) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + writeFileSync(join(workDir, filename), IDLE_SERVICE_SOURCE) + } + writeFileSync(join(workDir, 'looks-like-relay-watcher.js'), IDLE_SERVICE_SOURCE) + pgreplessBinDir = join(workDir, 'pgrepless-bin') + mkdirSync(pgreplessBinDir) + for (const tool of ['ps', 'tr']) { + symlinkSync((await sh(`command -v ${tool}`)).trim(), join(pgreplessBinDir, tool)) + } hasLsof = await sh('command -v lsof >/dev/null 2>&1 && echo yes || echo no').then( (out) => out.trim() === 'yes' ) @@ -105,13 +145,51 @@ posixOnly('relay endpoint probe against a real socket', () => { return } expect(incumbent.holders.map((holder) => holder.pid)).toEqual([relay.pid]) - expect(incumbent.holders[0]).toMatchObject({ matchesRelayArgv: true, childCount: 0 }) + expect(incumbent.holders[0]).toMatchObject({ + matchesRelayArgv: true, + childCount: 0, + unrecognizedChildCount: 0 + }) expect(isReapableRelayHusk(incumbent)).toBe(true) }) + it("counts the daemon's own service children but does not hold them against it", async () => { + const sockPath = join(workDir, 'services.sock') + await startFakeRelay(sockPath, { serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES }) + const incumbent = await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + + expect(incumbent.holders[0].childCount).toBe(RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + expect(incumbent.holders[0].unrecognizedChildCount).toBe(0) + expect(isReapableRelayHusk(incumbent)).toBe(true) + }) + + it('still retains a relay holding work alongside its service children', async () => { + const sockPath = join(workDir, 'services-and-work.sock') + await startFakeRelay(sockPath, { + withChild: true, + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + const incumbent = await waitForChildCount( + sockPath, + RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length + 1 + ) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + + it('does not excuse a child that merely mentions a service entry name', async () => { + const sockPath = join(workDir, 'lookalike.sock') + await startFakeRelay(sockPath, { serviceChildren: ['looks-like-relay-watcher.js'] }) + const incumbent = await waitForChildCount(sockPath, 1) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + it('refuses to call a relay with a live child an empty husk', async () => { const sockPath = join(workDir, 'busy.sock') - await startFakeRelay(sockPath, true) + await startFakeRelay(sockPath, { withChild: true }) const incumbent = await probe(sockPath) expect(incumbent.verdict).toBe('live') @@ -150,12 +228,34 @@ posixOnly('empty relay husk reap against a real process', () => { it('refuses to signal a relay that acquired a child after it was probed', async () => { const sockPath = join(workDir, 'raced.sock') - const relay = await startFakeRelay(sockPath, true) + const relay = await startFakeRelay(sockPath, { withChild: true }) const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) expect(output.trim()).toBe('BUSY') expect(relay.killed).toBe(false) }) + it('terminates a relay whose only children are its own service processes (#13614)', async () => { + const sockPath = join(workDir, 'service-husk.sock') + const relay = await startFakeRelay(sockPath, { + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) + expect(output.trim()).toBe('GONE') + }) + + it('refuses to signal when the host cannot enumerate children at all', async () => { + const sockPath = join(workDir, 'no-pgrep.sock') + const relay = await startFakeRelay(sockPath) + // A PATH carrying every tool the script needs except `pgrep`: the census answers + // `unknown`, which must reach BUSY rather than the zero a missing tool would imply. + const output = await sh( + `PATH=${pgreplessBinDir}\n${reapEmptyRelayHuskCommand(relay.pid!, sockPath)}` + ) + expect(output.trim()).toBe('BUSY') + expect(relay.killed).toBe(false) + }) + it('refuses to signal a pid whose argv is not this relay at this socket', async () => { const sockPath = join(workDir, 'mismatch.sock') await startFakeRelay(sockPath) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts index a65cb33fc57..de4cc28d170 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts @@ -32,17 +32,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('reports live when the socket accepted a connection', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=4242 yes 13']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=4242 yes 13 11' + ]) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('accepted-connection') - expect(incumbent.holders).toEqual([{ pid: 4242, matchesRelayArgv: true, childCount: 13 }]) + expect(incumbent.holders).toEqual([ + { pid: 4242, matchesRelayArgv: true, childCount: 13, unrecognizedChildCount: 11 } + ]) }) it('reports live when a process still holds an inode that refuses connections', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2']) + probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2 2']) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('holder-process') @@ -85,7 +92,12 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('drops holder lines that do not carry a usable pid', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=- no unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=refused', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=- no unknown unknown' + ]) ) expect(incumbent.holders).toEqual([]) expect(incumbent.verdict).toBe('exited') @@ -94,9 +106,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('keeps an unreadable child count as null rather than zero', () => { const [holder] = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=7 yes unknown unknown' + ]) ).holders expect(holder.childCount).toBeNull() + expect(holder.unrecognizedChildCount).toBeNull() + }) + + it('keeps a holder line with no unrecognized-child field unreapable', () => { + const incumbent = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes 0']) + ) + expect(incumbent.holders[0].unrecognizedChildCount).toBeNull() + expect(isReapableRelayHusk(incumbent)).toBe(false) }) }) @@ -172,27 +199,36 @@ describe('mayLaunchOverRelayEndpoint', () => { describe('isReapableRelayHusk', () => { const husk = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0']) + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0 0']) ) - it('accepts a single proven relay holder with zero children', () => { + it('accepts a single proven relay holder with no unaccounted-for children', () => { expect(isReapableRelayHusk(husk)).toBe(true) }) - it('refuses a relay that still holds children', () => { + it('accepts a relay whose only children are its own service processes (#13614)', () => { + const withServices = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 2 0']) + ) + expect(withServices.holders[0].childCount).toBe(2) + expect(isReapableRelayHusk(withServices)).toBe(true) + }) + + it('refuses a relay that still holds children it could not account for', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: 1 }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 3, unrecognizedChildCount: 1 }] }) ).toBe(false) }) - it('refuses a holder whose child count could not be read', () => { + it('refuses a holder whose unrecognized-child count could not be read', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: null }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: null }] }) ).toBe(false) }) @@ -201,7 +237,7 @@ describe('isReapableRelayHusk', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0 }] + holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0, unrecognizedChildCount: 0 }] }) ).toBe(false) }) @@ -211,8 +247,8 @@ describe('isReapableRelayHusk', () => { isReapableRelayHusk({ ...husk, holders: [ - { pid: 500, matchesRelayArgv: true, childCount: 0 }, - { pid: 501, matchesRelayArgv: true, childCount: 0 } + { pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 }, + { pid: 501, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 } ] }) ).toBe(false) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.ts index 2688267f4c7..628a9558793 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.ts @@ -20,6 +20,11 @@ */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_CHILD_COUNT_VAR, + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { isWindowsRemoteHost, type RemoteHostPlatform } from './ssh-remote-platform' @@ -38,6 +43,12 @@ export type RelayEndpointHolder = { matchesRelayArgv: boolean /** Direct children, or null when `pgrep` could not answer. Never guessed. */ childCount: number | null + /** + * Direct children *not* positively identified as the daemon's own service processes, or + * null when the host could not enumerate them. This — not `childCount` — is what says + * whether the relay holds anything; see relay-daemon-service-children.ts. + */ + unrecognizedChildCount: number | null } export type RelayEndpointIncumbent = { @@ -95,11 +106,9 @@ export function relayEndpointIncumbentProbeCommand(nodePath: string, sockPath: s ' args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', ' match=no', ' case "$args" in *relay.js*"$sock"*) match=yes ;; esac', - ' kids=unknown', - ' if command -v pgrep >/dev/null 2>&1; then', - ' kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - ' fi', - ' printf \'HOLDER=%s %s %s\\n\' "$pid" "$match" "$kids"', + ...relayDaemonChildCensusShell().map((line) => ` ${line}`), + ' printf \'HOLDER=%s %s %s %s\\n\' "$pid" "$match" ' + + `"$${RELAY_CHILD_COUNT_VAR}" "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}"`, ' done', 'else', " printf 'HOLDERS_SOURCE=unavailable\\n'", @@ -159,19 +168,25 @@ export function parseRelayEndpointIncumbentProbe( } function parseHolder(value: string): RelayEndpointHolder | null { - const [rawPid, rawMatch, rawKids] = value.split(/\s+/) + const [rawPid, rawMatch, rawKids, rawUnrecognized] = value.split(/\s+/) const pid = Number.parseInt(rawPid ?? '', 10) if (!Number.isInteger(pid) || pid <= 0) { return null } - const childCount = Number.parseInt(rawKids ?? '', 10) return { pid, matchesRelayArgv: rawMatch === 'yes', - childCount: Number.isInteger(childCount) && childCount >= 0 ? childCount : null + childCount: parseChildCount(rawKids), + unrecognizedChildCount: parseChildCount(rawUnrecognized) } } +/** `unknown`, a missing field, and anything unparseable are all "could not tell" — never 0. */ +function parseChildCount(raw: string | undefined): number | null { + const count = Number.parseInt(raw ?? '', 10) + return Number.isInteger(count) && count >= 0 ? count : null +} + function unverifiableEndpoint(sockPath: string): RelayEndpointIncumbent { return { sockPath, @@ -239,8 +254,12 @@ export function mayLaunchOverRelayEndpoint(incumbent: RelayEndpointIncumbent): b /** * A live relay that provably holds nothing: identity confirmed against its argv, exactly one - * holder, and zero children. Reaping it destroys no user work. Anything less is retained — - * killing the wrong pid on someone's remote host is the worst outcome available here. + * holder, and no child the host could not account for as one of the daemon's own service + * processes. Reaping it destroys no user work. Anything less is retained — killing the wrong + * pid on someone's remote host is the worst outcome available here. + * + * Why not `childCount === 0`: the daemon's AI Vault sidecar never exits once spawned, so that + * gate was unreachable for any relay that had ever served a vault request (#13614). */ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean { if (incumbent.verdict !== 'live' || !incumbent.holdersEnumerable) { @@ -250,12 +269,16 @@ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean return false } const [holder] = incumbent.holders - return holder.matchesRelayArgv && holder.childCount === 0 + return holder.matchesRelayArgv && holder.unrecognizedChildCount === 0 } export function describeRelayEndpointIncumbent(incumbent: RelayEndpointIncumbent): string { const holders = incumbent.holders - .map((holder) => `${holder.pid}(children=${holder.childCount ?? 'unknown'})`) + .map( + (holder) => + `${holder.pid}(children=${holder.childCount ?? 'unknown'},` + + `unrecognized=${holder.unrecognizedChildCount ?? 'unknown'})` + ) .join(',') return ( `${incumbent.sockPath} verdict=${incumbent.verdict} evidence=${incumbent.evidence} ` + diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts index 687d633b92b..d23f065f478 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts @@ -14,6 +14,7 @@ import { resolveRelayEndpointBeforeRelaunch } from './ssh-relay-endpoint-takeover' import { RelayVersionMismatchError } from './ssh-relay-version-mismatch-error' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' import type { SshConnection } from './ssh-connection' import { getRemoteHostPlatform } from './ssh-remote-platform' @@ -42,7 +43,7 @@ beforeEach(() => { describe('incumbent alive and refusing', () => { it('refuses to rebind a live relay holding PTYs, and signals nothing', async () => { execCommand.mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) // The whole point of #8585: the incumbent's socket must survive so it is not orphaned. @@ -52,9 +53,9 @@ describe('incumbent alive and refusing', () => { it('names the incumbent pid and the Reset Relay escape hatch in the error', async () => { execCommand.mockResolvedValue( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) - await expect(resolve()).rejects.toThrow(/3669803\(children=13\)/) + await expect(resolve()).rejects.toThrow(/3669803\(children=13,unrecognized=11\)/) await expect(resolve()).rejects.toThrow(/Reset Relay/) }) @@ -70,7 +71,7 @@ describe('incumbent alive and refusing', () => { it('reaps a live relay only when it provably holds nothing, and confirms it is gone', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') await expect(resolve()).resolves.toMatchObject({ verdict: 'live' }) @@ -80,7 +81,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over an empty relay whose death could not be confirmed', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -89,7 +90,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over a relay the host refused to signal on its own re-check', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('BUSY\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -138,9 +139,19 @@ describe('reapEmptyRelayHuskCommand', () => { }) it('aborts without signalling when the host cannot count children', () => { - expect(reapEmptyRelayHuskCommand(4242, SOCK)).toContain( - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }" - ) + const command = reapEmptyRelayHuskCommand(4242, SOCK) + // The census leaves both counters at `unknown` without pgrep, and the gate demands "0". + expect(command).toContain('unrecognized_kids=unknown') + expect(command).toContain('command -v pgrep >/dev/null 2>&1') + expect(command).toContain('[ "$unrecognized_kids" = "0" ] ||') + }) + + it('subtracts only the daemon service children it can name from the reap gate', () => { + const command = reapEmptyRelayHuskCommand(4242, SOCK) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + expect(command).toContain(`*'/${filename}'`) + } + expect(command).toContain('unrecognized_kids=$((unrecognized_kids+1))') }) }) diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.ts b/src/main/ssh/ssh-relay-endpoint-takeover.ts index f104aab5256..8f6130620cb 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.ts @@ -2,13 +2,17 @@ * Deciding whether a relay socket path is ours to take, and acting on the answer. * * The only destructive action available here is a SIGTERM to a relay that has been proven — - * by argv, by socket-holder enumeration, and by a zero child count re-checked on the host - * immediately before the signal — to hold nothing at all. Everything else is left running. + * by argv, by socket-holder enumeration, and by a child census re-run on the host immediately + * before the signal — to hold nothing at all. Everything else is left running. * Per docs/reference/ssh-execution-boundary.md, a relay we merely failed to reach is * `unverifiable`, and `unverifiable` never authorizes a kill or a rebind. */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { describeRelayEndpointIncumbent, @@ -39,9 +43,10 @@ export function reapEmptyRelayHuskCommand(pid: number, sockPath: string): string `sock=${shellEscape(sockPath)}`, 'args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', 'case "$args" in *relay.js*"$sock"*) ;; *) printf \'MISMATCH\\n\'; exit 0 ;; esac', - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }", - 'kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - '[ "$kids" = "0" ] || { printf \'BUSY\\n\'; exit 0; }', + // Why the same census as the probe: `unknown` (no pgrep) and any child this host could + // not account for as a relay service both land on BUSY, so nothing is signalled. + ...relayDaemonChildCensusShell(), + `[ "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}" = "0" ] || { printf 'BUSY\\n'; exit 0; }`, // SIGTERM only: the relay's own handler disposes and unlinks. SIGKILL would leave the // socket inode behind and skip that shutdown path for no gain on an empty daemon. 'kill -TERM "$pid" 2>/dev/null || true', diff --git a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts index 874d9aae3fe..9168f4688bd 100644 --- a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts +++ b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts @@ -67,7 +67,7 @@ describe('classifySupersededRelay', () => { 'PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', - 'HOLDER=3669803 yes 13' + 'HOLDER=3669803 yes 13 11' ]) ) ).toBe('retained-live-work') @@ -76,7 +76,7 @@ describe('classifySupersededRelay', () => { it('nominates only a proven empty relay for reaping', () => { expect( classifySupersededRelay( - incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) ).toBe('reap-candidate') }) @@ -101,7 +101,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) expect(findings).toHaveLength(1) @@ -114,7 +114,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) @@ -126,7 +126,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) diff --git a/src/main/ssh/ssh-remote-powershell.ts b/src/main/ssh/ssh-remote-powershell.ts index 8c94fd3c483..420223ced29 100644 --- a/src/main/ssh/ssh-remote-powershell.ts +++ b/src/main/ssh/ssh-remote-powershell.ts @@ -11,14 +11,24 @@ export { // to leave room for the `/c` wrapper sshd adds before cmd.exe counts the line. const WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS = 8_000 -export function powerShellCommand(script: string): string { - const inline = encodedPowerShellCommand(script) +/** + * `pwsh.exe` is PowerShell 7. It is not present on a stock Windows install, so it is only ever + * chosen after a probe — but where it exists it reads a redirected stdin correctly, which Windows + * PowerShell 5.1 does not (see `system-ssh-file-binary-transfer.ts`). + */ +export type WindowsPowerShellExecutable = 'powershell.exe' | 'pwsh.exe' + +export function powerShellCommand( + script: string, + executable: WindowsPowerShellExecutable = 'powershell.exe' +): string { + const inline = encodedPowerShellCommand(script, executable) if (inline.length <= WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { return inline } // Why: these scripts are repetitive enough that gzip beats the UTF-16LE tax by // ~4x, which is the difference between a line cmd.exe runs and one it refuses. - const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script)) + const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script), executable) if (compressed.length > WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { throw new Error( `Remote Windows command needs ${compressed.length} characters; Orca budgets ${WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS} for a line sshd hands to cmd.exe, which itself refuses more than ${CMD_EXE_COMMAND_LINE_MAX_CHARS}.` @@ -27,8 +37,8 @@ export function powerShellCommand(script: string): string { return compressed } -function encodedPowerShellCommand(script: string): string { - return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` +function encodedPowerShellCommand(script: string, executable: WindowsPowerShellExecutable): string { + return `${executable} -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` } /** Orca-prefixed names so the payload can never shadow the bootstrap's own state. */ diff --git a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts index c104078238e..fa34a8fbda7 100644 --- a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts +++ b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts @@ -4,6 +4,10 @@ import { CMD_EXE_COMMAND_LINE_MAX_CHARS } from '../providers/windows-shell-args' import { getRemoteHostPlatform } from './ssh-remote-platform' import { tryStealInstallLockCommand } from './ssh-relay-install-lock-commands' import { decodeRemotePowerShellScript, powerShellCommand } from './ssh-remote-powershell' +import { + makeWindowsPublishStagedFileCommand, + makeWindowsWriteFileCommand +} from './system-ssh-windows-file-write' import { cleanupOwnedRelayUploadStageCommand, promoteOwnedRelayUploadStageCommand, @@ -38,6 +42,17 @@ describe('Windows remote command line limit', () => { [ 'steal stale install lock', tryStealInstallLockCommand(windows, 'C:\\Users\\orca\\.orca-remote\\relay', 1_200) + ], + // F11 flagged these two as uncovered. They carry one path literal each, so they are the file + // commands whose length a caller can actually move. + ['write file', makeWindowsWriteFileCommand('C:\\Users\\orca\\.orca-remote\\relay.js')], + [ + 'publish staged file', + makeWindowsPublishStagedFileCommand( + 'C:\\Users\\orca\\.orca-remote\\relay.js.orca-partial-0123456789ab', + 'C:\\Users\\orca\\.orca-remote\\relay.js', + 'create' + ) ] ])('keeps the %s command inside what sshd\u2019s cmd.exe accepts', (_name, command) => { expect(command.length).toBeLessThanOrEqual(CMD_EXE_COMMAND_LINE_MAX_CHARS) @@ -76,3 +91,32 @@ describe('Windows remote command line limit', () => { ) }) }) + +/** + * F11 asked whether a pathological path could reach the budget, and what happens if it does. + * Measured: the inline encoding crosses 8000 at roughly 2500 high-entropy path characters — an + * order of magnitude past what Windows itself accepts — and the failure is a throw before any ssh + * is spawned, never a hang. + */ +describe('Windows file command budget headroom', () => { + it('absorbs a path far longer than Windows will accept', () => { + const deep = `C:\\Users\\orca\\${'segment\\'.repeat(30)}relay.js` + + expect(deep.length).toBeGreaterThan(260) + expect(makeWindowsWriteFileCommand(deep).length).toBeLessThanOrEqual( + CMD_EXE_COMMAND_LINE_MAX_CHARS + ) + }) + + it('throws rather than spawning a line cmd.exe would refuse', () => { + // Random segments so gzip cannot rescue it, which is the only way to reach the ceiling at all. + const incompressible = Array.from( + { length: 400 }, + (_unused, index) => `${index}-${Math.random().toString(36).slice(2)}` + ).join('\\') + + expect(() => makeWindowsWriteFileCommand(`C:\\${incompressible}\\f.bin`)).toThrow( + /Orca budgets 8000/ + ) + }) +}) diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index c366e899bf6..b477ad682ef 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -709,9 +709,13 @@ describe('spawnSystemSsh', () => { expect(args[standaloneControlIdx + 1]).toBe('none') }) - it('writes files to Windows system SSH targets with PowerShell stdin bytes', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('sends Windows file writes over sftp, not through a remote PowerShell stdin', async () => { + const spawned: EventedProcess[] = [] + spawnMock.mockImplementation(() => { + const proc = createEventedProcess() + spawned.push(proc) + return closeOnceSpawned(proc) + }) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -722,16 +726,24 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('0.1.0', 'utf-8')) + // #16432, re-measured: Windows PowerShell 5.1 can lose a redirected stdin for good when a read + // finds it momentarily empty, so the bytes must not travel that way at all. + const batch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(batch).toContain('put ') + expect(batch).toContain('/C:/Users/me/.orca-remote/relay/.version.orca-partial-') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + expect(sftpArgs).toContain('-b') + // The rename that publishes it reads the staged file, never a pipe. + const publish = (spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '' + expect(publish).toContain('powershell.exe') + expect(decodePowerShellCommand(publish)).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + expect(publish).not.toContain('/bin/sh') }) - it('writes binary buffers to Windows system SSH targets with CreateNew mode', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('enforces an exclusive Windows buffer write at the rename, where it is atomic', async () => { + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeBufferViaSystemSsh( @@ -742,12 +754,11 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(decodePowerShellCommand(remoteCommand)).toContain('CreateNew') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('png')) + const publish = decodePowerShellCommand((spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '') + // `File::Move` raising on an existing destination is what carries the exclusive contract now; + // a `CreateNew` on the staged file would only refuse a leftover of our own. + expect(publish).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish).not.toContain('[System.IO.File]::Delete($path)') }) it('downloads files from Windows system SSH targets with PowerShell stdout bytes', async () => { @@ -779,8 +790,7 @@ describe('spawnSystemSsh', () => { }) it('forces standalone SSH for Windows file writes when requested', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -791,10 +801,14 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + // sftp's own `-S` names a program to run, so the same request has to be spelled as an option. + expect(sftpArgs).not.toContain('-S') + expect(sftpArgs).toContain('ControlPath=none') + const publishArgs = spawnMock.mock.calls[1][1] as string[] + const standaloneControlIdx = publishArgs.indexOf('-S') expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + expect(publishArgs[standaloneControlIdx + 1]).toBe('none') }) it('uploads a Windows directory as a mkdir batch plus per-file writes, never one blob', async () => { @@ -819,19 +833,20 @@ describe('spawnSystemSsh', () => { rmSync(localDir, { recursive: true, force: true }) } + // #16432: directories first, then the file — but both over sftp now, so the only PowerShell + // left is the rename that publishes the staged file, which reads a file rather than a pipe. + const mkdirBatch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(mkdirBatch).toBe('-mkdir "/C:/Users/me/.orca-remote/relay"\n') + const putBatch = String(spawned[1]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(putBatch).toContain('put ') + expect(putBatch).toContain('/C:/Users/me/.orca-remote/relay/relay.js.orca-partial-') const commands = spawnMock.mock.calls.map((call) => (call[1] as string[]).at(-1) ?? '') - // #16432: directories first (metadata only), then the file bytes on their own stdin. One batch - // meant base64-ing the whole bundle into a single PowerShell string, which the remote never read. - expect(commands).toHaveLength(2) - expect(commands.every((command) => command.includes('powershell.exe'))).toBe(true) expect(commands.every((command) => !command.includes('/bin/sh'))).toBe(true) expect(commands.join('\n')).not.toContain('tar -xzf') - expect(JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string)).toEqual([ - 'C:/Users/me/.orca-remote/relay' - ]) - expect(Buffer.from(spawned[1].stdin.end.mock.calls[0]?.[0] as Buffer).toString('utf-8')).toBe( - 'console.log("relay")' - ) + // Nothing base64s the bundle into one PowerShell string any more, and nothing reads one. + expect( + commands.some((command) => decodePowerShellCommand(command).includes('OpenStandardInput')) + ).toBe(false) }) it('forces standalone SSH for Windows upload packages when requested', async () => { @@ -855,9 +870,9 @@ describe('spawnSystemSsh', () => { } const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') - expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + // The first spawn is the sftp client, whose own `-S` names a program to run. + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') }) it('throws when no system ssh is found', () => { diff --git a/src/main/ssh/system-ssh-file-binary-transfer.ts b/src/main/ssh/system-ssh-file-binary-transfer.ts index b0c5b662ed1..149d10dfbd3 100644 --- a/src/main/ssh/system-ssh-file-binary-transfer.ts +++ b/src/main/ssh/system-ssh-file-binary-transfer.ts @@ -1,5 +1,7 @@ import { constants, createWriteStream } from 'node:fs' -import { lstat, open } from 'node:fs/promises' +import { lstat, mkdtemp, open, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import type { Writable } from 'node:stream' import { pipeline } from 'node:stream/promises' import type { SshTarget } from '../../shared/ssh-types' @@ -16,6 +18,16 @@ import { throwIfAborted, waitForChannelClose } from './system-ssh-operation-lifecycle' +import { + writeWindowsRemoteFile, + type WindowsWriteSource +} from './system-ssh-windows-write-strategy' + +export { + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS +} from './system-ssh-windows-write-strategy' +export { WINDOWS_STAGED_WRITE_SUFFIX } from './system-ssh-windows-file-write' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -74,13 +86,16 @@ export async function writeBufferViaSystemSsh( ): Promise { throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - await writeWindowsBytesViaSystemSsh( + await writeWindowsRemoteFile( target, remotePath, - contents.length, - (offset, maxBytes) => - Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), - options + { + totalBytes: contents.length, + readChunk: (offset, maxBytes) => + Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), + withLocalFile: (send) => withTemporaryLocalFile(contents, send) + }, + options ?? {} ) return } @@ -127,20 +142,19 @@ export async function uploadFileViaSystemSsh( throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - // #16432: a Windows host cannot take a whole file through one stdin, however the local side - // paces it — see WINDOWS_STDIN_WRITE_CHUNK_BYTES. This is the path that carries the large - // files, so it is the one that has to be chunked and bounded. - await writeWindowsBytesViaSystemSsh( - target, - remotePath, - openedStat.size, - async (offset, maxBytes) => { + // This is the path that carries the large files, so it is the one the transport choice is + // made for; see the #16432 note below. + const source: WindowsWriteSource = { + totalBytes: openedStat.size, + readChunk: async (offset, maxBytes) => { const buffer = Buffer.allocUnsafe(Math.min(maxBytes, openedStat.size - offset)) const { bytesRead } = await handle.read(buffer, 0, buffer.length, offset) return buffer.subarray(0, bytesRead) }, - options - ) + // The verified local file is already exactly the payload, so sftp sends it as is. + withLocalFile: (send) => send(localPath) + } + await writeWindowsRemoteFile(target, remotePath, source, options ?? {}) return } @@ -173,158 +187,56 @@ export async function uploadFileViaSystemSsh( } /** - * #16432: Windows PowerShell 5.1 stops draining a redirected stdin over a non-pty ssh exec - * somewhere between 50KB and 1MB, depending on the host's `DefaultShell`, and it hangs rather than - * failing. The reporter measured that on both constructs he tried — `[Console]::In.ReadToEnd()` and - * `new IO.StreamReader([Console]::OpenStandardInput())`, the latter reading incrementally, which is - * why the limit cannot be attributed to materializing the payload. `Stream.CopyTo` reads the same - * `[Console]::OpenStandardInput()` object with the same incremental `Read` loop, so nothing in it - * escapes that limit either: no single write may exceed what one stdin is known to carry. + * #16432, re-measured: the constraint is not a size limit, and it is not cmd.exe's. * - * 32KB is an order of magnitude under the low end of the measured range, and under 50KB, which the - * reporter measured succeeding against a stream reader on the worse of the two `DefaultShell` - * settings. - */ -export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 - -/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ -export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 - -/** Suffix for the path a multi-exec Windows write lands on before it is published by rename. */ -export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' - -/** - * Splits one logical Windows write into stdin-sized execs. + * A read on Windows PowerShell 5.1's redirected-stdin handle over a non-pty ssh exec can die + * permanently when it finds the stream momentarily empty: no further bytes arrive, and no EOF ever + * does. It is probabilistic per such read — not a size threshold, and not certain on the first one. + * Measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2 with `DefaultShell = cmd.exe`, by + * replacing the copy loop with a counting reader: * - * A write that needs more than one exec cannot land on the destination directly: a chunk failing - * mid-file would leave a truncated artifact under the real name with nothing marking it incomplete, - * and the retry would then meet its own leftovers — under `exclusive` the retry's `CreateNew` fails - * on them. Multi-exec creates therefore land on a staging path and are published by a rename, which - * is also where `exclusive` is enforced: once, at the destination, instead of smeared across the - * first chunk. A caller-requested append cannot be staged without reading the remote file back, so - * it keeps writing straight through, as its own protocol already implies. + * - a 1.5s gap before any byte, which forces the first read to find nothing -> 0 bytes, 6 of 6 + * - one byte, a 1.5s gap, then 32767 more -> exactly 1 byte, then nothing + * - 32768, a 1.5s gap, then 32768 more -> exactly 32768, then nothing + * - a continuous 2MB -> 167936 / 270336 / 372736, then nothing + * + * Those three 2MB death points are one payload run three times under the same conditions, which is + * what rules out a threshold: a stream that died at a fixed point would not vary by 2x. Independently reproduced by + * a second harness, where one 1.9MB counted read survived 39 reads to completion and another died + * after 11 — same construct, same payload. + * + * A payload small enough to arrive in one burst usually presents only one read that can find the + * stream empty (the one waiting for EOF), which is why 32KB mostly works: it still failed 15 times + * in 120 with the host under load, and 1 in 40 on a quiet one. Neither rate is survivable across + * the 62 execs a 1.9MB file needs — even 2.5% compounds to roughly four uploads in five failing — + * and no chunk size helps, because the client does not control whether its bytes arrive together. + * + * The same host, same `DefaultShell`, same connection pattern contradicts every size-limit reading: + * `findstr` took 2,016,000 bytes through one exec's stdin, and PowerShell 7 took 2MB. So cmd.exe is + * not the ceiling and neither is ~50KB. Writes now go over sftp, which moves the whole payload + * without any remote process reading a pipe; see `system-ssh-windows-write-strategy.ts` for the + * fallback order. + * + * Successes are never partial. Across every run in both harnesses a failed write hung; not one + * produced a short file, so this defect cannot silently truncate an upload. */ -async function writeWindowsBytesViaSystemSsh( - target: SshTarget, - remotePath: string, - totalBytes: number, - readChunk: (offset: number, maxBytes: number) => Promise, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const staged = !options.append && totalBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES - const writePath = staged ? `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` : remotePath - let offset = 0 - // An empty write still has to run: it is what creates (or truncates) the file. - do { - const chunk = await readChunk(offset, WINDOWS_STDIN_WRITE_CHUNK_BYTES) - if (chunk.length === 0 && offset < totalBytes) { - throw new Error(`Source ran short during upload of ${remotePath}`) - } - await writeWindowsChunkViaSystemSsh( - target, - writePath, - chunk, - { - ...options, - append: staged ? offset > 0 : options.append === true || offset > 0, - exclusive: staged ? false : options.exclusive === true && offset === 0 - }, - offset - ) - offset += chunk.length - } while (offset < totalBytes) - if (staged) { - await publishWindowsStagedWrite(target, writePath, remotePath, options) + +/** A staged write is materialized locally first when the source is a buffer rather than a file. */ +async function withTemporaryLocalFile( + contents: Buffer, + send: (localPath: string) => Promise +): Promise { + const directory = await mkdtemp(join(tmpdir(), 'orca-win-upload-')) + const localPath = join(directory, 'payload.bin') + try { + // 0600: the payload can be repository content, and tmpdir is shared on every platform. + await writeFile(localPath, contents, { mode: 0o600 }) + return await send(localPath) + } finally { + await rm(directory, { recursive: true, force: true }).catch(() => {}) } } -async function writeWindowsChunkViaSystemSsh( - target: SshTarget, - remotePath: string, - chunk: Buffer, - options: SystemSshWriteBufferOptions, - offset: number -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsWriteFileCommand(remotePath, options), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose( - channel, - `write ${remotePath} at offset ${offset}`, - WINDOWS_STDIN_WRITE_TIMEOUT_MS - ) - ) - if (!options.signal?.aborted) { - channel.stdin.end(chunk) - } - await closePromise -} - -async function publishWindowsStagedWrite( - target: SshTarget, - stagingPath: string, - remotePath: string, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand( - target, - makeWindowsPublishStagedFileCommand(stagingPath, remotePath, options.exclusive === true), - { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } - ) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, `publish ${remotePath}`, WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ) - if (!options.signal?.aborted) { - channel.stdin.end() - } - await closePromise -} - -function makeWindowsWriteFileCommand( - remotePath: string, - options?: { append?: boolean; exclusive?: boolean } -): string { - const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$path = ${powerShellLiteral(remotePath)}`, - '$parent = [System.IO.Path]::GetDirectoryName($path)', - 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', - '$inputStream = [Console]::OpenStandardInput()', - `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, - 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' - ].join('; ') - ) -} - -// `File::Move` throws when the destination exists, which is exactly the exclusive contract; the -// non-exclusive caller asked to replace, so it deletes first (a no-op on an absent path). -function makeWindowsPublishStagedFileCommand( - stagingPath: string, - remotePath: string, - exclusive: boolean -): string { - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$staging = ${powerShellLiteral(stagingPath)}`, - `$path = ${powerShellLiteral(remotePath)}`, - ...(exclusive ? [] : ['[System.IO.File]::Delete($path)']), - '[System.IO.File]::Move($staging, $path)' - ].join('; ') - ) -} - function makePosixWriteFileCommand( remotePath: string, options?: { append?: boolean; exclusive?: boolean } diff --git a/src/main/ssh/system-ssh-file-transfer.ts b/src/main/ssh/system-ssh-file-transfer.ts index f728c0eab02..d5757904356 100644 --- a/src/main/ssh/system-ssh-file-transfer.ts +++ b/src/main/ssh/system-ssh-file-transfer.ts @@ -27,6 +27,12 @@ import { WINDOWS_STDIN_WRITE_TIMEOUT_MS, writeBufferViaSystemSsh } from './system-ssh-file-binary-transfer' +import { + isSftpPathUnsupportedError, + isSftpUnavailableError, + makeDirectoriesViaSftp +} from './system-ssh-sftp-transfer' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -161,9 +167,14 @@ async function collectWindowsUploadPlan( return plan } -// Why the JSON envelope survives here: a path list is metadata, so this payload stays in the -// hundreds of bytes even for a deep tree. Batched anyway, so a pathological tree cannot walk back -// into the same stdin size that wedges PowerShell. +/** + * Creates the upload's directories, preferring sftp's own `mkdir`. + * + * The PowerShell fallback keeps the JSON envelope, batched under one stdin's worth: a path list is + * metadata, so it stays in the hundreds of bytes even for a deep tree. It is still a redirected + * stdin read though, so on Windows PowerShell 5.1 it carries the same defect as any other — which + * is why sftp is tried first even for a payload this small. + */ async function createWindowsUploadDirectories( target: SshTarget, directories: readonly string[], @@ -175,23 +186,27 @@ async function createWindowsUploadDirectories( if (batch.length === 0) { return } + const pending = batch const payload = JSON.stringify(batch) batch = [] batchBytes = 0 throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + await getWindowsRemoteWriteCapabilities(target).runWithFallback( + 'sftp-subsystem', + async () => { + try { + await makeDirectoriesViaSftp(target, pending, options) + } catch (error) { + // A directory sftp cannot address is this batch's problem, not the host's verdict. + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await createWindowsUploadDirectoriesViaPowerShell(target, payload, options) + } + }, + () => createWindowsUploadDirectoriesViaPowerShell(target, payload, options), + isSftpUnavailableError ) - if (!options.signal?.aborted) { - channel.stdin.end(payload) - } - await closePromise } for (const directory of directories) { const entryBytes = Buffer.byteLength(directory) + 4 @@ -204,17 +219,44 @@ async function createWindowsUploadDirectories( await flush() } +async function createWindowsUploadDirectoriesViaPowerShell( + target: SshTarget, + payload: string, + options: SystemSshOperationOptions +): Promise { + const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end(payload) + } + await closePromise +} + function makeWindowsCreateDirectoriesCommand(): string { return powerShellCommand( [ '$ErrorActionPreference = "Stop"', - // The reporter measured this reader surviving 50KB where `[Console]::In` wedged at the same - // size (#16432); the batch above stays under that. + // Reached only where the host has no sftp subsystem. Windows PowerShell 5.1 can lose a + // redirected stdin for good when a read finds it empty (#16432); a batch this small usually + // arrives in one piece, and "usually" is exactly why sftp is preferred. '$reader = New-Object System.IO.StreamReader([Console]::OpenStandardInput())', 'try { $json = $reader.ReadToEnd() } finally { $reader.Dispose() }', 'if ([string]::IsNullOrWhiteSpace($json)) { return }', - 'foreach ($path in @($json | ConvertFrom-Json)) {', - ' $null = [System.IO.Directory]::CreateDirectory([string]$path)', + // `[string[]]`, not `@(...)`: ConvertFrom-Json emits the parsed array as a single pipeline + // object, so `@(...)` wraps it in *another* array and the loop variable binds to the whole + // thing. `[string]` of that is the paths joined by spaces, which CreateDirectory rejects with + // "The given path's format is not supported". It only ever worked for a one-element batch, + // where stringifying a single-element array happens to yield the element. Measured on + // WindowsPowerShell 5.1.26100 against a three-directory tree. + 'foreach ($path in [string[]]($json | ConvertFrom-Json)) {', + ' $null = [System.IO.Directory]::CreateDirectory($path)', '}' ].join('; ') ) diff --git a/src/main/ssh/system-ssh-sftp-args.test.ts b/src/main/ssh/system-ssh-sftp-args.test.ts new file mode 100644 index 00000000000..d971390f8dd --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.test.ts @@ -0,0 +1,141 @@ +/** + * `buildSshArgs` is shared with the sftp client, and three of its flags mean something else there. + * Every case below is a silent wrong-target rather than an error if the translation is skipped, + * which is why the fallback is "refuse and use another transport", never "pass it through". + */ +import { describe, expect, it } from 'vitest' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' + +describe('translateSshArgsToSftpArgs', () => { + it('sends the port as an option, since sftp -p preserves mtimes instead', () => { + const args = translateSshArgsToSftpArgs(['-p', '2222', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'Port=2222', '--', 'dev@win.example']) + }) + + it('sends the login name as an option, since sftp has no -l', () => { + // `buildSshArgs` emits `-l` for a config alias no Host block claims. Throwing here would send + // exactly those hosts to the transport this PR exists to stop using, silently. + const args = translateSshArgsToSftpArgs(['-l', 'neil', '--', 'awin']) + + expect(args).toEqual(['-o', 'User=neil', '--', 'awin']) + }) + + it('translates the whole unclaimed-alias shape buildSshArgs emits', () => { + const args = translateSshArgsToSftpArgs([ + '-o', + 'BatchMode=no', + '-T', + '-S', + 'none', + '-o', + 'Hostname=192.168.0.186', + '-p', + '2222', + '-l', + 'neil', + '--', + 'awin' + ]) + + expect(args).toEqual([ + '-o', + 'BatchMode=no', + '-o', + 'ControlPath=none', + '-o', + 'Hostname=192.168.0.186', + '-o', + 'Port=2222', + '-o', + 'User=neil', + '--', + 'awin' + ]) + }) + + it('spells ControlPath=none out, since sftp -S names a program to run', () => { + // `sftp -S none` would try to exec a binary called `none`. + const args = translateSshArgsToSftpArgs(['-S', 'none', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'ControlPath=none', '--', 'dev@win.example']) + }) + + it('refuses any other -S, which would hand sftp an ssh binary Orca did not choose', () => { + expect(() => translateSshArgsToSftpArgs(['-S', '/tmp/ctl.sock'])).toThrow( + SftpArgTranslationError + ) + }) + + it('drops -T, which sftp does not have', () => { + expect(translateSshArgsToSftpArgs(['-T', '--', 'host'])).toEqual(['--', 'host']) + }) + + it('passes through the flags both clients spell the same way', () => { + const args = translateSshArgsToSftpArgs([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + + expect(args).toEqual([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + }) + + it('takes everything after -- as the destination without reinterpreting it', () => { + // A host literally named `-p` is not a flag once `--` has been seen. + expect(translateSshArgsToSftpArgs(['--', '-p'])).toEqual(['--', '-p']) + }) + + it('refuses an unknown flag rather than guessing what sftp would do with it', () => { + // The point of the throw: a flag added to buildSshArgs later must degrade to another + // transport, not reach sftp carrying a different meaning. + expect(() => translateSshArgsToSftpArgs(['-A', '--', 'host'])).toThrow(SftpArgTranslationError) + }) + + it('refuses a value flag with no value', () => { + expect(() => translateSshArgsToSftpArgs(['-o'])).toThrow(SftpArgTranslationError) + }) +}) + +describe('withSftpKeepalive', () => { + it('asks OpenSSH to notice a dead peer, since the transfer itself has no wall-clock bound', () => { + expect(withSftpKeepalive(['--', 'host'])).toEqual([ + '-o', + 'ServerAliveInterval=15', + '-o', + 'ServerAliveCountMax=3', + '--', + 'host' + ]) + }) + + it('leaves a caller-stated keepalive policy alone', () => { + const args = withSftpKeepalive(['-o', 'ServerAliveInterval=60', '--', 'host']) + + expect(args.filter((arg) => arg.startsWith('ServerAliveInterval'))).toEqual([ + 'ServerAliveInterval=60' + ]) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-args.ts b/src/main/ssh/system-ssh-sftp-args.ts new file mode 100644 index 00000000000..17fa37e78bb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.ts @@ -0,0 +1,95 @@ +/** + * Rewrites `buildSshArgs` output for the sftp(1) client. + * + * Three flags ssh and sftp share spell different things: sftp's `-p` is "preserve mtime", its `-S` + * names the ssh binary to run, and it has no `-T` at all. Passing ssh's list through unchanged + * would silently connect to the wrong port and try to exec a program called `none`. + * + * Anything this table does not recognize throws. A flag added to `buildSshArgs` later must degrade + * to the non-sftp transfer path, never reach sftp carrying a different meaning. + */ + +/** `buildSshArgs` emitted a flag with no sftp equivalent; the caller should use another transport. */ +export class SftpArgTranslationError extends Error { + constructor(flag: string) { + super(`No sftp equivalent for system ssh argument ${JSON.stringify(flag)}`) + this.name = 'SftpArgTranslationError' + } +} + +/** Flags whose spelling and meaning are identical in both clients. */ +const PASSTHROUGH_VALUE_FLAGS = new Set(['-F', '-o', '-i', '-J']) + +export function translateSshArgsToSftpArgs(sshArgs: readonly string[]): string[] { + const sftpArgs: string[] = [] + let index = 0 + while (index < sshArgs.length) { + const flag = sshArgs[index]! + if (flag === '--') { + // Everything after `--` is the destination, which both clients spell the same way. + sftpArgs.push(...sshArgs.slice(index)) + return sftpArgs + } + const value = sshArgs[index + 1] + if (PASSTHROUGH_VALUE_FLAGS.has(flag)) { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push(flag, value) + index += 2 + continue + } + if (flag === '-T') { + // sftp never allocates a tty, so ssh's "no tty" request has nothing to translate to. + index += 1 + continue + } + if (flag === '-p') { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `Port=${value}`) + index += 2 + continue + } + if (flag === '-l') { + // sftp has no `-l`; the login name is an option there. `buildSshArgs` emits this for an + // unclaimed config alias, so throwing would route those hosts down the defective path and + // then cache the refusal against them for half an hour. + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `User=${value}`) + index += 2 + continue + } + if (flag === '-S') { + // ssh's `-S none` is ControlPath=none; sftp's `-S` would run a binary called `none`. + if (value !== 'none') { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', 'ControlPath=none') + index += 2 + continue + } + throw new SftpArgTranslationError(flag) + } + return sftpArgs +} + +/** + * A transfer that stalls mid-stream has no per-write bound to catch it, so ask OpenSSH to notice a + * dead peer itself. Only added when the caller has not already stated a keepalive policy. + */ +export function withSftpKeepalive(sftpArgs: readonly string[]): string[] { + const hasOption = (name: string): boolean => + sftpArgs.some((arg, position) => sftpArgs[position - 1] === '-o' && arg.startsWith(`${name}=`)) + const keepalive: string[] = [] + if (!hasOption('ServerAliveInterval')) { + keepalive.push('-o', 'ServerAliveInterval=15') + } + if (!hasOption('ServerAliveCountMax')) { + keepalive.push('-o', 'ServerAliveCountMax=3') + } + return [...keepalive, ...sftpArgs] +} diff --git a/src/main/ssh/system-ssh-sftp-path.test.ts b/src/main/ssh/system-ssh-sftp-path.test.ts new file mode 100644 index 00000000000..e196c2e3e8f --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.test.ts @@ -0,0 +1,59 @@ +/** + * Both functions here guard against the same measured failure: sftp's batch lexer treats `\` as an + * escape, so a Windows path handed over raw is silently mis-targeted *and the client still exits + * 0*. On Windows 11 / OpenSSH 10.0p2, `put src C:\Users\neil\qt\a.bin` created a file literally + * named `C` in the start directory and reported success. + */ +import { describe, expect, it } from 'vitest' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' + +describe('toSftpRemotePath', () => { + it('roots a drive path under /, which is the namespace the Windows sftp-server exposes', () => { + // `pwd` in that session reports `/C:/Users/dev`. + expect(toSftpRemotePath('C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('accepts a path already in that namespace unchanged', () => { + expect(toSftpRemotePath('/C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('converts the separators Orca stores paths with', () => { + expect(toSftpRemotePath('C:\\Users\\dev\\f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('declines a UNC path rather than guessing where it lands', () => { + // A guess here writes real bytes to the wrong place; declining falls back to another transport. + expect(() => toSftpRemotePath('//server/share/f.bin')).toThrow(UnsupportedSftpPathError) + }) + + it('declines a relative path, which would resolve against the session start directory', () => { + expect(() => toSftpRemotePath('Users/dev/f.bin')).toThrow(UnsupportedSftpPathError) + }) +}) + +describe('quoteSftpBatchArgument', () => { + it('escapes the backslashes in a Windows client local path', () => { + // Unescaped, sftp reads this as C:srcf.bin and fails to find the source. + expect(quoteSftpBatchArgument('C:\\src\\f.bin')).toBe('"C:\\\\src\\\\f.bin"') + }) + + it('keeps a path with spaces as one argument', () => { + expect(quoteSftpBatchArgument('/tmp/two words.bin')).toBe('"/tmp/two words.bin"') + }) + + it('escapes an embedded quote, which would otherwise end the argument early', () => { + expect(quoteSftpBatchArgument('/tmp/dq".bin')).toBe('"/tmp/dq\\".bin"') + }) + + it('refuses a line break, which would split one batch command into two', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\nrm -rf b')).toThrow(UnsupportedSftpPathError) + }) + + it('refuses a NUL, which truncates the argument', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\0b')).toThrow(UnsupportedSftpPathError) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-path.ts b/src/main/ssh/system-ssh-sftp-path.ts new file mode 100644 index 00000000000..2b5bfe53f02 --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.ts @@ -0,0 +1,46 @@ +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * A path this transfer cannot express to sftp. Callers treat it as "use another transport", never + * as a transfer failure. + */ +export class UnsupportedSftpPathError extends Error { + constructor(path: string) { + super(`Path cannot be addressed over sftp: ${JSON.stringify(path)}`) + this.name = 'UnsupportedSftpPathError' + } +} + +/** + * Converts a Windows remote path to the namespace OpenSSH's Windows sftp-server exposes, which + * roots every drive under `/`: `C:/Users/dev/f` is `/C:/Users/dev/f`, and `pwd` there reports + * `/C:/Users/dev`. + */ +export function toSftpRemotePath(remotePath: string): string { + const normalized = normalizeWindowsRemotePath(remotePath) + if (/^\/[a-zA-Z]:\//.test(normalized)) { + return normalized + } + if (/^[a-zA-Z]:\//.test(normalized)) { + return `/${normalized}` + } + // UNC (`//server/share`) and relative paths have no settled mapping in this namespace, and a + // guess here writes real bytes to the wrong place. Decline instead. + throw new UnsupportedSftpPathError(remotePath) +} + +/** + * Quotes one argument of an sftp batch line. + * + * Escaping is load-bearing, not cosmetic: sftp's batch lexer treats `\` as an escape even inside + * double quotes, so an unescaped Windows local path `C:\src\f.bin` is read as `C:srcf.bin`, and an + * unescaped destination `C:\Users\dev\f.bin` writes a file literally named `C` in the start + * directory — while sftp still exits 0. Both measured on Windows 11 / OpenSSH 10.0p2. + */ +export function quoteSftpBatchArgument(value: string): string { + if (/[\n\r\0]/.test(value)) { + // A line break would split one batch command into two; NUL truncates the argument. + throw new UnsupportedSftpPathError(value) + } + return `"${value.replace(/([\\"])/g, '\\$1')}"` +} diff --git a/src/main/ssh/system-ssh-sftp-transfer.ts b/src/main/ssh/system-ssh-sftp-transfer.ts new file mode 100644 index 00000000000..c50375fa8eb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-transfer.ts @@ -0,0 +1,191 @@ +import { accessSync, constants, existsSync, statSync } from 'node:fs' +import { posix, win32 } from 'node:path' +import type { SshTarget } from '../../shared/ssh-types' +import { buildSshArgs, type SystemSshBuildArgsOptions } from './system-ssh-args' +import { findSystemSsh } from './system-ssh-binary' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' +import { throwIfAborted } from './system-ssh-operation-lifecycle' +import { runProcess } from '../../shared/child-process/run-process' + +/** The host answered, but not with an sftp subsystem. The caller must fall back, not fail. */ +export class SftpSubsystemUnavailableError extends Error { + constructor(detail: string) { + super(`Remote host has no usable sftp subsystem: ${detail}`) + this.name = 'SftpSubsystemUnavailableError' + } +} + +/** + * True for the errors that mean "this host cannot serve sftp at all". + * + * Host-scoped, and therefore the only errors safe to remember: a capability cache keyed by host + * turns anything it accepts into a verdict about every later write to that host. Deliberately + * narrow — a permission denial or a missing directory is a real failure that must surface, not a + * reason to retry the whole upload down a slower path. + */ +export function isSftpUnavailableError(error: unknown): boolean { + return error instanceof SftpSubsystemUnavailableError || error instanceof SftpArgTranslationError +} + +/** + * True when *this path* cannot be spelled for sftp, which says nothing about the host. + * + * Kept apart from the host verdict on purpose. A UNC destination, or a local file whose name + * contains a newline, is a property of one operation; caching it would degrade every subsequent + * write to that host for the cache's whole retry window on the strength of one odd filename. + */ +export function isSftpPathUnsupportedError(error: unknown): boolean { + return error instanceof UnsupportedSftpPathError +} + +/** Neither kind of refusal moves a byte, so a staged file cannot exist to sweep. */ +export function isSftpRefusalBeforeStaging(error: unknown): boolean { + return isSftpUnavailableError(error) || isSftpPathUnsupportedError(error) +} + +function systemSftpCandidates(sshPath: string | null, platform: NodeJS.Platform): string[] { + const pathApi = platform === 'win32' ? win32 : posix + const executable = platform === 'win32' ? 'sftp.exe' : 'sftp' + const candidates: string[] = [] + // Why the ssh binary's own directory first: a host with two OpenSSH installs must pair the sftp + // client with the ssh that `buildSshArgs` was built for, not whichever one PATH happens to reach. + if (sshPath) { + candidates.push(pathApi.join(pathApi.dirname(sshPath), executable)) + } + if (platform === 'win32') { + const systemRoot = process.env.SystemRoot || process.env.WINDIR + if (systemRoot) { + candidates.push(win32.join(systemRoot, 'System32', 'OpenSSH', executable)) + } + } else { + candidates.push('/usr/bin/sftp', '/usr/local/bin/sftp', '/opt/homebrew/bin/sftp') + } + return candidates +} + +/** Locate the sftp client paired with the system ssh binary. Returns null when there is none. */ +export function findSystemSftp(): string | null { + if (process.env.ORCA_SYSTEM_SFTP_PATH) { + return process.env.ORCA_SYSTEM_SFTP_PATH + } + const sshPath = findSystemSsh() + for (const candidate of systemSftpCandidates(sshPath, process.platform)) { + try { + if (!statSync(candidate).isFile()) { + continue + } + if (process.platform !== 'win32') { + accessSync(candidate, constants.X_OK) + } + return candidate + } catch { + continue + } + } + return findSftpOnPath() +} + +function findSftpOnPath(): string | null { + const pathValue = process.env.PATH + if (!pathValue) { + return null + } + const pathApi = process.platform === 'win32' ? win32 : posix + const executable = process.platform === 'win32' ? 'sftp.exe' : 'sftp' + for (const entry of pathValue.split(pathApi.delimiter)) { + const directory = entry.trim().replace(/^"|"$/g, '') + if (!directory) { + continue + } + const candidate = pathApi.join(directory, executable) + if (existsSync(candidate)) { + return candidate + } + } + return null +} + +/** + * OpenSSH prints this when the server refuses the subsystem — a host with `Subsystem sftp` + * commented out, or an internal-sftp block that does not apply to this user. + */ +const SUBSYSTEM_REFUSED_PATTERN = /subsystem request failed|no such file or directory.*sftp-server/i + +export type SftpBatchOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal } + +/** + * Runs one sftp batch script. + * + * The script goes to the *local* sftp client's stdin, which is the point: no remote process ever + * reads a redirected stdin, so none of this rides the Windows PowerShell stdin defect. + */ +export async function runSftpBatch( + target: SshTarget, + commands: readonly string[], + options?: SftpBatchOptions +): Promise { + throwIfAborted(options?.signal) + const sftpPath = findSystemSftp() + if (!sftpPath) { + throw new SftpSubsystemUnavailableError('no sftp client binary found alongside ssh') + } + const args = withSftpKeepalive(translateSshArgsToSftpArgs(buildSshArgs(target, options))) + let result + try { + result = await runProcess({ + program: sftpPath, + args: ['-b', '-', ...args], + // `-b -` takes the script on stdin, and that stdin is the *local* client's — no remote + // process reads a pipe anywhere in this transfer, which is the whole point of preferring it. + input: `${commands.join('\n')}\n`, + // Why no timeout: a large upload is legitimately slow, and a wall-clock cap would fail a + // healthy transfer on a slow link. A dead peer is caught by the ServerAlive options instead. + timeoutMs: null, + signal: options?.signal + }) + } catch (error) { + // A client that will not start is "this host cannot do sftp" from the caller's side, not a + // transfer failure: the payload never left. Falling back is the only useful answer. + throw new SftpSubsystemUnavailableError( + `sftp client at ${sftpPath} could not be started: ${error instanceof Error ? error.message : String(error)}` + ) + } + if (result.code === 0) { + return + } + throwIfAborted(options?.signal) + const detail = result.stderr.trim() + if (SUBSYSTEM_REFUSED_PATTERN.test(detail)) { + throw new SftpSubsystemUnavailableError(detail) + } + throw new Error(`sftp batch failed (exit ${result.code}): ${detail}`) +} + +/** + * Creates remote directories, parents first. + * + * `-mkdir` keeps sftp going when a directory is already there; batch mode otherwise aborts the + * whole script on the first non-zero status, which for an idempotent tree walk is not a failure. + */ +export function makeDirectoriesViaSftp( + target: SshTarget, + remoteDirectories: readonly string[], + options?: SftpBatchOptions +): Promise { + const commands = remoteDirectories.map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + if (commands.length === 0) { + return Promise.resolve() + } + return runSftpBatch(target, commands, options) +} diff --git a/src/main/ssh/system-ssh-windows-file-write.ts b/src/main/ssh/system-ssh-windows-file-write.ts new file mode 100644 index 00000000000..8b98d4aec50 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-file-write.ts @@ -0,0 +1,138 @@ +import { randomBytes } from 'node:crypto' +import { powerShellCommand, powerShellLiteral } from './ssh-remote-powershell' +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * Suffix marking the path a Windows write lands on before it is published by rename. + * + * The random tail is the fix for a measured harm, not decoration. A write that loses contact with + * the host leaves a remote process that may still hold the staging file open exclusively, and + * `docs/reference/ssh-execution-boundary.md` is explicit that losing contact is not evidence that + * process died — so the retry must not reuse the name it may still own. A fresh name per attempt + * means a retry never meets its predecessor's lock; the abandoned file is cleaned up best-effort + * and never treated as proof of anything. + */ +export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' + +export function makeWindowsStagingPath(remotePath: string): string { + return `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}-${randomBytes(6).toString('hex')}` +} + +export type WindowsPublishMode = 'create' | 'exclusive' | 'append' + +/** + * Publishes a staged upload onto its real name. + * + * Every branch reads the staged *file*, never a redirected stdin, which is what makes this safe on + * a host whose Windows PowerShell 5.1 cannot drain a piped stdin. + * + * The replacing branch must never delete the destination first. Deleting and then moving loses the + * user's existing file outright if the move fails, and exposes a window where a reader sees no file + * at all — a worse outcome than the truncated-partial this staging discipline exists to prevent. + * `File.Replace` is the atomic swap (Win32 `ReplaceFile`), and it requires the destination to + * exist, so an absent one falls back to a plain `Move`. That fallback is raced deliberately: if the + * destination appears in between, `Move` throws, the staged file survives, and the destination is + * left exactly as whoever created it left it. + * + * `File::Move` throwing on an existing destination is also precisely the exclusive contract, which + * is why that branch needs nothing else. + */ +export function makeWindowsPublishStagedFileCommand( + stagingPath: string, + remotePath: string, + mode: WindowsPublishMode +): string { + const preamble = [ + '$ErrorActionPreference = "Stop"', + `$staging = ${powerShellLiteral(stagingPath)}`, + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }' + ] + if (mode === 'append') { + return powerShellCommand( + [ + ...preamble, + // Not atomic, and cannot cheaply be: appending is defined as extending the destination, so + // a failure part-way leaves it longer than it was rather than destroyed. The caller's + // chunked-append protocol already restarts from its own offset. + '$in = [System.IO.File]::OpenRead($staging)', + '$out = [System.IO.File]::Open($path, [System.IO.FileMode]::Append, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)', + 'try { $in.CopyTo($out) } finally { $out.Dispose(); $in.Dispose() }', + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) + } + if (mode === 'exclusive') { + return powerShellCommand([...preamble, '[System.IO.File]::Move($staging, $path)'].join('; ')) + } + return powerShellCommand( + [ + ...preamble, + // `[NullString]::Value`, not `$null`: PowerShell coerces a bare `$null` to an empty string + // when binding a .NET `string` parameter, and `Replace` rejects that with "The path is not + // of a legal form" — so every publish would fail. Measured on WindowsPowerShell 5.1.26100. + 'try { [System.IO.File]::Replace($staging, $path, [NullString]::Value) } catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ].join('; ') + ) +} + +/** Best-effort removal of a staged file whose write was abandoned. Never asserts the writer died. */ +export function makeWindowsDiscardStagedFileCommand(stagingPath: string): string { + return powerShellCommand( + [ + // Deliberately not `Stop`: the previous writer may still hold this file, and that is a + // possibility to tolerate, not an error to report. The unique staging name means a leftover + // blocks nothing; sweeping it is housekeeping. + '$ErrorActionPreference = "SilentlyContinue"', + `$staging = ${powerShellLiteral(stagingPath)}`, + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) +} + +/** + * The ancestor directories of a Windows remote path, drive root first. + * + * sftp's `mkdir` creates one level, so a batch has to name each level itself. The drive root is + * excluded: `-mkdir "/C:/"` is not a directory anyone creates. + */ +export function windowsRemoteAncestorDirectories(remotePath: string): string[] { + const normalized = normalizeWindowsRemotePath(remotePath) + const segments = normalized.split('/') + segments.pop() + const ancestors: string[] = [] + // Start past the drive (`C:`) or the UNC host, which are never created. + for (let depth = 2; depth <= segments.length; depth += 1) { + const directory = segments.slice(0, depth).join('/') + if (directory) { + ancestors.push(directory) + } + } + return ancestors +} + +/** + * `[Console]::OpenStandardInput()` into a `FileStream`, used only by the two stdin fallbacks. + * + * On Windows PowerShell 5.1 this is the defective read; see the strategy comment in + * `system-ssh-file-binary-transfer.ts`. It is correct under PowerShell 7. + */ +export function makeWindowsWriteFileCommand( + remotePath: string, + options?: { append?: boolean; exclusive?: boolean; executable?: 'powershell.exe' | 'pwsh.exe' } +): string { + const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' + return powerShellCommand( + [ + '$ErrorActionPreference = "Stop"', + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', + '$inputStream = [Console]::OpenStandardInput()', + `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, + 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' + ].join('; '), + options?.executable ?? 'powershell.exe' + ) +} diff --git a/src/main/ssh/system-ssh-windows-upload.test.ts b/src/main/ssh/system-ssh-windows-upload.test.ts index 207c3e2df8e..c3a5ff80276 100644 --- a/src/main/ssh/system-ssh-windows-upload.test.ts +++ b/src/main/ssh/system-ssh-windows-upload.test.ts @@ -1,29 +1,40 @@ /** - * #16432: the Windows relay upload pushed the whole bundle into one PowerShell stdin, which - * Windows PowerShell 5.1 cannot drain over a non-pty ssh exec — the remote blocks forever, and - * `waitForChannelClose()` had no timeout, so the UI sat at "Connecting…" with no error. Covered - * here: no write exceeds one stdin's worth on any Windows path (bundle upload *and* single-file - * upload, which is the one that carries large files), a partial write never lands under the real - * name, and a remote that never closes fails instead of hanging. + * #16432. The original fix chunked the payload because the constraint was believed to be a ~50KB + * cmd.exe stdin ceiling. Re-measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2, it is + * not a size limit and not cmd.exe's: a read on Windows PowerShell 5.1's redirected-stdin handle + * over a non-pty ssh exec can die permanently when it finds the stream momentarily empty, taking + * both the remaining data and the EOF with it. It is probabilistic per such read — identical 2MB + * payloads died at 167936, 270336 and 372736 — so a 32KB chunk still failed 15 times in 120 under + * load, while `findstr` took 2,016,000 bytes through one exec on the same host. + * + * So the covering property is no longer "every write is small". It is "the bytes do not cross a + * remote process's stdin at all": sftp first, PowerShell 7 next, and Windows PowerShell 5.1 last, + * bounded and loud. The staging-and-rename discipline is kept on every path, with a unique staging + * name per attempt so a retry never meets a predecessor's lock. */ import { EventEmitter } from 'node:events' import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' -import { rm } from 'node:fs/promises' +import { readFile, rm, stat } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { PassThrough, Writable } from 'node:stream' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as SystemSshOperationLifecycle from './system-ssh-operation-lifecycle' -const { spawnSystemSshCommandMock, waitForChannelCloseSpy } = vi.hoisted(() => ({ +const { spawnSystemSshCommandMock, waitForChannelCloseSpy, runProcessMock } = vi.hoisted(() => ({ spawnSystemSshCommandMock: vi.fn(), - waitForChannelCloseSpy: vi.fn() + waitForChannelCloseSpy: vi.fn(), + runProcessMock: vi.fn() })) vi.mock('./system-ssh-command', () => ({ spawnSystemSshCommand: spawnSystemSshCommandMock })) +vi.mock('../../shared/child-process/run-process', () => ({ + runProcess: runProcessMock +})) + // Delegates to the real implementation; the spy only records whether each wait was given a bound. vi.mock('./system-ssh-operation-lifecycle', async (importActual) => { const actual = (await importActual()) as typeof SystemSshOperationLifecycle @@ -41,6 +52,11 @@ import { } from './system-ssh-file-binary-transfer' import { waitForChannelClose } from './system-ssh-operation-lifecycle' import { getRemoteHostPlatform } from './ssh-remote-platform' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities +} from './system-ssh-windows-write-capabilities' +import { explainWindowsPowerShellStdinFailure } from './system-ssh-windows-write-strategy' import type { SshTarget } from '../../shared/ssh-types' type FakeChannel = EventEmitter & { @@ -50,7 +66,12 @@ type FakeChannel = EventEmitter & { written: Buffer } -const target = { id: 'win-1', host: 'win.example', username: 'dev' } as unknown as SshTarget +const target = { + id: 'win-1', + host: 'win.example', + username: 'dev', + port: 22 +} as unknown as SshTarget const hostPlatform = getRemoteHostPlatform('win32-x64') const remoteRoot = 'C:/Users/dev/.orca-remote' @@ -78,96 +99,238 @@ function createFakeChannel(onEnd: (channel: FakeChannel) => void): FakeChannel { return channel } -type RecordedCommand = { script: string; stdin: Buffer } +type RecordedCommand = { script: string; executable: string; stdin: Buffer } +type RecordedSftpBatch = { args: string[]; script: string } -describe('Windows upload stdin framing', () => { - let localDir: string - const commands: RecordedCommand[] = [] - /** Index of the spawn that should report a non-zero exit, to model a chunk failing mid-file. */ - let failAtSpawn = -1 +const sftpBatches: RecordedSftpBatch[] = [] +const commands: RecordedCommand[] = [] +/** Index of the exec that should report a non-zero exit, to model a chunk failing mid-file. */ +let failAtSpawn = -1 +let localDir: string - const fileWrites = (): RecordedCommand[] => - commands.filter((command) => command.script.includes('FileMode]::')) - const writtenPath = (command: RecordedCommand): string => - /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1].replace(/''/g, "'") ?? '' - const fileMode = (command: RecordedCommand): string | undefined => - /FileMode\]::(\w+)/.exec(command.script)?.[1] +const fileWrites = (): RecordedCommand[] => + commands.filter((command) => command.script.includes('OpenStandardInput')) +const writtenPath = (command: RecordedCommand): string => + /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1]?.replace(/''/g, "'") ?? '' +const fileMode = (command: RecordedCommand): string | undefined => + /FileMode\]::(\w+)/.exec(command.script)?.[1] +const putLines = (): string[] => + sftpBatches.flatMap((batch) => batch.script.split('\n').filter((line) => line.startsWith('put '))) +const putDestination = (line: string): string => /put "(?:[^"]*)" "([^"]*)"/.exec(line)?.[1] ?? '' +const putSource = (line: string): string => /put "([^"]*)"/.exec(line)?.[1] ?? '' - beforeEach(() => { - commands.length = 0 - failAtSpawn = -1 - waitForChannelCloseSpy.mockClear() - localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) - spawnSystemSshCommandMock.mockReset() - spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { - const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 - return createFakeChannel((channel) => { - commands.push({ script: decodePowerShellCommand(command), stdin: channel.written }) - setImmediate(() => - spawnIndex === failAtSpawn - ? channel.emit('close', 1, null) - : channel.emit('close', 0, null) - ) +/** Makes every sftp batch succeed, recording what it was asked to do. */ +function acceptSftp(): void { + runProcessMock.mockImplementation( + async (spec: { args: string[]; input: string; program: string }) => { + const script = spec.input + sftpBatches.push({ args: spec.args, script }) + // Model the real client: `put` copies the local file, so read it while it still exists. + for (const line of script.split('\n').filter((entry) => entry.startsWith('put '))) { + await readFile(putSource(line)) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + } + ) +} + +/** Models a host whose sshd has no `Subsystem sftp` line. */ +function refuseSftp(): void { + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + return { + code: 255, + signal: null, + stdout: '', + stderr: 'subsystem request failed on channel 0\nConnection closed', + timedOut: false + } + }) +} + +/** Models a host with no PowerShell 7, which cmd.exe reports as an unrecognized command. */ +function refusePwsh(): void { + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + const executable = command.split(' ')[0] ?? '' + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable, + stdin: channel.written + }) + setImmediate(() => { + if (executable === 'pwsh.exe') { + channel.stderr.write( + "'pwsh.exe' is not recognized as an internal or external command,\noperable program or batch file." + ) + channel.emit('close', 9009, null) + return + } + channel.emit('close', spawnIndex === failAtSpawn ? 1 : 0, null) }) }) }) +} - afterEach(async () => { - await rm(localDir, { recursive: true, force: true }) +beforeEach(() => { + commands.length = 0 + sftpBatches.length = 0 + failAtSpawn = -1 + clearWindowsRemoteWriteCapabilitiesForTests() + waitForChannelCloseSpy.mockClear() + localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) + process.env.ORCA_SYSTEM_SFTP_PATH = '/usr/bin/sftp' + runProcessMock.mockReset() + acceptSftp() + spawnSystemSshCommandMock.mockReset() + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable: command.split(' ')[0] ?? '', + stdin: channel.written + }) + setImmediate(() => + spawnIndex === failAtSpawn ? channel.emit('close', 1, null) : channel.emit('close', 0, null) + ) + }) }) +}) - it('never pushes a whole artifact bundle into one PowerShell stdin', async () => { - mkdirSync(join(localDir, 'node'), { recursive: true }) - // Comfortably past the ~50KB point at which the reporter measured PowerShell 5.1 wedging. - writeFileSync(join(localDir, 'node', 'relay.js'), Buffer.alloc(600 * 1024, 0x61)) - writeFileSync(join(localDir, 'index.js'), Buffer.alloc(300 * 1024, 0x62)) +afterEach(async () => { + delete process.env.ORCA_SYSTEM_SFTP_PATH + await rm(localDir, { recursive: true, force: true }) +}) - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - - const largest = Math.max(...commands.map((command) => command.stdin.length)) - expect(largest).toBeLessThanOrEqual(WINDOWS_STDIN_WRITE_CHUNK_BYTES) - // The base64 + JSON envelope is gone entirely: nothing reads the bundle as one string. - expect(commands.some((command) => command.script.includes('FromBase64String'))).toBe(false) - // `[Console]::In` wedged at 50KB where the stream reader did not, so the mkdir batch — the one - // payload still read as a string — must use the reader the reporter measured surviving. - expect(commands.some((command) => command.script.includes('[Console]::In.ReadToEnd()'))).toBe( - false - ) - expect( - commands.filter((command) => command.script.includes('StreamReader([Console]::')) - ).toHaveLength(1) - }) - - it('bounds the single-file upload too, which is the path large files take', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) +describe('Windows upload over sftp', () => { + it('moves the payload without any remote process reading a stdin', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 60 + 11, 0x64) const localPath = join(localDir, 'big.node') writeFileSync(localPath, contents) await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/big.node`, { hostPlatform }) - const writes = fileWrites() - expect(writes).toHaveLength(4) - expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( - WINDOWS_STDIN_WRITE_CHUNK_BYTES - ) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. - expect( - waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ).toBe(true) + // The defect is a remote stdin read; the fix is that there is not one. + expect(fileWrites()).toHaveLength(0) + expect(putLines()).toHaveLength(1) + // One transfer, not 61 execs: the whole point of the change. + expect(sftpBatches).toHaveLength(1) }) - it('writes every byte of every artifact across the chunked writes', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 2 + 17, 0x63) - writeFileSync(join(localDir, 'relay.js'), contents) + it('creates the parent chain and sends the payload in one round trip', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/a/b/relay.js`, { + hostPlatform + }) - const writes = fileWrites() - expect(writes).toHaveLength(3) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // Only the first write creates the staging file; the rest must extend it or it is truncated. - expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append']) + expect(sftpBatches).toHaveLength(1) + expect(sftpBatches[0]!.script.split('\n').filter(Boolean)).toEqual([ + '-mkdir "/C:/Users"', + '-mkdir "/C:/Users/dev"', + '-mkdir "/C:/Users/dev/.orca-remote"', + '-mkdir "/C:/Users/dev/.orca-remote/a"', + '-mkdir "/C:/Users/dev/.orca-remote/a/b"', + expect.stringContaining('put ') as unknown as string + ]) + }) + + it('addresses the destination in the drive-rooted namespace sftp exposes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A backslash destination silently writes a file named `C` and still exits 0, so the leading + // slash and forward separators are correctness, not style. + expect(putDestination(putLines()[0]!)).toMatch( + /^\/C:\/Users\/dev\/\.orca-remote\/relay\.js\.orca-partial-[0-9a-f]{12}$/ + ) + }) + + it('never lands a partial under the real name, and publishes by rename', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const remotePath = `${remoteRoot}/relay.js` + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), remotePath, { hostPlatform }) + + const destination = putDestination(putLines()[0]!) + // Assert the positive first: an unmatched regex yields '', which would satisfy the `not.toBe` + // below without this test ever having seen a destination. + expect(destination).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + expect(destination).not.toBe(`/C:${remotePath.slice(2)}`) + const publish = commands.at(-1)! + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // The publish reads the staged file, never a pipe, so it is safe on PowerShell 5.1. + expect(publish.script).not.toContain('OpenStandardInput') + }) + + it('never deletes the destination it is replacing', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const publish = commands.at(-1)! + // Delete-then-move destroys the user's existing file outright if the move then fails, and + // exposes a window where a reader sees no file at all — worse than the truncated partial the + // staging discipline exists to prevent. `File.Replace` is the atomic swap. + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // An absent destination cannot be Replaced, so that case falls back to a plain Move. + expect(publish.script).toContain( + 'catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ) + }) + + it('gives every attempt its own staging name, so a retry cannot meet a predecessor lock', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const [first, second] = putLines().map(putDestination) + expect(first).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + // Losing contact is not evidence the previous writer died, so the name must not be reused. + expect(second).not.toBe(first) + }) + + it('enforces exclusive at the rename, where it is atomic', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + }) + + it('appends by concatenating the staged file, not by piping bytes to the remote', async () => { + await writeBufferViaSystemSsh(target, `${remoteRoot}/log.bin`, Buffer.from('tail'), { + hostPlatform, + append: true + }) + + expect(fileWrites()).toHaveLength(0) + const publish = commands.at(-1)! + expect(publish.script).toContain('FileMode]::Append') + expect(publish.script).toContain('$in.CopyTo($out)') + expect(publish.script).toContain('[System.IO.File]::Delete($staging)') }) it('still creates an empty artifact on the host', async () => { @@ -175,80 +338,310 @@ describe('Windows upload stdin framing', () => { await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - expect(fileWrites().map(writtenPath)).toEqual([`${remoteRoot}/empty.txt`]) - expect(fileWrites()[0].stdin).toHaveLength(0) - expect(fileMode(fileWrites()[0])).toBe('Create') + expect(putLines()).toHaveLength(1) + expect(commands.at(-1)!.script).toContain('[System.IO.File]::Move($staging, $path)') }) - it('lands a multi-chunk write on a staging path and publishes it by rename', async () => { - const remotePath = `${remoteRoot}/relay.js` - writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + it('writes a buffer through a 0600 temp file that does not outlive the transfer', async () => { + const seen: { path: string; contents: Buffer; mode: number }[] = [] + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + for (const line of spec.input.split('\n').filter((entry) => entry.startsWith('put '))) { + const path = putSource(line) + seen.push({ + path, + contents: await readFile(path), + mode: (await stat(path)).mode & 0o777 + }) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + }) + + await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { + hostPlatform + }) + + expect(seen).toHaveLength(1) + expect(seen[0]!.contents.toString()).toBe('1.2.3') + // The payload can be repository content and tmpdir is world-readable on every platform, so the + // window between write and upload must not be group- or world-readable. + expect(seen[0]!.mode).toBe(0o600) + await expect(readFile(seen[0]!.path)).rejects.toThrow() + }) + + it('creates upload directories over sftp rather than a PowerShell stdin batch', async () => { + mkdirSync(join(localDir, 'node'), { recursive: true }) + writeFileSync(join(localDir, 'node', 'relay.js'), 'x') await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - // Nothing touches the real name until every byte is on the host. - expect(fileWrites().map(writtenPath)).toEqual([ - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`, - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` - ]) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).toContain('[System.IO.File]::Delete($path)') + // Anchor on a non-empty observation: `some` is false of an empty list, so this would pass even + // if no command had been recorded at all. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('StreamReader([Console]::'))).toBe( + false + ) + expect(sftpBatches[0]!.script).toContain('-mkdir "/C:/Users/dev/.orca-remote"') + }) + + it('sweeps the staged bytes when the publish is the thing that fails', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + // An exclusive conflict is the ordinary way to get here: the payload is on the host, and the + // rename that would have given it a name refuses. + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const script = decodePowerShellCommand(command) + return createFakeChannel((channel) => { + commands.push({ script, executable: command.split(' ')[0] ?? '', stdin: channel.written }) + const failed = script.includes('::Move($staging, $path)') + setImmediate(() => channel.emit('close', failed ? 1 : 0, null)) + }) + }) + + await expect( + uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + ).rejects.toThrow() + + const sweep = commands.at(-1)! + expect(sweep.script).toContain('[System.IO.File]::Delete($staging)') + // Tolerated, not asserted: the previous writer may still hold the file, and losing contact is + // not evidence it died. + expect(sweep.script).toContain('$ErrorActionPreference = "SilentlyContinue"') + }) + + it('reports a cancelled transfer as an abort, not as a failed one', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const controller = new AbortController() + // runProcess reports the kill as a non-zero exit rather than throwing, so without checking the + // signal first a user pressing cancel is indistinguishable from the transfer genuinely failing. + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + controller.abort() + return { code: 255, signal: 'SIGTERM', stdout: '', stderr: '', timedOut: false } + }) + + let error: Error | undefined + try { + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + signal: controller.signal + }) + } catch (thrown) { + error = thrown as Error + } + + expect(error?.name).toBe('AbortError') + expect(error?.message).not.toContain('sftp batch failed') + // A cancel is also not evidence about the host, so it must not send later writes to the slow + // path, and must not fall through to the defective reader now. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + expect(fileWrites()).toHaveLength(0) + }) + + it('does not let one unaddressable path become a verdict about the host', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + // A UNC destination has no settled mapping in sftp's drive-rooted namespace, so this write + // falls back — but the host still serves sftp perfectly well for every other path. + await uploadFileViaSystemSsh( + target, + join(localDir, 'relay.js'), + '//fileserver/share/relay.js', + { hostPlatform } + ) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(sftpBatches).toHaveLength(0) + // The 30-minute capability cache is keyed by host; caching this would send every later write + // to the same machine down the defective path on the strength of one odd destination. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps using sftp for the next file after one path it could not spell', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), '//fileserver/share/a.js', { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/b.js`, { + hostPlatform + }) + + expect(putLines()).toHaveLength(1) + expect(putDestination(putLines()[0]!)).toContain('/C:/Users/dev/.orca-remote/b.js') + }) + + it('does not let a local filename sftp cannot quote become a verdict either', async () => { + // POSIX clients allow a newline in a filename, and sftp's batch lexer would read it as the end + // of one command and the start of another. + const awkward = join(localDir, 'two\nlines.js') + writeFileSync(awkward, 'x') + + await uploadFileViaSystemSsh(target, awkward, `${remoteRoot}/relay.js`, { hostPlatform }) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('translates the ssh argument list rather than passing it to a client that reads it differently', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + disableControlMaster: true + }) + + const args = sftpBatches[0]!.args + // sftp's `-T` does not exist, its `-p` preserves mtime, and its `-S` names a program to run. + expect(args).not.toContain('-T') + expect(args).not.toContain('-p') + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') + expect(args).toContain('ServerAliveInterval=15') + }) +}) + +describe('Windows upload on a host with no sftp subsystem', () => { + beforeEach(() => { + refuseSftp() + }) + + it('creates a multi-directory tree, which the one-element case never exercised', async () => { + mkdirSync(join(localDir, 'node', 'deep'), { recursive: true }) + writeFileSync(join(localDir, 'index.js'), 'a') + writeFileSync(join(localDir, 'node', 'deep', 'x.js'), 'b') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const mkdir = commands.find((command) => command.script.includes('ConvertFrom-Json'))! + // `@($json | ConvertFrom-Json)` wraps the parsed array in another array, so the loop variable + // binds to the whole thing and `[string]` of it is the paths joined by spaces — which + // CreateDirectory rejects. It only ever worked for a single directory, where stringifying a + // one-element array happens to yield the element, so no batch of one can catch this. + expect(mkdir.script).toContain('[string[]]($json | ConvertFrom-Json)') + expect(mkdir.script).not.toContain('@($json | ConvertFrom-Json)') + const batch = JSON.parse(mkdir.stdin.toString('utf-8')) as string[] + expect(batch.length).toBeGreaterThan(1) + }) + + it('falls back rather than failing the transfer', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 5, 0x61) + writeFileSync(join(localDir, 'relay.js'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(Buffer.concat(fileWrites().map((write) => write.stdin)).equals(contents)).toBe(true) + }) + + it('remembers the refusal, so a multi-file upload probes once', async () => { + writeFileSync(join(localDir, 'a.js'), 'a') + writeFileSync(join(localDir, 'b.js'), 'b') + writeFileSync(join(localDir, 'c.js'), 'c') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + // One refusal is enough; re-probing per file is a wasted round trip on every file. + expect(sftpBatches).toHaveLength(1) + }) + + it('does not spend a sweep round trip when sftp declined before moving any bytes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A refused subsystem staged nothing, so there is nothing to delete — and on a host without + // sftp that sweep would otherwise be paid on every single write. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('Delete($staging)'))).toBe(false) + }) + + it('prefers PowerShell 7, which reads a redirected stdin correctly', async () => { + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(fileWrites().map((write) => write.executable)).toEqual(['pwsh.exe']) + // PowerShell 7 took 2MB through one exec when measured, so chunking it buys nothing. + expect(fileWrites()[0]!.stdin).toHaveLength(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3) + }) + + it('bounds every write when only Windows PowerShell 5.1 is available', async () => { + refusePwsh() + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) + writeFileSync(join(localDir, 'big.node'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + const writes = fileWrites().filter((write) => write.executable === 'powershell.exe') + expect(writes).toHaveLength(4) + expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( + WINDOWS_STDIN_WRITE_CHUNK_BYTES + ) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append', 'Append']) + // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. + // Count first: `every` is true of zero calls, so a wait that moved to a different helper would + // pass this silently. + expect(waitForChannelCloseSpy.mock.calls.length).toBeGreaterThan(0) + expect( + waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ).toBe(true) + }) + + it('remembers that PowerShell 7 is absent instead of re-probing per chunk', async () => { + refusePwsh() + writeFileSync(join(localDir, 'big.node'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + expect(fileWrites().filter((write) => write.executable === 'pwsh.exe')).toHaveLength(1) }) it('leaves no truncated file under the real name when a chunk fails mid-file', async () => { writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) - // Spawns: 0 = mkdir batch, 1..3 = chunk writes. Fail the second chunk. - failAtSpawn = 2 + // Spawn 0 is the pwsh write; fail it and every retry beneath it. + failAtSpawn = 0 await expect( - uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) ).rejects.toThrow() + expect(fileWrites().length).toBeGreaterThan(0) expect(fileWrites().map(writtenPath)).not.toContain(`${remoteRoot}/relay.js`) expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) }) +}) - it('enforces exclusive once at the rename, so a retry is not blocked by its own leftovers', async () => { - const localPath = join(localDir, 'import.bin') - writeFileSync(localPath, Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) +describe('last-resort Windows PowerShell failure reporting', () => { + it('names the host limitation and its remedy, not just the timeout', () => { + const timeout = new Error('write C:/x at offset 0 timed out after 60000ms with no response') - await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/import.bin`, { - hostPlatform, - exclusive: true - }) + const explained = explainWindowsPowerShellStdinFailure(timeout) as Error - // CreateNew on chunk one would fail against a leftover staging file from a failed attempt; - // `File::Move` raising on an existing destination is what carries the exclusive contract. - expect(fileWrites().map(fileMode)).toEqual(['Create', 'Append']) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + // "timed out" alone sends the user to retry a network they cannot fix; the fix is host-side. + expect(explained.message).toContain('Windows PowerShell 5.1') + expect(explained.message).toContain('Subsystem sftp sftp-server.exe') + expect(explained.cause).toBe(timeout) }) - it('keeps a single-chunk write on the destination, with the caller mode intact', async () => { - await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { - hostPlatform, - exclusive: true - }) + it('leaves a real failure alone, so a permission error is not reported as a host limitation', () => { + const denied = new Error('write C:/x at offset 0 failed (exit 1): Access to the path is denied') - expect(fileWrites()).toHaveLength(1) - expect(writtenPath(fileWrites()[0])).toBe(`${remoteRoot}/version`) - expect(fileMode(fileWrites()[0])).toBe('CreateNew') - expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) - }) - - it('appends onto the destination rather than staging, since append cannot be staged', async () => { - const remotePath = `${remoteRoot}/log.bin` - await writeBufferViaSystemSsh( - target, - remotePath, - Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1), - { hostPlatform, append: true } - ) - - expect(fileWrites().map(writtenPath)).toEqual([remotePath, remotePath]) - expect(fileWrites().map(fileMode)).toEqual(['Append', 'Append']) + expect(explainWindowsPowerShellStdinFailure(denied)).toBe(denied) }) }) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.test.ts b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts new file mode 100644 index 00000000000..ad723d592b0 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts @@ -0,0 +1,79 @@ +/** + * Whether a Windows host has an sftp subsystem is a fact about that host, so the cache is keyed by + * the endpoint that executes rather than by Orca's target id — otherwise a hardened host is + * re-probed once per file, and two targets pointing at one machine learn the same fact twice. + */ +import { afterEach, describe, expect, it } from 'vitest' +import type { SshTarget } from '../../shared/ssh-types' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities, + getWindowsRemoteWriteExecutionHostKey +} from './system-ssh-windows-write-capabilities' + +const asTarget = (fields: Partial): SshTarget => fields as SshTarget + +afterEach(() => { + clearWindowsRemoteWriteCapabilitiesForTests() +}) + +describe('getWindowsRemoteWriteExecutionHostKey', () => { + it('gives two targets on one endpoint the same key', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + // A target re-created under a new id has not changed what the host supports. + expect(getWindowsRemoteWriteExecutionHostKey(first)).toBe( + getWindowsRemoteWriteExecutionHostKey(second) + ) + }) + + it('separates hosts, ports and users', () => { + const base = { id: 'a', host: 'win.example', username: 'dev', port: 22 } + const keys = [ + asTarget(base), + asTarget({ ...base, host: 'other.example' }), + asTarget({ ...base, port: 2222 }), + asTarget({ ...base, username: 'ops' }) + ].map(getWindowsRemoteWriteExecutionHostKey) + + expect(new Set(keys).size).toBe(4) + }) + + it('keys a config alias by the alias, since ssh_config decides where it lands', () => { + const alias = asTarget({ id: 'a', host: 'stale.example', configHost: 'winbox' }) + + expect(getWindowsRemoteWriteExecutionHostKey(alias)).toBe('config:winbox') + }) +}) + +describe('getWindowsRemoteWriteCapabilities', () => { + it('shares one cache across targets that reach the same host', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(first).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(second).shouldTry('sftp-subsystem')).toBe(false) + }) + + it('does not let one host answer for another', () => { + const hardened = asTarget({ id: 'a', host: 'hardened.example', username: 'dev', port: 22 }) + const ordinary = asTarget({ id: 'b', host: 'ordinary.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(hardened).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(ordinary).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps the two capabilities independent', () => { + const target = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const capabilities = getWindowsRemoteWriteCapabilities(target) + + capabilities.rememberUnsupported('pwsh') + + // No PowerShell 7 says nothing about whether the host will serve sftp. + expect(capabilities.shouldTry('sftp-subsystem')).toBe(true) + expect(capabilities.shouldTry('pwsh')).toBe(false) + }) +}) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.ts b/src/main/ssh/system-ssh-windows-write-capabilities.ts new file mode 100644 index 00000000000..dcd03f19807 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.ts @@ -0,0 +1,52 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { CapabilityProbeCache } from '../../shared/capability-probe-cache' + +/** + * Whether a Windows host can take a file write over the sftp subsystem, and whether it has a + * PowerShell 7 to fall back to. Both are host facts, so they are cached per execution host rather + * than per transfer — a hardened host with `Subsystem sftp` removed must not be re-probed on every + * file of a multi-file upload. + */ +export type WindowsRemoteWriteCapability = 'sftp-subsystem' | 'pwsh' + +// Why re-probe at all: an admin can enable the subsystem, or install PowerShell 7, without the +// user restarting Orca. Long enough that a hardened host costs one failed probe per half hour. +export const WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS = 30 * 60_000 + +const capabilitiesByExecutionHost = new Map< + string, + CapabilityProbeCache +>() + +/** + * Keyed by the endpoint that executes, not by target id: two Orca targets pointing at one host + * describe the same sshd, and a target re-created under a new id has not changed what that host + * supports. A config alias is its own key because ssh_config, not Orca, resolves where it lands. + */ +export function getWindowsRemoteWriteExecutionHostKey(target: SshTarget): string { + if (target.configHost) { + return `config:${target.configHost}` + } + const port = target.port ?? 22 + return target.username + ? `host:${target.username}@${target.host}:${port}` + : `host:${target.host}:${port}` +} + +export function getWindowsRemoteWriteCapabilities( + target: SshTarget +): CapabilityProbeCache { + const key = getWindowsRemoteWriteExecutionHostKey(target) + let cache = capabilitiesByExecutionHost.get(key) + if (!cache) { + cache = new CapabilityProbeCache( + WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS + ) + capabilitiesByExecutionHost.set(key, cache) + } + return cache +} + +export function clearWindowsRemoteWriteCapabilitiesForTests(): void { + capabilitiesByExecutionHost.clear() +} diff --git a/src/main/ssh/system-ssh-windows-write-strategy.ts b/src/main/ssh/system-ssh-windows-write-strategy.ts new file mode 100644 index 00000000000..f2cdca12516 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-strategy.ts @@ -0,0 +1,329 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { getSystemSshBuildArgsFromOperationOptions } from './system-ssh-args' +import { spawnSystemSshCommand } from './system-ssh-command' +import { + awaitWithSystemSshAbort, + throwIfAborted, + waitForChannelClose +} from './system-ssh-operation-lifecycle' +import { + isSftpPathUnsupportedError, + isSftpRefusalBeforeStaging, + isSftpUnavailableError, + runSftpBatch +} from './system-ssh-sftp-transfer' +import { quoteSftpBatchArgument, toSftpRemotePath } from './system-ssh-sftp-path' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' +import { + makeWindowsDiscardStagedFileCommand, + makeWindowsPublishStagedFileCommand, + makeWindowsStagingPath, + makeWindowsWriteFileCommand, + windowsRemoteAncestorDirectories, + type WindowsPublishMode +} from './system-ssh-windows-file-write' + +/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ +export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 + +/** + * Bound on one stdin write for the last-resort Windows PowerShell 5.1 path. + * + * Measured on Windows 11 26200 / OpenSSH 10.0p2: a 32KB write still hangs 15 times in 120 under + * load, and no smaller value removes the risk. The defect is per blocking read, not per byte, so + * shrinking the chunk trades one risky read for more execs that each carry their own. This is a + * damage bound on a path known to be unreliable, not a safe size. + */ +export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 + +export type WindowsWriteOptions = Parameters< + typeof getSystemSshBuildArgsFromOperationOptions +>[0] & { + signal?: AbortSignal + append?: boolean + exclusive?: boolean +} + +/** Bytes to write, plus a way to present them to sftp, which can only send a local file. */ +export type WindowsWriteSource = { + totalBytes: number + readChunk: (offset: number, maxBytes: number) => Promise + withLocalFile: (send: (localPath: string) => Promise) => Promise +} + +function publishMode(options: WindowsWriteOptions): WindowsPublishMode { + return options.append ? 'append' : options.exclusive === true ? 'exclusive' : 'create' +} + +/** + * Writes one file to a Windows host, preferring transports that do not push bytes through a remote + * PowerShell's stdin. + * + * Order, and why: sftp carries the whole payload in one transfer and never has a remote process + * read a pipe. Measured on Windows 11 / OpenSSH 10.0p2: 1.9MB in a median 315ms over sftp against + * 0 of 6 completions on the chunked path, whose best case was ~62 execs at ~350ms each. PowerShell + * 7 reads a redirected stdin correctly but is not installed by default. Windows PowerShell 5.1 is + * always present and is the defective reader, so it is last and it is bounded. + * + * Every transport stages under a unique name and publishes by rename, so no partial write is ever + * visible under the real name and no retry inherits a predecessor's lock. + */ +export async function writeWindowsRemoteFile( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + throwIfAborted(options.signal) + const capabilities = getWindowsRemoteWriteCapabilities(target) + await capabilities.runWithFallback( + 'sftp-subsystem', + () => writeViaSftp(target, remotePath, source, options), + () => writeViaRemoteStdin(target, remotePath, source, options), + isSftpUnavailableError + ) +} + +/** + * Stages under a name nothing else can own, publishes it, and sweeps the staging file if either + * step fails. + * + * Shared by both transports so the cleanup contract cannot drift between them: a failed publish — + * an exclusive conflict is the ordinary case — leaves bytes on the host that no longer have a + * purpose, and the sweep is what stops them accumulating. + */ +async function stageThenPublish( + target: SshTarget, + remotePath: string, + options: WindowsWriteOptions, + stage: (stagingPath: string) => Promise, + nothingStaged: (error: unknown) => boolean = () => false +): Promise { + const stagingPath = makeWindowsStagingPath(remotePath) + try { + await stage(stagingPath) + await publishStagedWrite(target, stagingPath, remotePath, options) + } catch (error) { + // A transport that declined before it moved any bytes has nothing to sweep, and sweeping + // anyway would spend a round trip on every write to a host that has no sftp subsystem. + if (!nothingStaged(error)) { + await discardStagedWrite(target, stagingPath, options) + } + throw error + } +} + +/** + * A path sftp cannot address falls back for this write alone, without touching the host verdict. + * + * The distinction matters because the capability cache is keyed by host and holds for half an hour: + * routing one UNC destination, or one local filename containing a newline, into + * `rememberUnsupported` would send every later write to that host down the defective path too. + */ +async function writeViaSftp( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + try { + await attemptSftpWrite(target, remotePath, source, options) + } catch (error) { + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await writeViaRemoteStdin(target, remotePath, source, options) + } +} + +function attemptSftpWrite( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const mkdirs = windowsRemoteAncestorDirectories(remotePath).map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + return stageThenPublish( + target, + remotePath, + options, + (stagingPath) => + source.withLocalFile((localPath) => + // One round trip: the parent chain and the payload travel in the same batch. + runSftpBatch( + target, + [ + ...mkdirs, + `put ${quoteSftpBatchArgument(localPath)} ${quoteSftpBatchArgument(toSftpRemotePath(stagingPath))}` + ], + options + ) + ), + isSftpRefusalBeforeStaging + ) +} + +function writeViaRemoteStdin( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const capabilities = getWindowsRemoteWriteCapabilities(target) + return stageThenPublish(target, remotePath, options, (stagingPath) => + capabilities.runWithFallback( + 'pwsh', + () => writeStdinChunks(target, stagingPath, source, options, 'pwsh.exe'), + () => writeStdinChunks(target, stagingPath, source, options, 'powershell.exe'), + isPwshUnavailableError + ) + ) +} + +/** + * PowerShell 7 takes the whole payload in one exec — measured at 2MB — so only the 5.1 path pays + * for chunking, and only because a bounded write is the most that path can be trusted with. + */ +async function writeStdinChunks( + target: SshTarget, + stagingPath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + const chunkBytes = + executable === 'pwsh.exe' ? Math.max(source.totalBytes, 1) : WINDOWS_STDIN_WRITE_CHUNK_BYTES + let offset = 0 + // An empty write still has to run: it is what creates the staged file. + do { + const chunk = await source.readChunk(offset, chunkBytes) + if (chunk.length === 0 && offset < source.totalBytes) { + throw new Error(`Source ran short during upload of ${stagingPath}`) + } + await writeOneStdinChunk( + target, + stagingPath, + chunk, + { ...options, append: offset > 0, exclusive: false }, + offset, + executable + ) + offset += chunk.length + } while (offset < source.totalBytes) +} + +async function writeOneStdinChunk( + target: SshTarget, + stagingPath: string, + chunk: Buffer, + options: WindowsWriteOptions, + offset: number, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand( + target, + makeWindowsWriteFileCommand(stagingPath, { + append: options.append, + exclusive: options.exclusive, + executable + }), + { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } + ) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose( + channel, + `write ${stagingPath} at offset ${offset}`, + WINDOWS_STDIN_WRITE_TIMEOUT_MS + ) + ).catch((error: unknown) => { + throw executable === 'powershell.exe' ? explainWindowsPowerShellStdinFailure(error) : error + }) + if (!options.signal?.aborted) { + channel.stdin.end(chunk) + } + await closePromise +} + +/** + * Names the cause on the one path that can hang, so the failure is not just "timed out". + * + * A user seeing this needs to know it is a host limitation with a host-side remedy, not a network + * fault they should retry into. + */ +export function explainWindowsPowerShellStdinFailure(error: unknown): unknown { + const message = error instanceof Error ? error.message : String(error) + if (!/timed out/i.test(message)) { + return error + } + return new Error( + `${message}\nWindows PowerShell 5.1 can lose a redirected stdin permanently when a read finds it momentarily empty, so this write cannot be made reliable from the client. Enable the sftp subsystem on the host (sshd_config: "Subsystem sftp sftp-server.exe"), or install PowerShell 7, and Orca will use it automatically.`, + { cause: error instanceof Error ? error : undefined } + ) +} + +function isPwshUnavailableError(error: unknown): boolean { + const message = error instanceof Error ? error.message : String(error) + // cmd.exe's "not recognized" and sshd's exit 9009 both mean "no pwsh here". A timeout does not: + // that is the stdin defect, and PowerShell 7 does not have it, so it must not be cached as absent. + return /is not recognized as an internal or external command|9009|CommandNotFoundException/i.test( + message + ) +} + +async function publishStagedWrite( + target: SshTarget, + stagingPath: string, + remotePath: string, + options: WindowsWriteOptions +): Promise { + await runWindowsCommandWithoutStdin( + target, + makeWindowsPublishStagedFileCommand(stagingPath, remotePath, publishMode(options)), + `publish ${remotePath}`, + options + ) +} + +async function discardStagedWrite( + target: SshTarget, + stagingPath: string, + options: WindowsWriteOptions +): Promise { + try { + await runWindowsCommandWithoutStdin( + target, + makeWindowsDiscardStagedFileCommand(stagingPath), + `discard ${stagingPath}`, + { ...options, signal: undefined } + ) + } catch { + // Housekeeping only. The staging name is unique, so a leftover blocks nothing, and a failure + // here says nothing about whether the abandoned writer is still alive. + } +} + +function runWindowsCommandWithoutStdin( + target: SshTarget, + command: string, + label: string, + options: WindowsWriteOptions +): Promise { + const channel = spawnSystemSshCommand(target, command, { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, label, WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end() + } + return closePromise +} diff --git a/src/main/startup/configure-process.test.ts b/src/main/startup/configure-process.test.ts index ef7ec6a9b69..7d7e2b0cc1a 100644 --- a/src/main/startup/configure-process.test.ts +++ b/src/main/startup/configure-process.test.ts @@ -1,4 +1,4 @@ -import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { homedir, tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' @@ -137,6 +137,29 @@ describe('patchPackagedProcessPath', () => { expect(segments).toContain('/usr/local/bin') }) + // Why derived, not a second literal: system-cli-install-dirs.ts documents its + // order as matching this seed's system block, and hardcoding the order in the + // fallback's own test lets a reorder here break that parity while both stay green. + it('seeds the system block in the order the install-dir fallback expects', async () => { + const { app } = await import('electron') + const { patchPackagedProcessPath } = await import('./configure-process') + const { getSystemCliInstallDirectories } = await import('../../shared/system-cli-install-dirs') + + setPlatform('linux') + Object.defineProperty(app, 'isPackaged', { configurable: true, value: true }) + process.env.HOME = '/home/tester' + process.env.PATH = '/usr/bin:/bin' + + patchPackagedProcessPath() + + const segments = (process.env.PATH ?? '').split(':') + const offsets = getSystemCliInstallDirectories('linux', '/home/tester').map((directory) => + segments.indexOf(directory) + ) + expect(offsets.every((offset) => offset >= 0)).toBe(true) + expect([...offsets].sort((a, b) => a - b)).toEqual(offsets) + }) + // Why this ordering is load-bearing (#18234): a seed exists so a GUI-launched // Electron can *find* a tool, not to re-rank tools the user already has. // `~/.local/bin` is user-writable and can hold a wrapper for any system tool. @@ -444,6 +467,43 @@ describe('configureElectronNetworkCompatibility', () => { ).toBe(false) }) + it('answers from the marker without reading the settings file', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const { writeHttp1CompatibilityMarker } = await import('./http1-compatibility-marker') + const userDataPath = createUserDataDir({ electronHttp1CompatibilityMode: false }) + writeHttp1CompatibilityMarker(userDataPath, true) + rmSync(join(userDataPath, 'orca-data.json'), { force: true }) + + expect(shouldDisableHttp2ForElectronNetworking({ env: {}, userDataPath })).toBe(true) + }) + + it('falls back to the settings file when no marker has been written yet', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const userDataPath = createUserDataDir({ electronHttp1CompatibilityMode: true }) + + expect(existsSync(join(userDataPath, 'http1-compatibility.json'))).toBe(false) + expect(shouldDisableHttp2ForElectronNetworking({ env: {}, userDataPath })).toBe(true) + }) + + it('falls back to the settings file when the marker is corrupt', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const userDataPath = createUserDataDir({ electronHttp1CompatibilityMode: true }) + writeFileSync(join(userDataPath, 'http1-compatibility.json'), '{ not json', 'utf-8') + + expect(shouldDisableHttp2ForElectronNetworking({ env: {}, userDataPath })).toBe(true) + }) + + it('lets the environment override the marker', async () => { + const { shouldDisableHttp2ForElectronNetworking } = await import('./configure-process') + const { writeHttp1CompatibilityMarker } = await import('./http1-compatibility-marker') + const userDataPath = createUserDataDir({}) + writeHttp1CompatibilityMarker(userDataPath, true) + + expect( + shouldDisableHttp2ForElectronNetworking({ env: { ORCA_DISABLE_HTTP2: '0' }, userDataPath }) + ).toBe(false) + }) + it('appends Electron disable-http2 before sessions are created', async () => { const { app } = await import('electron') const { configureElectronNetworkCompatibility } = await import('./configure-process') diff --git a/src/main/startup/configure-process.ts b/src/main/startup/configure-process.ts index 6b167d7fe14..13dbd70c2c3 100644 --- a/src/main/startup/configure-process.ts +++ b/src/main/startup/configure-process.ts @@ -5,6 +5,7 @@ import { join, resolve } from 'node:path' import { getVersionManagerBinPaths } from '../codex-cli/command' import { getMainE2EConfig } from '../e2e-config' import { DISABLED_CHROMIUM_FEATURES } from './disabled-chromium-features' +import { readHttp1CompatibilityMarker } from './http1-compatibility-marker' const DEV_PARENT_SHUTDOWN_GRACE_MS = 3000 const HTTP1_COMPATIBILITY_ENV_VAR = 'ORCA_DISABLE_HTTP2' @@ -54,7 +55,13 @@ export function shouldDisableHttp2ForElectronNetworking( if (envValue !== null) { return envValue } - return readPersistedHttp1CompatibilityMode(options.userDataPath ?? app.getPath('userData')) + const userDataPath = options.userDataPath ?? app.getPath('userData') + // Why the marker first: this runs before app.whenReady(), and the settings file is the multi-MB + // orca-data.json the Store parses again moments later. The marker is refreshed whenever settings + // change, so the full read only happens on a profile that has never written one. + return ( + readHttp1CompatibilityMarker(userDataPath) ?? readPersistedHttp1CompatibilityMode(userDataPath) + ) } export function configureElectronNetworkCompatibility( diff --git a/src/main/startup/http1-compatibility-marker.ts b/src/main/startup/http1-compatibility-marker.ts new file mode 100644 index 00000000000..9e85a83be66 --- /dev/null +++ b/src/main/startup/http1-compatibility-marker.ts @@ -0,0 +1,51 @@ +import { readFileSync, writeFileSync } from 'node:fs' +import { join } from 'node:path' + +/** + * Cached copy of `settings.electronHttp1CompatibilityMode` for pre-`ready` startup. + * + * Why a standalone file (not the Store): app.commandLine.appendSwitch('disable-http2') must run + * before the first Electron session exists, which is before the settings Store is constructed. + * Reading it from the settings file meant a synchronous read + JSON.parse of the whole multi-MB + * orca-data.json on the critical path of every cold start, duplicating the parse the Store does a + * moment later. This marker is a few bytes, mirroring gpu-fallback-marker.ts. + */ + +export const HTTP1_COMPATIBILITY_MARKER_FILE = 'http1-compatibility.json' +const MARKER_SCHEME_VERSION = 1 + +type Http1CompatibilityMarker = { + schemeVersion: number + enabled: boolean +} + +function markerPath(userDataPath: string): string { + return join(userDataPath, HTTP1_COMPATIBILITY_MARKER_FILE) +} + +/** Returns null when the marker is missing or unreadable, so callers fall back to the settings file. */ +export function readHttp1CompatibilityMarker(userDataPath: string): boolean | null { + try { + const parsed = JSON.parse( + readFileSync(markerPath(userDataPath), 'utf-8') + ) as Partial + if (parsed.schemeVersion !== MARKER_SCHEME_VERSION || typeof parsed.enabled !== 'boolean') { + return null + } + return parsed.enabled + } catch { + return null + } +} + +export function writeHttp1CompatibilityMarker(userDataPath: string, enabled: boolean): void { + if (readHttp1CompatibilityMarker(userDataPath) === enabled) { + return + } + const marker: Http1CompatibilityMarker = { schemeVersion: MARKER_SCHEME_VERSION, enabled } + try { + writeFileSync(markerPath(userDataPath), JSON.stringify(marker)) + } catch { + // Best effort: a missing marker just costs the next launch the settings-file fallback. + } +} diff --git a/src/main/startup/main-process-ready-foundation.ts b/src/main/startup/main-process-ready-foundation.ts index 171aaf50421..3c0e01fe09e 100644 --- a/src/main/startup/main-process-ready-foundation.ts +++ b/src/main/startup/main-process-ready-foundation.ts @@ -41,6 +41,7 @@ import { registerDocPreviewGrantHandlers } from '../ipc/doc-preview-grant-ipc' import { initializeBrowserSessionsForApp } from '../browser/browser-session-startup' import { browserSessionRegistry } from '../browser/browser-session-registry' import { logStartupMilestone } from './startup-diagnostics' +import { writeHttp1CompatibilityMarker } from './http1-compatibility-marker' import { mainProcessState as state } from './main-process-state' import { recordDurableCrashBreadcrumb } from '../crash-reporting/durable-crash-breadcrumb' import { syncMacMenuBarIcon } from './main-window-actions' @@ -193,9 +194,20 @@ export async function initializeReadyFoundation(): Promise { } wslHookRelayManager.setManagedHookSettingsResolver(() => state.store?.getSettings() ?? null) logStartupMilestone('store-loaded') + // Why: pre-`ready` startup reads this flag from a marker so it never has to parse orca-data.json. + writeHttp1CompatibilityMarker( + canonicalUserDataPath, + store.getSettings().electronHttp1CompatibilityMode === true + ) // Why: apply initial fallback WSL distro from store settings for global git/CLI calls. setDefaultWslDistroOverride(store.getSettings().terminalWindowsWslDistro ?? null) store.onSettingsChanged((updates, settings) => { + if ('electronHttp1CompatibilityMode' in updates) { + writeHttp1CompatibilityMarker( + canonicalUserDataPath, + settings.electronHttp1CompatibilityMode === true + ) + } if ('terminalWindowsWslDistro' in updates) { // Why: synchronize fallback WSL distro updates to runner. setDefaultWslDistroOverride(settings.terminalWindowsWslDistro ?? null) diff --git a/src/main/startup/main-process-ready-runtime.ts b/src/main/startup/main-process-ready-runtime.ts index 75784a4c196..26b920652df 100644 --- a/src/main/startup/main-process-ready-runtime.ts +++ b/src/main/startup/main-process-ready-runtime.ts @@ -9,7 +9,7 @@ import { RpcDispatcher } from '../runtime/rpc/dispatcher' import { browserManager } from '../browser/browser-manager' import { configureBrowserClientPageAutomationRuntime } from '../browser/browser-client-page-automation-runtime' import { BrowserClientPageCommandError } from '../browser/browser-client-page-command-failure' -import { startPreGoneProcessMetricsSampling } from '../crash-reporting/process-gone-diagnostics' +import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics' import { recordProcessGoneCrash } from './main-window-lifecycle-flags' import { handleGpuChildCrash } from './gpu-lifecycle' import { isGpuFallbackCrashCandidate } from '../crash-reporting/gpu-crash-fallback-decision' @@ -130,9 +130,10 @@ export async function initializeReadyRuntimeServices(): Promise { console.warn('[agent-hooks] failed to reconcile managed hooks on startup:', error) ) } - // Why: process-gone metrics only see survivors; retain a recent whole-app - // snapshot for comparison in crash reports. - startPreGoneProcessMetricsSampling() + // Why: process-gone metrics only see survivors, and the gone-time host memory + // read lands after the corpse released its pages; both need a live pre-gone + // sample to compare against in crash reports. + startPreGoneCrashSampling() app.on('child-process-gone', (_event, details) => { recordProcessGoneCrash('child', details.type, details.reason, details.exitCode ?? null, { name: details.name, diff --git a/src/main/startup/main-window-core-services.ts b/src/main/startup/main-window-core-services.ts index 759a4b2b100..d3ece383ee3 100644 --- a/src/main/startup/main-window-core-services.ts +++ b/src/main/startup/main-window-core-services.ts @@ -15,6 +15,7 @@ import { import { prepareCodexRuntimeHomeForLaunch } from './codex-launch-preparation' import { prepareCodexSessionResumeForLaunch } from './codex-session-resume-launch' import { isRecoveryReloadInFlight } from './main-window-lifecycle-flags' +import { RELAY_HOST_CLOSE_REASON } from '../../shared/relay-host-close-reason' export function attachMainWindowCoreServices( window: BrowserWindow, @@ -90,7 +91,10 @@ export function attachMainWindowCoreServices( }) }, onOrcaProfileAuthMutation: () => state.desktopRelayService?.authMutated(), - onBeforeOrcaProfileSignOut: () => state.desktopRelayService?.fenceAndCloseNow() + // Sign-out is the one fence a paired phone can be told about; quit and + // relaunch above stay reasonless so a restart never reads as signed out. + onBeforeOrcaProfileSignOut: () => + state.desktopRelayService?.fenceAndCloseNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) }, state.pluginService ?? undefined, state.pluginMarketplaceService && state.pluginMarketplaceInstaller diff --git a/src/main/startup/pre-gone-crash-sampling-wiring.test.ts b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts new file mode 100644 index 00000000000..2a8e008c0b3 --- /dev/null +++ b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts @@ -0,0 +1,49 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * Guards the one line that arms pre-gone crash sampling. + * + * That branch is pure instrumentation, so this line is the whole of its value in + * the shipped app: deleting it left all 691 tests across `src/main/crash-reporting/` + * and `src/main/startup/` green while every crash report silently lost its only + * host reading taken before the dying process returned its pages. + * + * Source-level because that is the property: the sampler is armed once inside the + * ready-phase composition, which has no runtime seam to assert against. + */ +describe('pre-gone crash sampling startup wiring', () => { + // Why normalize: the indent anchors below are `\n`-prefixed, and nothing pins + // src/**/*.ts to LF, so a CRLF Windows checkout would fail them spuriously. + const readSource = (name: string): string => + readFileSync(join(process.cwd(), 'src/main/startup', name), 'utf8').replace(/\r\n/g, '\n') + + const readyRuntimeSource = readSource('main-process-ready-runtime.ts') + const readySource = readSource('main-process-ready.ts') + + const READY_ENTRY = 'export async function initializeReadyRuntimeServices(' + // Why the entry's body and not the file: the call satisfies a whole-file grep + // just as well from a sibling export nothing calls, which arms nothing. + const readyRuntimeEntryBody = readyRuntimeSource + .slice(readyRuntimeSource.indexOf(READY_ENTRY) + READY_ENTRY.length) + .split('\nexport ')[0] + + it('arms the sampler unconditionally inside the function app readiness runs', () => { + expect(readyRuntimeSource).toContain( + "import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics'" + ) + expect(readyRuntimeSource).toContain(READY_ENTRY) + expect(readyRuntimeEntryBody.split('startPreGoneCrashSampling()').length - 1).toBe(1) + // Why pin the indent: the call also matches as the body of an added + // `if (...)` guard, which keeps every other assertion here true while the + // sampler silently stops arming on most startups. + expect(readyRuntimeEntryBody).toContain('\n startPreGoneCrashSampling()') + + // ...and that this really is the function app readiness runs. + expect(readySource).toContain( + "import { initializeReadyRuntimeServices } from './main-process-ready-runtime'" + ) + expect(readySource).toContain('\n await initializeReadyRuntimeServices()') + }) +}) diff --git a/src/main/windows/windows-pty-job.ts b/src/main/windows/windows-pty-job.ts index 193b1d8825a..169db375f3b 100644 --- a/src/main/windows/windows-pty-job.ts +++ b/src/main/windows/windows-pty-job.ts @@ -118,8 +118,10 @@ export function terminatePtyJob(proc: IPty): JobTerminationOutcome { /** * Pids still alive in a PTY's tree, or null when there is no answer. * - * Measured on Windows 11: once the shell exits, node-pty drops its handle - * record and closes the job, so a terminated tree reports **null**, not `[]`. + * Measured on Windows 11: once the shell exits, node-pty closes the job, so a + * terminated tree reports **null**, not `[]`. (Its handle record now outlives + * the shell until `kill()` runs — see config/patches/node-pty@1.1.0.patch — but + * the nulled job handle is what makes the answer null either way.) * Null therefore means "unverifiable" in the sense of * docs/reference/ssh-execution-boundary.md — this build has no job support, * the terminal is not a ConPTY, or it is no longer tracked. It is never diff --git a/src/preload/api/orca-profile-api.ts b/src/preload/api/orca-profile-api.ts index 16c2a575078..80e9f08fc8d 100644 --- a/src/preload/api/orca-profile-api.ts +++ b/src/preload/api/orca-profile-api.ts @@ -28,6 +28,8 @@ import type { export type OrcaProfileApi = { list: () => Promise authStatus: () => Promise + /** Fires when main changed the stored auth status on its own (e.g. a revoked session). */ + onAuthStatusChanged: (callback: () => void) => () => void createLocal: (args?: CreateLocalOrcaProfileArgs) => Promise createCloudLinked: ( args?: CreateCloudLinkedOrcaProfileArgs diff --git a/src/preload/api/orca-profiles-bridge.ts b/src/preload/api/orca-profiles-bridge.ts index 0b2897f8ab1..da58b2d9def 100644 --- a/src/preload/api/orca-profiles-bridge.ts +++ b/src/preload/api/orca-profiles-bridge.ts @@ -1,9 +1,15 @@ import { ipcRenderer } from 'electron' import type { PreloadApi } from '../api-types' +import { ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL } from '../../shared/orca-profiles' export const orcaProfilesApi = { list: () => ipcRenderer.invoke('orcaProfiles:list'), authStatus: () => ipcRenderer.invoke('orcaProfiles:authStatus'), + onAuthStatusChanged: (callback: () => void): (() => void) => { + const listener = (): void => callback() + ipcRenderer.on(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL, listener) + return () => ipcRenderer.removeListener(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL, listener) + }, createLocal: (args) => ipcRenderer.invoke('orcaProfiles:createLocal', args), createCloudLinked: (args) => ipcRenderer.invoke('orcaProfiles:createCloudLinked', args), switchProfile: (args) => ipcRenderer.invoke('orcaProfiles:switch', args), diff --git a/src/preload/api/pty-api.ts b/src/preload/api/pty-api.ts index bf012306675..a1850398357 100644 --- a/src/preload/api/pty-api.ts +++ b/src/preload/api/pty-api.ts @@ -112,7 +112,7 @@ export type PtyApi = { getForegroundProcess: (id: string) => Promise inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ) => Promise confirmForegroundProcess: (id: string) => Promise getCwd: (id: string) => Promise diff --git a/src/preload/api/pty-bridge-stream-and-serialization.ts b/src/preload/api/pty-bridge-stream-and-serialization.ts index 7e2d7bbe4ff..414a5514bfa 100644 --- a/src/preload/api/pty-bridge-stream-and-serialization.ts +++ b/src/preload/api/pty-bridge-stream-and-serialization.ts @@ -7,7 +7,7 @@ import type { TerminalProcessInspection } from '../../shared/terminal-process-in export const ptyStreamAndSerializationApi = { inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise => ipcRenderer.invoke('pty:inspectProcess', { id, ...options }), confirmForegroundProcess: (id: string): Promise => diff --git a/src/relay/pty-child-process-inspection.ts b/src/relay/pty-child-process-inspection.ts new file mode 100644 index 00000000000..6d246817053 --- /dev/null +++ b/src/relay/pty-child-process-inspection.ts @@ -0,0 +1,81 @@ +/** + * Whether anything is running under a pane's shell. + * + * Split out of `pty-shell-utils` because it is a distinct question from "what is in front" and + * carries its own platform reasoning, its own cost budget, and the verdict vocabulary from + * docs/reference/ssh-execution-boundary.md. + */ +import { queryWindowsPaneProcessInventory } from '../main/providers/windows-foreground-process-rows' +import { getProcessTableIndex } from '../shared/process-table-index' +import { + getFreshProcessTableSnapshot, + getProcessTableSnapshot +} from '../shared/process-table-snapshot-reader' +import type { PtyChildProcessVerdict } from '../shared/terminal-process-inspection' +import { isProcessAlive } from './pty-shell-utils' + +/** + * Check whether a process has child processes. + * + * Why the shared snapshot and not `pgrep -P`: this answers one field of + * `pty.inspectProcess`, which every tracked pane polls on a 750ms/2000ms + * cadence, and the fork was neither cached nor coalesced. procps-ng opens six + * procfs files per process to resolve a ppid — including a `/proc//ctty` + * that never exists on Linux — so one call cost O(host process count) syscalls, + * ~4k opens per pgrep on a 690-process host, at up to 8 forks/sec (#13537). + * `getForegroundProcessName` in the same RPC already captured the TTL-cached + * `ps` table, whose index carries the parent/child map, so the answer is free. + * + * `fresh` opts out of that TTL. A poll can read a 500ms-old table because its + * next tick corrects it, but a close or cleanup decision acts on the answer + * once and destructively — a child that started inside the TTL would be killed + * with no confirmation. `pgrep` scanned per call, so anything that decides + * has to keep scanning per call. + */ +export async function inspectPtyChildProcesses( + pid: number, + options?: { fresh?: boolean } +): Promise { + if (process.platform === 'win32') { + // Windows has no `ps`, but it does have a process table, and the pane walk over it already + // exists for the foreground reader. Answering `false` from nothing was the older shape: a + // hardcoded negative is indistinguishable from a measurement, and every close guard reads it + // as "nothing is running here". + // + // Deliberately the TTL-cached table even when `fresh` is asked for: on a relay without the + // native binding this falls back to the CIM scan, whose own 1.36s runtime is longer than the + // 500ms TTL a fresh read would be refreshing, so a "fresh" answer is not meaningfully fresher + // while N sequential ones are an N x 1.36s stall. + const inventory = await queryWindowsPaneProcessInventory(pid) + if (inventory) { + return inventory.candidates.length > 0 ? 'children' : 'no-children' + } + // A null inventory is an unreadable table OR a snapshot that never showed the root, and + // neither of those looked at the pane. The one answer available without the table is a root + // the kernel says is gone: nothing runs under a shell that does not exist. + return isProcessAlive(pid) ? 'unverifiable' : 'no-children' + } + try { + const rows = options?.fresh + ? await getFreshProcessTableSnapshot() + : await getProcessTableSnapshot() + return (getProcessTableIndex(rows).childrenByPpid.get(pid)?.length ?? 0) > 0 + ? 'children' + : 'no-children' + } catch { + return 'unverifiable' + } +} + +/** + * The boolean the wire has always carried. `unverifiable` keeps spelling itself `false` here on + * purpose: this value reaches clients too old to know the third answer, and it is read both as + * "busy, do not close" and as "the agent has taken over, safe to type into", so no single mapping + * of `unverifiable` is safe for both. Callers that can act on the distinction read the verdict. + */ +export async function processHasChildren( + pid: number, + options?: { fresh?: boolean } +): Promise { + return (await inspectPtyChildProcesses(pid, options)) === 'children' +} diff --git a/src/relay/pty-handler-spawn-admission.test.ts b/src/relay/pty-handler-spawn-admission.test.ts index c25a623f406..045ee2e412d 100644 --- a/src/relay/pty-handler-spawn-admission.test.ts +++ b/src/relay/pty-handler-spawn-admission.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' +import * as ptyChildProcessInspection from './pty-child-process-inspection' import * as ptyShellUtils from './pty-shell-utils' import * as processTableSnapshotReader from '../shared/process-table-snapshot-reader' @@ -92,7 +93,7 @@ describe('PtyHandler', () => { }) it('rescans the process table for a close decision but not for a poll', async () => { - const hasChildren = vi.mocked(ptyShellUtils.processHasChildren) + const hasChildren = vi.mocked(ptyChildProcessInspection.processHasChildren) const snapshot = vi .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') .mockResolvedValue({ diff --git a/src/relay/pty-handler-test-harness.ts b/src/relay/pty-handler-test-harness.ts index e3d99213ea5..fe6e17c5d9c 100644 --- a/src/relay/pty-handler-test-harness.ts +++ b/src/relay/pty-handler-test-harness.ts @@ -1,6 +1,6 @@ import { vi } from 'vitest' import type { Mock } from 'vitest' -import * as ptyShellUtils from './pty-shell-utils' +import * as ptyChildProcessInspection from './pty-child-process-inspection' import { PtyHandler } from './pty-handler' import type { RelayDispatcher } from './dispatcher' @@ -115,7 +115,7 @@ export function beginPtyHandlerTest(mocks: PtyHandlerTestMocks): { notifyOutput: vi.fn(), dispose: vi.fn() }) - vi.spyOn(ptyShellUtils, 'processHasChildren').mockResolvedValue(false) + vi.spyOn(ptyChildProcessInspection, 'processHasChildren').mockResolvedValue(false) mockPtySpawn.mockReturnValue({ ...mockPtyInstance }) diff --git a/src/relay/pty-handler-windows-child-process-evidence.test.ts b/src/relay/pty-handler-windows-child-process-evidence.test.ts new file mode 100644 index 00000000000..6d723824a04 --- /dev/null +++ b/src/relay/pty-handler-windows-child-process-evidence.test.ts @@ -0,0 +1,132 @@ +// Regression guard for the Windows SSH child-process answer. The relay used to return a hardcoded +// `false` here, which every close guard reads as "nothing is running in this pane" -- so a Windows +// SSH pane running a build closed with no prompt. The answer now comes from the process table, and +// the one thing it may never do again is fabricate a negative. +// +// The second contract is cost. `pty.inspectProcess` is the polled path (750ms/2000ms per tracked +// pane) and a relay host has no `@vscode/windows-process-tree`, so its table read falls back to a +// 1.36s CIM scan. Polling that would reinstate the fork storm the shared table exists to prevent, +// so only a caller whose answer decides something asks for the scan. +import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' + +const { mockPtySpawn, mockPtyInstance, mockCreateShellPromptReadinessProbe } = vi.hoisted(() => ({ + mockPtySpawn: vi.fn(), + mockCreateShellPromptReadinessProbe: vi.fn(), + mockPtyInstance: { + pid: process.pid, + process: 'xterm-256color', + onData: vi.fn(), + onExit: vi.fn(), + write: vi.fn(), + resize: vi.fn(), + kill: vi.fn(), + clear: vi.fn(), + pause: vi.fn(), + resume: vi.fn() + } +})) + +vi.mock('node-pty', () => ({ spawn: mockPtySpawn })) + +vi.mock('../main/pty/posix-pty-process-groups', () => ({ + forceKillPosixPtyProcessGroups: vi.fn((_pid: number, fallback: () => void) => fallback()) +})) + +vi.mock('../main/shell-prompt-readiness-probe', () => ({ + createShellPromptReadinessProbe: mockCreateShellPromptReadinessProbe +})) + +import * as ptyChildProcessInspection from './pty-child-process-inspection' +import type { PtyHandler } from './pty-handler' +import { + beginPtyHandlerTest, + createPtyRequestHelpers, + endPtyHandlerTest +} from './pty-handler-test-harness' +import type { MockDispatcher } from './pty-handler-test-harness' + +type Inspection = { + foregroundProcess: string | null + hasChildProcesses: boolean + childProcessEvidence?: string +} + +describe('PtyHandler Windows child-process evidence', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + let inspectChildren: ReturnType + + const { spawnPty } = createPtyRequestHelpers(() => dispatcher) + + /** Spawn under the harness's POSIX platform, then answer as the Windows relay would. */ + async function spawnThenBecomeWindows(): Promise { + const { id } = await spawnPty() + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + return id + } + + async function inspect(params: Record): Promise { + return (await dispatcher.callRequest('pty.inspectProcess', params)) as Inspection + } + + beforeEach(() => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + inspectChildren = vi + .spyOn(ptyChildProcessInspection, 'inspectPtyChildProcesses') + .mockResolvedValue('no-children') + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('publishes what the host observed when the caller pays for the scan', async () => { + const id = await spawnThenBecomeWindows() + inspectChildren.mockResolvedValue('children') + + const result = await inspect({ id, scanChildProcesses: true }) + + expect(inspectChildren).toHaveBeenCalledWith(mockPtyInstance.pid) + expect(result.childProcessEvidence).toBe('children') + expect(result.hasChildProcesses).toBe(true) + }) + + it('reports an observed-empty pane as no-children, not merely false', async () => { + const id = await spawnThenBecomeWindows() + inspectChildren.mockResolvedValue('no-children') + + const result = await inspect({ id, scanChildProcesses: true }) + + // Asserted alongside the value so the case fails if the answer stops coming from a real read. + expect(inspectChildren).toHaveBeenCalledWith(mockPtyInstance.pid) + expect(result.childProcessEvidence).toBe('no-children') + expect(result.hasChildProcesses).toBe(false) + }) + + it('keeps the compatibility boolean false when the host could not observe the pane', async () => { + const id = await spawnThenBecomeWindows() + inspectChildren.mockResolvedValue('unverifiable') + + const result = await inspect({ id, scanChildProcesses: true }) + + expect(result.childProcessEvidence).toBe('unverifiable') + // Clients too old to read the verdict also read `true` as "an agent took the PTY, safe to + // type into it", so `unverifiable` must not be promoted to `true` on the shared boolean. + expect(result.hasChildProcesses).toBe(false) + }) + + it('never reads the process table for a poll, and says so instead of guessing', async () => { + const id = await spawnThenBecomeWindows() + + const result = await inspect({ id }) + + expect(inspectChildren).not.toHaveBeenCalled() + expect(result.childProcessEvidence).toBe('unverifiable') + expect(result.hasChildProcesses).toBe(false) + }) +}) diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 42f8bdc0a77..190bcabb3f9 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -10,11 +10,11 @@ import type { RelayDispatcher, RequestContext } from './dispatcher' import { resolveDefaultShell, resolveProcessCwd, - processHasChildren, getForegroundProcessName, isProcessAlive, listShellProfiles } from './pty-shell-utils' +import { inspectPtyChildProcesses, processHasChildren } from './pty-child-process-inspection' import { getRelayShellLaunchConfig, isRelayWslShell } from './pty-shell-launch' import { RetiredPaneSurfaceRegistry } from './retired-pane-surfaces' import { addWslEnvKeys } from '../shared/wsl-env' @@ -55,6 +55,7 @@ import { import { isTuiAgent } from '../shared/tui-agent-config' import type { TuiAgent } from '../shared/tui-agent' import { forceKillPosixPtyProcessGroups } from '../main/pty/posix-pty-process-groups' +import type { PtyChildProcessVerdict } from '../shared/terminal-process-inspection' import { terminatePtyJob } from '../main/windows/windows-pty-job' import { stripInheritedBuildModeEnv } from '../main/pty/build-mode-env' import { stripLegacyTerminalShimEnv } from '../main/pty/legacy-terminal-shim-dir' @@ -2610,6 +2611,7 @@ export class PtyHandler { private async inspectProcess(params: Record): Promise<{ foregroundProcess: string | null hasChildProcesses: boolean + childProcessEvidence?: PtyChildProcessVerdict foregroundProcessEvidence?: RemoteForegroundEvidence }> { pruneRetiredPtyIncarnations(this.retiredIncarnations) @@ -2716,14 +2718,28 @@ export class PtyHandler { evidence?.verdict === 'live' ? (evidence.processName ?? managed.pty.process) || null : managed.pty.process || null + // Derive child liveness from the same capture; do not fork a second process-table probe for + // each field/pane in an event burst. + // + // Why Windows is gated on the caller asking: this is the one field whose Windows answer costs + // a process-table read, and `inspectProcess` is the polled path (750ms/2000ms per tracked + // pane). A relay host has no `@vscode/windows-process-tree`, so the read falls back to the + // 1.36s CIM scan, and polling that would reinstate exactly the fork storm the shared table + // exists to prevent (#15209, #15036). Close and cleanup decisions ask for the scan by name; + // a poll gets the honest `unverifiable` instead of a fabricated negative. + const childProcessEvidence: PtyChildProcessVerdict = rows + ? rows.some((row) => row.ppid === managed.pty.pid) + ? 'children' + : 'no-children' + : process.platform === 'win32' && params.scanChildProcesses !== true + ? 'unverifiable' + : await inspectPtyChildProcesses(managed.pty.pid) return { foregroundProcess, - // Derive child liveness from the same capture; do not fork a second - // process-table probe for each field/pane in an event burst. Windows - // has no evidence capture, so preserve the compatibility child probe. - hasChildProcesses: rows - ? rows.some((row) => row.ppid === managed.pty.pid) - : await processHasChildren(managed.pty.pid), + // `unverifiable` keeps spelling itself `false` on the compatibility field, which is what + // every client too old to read the verdict receives. + hasChildProcesses: childProcessEvidence === 'children', + childProcessEvidence, ...(evidence ? { foregroundProcessEvidence: evidence } : {}) } } diff --git a/src/relay/pty-shell-utils.test.ts b/src/relay/pty-shell-utils.test.ts index 95d6a8e050f..94953aae1bc 100644 --- a/src/relay/pty-shell-utils.test.ts +++ b/src/relay/pty-shell-utils.test.ts @@ -14,10 +14,10 @@ vi.mock('child_process', () => ({ import { resetWindowsProcessRowsSnapshotForTests } from '../main/providers/windows-foreground-process-rows' import { __setWindowsProcessTreeLoaderForTests } from '../main/windows/windows-process-table' import { resetProcessTableSnapshotForTests } from '../shared/process-table-snapshot-reader' +import { inspectPtyChildProcesses, processHasChildren } from './pty-child-process-inspection' import { getForegroundProcessName, isProcessAlive, - processHasChildren, resolveDefaultCwd, resolveWindowsDefaultShell } from './pty-shell-utils' @@ -42,6 +42,14 @@ function mockExecFile( * Feed the native Windows snapshot. A real snapshot always contains the * querying process, and the reader rejects a table without it. */ +/** A native reader that answers, but with no snapshot -- an unreadable table, not an empty one. */ +function mockUnreadableWindowsProcessTable(): void { + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2 }, + getAllProcesses: (cb: (value: undefined) => void) => cb(undefined) + })) +} + function mockWindowsProcessTable( rows: { pid: number; ppid: number; name: string; commandLine?: string }[] ): void { @@ -594,19 +602,79 @@ describe('processHasChildren', () => { }) }) - it('reports no children when the process table is unreadable', async () => { + it('reports an unreadable POSIX table as unverifiable, and still spells it false on the wire', async () => { await withProcessPlatform('linux', async () => { mockExecFile(() => new Error('ps table unavailable')) + await expect(inspectPtyChildProcesses(100)).resolves.toBe('unverifiable') await expect(processHasChildren(100)).resolves.toBe(false) }) }) +}) - it('spawns nothing on Windows, where the answer was always false', async () => { +describe('inspectPtyChildProcesses on Windows', () => { + // Why this describe exists: the relay used to `return false` here unconditionally, and a + // hardcoded negative is indistinguishable from a measurement. Every close guard reads it as + // "nothing is running here", so a Windows SSH pane running a build closed with no prompt. + it('walks the process table rather than answering from nothing', async () => { await withProcessPlatform('win32', async () => { - await expect(processHasChildren(100)).resolves.toBe(false) + mockWindowsProcessTable([ + { pid: 100, ppid: 99, name: 'cmd.exe', commandLine: 'cmd.exe' }, + { pid: 101, ppid: 100, name: 'PING.EXE', commandLine: 'ping -n 40 127.0.0.1' } + ]) - expect(execFileMock).not.toHaveBeenCalled() + await expect(inspectPtyChildProcesses(100)).resolves.toBe('children') + await expect(processHasChildren(100)).resolves.toBe(true) + }) + }) + + it('finds a grandchild the shell backgrounded, not just direct children', async () => { + await withProcessPlatform('win32', async () => { + mockWindowsProcessTable([ + { pid: 100, ppid: 99, name: 'cmd.exe', commandLine: 'cmd.exe' }, + { pid: 101, ppid: 100, name: 'node.exe', commandLine: 'node build.js' }, + { pid: 102, ppid: 101, name: 'tsc.exe', commandLine: 'tsc --watch' } + ]) + + await expect(inspectPtyChildProcesses(102)).resolves.toBe('no-children') + await expect(inspectPtyChildProcesses(101)).resolves.toBe('children') + }) + }) + + it('separates an observed-empty shell from a table it could not read', async () => { + await withProcessPlatform('win32', async () => { + mockWindowsProcessTable([{ pid: 100, ppid: 99, name: 'cmd.exe', commandLine: 'cmd.exe' }]) + await expect(inspectPtyChildProcesses(100)).resolves.toBe('no-children') + + resetWindowsProcessRowsSnapshotForTests() + mockUnreadableWindowsProcessTable() + const alive = vi.spyOn(process, 'kill').mockReturnValue(true as never) + try { + await expect(inspectPtyChildProcesses(100)).resolves.toBe('unverifiable') + // The compatibility boolean keeps spelling unverifiable `false`: it reaches clients that + // cannot read the verdict, and they read `true` as "an agent took the PTY, safe to type". + await expect(processHasChildren(100)).resolves.toBe(false) + } finally { + alive.mockRestore() + } + }) + }) + + it('does not read a missing shell as unverifiable when the kernel says it is gone', async () => { + await withProcessPlatform('win32', async () => { + // The root is absent from the snapshot, which on its own cannot distinguish a filtered + // table from an exited shell. Only ESRCH settles it. + mockWindowsProcessTable([{ pid: 900, ppid: 1, name: 'explorer.exe' }]) + const gone = vi.spyOn(process, 'kill').mockImplementation(() => { + const error = new Error('no such process') as NodeJS.ErrnoException + error.code = 'ESRCH' + throw error + }) + try { + await expect(inspectPtyChildProcesses(100)).resolves.toBe('no-children') + } finally { + gone.mockRestore() + } }) }) }) diff --git a/src/relay/pty-shell-utils.ts b/src/relay/pty-shell-utils.ts index d84faad6ae9..9c8585933a1 100644 --- a/src/relay/pty-shell-utils.ts +++ b/src/relay/pty-shell-utils.ts @@ -11,10 +11,7 @@ import { import { getFirstCommandToken } from '../shared/command-token-scanner' import { getProcessTableIndex, type ProcessTableIndex } from '../shared/process-table-index' import { PS_MAX_BUFFER_BYTES, type ProcessTableRow } from '../shared/process-table-snapshot' -import { - getFreshProcessTableSnapshot, - getProcessTableSnapshot -} from '../shared/process-table-snapshot-reader' +import { getProcessTableSnapshot } from '../shared/process-table-snapshot-reader' import { selectForegroundProcessCandidate } from '../shared/foreground-process-selection' import { resolveOuterWrapperForegroundProcess, @@ -169,43 +166,6 @@ export async function resolveProcessCwd(pid: number, fallbackCwd: string): Promi return fallbackCwd } -/** - * Check whether a process has child processes. - * - * Why the shared snapshot and not `pgrep -P`: this answers one field of - * `pty.inspectProcess`, which every tracked pane polls on a 750ms/2000ms - * cadence, and the fork was neither cached nor coalesced. procps-ng opens six - * procfs files per process to resolve a ppid — including a `/proc//ctty` - * that never exists on Linux — so one call cost O(host process count) syscalls, - * ~4k opens per pgrep on a 690-process host, at up to 8 forks/sec (#13537). - * `getForegroundProcessName` in the same RPC already captured the TTL-cached - * `ps` table, whose index carries the parent/child map, so the answer is free. - * - * `fresh` opts out of that TTL. A poll can read a 500ms-old table because its - * next tick corrects it, but a close or cleanup decision acts on the answer - * once and destructively — a child that started inside the TTL would be killed - * with no confirmation. `pgrep` scanned per call, so anything that decides - * has to keep scanning per call. - */ -export async function processHasChildren( - pid: number, - options?: { fresh?: boolean } -): Promise { - // Windows has no `ps`; the previous `pgrep` fork always failed here too, so - // this keeps the same answer without spawning anything to reach it. - if (process.platform === 'win32') { - return false - } - try { - const rows = options?.fresh - ? await getFreshProcessTableSnapshot() - : await getProcessTableSnapshot() - return (getProcessTableIndex(rows).childrenByPpid.get(pid)?.length ?? 0) > 0 - } catch { - return false - } -} - // Why: signal 0 probes existence without delivering a signal. Only ESRCH ("no // such process") proves the pid is gone; EPERM means it exists but is // unsignalable, so treat every non-ESRCH outcome as alive. Kept conservative so diff --git a/src/renderer/src/assets/main.css b/src/renderer/src/assets/main.css index 8187fe496c7..e3d267cb353 100644 --- a/src/renderer/src/assets/main.css +++ b/src/renderer/src/assets/main.css @@ -392,9 +392,9 @@ z-index: 40 !important; } -/* Keep interruption controls above unrelated updater/onboarding chrome. */ +/* Above the z-40 updater/onboarding chrome, below the floating workspace panel's z-45. */ .native-chat-pane-shell:has([data-native-chat-working='true']) { - z-index: 50; + z-index: 44; } [data-sonner-toaster] [data-sonner-toast][data-styled='true'] { diff --git a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx index 52ddbf1ba5d..bb3ffbfd621 100644 --- a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx +++ b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx @@ -24,6 +24,7 @@ export function TerminalWorkspaceDialogs({ saveDialogFile, saveDialogFileId, setWindowCloseDialogOpen, + windowCloseDialogKind, windowCloseDialogOpen } = controller return ( @@ -82,10 +83,15 @@ export function TerminalWorkspaceDialogs({ {translate('auto.components.Terminal.2fa9c69ff3', 'Close Window?')} - {translate( - 'auto.components.Terminal.7958465754', - 'There are local terminals with running processes. Close the window anyway?' - )} + {windowCloseDialogKind === 'unverifiable' + ? translate( + 'auto.components.Terminal.b7c1f0a934', + 'A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?' + ) + : translate( + 'auto.components.Terminal.7958465754', + 'There are terminals with running processes. Close the window anyway?' + )} diff --git a/src/renderer/src/components/cmd-j/palette-live-status.test.tsx b/src/renderer/src/components/cmd-j/palette-live-status.test.tsx index a2656104d87..3cc6a9077dd 100644 --- a/src/renderer/src/components/cmd-j/palette-live-status.test.tsx +++ b/src/renderer/src/components/cmd-j/palette-live-status.test.tsx @@ -451,7 +451,7 @@ describe('palette live status', () => { expect(dotLabels()).toEqual(['Needs permission']) }) - it('cuts the pip out of the dialog surface, and out of accent when selected', async () => { + it('keeps the attention glyph knockout popover-colored when its row is selected', async () => { setAgentState('working') await act(async () => { testRoot.render( @@ -469,18 +469,11 @@ describe('palette live status', () => { ) }) - const pip = testContainer.querySelector('[aria-hidden="true"].rounded-full') + const pip = testContainer.querySelector('[aria-hidden="true"]') expect(pip).not.toBeNull() - // Why popover and not background: the CommandDialog surface is --popover (#171717 dark), while - // --background is the app canvas (#0a0a0a) — the mismatch punched a dark halo through each row. expect(pip?.className).toContain('bg-popover') expect(pip?.className).toContain('ring-popover') - expect(pip?.className).not.toContain('bg-background') - expect(pip?.className).toContain( - 'group-data-[selected=true]:bg-[var(--jump-palette-selection-surface)]' - ) - expect(pip?.className).toContain( - 'group-data-[selected=true]:ring-[var(--jump-palette-selection-surface)]' - ) + expect(pip?.className).toContain('rounded-full') + expect(pip?.className).not.toContain('group-data-[selected=true]') }) }) diff --git a/src/renderer/src/components/cmd-j/palette-live-status.tsx b/src/renderer/src/components/cmd-j/palette-live-status.tsx index 432b6c35210..da59663490f 100644 --- a/src/renderer/src/components/cmd-j/palette-live-status.tsx +++ b/src/renderer/src/components/cmd-j/palette-live-status.tsx @@ -9,7 +9,6 @@ import { buildExplicitEntriesByTabId, type TabPaneInputSources } from '@/components/sidebar/smart-attention' -import { cn } from '@/lib/utils' import { isExplicitAgentStatusFresh } from '@/lib/agent-status' import { getLiveAgentStatusByWorktreeId } from '@/lib/worktree-activity-state' import { @@ -255,15 +254,8 @@ export function PaletteRecentTabStatusDot({ {fallback}