diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index ed37cefb434..ca1651325c9 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -816,6 +816,7 @@ jobs: src/main/cli/wsl-cli-powershell-boundary.test.ts src/main/cursor/hook-service.test.ts src/main/orca-profiles/profile-index-store.test.ts + src/main/startup/windows-install-dir-acl-repair.win32.test.ts src/main/runtime/repo-worktree-admin-fingerprint.test.ts src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts src/shared/secure-file-fsync-flags.test.ts diff --git a/AGENTS.md b/AGENTS.md index 6915b246bcc..8b0156ba6b1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -53,18 +53,6 @@ Orca targets macOS, Linux, and Windows. Keep all platform-dependent behavior beh - **WSL commands**: build argv with `buildWslExecArgs` (always `--exec` — under `--`, `wsl.exe` expands `$name` in every argument and silently rewrites the script), and fence anything whose stdout you parse with `buildWslCapturedLoginShellCommand`, because the interactive login shell prints the distro banner to stdout. See [`docs/reference/wsl-command-execution.md`](./docs/reference/wsl-command-execution.md). - **Linux native modules**: keep the glibc floor at Ubuntu 20.04 / glibc 2.31. A module compiled from source on a newer runner can reference symbol versions absent on the floor and crash the app on startup. See [`docs/reference/linux-glibc-compatibility.md`](./docs/reference/linux-glibc-compatibility.md); packaging fails if a bundled native binary needs newer glibc. -## Localization (i18n) - -All user-facing copy is localized. `src/renderer/src/i18n/locales/en.json` is the source of truth; the `zh`, `ja`, `ko`, and `es` catalogs mirror its keys. Strings reach the UI through `translate('auto.', 'English fallback')` — never hardcode display text. - -When you touch user-facing copy, keep all five catalogs in sync: - -- **New strings** — wrap them in `translate(...)` with an English fallback, run `pnpm sync:localization-catalog` to register the keys in `en.json` and add placeholders to every other locale, then `pnpm bootstrap:-catalog` (e.g. `bootstrap:ja-catalog`) to translate the placeholders. -- **Reworded strings** — changing the value of an existing key updates only `en.json`. The other locales keep the key with its old translation, and **the lint checks will not catch this**: `verify:localization-catalog` enforces key _parity_, not translation _freshness_. Update the same key in `zh/ja/ko/es` by hand, or re-translate it via `bootstrap:-catalog`. -- **Removed strings** — delete the key from _every_ locale; the parity check rejects a key that exists in one catalog but not another. - -Before pushing copy changes, run `pnpm verify:localization-catalog` and `pnpm verify:localization-coverage` (both also run in `pnpm lint`). - ## SSH Use Case All changes must consider the SSH use case. Don't assume local-only execution. Before changing anything that reports on, stops, or lists remote work, follow [`docs/reference/ssh-execution-boundary.md`](./docs/reference/ssh-execution-boundary.md): the execution host owns everything that touches execution, and loss of contact is never evidence of process death — the verdict vocabulary is `live` / `unverifiable` / `exited`, with no synonyms. diff --git a/README.md b/README.md index 805c94295cc..2ae59035da8 100644 --- a/README.md +++ b/README.md @@ -12,7 +12,7 @@

- 中文 · 日本語 · 한국어 · Español · Français · Português · Українська + 中文 · 日本語 · 한국어 · Español · Français · Português

@@ -238,9 +238,9 @@ Pair with your desktop app to monitor and steer your agents from your phone. - **Discord:** Join the community on **[Discord](https://discord.gg/fzjDKHxv8Q)**. - **Twitter / X:** Follow **[@orca_build](https://x.com/orca_build)** for updates and announcements. -- **WeChat:** Scan to join the Orca community WeChat group 8. +- **WeChat:** Scan to join the Orca community WeChat group 8. Group 8 may be full; if so, scan the Group 9 QR code instead. - WeChat group 8 QR code for the Orca community + WeChat group 8 QR code for the Orca community  WeChat group 9 QR code for the Orca community - **Feedback & Ideas:** We ship fast. Missing something? [Request a new feature](https://github.com/stablyai/orca/issues). - **Privacy:** See the [privacy & telemetry docs](https://www.onorca.dev/docs/telemetry) for what anonymous usage data Orca collects and how to opt out. diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index ea5ad55b645..4e1da9fab26 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -111,27 +111,41 @@ describe('incident monitor evaluator', () => { }) }) - it('freezes when postgres retries exceed the recalibrated ceiling', () => { - const sample = healthySample() - sample.sources['relay-logs']!.signals['relay.postgres_retries'] = - signal(INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries + 1) - expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + // Why: the global relay_cells lock made retries a steady-state rate (24 h p99 + // 1320/5min on 2026-09-04); the bar fences only unbounded growth beyond that. + it('tolerates the measured healthy retry rate and freezes above the bar', () => { + const healthy = healthySample() + healthy.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(1504) + expect(evaluateIncidentSample(healthy, startedAt).status).toBe('green') + + const incident = healthySample() + incident.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(2001) + expect(evaluateIncidentSample(incident, startedAt)).toMatchObject({ status: 'freeze', failures: [ - expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 300 }) + expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 2000 }) ] }) }) - // Why: sweeps no longer reach the retry wrapper, so any exhaustion left in this - // counter is a request path that terminally failed. It must still freeze. - it('freezes on a single exhausted request-path transaction', () => { - const sample = healthySample() - sample.sources['relay-logs']!.signals['relay.postgres_retry_exhausted'] = signal(1) - expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + // Why: since #18521 the request path fails fast on the cell-inventory lock, so + // exhaustion is a steady contention rate (post-#18521 p90 147/5min, max 220), + // not an anomaly. The bar bounds it below the 2026-08-23 incident peak of 467. + it('tolerates the measured healthy exhaustion rate and freezes above the bar', () => { + const healthy = healthySample() + healthy.sources['relay-logs']!.signals['relay.postgres_retry_exhausted'] = signal(220) + expect(evaluateIncidentSample(healthy, startedAt).status).toBe('green') + + const atLimit = healthySample() + atLimit.sources['relay-logs']!.signals['relay.postgres_retry_exhausted'] = signal(300) + expect(evaluateIncidentSample(atLimit, startedAt).status).toBe('green') + + const incident = healthySample() + incident.sources['relay-logs']!.signals['relay.postgres_retry_exhausted'] = signal(301) + expect(evaluateIncidentSample(incident, startedAt)).toMatchObject({ status: 'freeze', failures: [ - expect.objectContaining({ signal: 'relay.postgres_retry_exhausted', threshold: 0 }) + expect.objectContaining({ signal: 'relay.postgres_retry_exhausted', threshold: 300 }) ] }) }) diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index a1936f5df6c..a121568d918 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -15,8 +15,8 @@ export const INCIDENT_MONITOR_THRESHOLDS = { // Why: healthy latest-sum backends idle near 100 but spike to 216 in 1-minute // bursts (~10 min/day exceeded the old bar of 160 on 2026-08-26, freezing a // pre-drain gate on baseline noise). 250 clears measured healthy peaks while - // firing well before the verified 400-connection ceiling; pool-wait and - // exhausted-retry signals keep their strict thresholds. + // firing well before the verified 400-connection ceiling; the retry signals + // below discriminate incident-class contention. cloudSqlBackends: 250, // Bound the observed recovery load; deadlocks remain zero-tolerance. cloudSqlLockWaits: 20, @@ -32,27 +32,35 @@ export const INCIDENT_MONITOR_THRESHOLDS = { relayPoolWaiting: 800, relayPoolWaitMs: 2_500, // Why: successful lock retries are the contention machinery working, not harm. - // Healthy 2026-08-26 baseline bursts to 234/5min (26% of windows crossed the old - // bar of 20, set unmeasured at the monitor's 2026-07-28 birth); the 2026-08-23 - // incident ran ~2,200-3,000/5min. 300 clears healthy bursts with ~10x incident - // margin; relayPostgresRetryExhausted below stays at zero tolerance, so any - // transaction that terminally fails still freezes the gate. - relayPostgresRetries: 300, - // Why: this bar stays at zero. Cell-inventory contention reaches the retry - // wrapper from exactly two kinds of caller, and neither is a sweep tick that - // can shrug the failure off: - // - request paths, which take a wait bounded at CELL_INVENTORY_LOCK_TIMEOUT_MS - // (assignment, control activation, activity, admin drain/evacuate/supersede); - // - sweep-reachable code that a request also enters, which keeps the pool - // lock_timeout so it cannot fail faster than before this change: the - // completeEvacuation site that waits, reconcileReservationAccounting, and - // placement re-entered from evacuateDeadCells. - // Sweep-only sites take the inventory NOWAIT, so their contention becomes - // database_lock_unavailable, which is not a retryable abort and never reaches - // this counter. The relay's cell-inventory-lock census test holds that split. - // Splitting the metric by the phase label PR #423 put on the log payload would - // need a labelled log-based metric, which this signal's counter does not carry. - relayPostgresRetryExhausted: 0, + // Recalibrated 2026-09-04 from 300, which was set 2026-08-26 when healthy bursts + // reached 234/5min. The global relay_cells FOR UPDATE lock has since become the + // fleet's steady state: measured fleet-wide (director + cells, summed per five + // minutes) 2026-09-03T05Z..2026-09-04T05Z p50 430 / p90 924 / p99 1320 / max + // 1504, with 55% of windows over 300 and only 22% of 15-minute gates clean, so + // the bar blocked the very cell roll that carries the 500 ms lock wait (#18521) + // and the beginProof crash guard to the cells. The 2026-08-23 lock incident on + // this same metric peaked at 1510 in one window and 646 in the next, so it is + // not separable from today's contention by retries alone; it is caught by + // relayPostgresRetryExhausted (467 at the peak vs a 300 bar), director + // concurrency, and the pool bars. 2000 passes every healthy 15-minute window + // measured in the last 24 h and still fences unbounded growth. Re-tighten once + // the fleet is on the 500 ms lock wait and the baseline is re-measured. + relayPostgresRetries: 2000, + // Why: 300 per five minutes, recalibrated 2026-09-04 from a bar of zero that no + // production window has cleared since #18521 shipped to the director. That + // change cut the request-path cell-inventory wait from the 1 s pool lock_timeout + // to 500 ms, so a contended waiter now fails fast (one /v1/assign 503 with + // Retry-After, which the client retries) instead of succeeding slowly, and the + // exhaustion count became a steady-state contention rate rather than an + // anomaly. Measured fleet-wide (director + cells) per five minutes over + // 2026-09-03T03Z..2026-09-04T02Z: every one of 236 windows was non-zero; + // quiet hours p50 2 / max 36; pre-#18521 daytime p50 10 / p90 25 / max 87; + // post-#18521 p50 42 / p90 147 / max 220. The 2026-08-23 lock incident peaked + // at 467. 300 clears every measured healthy window and still sits below the + // incident shape; retries above fence only unbounded growth. + // User-facing /v1/assign 503 share did not move with #18521 (13.9% old image + // vs 12.3% new, same evening), so exhaustion is not a proxy for user harm. + relayPostgresRetryExhausted: 300, // Why: public admission is a per-instance semaphore, so fleet assignment capacity is // concurrency x instances. A floor of 1 let the 2026-08-04 collapse from five instances // to two pass unnoticed, which is the exact failure this monitor exists to catch. Keep in diff --git a/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts index 6ac9521c3d6..80a74a47eeb 100644 --- a/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts +++ b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts @@ -44,6 +44,12 @@ describePostgres('PostgreSQL assignment connection headroom', () => { `DELETE FROM relay_assignments WHERE user_id LIKE 'connection-headroom-postgres-%'` ) + // A snapshot left by an aborted run rejects the replayed watermark + // with stale_connection_snapshot. + await database.query( + `DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, + [cell.id] + ) await database.query( `DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id] diff --git a/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts index 10193b78cc6..cf8819686b5 100644 --- a/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts +++ b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts @@ -38,6 +38,10 @@ describePostgres('PostgreSQL control supersession', () => { [identity.userId] ) await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [identity.userId]) + // A snapshot left by an aborted run rejects the replayed watermark with stale_connection_snapshot. + await database.query(`DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, [ + cell.id + ]) await database.query(`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id]) await database.query(`DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, [cell.id]) await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts index 96d66eb7dc2..9d240e304a7 100644 --- a/cloud/apps/relay/src/assignment-store.ts +++ b/cloud/apps/relay/src/assignment-store.ts @@ -337,8 +337,8 @@ export type CellInventoryLockMode = // Never queue: the caller handles database_lock_unavailable and moves on. | 'nowait' // A sweep can enter here, so keep the pool default. Failing sooner would turn - // ordinary contention into a 55P03 the retry wrapper reports as terminal, and - // one terminal failure freezes the incident gate. + // ordinary contention into a 55P03 the retry wrapper reports as terminal, which + // spends the incident gate's bounded exhausted-retry budget (300 per 5 min). | 'pool-default' // Why: stranded detection (issue #225) needs a grant old enough that a real // attach would have registered (the 90s activity lease covers dial + @@ -3202,8 +3202,7 @@ export class RelayAssignmentStore { ) const requestDelta = ACTIVITY_REQUEST_UNITS[kind] * (after - before) if (requestDelta !== 0) { - await this.lockCellInventory(transaction, 'request') - await this.adjustCellReservation(transaction, text(row, 'cell_id'), requestDelta) + await this.adjustCellReservationAtomically(transaction, text(row, 'cell_id'), requestDelta) } }) }) @@ -3263,9 +3262,12 @@ export class RelayAssignmentStore { } const units = ACTIVITY_REQUEST_UNITS[input.kind] if (existing) { - await this.lockCellInventory(transaction, 'request') + // Why: a client-chosen activity id can move between cells, so lock the + // one or two rows this path touches in cell_id order, the same order + // placement takes the inventory in, and no cycle can form. + await this.lockCellRows(transaction, [text(existing, 'cell_id'), input.cellId]) await this.removeActivityLease(transaction, identity, existing, now) - await this.adjustCellReservation(transaction, input.cellId, units) + await this.adjustCellReservationAtomically(transaction, input.cellId, units) } await this.adjustActivityCount(transaction, identity, input.kind, 1, expiresAt, now) await transaction.query( @@ -3580,8 +3582,7 @@ export class RelayAssignmentStore { ) await this.touchAssignment(transaction, identity, expiresAt, now) } else { - await this.lockCellInventory(transaction, 'request') - await this.adjustCellReservation(transaction, input.cellId, 1) + await this.adjustCellReservationAtomically(transaction, input.cellId, 1) await this.adjustActivityCount(transaction, identity, 'control', 1, expiresAt, now) await transaction.query( `INSERT INTO relay_assignment_activity_leases @@ -6954,6 +6955,19 @@ export class RelayAssignmentStore { return rows } + // Per-connection paths touch one or two cells. Locking exactly those rows, + // in the same ascending order the inventory lock uses (ORDER BY fixes the + // row-lock order), keeps them off the fleet-wide lock without a cycle. + private async lockCellRows(database: RelayDatabase, cellIds: string[]): Promise { + const distinct = [...new Set(cellIds)] + return await database.queryLocked( + `SELECT * FROM relay_cells WHERE cell_id IN (${distinct.map(() => '?').join(', ')}) + ORDER BY cell_id ASC`, + distinct, + { lockTimeoutMs: CELL_INVENTORY_LOCK_TIMEOUT_MS } + ) + } + private async lockGeneralCellInventory( database: RelayDatabase, mode: CellInventoryLockMode @@ -7590,7 +7604,10 @@ export class RelayAssignmentStore { ) { throw new Error('activity_lease_shape_mismatch') } - const cells = await this.lockCellInventory(database, 'request') + // Why: this recomputes one cell's reservation from its leases, so only that + // row needs to be held; the 23-row inventory lock here serialised every + // desktop control rebind in the fleet behind every other one. + const cellRow = (await this.lockCellRows(database, [cellId]))[0] await database.query( `DELETE FROM relay_assignment_activity_leases WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control' @@ -7611,7 +7628,6 @@ export class RelayAssignmentStore { [cellId] ) )[0]! - const cellRow = cells.find((cell) => text(cell, 'cell_id') === cellId) const cellUnits = integer(cellUnitsRow, 'request_units') if (!cellRow) throw new Error('assigned_cell_missing') if (cellUnits > integer(cellRow, 'capacity_requests')) { diff --git a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts index b2b5b65684c..a26e15f8e1d 100644 --- a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts +++ b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts @@ -25,10 +25,12 @@ const CENSUS: CensusEntry[] = [ { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'refreshDrainMigrationLeasesOnce', mode: 'request', reach: 'request' }, - // Reachable from neither: changeActivity has no production callers, only tests. - { method: 'changeActivity', mode: 'request', reach: 'orphan' }, - { method: 'acquireActivity', mode: 'request', reach: 'request' }, - { method: 'activateControl', mode: 'request', reach: 'request' }, + // changeActivity, acquireActivity, activateControl and + // removeSupersededSameCellControls no longer take the inventory: they lock + // only the one or two cell rows they touch, in cell_id order (lockCellRows), + // so they cannot cycle with placement's ordered inventory lock, and the + // 23-row lock there had serialised every reconnect in the fleet behind every + // other one. { method: 'startEvacuation', mode: 'request', reach: 'request' }, { method: 'completeEvacuationFromDeadSourceOnce', mode: 'request', reach: 'request' }, { method: 'completeEvacuationFromDeadSourceOnce', mode: 'nowait', reach: 'request' }, @@ -48,8 +50,31 @@ const CENSUS: CensusEntry[] = [ { method: 'releaseExpiredActivityLeases', mode: 'nowait', reach: 'sweep' }, { method: 'releaseExpiredActivity', mode: 'nowait', reach: 'sweep' }, { method: 'reconcileReservationAccounting', mode: 'pool-default', reach: 'both' }, - { method: 'leastLoadedCell', mode: 'pool-default', reach: 'both' }, - { method: 'removeSupersededSameCellControls', mode: 'request', reach: 'request' } + { method: 'leastLoadedCell', mode: 'pool-default', reach: 'both' } +] + +// Every inline `FROM relay_cells ... FOR UPDATE` outside the named lock helpers, +// in source order: whole-table locks in reconciliation and sticky placement, +// and single-row locks for a cell the method is already scoped to (heartbeat, +// fence, drain generation, configuration, or a reservation adjust that runs +// under a lock its caller already holds). A new inline lock fails the census +// below until it is listed here; per-connection paths that touch more than one +// cell go through lockCellRows so the order is fixed. +const NAMED_LOCK_HELPERS = ['lockCellInventory', 'lockGeneralCellInventory', 'lockCellRows'] + +const INLINE_CELL_LOCK_SITES = [ + 'reconcileCellsWithOptions', + 'assignStickyOnce', + 'recordCellHeartbeat', + 'attestCellFence', + 'adoptLegacyCellFence', + 'commitLegacyCellFenceAdoption', + 'prepareCellFenceAttempt', + 'attestCellFenceAttempt', + 'attestCellFenceAttempt', + 'configureCell', + 'assertDrainCellGeneration', + 'adjustCellReservation' ] // The background sweeps, and nothing else. A method reachable from one of these @@ -151,6 +176,42 @@ describe('cell inventory lock call-site census', () => { ) }) + // Why: the census only sees lockCellInventory calls, so a hand-written + // `relay_cells ... FOR UPDATE` would escape classification entirely. + it('routes every relay_cells row lock through a named lock helper', () => { + const lines = storeSource() + const rawSites: string[] = [] + // Whole statements, not a fixed window: a wide column list or a raw + // FOR UPDATE inside query() must not slip past. + const source = lines.join('\n') + const bounds: { name: string; start: number }[] = [] + lines.forEach((line, index) => { + const declaration = DECLARATION.exec(line) + if (declaration) bounds.push({ name: declaration[1]!, start: index }) + }) + const methodAt = (offset: number): string => { + const lineIndex = source.slice(0, offset).split('\n').length - 1 + let name = '' + for (const bound of bounds) if (bound.start <= lineIndex) name = bound.name + return name + } + const tick = String.fromCharCode(96) + const statementCall = new RegExp( + '\\.(queryLocked|query)\\(\\s*' + tick + '([^' + tick + ']*)' + tick, + 'g' + ) + for (const call of source.matchAll(statementCall)) { + const statement = call[2]! + if (!/\bFROM\s+relay_cells\b/.test(statement)) continue + const locks = call[1] === 'queryLocked' || /\bFOR\s+UPDATE\b/.test(statement) + if (!locks) continue + const method = methodAt(call.index) + if (NAMED_LOCK_HELPERS.includes(method)) continue + rawSites.push(method) + } + expect(rawSites).toEqual(INLINE_CELL_LOCK_SITES) + }) + it('leaves no call site taking the inventory without naming a mode', () => { const source = readFileSync(new URL('./assignment-store.ts', import.meta.url), 'utf8') const unclassified = source @@ -170,8 +231,8 @@ describe('cell inventory lock call-site census', () => { }) // Why: this is the whole point of the classification. A shorter wait on a - // sweep-reachable site turns contention into a terminal transaction failure, - // and relayPostgresRetryExhausted freezes the incident gate at zero. + // sweep-reachable site turns contention into a terminal transaction failure + // that counts against the incident gate's relayPostgresRetryExhausted bar. // Why: the hold distribution is what the 500ms bound will be tuned against, so // a mode that stops asking for it goes unmeasured in exactly the lane that // matters. Nothing else in the suite reads the pool-default branch. diff --git a/cloud/apps/relay/src/cell-inventory-lock-contention.test.ts b/cloud/apps/relay/src/cell-inventory-lock-contention.test.ts index 783d8a62e8b..23ec001c573 100644 --- a/cloud/apps/relay/src/cell-inventory-lock-contention.test.ts +++ b/cloud/apps/relay/src/cell-inventory-lock-contention.test.ts @@ -300,8 +300,8 @@ describe('bounded cell-inventory lock wait', () => { }) }) -// Why: the incident monitor freezes at zero exhausted transactions. A sweep that -// steps aside must not spend the retry budget or report a terminal failure. +// Why: exhausted transactions count against the incident monitor's bounded bar. +// A sweep that steps aside must not spend the retry budget or report a terminal failure. describe('sweep lock skips stay off the transaction retry counters', () => { it('reports neither a retry nor an exhaustion when NOWAIT finds the lock held', async () => { const database = await openFakePostgres() diff --git a/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts b/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts new file mode 100644 index 00000000000..e990ac1ed1a --- /dev/null +++ b/cloud/apps/relay/src/control-rebind-inventory-lock-postgres.test.ts @@ -0,0 +1,260 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +// Three cells: the inventory lock covers more than the rows a move touches, and +// a high-to-low move exposes any lock taken out of cell_id order. +const cells = [ + { + id: 'rebind-inventory-postgres-a', + url: 'https://rebind-inventory-postgres-a.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'rebind-inventory-postgres-b', + url: 'https://rebind-inventory-postgres-b.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'rebind-inventory-postgres-c', + url: 'https://rebind-inventory-postgres-c.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +] +const identity = { userId: 'rebind-inventory-postgres-user', relayHostId: 'rebindinvhost001' } + +function heartbeat(cell: (typeof cells)[number]) { + return { + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 1, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } +} + +// Why: every desktop control rebind used to take the fleet-wide relay_cells +// FOR UPDATE lock, so a rebind on one cell queued behind whatever held any +// other cell's row, until COMMIT (55P03 at the request bound). A rebind only +// touches its own cell row, so it must proceed while another cell's row is +// held elsewhere. +describePostgres('PostgreSQL control rebind under a held cell row', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push( + await openRelayDatabase({ databaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl, dataDir: '' }) + ) + }) + + async function removeTestRows(database: RelayDatabase): Promise { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, + [identity.userId] + ) + for (const table of [ + 'relay_assignment_activity_leases', + 'relay_post_drain_migration_pins', + 'relay_assignment_migration_incarnations', + 'relay_assignment_migrations', + 'relay_assignments' + ]) { + await database.query(`DELETE FROM ${table} WHERE user_id = ?`, [identity.userId]) + } + for (const cell of cells) { + for (const table of [ + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_connection_limits', + 'relay_cell_runtime', + 'relay_cells' + ]) { + await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [cell.id]) + } + } + } + + afterAll(async () => { + if (databases[0]) await removeTestRows(databases[0]) + for (const connection of databases) await connection.close() + }) + + it("rebinds and supersedes a control while another cell's row is held", async () => { + // A prior aborted run leaves connection snapshots that reject a replayed watermark. + await removeTestRows(databases[0]!) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + // Pin the host to cell A so placement is deterministic. + await store.setCellEnabled(cells[1]!.id, false) + await store.setCellEnabled(cells[2]!.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cells[0]!.id) + await store.setCellEnabled(cells[1]!.id, true) + await store.setCellEnabled(cells[2]!.id, true) + await store.activateControl(identity, { + cellId: cells[0]!.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 10 + }) + + // Hold only cell B's row on a second connection, the way a rebind on B + // does, for longer than the request-path lock bound. + let releaseInventory!: () => void + const inventoryReleased = new Promise((resolve) => { + releaseInventory = resolve + }) + let inventoryHeld!: () => void + const inventoryHeldPromise = new Promise((resolve) => { + inventoryHeld = resolve + }) + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cells[1]!.id]) + inventoryHeld() + await inventoryReleased + }) + await inventoryHeldPromise + + // A generation-2 rebind on cell A supersedes generation 1. It must not + // wait on cell B's row. + const startedAt = Date.now() + const blockedStatement = async (): Promise => { + const rows = await databases[1]!.query( + `SELECT left(query, 160) AS q FROM pg_stat_activity + WHERE datname = current_database() AND wait_event_type = 'Lock'` + ) + return rows.map((row) => String(row.q)).join(' | ') + } + const timeout = new Promise((_, reject) => + setTimeout( + () => + void blockedStatement().then((statement) => + reject(new Error(`rebind on cell A blocked behind cell B's row: ${statement}`)) + ), + 2_000 + ) + ) + const rebound = await Promise.race([ + store.activateControl(identity, { + cellId: cells[0]!.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 2, + connectionInclusionWatermark: 11 + }), + timeout + ]) + const elapsedMs = Date.now() - startedAt + releaseInventory() + await holder + + expect(rebound).toBe(`control:${cells[0]!.id}:2`) + expect(elapsedMs).toBeLessThan(2_000) + const controls = await databases[0]!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND activity_kind = 'control' ORDER BY activity_id`, + [identity.userId] + ) + expect(controls).toEqual([{ activity_id: `control:${cells[0]!.id}:2` }]) + const reserved = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cells[0]!.id] + ) + expect(Number(reserved[0]!.reserved_requests)).toBe(1) + }, 15_000) + + // Why: a phone's activity id is client-chosen and can follow the host across + // a migration, so acquireActivity may touch two cell rows. Moving from the + // higher cell to the lower one is where an unordered lock cycles with + // placement's ascending inventory lock (reproduced live before this fix). + it('moves an activity from a higher cell to a lower one in cell_id order', async () => { + await removeTestRows(databases[0]!) + const [cellA, cellB, cellC] = cells as [typeof cells[0], typeof cells[0], typeof cells[0]] + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells(cells) + for (const cell of cells) await store.recordCellHeartbeat(heartbeat(cell)) + await store.setCellEnabled(cellA.id, false) + await store.setCellEnabled(cellB.id, false) + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(cellC.id) + await store.setCellEnabled(cellA.id, true) + await store.setCellEnabled(cellB.id, true) + const activityId = 'splice:rebind-inventory-postgres' + await store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellC.id }) + // The migration makes B authoritative; the lease still sits on C. + const migration = await store.startEvacuation(identity, cellB.id) + expect(migration.targetCellId).toBe(cellB.id) + + // Hold B elsewhere. An ordered move locks B first and queues here holding + // nothing else. Locking C first (the old lease's row, as an unordered move + // does) or the whole inventory (which takes A) shows up as a held row. + let releaseRow!: () => void + const rowReleased = new Promise((resolve) => { + releaseRow = resolve + }) + let rowHeld!: () => void + const rowHeldPromise = new Promise((resolve) => { + rowHeld = resolve + }) + const heldWhileMoverWaits: string[] = [] + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellB.id]) + rowHeld() + await rowReleased + for (const cell of [cellA, cellC]) { + try { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id], { + failIfUnavailable: true + }) + } catch { + heldWhileMoverWaits.push(cell.id) + } + } + }) + await rowHeldPromise + const move = store.acquireActivity(identity, { activityId, kind: 'splice', cellId: cellB.id }) + let moved = false + void move.then(() => { + moved = true + }) + await new Promise((resolve) => setTimeout(resolve, 250)) + expect(moved).toBe(false) + releaseRow() + await holder + await move + expect(heldWhileMoverWaits).toEqual([]) + + const reservations = await databases[0]!.query( + `SELECT cell_id, reserved_requests FROM relay_cells + WHERE cell_id IN (?, ?, ?) ORDER BY cell_id ASC`, + [cellA.id, cellB.id, cellC.id] + ) + const reserved = reservations.map((row) => [String(row.cell_id), Number(row.reserved_requests)]) + expect(reserved).toEqual([ + [cellA.id, 0], + // Migration grant plus the moved splice, as in the SQLite origin-scoped + // reservation case: the lock change did not alter accounting. + [cellB.id, 6], + // The sticky grant stays on the source until the migration completes. + [cellC.id, 1] + ]) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/host-close-reason-memory.test.ts b/cloud/apps/relay/src/host-close-reason-memory.test.ts new file mode 100644 index 00000000000..2985e6f1a1d --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.test.ts @@ -0,0 +1,82 @@ +import { ASSIGNMENT_LIMITS, RELAY_HOST_CLOSE_REASON } from '@orca-cloud/relay-contract' +import { describe, expect, it } from 'vitest' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' + +function memoryAt(clock: { now: number }): HostCloseReasonMemory { + return new HostCloseReasonMemory(() => clock.now) +} + +describe('HostCloseReasonMemory', () => { + it('remembers only reasons it knows', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', 'quitting') + memory.record('c', Buffer.alloc(0)) + memory.record('d', undefined) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + expect(memory.read('b')).toBeNull() + expect(memory.read('c')).toBeNull() + expect(memory.read('d')).toBeNull() + }) + + it('accepts the reason as the Buffer a ws close delivers', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + + memory.record('a', Buffer.from(RELAY_HOST_CLOSE_REASON.SIGNED_OUT)) + + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('expires an entry once its host may have been rebalanced away', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += ASSIGNMENT_LIMITS.dormantTtlMs - 1 + expect(memory.read('a')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + clock.now += 1 + expect(memory.read('a')).toBeNull() + expect(memory.size()).toBe(0) + }) + + it('forgets on demand', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + memory.forget('a') + + expect(memory.read('a')).toBeNull() + }) + + it('drops the oldest survivors rather than growing without bound', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + for (let index = 0; index < 50_050; index++) { + memory.record(`host-${index}`, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + } + + expect(memory.size()).toBe(50_000) + expect(memory.read('host-0')).toBeNull() + expect(memory.read('host-50049')).toBe(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('re-recording refreshes recency so a live host is not evicted first', () => { + const clock = { now: 1_000 } + const memory = memoryAt(clock) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('b', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + memory.record('a', RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + expect([...['a', 'b'].map((key) => memory.read(key))]).toEqual([ + RELAY_HOST_CLOSE_REASON.SIGNED_OUT, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ]) + expect(memory.size()).toBe(2) + }) +}) diff --git a/cloud/apps/relay/src/host-close-reason-memory.ts b/cloud/apps/relay/src/host-close-reason-memory.ts new file mode 100644 index 00000000000..ed01aacd666 --- /dev/null +++ b/cloud/apps/relay/src/host-close-reason-memory.ts @@ -0,0 +1,72 @@ +import { + ASSIGNMENT_LIMITS, + relayHostCloseReasonFrom, + type RelayHostCloseReason +} from '@orca-cloud/relay-contract' + +// Retention matches the dormant assignment TTL: past it the host may have been +// rebalanced onto another cell, so this cell is no longer the one a phone asks. +const RETENTION_MS = ASSIGNMENT_LIMITS.dormantTtlMs +// A fleet-wide auth outage signs out every host at once; the cap bounds that +// burst well above any single cell's host count without becoming a leak. +const MAX_ENTRIES = 50_000 + +// Why in-memory and not Postgres: a phone reaches the cell its host's assignment +// row already names, which is the same cell that watched the control socket +// close. Losing this on a cell restart degrades to the pre-existing generic +// verdict, so the failure mode is the old behaviour rather than a wrong one. +export class HostCloseReasonMemory { + private readonly entries = new Map() + + constructor(private readonly now: () => number = Date.now) {} + + // Silently ignores anything that is not a known reason, which is every close + // from a host that predates the field and every abrupt 1006. + record(key: string, reason: unknown): void { + const parsed = relayHostCloseReasonFrom(reason) + if (!parsed) { + return + } + this.entries.delete(key) + this.entries.set(key, { reason: parsed, expiresAt: this.now() + RETENTION_MS }) + this.evict() + } + + forget(key: string): void { + this.entries.delete(key) + } + + read(key: string): RelayHostCloseReason | null { + const entry = this.entries.get(key) + if (!entry) { + return null + } + if (entry.expiresAt <= this.now()) { + this.entries.delete(key) + return null + } + return entry.reason + } + + size(): number { + return this.entries.size + } + + private evict(): void { + const now = this.now() + for (const [key, entry] of this.entries) { + if (entry.expiresAt > now) { + break + } + this.entries.delete(key) + } + // Insertion order is recency order (record deletes before setting), so the + // head is always the oldest survivor. + for (const key of this.entries.keys()) { + if (this.entries.size <= MAX_ENTRIES) { + break + } + this.entries.delete(key) + } + } +} diff --git a/cloud/apps/relay/src/host-session-registry.ts b/cloud/apps/relay/src/host-session-registry.ts index 11b7d1de030..5c53041e7ff 100644 --- a/cloud/apps/relay/src/host-session-registry.ts +++ b/cloud/apps/relay/src/host-session-registry.ts @@ -14,7 +14,8 @@ import { HostHelloSchema, InviteCreateSchema, RELAY_PROTOCOL_LIMITS, - RELAY_CLOSE_CODE + RELAY_CLOSE_CODE, + type RelayHostCloseReason } from '@orca-cloud/relay-contract' import nacl from 'tweetnacl' import type WebSocket from 'ws' @@ -25,6 +26,7 @@ import { RelayCredentialStore, type CredentialReservation } from './credential-store.js' +import { HostCloseReasonMemory } from './host-close-reason-memory.js' import { relayHostLogDigest } from './relay-host-log-digest.js' import type { RelayTokenClaims } from './relay-token-verifier.js' import type { RelayRuntimeObserver } from './relay-observability.js' @@ -130,6 +132,10 @@ const ACTIVATION_QUEUE_WAIT_MS = 30_000 export class HostSessionRegistry { private readonly sessions = new Map() private readonly activationQueues = new Map>() + // Why it outlives `sessions`: the orphan grace deletes the session within 30s, + // but a signed-out desktop never comes back, so the phone that asks minutes + // later would otherwise find nothing to explain its rejection with. + private readonly hostCloseReasons = new HostCloseReasonMemory(() => this.now()) private draining = false constructor( @@ -175,7 +181,8 @@ export class HostSessionRegistry { return } this.observer.recordAuth(true) - const session = this.sessions.get(this.key(reservation.userId, hostId)) + const sessionKey = this.key(reservation.userId, hostId) + const session = this.sessions.get(sessionKey) if ( !session || session.state !== 'active' || @@ -184,7 +191,13 @@ export class HostSessionRegistry { ) { capacityReservation?.release() await this.store.failReservation(reservation) - this.rejectClient(socket, RELAY_CLOSE_CODE.HOST_OFFLINE) + // The only rejection that can name a cause: the host is genuinely absent. + // The attach-deadline 4404 below fires while control is still connected. + this.rejectClient( + socket, + RELAY_CLOSE_CODE.HOST_OFFLINE, + this.hostCloseReasons.read(sessionKey) + ) return } if (session.activeConnIds.size + session.pendingConns.size >= 8) { @@ -793,7 +806,10 @@ export class HostSessionRegistry { regionalDrainTimer: null, regionalDrainExpiresAt: null } - this.sessions.set(this.key(identity.sub, identity.relayHostId), session) + const sessionKey = this.key(identity.sub, identity.relayHostId) + // A host that proved itself again is not signed out, whatever it said last. + this.hostCloseReasons.forget(sessionKey) + this.sessions.set(sessionKey, session) this.wireActiveControl(session) this.sendHelloAck(session) } @@ -813,6 +829,11 @@ export class HostSessionRegistry { }) socket.once('close', (code, reason) => { this.observer.recordControlClose?.(code) + // Guarded on identity: a predecessor retired by a rebind must not stamp a + // cause onto the live session that replaced it. + if (session.socket === socket) { + this.hostCloseReasons.record(this.key(session.identity.sub, session.relayHostId), reason) + } // One line per control close makes reconnect churners attributable by // host digest without exposing the raw relay host id. console.warn( @@ -1187,9 +1208,16 @@ export class HostSessionRegistry { if (session.socket) send(session.socket, 'control-error', { ...(reqId ? { reqId } : {}), code }) } - private rejectClient(socket: WebSocket, code: number): void { + // hostCloseReason rides the WebSocket close reason, never relay-hello: every + // shipped phone parses relay-hello with a strict schema that rejects an + // unknown key, and none of them read the close reason at all. + private rejectClient( + socket: WebSocket, + code: number, + hostCloseReason?: RelayHostCloseReason | null + ): void { send(socket, 'relay-hello', { ok: false, code }) - closeRelayWebSocket(socket, code, 'relay connection rejected') + closeRelayWebSocket(socket, code, hostCloseReason ?? 'relay connection rejected') } private releaseControlActivity(session: HostSession): void { diff --git a/cloud/apps/relay/src/host-signed-out-rejection.test.ts b/cloud/apps/relay/src/host-signed-out-rejection.test.ts new file mode 100644 index 00000000000..0f8542c960f --- /dev/null +++ b/cloud/apps/relay/src/host-signed-out-rejection.test.ts @@ -0,0 +1,206 @@ +import { EventEmitter } from 'node:events' +import { + CONTROL_CONTINUITY_LIMITS, + RELAY_CLOSE_CODE, + RELAY_HOST_CLOSE_REASON +} from '@orca-cloud/relay-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { RelayCredentialStore } from './credential-store.js' +import { HostSessionRegistry } from './host-session-registry.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close', 1006, Buffer.alloc(0)) + }) +} + +const config = { + port: 8080, + publicUrl: 'https://relay-c3.example.com', + cellUrl: 'https://relay-c3.example.com', + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [] +} as unknown as RelayConfig + +const identity = { + sub: 'user-1', + prof: 'profile-1', + org: 'org-1', + relayHostId: 'AbCdEf0123_-xyZ9' +} as unknown as RelayTokenClaims + +const reservation = { + userId: identity.sub, + relayHostId: identity.relayHostId, + credentialKind: 'resume', + relayDeviceId: 'device-1', + leaseExpiresAt: Date.now() + 60_000 +} + +function createRegistry() { + const store = { + resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }), + reserveCredential: vi.fn().mockResolvedValue(reservation), + failReservation: vi.fn().mockResolvedValue(undefined) + } + const assignments = { + activateControl: vi.fn().mockResolvedValue('control:production-gce-c3:1'), + markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined), + resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }), + acquireActivity: vi.fn().mockResolvedValue(undefined), + renewControlActivity: vi.fn().mockResolvedValue(undefined), + releaseActivity: vi.fn().mockResolvedValue(true) + } as unknown as RelayAssignmentStore + const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn(), + recordControlClose: vi.fn(), + recordSpliceClose: vi.fn() + } satisfies RelayRuntimeObserver + const registry = new HostSessionRegistry( + config, + vi.fn(), + store as unknown as RelayCredentialStore, + assignments, + new ProcessQueuedByteBudget(), + observer + ) + const activate = (socket: WebSocket, generation: number): Promise => + ( + registry as unknown as { + activate: ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string + ) => Promise + } + ).activate(socket, identity, null, generation, false, 1, '1.4.173') + return { registry, activate } +} + +async function dialPhone(registry: HostSessionRegistry): Promise { + const phone = new FakeSocket() + await registry.acceptClient(phone as unknown as WebSocket, identity.relayHostId, 'credential') + return phone +} + +// The 4404 hello body is unchanged: every shipped phone parses it with a strict +// schema, so the cause has to ride the close frame instead. +const HOST_OFFLINE_HELLO = JSON.stringify({ + type: 'relay-hello', + ok: false, + code: RELAY_CLOSE_CODE.HOST_OFFLINE +}) + +describe('host sign-out reason on phone rejection', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('names the sign-out to a phone that arrives after the host is gone', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.send).toHaveBeenCalledWith(HOST_OFFLINE_HELLO) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + RELAY_HOST_CLOSE_REASON.SIGNED_OUT + ) + }) + + it('says nothing when the host died without naming a cause', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('ignores a close reason the host invented', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + control.close(1000, 'signed-out-ish') + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + it('forgets the sign-out once the host proves itself again', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + control.close(1000, RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const reconnected = new FakeSocket() + await activate(reconnected as unknown as WebSocket, 2) + // Drop it abruptly, as a network death would, so only the stale memory + // could still name a cause. + reconnected.terminate() + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs + 1) + + const phone = await dialPhone(registry) + expect(phone.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.HOST_OFFLINE, + 'relay connection rejected' + ) + }) + + // A live host is present: the 4404 there is an attach deadline, not absence. + it('never names a cause while the host control is connected', async () => { + const { registry, activate } = createRegistry() + const control = new FakeSocket() + await activate(control as unknown as WebSocket, 1) + + const phone = await dialPhone(registry) + expect(phone.close).not.toHaveBeenCalled() + expect(control.send).toHaveBeenCalledWith(expect.stringContaining('"type":"conn-open"')) + }) +}) diff --git a/cloud/dev/fixtures/terraform-root-partition/families.json b/cloud/dev/fixtures/terraform-root-partition/families.json index 9664c1eb299..3c57b2ea7cb 100644 --- a/cloud/dev/fixtures/terraform-root-partition/families.json +++ b/cloud/dev/fixtures/terraform-root-partition/families.json @@ -133,7 +133,10 @@ "google_logging_metric.relay_snapshot", "google_monitoring_alert_policy.relay_assignment_5xx", "google_monitoring_alert_policy.relay_assignment_edge_429", + "google_monitoring_alert_policy.relay_cloud_nat_port_drops", "google_monitoring_alert_policy.relay_cloud_sql_backends", + "google_monitoring_alert_policy.relay_cloud_sql_checkpoint_loop", + "google_monitoring_alert_policy.relay_cloud_sql_disk", "google_monitoring_alert_policy.relay_custom", "google_monitoring_alert_policy.relay_gce_connection_headroom", "google_monitoring_alert_policy.relay_postgres_retry_exhausted", diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 3e5fc04836e..870c95dd413 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -99,8 +99,8 @@ durably marked consumed before mutation and cannot authorize another run. | Cloud SQL deadlocks | over 0 | | Relay pool waiters | over 800 | | Relay pool wait | over 2,500 ms | -| PostgreSQL retries in five minutes | over 300 | -| Exhausted PostgreSQL retries | over 0 | +| PostgreSQL retries in five minutes | over 2,000 | +| Exhausted PostgreSQL retries in five minutes | over 300 | | Director instances | outside 5–6 | | Director CPU or memory | over 80% | | Director concurrency | over 64 | @@ -132,15 +132,41 @@ heartbeats, and matching live admission. 10 minutes over the old bar of 160 — enough to freeze roughly one in ten 15-minute pre-drain gates on baseline noise. 250 clears measured healthy peaks and still fires well before the verified 400-connection ceiling; - pool waiters, pool wait latency, and exhausted retries keep their strict - thresholds. + pool waiters and pool wait latency keep their strict thresholds. - Recalibrated the PostgreSQL-retry freeze from 20 to 300 per five minutes (2026-08-26). Basis, measured from `jsonPayload.event="orca_relay_postgres_transaction_retry"` in production logs: healthy-day bursts reach 234/5min with zero exhausted retries and 26% of five-minute windows over 20, while the 2026-08-23 lock-contention - incident ran roughly 2,200–3,000/5min. Exhausted retries stay at zero - tolerance. + incident ran roughly 2,200–3,000/5min by raw log-line count (the gate's + own `orca_relay_postgres_retries` metric read 1,510 for that window; see the + 2026-09-04 entry). +- Recalibrated the PostgreSQL-retry freeze from 300 to 2,000 per five minutes + (2026-09-04). Basis: the global `relay_cells FOR UPDATE` lock made + successful retries a steady-state rate. Measured fleet-wide (director + + cells, summed per five minutes from the `orca_relay_postgres_retries` + log metric) over 2026-09-03T05Z..2026-09-04T05Z: p50 430 / p90 924 / + p99 1,320 / max 1,504; 55% of windows over 300; only 22% of 15-minute gates + clean at 300 versus 100% at 2,000. Three read-only dry-runs on 2026-09-04 + froze on this bar (runs 33836470590, 33838698725) or on a genuine six-cell + crash storm (33837160275), blocking the same-cap roll that carries #18521 + and the `beginProof` crash guard to the 23 cells. The 2026-08-23 incident + on this metric peaked at 1,510 then 646, so retries alone no longer + separate it from today's baseline; the exhausted-retry bar (incident peak + 467 vs bar 300), director concurrency, and the pool bars carry that role. + Re-tighten after the fleet is on the 500 ms lock wait. +- Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five + minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory + lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters + now fail fast (one `/v1/assign` 503 with `Retry-After`) instead of + succeeding slowly, and `orca_relay_postgres_transaction_exhausted` became + a steady contention rate. Measured fleet-wide per five minutes over + 2026-09-03T03Z..2026-09-04T02Z: 236 of 236 windows non-zero; quiet hours + p50 2 / max 36; pre-#18521 daytime p50 10 / p90 25 / max 87; post-#18521 + p50 42 / p90 147 / max 220; the 2026-08-23 incident peaked at 467. Every + pre-drain dry-run since the director deploy froze at minute one on this + bar, which blocked the cell roll that carries the same fix to the 23 GCE + cells. `/v1/assign` 503 share was unchanged by #18521 (13.9% vs 12.3%). - Added a fail-closed state machine with latched threshold freezes, generation-scoped checkpoint boundaries, continuity-reset evidence, cadence accounting, restart-gap recovery, and the 15-minute pre-drain gate. diff --git a/cloud/infra/terraform/relay-gce-foundation.tf b/cloud/infra/terraform/relay-gce-foundation.tf index aab3b4579eb..a8d64b3fcea 100644 --- a/cloud/infra/terraform/relay-gce-foundation.tf +++ b/cloud/infra/terraform/relay-gce-foundation.tf @@ -42,6 +42,12 @@ resource "google_compute_router_nat" "relay_gce" { router = google_compute_router.relay_gce[0].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce[0].id @@ -85,6 +91,12 @@ resource "google_compute_router_nat" "relay_gce_additional" { router = google_compute_router.relay_gce_additional[each.key].name nat_ip_allocate_option = "AUTO_ONLY" source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + # Cells reach Cloud SQL's public IP through this NAT. The static default of 64 ports per VM + # filled during the 2026-09-04 incident and every cell's proxy dial timed out at once. + enable_dynamic_port_allocation = true + enable_endpoint_independent_mapping = false + min_ports_per_vm = 64 + max_ports_per_vm = 4096 subnetwork { name = google_compute_subnetwork.relay_gce_additional[each.key].id diff --git a/cloud/infra/terraform/relay-observability.tf b/cloud/infra/terraform/relay-observability.tf index a8fa4276eb2..5467555ccba 100644 --- a/cloud/infra/terraform/relay-observability.tf +++ b/cloud/infra/terraform/relay-observability.tf @@ -37,6 +37,10 @@ locals { description = "Relay PostgreSQL transactions that exhausted bounded retry." filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR resource.type=\"gce_instance\") AND jsonPayload.event=\"orca_relay_postgres_transaction_exhausted\"" } + cloud_sql_wal_checkpoint = { + description = "Cloud SQL checkpoints triggered by WAL volume instead of the timed schedule; a sustained run is the fsync loop that stalled every relay process at once on 2026-09-04." + filter = "resource.type=\"cloudsql_database\" AND resource.labels.database_id=\"${var.project_id}:${local.relay_database_instance_name}\" AND textPayload:\"checkpoint starting: wal\"" + } } relay_runtime_metrics = { @@ -523,3 +527,109 @@ resource "google_monitoring_alert_policy" "relay_cloud_sql_backends" { mime_type = "text/markdown" } } + +resource "google_monitoring_alert_policy" "relay_cloud_sql_checkpoint_loop" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL checkpoint loop" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "WAL-triggered checkpoints above 3 in 5 minutes" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND metric.type=\"logging.googleapis.com/user/orca_relay_cloud_sql_wal_checkpoint\"" + comparison = "COMPARISON_GT" + threshold_value = 3 + duration = "300s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Healthy operation is one timed checkpoint every 5 minutes. Repeated `checkpoint starting: wal` lines mean WAL is outrunning `max_wal_size` and every checkpoint fsync stalls all relay SQL for seconds. Check `checkpoint complete` sync= times and disk write throughput against the PD-SSD ceiling; the fix is disk size and `max_wal_size` in the Terraform root that owns the instance (orca-cloud `infra/terraform-foundation`)." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +resource "google_monitoring_alert_policy" "relay_cloud_sql_disk" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL disk utilization" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cloud SQL disk above 70%" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND resource.label.\"database_id\"=\"${var.project_id}:${local.relay_database_instance_name}\" AND metric.type=\"cloudsql.googleapis.com/database/disk/utilization\"" + comparison = "COMPARISON_GT" + threshold_value = 0.7 + duration = "600s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_MAX" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "The shared auth/relay Cloud SQL disk is filling. `refresh_tokens` is the largest table and grows without pruning; grow the disk (IOPS scale with size) before it reaches the WAL checkpoint loop, and prune revoked token rows." + mime_type = "text/markdown" + } +} + +resource "google_monitoring_alert_policy" "relay_cloud_nat_port_drops" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + display_name = "Orca Relay: Cloud NAT port exhaustion" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "NAT packets dropped for lack of ports" + + condition_threshold { + filter = "resource.type=\"nat_gateway\" AND resource.label.\"gateway_name\"=monitoring.regex.full_match(\"${local.relay_gce_name}(-.*)?\") AND metric.type=\"router.googleapis.com/nat/dropped_sent_packets_count\" AND metric.label.\"reason\"=\"OUT_OF_RESOURCES\"" + comparison = "COMPARISON_GT" + threshold_value = 0 + duration = "120s" + + aggregations { + alignment_period = "60s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"gateway_name\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Relay cells reach Cloud SQL's public IP through this NAT. Port exhaustion makes every cell's Cloud SQL Auth Proxy dial time out at once, which reads as a fleet-wide SQL stall with a healthy database. Check `nat/port_usage` per VM and raise `max_ports_per_vm` in `relay-gce-foundation.tf`, or move the database to a private IP." + mime_type = "text/markdown" + } +} diff --git a/cloud/packages/relay-contract/src/host-close-reason.ts b/cloud/packages/relay-contract/src/host-close-reason.ts new file mode 100644 index 00000000000..3a5abde3f00 --- /dev/null +++ b/cloud/packages/relay-contract/src/host-close-reason.ts @@ -0,0 +1,18 @@ +// Mirror of src/shared/relay-host-close-reason.ts in the Orca app repo half. +// A host control socket may close with one of these as its WebSocket close +// reason; the cell records it so a later phone rejection can name the cause. +// Anything else (including the empty reason of an abrupt 1006) means "unknown", +// which is what every peer that predates this file sends. +export const RELAY_HOST_CLOSE_REASON = { + SIGNED_OUT: 'signed-out' +} as const + +export type RelayHostCloseReason = + (typeof RELAY_HOST_CLOSE_REASON)[keyof typeof RELAY_HOST_CLOSE_REASON] + +const REASONS: readonly string[] = Object.values(RELAY_HOST_CLOSE_REASON) + +export function relayHostCloseReasonFrom(value: unknown): RelayHostCloseReason | null { + const text = typeof value === 'string' ? value : (value?.toString() ?? '') + return REASONS.includes(text) ? (text as RelayHostCloseReason) : null +} diff --git a/cloud/packages/relay-contract/src/index.ts b/cloud/packages/relay-contract/src/index.ts index 2a7d7d0feda..aab3b53b5f3 100644 --- a/cloud/packages/relay-contract/src/index.ts +++ b/cloud/packages/relay-contract/src/index.ts @@ -5,6 +5,7 @@ export * from './control-messages.js' export * from './control-continuity.js' export * from './credential-messages.js' export * from './director-messages.js' +export * from './host-close-reason.js' export * from './host-proof-transcript.js' export * from './persistence-invariants.js' export * from './protocol-limits.js' diff --git a/config/electron-builder.config.cjs b/config/electron-builder.config.cjs index 06d41bad344..ebf4d275678 100644 --- a/config/electron-builder.config.cjs +++ b/config/electron-builder.config.cjs @@ -90,7 +90,19 @@ const bundledPluginResources = { // from package directories where pnpm's symlink farm is absent. Copy the exact // runtime dependency closure to Resources/node_modules so bare require() calls // do not fall through to a developer checkout's node_modules. -const commonExtraResources = [relayExtraResource, bundledPluginResources, skillFreshnessResources] +// Why the single file rather than the package root: app.asar carries no node_modules, so main's +// lazy require in deferred-emoji-shortcode-dataset.ts resolves only out of Resources/node_modules, +// but emojibase-data is 49 MB of locale datasets and worktree naming reads exactly this 166 KB file. +const emojiShortcodeDatasetResource = { + from: 'node_modules/emojibase-data/en/shortcodes/emojibase.json', + to: 'node_modules/emojibase-data/en/shortcodes/emojibase.json' +} +const commonExtraResources = [ + relayExtraResource, + bundledPluginResources, + skillFreshnessResources, + emojiShortcodeDatasetResource +] // Why: native speech addons must be real files outside app.asar; copy only the // package matching the artifact target instead of every optional variant. const macSpeechNativeResource = { diff --git a/config/patches/node-pty@1.1.0.patch b/config/patches/node-pty@1.1.0.patch index ff474f7d95e..8f5045b932a 100644 --- a/config/patches/node-pty@1.1.0.patch +++ b/config/patches/node-pty@1.1.0.patch @@ -603,7 +603,7 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..2ae787c5bd4f3eba470584dc658a01a5 } #endif diff --git a/src/win/conpty.cc b/src/win/conpty.cc -index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a6a4082ce 100644 +index 7b286d3d644c26141df516929703aa6e129df4b2..4aed260dd68e6a171dcfd349e9a7c5c97209248e 100644 --- a/src/win/conpty.cc +++ b/src/win/conpty.cc @@ -18,6 +18,7 @@ @@ -614,7 +614,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a #include #include #include -@@ -44,12 +45,29 @@ struct pty_baton { +@@ -44,12 +45,39 @@ struct pty_baton { HANDLE hOut; HPCON hpc; @@ -630,22 +630,32 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + // refused to create or assign one (an outer job without breakaway rights), + // in which case callers fall back to their pre-job behaviour. + HANDLE hJob = nullptr; ++ ++ // Orca: teardown needs BOTH the shell's death and an explicit kill() before ++ // the baton can be freed, so each side records that it has run. Whichever ++ // arrives second frees it. Freeing on the shell's death alone -- what this ++ // file did before -- destroyed the only record of `hpc` while ++ // ClosePseudoConsole was still owed, which is why a self-exiting shell ++ // leaked its pseudoconsole and the console host it reaps (#18601 / F24). ++ bool shellExited = false; ++ bool consoleClosed = false; pty_baton(int _id, HANDLE _hIn, HANDLE _hOut, HPCON _hpc) : id(_id), hIn(_hIn), hOut(_hOut), hpc(_hpc) {}; }; static std::vector> ptyHandles; -+// Orca: guards the job accessors below against the exit watcher thread. It does -+// NOT make the whole table safe -- PtyResize/PtyClear/PtyKill read it unlocked, -+// as they always have -- but it closes the window this patch opened, where the -+// watcher can close hShell/hJob and free the baton between a lookup and its use. ++// Orca: guards the job accessors below, and PtyKill, against the exit watcher ++// thread. It does NOT make the whole table safe -- PtyResize and PtyClear still ++// read it unlocked, as they always have -- but it closes the window this patch ++// opened, where the watcher can close hShell/hJob and free the baton between a ++// lookup and its use. +// Handle VALUES are recycled aggressively, so an unguarded read could pass the +// shell-pid check against an unrelated process and terminate the wrong job. +static std::mutex ptyJobMutex; static volatile LONG ptyCounter; static pty_baton* get_pty_baton(int id) { -@@ -102,8 +120,27 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) { +@@ -102,8 +130,31 @@ void SetupExitCallback(Napi::Env env, Napi::Function cb, pty_baton* baton) { // Get process exit code. GetExitCodeProcess(baton->hShell, (LPDWORD)(&exit_event->exit_code)); // Clean up handles @@ -665,9 +675,13 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + // Why inside the lock: erasing frees the baton the job accessors hold a + // pointer to. Note remove_pty_baton must not be an assert() argument -- + // NDEBUG would compile the call away and leak every baton. -+ const bool removed = remove_pty_baton(baton->id); -+ assert(removed); -+ (void)removed; ++ baton->shellExited = true; ++ if (baton->consoleClosed) { ++ const bool removed = remove_pty_baton(baton->id); ++ assert(removed); ++ (void)removed; ++ } ++ // Else PtyKill has not run yet and still owns hpc. It frees the baton. + } + // Why the lock ends here: BlockingCall below waits on the JS thread, and the + // JS thread can be waiting on ptyJobMutex inside PtyTerminateJob. Holding @@ -675,7 +689,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a auto status = tsfn.BlockingCall(exit_event, callback); // In main thread switch (status) { -@@ -409,6 +446,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -409,6 +460,15 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { throw errorWithCode(info, "UpdateProcThreadAttribute failed"); } @@ -691,7 +705,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a PROCESS_INFORMATION piClient{}; fSuccess = !!CreateProcessW( nullptr, -@@ -416,7 +462,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -416,7 +476,10 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { nullptr, // lpProcessAttributes nullptr, // lpThreadAttributes false, // bInheritHandles VERY IMPORTANT that this is false @@ -703,7 +717,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a envArg, // lpEnvironment mutableCwd.get(), // lpCurrentDirectory &siEx.StartupInfo, // lpStartupInfo -@@ -426,8 +475,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -426,8 +489,47 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { throw errorWithCode(info, "Cannot create process"); } @@ -753,7 +767,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a if (useConptyDll && fLoadedDll) { PFNRELEASEPSEUDOCONSOLE const pfnReleasePseudoConsole = (PFNRELEASEPSEUDOCONSOLE)GetProcAddress( -@@ -440,6 +528,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { +@@ -440,6 +542,8 @@ static Napi::Value PtyConnect(const Napi::CallbackInfo& info) { // Update handle handle->hShell = piClient.hProcess; @@ -762,7 +776,91 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a // Close the thread handle to avoid resource leak CloseHandle(piClient.hThread); -@@ -567,6 +657,143 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) { +@@ -544,29 +648,215 @@ static Napi::Value PtyKill(const Napi::CallbackInfo& info) { + int id = info[0].As().Int32Value(); + const bool useConptyDll = info[1].As().Value(); + +- const pty_baton* handle = get_pty_baton(id); ++ // Orca: resolve the DLL BEFORE touching any baton state, for the same reason ++ // PtyConnect does it before creating anything. LoadConptyDll throws when ++ // conpty.dll is missing, and a throw after consoleClosed was set would strand ++ // the pseudoconsole permanently: the retry would find the work already ++ // claimed and do nothing. Only the useConptyDll path can throw here; the ++ // other returns kernel32. ++ HANDLE hLibrary = LoadConptyDll(info, useConptyDll); ++ PFNCLOSEPSEUDOCONSOLE pfnClosePseudoConsole = nullptr; ++ if (hLibrary != nullptr) { ++ pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress( ++ (HMODULE)hLibrary, ++ useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole"); ++ } + +- if (handle != nullptr) { +- HANDLE hLibrary = LoadConptyDll(info, useConptyDll); +- bool fLoadedDll = hLibrary != nullptr; +- if (fLoadedDll) +- { +- PFNCLOSEPSEUDOCONSOLE const pfnClosePseudoConsole = (PFNCLOSEPSEUDOCONSOLE)GetProcAddress( +- (HMODULE)hLibrary, +- useConptyDll ? "ConptyClosePseudoConsole" : "ClosePseudoConsole"); +- if (pfnClosePseudoConsole) +- { +- pfnClosePseudoConsole(handle->hpc); ++ // Orca: the baton now outlives the shell, so this runs on a self-exited pty ++ // too -- that is the whole point. Take what we need under the lock: the ++ // watcher thread nulls hShell the moment the shell dies, and TerminateProcess ++ // on a handle it just closed is an invalid-handle operation. Duplicating ++ // rather than reordering keeps upstream's close-then-terminate sequence. ++ HPCON hpc = nullptr; ++ HANDLE hShellDup = nullptr; ++ bool owed = false; ++ { ++ std::lock_guard guard(ptyJobMutex); ++ pty_baton* handle = get_pty_baton(id); ++ // Why the consoleClosed check: a second kill() would otherwise close the ++ // same pseudoconsole twice. Upstream relied on the baton being gone. ++ if (handle != nullptr && !handle->consoleClosed) { ++ hpc = handle->hpc; ++ owed = true; ++ handle->consoleClosed = true; ++ // Null hShell means a self-exited pty, where there is nothing to kill. ++ if (useConptyDll && handle->hShell != nullptr) { ++ if (!DuplicateHandle(GetCurrentProcess(), handle->hShell, GetCurrentProcess(), ++ &hShellDup, 0, FALSE, DUPLICATE_SAME_ACCESS)) { ++ // Why terminate here instead of skipping: a failed duplication leaves ++ // hShellDup null, which is indistinguishable from the self-exit case, ++ // and skipping would leave the shell RUNNING after its pane closed -- ++ // a worse outcome than the leak this all exists to fix. hShell is ++ // valid under this lock and TerminateProcess does not block, so the ++ // only cost is that this rare path kills before the console closes. ++ hShellDup = nullptr; ++ TerminateProcess(handle->hShell, 1); ++ } ++ } ++ if (handle->shellExited) { ++ const bool removed = remove_pty_baton(id); ++ assert(removed); ++ (void)removed; + } ++ // Else the shell is still running and the watcher frees the baton. + } +- if (useConptyDll) { +- TerminateProcess(handle->hShell, 1); ++ } ++ ++ // Why outside the lock: ClosePseudoConsole blocks until the conout side has ++ // drained, and the watcher must be able to take the lock while it does. ++ if (owed) { ++ if (pfnClosePseudoConsole) ++ { ++ pfnClosePseudoConsole(hpc); ++ } ++ if (hShellDup != nullptr) { ++ TerminateProcess(hShellDup, 1); ++ CloseHandle(hShellDup); + } + } + return env.Undefined(); } @@ -808,9 +906,11 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a + * Orca: the pids still alive in this pty's tree, straight from the kernel. + * + * Descendant liveness for a tree that is still tracked, including children that -+ * detached from the console. Once the shell exits the baton is gone, so this -+ * returns null rather than an empty list -- null means "no answer", never -+ * "they died". Also returns null when no job was assigned. ++ * detached from the console. Once the shell exits the watcher nulls hJob, which ++ * ownsShell rejects, so this returns null rather than an empty list -- null ++ * means "no answer", never "they died". (The baton itself now outlives the ++ * shell, until kill() runs; hJob is what makes the answer null.) Also returns ++ * null when no job was assigned. + * + * Does not include the ConPTY console host: CreatePseudoConsole spawns it + * before this job exists, so it is not a member and ClosePseudoConsole is what @@ -906,7 +1006,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a /** * Init */ -@@ -577,6 +804,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) { +@@ -577,6 +867,9 @@ Napi::Object init(Napi::Env env, Napi::Object exports) { exports.Set("resize", Napi::Function::New(env, PtyResize)); exports.Set("clear", Napi::Function::New(env, PtyClear)); exports.Set("kill", Napi::Function::New(env, PtyKill)); @@ -917,7 +1017,7 @@ index 7b286d3d644c26141df516929703aa6e129df4b2..ec6bf3932c65b89c013ff133dc6bf46a }; diff --git a/lib/windowsPtyAgent.js b/lib/windowsPtyAgent.js -index a358ffb..fb3a96f 100644 +index a358ffb177357e177661033c1b092f9c9d0e5f5a..26c2a4c58799ce649f5113131e4c52f7ed2d87ad 100644 --- a/lib/windowsPtyAgent.js +++ b/lib/windowsPtyAgent.js @@ -136,6 +136,9 @@ var WindowsPtyAgent = /** @class */ (function () { @@ -930,6 +1030,20 @@ index a358ffb..fb3a96f 100644 this._outSocket.readable = false; this._getConsoleProcessList().then(function (consoleProcessList) { consoleProcessList.forEach(function (pid) { +@@ -154,9 +157,10 @@ var WindowsPtyAgent = /** @class */ (function () { + // Close the input write handle to signal the end of session. + this._inSocket.destroy(); + this._ptyNative.kill(this._pty, this._useConptyDll); +- this._outSocket.on('data', function () { +- _this._conoutSocketWorker.dispose(); +- }); ++ // Orca: dispose unconditionally, as the non-DLL branch above does. ++ // Waiting for another 'data' event leaks the conout worker on every ++ // self-exiting shell, because no more data ever arrives (F24). ++ this._conoutSocketWorker.dispose(); + } + } + else { diff --git a/lib/windowsTerminal.js b/lib/windowsTerminal.js index 3c38f89..e20b3e6 100644 --- a/lib/windowsTerminal.js @@ -1015,7 +1129,7 @@ index 3c38f89..e20b3e6 100644 \ No newline at end of file +//# sourceMappingURL=windowsTerminal.js.map diff --git a/src/windowsPtyAgent.ts b/src/windowsPtyAgent.ts -index d705444..ce611b8 100644 +index d7054449516f0c9a62af351c2caa17331206d530..0c28a32e2e1db2b3f208ddde8443cd4e67bb1ad6 100644 --- a/src/windowsPtyAgent.ts +++ b/src/windowsPtyAgent.ts @@ -143,6 +143,9 @@ export class WindowsPtyAgent { @@ -1028,6 +1142,20 @@ index d705444..ce611b8 100644 this._outSocket.readable = false; this._getConsoleProcessList().then(consoleProcessList => { consoleProcessList.forEach((pid: number) => { +@@ -159,9 +162,10 @@ export class WindowsPtyAgent { + // Close the input write handle to signal the end of session. + this._inSocket.destroy(); + (this._ptyNative as IConptyNative).kill(this._pty, this._useConptyDll); +- this._outSocket.on('data', () => { +- this._conoutSocketWorker.dispose(); +- }); ++ // Orca: dispose unconditionally, as the non-DLL branch above does. ++ // Waiting for another 'data' event leaks the conout worker on every ++ // self-exiting shell, because no more data ever arrives (F24). ++ this._conoutSocketWorker.dispose(); + } + } else { + // Because pty.kill closes the handle, it will kill most processes by itself. diff --git a/src/windowsTerminal.ts b/src/windowsTerminal.ts index 13f6c6d..eda63c8 100644 --- a/src/windowsTerminal.ts diff --git a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs new file mode 100644 index 00000000000..dd26784ee46 --- /dev/null +++ b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs @@ -0,0 +1,158 @@ +const { createHash } = require('node:crypto') +const { readFileSync, renameSync, rmSync, writeFileSync } = require('node:fs') +const { join, resolve } = require('node:path') + +/** + * Release the ConPTY teardown handles a relay's npm-installed node-pty never releases. + * + * Two files, and the ORDER of one of the edits is the whole fix. + * + * `windowsPtyAgent.js` -- `kill()` flips `readable` on both sockets and destroys neither. + * `_cleanUpProcess` destroys `_outSocket`, so the conout handle comes back; nothing ever destroys + * `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`. + * Every terminal leaks one File handle for the life of the host process. + * + * The obvious fix -- and the one the desktop patch ships -- releases it at the TOP of the branch, + * before `_getConsoleProcessList()` forks and before the native kill. That is measurably worse than + * leaving the leak alone: teardown aborts partway, the forked console-list agent is never reaped, + * and both pipe handles stay alive instead of one. This asset releases it at the END of the branch + * instead, after the fork and the kill have already happened. + * + * Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type + * (identical numbers standalone and through a real relay): + * + * published node-pty File +1/terminal, Process flat + * desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE + * released last (here) File flat, Process flat + * + * `windowsTerminal.js` carries the desktop's error-listener hunks verbatim. The conin listener is + * what keeps a pipe error retiring one terminal instead of the host -- its own comment names the + * failure mode: "Without a listener, Node promotes errors such as write EAGAIN to uncaughtException". + * It is not what fixes the leak (adding it changed nothing on its own), but it is the guard that + * makes destroying conin safe at all. + * + * Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm + * patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there. + * + * DELIBERATE DIVERGENCE FROM THE DESKTOP: the desktop patch has the early placement and therefore + * the +2 File / +1 Process regression, measured against its exact installed tree. Correcting it + * there is a separate change with its own verification, so the two trees differ on this one hunk on + * purpose, and the test pins that so a future "sync the patches" does not copy the bug back. + * + * NOT ADDRESSED, AND A SEPARATE DEFECT THAT IS STILL OPEN: a terminal that exits on its own is + * still torn down through `kill()` -- both hosts call `destroy()` on natural exit and + * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, and the + * ordering this patch relies on does not hold. Measured over 20 self-exit cycles with that + * `destroy()` issued: published +3 File/+1 Process per terminal, desktop-patched +2/+1, this tree + * +2/+1. So this patch does not close it and the desktop patch does not either. It is reachable + * for every Windows user, local and relay, on every terminal closed by typing `exit`. + */ + +const EXPECTED_NODE_PTY_VERSION = '1.1.0' + +/** Each entry is one published file, its patched form, and the edits between them. */ +const PATCH_TARGETS = [ + { + relativePath: ['lib', 'windowsPtyAgent.js'], + originalSha256: '8636d16b38266112204061a22b135734177c242837982fd3a4055be726efa64a', + patchedSha256: '1e23ef480569e73706e3ab4f5482c7e553c76f51414ae8e7b0bdcc2fd75f7280', + replacements: [ + [ + ' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n', + ' this._ptyNative.kill(this._pty, this._useConptyDll);\n this._conoutSocketWorker.dispose();\n // Orca: released AFTER the console-list fork and the native kill, not before them.\n // Destroying conin first aborts teardown partway -- measured on a Windows SSH relay\n // as +2 File and +1 Process handles per terminal, against +1 File unpatched.\n this._inSocket.destroy();\n' + ] + ] + }, + { + relativePath: ['lib', 'windowsTerminal.js'], + originalSha256: 'c3a65716f53fed0135a8a633373d5f9c2ab092544d651f27ef0a67096dd3bcd9', + patchedSha256: '8247ecd69be8b18257050fb026b290024612c5ffc6d492ff1d46f81e613be2cf', + replacements: [ + [ + ' _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;', + " _this._agent = new windowsPtyAgent_1.WindowsPtyAgent(file, args, parsedEnv, cwd, _this._cols, _this._rows, false, opt.useConpty, opt.useConptyDll, opt.conptyInheritCursor);\n _this._socket = _this._agent.outSocket;\n // Attach before readiness so a broken ConPTY output pipe cannot be unhandled.\n _this._socket.on('error', function (err) {\n var code = err && err.code;\n // PTY output can report EPIPE before `_close()` wins the race.\n _this._close();\n if (code === 'EPIPE' || code === 'ERR_STREAM_PUSH_AFTER_EOF' || code === 'ERR_STREAM_DESTROYED') {\n return;\n }\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (typeof code === 'string') {\n if (~code.indexOf('errno 5') || ~code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Not available until `ready` event emitted.\n _this._pid = _this._agent.innerPid;" + ], + [ + " }\n });\n // Shutdown if `error` event is emitted.\n _this._socket.on('error', function (err) {\n // Close terminal session.\n _this._close();\n // EIO, happens when someone closes our child process: the only process\n // in the terminal.\n // node < 0.6.14: errno 5\n // node >= 0.6.14: read EIO\n if (err.code) {\n if (~err.code.indexOf('errno 5') || ~err.code.indexOf('EIO'))\n return;\n }\n // Throw anything else.\n if (_this.listeners('error').length < 2) {\n throw err;\n }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {", + " }\n });\n // Cleanup after the socket is closed.\n _this._socket.on('close', function () {" + ], + [ + ' _this._readable = true;\n _this._writable = true;\n _this._forwardEvents();\n return _this;', + " _this._readable = true;\n _this._writable = true;\n // A ConPTY input-pipe error must retire only this terminal. Without a listener, Node promotes\n // errors such as write EAGAIN to uncaughtException and kills every PTY in the daemon.\n _this._agent.inSocket.on('error', function () {\n if (!_this._writable) {\n return;\n }\n _this._close();\n try {\n _this._agent.kill();\n }\n catch (_a) {\n // The failing pipe may have raced process exit; the terminal is already unwritable.\n }\n });\n _this._forwardEvents();\n return _this;" + ], + [ + 'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map', + 'exports.WindowsTerminal = WindowsTerminal;\n//# sourceMappingURL=windowsTerminal.js.map\n' + ] + ] + } +] + +function inspectTarget(relayDir, target) { + const nodePtyDir = resolve(relayDir, 'node_modules', 'node-pty') + const packageJson = JSON.parse(readFileSync(join(nodePtyDir, 'package.json'), 'utf8')) + if (packageJson.version !== EXPECTED_NODE_PTY_VERSION) { + throw new Error( + `Refusing to patch node-pty ${packageJson.version}; expected ${EXPECTED_NODE_PTY_VERSION}` + ) + } + const filePath = join(nodePtyDir, ...target.relativePath) + return { filePath, source: readFileSync(filePath, 'utf8') } +} + +function assertPatchedNodePtyWindowsTeardown(relayDir = process.cwd()) { + for (const target of PATCH_TARGETS) { + const inspected = inspectTarget(relayDir, target) + if (sourceSha256(inspected.source) !== target.patchedSha256) { + throw new Error( + `node-pty ConPTY teardown release is not installed in ${target.relativePath.join('/')}` + ) + } + } +} + +function patchNodePtyWindowsTeardown(relayDir = process.cwd()) { + for (const target of PATCH_TARGETS) { + const inspected = inspectTarget(relayDir, target) + const sourceHash = sourceSha256(inspected.source) + if (sourceHash === target.patchedSha256) { + continue + } + if (sourceHash !== target.originalSha256) { + throw new Error( + `Refusing to patch unexpected node-pty source in ${target.relativePath.join('/')}` + ) + } + let patchedSource = inspected.source + for (const [from, to] of target.replacements) { + // Why the count check: an anchor that matched twice would patch the wrong site silently, and + // the hash below would then reject a tree this script had already rewritten. + if (patchedSource.split(from).length - 1 !== 1) { + throw new Error(`Refusing to patch ${target.relativePath.join('/')}; anchor is not unique`) + } + patchedSource = patchedSource.replace(from, to) + } + const temporaryPath = `${inspected.filePath}.orca-patch-${process.pid}` + // Why: a terminated remote install must leave either known source version recoverable on reconnect. + try { + writeFileSync(temporaryPath, patchedSource) + renameSync(temporaryPath, inspected.filePath) + } finally { + rmSync(temporaryPath, { force: true }) + } + } + assertPatchedNodePtyWindowsTeardown(relayDir) +} + +function sourceSha256(source) { + return createHash('sha256').update(source).digest('hex') +} + +if (require.main === module) { + patchNodePtyWindowsTeardown() +} + +module.exports = { + assertPatchedNodePtyWindowsTeardown, + patchNodePtyWindowsTeardown +} diff --git a/config/scripts/build-relay.mjs b/config/scripts/build-relay.mjs index 289c7a957bd..4d408712f97 100644 --- a/config/scripts/build-relay.mjs +++ b/config/scripts/build-relay.mjs @@ -57,6 +57,13 @@ const NODE_PTY_CONSOLE_LIST_PATCH_SOURCE = join( 'relay-assets', NODE_PTY_CONSOLE_LIST_PATCH_FILENAME ) +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME = 'node-pty-1.1.0-windows-pty-teardown-patch.cjs' +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_SOURCE = join( + ROOT, + 'config', + 'relay-assets', + NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME +) const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' const NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE = join( ROOT, @@ -132,6 +139,10 @@ for (const platform of RELAY_BUILD_PLATFORMS) { NODE_PTY_CONSOLE_LIST_PATCH_SOURCE, join(outDir, NODE_PTY_CONSOLE_LIST_PATCH_FILENAME) ) + copyFileSync( + NODE_PTY_WINDOWS_TEARDOWN_PATCH_SOURCE, + join(outDir, NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME) + ) } copyFileSync( NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE, diff --git a/config/scripts/electron-builder-runtime-resources.test.mjs b/config/scripts/electron-builder-runtime-resources.test.mjs index d2407776fa7..453d5702cb0 100644 --- a/config/scripts/electron-builder-runtime-resources.test.mjs +++ b/config/scripts/electron-builder-runtime-resources.test.mjs @@ -1,14 +1,18 @@ +import { readFileSync, readdirSync } from 'node:fs' import { cp, mkdir, mkdtemp, readFile, readdir, rm, stat, writeFile } from 'node:fs/promises' import { createRequire } from 'node:module' import { tmpdir } from 'node:os' -import { join } from 'node:path' +import { dirname, join, relative, resolve } from 'node:path' import { describe, expect, it } from 'vitest' const require = createRequire(import.meta.url) +const projectRoot = resolve(import.meta.dirname, '..', '..') const electronBuilderConfig = require('../electron-builder.config.cjs') const { createPackagedRuntimeNodeModuleResources, findAsarEntry, + isPackagedExternalSpecifier, + packageNameFromSpecifier, prunePackagedNodePty, prunePackagedParcelWatcher, prunePackagedSherpaOnnx, @@ -306,3 +310,91 @@ describe('packaged runtime resources', () => { } ) }) + +// Why source-anchored: the bundler renames a createRequire()'d require, so +// verifyPackagedMainRuntimeDeps' `require("x")` scan cannot see these specifiers — packaging +// stays green while the packaged app throws MODULE_NOT_FOUND the first time the path runs. +function collectLazyRequireSpecifiers(directory, found = new Map()) { + for (const entry of readdirSync(directory, { withFileTypes: true })) { + const entryPath = join(directory, entry.name) + if (entry.isDirectory()) { + collectLazyRequireSpecifiers(entryPath, found) + continue + } + if (!entry.isFile() || !entry.name.endsWith('.ts') || entry.name.includes('.test.')) { + continue + } + const source = readFileSync(entryPath, 'utf8') + if (!source.includes('createRequire(')) { + continue + } + for (const match of source.matchAll(/\brequire[A-Za-z0-9_]*\(\s*'([^']+)'\s*\)/g)) { + if (isPackagedExternalSpecifier(match[1])) { + found.set(match[1], relative(projectRoot, entryPath).replaceAll('\\', '/')) + } + } + } + return found +} + +function packagedResourceDestinations(platform) { + return new Set( + (electronBuilderConfig[platform].extraResources ?? []).map((resource) => + String(resource.to).replaceAll('\\', '/') + ) + ) +} + +describe('lazily required packages reach Resources/node_modules', () => { + it('copies every createRequire specifier main uses into the packaged resource plan', () => { + const specifiers = collectLazyRequireSpecifiers(join(projectRoot, 'src', 'main')) + expect(specifiers.size).toBeGreaterThan(0) + + const destinations = { + win: packagedResourceDestinations('win'), + mac: packagedResourceDestinations('mac'), + linux: packagedResourceDestinations('linux') + } + for (const [specifier, source] of specifiers) { + const packageName = packageNameFromSpecifier(specifier) + const covered = (platform) => + destinations[platform].has(`node_modules/${packageName}`) || + destinations[platform].has(`node_modules/${specifier}`) + // Windows carries the full closure, so an uncovered specifier is uncovered everywhere. + expect( + covered('win'), + `${source} lazily requires '${specifier}', but nothing copies it to Resources/node_modules` + ).toBe(true) + if (covered('mac') && covered('linux')) { + continue + } + // Only the Windows-native loaders may be absent from the mac/linux plans. + expect(source, `'${specifier}' is packaged for Windows only`).toContain('windows') + } + }) + + it('resolves the copied emoji dataset the way the packaged main bundle does', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-lazy-require-')) + try { + const datasetPath = 'node_modules/emojibase-data/en/shortcodes/emojibase.json' + const entry = electronBuilderConfig.mac.extraResources.find( + (resource) => String(resource.to) === datasetPath + ) + expect(entry).toBeDefined() + const destination = join(resourcesDir, ...datasetPath.split('/')) + await mkdir(dirname(destination), { recursive: true }) + await cp(join(projectRoot, ...String(entry.from).split('/')), destination) + + // app.asar's parent is Resources, so main's bare require walks into Resources/node_modules. + const packagedMainDir = join(resourcesDir, 'app.asar', 'out', 'main') + await mkdir(packagedMainDir, { recursive: true }) + const probe = join(packagedMainDir, 'probe.cjs') + await writeFile(probe, 'module.exports = require', 'utf8') + + const dataset = require(probe)('emojibase-data/en/shortcodes/emojibase.json') + expect(Object.keys(dataset).length).toBeGreaterThan(1000) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) +}) diff --git a/config/scripts/locale-ko-key-overrides.json b/config/scripts/locale-ko-key-overrides.json index f368ecc3cbc..bf5f62d1fa5 100644 --- a/config/scripts/locale-ko-key-overrides.json +++ b/config/scripts/locale-ko-key-overrides.json @@ -492,7 +492,7 @@ "ko": "agent CLI를 찾지 못했습니다. 하나를 설치하거나 설정에서 기본 agent를 선택하세요." }, "auto.components.Terminal.7958465754": { - "ko": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" + "ko": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" }, "auto.components.Terminal.cdc9ac4b2d": { "ko": "편집기" diff --git a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs new file mode 100644 index 00000000000..64fb1b056b8 --- /dev/null +++ b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs @@ -0,0 +1,213 @@ +// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with the +// desktop's own node-pty patch. pnpm patches do not cross the SSH boundary, so a relay runs the tree +// `npm install` put there; the desktop had this fix and the relay did not, and every terminal on a +// Windows SSH host leaked one File handle for the life of the relay process. +// +// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- what the +// desktop patch does -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a new +// Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +import { createRequire } from 'node:module' +import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const { + assertPatchedNodePtyWindowsTeardown, + patchNodePtyWindowsTeardown +} = require('../relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs') +const projectDir = resolve(import.meta.dirname, '..', '..') +const cleanupDirs = [] + +const PATCHED_FILES = ['windowsPtyAgent.js', 'windowsTerminal.js'] + +/** The hunks config/patches/node-pty@1.1.0.patch adds to the installed desktop tree. */ +const DESKTOP_HUNKS = { + 'windowsPtyAgent.js': [ + [ + [ + ' this._inSocket.readable = false;', + ' // The non-DLL path previously only flipped `readable`, leaving the', + ' // conin PipeWrap alive until the host exited (#947).', + ' this._inSocket.destroy();', + ' this._outSocket.readable = false;', + '' + ].join('\n'), + [ + ' this._inSocket.readable = false;', + ' this._outSocket.readable = false;', + '' + ].join('\n') + ], + // The useConptyDll branch, which only the DESKTOP runs -- the relay takes the + // non-DLL branch above, where the dispose is already unconditional. Listed here + // so un-applying still yields published; the relay asset needs no counterpart. + [ + [ + ' // Orca: dispose unconditionally, as the non-DLL branch above does.', + " // Waiting for another 'data' event leaks the conout worker on every", + ' // self-exiting shell, because no more data ever arrives (F24).', + ' this._conoutSocketWorker.dispose();', + '' + ].join('\n'), + [ + " this._outSocket.on('data', function () {", + ' _this._conoutSocketWorker.dispose();', + ' });', + '' + ].join('\n') + ] + ], + 'windowsTerminal.js': [ + [ + ' // Attach before readiness so a broken ConPTY output pipe cannot be unhandled.', + null + ], + [' // A ConPTY input-pipe error must retire only this terminal.', null] + ] +} + +function desktopPath(file) { + return join(projectDir, 'node_modules', 'node-pty', 'lib', file) +} + +afterEach(() => { + for (const dir of cleanupDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +describe('Windows SSH relay node-pty ConPTY teardown patch', () => { + // Why reconstruct rather than vendor upstream: the installed tree IS the published file plus the + // desktop's hunks, so un-applying them yields upstream exactly -- and pinning that against this + // asset's own hashes is what fails loudly if either side of the pair moves. + it('takes the desktop error listeners verbatim', () => { + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + + expect(readFileSync(join(fixture.libDir, 'windowsTerminal.js'), 'utf8')).toBe( + readFileSync(desktopPath('windowsTerminal.js'), 'utf8') + ) + }) + + // The one hunk that must NOT match the desktop, and the reason is measured, not stylistic: + // releasing conin before `_getConsoleProcessList()` forks aborts teardown partway. + it('releases conin after the console-list fork, not before it like the desktop patch', () => { + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + const patched = readFileSync(join(fixture.libDir, 'windowsPtyAgent.js'), 'utf8') + + const branch = patched.slice( + patched.indexOf('if (!this._useConptyDll) {'), + patched.indexOf('else {', patched.indexOf('if (!this._useConptyDll) {')) + ) + expect(branch).toContain('this._inSocket.destroy();') + expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( + branch.indexOf('this._conoutSocketWorker.dispose();') + ) + expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( + branch.indexOf('this._getConsoleProcessList()') + ) + // Pinned so a future "sync the relay asset to config/patches" cannot copy the regression back. + expect(patched).not.toBe(readFileSync(desktopPath('windowsPtyAgent.js'), 'utf8')) + }) + + it('installs and verifies idempotently', () => { + const fixture = writeNodePtyFixture('1.1.0') + + patchNodePtyWindowsTeardown(fixture.root) + const once = PATCHED_FILES.map((file) => readFileSync(join(fixture.libDir, file), 'utf8')) + for (const file of PATCHED_FILES) { + expect(existsSync(`${join(fixture.libDir, file)}.orca-patch-${process.pid}`)).toBe(false) + } + expect(() => assertPatchedNodePtyWindowsTeardown(fixture.root)).not.toThrow() + + patchNodePtyWindowsTeardown(fixture.root) + expect(PATCHED_FILES.map((file) => readFileSync(join(fixture.libDir, file), 'utf8'))).toEqual( + once + ) + }) + + it('refuses a different package version or unexpected source', () => { + const wrongVersion = writeNodePtyFixture('1.2.0-beta.11') + expect(() => patchNodePtyWindowsTeardown(wrongVersion.root)).toThrow('expected 1.1.0') + + for (const file of PATCHED_FILES) { + const drifted = writeNodePtyFixture('1.1.0') + const path = join(drifted.libDir, file) + writeFileSync(path, `${readFileSync(path, 'utf8')}\n// drift`) + expect(() => patchNodePtyWindowsTeardown(drifted.root)).toThrow('unexpected node-pty') + } + }) + + it('refuses a half-applied tree, so one file cannot pass for both', () => { + for (const file of PATCHED_FILES) { + const partial = writeNodePtyFixture('1.1.0') + const fixture = writeNodePtyFixture('1.1.0') + patchNodePtyWindowsTeardown(fixture.root) + writeFileSync(join(partial.libDir, file), readFileSync(join(fixture.libDir, file), 'utf8')) + expect(() => assertPatchedNodePtyWindowsTeardown(partial.root)).toThrow('is not installed') + } + }) +}) + +/** A published node-pty tree, rebuilt by un-applying the desktop hunks from the installed one. */ +function writeNodePtyFixture(version) { + const root = mkdtempSync(join(projectDir, '.node-pty-teardown-patch-test-')) + cleanupDirs.push(root) + const libDir = join(root, 'node_modules', 'node-pty', 'lib') + mkdirSync(libDir, { recursive: true }) + writeFileSync(join(root, 'node_modules', 'node-pty', 'package.json'), JSON.stringify({ version })) + for (const file of PATCHED_FILES) { + const desktop = readFileSync(desktopPath(file), 'utf8') + for (const [marker] of DESKTOP_HUNKS[file]) { + expect(desktop).toContain(marker) + } + writeFileSync(join(libDir, file), unapplyDesktopHunks(file, desktop)) + } + return { root, libDir } +} + +/** + * Reverse of the published-to-desktop transform. + * + * `windowsTerminal.js` is taken verbatim from the desktop, so the asset's own replacement table is + * the transform and reversing it is exact. `windowsPtyAgent.js` deliberately diverges, so its + * published form is rebuilt from the desktop hunk instead -- which is also what makes this file the + * place that notices if the desktop hunk itself ever moves. + */ +function unapplyDesktopHunks(file, desktop) { + if (file === 'windowsPtyAgent.js') { + let published = desktop + for (const [patched, original] of DESKTOP_HUNKS[file]) { + expect(published.split(patched).length - 1).toBe(1) + published = published.replace(patched, original) + } + return published + } + const asset = readFileSync( + join(projectDir, 'config', 'relay-assets', 'node-pty-1.1.0-windows-pty-teardown-patch.cjs'), + 'utf8' + ) + const { PATCH_TARGETS } = loadPatchTargets(asset) + const target = PATCH_TARGETS.find((entry) => entry.relativePath.at(-1) === file) + expect(target).toBeDefined() + let published = desktop + for (const [from, to] of target.replacements.toReversed()) { + expect(published.split(to).length - 1).toBe(1) + published = published.replace(to, from) + } + return published +} + +function loadPatchTargets(assetSource) { + const module = { exports: {} } + const factory = new Function( + 'module', + 'exports', + 'require', + `${assetSource}\nmodule.exports.PATCH_TARGETS = PATCH_TARGETS` + ) + factory(module, module.exports, require) + return module.exports +} diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index 7111531c35c..f5a73f6239f 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -228,6 +228,7 @@ const WINDOWS_PACKAGE_TESTS = [ 'src/main/cli/wsl-cli-powershell-boundary.test.ts', 'src/main/cursor/hook-service.test.ts', 'src/main/orca-profiles/profile-index-store.test.ts', + 'src/main/startup/windows-install-dir-acl-repair.win32.test.ts', 'src/main/runtime/repo-worktree-admin-fingerprint.test.ts', 'src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts', 'src/shared/secure-file-fsync-flags.test.ts', diff --git a/config/scripts/skill-description-length.test.mjs b/config/scripts/skill-description-length.test.mjs new file mode 100644 index 00000000000..e7a9db79541 --- /dev/null +++ b/config/scripts/skill-description-length.test.mjs @@ -0,0 +1,39 @@ +import { readdirSync, readFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' + +const skillsDir = resolve(import.meta.dirname, '../../skills') +// Why: the Agent Skills spec caps `description` at 1024 chars and conforming installers +// reject the whole skill (#17935); the frontmatter is what the installer parses, so check it. +const MAX_DESCRIPTION_LENGTH = 1024 + +function readDescription(skillName) { + const skillMarkdown = readFileSync(join(skillsDir, skillName, 'SKILL.md'), 'utf8') + const frontmatter = /^---\r?\n([\s\S]*?)\r?\n---\r?\n/u.exec(skillMarkdown)?.[1] + + expect(frontmatter, `${skillName}: missing frontmatter`).toBeDefined() + + return parse(frontmatter ?? '').description +} + +describe('bundled skill descriptions', () => { + const skillNames = readdirSync(skillsDir, { withFileTypes: true }) + .filter((entry) => entry.isDirectory()) + .map((entry) => entry.name) + + it('discovers the bundled skills', () => { + expect(skillNames).toContain('orchestration') + }) + + it.each(skillNames)('%s keeps description within the Agent Skills spec limit', (name) => { + const description = readDescription(name) + + expect(typeof description, `${name}: description must be a string`).toBe('string') + expect(description.trim().length, `${name}: description is empty`).toBeGreaterThan(0) + expect( + description.length, + `${name}: description is ${description.length} chars` + ).toBeLessThanOrEqual(MAX_DESCRIPTION_LENGTH) + }) +}) diff --git a/config/tsconfig.tc.web.json b/config/tsconfig.tc.web.json index 56253527c69..2caf2149f73 100644 --- a/config/tsconfig.tc.web.json +++ b/config/tsconfig.tc.web.json @@ -19,6 +19,7 @@ "../src/preload/usage-provider-api.ts", "../src/shared/**/*", "../src/main/gitlab/mappers.ts", + "../src/main/ipc/deferred-emoji-shortcode-dataset.ts", "../src/main/ipc/worktree-branch-name.ts", "../src/main/ipc/worktree-logic.ts", "../src/main/ipc/worktree-display-name.ts", diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index ef8ebb61bb4..c240c965fee 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 38m + + downloads: 39m @@ -15,7 +15,7 @@ downloads downloads - 38m - 38m + 39m + 39m diff --git a/docs/assets/wechat-qr-group9.jpg b/docs/assets/wechat-qr-group9.jpg new file mode 100644 index 00000000000..2bf46a28c3d Binary files /dev/null and b/docs/assets/wechat-qr-group9.jpg differ diff --git a/docs/readme/README.es.md b/docs/readme/README.es.md index c8333ec31c9..f2247e0900d 100644 --- a/docs/readme/README.es.md +++ b/docs/readme/README.es.md @@ -12,7 +12,7 @@

- English · 中文 · 日本語 · 한국어 · Français · Português · Українська + English · 中文 · 日本語 · 한국어 · Français · Português

diff --git a/docs/readme/README.fr.md b/docs/readme/README.fr.md index 7a69e1acffe..e601abc2344 100644 --- a/docs/readme/README.fr.md +++ b/docs/readme/README.fr.md @@ -12,7 +12,7 @@

- English · 中文 · 日本語 · 한국어 · Español · Português · Українська + English · 中文 · 日本語 · 한국어 · Español · Português

@@ -243,9 +243,9 @@ Associez-la à l'app de bureau pour surveiller et piloter vos agents depuis votr - **Discord :** Rejoignez la communauté sur **[Discord](https://discord.gg/fzjDKHxv8Q)**. - **Twitter / X :** Suivez **[@orca_build](https://x.com/orca_build)** pour les news et annonces. -- **WeChat :** Scannez pour rejoindre le groupe WeChat 8 de la communauté Orca. +- **WeChat :** Scannez pour rejoindre le groupe WeChat 8 de la communauté Orca. Le groupe 8 est peut-être complet ; dans ce cas, scannez plutôt le QR code du groupe 9. - QR code WeChat groupe 8 de la communauté Orca + QR code WeChat groupe 8 de la communauté Orca  QR code WeChat groupe 9 de la communauté Orca - **Feedback & idées :** On ship vite. Il manque quelque chose ? [Demandez une feature](https://github.com/stablyai/orca/issues). - **Confidentialité :** Voir la [doc confidentialité & télémétrie](https://www.onorca.dev/docs/telemetry) pour ce qu'Orca collecte en anonyme et comment désactiver la télémétrie. diff --git a/docs/readme/README.ja.md b/docs/readme/README.ja.md index a256f5239c4..cce2032a67c 100644 --- a/docs/readme/README.ja.md +++ b/docs/readme/README.ja.md @@ -12,7 +12,7 @@

- English · 中文 · 한국어 · Español · Français · Português · Українська + English · 中文 · 한국어 · Español · Français · Português

diff --git a/docs/readme/README.ko.md b/docs/readme/README.ko.md index d2dd2a62747..837ecf2133f 100644 --- a/docs/readme/README.ko.md +++ b/docs/readme/README.ko.md @@ -12,7 +12,7 @@

- English · 中文 · 日本語 · Español · Français · Português · Українська + English · 中文 · 日本語 · Español · Français · Português

@@ -238,9 +238,9 @@ yay -S stably-orca-bin - **Discord:** **[Discord](https://discord.gg/fzjDKHxv8Q)** 커뮤니티에 참여하세요. - **Twitter / X:** 업데이트와 공지는 **[@orca_build](https://x.com/orca_build)** 를 팔로우하세요. -- **WeChat:** QR 코드를 스캔해 Orca 커뮤니티 WeChat 그룹 8에 참여하세요. +- **WeChat:** QR 코드를 스캔해 Orca 커뮤니티 WeChat 그룹 8에 참여하세요. 그룹 8이 가득 찼을 수 있으니, 그런 경우 그룹 9 QR 코드를 스캔하세요. - Orca 커뮤니티 WeChat 그룹 8 QR 코드 + Orca 커뮤니티 WeChat 그룹 8 QR 코드  Orca 커뮤니티 WeChat 그룹 9 QR 코드 - **피드백과 아이디어:** 우리는 빠르게 출시합니다. 필요한 기능이 있나요? [새 기능을 요청](https://github.com/stablyai/orca/issues)하세요. - **개인정보 보호:** Orca가 수집하는 익명 사용 데이터와 수집 거부 방법은 [개인정보 및 텔레메트리 문서](https://www.onorca.dev/docs/telemetry)를 참고하세요. diff --git a/docs/readme/README.pt.md b/docs/readme/README.pt.md index 667cebb685f..86d998a4e5f 100644 --- a/docs/readme/README.pt.md +++ b/docs/readme/README.pt.md @@ -12,7 +12,7 @@

- English · 中文 · 日本語 · 한국어 · Español · Français · Українська + English · 中文 · 日本語 · 한국어 · Español · Français

diff --git a/docs/readme/README.uk.md b/docs/readme/README.uk.md deleted file mode 100644 index b3e2a42b868..00000000000 --- a/docs/readme/README.uk.md +++ /dev/null @@ -1,269 +0,0 @@ -

- Orca Orca -

- -

- Зірки на GitHub - Загальна кількість завантажень усіх релізів - Ліцензія: MIT - Приєднатися до Discord Orca - Стежити за Orca в X - Підтримувані платформи: macOS, Windows і Linux -

- -

- English · 中文 · 日本語 · 한국어 · Español · Français · Português -

- -

- AI-оркестратор для розробників рівня 100x.
- Запускайте Codex, Claude Code, OpenCode або Pi паралельно — кожен у власному worktree, усі під контролем в одному місці. -

- -

Завантажити Orca

- -

- Десктопний застосунок Orca запускає агентів у паралельних worktree, у кутку — супутній мобільний застосунок Orca -

- -## Можливості - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
- -### Супутній мобільний застосунок - -Стежте за агентами та керуйте ними з телефону — отримуйте сповіщення про завершення роботи агента та надсилайте подальші вказівки, де б ви не були. - -[App Store для iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.44](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.44/app-release.apk) · [Документація →](https://www.onorca.dev/docs/mobile) - - - Десктоп Orca із супутнім мобільним застосунком -
- -### Паралельні worktree - -Надішліть один промпт одразу п’ятьом агентам, кожен із яких працюватиме у власному ізольованому git worktree, — порівняйте результати та виконайте злиття найкращого з них. - -[Документація →](https://www.onorca.dev/docs/model/worktrees) - - - Оркестрація паралельних worktree -
- -### Розділені термінали - -Термінали рівня Ghostty з рендерингом на WebGL, необмеженою кількістю розділень і буфером прокручування, який зберігається після перезапуску. - -[Документація →](https://www.onorca.dev/docs/terminal) - - - Розділені термінали -
- -### Режим дизайну - -Клацніть на будь-якому елементі інтерфейсу у справжньому вікні Chromium, щоб надіслати його HTML, CSS і обрізаний скриншот прямо в промпт агента. - -[Документація →](https://www.onorca.dev/docs/browser/design-mode) - - - Вбудований браузер і режим дизайну -
- -### GitHub і Linear, нативно - -Переглядайте PR, issue та дошки проєктів прямо в застосунку — відкривайте worktree з будь-якої задачі та рев'юйте без перемикання контексту. - -[Документація →](https://www.onorca.dev/docs/review/linear) - - - Робочі процеси GitHub і Linear в Orca -
- -### SSH worktree - -Запускайте агентів на потужній віддаленій машині з повноцінним редагуванням файлів, git і терміналами — з автоперепідключенням і прокиданням портів. - -[Документація →](https://www.onorca.dev/docs/ssh) - - - Віддалені worktree через SSH -
- -### Анотуйте diff-и агентів - -Залишайте коментарі на будь-якому рядку diff-у й надсилайте їх агенту — рев'юйте, редагуйте та комітьте, не виходячи з Orca. - -[Документація →](https://www.onorca.dev/docs/review/annotate-ai-diff) - - - Анотування diff-ів, згенерованих AI -
- -### Перетягуйте файли агентам - -Редактор на базі VS Code з автозбереженням усюди — перетягуйте файли чи зображення прямо в промпт агента. - -[Документація →](https://www.onorca.dev/docs/editing/file-explorer) - - - Перетягування файлів і зображень у промпт агента -
- -### Orca CLI - -Агенти теж керують Orca — автоматизуйте будь-який робочий процес командами `orca worktree create`, `snapshot`, `click` і `fill`. - -[Документація →](https://www.onorca.dev/docs/cli/overview) - - - Керування Orca з CLI -
- -**Також у комплекті:** - -- **[Швидкий пошук](https://www.onorca.dev/docs/model/quick-open)** — Шукайте серед worktree, файлів, агентів, команд і контексту репозиторію, не відриваючись від роботи. -- **[Перемикач акаунтів і відстеження використання](https://www.onorca.dev/docs/agents/usage-tracking)** — Стежте за використанням Claude і Codex та скиданням лімітів, перемикайте акаунти на льоту без повторного входу. -- **[Розширені перегляди репозиторію](https://www.onorca.dev/docs/editing/markdown)** — Переглядайте Markdown, зображення, PDF та документацію репозиторію прямо в робочому просторі. -- **[Computer Use](https://www.onorca.dev/docs/cli/computer-use)** — Дозвольте агентам керувати десктопними застосунками та видимим інтерфейсом, коли робочий процес потребує реальної взаємодії. -- **[Сповіщення та статус непрочитаного](https://www.onorca.dev/docs/notifications)** — Дізнавайтеся, коли агент завершив роботу або потребує уваги, і позначайте треди як непрочитані, щоб повернутися пізніше. -- **І багато іншого** — ми випускаємо оновлення щодня, тож цей список завжди відстає. Справжній перелік можливостей — це [changelog](https://github.com/stablyai/orca/releases). - ---- - -## Підтримувані агенти - -Працює з **будь-яким CLI-агентом** — якщо він запускається в терміналі, він запуститься і в Orca. - -

- Claude Code logo Claude Code   - Codex logo Codex   - Grok logo Grok   - Cursor logo Cursor   - GitHub Copilot logo GitHub Copilot   - OpenCode logo OpenCode   - MiMo Code logo MiMo Code   - Amp logo Amp   - OpenClaude logo OpenClaude   - Antigravity logo Antigravity   - Pi logo Pi   - oh-my-pi logo oh-my-pi   - Hermes Agent logo Hermes Agent   - Devin logo Devin   - Goose logo Goose   - Auggie logo Auggie   - Autohand Code logo Autohand Code   - Charm logo Charm   - Cline logo Cline   - Codebuff logo Codebuff   - Command Code logo Command Code   - Continue logo Continue   - Droid logo Droid   - Kilocode logo Kilocode   - Kimi logo Kimi   - Kiro logo Kiro   - Mistral Vibe logo Mistral Vibe   - Qwen Code logo Qwen Code   - Rovo Dev logo Rovo Dev   - + будь-який CLI-агент -

- ---- - -## Встановлення - -### Десктоп — macOS, Windows, Linux - -- **[Завантажити з onOrca.dev](https://onorca.dev/download)** -- Або завантажте білд напряму: [macOS Apple Silicon](https://github.com/stablyai/orca/releases/latest/download/orca-macos-arm64.dmg) · [macOS Intel](https://github.com/stablyai/orca/releases/latest/download/orca-macos-x64.dmg) · [Windows (.exe)](https://github.com/stablyai/orca/releases/latest/download/orca-windows-setup.exe) · [Linux AppImage](https://github.com/stablyai/orca/releases/latest/download/orca-linux.AppImage) · [Усі білди](https://github.com/stablyai/orca/releases/latest) -- Запускаєте `orca serve` на headless Linux-сервері? Дивіться [посібник із headless Linux-сервера](../reference/headless-linux-server.md). - -_Або через пакетний менеджер:_ - -```bash -# macOS (Homebrew) -brew install --cask stablyai/orca/orca - -# Arch Linux (AUR) — або stably-orca-git для збірки з джерела -yay -S stably-orca-bin -``` - -### Супутній мобільний застосунок — iOS, Android - -Під’єднайте мобільний застосунок до десктопного, щоб стежити за агентами та керувати ними з телефону. - -- **iOS:** [Завантажити з App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) або [приєднатися до TestFlight](https://testflight.apple.com/join/YjeGMQBA) -- **Android:** [Завантажити APK 0.0.44](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.44/app-release.apk) · [Інструкція зі встановлення](https://www.onorca.dev/docs/android-apk) - ---- - -## Спільнота та підтримка - -- **Discord:** Приєднуйтеся до спільноти в **[Discord](https://discord.gg/fzjDKHxv8Q)**. -- **Twitter / X:** Стежте за **[@orca_build](https://x.com/orca_build)**, щоб бути в курсі оновлень і анонсів. -- **WeChat:** Відскануйте QR-код, щоб приєднатися до групи № 7 спільноти Orca у WeChat. Якщо вона заповнена, приєднайтеся до групи № 8. - - QR-код групи WeChat 7 спільноти Orca   - QR-код групи WeChat 8 спільноти Orca - -- **Зворотний зв'язок та ідеї:** Ми випускаємо оновлення швидко. Чогось бракує? [Запропонуйте нову функцію](https://github.com/stablyai/orca/issues). -- **Конфіденційність:** Перегляньте [документацію про конфіденційність і телеметрію](https://www.onorca.dev/docs/telemetry), щоб дізнатися, які анонімні дані про використання збирає Orca і як від цього відмовитися. -- **Підтримайте нас:** Поставте [зірку](https://github.com/stablyai/orca) цьому репозиторію, щоб стежити за нашими щоденними релізами. - ---- - -## Розробка - -Хочете зробити внесок або запустити проєкт локально? Перегляньте наш посібник [CONTRIBUTING.md](../../.github/CONTRIBUTING.md). - - - Контриб'ютори Orca - - -

- Графік історії зірок на GitHub для stablyai/orca -

- -## Підписані білди -Підписання коду для Windows надано за підтримки [SignPath.io](https://signpath.io), сертифікат надано [SignPath Foundation](https://signpath.org). - -## Ліцензія - -Orca — безкоштовний проєкт із відкритим кодом за ліцензією [MIT](../../LICENSE). diff --git a/docs/readme/README.zh-CN.md b/docs/readme/README.zh-CN.md index 160f5b05611..10f47e20fe6 100644 --- a/docs/readme/README.zh-CN.md +++ b/docs/readme/README.zh-CN.md @@ -12,7 +12,7 @@

- English · 日本語 · 한국어 · Español · Français · Português · Українська + English · 日本語 · 한국어 · Español · Français · Português

@@ -235,9 +235,9 @@ yay -S stably-orca-bin - **Discord:** 加入 **[Discord](https://discord.gg/fzjDKHxv8Q)** 社区。 - **Twitter / X:** 关注 **[@orca_build](https://x.com/orca_build)** 获取更新和公告。 -- **微信:** 扫码加入 Orca 社区微信第 8 群。 +- **微信:** 扫码加入 Orca 社区微信第 8 群。第 8 群可能已满,如遇这种情况请扫描第 9 群二维码。 - Orca 社区微信第 8 群二维码 + Orca 社区微信第 8 群二维码  Orca 社区微信第 9 群二维码 - **反馈与想法:** 我们发布很快。缺少什么功能?[提交功能请求](https://github.com/stablyai/orca/issues)。 - **隐私:** 查看[隐私与遥测文档](https://www.onorca.dev/docs/telemetry),了解 Orca 收集哪些匿名使用数据以及如何退出。 diff --git a/mobile/src/home/MobileHomeHostList.tsx b/mobile/src/home/MobileHomeHostList.tsx index 3907df16f03..41d1f07156d 100644 --- a/mobile/src/home/MobileHomeHostList.tsx +++ b/mobile/src/home/MobileHomeHostList.tsx @@ -19,6 +19,7 @@ type MobileHomeHostListProps = { hostAttempts: Record hostLastConnected: Record hostPairingRejected: Record + hostSignedOut: Record hostPaths: Record hostPendingPaths: Record hosts: HostCatalogEntry[] @@ -40,6 +41,7 @@ export function MobileHomeHostList(props: MobileHomeHostListProps) { hostAttempts={props.hostAttempts} hostLastConnected={props.hostLastConnected} hostPairingRejected={props.hostPairingRejected} + hostSignedOut={props.hostSignedOut} hostPaths={props.hostPaths} hostPendingPaths={props.hostPendingPaths} hostStates={props.hostStates} @@ -54,6 +56,7 @@ export function MobileHomeHostList(props: MobileHomeHostListProps) { props.hostAttempts, props.hostLastConnected, props.hostPairingRejected, + props.hostSignedOut, props.hostPaths, props.hostPendingPaths, props.hostStates, @@ -91,6 +94,7 @@ type MobileHomeHostRowProps = Pick< | 'hostAttempts' | 'hostLastConnected' | 'hostPairingRejected' + | 'hostSignedOut' | 'hostPaths' | 'hostPendingPaths' | 'hostStates' @@ -113,7 +117,8 @@ const MobileHomeHostRow = memo(function MobileHomeHostRow(props: MobileHomeHostR lastConnectedAt: props.hostLastConnected[item.id] ?? null, endpoint: item.endpoint, pendingPath: props.hostPendingPaths[item.id] ?? null, - pairingRejected: props.hostPairingRejected[item.id] ?? false + pairingRejected: props.hostPairingRejected[item.id] ?? false, + hostSignedOut: props.hostSignedOut[item.id] ?? false }) const open = useCallback(() => onOpen(item), [item, onOpen]) const longPress = useCallback(() => onLongPress(item), [item, onLongPress]) diff --git a/mobile/src/home/MobileHomeScreen.tsx b/mobile/src/home/MobileHomeScreen.tsx index 83cf3de4b8a..7772f5152c6 100644 --- a/mobile/src/home/MobileHomeScreen.tsx +++ b/mobile/src/home/MobileHomeScreen.tsx @@ -143,6 +143,7 @@ export function MobileHomeScreen() { hostAttempts={data.hostAttempts} hostLastConnected={data.hostLastConnected} hostPairingRejected={data.hostPairingRejected} + hostSignedOut={data.hostSignedOut} hostPaths={data.hostPaths} hostPendingPaths={data.hostPendingPaths} hosts={data.sortedHostCatalog} diff --git a/mobile/src/home/home-host-connection-projection.ts b/mobile/src/home/home-host-connection-projection.ts index f8fdfd4bcdf..9a49f6186c4 100644 --- a/mobile/src/home/home-host-connection-projection.ts +++ b/mobile/src/home/home-host-connection-projection.ts @@ -5,12 +5,14 @@ export type HomeHostConnectionProjectionEntry = { path: MobileConnectionPath pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean } export type HomeHostConnectionProjection = { hostPaths: Record hostPendingPaths: Record hostPairingRejected: Record + hostSignedOut: Record } /** Build all host lookup maps while reading each connection entry once. */ @@ -22,16 +24,19 @@ export function projectHomeHostConnections( const hostPaths = Object.create(null) as Record const hostPendingPaths = Object.create(null) as Record const hostPairingRejected = Object.create(null) as Record + const hostSignedOut = Object.create(null) as Record - for (const { hostId, path, pendingPath, pairingRejected } of entries) { + for (const { hostId, path, pendingPath, pairingRejected, hostSignedOut: signedOut } of entries) { hostPaths[hostId] = path hostPendingPaths[hostId] = pendingPath hostPairingRejected[hostId] = pairingRejected + hostSignedOut[hostId] = signedOut } Object.setPrototypeOf(hostPaths, Object.prototype) Object.setPrototypeOf(hostPendingPaths, Object.prototype) Object.setPrototypeOf(hostPairingRejected, Object.prototype) + Object.setPrototypeOf(hostSignedOut, Object.prototype) - return { hostPaths, hostPendingPaths, hostPairingRejected } + return { hostPaths, hostPendingPaths, hostPairingRejected, hostSignedOut } } diff --git a/mobile/src/home/use-mobile-home-data.ts b/mobile/src/home/use-mobile-home-data.ts index c77d024158e..6b28354a86d 100644 --- a/mobile/src/home/use-mobile-home-data.ts +++ b/mobile/src/home/use-mobile-home-data.ts @@ -179,6 +179,7 @@ export function useMobileHomeData() { connectedHosts, hostCatalog, hostPairingRejected: hostConnectionProjection.hostPairingRejected, + hostSignedOut: hostConnectionProjection.hostSignedOut, hostPaths: hostConnectionProjection.hostPaths, hostPendingPaths: hostConnectionProjection.hostPendingPaths, primaryHost, diff --git a/mobile/src/session/mobile-image-attachment.test.ts b/mobile/src/session/mobile-image-attachment.test.ts index 9d725d5fe60..eead9691303 100644 --- a/mobile/src/session/mobile-image-attachment.test.ts +++ b/mobile/src/session/mobile-image-attachment.test.ts @@ -50,7 +50,9 @@ describe('attachMobileImageToTerminal', () => { const sendCall = client.calls.find((c) => c.method === 'terminal.send') expect(sendCall?.params).toEqual({ terminal: 'term-1', - text: '\x1b[200~/tmp/orca-attach.png\x1b[201~', + // Trailing space: the user types on this same line next, so a bare + // `…\x1b[201~` would arrive as `…pngadd` (STA-4847). + text: '\x1b[200~/tmp/orca-attach.png\x1b[201~ ', enter: false, client: { id: 'device-9', type: 'mobile' } }) diff --git a/mobile/src/session/mobile-image-attachment.ts b/mobile/src/session/mobile-image-attachment.ts index 567be99a9a1..9cb7d60aa8e 100644 --- a/mobile/src/session/mobile-image-attachment.ts +++ b/mobile/src/session/mobile-image-attachment.ts @@ -1,4 +1,5 @@ import type { RpcClient } from '../transport/rpc-client' +import { separateImagePasteFromFollowingText } from '../../../src/shared/image-paste-following-text' import { buildMobileImagePastePayload, saveMobileClipboardImageAsTempFile @@ -47,7 +48,10 @@ export async function attachMobileImageToTerminal( }) // Why: a generated image path is terminal image injection, so it's always // bracketed (matching desktop paste) regardless of terminal mode. - const payload = buildMobileImagePastePayload(imagePath) + // Always separated: attach-then-type is the whole interaction here, so the user's + // next keystroke would otherwise glue onto the path (`…pngadd`). Unlike native + // chat there is no batch to look ahead in, and a trailing space is inert. + const payload = separateImagePasteFromFollowingText(buildMobileImagePastePayload(imagePath), true) if (beforeTerminalSend && !(await beforeTerminalSend(terminal))) { return false } diff --git a/mobile/src/session/mobile-native-chat-image-send.test.ts b/mobile/src/session/mobile-native-chat-image-send.test.ts index cf1c59adf0f..a41b3fca3f1 100644 --- a/mobile/src/session/mobile-native-chat-image-send.test.ts +++ b/mobile/src/session/mobile-native-chat-image-send.test.ts @@ -38,7 +38,8 @@ describe('pasteMobileNativeChatImagePaths', () => { client, terminal: 'term-1', deviceToken: 'device-9', - imagePaths: ['/tmp/a.png', '/tmp/b.png', '/tmp/c.png'] + imagePaths: ['/tmp/a.png', '/tmp/b.png', '/tmp/c.png'], + followedByText: true }) expect(ok).toBe(true) @@ -55,7 +56,7 @@ describe('pasteMobileNativeChatImagePaths', () => { }) expect(client.calls[1]?.params.text).toBe('\x1b[200~/tmp/a.png\x1b[201~') expect(client.calls[2]?.params.text).toBe('\x1b[200~/tmp/b.png\x1b[201~') - expect(client.calls[3]?.params.text).toBe('\x1b[200~/tmp/c.png\x1b[201~') + expect(client.calls[3]?.params.text).toBe('\x1b[200~/tmp/c.png\x1b[201~ ') }) it('stops and reports failure as soon as a paste is rejected', async () => { @@ -66,7 +67,8 @@ describe('pasteMobileNativeChatImagePaths', () => { client, terminal: 'term-1', deviceToken: null, - imagePaths: ['/tmp/a.png', '/tmp/b.png'] + imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: true }) expect(ok).toBe(false) @@ -94,7 +96,8 @@ describe('pasteMobileNativeChatImagePaths', () => { client, terminal: 'term-1', deviceToken: null, - imagePaths: ['/tmp/a.png', '/tmp/b.png'] + imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: true }) expect(ok).toBe(false) @@ -121,6 +124,7 @@ describe('clearing a parked multi-line launch draft before the image paste', () terminal: 'term-1', deviceToken: null, imagePaths: ['/tmp/a.png'], + followedByText: true, clearInput }) @@ -137,6 +141,7 @@ describe('clearing a parked multi-line launch draft before the image paste', () terminal: 'term-1', deviceToken: null, imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: true, clearInput }) @@ -151,9 +156,27 @@ describe('clearing a parked multi-line launch draft before the image paste', () client, terminal: 'term-1', deviceToken: null, - imagePaths: ['/tmp/a.png'] + imagePaths: ['/tmp/a.png'], + followedByText: true }) expect(client.calls[0]?.params.text).toBe('\x15') }) + + it('keeps image writes byte-clean when no text or submit follows', async () => { + const client = clientWithResponses([sendResult(true), sendResult(true), sendResult(true)]) + + await pasteMobileNativeChatImagePaths({ + client, + terminal: 'term-1', + deviceToken: null, + imagePaths: ['/tmp/a.png', '/tmp/b.png'], + followedByText: false + }) + + expect(client.calls.slice(1).map((call) => call.params.text)).toEqual([ + '\x1b[200~/tmp/a.png\x1b[201~', + '\x1b[200~/tmp/b.png\x1b[201~' + ]) + }) }) diff --git a/mobile/src/session/mobile-native-chat-image-send.ts b/mobile/src/session/mobile-native-chat-image-send.ts index adb996b7612..9f3c555062f 100644 --- a/mobile/src/session/mobile-native-chat-image-send.ts +++ b/mobile/src/session/mobile-native-chat-image-send.ts @@ -1,4 +1,5 @@ import type { RpcClient } from '../transport/rpc-client' +import { imagePasteWritesFollowedByText } from '../../../src/shared/image-paste-following-text' import { buildMobileImagePastePayload } from './mobile-clipboard-image' import { MOBILE_NATIVE_CHAT_MIN_WRITE_TIMEOUT_MS, @@ -23,6 +24,7 @@ type PasteImagesArgs = { readonly terminal: string readonly deviceToken: string | null readonly imagePaths: readonly string[] + readonly followedByText: boolean /** Budget shared with the rest of the user action (the text body that follows, or * the send this is healing for). Omit to open a fresh one for this paste alone. */ readonly deadline?: number @@ -42,6 +44,7 @@ export async function pasteMobileNativeChatImagePaths({ terminal, deviceToken, imagePaths, + followedByText, deadline: sharedDeadline, clearInput }: PasteImagesArgs): Promise { @@ -55,7 +58,7 @@ export async function pasteMobileNativeChatImagePaths({ const deadline = sharedDeadline ?? openMobileNativeChatSendBudget() for (const text of [ clearInput ?? MOBILE_NATIVE_CHAT_CLEAR_UNSUBMITTED_INPUT, - ...imagePaths.map(buildMobileImagePastePayload) + ...imagePasteWritesFollowedByText(imagePaths.map(buildMobileImagePastePayload), followedByText) ]) { const remainingMs = deadline - Date.now() // Why: the budget is the whole sequence's — starting a write it can't fund would diff --git a/mobile/src/session/mobile-native-chat-stale-input.ts b/mobile/src/session/mobile-native-chat-stale-input.ts index 18d84d2e503..cdc21067683 100644 --- a/mobile/src/session/mobile-native-chat-stale-input.ts +++ b/mobile/src/session/mobile-native-chat-stale-input.ts @@ -55,6 +55,7 @@ export async function healMobileNativeChatStaleInput(args: { terminal: args.terminal, deviceToken: args.deviceToken, imagePaths: [], + followedByText: false, ...(args.deadline === undefined ? {} : { deadline: args.deadline }) }) } catch { diff --git a/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts b/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts index cb8522bae7a..022bb8c3953 100644 --- a/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts +++ b/mobile/src/session/use-mobile-native-chat-image-attachments.test.ts @@ -182,9 +182,12 @@ describe('useMobileNativeChatImageAttachments', () => { expect(sendCalls).toHaveLength(2) expect(sendCalls[0]?.params).toMatchObject({ text: '\x15', enter: false }) expect(sendCalls[1]?.params).toMatchObject({ - text: '\x1b[200~/tmp/a.png\x1b[201~', + text: '\x1b[200~/tmp/a.png\x1b[201~ ', enter: false }) + const combined = String(sendCalls[1]?.params.text ?? '') + 'look at this' + expect(combined).toContain('.png\x1b[201~ look') + expect(combined).not.toContain('.png\x1b[201~look') // Clear, then paste, then settle, then the text send — in that order. expect(order).toEqual(['clear', 'paste', 'settle', 'text:look at this']) // The local preview URI rides along so the sent bubble shows the photo. @@ -267,7 +270,10 @@ describe('useMobileNativeChatImageAttachments', () => { } }) - it('routes an attachments-only send through baseSend with empty text so the echo still shows the photo', async () => { + it.each([ + ['empty', ''], + ['whitespace-only', ' '] + ])('routes an attachments-only send through baseSend with %s text', async (_label, text) => { pick.mockResolvedValue([{ base64: 'AAAA', uri: 'file:///a.jpg' }]) const client = makeClient([ methodNotFound('start'), @@ -283,16 +289,17 @@ describe('useMobileNativeChatImageAttachments', () => { }) let accepted = false await act(async () => { - accepted = await hook!.sendNativeChat('') + accepted = await hook!.sendNativeChat(text) }) expect(accepted).toBe(true) - // Empty text still goes through baseSend (which submits the bare Enter) so the + // Attachment-only text still goes through baseSend (which submits Enter) so the // optimistic echo carries the preview URI. - expect(baseSend).toHaveBeenCalledWith('', ['file:///a.jpg'], expect.any(Number)) + expect(baseSend).toHaveBeenCalledWith(text, ['file:///a.jpg'], expect.any(Number)) const sendCalls = client.calls.filter((c) => c.method === 'terminal.send') // Only the clear + image paste hit the wire here; baseSend owns the submit. expect(sendCalls).toHaveLength(2) + expect(sendCalls[1]?.params.text).toBe('\x1b[200~/tmp/a.png\x1b[201~') expect(hook!.attachments).toEqual([]) }) diff --git a/mobile/src/session/use-mobile-native-chat-image-attachments.ts b/mobile/src/session/use-mobile-native-chat-image-attachments.ts index 77d839adf34..c36e30f44d7 100644 --- a/mobile/src/session/use-mobile-native-chat-image-attachments.ts +++ b/mobile/src/session/use-mobile-native-chat-image-attachments.ts @@ -240,6 +240,7 @@ export function useMobileNativeChatImageAttachments({ terminal: handle, deviceToken: deviceTokenRef.current, imagePaths: pendingImages.map((attachment) => attachment.path), + followedByText: text.trim().length > 0, deadline, ...(seededLaunchDraft ? { clearInput: buildAgentTuiClearInputForText(seededLaunchDraft) } diff --git a/mobile/src/transport/client-context-connection-metrics.ts b/mobile/src/transport/client-context-connection-metrics.ts index 1e2ff6f7919..1fe72687bd2 100644 --- a/mobile/src/transport/client-context-connection-metrics.ts +++ b/mobile/src/transport/client-context-connection-metrics.ts @@ -30,14 +30,16 @@ export function useConnectionPathStatus(hostId: string | undefined): { export function useRelayRecoveryStatus(hostId: string | undefined): { pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean } { return useHostMetric( hostId, (context, id) => ({ pendingPath: context.getPendingPath(id), - pairingRejected: context.isPairingRejected(id) + pairingRejected: context.isPairingRejected(id), + hostSignedOut: context.isHostSignedOut(id) }), - { pendingPath: null, pairingRejected: false } + { pendingPath: null, pairingRejected: false, hostSignedOut: false } ) } diff --git a/mobile/src/transport/client-context.test.ts b/mobile/src/transport/client-context.test.ts index 4a7d5d7b0e8..56b227bdff7 100644 --- a/mobile/src/transport/client-context.test.ts +++ b/mobile/src/transport/client-context.test.ts @@ -533,12 +533,12 @@ describe('useAllHostClients', () => { await Promise.resolve() }) act(() => client.emitPendingPath('relay')) - expect(status).toEqual({ pendingPath: 'relay', pairingRejected: false }) + expect(status).toEqual({ pendingPath: 'relay', pairingRejected: false, hostSignedOut: false }) // Why: the desktop refusing the credential is a status-only change — no // transport state moves, so only the connection-path signal can carry it. act(() => client.emitPairingRejected(true)) - expect(status).toEqual({ pendingPath: 'relay', pairingRejected: true }) + expect(status).toEqual({ pendingPath: 'relay', pairingRejected: true, hostSignedOut: false }) act(() => renderer.unmount()) }) diff --git a/mobile/src/transport/connection-health.ts b/mobile/src/transport/connection-health.ts index 858b13a8b24..1a9e282047f 100644 --- a/mobile/src/transport/connection-health.ts +++ b/mobile/src/transport/connection-health.ts @@ -29,6 +29,10 @@ const STALE_SINCE_LAST_CONNECT_MS = 60_000 // instead of leaving the user staring at a generic "Can't connect". const TAILSCALE_HINT = 'check Tailscale' +// No hint field: the remedy is the label, and appending "— check Tailscale" to +// it would be wrong advice for a desktop that is reachable but signed out. +const SIGNED_OUT_LABEL = 'Desktop signed out — sign in to Orca on your desktop to reconnect' + export type ConnectionVerdict = | { kind: 'normal'; label: string } | { kind: 'warning'; label: string; hint?: string } // "Can't connect" @@ -54,6 +58,10 @@ export function classifyConnection(args: { // The desktop has repeatedly refused this device's relay credential — retrying // cannot fix it, so it outranks any "still connecting" reading (STA-4681). pairingRejected?: boolean + // The relay says the desktop's last control close named its own Orca Cloud + // sign-out. Retrying is still correct and still happens on the same cadence, + // but only the desktop's owner can end it, so the label has to say so. + hostSignedOut?: boolean nowMs?: number }): ConnectionVerdict { const { state, reconnectAttempts, lastConnectedAt } = args @@ -70,6 +78,17 @@ export function classifyConnection(args: { return { kind: 'normal', label: 'Connected' } } + // Ahead of the attempt thresholds: this is evidence, not an inference from a + // failure streak, and waiting twelve dials to show it wastes the whole point. + // Below auth-failed because a revoked pairing cannot be fixed by signing in. + if (args.hostSignedOut) { + return { + kind: 'unreachable', + label: SIGNED_OUT_LABEL, + reason: lastConnectedAt == null ? 'never-connected' : 'stale' + } + } + // A disconnected pending path can survive a cleared retry timer during a // lifecycle race. Only narrate Relay while dialing or after a retry has // recorded progress; otherwise the idle transport must read Disconnected. diff --git a/mobile/src/transport/host-client-context-state.ts b/mobile/src/transport/host-client-context-state.ts index 859e5d847f9..db4c7908b41 100644 --- a/mobile/src/transport/host-client-context-state.ts +++ b/mobile/src/transport/host-client-context-state.ts @@ -89,10 +89,16 @@ export function createHostClientSelectors( getPendingPath: (hostId: string): MobileConnectionPath | null => clientPendingPath(entries.get(hostId)?.client), isPairingRejected: (hostId: string): boolean => - clientPairingRejected(entries.get(hostId)?.client) + clientPairingRejected(entries.get(hostId)?.client), + isHostSignedOut: (hostId: string): boolean => clientHostSignedOut(entries.get(hostId)?.client) } } +export function clientHostSignedOut(client: RpcClient | undefined): boolean { + const logical = client as Partial | undefined + return logical?.isHostSignedOut?.() ?? false +} + export function clientPairingRejected(client: RpcClient | undefined): boolean { const logical = client as Partial | undefined return logical?.isPairingRejected?.() ?? false diff --git a/mobile/src/transport/logical-client-connection-path.ts b/mobile/src/transport/logical-client-connection-path.ts index 55c6d1d28f3..b7f02b40b88 100644 --- a/mobile/src/transport/logical-client-connection-path.ts +++ b/mobile/src/transport/logical-client-connection-path.ts @@ -5,6 +5,7 @@ export class LogicalClientConnectionPath { private recovery: MobileConnectionPath | null = null private recoveryAttempt = 0 private pairingRejected = false + private hostSignedOut = false private readonly listeners = new Set<() => void>() constructor(private readonly isConnected: () => boolean) {} @@ -35,12 +36,23 @@ export class LogicalClientConnectionPath { }) } + isHostSignedOut(): boolean { + return this.hostSignedOut + } + + setHostSignedOut(signedOut: boolean): void { + this.update(() => { + this.hostSignedOut = signedOut + }) + } + clearAfterConnected(): void { this.migration = null this.recovery = null this.recoveryAttempt = 0 // Why: an authenticated session is the desktop accepting this device. this.pairingRejected = false + this.hostSignedOut = false } setRecovery(path: MobileConnectionPath | null, attempt?: number): void { @@ -69,11 +81,13 @@ export class LogicalClientConnectionPath { const previousPath = this.pending() const previousAttempt = this.reconnectAttempt(0) const previousRejected = this.pairingRejected + const previousSignedOut = this.hostSignedOut apply() if ( previousPath === this.pending() && previousAttempt === this.reconnectAttempt(0) && - previousRejected === this.pairingRejected + previousRejected === this.pairingRejected && + previousSignedOut === this.hostSignedOut ) { return } diff --git a/mobile/src/transport/mobile-endpoint-lifecycle.ts b/mobile/src/transport/mobile-endpoint-lifecycle.ts index 8ee8df6948e..7ec5f28b945 100644 --- a/mobile/src/transport/mobile-endpoint-lifecycle.ts +++ b/mobile/src/transport/mobile-endpoint-lifecycle.ts @@ -86,7 +86,7 @@ function createSupervisor( ): MobileEndpointSupervisor { return new MobileEndpointSupervisor(logical, host, { openDirect: (endpoint) => connect(endpoint, host.deviceToken, host.publicKeyB64, { onLog }), - openRelay: (relay, credential, confirmReqId) => + openRelay: (relay, credential, confirmReqId, onHostCloseReason) => connectMobileRelayRpcSession({ relay, resumeToken: credential.token, @@ -94,6 +94,7 @@ function createSupervisor( resumeConfirmReqId: confirmReqId, deviceToken: host.deviceToken, desktopPublicKeyB64: host.publicKeyB64, + onHostCloseReason, onLog }), resolveRelay: resolveMobileRelayEndpoint, diff --git a/mobile/src/transport/mobile-endpoint-supervisor-contract.ts b/mobile/src/transport/mobile-endpoint-supervisor-contract.ts index 0098de6e079..2a784fd8895 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-contract.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-contract.ts @@ -1,4 +1,5 @@ import type { MobileRelayEndpoint } from '../../../src/shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../src/shared/relay-host-close-reason' import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bundle' import type { MobileRelayRpcSession } from './mobile-relay-rpc-session' import type { resolveMobileRelayEndpoint } from './mobile-relay-resume-director' @@ -10,7 +11,8 @@ export type MobileEndpointSupervisorDependencies = { openRelay: ( relay: MobileRelayEndpoint, credential: { token: string; version: number }, - confirmReqId: string + confirmReqId: string, + onHostCloseReason?: (reason: RelayHostCloseReason) => void ) => MobileRelayRpcSession resolveRelay: typeof resolveMobileRelayEndpoint readBundle: (hostId: string) => Promise diff --git a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts index 1dc1473d9db..80f4438c160 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts @@ -134,10 +134,22 @@ export class FakeLogicalClient extends FakeSession implements StableLogicalRpcCl } }) isPairingRejected = () => this.pairingRejected + private hostSignedOut = false + setHostSignedOut = vi.fn((signedOut: boolean) => { + if (this.hostSignedOut === signedOut) { + return + } + this.hostSignedOut = signedOut + for (const listener of this.pathListeners) { + listener() + } + }) + isHostSignedOut = () => this.hostSignedOut // Mirrors LogicalClientConnectionPath.clearAfterConnected. publishState(state: ConnectionState): void { if (state === 'connected') { this.pairingRejected = false + this.hostSignedOut = false } super.publishState(state) } diff --git a/mobile/src/transport/mobile-endpoint-supervisor.test.ts b/mobile/src/transport/mobile-endpoint-supervisor.test.ts index aeb9cddef63..10ef892a479 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor.test.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor.test.ts @@ -185,7 +185,12 @@ describe('mobile endpoint supervisor', () => { await supervisor.start() expect(deps.resolveRelay).toHaveBeenCalledOnce() - expect(openRelay).toHaveBeenLastCalledWith(resolved, expect.any(Object), expect.any(String)) + expect(openRelay).toHaveBeenLastCalledWith( + resolved, + expect.any(Object), + expect.any(String), + expect.any(Function) + ) expect(deps.saveHost).toHaveBeenCalledWith( expect.objectContaining({ relay: resolved, endpoint: host.endpoint }) ) @@ -556,7 +561,8 @@ describe('mobile endpoint supervisor', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) supervisor.stop() }) @@ -603,7 +609,8 @@ describe('mobile endpoint supervisor', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) supervisor.stop() }) diff --git a/mobile/src/transport/mobile-relay-e2ee-link.ts b/mobile/src/transport/mobile-relay-e2ee-link.ts index 7743deb23dd..9b1f7a9a355 100644 --- a/mobile/src/transport/mobile-relay-e2ee-link.ts +++ b/mobile/src/transport/mobile-relay-e2ee-link.ts @@ -2,6 +2,10 @@ import { RelayPhoneHelloSchema, type RelayPhoneHello } from '../../../src/shared/mobile-relay-phone-protocol' +import { + relayHostCloseReasonFrom, + type RelayHostCloseReason +} from '../../../src/shared/relay-host-close-reason' import { MobileE2EEV2ClientSession } from './mobile-e2ee-v2-client-session' import { MobileE2EEV2PhysicalChannel } from './mobile-e2ee-v2-physical-channel' import { websocketPayloadToUint8 } from './websocket-payload-bytes' @@ -26,6 +30,12 @@ type MobileRelayE2eeLinkOptions = { onText: (plaintext: string) => void onBinary: (plaintext: Uint8Array) => void onHello?: (hello: Extract) => void + // The cell's account of why the desktop is absent, read off the close frame. + // Reported separately from onError because a rejection is delivered as both a + // relay-hello and a close, and which one the runtime dispatches first is not + // ordered — only the close carries the reason, and it must not be lost to + // that race. + onHostCloseReason?: (reason: RelayHostCloseReason) => void // Fired once relay-auth is on the wire: from here the cell owns the wait. onOpen?: () => void onError: (error: Error) => void @@ -129,6 +139,11 @@ export class MobileRelayE2eeLink { clearTimeout(this.transportErrorTimer) this.transportErrorTimer = null } + // Ahead of fail(), which no-ops once the hello already reported this close. + const hostCloseReason = relayHostCloseReasonFrom(event.reason) + if (hostCloseReason) { + this.options.onHostCloseReason?.(hostCloseReason) + } this.fail(new RelayOuterError(event.code || 1006)) } } diff --git a/mobile/src/transport/mobile-relay-rpc-session.ts b/mobile/src/transport/mobile-relay-rpc-session.ts index 947a1d23ce8..203a0329192 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.ts @@ -13,6 +13,7 @@ import { RelayDialStageTracker, type RelayDialStageSource } from './relay-dial-s import { RelayPendingRequests } from './relay-pending-requests' import { RpcSessionLivenessWatchdog } from './rpc-session-liveness-watchdog' import { settleMobileRuntimeCapabilities } from './mobile-runtime-capability-negotiation' +import type { RelayHostCloseReason } from '../../../src/shared/relay-host-close-reason' import type { RpcClient } from './rpc-client' import type { ConnectionLogSink, ConnectionState, RpcResponse } from './types' @@ -40,6 +41,7 @@ export function connectMobileRelayRpcSession(args: { desktopPublicKeyB64: string requestTimeoutMs?: number createSocket?: (url: string) => WebSocket + onHostCloseReason?: (reason: RelayHostCloseReason) => void onLog?: ConnectionLogSink }): MobileRelayRpcSession { const requestTimeoutMs = args.requestTimeoutMs ?? 30_000 @@ -69,6 +71,7 @@ export function connectMobileRelayRpcSession(args: { deviceToken: args.deviceToken, desktopPublicKeyB64: args.desktopPublicKeyB64, createSocket: args.createSocket, + onHostCloseReason: args.onHostCloseReason, onOpen: () => dialStage.advance('awaiting-hello'), onHello: (hello) => { if ( diff --git a/mobile/src/transport/mobile-relay-runtime-failover.test.ts b/mobile/src/transport/mobile-relay-runtime-failover.test.ts index 01f4d45feb0..ce7cca3fd9f 100644 --- a/mobile/src/transport/mobile-relay-runtime-failover.test.ts +++ b/mobile/src/transport/mobile-relay-runtime-failover.test.ts @@ -145,10 +145,22 @@ class FakeLogicalClient extends FakeSession implements StableLogicalRpcClient { } }) isPairingRejected = () => this.pairingRejected + private hostSignedOut = false + setHostSignedOut = vi.fn((signedOut: boolean) => { + if (this.hostSignedOut === signedOut) { + return + } + this.hostSignedOut = signedOut + for (const listener of this.pathListeners) { + listener() + } + }) + isHostSignedOut = () => this.hostSignedOut // Mirrors LogicalClientConnectionPath.clearAfterConnected. publishState(state: ConnectionState): void { if (state === 'connected') { this.pairingRejected = false + this.hostSignedOut = false } super.publishState(state) } @@ -264,7 +276,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 3 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() @@ -353,7 +366,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(deps.openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 2 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() @@ -382,7 +396,8 @@ describe('relay runtime recovery without direct connectivity', () => { expect(openRelay).toHaveBeenLastCalledWith( relay, expect.objectContaining({ version: 1 }), - expect.any(String) + expect.any(String), + expect.any(Function) ) expect(logical.getActivePath()).toBe('relay') supervisor.stop() diff --git a/mobile/src/transport/mobile-relay-session-establisher.ts b/mobile/src/transport/mobile-relay-session-establisher.ts index 7a8ce372155..9a04ae44137 100644 --- a/mobile/src/transport/mobile-relay-session-establisher.ts +++ b/mobile/src/transport/mobile-relay-session-establisher.ts @@ -10,6 +10,7 @@ import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bund import type { RelayReconnectController } from './mobile-relay-reconnect-controller' import type { StableLogicalRpcClient } from './stable-logical-rpc-client' import type { MobileRelayEndpoint } from '../../../src/shared/mobile-relay-credential-contract' +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' import type { HostProfile } from './types' type EstablishResult = { ok: true } | { ok: false; error: Error } @@ -100,7 +101,16 @@ export class MobileRelaySessionEstablisher { const session = args.openRelay( relay, credential, - `confirm-${encodeBase64Url(args.randomBytes(16))}` + `confirm-${encodeBase64Url(args.randomBytes(16))}`, + // Latched on the logical client, not on the dial result: the close that + // carries the reason can land after this dial has already reported its + // failure. Clearing is clearAfterConnected's job, so any path that + // reaches connected retires it. + (reason) => { + if (reason === RELAY_HOST_CLOSE_REASON.SIGNED_OUT) { + args.logical.setHostSignedOut(true) + } + } ) try { // Why: backgrounding or a direct winner withdraws this dial before cutover. diff --git a/mobile/src/transport/relay-host-signed-out-supervisor.test.ts b/mobile/src/transport/relay-host-signed-out-supervisor.test.ts new file mode 100644 index 00000000000..cc7a9ac6c35 --- /dev/null +++ b/mobile/src/transport/relay-host-signed-out-supervisor.test.ts @@ -0,0 +1,60 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { MOBILE_RELAY_CLOSE_CODE } from '../../../src/shared/mobile-relay-close-codes' +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' +import { RelayOuterError } from './mobile-relay-e2ee-link' +import { + dependencies, + FakeLogicalClient, + FakeRelaySession, + host +} from './mobile-endpoint-supervisor-test-fakes' +import { MobileEndpointSupervisor } from './mobile-endpoint-supervisor' + +vi.mock('react-native', () => ({ Platform: { OS: 'ios' } })) +vi.mock('expo-secure-store', () => ({ WHEN_UNLOCKED_THIS_DEVICE_ONLY: 'when-unlocked' })) +vi.mock('expo-crypto', () => ({ getRandomBytes: (length: number) => new Uint8Array(length) })) + +// The reason travels from the cell's close frame to the screens. This covers +// the production wiring between them: the supervisor's own openRelay callback. +describe('a signed-out desktop reaches the phone verdict', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-07-13T12:00:00Z')) + }) + afterEach(() => vi.useRealTimers()) + + function supervisorOver(closeReason: string | null) { + const logical = new FakeLogicalClient('disconnected', 'lan') + const deps = dependencies({ + openDirect: vi.fn(() => new FakeRelaySession('disconnected')), + openRelay: vi.fn((_relay, _credential, _confirmReqId, onHostCloseReason) => { + if (closeReason) { + onHostCloseReason?.(closeReason as never) + } + return new FakeRelaySession( + 'disconnected', + new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE) + ) + }) + }) + return { logical, supervisor: new MobileEndpointSupervisor(logical, host, deps) } + } + + it('latches the sign-out the cell reported', async () => { + const { logical, supervisor } = supervisorOver(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + await supervisor.start() + await vi.waitFor(() => expect(logical.isHostSignedOut()).toBe(true)) + + supervisor.stop() + }) + + it('stays quiet for an ordinary host-offline rejection', async () => { + const { logical, supervisor } = supervisorOver(null) + + await supervisor.start() + + expect(logical.isHostSignedOut()).toBe(false) + supervisor.stop() + }) +}) diff --git a/mobile/src/transport/relay-host-signed-out-verdict.test.ts b/mobile/src/transport/relay-host-signed-out-verdict.test.ts new file mode 100644 index 00000000000..2607b922b58 --- /dev/null +++ b/mobile/src/transport/relay-host-signed-out-verdict.test.ts @@ -0,0 +1,216 @@ +import { describe, expect, it, vi } from 'vitest' + +vi.mock('./mobile-e2ee-v2-client-session', () => ({ + MobileE2EEV2ClientSession: { create: () => ({}) } +})) + +vi.mock('./mobile-e2ee-v2-physical-channel', () => ({ + MobileE2EEAuthenticationError: class extends Error {}, + MobileE2EEV2PhysicalChannel: class { + start = vi.fn() + handleMessage = vi.fn(async () => {}) + sendText = vi.fn(() => true) + sendBinary = vi.fn(() => true) + dispose = vi.fn() + } +})) + +import { RELAY_HOST_CLOSE_REASON } from '../../../src/shared/relay-host-close-reason' +import { MOBILE_RELAY_CLOSE_CODE } from '../../../src/shared/mobile-relay-close-codes' +import { classifyConnection, verdictDisplayLabel } from './connection-health' +import { MobileRelayE2eeLink, RelayOuterError } from './mobile-relay-e2ee-link' +import { LogicalClientConnectionPath } from './logical-client-connection-path' +import { RelayReconnectController } from './mobile-relay-reconnect-controller' + +const SIGNED_OUT_LABEL = 'Desktop signed out — sign in to Orca on your desktop to reconnect' + +class FakeSocket { + static readonly OPEN = 1 + readonly OPEN = FakeSocket.OPEN + readyState = FakeSocket.OPEN + bufferedAmount = 0 + onopen: (() => void) | null = null + onmessage: ((event: { data: unknown }) => void) | null = null + onerror: (() => void) | null = null + onclose: ((event: { code: number; reason: string }) => void) | null = null + send = vi.fn() + close = vi.fn() +} + +function linkOver( + socket: FakeSocket, + onHostCloseReason: (reason: string) => void, + onError: (error: Error) => void +): MobileRelayE2eeLink { + return new MobileRelayE2eeLink({ + endpoint: { cellUrl: 'https://relay-c1.onorca.dev', relayHostId: 'AbCdEf0123_-xyZ9' }, + credential: 'credential', + expectedCredentialKind: 'resume', + deviceToken: 'device-token', + desktopPublicKeyB64: 'desktop-key', + onAuthenticated: vi.fn(), + onText: vi.fn(), + onBinary: vi.fn(), + onHostCloseReason, + onError, + createSocket: () => socket as unknown as WebSocket + }) +} + +describe('relay close reason on the phone', () => { + it('reports the cell close reason and still fails with 4404', () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + const onError = vi.fn() + linkOver(socket, onHostCloseReason, onError) + + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + + expect(onHostCloseReason).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + expect(onError).toHaveBeenCalledWith(new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE)) + }) + + // An old cell sends its constant, and every other close sends nothing. + it('reports nothing for a reason it does not know', () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + linkOver(socket, onHostCloseReason, vi.fn()) + + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: 'relay connection rejected' + }) + + expect(onHostCloseReason).not.toHaveBeenCalled() + }) + + // The rejection arrives as a relay-hello AND a close, in an unordered pair. + // Whichever lands first, the reason must survive. + it('still reports the reason when the hello already failed the link', async () => { + const socket = new FakeSocket() + const onHostCloseReason = vi.fn() + linkOver(socket, onHostCloseReason, vi.fn()) + + socket.onmessage?.({ + data: JSON.stringify({ + type: 'relay-hello', + ok: false, + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE + }) + }) + await Promise.resolve() + await Promise.resolve() + socket.onclose?.({ + code: MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + + expect(onHostCloseReason).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) +}) + +describe('the signed-out signal on the logical client', () => { + it('publishes on change and retires when any path reaches connected', () => { + const path = new LogicalClientConnectionPath(() => false) + const changes = vi.fn() + path.subscribe(changes) + + path.setHostSignedOut(true) + path.setHostSignedOut(true) + expect(path.isHostSignedOut()).toBe(true) + expect(changes).toHaveBeenCalledTimes(1) + + path.clearAfterConnected() + expect(path.isHostSignedOut()).toBe(false) + }) +}) + +describe('RelayReconnectController cadence', () => { + // The reason changes no recovery decision; 4404 keeps the host-offline + // backoff it has always had, so a phone on this build retries exactly as + // often as one that never hears the reason. + it('keeps the host-offline retry delay for a 4404', () => { + const delays: number[] = [] + const controller = new RelayReconnectController( + { + now: () => 0, + randomBytes: () => new Uint8Array([0, 0]), + setTimer: ((callback: () => void, delay: number) => { + delays.push(delay) + return 1 as unknown as ReturnType + }) as unknown as typeof setTimeout, + clearTimer: (() => {}) as unknown as typeof clearTimeout + }, + vi.fn() + ) + + controller.registerFailure(new RelayOuterError(MOBILE_RELAY_CLOSE_CODE.HOST_OFFLINE)) + + // hostOfflineDelayMs' 5s floor, not the 250ms transport-backoff floor. + expect(delays.at(-1)).toBe(5_000) + }) +}) + +describe('classifyConnection with a signed-out desktop', () => { + const base = { reconnectAttempts: 0, lastConnectedAt: null, hostSignedOut: true } + + it('says so from the first failed dial instead of "Connecting via Relay…"', () => { + const verdict = classifyConnection({ + ...base, + state: 'connecting', + pendingPath: 'relay' + }) + + expect(verdict).toEqual({ + kind: 'unreachable', + label: SIGNED_OUT_LABEL, + reason: 'never-connected' + }) + expect(verdictDisplayLabel(verdict)).toBe(SIGNED_OUT_LABEL) + }) + + it('replaces "Can\'t reach desktop" on the direct path too', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', reconnectAttempts: 20 }).label + ).toBe(SIGNED_OUT_LABEL) + }) + + it('reads as stale once this session had been connected', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', lastConnectedAt: 1, nowMs: 2 }).reason + ).toBe('stale') + }) + + // A Tailscale endpoint cannot make "sign in on your desktop" better advice. + it('never appends the Tailscale hint', () => { + expect( + classifyConnection({ ...base, state: 'reconnecting', endpoint: '100.64.0.1' }) + ).not.toHaveProperty('hint') + }) + + it('never outranks a connected session', () => { + expect(classifyConnection({ ...base, state: 'connected' }).label).toBe('Connected') + }) + + // Re-pairing, not signing in, is the remedy when the pairing itself is dead. + it('never outranks a revoked pairing', () => { + expect(classifyConnection({ ...base, state: 'reconnecting', pairingRejected: true }).kind).toBe( + 'auth-failed' + ) + }) + + it('leaves every other verdict alone when the desktop is not signed out', () => { + expect( + classifyConnection({ + state: 'connecting', + reconnectAttempts: 0, + lastConnectedAt: null, + pendingPath: 'relay', + hostSignedOut: false + }).label + ).toBe('Connecting via Relay…') + }) +}) diff --git a/mobile/src/transport/rpc-client-context-contract.ts b/mobile/src/transport/rpc-client-context-contract.ts index 54e25973c7f..65262a6fc15 100644 --- a/mobile/src/transport/rpc-client-context-contract.ts +++ b/mobile/src/transport/rpc-client-context-contract.ts @@ -24,6 +24,7 @@ export type RpcClientContextValue = { getActivePath: (hostId: string) => MobileConnectionPath getPendingPath: (hostId: string) => MobileConnectionPath | null isPairingRejected: (hostId: string) => boolean + isHostSignedOut: (hostId: string) => boolean subscribeHostState: (hostId: string, listener: (state: ConnectionState) => void) => () => void getAllClients: () => { hostId: string; client: RpcClient }[] subscribeAllHosts: (listener: () => void) => () => void diff --git a/mobile/src/transport/stable-logical-rpc-client.ts b/mobile/src/transport/stable-logical-rpc-client.ts index d1f1701aed9..fb514128382 100644 --- a/mobile/src/transport/stable-logical-rpc-client.ts +++ b/mobile/src/transport/stable-logical-rpc-client.ts @@ -55,6 +55,9 @@ export type StableLogicalRpcClient = RpcClient & { // Latched when the desktop has repeatedly refused this device's relay credential. setPairingRejected(rejected: boolean): void isPairingRejected(): boolean + // Latched when the relay named the desktop's own sign-out as the reason it is absent. + setHostSignedOut(signedOut: boolean): void + isHostSignedOut(): boolean // Recovery attempts share this signal so status-only changes rerender. onConnectionPathChange(listener: () => void): () => void getGeneration(): number @@ -282,6 +285,8 @@ export function createStableLogicalRpcClient( setRecoveryAttempt: (attempt) => connectionPath.setRecoveryAttempt(attempt), setPairingRejected: (rejected) => connectionPath.setPairingRejected(rejected), isPairingRejected: () => connectionPath.isPairingRejected(), + setHostSignedOut: (signedOut) => connectionPath.setHostSignedOut(signedOut), + isHostSignedOut: () => connectionPath.isHostSignedOut(), onConnectionPathChange: (listener) => connectionPath.subscribe(listener), getGeneration: () => generation } diff --git a/mobile/src/transport/use-all-host-clients.ts b/mobile/src/transport/use-all-host-clients.ts index 03ac5890015..70c709d965f 100644 --- a/mobile/src/transport/use-all-host-clients.ts +++ b/mobile/src/transport/use-all-host-clients.ts @@ -138,6 +138,7 @@ export function useAllHostClients(hostIds: string[], options?: UseAllHostClients path: MobileConnectionPath pendingPath: MobileConnectionPath | null pairingRejected: boolean + hostSignedOut: boolean }>((hostId) => { const client = clientsByHostId.get(hostId) return client @@ -148,7 +149,8 @@ export function useAllHostClients(hostIds: string[], options?: UseAllHostClients state: ctx.getState(hostId), path: ctx.getActivePath(hostId), pendingPath: ctx.getPendingPath(hostId), - pairingRejected: ctx.isPairingRejected(hostId) + pairingRejected: ctx.isPairingRejected(hostId), + hostSignedOut: ctx.isHostSignedOut(hostId) } ] : [] diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 0b72bb101ba..6b59d23e026 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -116,7 +116,7 @@ patchedDependencies: '@xterm/addon-webgl@0.20.0-beta.299': 94687e89a0115e6e6aa102837f986debdc029c091527ee5eb4a4e17ceaf9473e '@xterm/xterm@6.1.0-beta.303': 98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d lint-staged@16.4.0: 7333b3837f80a7fbd045964db6d76ba4fc118e49134bdbabb00585b6b7b60673 - node-pty@1.1.0: e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa + node-pty@1.1.0: 7cc9d45f3d2c38f142490d0805e75db55f0eef5174ad41c4b52abc5fbe079ad1 importers: @@ -160,7 +160,7 @@ importers: version: 3.3.1 node-pty: specifier: ^1.1.0 - version: 1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa) + version: 1.1.0(patch_hash=7cc9d45f3d2c38f142490d0805e75db55f0eef5174ad41c4b52abc5fbe079ad1) posthog-node: specifier: ^5.33.3 version: 5.33.3 @@ -12285,7 +12285,7 @@ snapshots: node-int64@0.4.0: {} - node-pty@1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa): + node-pty@1.1.0(patch_hash=7cc9d45f3d2c38f142490d0805e75db55f0eef5174ad41c4b52abc5fbe079ad1): dependencies: node-addon-api: 7.1.1 diff --git a/resources/skills/current-manifest.json b/resources/skills/current-manifest.json index c8b204f00c5..fdb54016a8f 100644 --- a/resources/skills/current-manifest.json +++ b/resources/skills/current-manifest.json @@ -131,17 +131,17 @@ "name": "orchestration", "sourcePath": "skills/orchestration", "releaseRevision": 29, - "packageDigest": "7a386ce558ba54abe02b4a0de5d71fe3d63c944ef0888ddde130c729b37f7cc8", - "gitTreeSha": "4199ec6988801dd491631706cba62631b4bed8fb", + "packageDigest": "689e31d84256aded123c801eaa87413474943a9a30d96bff9a19d0a321aefb54", + "gitTreeSha": "902cc33dd65730b32ac234dd0ae7166d75498b46", "files": [ { "path": "SKILL.md", - "size": 4451, + "size": 4398, "executable": false, "classification": "text", - "exactSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "textNormalizedSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "identitySha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f" + "exactSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "textNormalizedSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "identitySha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18" } ] } diff --git a/resources/skills/snapshot-registry.json b/resources/skills/snapshot-registry.json index 5842b48858a..5b3412a497b 100644 --- a/resources/skills/snapshot-registry.json +++ b/resources/skills/snapshot-registry.json @@ -1046,17 +1046,17 @@ }, { "releaseRevision": 29, - "packageDigest": "7a386ce558ba54abe02b4a0de5d71fe3d63c944ef0888ddde130c729b37f7cc8", - "gitTreeSha": "4199ec6988801dd491631706cba62631b4bed8fb", + "packageDigest": "689e31d84256aded123c801eaa87413474943a9a30d96bff9a19d0a321aefb54", + "gitTreeSha": "902cc33dd65730b32ac234dd0ae7166d75498b46", "files": [ { "path": "SKILL.md", - "size": 4451, + "size": 4398, "executable": false, "classification": "text", - "exactSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "textNormalizedSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f", - "identitySha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f" + "exactSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "textNormalizedSha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18", + "identitySha256": "19ffdc1fe0d2c97dae845e8d636edb16781453ce2ec26f65a323c492ef90da18" } ] } diff --git a/skill-guides/orchestration.md b/skill-guides/orchestration.md index b9b7ca79442..0878532c447 100644 --- a/skill-guides/orchestration.md +++ b/skill-guides/orchestration.md @@ -3,18 +3,18 @@ name: orchestration description: >- Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation - waits, task DAGs, decision gates, coordinator loops, or decomposing work - across agents. Use `orca-cli` instead for full ownership handoffs, including - requests phrased as "hand off", "handoff", "handover", "give this to another - agent", or "another worktree" when the user did not explicitly ask to - supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for - terminal control, lightweight terminal prompts, shell commands, Orca - worktree management, reading or waiting on terminals, and automation of the - browser embedded inside Orca. Use Computer Use for external browser windows, - webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when - the task requires OS/window-level control such as focus, menus, dialogs, - coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a - page-automation tool such as Playwright or CDP for external pages. + waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli` + instead for full ownership handoffs, including requests phrased as "hand + off", "handoff", "handover", "give this to another agent", or "another + worktree" when the user did not explicitly ask to supervise, monitor, wait + for results, or coordinate a DAG. Use `orca-cli` for terminal control, + lightweight terminal prompts, shell commands, Orca worktree management, + reading or waiting on terminals, and the Orca embedded browser. Use Computer + Use for external browser windows, webviews, Orca app UI, or desktop UI + outside Orca's embedded browser only when the task requires OS/window-level + control such as focus, menus, dialogs, coordinates, or screenshots. Use + `orca-cli` for Orca's embedded pages and a page-automation tool such as + Playwright or CDP for external pages. --- # Orca Inter-Agent Orchestration diff --git a/skills/orchestration/SKILL.md b/skills/orchestration/SKILL.md index 85a0ff8c4b0..5725a8f5512 100644 --- a/skills/orchestration/SKILL.md +++ b/skills/orchestration/SKILL.md @@ -3,18 +3,18 @@ name: orchestration description: >- Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation - waits, task DAGs, decision gates, coordinator loops, or decomposing work - across agents. Use `orca-cli` instead for full ownership handoffs, including - requests phrased as "hand off", "handoff", "handover", "give this to another - agent", or "another worktree" when the user did not explicitly ask to - supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for - terminal control, lightweight terminal prompts, shell commands, Orca - worktree management, reading or waiting on terminals, and automation of the - browser embedded inside Orca. Use Computer Use for external browser windows, - webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when - the task requires OS/window-level control such as focus, menus, dialogs, - coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a - page-automation tool such as Playwright or CDP for external pages. + waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli` + instead for full ownership handoffs, including requests phrased as "hand + off", "handoff", "handover", "give this to another agent", or "another + worktree" when the user did not explicitly ask to supervise, monitor, wait + for results, or coordinate a DAG. Use `orca-cli` for terminal control, + lightweight terminal prompts, shell commands, Orca worktree management, + reading or waiting on terminals, and the Orca embedded browser. Use Computer + Use for external browser windows, webviews, Orca app UI, or desktop UI + outside Orca's embedded browser only when the task requires OS/window-level + control such as focus, menus, dialogs, coordinates, or screenshots. Use + `orca-cli` for Orca's embedded pages and a page-automation tool such as + Playwright or CDP for external pages. --- # Orca Orchestration diff --git a/src/cli/bundled-skill-guides.ts b/src/cli/bundled-skill-guides.ts index 66625fc9a1f..06aae7bdd7d 100644 --- a/src/cli/bundled-skill-guides.ts +++ b/src/cli/bundled-skill-guides.ts @@ -30,7 +30,7 @@ const ORCA_LINEAR_MARKDOWN = "---\nname: orca-linear\ndescription: >-\n Use Orc const ORCA_PER_WORKSPACE_ENV_MARKDOWN = "---\nname: orca-per-workspace-env\ndescription: >-\n Set up, review, debug, or validate Orca per-workspace environment recipes —\n on-demand, disposable runtimes (cloud sandboxes, VMs, or local) created fresh\n for each workspace. Covers first-time setup (provider prerequisites, the\n reusable base snapshot, the coding-agent auth snapshot, credentials, and\n state), not just the per-workspace lifecycle scripts. Use to stand up\n per-workspace environments, fix an `environmentRecipes` entry in `orca.yaml`, scaffold\n provider lifecycle scripts, or resolve an `orca vm recipe doctor` failure.\n---\n\n# Per-Workspace Environments\n\nHelp a user stand up and maintain a repo-owned per-workspace environment recipe end to end. Each\nworkspace gets its own on-demand, disposable runtime (a cloud sandbox, a VM, or a local one),\ncreated fresh and torn down after.\n\nOrca is a **thin wrapper**: you guide, detect, and scaffold; you never own the user's cloud account,\nbilling, images, or credentials.\n\n- **You DO:** sequence the setup, detect what's detectable (provider CLI present/logged-in? recipe\n present? `doctor` passing?), scaffold provider-templated scripts the user fills in, drive the slow\n snapshot/auth phases with the user, and always show the next action.\n- **You DO NOT:** create accounts, choose plans/regions, invent org/project/scope ids, store or print\n secrets, or run anything that spends money without an explicit user OK.\n\nFirst-time setup has **four phases before the per-workspace recipe runs** — easy to miss, so walk\nthem in order:\n\n1. **Prerequisites** — cloud account, provider CLI, scope/project, plan limits, git token (§2).\n2. **Base snapshot** — reusable image: tools + repo + headless build, snapshotted once (§3).\n3. **Agent-auth snapshot** — boot the base, run interactive device-auth, re-snapshot (§4).\n4. **State** — thread snapshot id / scope / project / port between phases via a state file (§6).\n\nThen the **per-workspace contract** (create/suspend/resume/destroy) runs fast (§8).\n\n**The one branch that shapes everything — connection mode:** **Orca-server** (`create` runs `orca serve`\nin the env and emits a `pairingCode`; §7c/§7f) vs **SSH** (`create` runs no server and emits a\n`connection.type:\"ssh\"` block Orca dials into; §7g/§7h). Settle this first — it changes the `create`\noutput shape and half the templates.\n\nKeep Orca's checkout behavior unchanged by default: omit `checkoutMode`, emit schema version 1, and\nlet Orca create a linked worktree. Only use `checkoutMode: provisioned-root` when the user explicitly\nwants one ephemeral machine to clone the finished workspace itself. This niche mode currently requires\ndirect SSH, an ordinary non-bare/non-sparse primary checkout at `projectRoot`, and schema version 2.\n\n**Quick-start (happy path):** interview the user (connection mode Orca-server vs SSH, provider, agent CLI,\ngit auth — §1.2) + read the provider's CLI docs → scaffold `scripts/orca-vm/` from §7 → run the\nbase-snapshot script, then the auth script (you invoke these by hand; not via `orca.yaml`) → wire\n`environmentRecipes` in `orca.yaml` → `orca vm recipe doctor --json` (free) → then the `--provision`\nself-test loop (§9) until it passes.\n\n---\n\n## 1. Setup workflow\n\nDrive these with the user. **[CHECKPOINT]** steps need explicit confirmation — they spend money, take\na long time, or need the user at the keyboard. Never create an Orca workspace or commit unless asked.\n\n1. **Inspect the repo** for an existing `environmentRecipes` entry, `scripts/orca-vm/`, a state file, or setup\n notes. If a working recipe exists, jump to Doctor (§9) instead of rebuilding.\n2. **Interview the user up front** — gather these choices and confirm them back before scaffolding\n anything. Don't pick for them (§11); don't guess.\n - **Connection mode:** how Orca attaches to the environment — an **Orca server** (the VM runs\n `orca serve` and Orca pairs over its pairing URL; worked example §7f) or **SSH** (Orca connects to\n the host over SSH; §7g). This decides the recipe's connection shape, so settle it first.\n - **Checkout ownership:** do not ask by default. Only when the user requires the environment to\n create the exact final checkout, confirm `provisioned-root` and direct SSH; otherwise omit it.\n - **Provider:** Vercel Sandbox, Fly, Modal, an existing SSH host, … For non-obvious providers, also\n ask scope/project/region and plan limits (§2). Then **read that provider's CLI/SDK docs** (or\n ` --help`) before scaffolding — you need its exact create/exec/snapshot/remove verbs.\n If a provider advertises `ssh`, verify whether it exposes a real dialable SSH target\n (host/port/user/key or proxy command) or only a provider-mediated interactive shell; Orca SSH mode\n needs the former.\n - **Coding-agent CLI + account:** which agent runs in the VM (`codex`, `claude`, …) and that the user\n has an account for it — it gets logged in during the Phase-3 auth snapshot (§4).\n - **Git auth:** the token source for cloning a private repo (`GH_TOKEN`/`GITHUB_TOKEN` or `gh auth\ntoken`; §5).\n3. **Check prerequisites (§2)** — detect the provider CLI + auth and confirm the items above are in\n place before any paid step.\n4. **Scaffold scripts + state file** from §7 (worked Vercel example: §7f; SSH host: §7g; Docker SSH:\n §7h; Windows: §7i), filling in the provider's real commands. Make them executable.\n5. **[CHECKPOINT] Build the base snapshot (§3)** — paid, slow.\n6. **[CHECKPOINT] Authenticate the agent (§4)** — interactive; the user follows a URL/code. **You cannot\n drive this step** — you run commands non-interactively, so there's no TTY for `docker exec -it` /\n `ssh -t` to prompt against. The **user** runs the Phase-3 login in their own terminal (or via the\n Claude Code harness bang-prefix — `! `, with the required space after `!`); you scaffold and drive\n the non-interactive phases around it. After kicking it off, **ask the user to report back once the login\n finishes** — you can't observe it completing, and you need that confirmation before resuming the\n non-interactive steps (base/auth commit, doctor, provision).\n7. **Wire the recipe** so `orca.yaml` points create/suspend/resume/destroy at the scripts (§8). The\n workspace composer reads `environmentRecipes` from the project's primary checkout of `orca.yaml`, **not** from\n a feature branch or worktree. So a recipe added only on a branch won't appear as a \"Run on\" option\n until that `orca.yaml` change is committed and merged to the project's primary branch. Tell the user\n this up front: `doctor`/`--provision` validate the scripts from the working copy on any branch, but\n creating a workspace from the recipe in the picker needs it on primary.\n8. **Dry-run doctor** — `orca vm recipe doctor --repo-path --json` (free, static; §9).\n Fix every failure before going live.\n9. **[CHECKPOINT] Live self-test** — get the user's OK once, then run\n `orca vm recipe doctor --provision --json` as a loop: it runs create → validates →\n destroys, and on failure returns a full transcript. Read it, fix the scripts, and re-run yourself until\n it passes (§9). Spends cloud money; the one approval covers the loop.\n10. **[CHECKPOINT] Optional workspace test** — only if asked: create a workspace via the picker, then\n verify sleep/wake/delete.\n\n---\n\n## 2. Phase 1 — Prerequisites\n\nThe user's responsibility; verify what's verifiable, ask for the rest, invent nothing. State which\nitems you verified vs. which the user asserted.\n\n- **Connection mode** (Orca server vs SSH) confirmed with the user — see §1 step 2; it shapes the recipe.\n- **Cloud account + plan** that allows sandboxes/VMs. Ask.\n- **Provider CLI installed + authenticated** — detect (`command -v `), check auth (e.g.\n `vercel whoami`). If missing, point at the provider's docs; don't log them in.\n- **Scope / project / region** the sandboxes live under. Ask; flows into every script via state.\n- **Plan / timeout / RAM caps.** Record them — e.g. Vercel Hobby caps sandbox timeout at **45m**,\n which limits both the base build and per-workspace runtime (see §10).\n- **Git token for private repos** (`GH_TOKEN`/`GITHUB_TOKEN`, or the provider's git auth; can fall back\n to `gh auth token`). See §5.\n- **Coding-agent CLI choice** (`codex`, `claude`…) and that the user has an account — it gets\n authenticated into the VM in Phase 3.\n\n---\n\n## 3. Phase 2 — Base snapshot (the reusable image)\n\nBuild **once**, snapshot, and every workspace boots from it in seconds instead of rebuilding.\nProvisioning + building takes a while (often ~20–30 min), so it runs behind a checkpoint. The script\nshape is §7a; key points:\n\n- Build the **headless Electron main only** (not the renderer) so it fits in plan RAM.\n- Use the VM image's package manager (`apt`/`dnf`/`apk`, per the base distro — not the provider brand).\n- Clone with the git token via `GIT_ASKPASS` (§5).\n- **Trap errors and remove the half-built sandbox** so a crash doesn't leave a paid resource running.\n- **Never snapshot a machine on which the Orca runtime has already run.** The first `orca serve` creates\n the runtime's user-data dir, and everything in it gets baked into the image and shared by every VM\n booted from it: the pairing keypair and device-token registry (`orca-devices.json`,\n `orca-e2ee-keypair.json`), `agent-session-authority.key`, and the build box's logs, terminal history\n and orchestration db. Confirmed: two VMs from one such snapshot emitted **identical `deviceToken` and\n `pairedDeviceId`**. Snapshot **before** the runtime has ever run, or delete the resolved user-data\n directory first: `orca_user_data_path=\"${ORCA_USER_DATA_PATH:-${XDG_CONFIG_HOME:-$HOME/.config}/orca}\"; rm -rf -- \"$orca_user_data_path\"`.\n This matches Orca's Linux precedence for custom and default paths; deleting a named file list will\n drift as Orca adds state.\n- Snapshot the stopped sandbox, parse the snapshot id, and write it + scope/project/port/repo to state.\n\n---\n\n## 4. Phase 3 — Agent-auth snapshot (interactive)\n\nThe base snapshot has the agent CLI installed but **not logged in**, and per-workspace VMs are\nephemeral — so authenticate once and bake it into a second snapshot layer. Script shape is §7b:\n\n1. Boot a sandbox from the base `snapshotId` (from state).\n2. Run the agent's login **interactively** (`--interactive --tty`); the user completes the URL/code in\n their browser. On a **headless VM this must be the device-auth flow** (e.g. `codex login --device-auth`),\n **not** plain `codex login`: the default OAuth login starts a loopback callback server on a container\n port the host browser can't reach, so it hangs. Device-auth instead prints a URL + code the user opens\n on the **host**.\n3. Verify login; **refuse to snapshot an unauthenticated VM.** Prefer the status command's **exit code**\n (most agent CLIs exit non-zero when unauthenticated). If you grep instead, agent status often goes to\n **stderr** (e.g. `codex login status` prints \"Logged in using ChatGPT\" there), so **fold stderr first**\n (`... 2>&1 | grep …`) and match the agent's **exact success line** — never `grep -qi 'logged in'`, which\n also matches \"**not** logged in\" and would commit an unauthenticated image.\n4. Re-snapshot, parse the new id, and overwrite `snapshotId` in state to the authenticated image\n (recording `authSourceSnapshotId`). Remove the auth sandbox.\n\n**You can't drive step 2 yourself** (you run commands non-interactively — no TTY). The **user** runs it in\ntheir own terminal, or via the Claude Code harness bang-prefix (`! `, with the required space after\n`!`). You scaffold/boot the sandbox and run steps 3–4, but **you cannot observe the interactive login\nfinishing** — so **ask the user to tell you when it's done** before you verify and re-snapshot.\n\nThis layer inherits §3's rule: if you started `orca serve` on the base or auth sandbox to smoke-test it,\ndelete the runtime's user-data dir (`~/.config/orca` on Linux) before re-snapshotting, or every workspace\nbooted from this image shares one pairing identity and one `agent-session-authority.key`.\n\nIf the agent's credentials are short-lived, warn that the snapshot may need periodic re-auth (§10).\n\nFor disposable runtimes, do **not** treat a host agent config directory (for example `~/.codex`) as the\nauth snapshot by bind-mounting or copying it wholesale. Agent homes often contain sqlite state, hook\napproval state, caches, logs, and host-specific env/config. Instead, authenticate/configure the agent\ninside the disposable runtime and snapshot/commit that runtime layer.\n\n---\n\n## 5. Credentials\n\n- **Never** commit secrets or put them in `userData`, recipe JSON, comments, docs, or the state file.\n- **Git token:** read from env (`GH_TOKEN`/`GITHUB_TOKEN`), falling back to `gh auth token`. Pass to the\n VM only via the provider's ephemeral `--env`. Inside the VM, use a `GIT_ASKPASS` helper with\n `x-access-token` (not the token in the clone URL) and `GIT_TERMINAL_PROMPT=0` so a missing token fails\n fast instead of hanging. When you write the helper from inside `bash -lc` under `set -u`, escape the\n positional arg and the token (`\\$1`, `\\$GH_TOKEN`) so they land **literally** and resolve at git-runtime\n — an unescaped `$1` aborts with \"unbound variable\", and a literal `$GH_TOKEN` keeps the real token out of\n the written file. `rm -f` the helper after the clone/fetch.\n- **Provider auth:** rely on the provider CLI's logged-in session, not checked-in keys.\n- **Agent auth:** lives in the authenticated snapshot (Phase 3) — never a file you write or commit.\n- State holds only **non-secret** wiring (snapshot ids, scope, project, port, repo url/ref).\n\n---\n\n## 6. State file\n\nA repo-local JSON file (e.g. `scripts/orca-vm/-state.json`) threads non-secret values between\nphases. Each script resolves values as **env var → state → built-in fallback**, and merges its outputs\nback. Phase 2 writes the base `snapshotId`; Phase 3 overwrites it with the authenticated snapshot;\nper-workspace `create` boots from `snapshotId`.\n\n```json\n{\n \"baseName\": \"orca-base\",\n \"snapshotId\": \"snap_authenticated_image_id\",\n \"authSourceSnapshotId\": \"snap_base_image_id\",\n \"scope\": \"\",\n \"project\": \"\",\n \"port\": 7331,\n \"repoUrl\": \"https://host/org/repo.git\",\n \"repoRef\": \"main\",\n \"projectRoot\": \"/abs/path/on/remote/repo\"\n}\n```\n\n---\n\n## 7. Script templates (provider-agnostic shapes)\n\nScaffold under `scripts/orca-vm/`. These are **shapes** — fill in the provider's real commands. All\nreserve stdout for the final JSON and log progress to stderr. Include a shared `json_value ` /\n`env_value ` reader (env → state → fallback) in each.\n\n**Where each script runs:**\n\n- **Local-side** (`create`/`suspend`/`resume`/`destroy` + the base-snapshot/auth scripts the user\n invokes) runs **on the user's desktop**, so it must run on their OS. macOS/Linux: `#!/usr/bin/env\nbash`, `set -euo pipefail`, quoted paths. **Windows:** a bare `.sh` won't run — scaffold `.ps1`/`.cmd`\n or require WSL/Git-Bash and point `orca.yaml` at the right launcher.\n- **Remote-side** (commands you `exec` _inside_ the Linux VM) always runs in the VM's Linux shell, so\n bash is fine there regardless of the user's OS.\n\n### 7a. Base-snapshot (`-base-snapshot.sh`) — Phase 2\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve base_name/repo_url/repo_ref/project_root/port/scope/project/timeout (env→state→fallback)\n# resolve gh token: GH_TOKEN | GITHUB_TOKEN | `gh auth token`\n# 1. provision a sandbox (timeout/vcpus/published port/snapshot retention); trap: remove on error\n# 2. remote exec (long timeout): install pkgs + gh + corepack/pnpm + agent CLI;\n# clone with GIT_ASKPASS(token); write headless main-only build config;\n# dev setup; pnpm install; build CLI; build headless electron main; smoke-check tools\n# 3. snapshot stopped sandbox; parse snapshot id (fail if unparseable)\n# 4. merge { baseName, snapshotId, projectRoot, repoUrl, repoRef, port, scope, project } into state\n# print only the state JSON to stdout\n```\n\nWorked Vercel commands for this phase are in §7f. You run this script by hand (not via `orca.yaml`),\nafter exporting the first-run inputs the state file doesn't have yet — e.g. provider scope/project, the\nrepo URL/ref, and a git token (`GH_TOKEN`); later runs read them back from state.\n\n### 7b. Auth (`-base-auth.sh`) — Phase 3\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read source snapshot from state.snapshotId (fail if absent); auth_name=\"${base_name}-auth\"\n# 1. boot sandbox from source snapshot; trap: remove on error\n# 2. INTERACTIVE/TTY remote exec: agent login — user completes URL/code. Headless VM: MUST use the\n# device-auth flow (e.g. `codex login --device-auth`) — plain OAuth login binds a loopback callback\n# port the host can't reach and hangs. User runs this themselves (you have no interactive TTY); ask\n# them to report back when it's done before continuing.\n# 3. verify login, then refuse to snapshot if not logged in. Prefer the status command's EXIT CODE (most\n# agent CLIs exit non-zero when unauthenticated) over string-matching. If you must grep, fold stderr\n# first (`status 2>&1 | grep …` — many agents print the success line there) and match the agent's exact\n# success line; never `grep -qi 'logged in'`, which also matches \"not logged in\". Codex example: §7f.\n# 4. snapshot; parse new id\n# 5. merge { snapshotId:, authSourceSnapshotId: } into state; remove auth sandbox\n# print only the state JSON to stdout\n```\n\n### 7c. Create (`-create.sh`) — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# read authenticated snapshotId/scope/project/port/repo*/project_root (env→state→fallback)\n# fail clearly if snapshotId is missing (point back to Phases 2–3)\n# name = orca-${ORCA_RECIPE_ID}-${ORCA_VM_INSTANCE_ID} (sanitized, length-capped)\n# 1. boot sandbox from snapshotId with a published port; capture the public URL → pairing address\n# (an externally reachable wss:// URL); trap: remove sandbox on error\n# 2. remote exec: ensure repo at desired commit; rebuild only if commit changed (cache marker)\n# 3. remote exec: start orca serve in the background and read the recipe JSON it writes (see below)\n# 4. print serve's JSON to stdout, optionally enriched with userData:\n# { schemaVersion:1, pairingCode, projectRoot, userData:{ provider, resourceId:name, snapshotId } }\n```\n\n**The exact `orca serve` invocation and its output (verified — do not improvise the flags).** Inside the\nVM, run:\n\n```bash\norca serve \\\n --port \"$PORT\" \\\n --project-root \"$ABS_REPO_PATH_ON_REMOTE\" \\\n --pairing-address \"$EXTERNAL_WSS_URL\" \\\n --recipe-json\n```\n\n**Binary name:** in a VM built from source (the Phase-2 flow), run it as `pnpm exec orca-dev serve …`\nfrom the repo root — `orca-dev` is the in-repo entrypoint and is what the §7f example uses. Plain\n`orca serve …` is the same command when the built CLI is installed on the VM's PATH. The flags/output\nare identical either way.\n\nThere is **no `--host` flag**. `--project-root` must be an absolute directory on the remote. With\n`--recipe-json` the server **stays running** and prints exactly this single object to **stdout**, then\nkeeps serving:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"\",\n \"projectRoot\": \"\"\n}\n```\n\n`pairingCode` is the pairing URL, already pointing at whatever you passed as `--pairing-address` — so set\n`--pairing-address` to the externally reachable address and **pass `pairingCode` through unchanged; never\nhand-rewrite it**. Because serve runs in the foreground and doesn't exit, redirect its stdout to a file\nand poll until that file parses as JSON (and bail if the process dies — dump its stderr log). Your\n`create` script then prints that JSON (optionally merging `userData`). Concrete pattern: §7f.\n\n### 7d. Suspend / resume / destroy — per workspace\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\npayload=\"$(cat)\" # Orca passes lifecycle JSON on stdin\nresource_id=\"$(node -e 'const d=JSON.parse(process.argv[1]); process.stdout.write(d.recipeResult?.userData?.resourceId ?? \"\")' \"$payload\")\"\n[ -n \"$resource_id\" ] || { echo \"No resource id in lifecycle payload\" >&2; exit 1; }\n# suspend: provider suspend \"$resource_id\"\n# resume: provider resume \"$resource_id\"; then RE-EMIT fresh recipe JSON (pairing may change)\n# destroy: provider remove \"$resource_id\" (or set destroy: none in orca.yaml)\n```\n\n### 7e. State file — scaffold with scope/project/repo filled in and snapshot ids empty (§6).\n\n### 7f. Worked example — Vercel Sandbox (all three phases)\n\nA real, working shape (the Vercel surface is a CLI: `vercel sandbox create|exec|snapshot|remove`). Adapt\nnames; verify flags against `vercel sandbox --help` for the user's CLI version before relying on them.\nThese ground §7a (base snapshot) and §7b (auth), which are otherwise generic skeletons.\n\n**Phase 2 — base snapshot (§7a):** provision → install tools + clone + headless build → snapshot.\n\n```bash\n# provision a fresh build sandbox (retain a couple of snapshots); trap-remove on error\nvercel sandbox create --name \"$base\" --runtime node24 --timeout 30m --vcpus 4 --publish-port \"$port\" \\\n --snapshot-expiration 30d --keep-last-snapshots 2 \"${vercel_args[@]}\" >&2\n# remote build (long timeout): install pkgs+gh+pnpm+agent CLI, clone with GIT_ASKPASS (write the helper\n# with LITERAL \\$1/\\$GH_TOKEN so they resolve at git-runtime, not write-time — see §5/§7f create — then\n# `rm -f /tmp/askpass.sh`), write the headless main-only build config (drop the renderer), dev setup,\n# build CLI + headless main, smoke-check\nvercel sandbox exec \"$base\" \"${vercel_args[@]}\" --timeout 25m --env \"GH_TOKEN=$gh_token\" … -- bash -lc '…build…' >&2\n# snapshot the STOPPED sandbox and parse the id from CLI output (fail if unparseable)\nout=\"$(vercel sandbox snapshot \"$base\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nsnapshot_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# merge { baseName, snapshotId, scope, project, port, repoUrl, repoRef, projectRoot } into state; print state JSON\n```\n\n**Phase 3 — agent-auth snapshot (§7b):** boot the base, log the agent in interactively, re-snapshot.\n(`codex` below is an example — substitute the user's chosen agent's login/status verbs, e.g. `claude`.)\n\n```bash\nvercel sandbox create --name \"$auth\" --snapshot \"$snapshot_id\" --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" >&2\n# INTERACTIVE — the USER runs this in their own terminal (you have no interactive TTY) and completes the\n# URL/code on the HOST. --device-auth is MANDATORY on a headless VM: plain `codex login` binds a loopback\n# callback port the host browser can't reach and hangs. Ask the user to report back when login finishes.\nvercel sandbox exec --interactive --tty \"$auth\" \"${vercel_args[@]}\" -- bash -lc 'codex login --device-auth'\n# refuse to snapshot an unauthenticated VM — fold stderr, match codex's exact success line (§4)\nvercel sandbox exec \"$auth\" \"${vercel_args[@]}\" --timeout 30s -- bash -lc 'codex login status 2>&1' | grep -Eqi 'Logged in using ChatGPT|Logged in via device' \\\n || { echo \"agent not logged in; not snapshotting\" >&2; exit 1; }\nout=\"$(vercel sandbox snapshot \"$auth\" --stop --expiration 30d \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$out\" >&2\nnew_id=\"$(printf '%s\\n' \"$out\" | sed -nE 's/.*(snap_[A-Za-z0-9]+).*/\\1/p' | tail -1)\"\n# overwrite state.snapshotId = new_id, record authSourceSnapshotId = snapshot_id; remove the auth sandbox\n```\n\n**Per-workspace `create`** (the fast path):\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback: snapshot_id, scope, project, port, repo_url, repo_ref, project_root\nvercel_args=(); [ -n \"$scope\" ] && vercel_args+=(--scope \"$scope\"); [ -n \"$project\" ] && vercel_args+=(--project \"$project\")\n[ -n \"$snapshot_id\" ] || { echo \"snapshotId missing — run Phases 2–3 first\" >&2; exit 1; }\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nrecipe_id=\"${ORCA_RECIPE_ID:-vercel-sandbox}\"\nrecipe_id=\"${recipe_id//./-}\" # Vercel names forbid dots.\ninstance_id=\"${ORCA_VM_INSTANCE_ID:-$(date +%s)}\"\nmax_recipe_id_length=$((128 - ${#instance_id} - 6)) # Preserve the unique instance suffix.\n[ \"$max_recipe_id_length\" -gt 0 ] || { echo \"ORCA_VM_INSTANCE_ID is too long for a Vercel sandbox name\" >&2; exit 1; }\nname=\"orca-${recipe_id:0:max_recipe_id_length}-${instance_id}\"\n\n# Arm cleanup BEFORE create so a failing create can't leak a half-built paid sandbox.\ncleanup_on_error() { [ \"$?\" -ne 0 ] && vercel sandbox remove \"$name\" \"${vercel_args[@]}\" >/dev/null 2>&1 || true; }\ntrap cleanup_on_error EXIT\n\n# 1. boot from the authenticated snapshot, publish the serve port\ncreate_output=\"$(vercel sandbox create --name \"$name\" --snapshot \"$snapshot_id\" \\\n --timeout 30m --publish-port \"$port\" \"${vercel_args[@]}\" 2>&1)\"; printf '%s\\n' \"$create_output\" >&2\n# Vercel prints the published https URL; derive the external wss:// pairing address from it\npublic_url=\"$(printf '%s\\n' \"$create_output\" | sed -nE 's#.*(https://[^[:space:]]+\\.vercel\\.run).*#\\1#p' | head -1)\"\n[ -n \"$public_url\" ] || { echo \"no published URL in create output\" >&2; exit 1; }\npairing_ws=\"${public_url/https:\\/\\//wss://}\"\n\n# 2. (remote) ensure the repo is at the right commit; rebuild only if the commit changed (cache marker)\nvercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 20m \\\n --env \"GH_TOKEN=$gh_token\" --env \"ORCA_PROJECT_ROOT=$project_root\" \\\n --env \"ORCA_REPO_URL=$repo_url\" --env \"ORCA_REPO_REF=$repo_ref\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; \\\n # Re-establish git auth for the private-repo fetch (why + full rationale: §5); else it hangs on a prompt.\n # Load-bearing escaping: \\$1 and \\$GH_TOKEN must land LITERALLY and resolve at git-runtime. Test after\n # any edit here — reformatting the nested printf/node quoting silently breaks the fetch or leaks the token.\n if [ -n \"${GH_TOKEN:-}\" ]; then \\\n printf \"%s\\n\" \"#!/usr/bin/env bash\" \"case \\\"\\$1\\\" in *Username*) echo x-access-token;; *Password*) echo \\\"\\$GH_TOKEN\\\";; esac\" > /tmp/askpass.sh; \\\n chmod 700 /tmp/askpass.sh; export GIT_ASKPASS=/tmp/askpass.sh GIT_TERMINAL_PROMPT=0; fi; \\\n git fetch origin \"$ORCA_REPO_REF\"; \\\n git checkout -B \"$ORCA_REPO_REF\" FETCH_HEAD; \\\n rm -f /tmp/askpass.sh; \\\n c=\"$(git rev-parse HEAD)\"; [ -f .orca-built ] && [ \"$(cat .orca-built)\" = \"$c\" ] || { \\\n pnpm install --prefer-offline && pnpm run build:cli && \\\n node config/scripts/run-electron-vite-build.mjs --config config/electron-vite.vm-serve.config.ts && \\\n printf \"%s\" \"$c\" > .orca-built; }' >&2\n\n# 3. (remote) start orca serve in the background, writing recipe JSON to a file; poll until it parses\nrecipe_json=\"$(vercel sandbox exec \"$name\" \"${vercel_args[@]}\" --timeout 60s \\\n --env \"ORCA_PORT=$port\" --env \"ORCA_PROJECT_ROOT=$project_root\" --env \"ORCA_PAIRING_ADDRESS=$pairing_ws\" \\\n -- bash -lc 'set -euo pipefail; cd \"$ORCA_PROJECT_ROOT\"; rm -f /tmp/orca-recipe.json /tmp/orca-serve.log; \\\n nohup pnpm exec orca-dev serve --port \"$ORCA_PORT\" --project-root \"$ORCA_PROJECT_ROOT\" \\\n --pairing-address \"$ORCA_PAIRING_ADDRESS\" --recipe-json >/tmp/orca-recipe.json 2>/tmp/orca-serve.log /dev/null 2>&1 && { cat /tmp/orca-recipe.json; exit 0; }; \\\n kill -0 \"$pid\" 2>/dev/null || { cat /tmp/orca-serve.log >&2; exit 1; }; sleep 0.25; \\\n done; cat /tmp/orca-serve.log >&2; echo \"serve recipe JSON timed out\" >&2; exit 1')\"\n\n# 4. print serve's JSON enriched with userData (single object on stdout)\nnode -e 'const p=JSON.parse(process.argv[1]); console.log(JSON.stringify({...p, schemaVersion:1,\n userData:{...p.userData, provider:\"vercel-sandbox\", resourceId:process.argv[2], snapshotId:process.argv[3]}}))' \\\n \"$recipe_json\" \"$name\" \"$snapshot_id\"\ntrap - EXIT\n```\n\n`suspend`/`resume`/`destroy` use `vercel sandbox stop|...|remove \"$resource_id\"` reading\n`userData.resourceId` from stdin (§7d). This is the **Orca-server** connection mode (the recipe emits a\npairing URL). If the user chose **SSH** in the §1 interview, use §7g instead.\n\n### 7g. Worked example — existing SSH host (SSH connection mode)\n\nSSH mode is **fundamentally different from §7c/§7f**, not a relabeling of them:\n\n- **`create` does NOT run `orca serve` and does NOT emit a `pairingCode`.** Orca itself connects to the\n host over its SSH relay, brings up the git + filesystem providers, and imports the repo. The script's\n only job is to make the host ready and **print SSH connection details** Orca will dial.\n- The result uses a `connection` block with `type: \"ssh\"` and a `target`, **not** the flat\n `pairingCode`/`projectRoot` shape. Exact shape (Orca rejects anything else):\n\n```json\n{\n \"schemaVersion\": 1,\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/path/to/repo/on/host\",\n \"target\": {\n \"label\": \"my-box\",\n \"host\": \"192.0.2.10\",\n \"port\": 22,\n \"username\": \"ubuntu\",\n \"identityFile\": \"~/.ssh/id_ed25519\",\n \"jumpHost\": \"bastion.example.com\",\n \"proxyCommand\": \"cloudflared access ssh --hostname %h\",\n \"relayGracePeriodSeconds\": 0,\n \"portForwards\": []\n }\n }\n}\n```\n\n`label`, `host`, `port`, `username` are required; the rest are optional — omit any you don't need.\n\nFor an explicitly requested one-VM-per-workspace checkout, the create script must read\n`ORCA_RECIPE_RESULT_SCHEMA_VERSION`, `ORCA_REPO_URL`, `ORCA_REPO_REF`, `ORCA_REPO_REF_HEAD`, and\n`ORCA_REPO_BRANCH`. Use `ORCA_REPO_REF` to fetch the selected source, but create\n`ORCA_REPO_BRANCH` at the exact `ORCA_REPO_REF_HEAD` commit; resolving the symbolic ref again can race\nwith an upstream update. `ORCA_REPO_URL` and `ORCA_REPO_REF` are a matched fetch pair, including when\nthe desktop source uses multiple remotes. Return that primary checkout at `projectRoot` and emit the\nsame SSH result with:\n\n```bash\n[ -n \"${ORCA_REPO_REF_HEAD:-}\" ] || { echo \"missing pinned source commit\" >&2; exit 1; }\ngit fetch origin \"$ORCA_REPO_REF\"\ngit cat-file -e \"${ORCA_REPO_REF_HEAD}^{commit}\"\ngit checkout -B \"$ORCA_REPO_BRANCH\" \"$ORCA_REPO_REF_HEAD\"\n```\n\n```json\n{\n \"schemaVersion\": 2,\n \"checkoutMode\": \"provisioned-root\",\n \"connection\": {\n \"type\": \"ssh\",\n \"projectRoot\": \"/abs/repo\",\n \"target\": { \"label\": \"my-box\", \"host\": \"192.0.2.10\", \"port\": 22, \"username\": \"ubuntu\" }\n }\n}\n```\n\nFail if the requested schema is not `2`; do not silently fall back to the ordinary recipe shape.\n\n**Networking → which `target` fields to set** (how _your desktop_ reaches the box — there is no\n`orca serve` URL in SSH mode):\n\n- Public IP / DNS, or a Tailscale/VPN address → `host`; SSH port → `port` (usually 22).\n- Key auth → `identityFile` (add `identitiesOnly: true` if the agent has many keys).\n- Through a bastion → `jumpHost` (a `user@host` ProxyJump) **or** a full `proxyCommand` (e.g. an access\n proxy). Use one, not both.\n- A service port the workspace needs → add entries to `portForwards`.\n- `relayGracePeriodSeconds` (optional): how long Orca keeps the SSH relay alive after the workspace\n detaches before tearing it down; `0` = tear down immediately. Leave it off unless the user wants a\n reconnect grace window.\n\n**Toolchain & agent auth on a persistent (no-snapshot) host — do this ONCE, by hand, before wiring the\nrecipe** (there's no base image to bake; the host _is_ the base). Run the §7f Phase-2 install steps and\nthe §7f Phase-3 ` login --device-auth` **directly over SSH on the host** (interactive, e.g.\n`ssh -t user@host ' login --device-auth'`). After that the host stays ready across workspaces.\n\n```bash\n#!/usr/bin/env bash\nset -euo pipefail\n# resolve from env→state→fallback (default unset optionals to \"\"): ssh_username, host,\n# ssh_port (default 22), identity_file, jump_host, proxy_command, project_root, repo_url, repo_ref\n: \"${identity_file:=}\"; : \"${jump_host:=}\"; : \"${proxy_command:=}\" # avoid set -u aborts on optionals\ngh_token=\"${GH_TOKEN:-${GITHUB_TOKEN:-$(command -v gh >/dev/null 2>&1 && gh auth token 2>/dev/null || true)}}\"\nssh_target=\"${ssh_username}@${host}\"\nssh_opts=(-p \"$ssh_port\"); [ -n \"$identity_file\" ] && ssh_opts+=(-i \"$identity_file\")\n# Why: a fresh host's key isn't in known_hosts; a StrictHostKeyChecking prompt would HANG a\n# non-interactive create. Pre-add the key (or set the option) so it can't block.\nssh-keyscan -p \"$ssh_port\" \"$host\" >> \"$HOME/.ssh/known_hosts\" 2>/dev/null || true\n\n# 1. ensure the repo is present and at the right commit on the host (NO orca serve here)\nssh \"${ssh_opts[@]}\" \"$ssh_target\" \\\n \"GH_TOKEN='$gh_token' GIT_TERMINAL_PROMPT=0 bash -lc '\n set -euo pipefail\n [ -d \\\"$project_root/.git\\\" ] || git clone \\\"$repo_url\\\" \\\"$project_root\\\"\n cd \\\"$project_root\\\" && git fetch origin \\\"$repo_ref\\\" && git checkout -B \\\"$repo_ref\\\" FETCH_HEAD\n '\" >&2\n\n# 2. print the SSH connection block (NO pairingCode, NO orca serve). host/port/username tell Orca's\n# relay how to dial in; identityFile/jumpHost/proxyCommand/portForwards are emitted when set.\nnode -e 'const [host,port,user,idf,jh,pc,root]=process.argv.slice(1);\n const target={ label:\"per-workspace-host\", host, port:Number(port), username:user };\n if(idf) target.identityFile=idf; if(jh) target.jumpHost=jh; if(pc) target.proxyCommand=pc;\n // add target.portForwards=[...] here if the workspace needs forwarded service ports\n console.log(JSON.stringify({ schemaVersion:1, connection:{ type:\"ssh\", projectRoot:root, target } }))' \\\n \"$host\" \"$ssh_port\" \"$ssh_username\" \"$identity_file\" \"$jump_host\" \"$proxy_command\" \"$project_root\"\n```\n\n`suspend`/`resume`/`destroy`: on a persistent host there's usually nothing to tear down — set\n`destroy: none` and omit suspend/resume. (Orca still disconnects/reconnects its own SSH relay on\nsleep/wake/delete — that's separate from these scripts.)\n\nIf the SSH host is instead an **ephemeral/snapshot-capable VM** (your hypervisor, or a cloud VM with\nimage support), keep the §7f Phase-2/3 base-image model for provisioning, but still emit the\n`connection.type:\"ssh\"` block above instead of starting `orca serve`.\n\n### 7h. Worked example — local Docker SSH (SSH connection mode)\n\nLocal Docker can model an ephemeral SSH VM without cloud cost: build a base image with `sshd`, tools,\nrepo prerequisites, and the agent CLI; run an **interactive auth container** once; then `docker commit`\nthat container as the authenticated image used by per-workspace `create`.\n\nKey points:\n\n- Publish container SSH to a random localhost port (`-p 127.0.0.1::22`) and emit\n `connection.type:\"ssh\"` with `host:\"127.0.0.1\"`, that port, `username`, `identityFile`, and\n `identitiesOnly:true`.\n- Generate a repo-local SSH key if needed, but gitignore the private/public key files.\n- **Bake SSH host keys into the base image** (`ssh-keygen -A` at **build** time; at runtime only generate\n if absent). Ephemeral containers all present the **same** host key, so `known_hosts` on `127.0.0.1`\n doesn't churn as the published port rotates across workspaces (otherwise every container's freshly\n generated key collides on `localhost` and trips host-key-changed warnings).\n- The auth image is the Docker equivalent of Phase 3: the **user** runs the agent login **inside** the\n container (you can't drive it — you have no interactive TTY), configures proxy env/config, approves\n hooks, and you commit once they report it's done. On a headless container use the **device-auth** flow\n (§4). Verify login before committing — exit code, or fold stderr and match the exact success line (§4).\n- Do not bind-mount or copy the host's full agent home into the image. Let each container have writable\n agent state; only the committed auth image should carry reusable authenticated state.\n- If committing from an interactive shell, force the runtime entrypoint back to `sshd`:\n `docker commit --change='ENTRYPOINT [\"/usr/local/bin/orca-docker-ssh-entrypoint\"]' …`.\n- `destroy` should read `recipeResult.userData.resourceId` and run `docker rm -f \"$resource_id\"`.\n\nValidation before wiring/live use:\n\n```bash\ndocker image inspect \"$auth_image\" --format '{{json .Config.Entrypoint}}'\ndocker run -d --name \"$name\" -p 127.0.0.1::22 -e \"ORCA_SSH_PUBLIC_KEY=$pubkey\" \"$auth_image\"\ndocker ps -a --filter \"name=$name\"\ndocker logs \"$name\"\nssh -i \"$key\" -p \"$port\" -o IdentitiesOnly=yes user@127.0.0.1 'codex --version'\n```\n\nIf the container exits immediately, inspect logs before the cleanup trap removes it; a committed\ninteractive image with `ENTRYPOINT [\"bash\"]` is a common cause.\n\nAlso confirm the **host key is stable** across containers: the SSH `ssh -i … 127.0.0.1` dial should not\ntrigger a host-key-changed warning when a second container reuses the port. If it does, the host keys\nweren't baked into the base image (see the `ssh-keygen -A` point above).\n\n### 7i. Windows local-side scripts\n\nThe local-side scripts run on the user's desktop. On **Windows**, a bare `.sh` won't execute. Either\nrequire WSL/Git-Bash (and point `orca.yaml` at e.g. `bash ./scripts/orca-vm/.sh` via a `.cmd`\nlauncher), or scaffold PowerShell equivalents. Minimal PowerShell shape:\n\n```powershell\n#requires -Version 5\n$ErrorActionPreference = 'Stop'\n# resolve env→state→fallback; run the provider CLI / ssh the same way;\n# capture provider output; build the result object for the chosen mode and write ONE line of JSON to stdout.\n# Orca-server mode: @{ schemaVersion=1; pairingCode=$pairingCode; projectRoot=$projectRoot; userData=@{...} }\n# SSH mode: @{ schemaVersion=1; connection=@{ type=\"ssh\"; projectRoot=$projectRoot;\n# target=@{ label=$label; host=$host; port=$port; username=$user } } } (see §7g/§7h)\n($result | ConvertTo-Json -Compress -Depth 6)\n# progress/errors → Write-Error / the error stream, never stdout.\n```\n\nThe remote-side commands you run _inside_ the Linux VM stay bash regardless of the desktop OS.\n\n---\n\n## 8. Per-workspace recipe contract (the fast path)\n\nOnce the authenticated snapshot exists, this runs on every workspace create. Define recipes in\n`orca.yaml`:\n\n```yaml\nenvironmentRecipes:\n - id: cloud-sandbox\n name: Cloud Sandbox\n create: ./scripts/orca-vm/cloud-sandbox-create.sh\n suspend: ./scripts/orca-vm/cloud-sandbox-suspend.sh\n resume: ./scripts/orca-vm/cloud-sandbox-resume.sh\n destroy: ./scripts/orca-vm/cloud-sandbox-destroy.sh\n```\n\n`create` runs **locally from the repo root** and prints **one** JSON object to stdout. Its shape depends\non the connection mode chosen in §1:\n\n**Orca-server mode** — boot the env, start `orca serve` in it, and print serve's result:\n\n```json\n{\n \"schemaVersion\": 1,\n \"pairingCode\": \"orca-pairing-code-or-url\",\n \"projectRoot\": \"/absolute/path/to/repo/on/remote\",\n \"userData\": { \"provider\": \"example\", \"resourceId\": \"provider-resource-id\" }\n}\n```\n\nHere `pairingCode` (from `orca serve --recipe-json`) and `projectRoot` are required; `schemaVersion` (`1`)\nand `userData` are optional.\n\n**SSH mode** — do **not** run `orca serve`; print the `connection.type:\"ssh\"` block instead (full shape +\nworked script in §7g). `pairingCode` is **not** used in SSH mode.\n\n**Optional provisioned root** — only for direct SSH and only when explicitly requested. Add\n`checkoutMode: provisioned-root` to the recipe, require `ORCA_RECIPE_RESULT_SCHEMA_VERSION=2`, create\nthe requested `ORCA_REPO_BRANCH` at the pinned `ORCA_REPO_REF_HEAD` commit (use `ORCA_REPO_REF` only\nto fetch that commit) at the returned `projectRoot`, and emit schema version 2 with\n`checkoutMode: \"provisioned-root\"`. All recipes without this field retain the schema-v1 behavior above.\n\nLifecycle hooks (all run locally):\n\n- `create`: required. Prints recipe result JSON.\n- `suspend`: optional. Sleep; reads lifecycle payload on stdin.\n- `resume`: optional. Wake; reads payload on stdin and **prints fresh recipe JSON** (pairing may change).\n- `destroy`: optional unless `destroy: none`. Delete/cleanup; reads payload on stdin.\n\nStart Orca remotely with `orca serve --port \"$PORT\" --project-root \"$ABS_ROOT\" --pairing-address\n\"$EXTERNAL_WSS_URL\" --recipe-json` (exact flags + output in §7c). Set `--pairing-address` to the\nexternally reachable address so the emitted `pairingCode` is reachable; tunneling/port mapping is the\nscript's job.\n\nBackward compatibility: `command`→`create`, `cleanup`→`destroy`, `cleanup: none`→`destroy: none`.\nPrefer the lifecycle names.\n\n---\n\n## 9. Doctor and validation\n\nValidate in two stages — the cheap dry run first, then the live self-test.\n\n### Dry run (free, non-destructive) — always do this first\n\n`orca vm recipe doctor --repo-path --json` validates **static wiring only** — it does\n**not** boot anything. It checks: local-host execution (v1), repo path, recipe id exists,\ncreate/destroy/suspend/resume command paths resolve, suspend/resume are paired, and each script is\nexecutable (POSIX exec bit; skipped on Windows). Fix every failure here before spending any cloud money.\n\n### Live self-test (`--provision`) — diagnose and iterate yourself\n\n`orca vm recipe doctor --repo-path --provision --json` actually runs the recipe end\nto end: it executes `create`, validates the returned recipe JSON, then runs `destroy` to **tear the\nenvironment back down** (so the test leaves nothing running, as long as `destroy` works). It spends real\ncloud money, so get the user's OK **once** before starting — that one approval covers the whole loop\nbelow; do not re-ask before each run.\n\nOn failure, the JSON result includes a `provisionTranscript` with the **complete** captured output of\neach stage so you can self-diagnose without asking the user to relay logs:\n\n```json\n{\n \"ok\": false,\n \"checks\": [{ \"id\": \"recipe.provision\", \"status\": \"fail\", \"message\": \"…\" }],\n \"provisionTranscript\": {\n \"provision\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\", \"parseError\": \"…\" },\n \"destroy\": { \"exitCode\": 0, \"signal\": null, \"stdout\": \"…\", \"stderr\": \"…\" }\n }\n}\n```\n\n**Run it as a loop:** read `provisionTranscript.provision.stderr` / `.stdout` / `.parseError` (and\n`destroy.*`), fix the script, and re-run `--provision` until `ok` is `true` — iterating on your own\nrather than waiting for the user to paste errors. Common reads: a non-empty `stderr` with `exitCode 0`\nplus a `parseError` means `create` ran but printed something other than the single recipe-result JSON on\nstdout (often a stray `echo` — route it to stderr, see §10); a non-zero `exitCode` is a provider/script\nfailure described in `stderr`. Each stream is redacted and capped (head+tail) — large logs keep both the\nsetup context and the failure.\n\nThe self-test cannot see provider-side truth beyond what the scripts print, so still confirm: state has a\npopulated **authenticated** `snapshotId` (Phases 2–3 done), and `destroy` is implemented/tested (or\nexplicitly `none` — in which case the self-test won't tear down, so clean up manually).\n\nFor SSH recipes, also smoke-test the exact emitted target before declaring success: dial the host/port\nwith the identity/proxy settings, run `pwd`, verify the repo path, check the agent binary, and confirm\n`destroy` removes the provider resource/container. For Docker, inspect the auth image entrypoint and do a\nstartup-only `docker run` before the full clone/install path.\n\n---\n\n## 10. Failure modes\n\n- **Build exceeds plan timeout (e.g. Hobby 45m).** Use enough vCPUs and a timeout covering the build;\n else split work or use a higher plan. The cap also limits per-workspace runtime — surface it.\n- **Build exceeds plan RAM.** Build the **headless main only** (drop the renderer) — the biggest fitter.\n- **Private-repo clone hangs/fails.** Wrong/missing token. Use `GIT_ASKPASS` + `GIT_TERMINAL_PROMPT=0`\n so it fails fast instead of prompting.\n- **`GIT_ASKPASS` helper aborts the clone with \"`$1: unbound variable`\".** The `printf`/heredoc that writes\n the helper inside `bash -lc` under `set -u` expanded `$1`/`$GH_TOKEN` at **write** time. Escape them\n (`\\$1`, `\\$GH_TOKEN`) so they land literally and resolve at git-runtime; this also keeps the real token\n out of the file. `rm -f` the helper afterward (§5, §7f).\n- **Agent verified as \"not logged in\" despite a good login.** `codex login status` (and similar) print\n \"Logged in …\" to **stderr**; an stdout-only `grep` misses it. Prefer the status **exit code**; if you\n grep, fold stderr first (`status 2>&1 | grep …`) and match the exact success line — not `grep -qi\n'logged in'`, which also matches \"not logged in\".\n- **Headless agent login hangs.** Plain OAuth `login` starts a loopback callback server on a VM/container\n port the host browser can't reach. Use the **device-auth** flow (`login --device-auth`) — it prints a\n URL + code the user opens on the host.\n- **`known_hosts` host-key churn on local Docker.** Each ephemeral container regenerating its SSH host key\n collides on `127.0.0.1` as the published port rotates. Bake host keys into the base image at build time\n (`ssh-keygen -A`; runtime generates only if absent) so all containers share one stable key (§7h).\n- **Snapshot expired/evicted.** If `create` hits an unknown snapshot id, rerun Phases 2–3 and update\n `snapshotId`.\n- **Agent auth didn't persist.** Confirm `snapshotId` points at the **authenticated** snapshot; re-run\n Phase 3. Warn that short-lived tokens may need periodic re-auth.\n- **Agent auth copied from the host breaks.** Do not bind-mount/copy a full host agent home; sqlite\n files can be unwritable or host-specific, hooks may need approval again, and config may reference\n local-only env vars. Authenticate inside the runtime and snapshot/commit that layer.\n- **Docker auth image exits immediately.** Inspect `docker image inspect … .Config.Entrypoint` and\n `docker logs`. If the image was committed from an interactive shell, reset the entrypoint to the SSH\n entrypoint during `docker commit`.\n- **Leaked paid resource.** Every long script must trap errors and remove the sandbox it created.\n- **`create` emits non-JSON on stdout.** A stray `echo` corrupts the result — stdout is for the final\n JSON only; everything else to stderr. The `--provision` self-test surfaces this as `exitCode 0` + a\n `parseError` with the offending stdout in `provisionTranscript` (§9).\n\n---\n\n## 11. Boundaries\n\n- Don't create accounts, choose plans/regions, or invent scope/project/org/image/billing ids.\n- Don't invent or store credentials; no secrets in `userData`, state, comments, docs, or commits.\n- Don't run paid/long phases (base snapshot, auth, live test) without an explicit OK.\n- Don't hide provider errors behind generic messages — preserve actionable stderr.\n- Don't make Orca own provider lifecycle beyond invoking the configured scripts.\n- Don't commit or create an Orca workspace unless asked.\n" // oxfmt-ignore -const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, coordinator loops, or decomposing work\n across agents. Use `orca-cli` instead for full ownership handoffs, including\n requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", or \"another worktree\" when the user did not explicitly ask to\n supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for\n terminal control, lightweight terminal prompts, shell commands, Orca\n worktree management, reading or waiting on terminals, and automation of the\n browser embedded inside Orca. Use Computer Use for external browser windows,\n webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when\n the task requires OS/window-level control such as focus, menus, dialogs,\n coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a\n page-automation tool such as Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" +const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Use Orca orchestration for structured multi-agent coordination: threaded\n messages, blocking ask/reply flows, task dispatch, worker_done/escalation\n waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli`\n instead for full ownership handoffs, including requests phrased as \"hand\n off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another\n worktree\" when the user did not explicitly ask to supervise, monitor, wait\n for results, or coordinate a DAG. Use `orca-cli` for terminal control,\n lightweight terminal prompts, shell commands, Orca worktree management,\n reading or waiting on terminals, and the Orca embedded browser. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI\n outside Orca's embedded browser only when the task requires OS/window-level\n control such as focus, menus, dialogs, coordinates, or screenshots. Use\n `orca-cli` for Orca's embedded pages and a page-automation tool such as\n Playwright or CDP for external pages.\n---\n\n# Orca Inter-Agent Orchestration\n\nOrchestration is Orca's structured coordination layer for agent messages, task ownership, dispatch state, and worker completion tracking.\n\nUse this skill when coordination state matters. For lightweight terminal prompts or basic worktree/terminal/built-in-browser control, use `orca-cli`.\n\n## Tool Boundary\n\nIf a task says to use Orca orchestration, the coordinator must create or bind a Run, create the Task with `orca orchestration task-create`, then attach the worker with either the preferred `orca orchestration worker-start` composition or the low-level `orca orchestration dispatch --inject` path.\n\nDo not substitute non-Orca subagent tools, generic agent-spawn APIs, or chat-only parallel worker features. Those may create useful workers, but they do not create Orca task/dispatch provenance, injected lifecycle preambles, `worker_done` authority, or decision gates.\n\nBefore claiming a worker was orchestrated, verify the task/dispatch exists:\n\n```bash\norca orchestration task-list --json\norca orchestration dispatch-show --task --json\n```\n\nIf the work was accidentally run outside Orca orchestration, say so plainly. To repair provenance, rerun or revalidate the needed work through a fresh Orca terminal plus injected dispatch; do not retroactively describe the external worker as orchestrated.\n\n## When To Use\n\n- Send/reply/ask between agent terminals with persistent messages.\n- Dispatch structured tasks to workers and wait for `worker_done` or `escalation`.\n- Track task DAGs with dependencies.\n- Run coordinator loops or decision gates.\n\nDo not use orchestration merely because the user says \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or asks for another worktree/agent/model/effort. Those are full ownership transfers unless the user explicitly asks to supervise, monitor, wait for worker completion/results, coordinate a DAG, use decision gates, or keep a blocking ask/reply loop.\n\n## Preconditions\n\n- `orca status --json` should show a running runtime.\n- `orca` must be on PATH (`orca-ide` on Linux).\n- The orchestration experimental feature must be enabled in Settings > Experimental.\n- `orca orchestration` commands are RPC calls to the running Orca runtime.\n\n## Contract Migration\n\nOrca adopts a live pre-update orchestration assignment into an ordinary Run. Adoption preserves the existing agent process, PTY/session, terminal handle, tab/leaf/pane, worktree or folder workspace, Task, and Dispatch; it never restarts or replaces the worker. The retired scheduler is not revived, and a newly created attempt uses the current grammar.\n\nTreat the authority label on injected or formatted messages as definitive:\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported command printed with the message, using the same CLI executable and arguments that the original prompt supplied.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded, at-least-once cutover replay. Process it idempotently and acknowledge it only through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or lifecycle action.\n- An unlabeled current message uses the current guide and current grammar.\n\nAn explicitly selected current Run, attested current Run binding, current Dispatch, or federated attachment takes precedence over legacy fallback. A retained adoption record alone never turns a current command into a legacy call.\n\nDatabase provenance, an old-looking terminal, or a legacy Run ID does not prove mutation authority. If the runtime cannot prove liveness, principal ownership, capability, or the exact legacy contract, it degrades to read-only inspection and must not fall back to local execution. Exact recovery may restore the already-live PTY once in its original inactive background tab. It must not spawn, write, signal, stop, switch, focus, split, or inject a terminal. Loss of lifecycle authority does not invalidate the existing assignment, process, or filesystem work.\n\nCompatibility retries have narrow guarantees. A pending ask, a reply, a final Dispatch settlement, and a consuming check have durable recovery identities. A-era heartbeat and escalation calls remain at-least-once across a manual A-to-B retry because identical later signals may be intentional. If an A-era ask may already have been answered, run the exact non-consuming recovery check printed by the runtime first; after its answer is printed and acknowledged, a new invocation with the same question creates a new question. Never guess among multiple identical question threads.\n\nWhen a compatibility or recovery command returns structured next-step arguments, run those exact arguments with the same CLI executable. The arguments intentionally omit the executable name so the guidance works with `orca`, `orca-ide`, `orca-dev`, or another configured Orca CLI command. Do not translate the command from memory, broaden its recipient, or retry it as a current mutation unless the returned guidance explicitly says to.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The initial command durably commits the question, prints its exact `ask --resume ` command, and exits with launcher status `75`; it does not wait for the answer. Run that exact resume command after the launcher or update boundary. Resume is idempotent and read-oriented: it waits for the already-committed question and does not create another one. For a WSL process that received compatibility proof at launch, use the printed executable `orca-ide` WSL resume command so the same distro and packaged launcher authority are preserved; do not substitute a PATH-resolved local CLI. Older WSL processes that never received the hidden launch token remain lifecycle read-only after the update, even while their terminal and filesystem work continue.\n\nLegacy inspection remains available without consuming mail:\n\n```bash\norca orchestration run-list --json\n# run_legacy_local is an empty audit tombstone after adoption.\norca orchestration run-show --id run_legacy_local --json\n# In run-list, find the ordinary Run whose objective is:\n# \"Recovered orchestration work from a contract update\"\norca orchestration run-show --id --json\norca orchestration task-list --run --json\norca orchestration inbox --full --json\norca orchestration check --terminal --peek --format --json\norca terminal read --terminal --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\nIf the original coordinator is unavailable or cannot prove its retained authority, a current coordinator may explicitly take over the adopted Run from its own live agent terminal:\n\n```bash\norca orchestration run-use --id --takeover-legacy --json\norca orchestration check --run --json\n```\n\nTakeover fences only the old coordinator, binds the current one, and moves pending worker mail into current Run Delivery. It is bound to the authenticated invoking terminal; `--from` cannot name another coordinator. Live legacy workers keep their original Tasks, Dispatches, processes, filesystems, and old prompt commands; their later questions, escalations, and completion reports route to the current coordinator. Do not use takeover while the original coordinator is still actively coordinating, because its later lifecycle mutations are rejected.\n\nDo not launch a replacement editor merely because the desktop app or runtime was updated. If adoption cannot prove continuing authority, keep the original worker as the only editor until it reaches a stable handoff point, then use a new current Dispatch in a conflict-free placement for any remaining work.\n\n## Ownership\n\nNew orchestration messages and tasks belong to one explicitly bound Run. A Run is only a durable namespace and coordinator inbox; it never schedules or places workers. Lifecycle authority comes from the active Dispatch, and terminal handles remain routing metadata rather than durable identity. Send `worker_done` and `heartbeat` from the worker's own terminal; Orca routes them to that Dispatch's Run.\n\nClassify inherited context before sending lifecycle messages:\n\n- Coordinated subtask: a live coordinator owns the DAG and waits on this dispatch. Follow the preamble exactly, including `worker_done`, heartbeat/status, `ask`, and `escalation`.\n- Full handoff means ownership transfer, not supervised dispatch. The original actor is not monitoring a DAG, so do not create lifecycle obligations unless the user explicitly asks you to supervise.\n- Classify requests containing \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"another agent\", or \"another worktree\" as full handoffs by default, even when the user names a custom model or reasoning effort.\n- Use supervised orchestration only when the user explicitly asks you to \"supervise\", \"monitor\", \"wait\", \"track completion\", \"wait for worker_done\", return results, coordinate a DAG, use a decision gate, or manage ask/reply flow.\n- Do not use `orca orchestration dispatch --inject` for full handoffs. It injects a coordinator preamble that tells the worker to send `worker_done`, heartbeat, and `ask` messages, then end its turn under the original terminal's dispatch lifecycle.\n- Do not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. Do not peek at terminal output after prompt delivery to monitor progress.\n- A review-only `worker_done` reports findings; it does not authorize coordinator file edits. After a review-only completion, synthesize findings, ask a decision gate if ownership is unclear, and dispatch or hand off fixes unless the user explicitly asked the coordinator to own fixes.\n- If the user's plan names a next owner agent (for example, \"then use opencode to create a PR\"), post-review corrections and PR prep belong to that named owner. The coordinator routes, synthesizes, asks decision gates when needed, and supervises; the named owner edits files and creates the PR.\n\nIf unclear, inspect orchestration state before sending lifecycle messages:\n\n```bash\norca orchestration task-list --json\norca terminal list --json\n# If inherited context includes a task id:\norca orchestration dispatch-show --task --json\n```\n\n## Messaging\n\n```bash\norca orchestration send --subject [--to ] [--from ] [--body ] [--type ] [--priority ] [--thread-id ] [--payload ] [--json]\norca orchestration check [--terminal ] [--ack ] [--peek|--all] [--types ] [--format] [--wait] [--timeout-ms ] [--json]\norca orchestration reply --id --body [--from ] [--json]\norca orchestration ask (--question |--resume ) [--options ] [--timeout-ms ] [--from ] [--json]\norca orchestration inbox [--limit ] [--json]\n```\n\nRules:\n\n- Omit `--from` unless impersonating another terminal; Orca auto-resolves it from the current terminal.\n- A coordinator `check` returns the bound Run's oldest FIFO Delivery (up to 50 messages) and replays that exact batch until `--ack `. Process every message before acknowledging; `check --ack --wait` acknowledges, checks, and waits in one operation.\n- Use `--peek` and `--all` only for read-only history/debugging. Type filters decide when a waiter wakes; the returned actionable Delivery is still the oldest full batch.\n- Use `dispatch:` for coordinator guidance to one supervised worker. Orca routes that stable address locally or through the connected-server relay; do not substitute a remote terminal handle.\n- Terminal handles remain appropriate for low-level pre-Dispatch messaging. Prefer `agentTerminalHandle` from the create response, fall back to `startupTerminal.handle` for older runtimes, then re-resolve with `orca terminal list --worktree ... --json` if missing or stale. Continue with the replacement handle only; never dual-send to old and new handles.\n- `terminal list --json` omits `visualLayouts` because handle recovery does not need topology. Add `--include-visual-layouts` only for explicit tab and pane inspection.\n- `orca orchestration check --peek --format --json` returns locally formatted unread mail without consuming it; it never writes to terminal input or remotely wakes another terminal. Use `orchestration dispatch --inject` to deliver a tracked task, or `terminal send` when an existing agent needs a free-form prompt.\n- While supervising workers manually, use `check --wait --types worker_done,escalation,question --timeout-ms ` instead of sleep/poll loops. Process the whole Delivery, reply to `question` messages with `orca orchestration reply --id --body --json`, then acknowledge and keep waiting.\n- `check --json` prints exactly one JSON document on stdout. While `--wait` blocks it also prints keepalive lines (`{\"_keepalive\":true,...}`) to stderr so you can tell the process is alive; those are never on stdout. Do not merge the streams before a parser — `check --wait --json 2>&1 | ` fails with \"Extra data: line 2\". Pipe stdout only.\n- Treat a `check --wait` timeout or `{count:0}` as a checkpoint, not a worker failure. Long coding tasks routinely run 15-60 minutes; keep using rolling waits unless you receive `worker_done`/`escalation`, the terminal exits or disappears, or the user explicitly asks you to stop.\n- Heartbeats and visible terminal activity mean the worker is alive, not done. Do not stop, close, kill, or restart a worker just because it has not produced a completion message yet.\n- Use `ask` when a worker needs a blocking answer from the coordinator; it defaults to the active Dispatch's Run. Timeout or disconnect leaves the question pending, so resume by its original message ID instead of asking again.\n- `check --wait` returns one bounded Delivery, not every future completion. Process every message, acknowledge it, then keep waiting until every expected Dispatch settles.\n- Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`.\n- Message types include `status`, `dispatch`, `worker_done`, `merge_ready`, `escalation`, `handoff`, `question`, `decision_gate` (legacy/gates), and `heartbeat`.\n- Use group addresses only for messages that are genuinely useful to many terminals, such as `status` broadcasts or intentional fan-out questions. Do not send dispatch lifecycle messages to groups.\n- `worker_done` belongs to the active Dispatch and defaults to its Run mailbox; never target a group.\n- A valid `worker_done` for the active `taskId` + `dispatchId` marks the task and dispatch completed automatically. Do not follow it with `task-update --status completed`; reserve manual updates for explicit recovery or overrides.\n- `heartbeat` is also Dispatch-scoped. Include both IDs and omit `--to` so Orca uses the owning Run; use `status` for broad progress updates.\n\n## Tasks And Dispatch\n\nA Run is the namespace/inbox, a Task is the work item, and a Dispatch assigns one Task attempt to a terminal. Create or bind a Run once before the common loop.\n\n```bash\norca orchestration run-create --objective --json\norca orchestration task-create --spec [--deps ] [--parent ] [--json]\norca orchestration task-list [--status ] [--ready] [--brief] [--json]\norca orchestration task-update --id --status [--result ] [--json]\norca orchestration dispatch --task --to [--from ] [--inject] [--json]\norca orchestration dispatch-show --task [--json]\n```\n\nTask statuses: `pending`, `ready`, `dispatched`, `completed`, `failed`, `blocked`.\n\nDispatch rules:\n\n- `--inject` sends the task spec plus preamble into a recognized agent CLI so it can report `worker_done`.\n- If the target is a bare shell, omit `--inject`, dispatch for tracking if needed, then send the prompt manually with `orca terminal send --terminal --text --enter --json`.\n- After 3 consecutive failures on one task, the dispatch context circuit-breaks and the task is marked failed.\n- Use `task-list --brief --json` for coordinator sweeps; it collapses whitespace and caps each echoed spec at 160 characters (`spec_truncated` marks shortened rows). Omit `--brief` when the full spec is required, or when an older CLI rejects it as an unknown flag.\n\n## How deep workers can nest\n\nA dispatched worker normally cannot dispatch sub-workers. Attempting it fails with\n`nested_worker_depth_exceeded` and a message telling the worker to complete the task\nitself. Do that — do not try to route around it.\n\nThe limit is a number, not an on/off switch. `Settings -> Orchestration -> Nested worker depth`\nsets how many generations are allowed:\n\n- `1` (default): a coordinator dispatches workers; those workers do not dispatch.\n- `2`: workers may dispatch one further generation.\n\nDepth is counted from the terminal that issues the command, not from the Run. Creating a\nnew Run does not reset it — a worker that runs `run-create` then `worker-start` is still a\nworker, and still counted. This is the part that changed: the old behaviour rejected\nsub-dispatch only because a worker's terminal was not bound to a Run, so creating a Run was\nenough to slip past it.\n\nTwo limits worth knowing:\n\n- **It is a guardrail, not a security boundary.** A caller that declares another terminal's\n handle while its own launch evidence is unverifiable (an ordinary restored terminal, for\n example) can be counted as that terminal instead. Orca does not treat workers as hostile.\n- **It applies while a Dispatch is active.** After `worker_done`, or after a coordinator\n settles the task, the terminal is no longer a worker and is counted as a root again. The\n process may still be alive; that is the documented boundary, not an accident.\n\n## Preferred Supervised Worker Loop\n\nUse `worker-start` for the normal supervised path. It composes the existing worktree, terminal, readiness, and dispatch primitives while returning exact created/reused effects. Agents still choose placement and concurrency; Orca does not schedule workers or infer conflicts.\n\nCreate the Run and every independent Task first, then start all independent workers before waiting:\n\n```bash\norca orchestration run-create --objective \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration task-create --spec \"\" --json\norca orchestration worker-start --task --worktree current --agent codex --json\norca orchestration worker-start --task --worktree current --agent claude --json\n```\n\n`current` and exact existing worktrees create a fresh agent terminal and do not rerun setup. Reuse an existing agent only with `--terminal `.\n\nFor a per-invocation Claude, Codex, or Cursor launch, pass an opaque provider model id with `--model`; add `--effort` only when that agent/model supports the level. These options apply only to fresh agent terminals, override general agent default arguments, and are reported under `launch.requested` and `launch.effective` in the receipt:\n\n```bash\norca orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`, and neither option can combine with `--terminal`. A connected worker server must advertise launch-preference support before Orca forwards either option.\n\nFor a new worktree, setup runs by default and agent-first creation reuses the returned startup agent terminal:\n\n```bash\norca orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n# Independent/top-level:\norca orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nSetup normally starts alongside the agent. Only a repository explicitly configured with `wait-for-setup` delays agent launch until setup succeeds. Use `--setup skip` or `--setup inherit` only for a concrete reason.\n\nRead the returned receipt before continuing: `ready` plus setup `running` is normal for start-immediately, while wait-for-setup returns setup `succeeded` before accepting task input. A failed or unknown start exits nonzero; inspect its `stage`, `effects`, and `residualResources` instead of guessing or automatically retrying. A wait-for-setup timeout can honestly leave setup `running`, which is not proof of failure.\n\nTo run the worker on another connected Orca server, add `--on `. The Run and Tasks remain authoritative on the current server; later commands route by Dispatch ID, so never repeat `--on`:\n\n```bash\n# Mac Run home -> Windows worker (the reverse is identical from a Windows Run home)\norca orchestration worker-start --task --on windows --worktree new-top-level --repo --name --agent codex --setup run --json\norca orchestration worker-show --dispatch --json\norca orchestration worker-read --dispatch --limit 50 --json\norca orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nRemote `current` and `new-child` are intentionally invalid because those words are ambiguous across servers. Use an exact discovered remote worktree selector or `new-top-level` with an explicit remote repo selector.\n\nThe follow-up is structured inbox mail, not prompt injection. The worker's next\n`orchestration check` receives it even when the Dispatch is on another connected Orca server.\n\n`worker-read` defaults to `--source auto`: Orca returns the exact hook-reported Codex, Claude, OpenClaude, or Grok transcript when it can prove the worker session, otherwise it returns bounded terminal output with `source: \"terminal\"` and a typed `fallbackReason`. Continue with the returned top-level `cursor`; it stays pinned to that exact source. If Orca reports `source_changed`, start a fresh read without the old cursor. Never supply or guess a provider session ID or transcript path.\n\nWait until every expected Dispatch settles, not for a fixed number of batches:\n\n```bash\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n# Process every message. For each accepted worker_done that is not immediately reused:\norca orchestration worker-release --dispatch --json\n# Acknowledge only after every message and required release decision is handled:\norca orchestration check --ack --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\nAfter processing each accepted `worker_done`, choose the terminal's next owner before you acknowledge the Delivery or wait again. If the same exact agent has an immediate follow-up Task, read the `worker.agent_terminal_handle` field of `worker-show --dispatch --json`, then run `orca orchestration worker-start --task --terminal --json` so Orca transfers cleanup ownership to the new Dispatch. Otherwise run `orca orchestration worker-release --dispatch --json`.\n\nRun `worker-release` after both succeeded and failed `worker_done` reports unless the user explicitly asked to keep that worker live. Release is post-completion cleanup, not cancellation: Orca first preserves inspectable output, then closes only the exact agent terminal owned by that settled Dispatch. Reused or pre-existing terminals, setup terminals, coordinators, active workers, user-taken-over terminals, and identities Orca cannot prove are retained. If the user explicitly asks to keep the live terminal for debugging, record that exception with `orca orchestration worker-retain --dispatch --json` instead of silently skipping cleanup. When the user is finished, the same Dispatch can be passed to `worker-release`, which clears the requested retention and releases the terminal.\n\nDo not release a worker because of a timeout, TUI idle state, heartbeat, status, question, escalation, or rejected/stale `worker_done`. If release returns `release_pending` or `release_unknown`, do not substitute `terminal close`; follow the exact recovery action in the receipt. A replayed Delivery may repeat `worker-release` safely.\n\nWorkers report exactly once using the IDs and capability injected by Orca; they do not supply Run/server/terminal identity:\n\n```bash\norca orchestration send --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a,path/b\" --json\n# On failure, use --outcome failed; never encode failure only in prose.\n```\n\nA worker question defaults to its owning Run. Timeout leaves it pending:\n\n```bash\norca orchestration ask --question \"\" --options \"yes,no\" --timeout-ms 600000 --json\norca orchestration ask --resume --timeout-ms 600000 --json\n# Coordinator:\norca orchestration reply --id --body \"\" --json\n```\n\nRecovery is conditional, never a fixed destructive sequence:\n\n- The response was lost and named no Dispatch: run `orca orchestration request-show --request --json` first. It is read-only. `completed` means the mutation already took effect. `pending` means the original mutation is still running or Orca restarted before recording its outcome. For either state, replaying the original command with `--retry-request ` reuses the same operation identity so Orca can replay, join, or safely recover it without starting a separate duplicate. `absent` means this runtime holds no receipt under your caller identity and is not proof that nothing happened; inspect the affected state before deciding whether to retry.\n- `worker-show --dispatch ` says `ready`: keep waiting or read bounded output.\n- It proves `failed` or `stopped`: start a replacement with `worker-start --task --retry-of ` plus an explicit `--on`/`--worktree` and `--agent`/`--terminal` choice. Retry does not silently inherit placement.\n- It remains `outcome_unknown`: either `worker-stop --dispatch ` and inspect again, or explicitly `worker-abandon --dispatch ` while accepting that resources may still be live. Abandon performs no remote, process, or filesystem action.\n- `worker-stop` closes only the exact supervised agent terminal. It never deletes the worktree, setup terminal, configured tabs, or unrelated processes.\n\nLow-level `worktree create`, `terminal create`, and `dispatch --inject` remain valid recipes for custom argv or topology that `worker-start` does not express.\n\n`dispatch --inject` deliberately keeps an operator-started terminal unsupervised: it never creates a `worker_dispatches` row and `worker-stop`/`worker-abandon` never close that process. The dispatch context is still authoritative, so `worker-show`, `worker-read`, and `worker-list` report it as `unsupervised`; settled `worker-retain` and `worker-release` report `retained` with `no_owned_resource` and take no process action. Use `worker-start --terminal ` when supervision and worker lifecycle state are required.\n\n## Gates And Legacy Inspection\n\n```bash\norca orchestration gate-create --task --question [--options ] [--json]\norca orchestration gate-resolve --id --resolution [--json]\norca orchestration gate-list [--task ] [--status ] [--json]\n```\n\nUse `ask` for worker-to-coordinator questions; it creates a `question` message that the coordinator answers with `reply`. Use `gate-create` only for coordinator-managed task DAG decisions, not for answering a worker's `ask`.\n\n`coordinator-start`, `coordinator-stop`, `run`, and `run-stop` are retired scheduler commands. They perform no effects and return the current-skill recovery action. They are not aliases for lightweight Run creation or binding.\n\nRecovery only: `orca orchestration reset --tasks|--messages|--all --json` clears the selected local orchestration database state. Do not run it during active coordination unless explicitly abandoning that state.\n\n## Full Handoffs\n\nFor full ownership transfer, use non-lifecycle terminal/worktree commands and then stop monitoring unless the user asks for supervision.\n\nTreat these as full handoff requests by default: \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", \"give this to another worktree\", \"send this to another agent\", \"another agent\", \"another worktree\", or \"launch another agent to own this.\" Custom model or reasoning effort words such as `gpt-5.5`, `high`, or `xhigh` do not make the handoff supervised.\n\nSupervised orchestration remains available only when the user explicitly asks for supervision or coordination: \"supervise\", \"monitor\", \"wait for worker_done\", \"wait for results\", \"track completion\", \"DAG\", \"decision gate\", \"ask/reply\", or \"coordinate workers.\"\n\nDo not run `orca orchestration task-create`, `orca orchestration dispatch --inject`, or `orca orchestration check --wait` for full handoffs. `task-create` is also forbidden because it records coordinator-owned tracking state; if a task row is needed, the user asked for supervised orchestration. Do not create a `taskId`/`dispatchId`, inject a lifecycle preamble, wait for completion, or read the worker terminal after prompt delivery except to avoid losing the initial prompt.\n\nNew top-level worktree handoff:\n\n```bash\norca worktree create --name --no-parent --agent codex --prompt \"\" --setup run --json\n```\n\nBefore creating a new worktree from an active feature branch, decide and state whether the desired Orca lineage is child or top-level. Use child worktree lineage only when the new work is conceptually stacked under or dependent on the active worktree. For independent repo-wide fixes, standalone feature work, or unrelated follow-up tasks, create a top-level worktree with `--no-parent`.\n\nExisting terminal handoff:\n\n```bash\norca terminal send --terminal --text \"\" --enter --json\n```\n\nCustom Codex model/effort handoff:\n\n`orca worktree create --agent codex --prompt ...` launches the known Codex agent but does not accept Codex-specific `--model` or `-c model_reasoning_effort=...` arguments. When the user asks for a specific Codex model or effort, create the independent worktree first, launch Codex with the requested command in that worktree, wait only for TUI readiness if prompt delivery would otherwise race startup, send the prompt, and stop.\n\nThe two-step custom-argv path cannot enforce a repository's explicit `wait-for-setup` startup policy because the later `terminal create` is not the startup owned by `worktree create`. Use it only when the repository starts agents immediately. If the repository requires `wait-for-setup`, use an agent-first configured launcher that can preserve sequencing, or stop and ask rather than silently bypassing the policy.\n\nNote: when no repo default-terminal configuration supplies a primary terminal, bare create opens a fallback shell before `terminal create` adds the agent. Configured default tabs are materialized instead and may run real commands. Prefer `--agent` whenever custom argv is not required. With the two-step path, target only the agent handle; close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell.\n\nUse the exact full `::` worktree id returned by `orca worktree create --json`; a bare repo id cannot target the new worktree.\n\n```bash\norca worktree create --name --no-parent --setup run --json\norca terminal create --worktree id: --title --command 'codex --model gpt-5.5 -c model_reasoning_effort=\"xhigh\"' --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca terminal send --terminal --text \"\" --enter --json\n```\n\nWait only for `tui-idle` when needed to avoid losing the prompt. Do not monitor task completion.\n\n`--no-parent` only controls Orca lineage; it does not choose the Git base. If the work should start from the repo default base, omit `--base-branch` so Orca uses that default, or explicitly pass the repo default base (`origin/main`, `origin/master`, or the `orca repo show --repo --json` value); never base it on the current feature branch unless the user explicitly asks for stacked work or \"branch from current\". Put current-branch context in the prompt instead.\n\n## Worker Terminals\n\nChoose the worker location before creating a terminal. `Fresh worker` means a fresh agent session, not a new git worktree. For parallel work, create one fresh agent terminal per worker in the same required worktree, falling back to the active worktree when none is named. If the task says current worktree only, depends on uncommitted files/artifacts, or must validate/PR the current branch, keep every worker in the active worktree:\n\n```bash\norca terminal create --worktree active --title --command \"codex\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nReuse an idle agent in the required worktree only if the prompt allows reuse; otherwise create a fresh terminal there. Create a new worktree only when the user explicitly requests one or a concrete checkout or filesystem conflict makes sharing unsafe or impossible; if the user did not request it, state that conflict before running `worktree create`. Independent tasks, parallel execution, convenience, or a preference for separate checkouts are not isolation requirements.\n\nWhen a new worktree is allowed, use child lineage for isolated work that is stacked under or dependent on the active worktree, and use `--no-parent` when it is not stacked. Decide the Git base separately: `--no-parent` makes the worktree top-level in Orca, while omitted `--base-branch` uses the repo default base.\n\nFor every new worktree, pass `--setup run` so any configured repository setup hook runs. This does not mean waiting for setup before agent launch: preserve the repository's startup policy, whose default starts setup and the agent side by side. Use `--setup skip` or `--setup inherit` only when there is a concrete task-specific reason, and state that reason before creating the worktree. This rule does not rerun setup for current or existing worktrees.\n\n```bash\norca worktree create --name --agent codex --setup run --json\n# or: --agent claude | omp | pi | grok | ...\n# Read from agentTerminalHandle, falling back to startupTerminal.handle.\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration dispatch --task --to --inject --json\n```\n\nFor new-worktree workers, read the id and `agentTerminalHandle` from `worktree create`, falling back to `startupTerminal.handle` for older runtimes. Use that as the sole worker handle when present; otherwise use `terminal list` to resolve the agent handle. Omit `--repo` only inside an Orca-managed worktree; otherwise pass `--repo `.\n\n**For an allowed new worktree, use agent-first:** `--agent` reveals the new worktree and launches the selected agent **in its first terminal**, without adding a separate fallback shell for that worker. Pass `--setup run`; repo setup and default-terminal settings may add intentional tabs or splits. Do **not** run bare `worktree create` and then `terminal create --command ` for the same worker when agent-first create is available: without configured default tabs, that two-step path leaves a fallback shell + agent pair. Only use it when custom agent argv is required (for example Codex model/effort flags) or when an older CLI rejects `--agent`; if you must, message only the agent handle. Configured default tabs are intentional surfaces, so close a prior terminal only after `terminal list` or `terminal show` confirms it is an unused shell. Do not run `worktree create` when the task must stay in the current worktree.\n\nUse `orca worktree create --prompt ...` or `orca terminal send ...` for full handoffs or untracked/lightweight prompts. Those paths do not attach `taskId`/`dispatchId`; the worker should not send lifecycle messages unless the prompt supplies a live orchestration preamble.\n\nSidebar lineage and orchestration lifecycle are related but not identical. A same-worktree worker may appear as a peer under that worktree in the sidebar while remaining a child dispatch in orchestration state; only an actual child worktree creates visible parent/child worktree lineage.\n\nOther terminal commands coordinators often need:\n\n```bash\norca terminal list [--worktree ] [--include-visual-layouts] [--json]\norca terminal create [--worktree ] [--title ] [--command ] [--json]\norca terminal split --terminal [--direction horizontal|vertical] [--command ] [--json]\norca terminal wait --terminal --for tui-idle --timeout-ms --json\norca terminal read --terminal --json\norca terminal send --terminal --text --enter --json\n```\n\nIf an older CLI rejects `worktree create --agent`, create the worktree normally, then run `orca terminal create --worktree --command \"codex\" --json` or `--command \"claude\"`.\n\nWait for `tui-idle` before dispatching. Always pass `--timeout-ms`; real coding tasks can take 15-60 minutes. During supervision, use rolling `check --wait` windows. If a window returns no matching message, inspect `task-list`, `terminal read`, or `terminal wait --for tui-idle` as a liveness checkpoint; if the terminal is still working or producing activity, keep waiting instead of retrying the task.\n\n## Agent Guidance\n\n- Workers with a valid live preamble must send `worker_done` exactly once from their own terminal with an explicit `--outcome succeeded` or `--outcome failed`:\n `orca orchestration send --type worker_done --subject \"\" --body \"<3-sentence summary: what you did, what you found, what's left>\" --task-id --dispatch-id --outcome succeeded --files-modified \"path/a\" --report-path \"\" --json`\n- A failed outcome is still a terminal report, but Orca records both the Dispatch and Task as failed. Never encode failure only in the subject/body.\n- After sending `worker_done`, end that dispatched turn and idle at the agent prompt. Do not autonomously start more work, poll, or attempt to close the terminal yourself. A direct user instruction takes precedence and starts ordinary user-owned work: follow it without coordinator approval or a fresh Dispatch, never refuse it because of worker/coordinator roles, and do not reuse the settled Dispatch's lifecycle IDs. A coordinator-supervised follow-up still arrives with a fresh preamble + TASK block.\n- For long tasks, send heartbeat/status only when the preamble asks for it, including both IDs:\n `orca orchestration send --type heartbeat --subject \"alive\" --payload '{\"taskId\":\"\",\"dispatchId\":\"\",\"phase\":\"implementing\"}' --json`\n- If blocked before completion, use `ask`; use `escalation` only when ownership is valid and the coordinator must intervene.\n- Treat preambles inherited through terminal history or full handoffs as stale unless the current prompt explicitly keeps that coordinator in the loop.\n- Coordinators must account for every settled worker terminal before waiting again or ending the turn: immediately reuse the exact worker for a new Dispatch, explicitly retain it at the user's request with `worker-retain`, or run `worker-release`. Do not leave a completed worker live merely to inspect output; released workers remain readable through `worker-read`.\n- Coordinators should use `task-list --ready` as external memory, dispatch parallel waves, and avoid dependency chains deeper than 3-4 steps.\n\n## Example\n\n```bash\norca terminal create --worktree active --title login-css-worker --command \"claude\" --json\norca terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\norca orchestration task-create --spec \"Fix the login button CSS\" --json\norca orchestration dispatch --task --to --inject --json\norca orchestration check --wait --types worker_done,escalation,question --timeout-ms 900000 --json\n```\n\n## Next Action\n\nCoordinator: confirm `orca status --json`, create or bind a Run, inspect `task-list`/`dispatch-show` if inheriting state, then use the explicit supervised loop (`task-create` -> `worker-start` -> `check --wait`). Use low-level terminal creation plus `dispatch --inject` only when the composed start does not express the needed topology. After every accepted `worker_done`, either transfer the exact terminal to an immediate follow-up Dispatch or run `worker-release` before the next wait.\n\nWorker: if the current prompt contains a live dispatch preamble, do the task, use `ask` for blocking questions, and send `worker_done` once with the required payload. If the preamble is stale or absent, do not send lifecycle messages; inspect state or treat the prompt as an ordinary handoff.\n" // Why: no current guide has bundled reference documents, so --full is byte-identical for now. // oxfmt-ignore @@ -86,7 +86,7 @@ export const BUNDLED_SKILL_GUIDES = [ }, { name: "orchestration", - description: "Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation waits, task DAGs, decision gates, coordinator loops, or decomposing work across agents. Use `orca-cli` instead for full ownership handoffs, including requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another worktree\" when the user did not explicitly ask to supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for terminal control, lightweight terminal prompts, shell commands, Orca worktree management, reading or waiting on terminals, and automation of the browser embedded inside Orca. Use Computer Use for external browser windows, webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when the task requires OS/window-level control such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages.", + description: "Use Orca orchestration for structured multi-agent coordination: threaded messages, blocking ask/reply flows, task dispatch, worker_done/escalation waits, task DAGs, decision gates, or coordinator loops. Use `orca-cli` instead for full ownership handoffs, including requests phrased as \"hand off\", \"handoff\", \"handover\", \"give this to another agent\", or \"another worktree\" when the user did not explicitly ask to supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for terminal control, lightweight terminal prompts, shell commands, Orca worktree management, reading or waiting on terminals, and the Orca embedded browser. Use Computer Use for external browser windows, webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when the task requires OS/window-level control such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages.", markdown: ORCHESTRATION_MARKDOWN, fullMarkdown: ORCHESTRATION_MARKDOWN, aliases: [] diff --git a/src/main/asar-transparent-fs.test.ts b/src/main/asar-transparent-fs.test.ts new file mode 100644 index 00000000000..f416450599d --- /dev/null +++ b/src/main/asar-transparent-fs.test.ts @@ -0,0 +1,43 @@ +import { mkdir, mkdtemp, writeFile } from 'node:fs/promises' +import { existsSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, describe, expect, it } from 'vitest' +import { rm } from './asar-transparent-fs' + +// Why not an asar fixture here: plain Node has no asar shim to see through, so the archive case can +// only be settled by the real binary — `host-tree-removal-asar.electron.test.ts` does that. What +// this pins is the other half: outside Electron `original-fs` does not resolve, and the helper has +// to degrade to `node:fs/promises` rather than throw at first use. +const roots: string[] = [] + +afterAll(async () => { + for (const root of roots) { + await rm(root, { recursive: true, force: true }).catch(() => {}) + } +}) + +describe('asar-transparent rm', () => { + it('removes a tree recursively where `original-fs` is unresolvable', async () => { + expect(process.versions.electron).toBeUndefined() + const root = await mkdtemp(join(tmpdir(), 'orca-asar-transparent-')) + roots.push(root) + const target = join(root, 'wt-1700000000000-abcdef01') + await mkdir(join(target, 'nested'), { recursive: true }) + await writeFile(join(target, 'nested', 'file.txt'), 'x', 'utf8') + + await expect(rm(target, { recursive: true, force: true })).resolves.toBeUndefined() + + expect(existsSync(target)).toBe(false) + expect(existsSync(root)).toBe(true) + }) + + it('honours `force: false` rather than swallowing a missing path', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-asar-transparent-')) + roots.push(root) + + await expect(rm(join(root, 'absent'), { recursive: true })).rejects.toMatchObject({ + code: 'ENOENT' + }) + }) +}) diff --git a/src/main/asar-transparent-fs.ts b/src/main/asar-transparent-fs.ts new file mode 100644 index 00000000000..cddab843434 --- /dev/null +++ b/src/main/asar-transparent-fs.ts @@ -0,0 +1,35 @@ +// Why: Electron patches `fs` so a `*.asar` file reports `isDirectory() === true`, so Node's +// recursive `rm` descends into the archive, tries to `rmdir` a real file, and fails the parent with +// ENOTEMPTY. Every worktree that has ever run `pnpm install` carries at least one +// (`node_modules/.pnpm/electron@…/…/Electron.app/Contents/Resources/default_app.asar`), so a +// worktree removal aborts there deterministically — the residue is not a concurrent-writer race and +// no amount of retrying clears it. `original-fs` is Electron's unpatched `fs`; unlike +// `process.noAsar` it is scoped to this call rather than to the whole process, which matters because +// a multi-GB removal runs for seconds while the main process may still be loading modules out of +// `app.asar`. See `cli/appimage-payload-removal.ts` for the same bug at a call site short enough to +// use the process-global flag. + +import { rm as nodeRm } from 'node:fs/promises' +import { createRequire } from 'node:module' + +type Rm = typeof nodeRm + +let resolvedRm: Rm | undefined + +function resolveRm(): Rm { + try { + // Why require and not an import: `original-fs` only exists inside Electron, so vitest, the + // `orca` CLI and the plain-node entrypoints must resolve `node:fs/promises` instead — and there + // the shim does not exist either, so plain `fs` is already asar-transparent. + const originalFs = createRequire(__filename)('original-fs') as { promises?: { rm?: Rm } } + return typeof originalFs.promises?.rm === 'function' ? originalFs.promises.rm : nodeRm + } catch { + return nodeRm + } +} + +/** `fs.promises.rm` that sees a `*.asar` as the file it is rather than as a directory. */ +export const rm: Rm = (path, options) => { + resolvedRm ??= resolveRm() + return resolvedRm(path, options) +} diff --git a/src/main/codex/codex-app-server-process-teardown.test.ts b/src/main/codex/codex-app-server-process-teardown.test.ts index 1ddb672a531..cec8f91d081 100644 --- a/src/main/codex/codex-app-server-process-teardown.test.ts +++ b/src/main/codex/codex-app-server-process-teardown.test.ts @@ -1,7 +1,14 @@ import type { ChildProcess } from 'node:child_process' -import { describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + findSelfInitiatedTreeKills, + resetSelfInitiatedTreeKillLogForTest +} from '../crash-reporting/self-initiated-tree-kill-log' import { terminateCodexAppServerProcessTree } from './codex-app-server-process-teardown' +/** Above pid_max on every supported POSIX host, so the group signal is a real ESRCH. */ +const UNREACHABLE_PGID = 2_147_483_647 + function child() { return { pid: 1234, @@ -10,6 +17,10 @@ function child() { } describe('terminateCodexAppServerProcessTree', () => { + beforeEach(() => { + resetSelfInitiatedTreeKillLogForTest() + }) + it('waits for the Windows tree kill before releasing the wrapper', async () => { const target = child() const release = Promise.withResolvers() @@ -120,6 +131,55 @@ describe('terminateCodexAppServerProcessTree', () => { expect(target.kill).not.toHaveBeenCalled() }) + /** + * `selfInitiatedTreeKillCount` decides whether a `render-process-gone` was + * ours. A group that had already exited was killed by nobody, so crediting it + * puts a suspect in the five-second window that Orca never issued. Exercised + * through the real `process.kill(-pgid)` because the swallow being tested + * lives in the production default, not in an injectable seam. + */ + it('does not claim a snapshot group that was already gone', async () => { + const target = { pid: UNREACHABLE_PGID, kill: vi.fn(() => true) as ChildProcess['kill'] } + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ + rootPgid: UNREACHABLE_PGID, + descendants: [], + capturedAtMs: 1 + }), + terminateDescendants: async () => true + }) + ).resolves.toBe(true) + + expect(target.kill).toHaveBeenLastCalledWith('SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([]) + }) + + it('claims a snapshot group the signal actually reached', async () => { + const target = child() + const signalProcessGroup = vi.fn() + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ rootPgid: 1234, descendants: [], capturedAtMs: 1 }), + terminateDescendants: async () => true, + signalProcessGroup + }) + ).resolves.toBe(true) + + expect(signalProcessGroup).toHaveBeenCalledWith(1234, 'SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([ + expect.objectContaining({ + pid: 1234, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' + }) + ]) + }) + it('tears down 40 dedicated groups without process-table scans or cross-group fanout', async () => { const killMocks = Array.from({ length: 40 }, () => vi.fn(() => true)) const targets = killMocks.map((kill, index) => ({ diff --git a/src/main/codex/codex-app-server-process-teardown.ts b/src/main/codex/codex-app-server-process-teardown.ts index a35ad4d3164..5a9c6e3574b 100644 --- a/src/main/codex/codex-app-server-process-teardown.ts +++ b/src/main/codex/codex-app-server-process-teardown.ts @@ -128,19 +128,24 @@ async function terminatePosixTree( if (descendantsExited && snapshot.rootPgid === rootPid) { const signalGroup = deps.signalProcessGroup ?? - ((pgid: number, signal: NodeJS.Signals) => { - try { - process.kill(-pgid, signal) - } catch { - // Group already exited. - } + ((pgid: number, signal: NodeJS.Signals) => process.kill(-pgid, signal)) + let groupSignalled = false + try { + signalGroup(snapshot.rootPgid, 'SIGKILL') + groupSignalled = true + } catch { + // Already-gone is still the desired outcome, but nothing here killed it, + // and a crumb for a kill we never landed is a false render-process-gone suspect. + } + if (groupSignalled) { + // Outside the try, as in terminateDedicatedPosixGroup: that catch is the + // already-gone contract, not a breadcrumb handler. + recordSelfInitiatedTreeKill({ + pid: snapshot.rootPgid, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' }) - signalGroup(snapshot.rootPgid, 'SIGKILL') - recordSelfInitiatedTreeKill({ - pid: snapshot.rootPgid, - site: 'codex-app-server-teardown', - scope: 'posix-process-group' - }) + } } if (!descendantsExited) { child.kill('SIGCONT') diff --git a/src/main/crash-reporting/gone-time-system-memory.ts b/src/main/crash-reporting/gone-time-system-memory.ts deleted file mode 100644 index 7cca89d2d4b..00000000000 --- a/src/main/crash-reporting/gone-time-system-memory.ts +++ /dev/null @@ -1,77 +0,0 @@ -import type { CrashReportDetailValue } from '../../shared/crash-reporting' - -// ─── System memory at gone time ───────────────────────────────────── -// Why: the system outlives the crashed process, so this IS sampleable at -// process-gone — it separates "renderer grew huge" from "machine out of -// memory/commit", which the per-process buckets alone cannot. -// Timing honesty: this reads AFTER the crashed process's memory returned to -// the OS, so free/swapFree can look healthier than they were at kill time. -// Platform honesty: swap* exist on Windows/Linux only. On Linux `free` is -// /proc/meminfo MemFree and is NOT the pressure signal — it excludes page cache -// and other reclaimable memory; `available` (MemAvailable, Linux-only) is. On -// macOS `free` is near-meaningless (file cache and compression keep it low on -// healthy machines); fileBacked/purgeable are the only reclaimability proxy this -// API gives there, and none of these fields answers "was the machine under -// pressure" on macOS — that needs a signal Electron does not expose. - -type CrashReportDetails = Record - -export function memoryKBFieldMB(value: unknown): number | undefined { - const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined - return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) -} - -type SystemMemoryInfoLike = { - total?: unknown - free?: unknown - available?: unknown - swapTotal?: unknown - swapFree?: unknown - fileBacked?: unknown - purgeable?: unknown -} - -type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null - -function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { - const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) - .getSystemMemoryInfo - if (typeof read !== 'function') { - return null - } - try { - return read.call(process) - } catch { - return null - } -} - -let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo - -export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { - systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo -} - -export function getSystemMemoryAtGoneDetails(): CrashReportDetails { - const info = systemMemoryInfoReader() - if (!info) { - return {} - } - const details: CrashReportDetails = {} - const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ - ['total', 'systemMemoryTotalMB'], - ['free', 'systemMemoryFreeMB'], - ['available', 'systemMemoryAvailableMB'], - ['swapTotal', 'systemMemorySwapTotalMB'], - ['swapFree', 'systemMemorySwapFreeMB'], - ['fileBacked', 'systemMemoryFileBackedMB'], - ['purgeable', 'systemMemoryPurgeableMB'] - ] - for (const [field, key] of fields) { - const mb = memoryKBFieldMB(info[field]) - if (mb !== undefined) { - details[key] = mb - } - } - return details -} diff --git a/src/main/crash-reporting/gpu-crash-fallback-decision.ts b/src/main/crash-reporting/gpu-crash-fallback-decision.ts index e962952c695..ed2034a62be 100644 --- a/src/main/crash-reporting/gpu-crash-fallback-decision.ts +++ b/src/main/crash-reporting/gpu-crash-fallback-decision.ts @@ -50,6 +50,8 @@ export class GpuCrashFallbackTracker { crashesInWindow: number } { if (this.engaged || !Number.isFinite(msSinceLaunch) || msSinceLaunch < 0) { + // Crashes landing while engaged (e.g. during a verdict wait before `disengage`) are + // not recorded, so a reported crashesInWindow can understate the actual burst. return { shouldEngageFallback: false, crashesInWindow: this.recentCrashes.length } } // Why: out-of-order arrivals would corrupt the sorted window, and a clock @@ -73,6 +75,17 @@ export class GpuCrashFallbackTracker { return this.engaged } + /** + * Re-arm after an engagement the caller decided not to act on. `recordGpuCrash` + * latches `engaged` and reports the threshold crossing exactly once, so a caller + * that discards that one report would otherwise silence safe graphics for the + * rest of the process — including a later burst it would have acted on. + * Leaves the crash window intact; only the one-shot latch is released. + */ + disengage(): void { + this.engaged = false + } + /** Crash times currently inside the window. Exposed to assert the pruning invariant. */ windowSnapshot(): readonly number[] { return [...this.recentCrashes] diff --git a/src/main/crash-reporting/pre-gone-host-memory.test.ts b/src/main/crash-reporting/pre-gone-host-memory.test.ts new file mode 100644 index 00000000000..0df13d4fee5 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.test.ts @@ -0,0 +1,379 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + getSystemMemoryDetails, + setSystemMemoryInfoReaderForTest, + withSwapVolumeFreeSpace +} from './system-memory-details' +import { + readSwapVolumeFreeSpace, + setSwapVolumeFreeSpaceReaderForTest, + type SwapVolumeFreeSpace +} from './swap-volume-free-space' +import { samplePreGoneSystemMemory } from './pre-gone-host-memory' +import { + buildProcessGoneCrashDetails, + resetPreGoneCrashSamplingForTest, + samplePreGoneProcessMetrics, + startPreGoneCrashSampling +} from './process-gone-diagnostics' + +type MetricFixture = { + pid: number + creationTime: number + type: string + memory: { workingSetSize: number; peakWorkingSetSize?: number; privateBytes?: number } +} + +const { appMetricsMock } = vi.hoisted(() => ({ + appMetricsMock: vi.fn<() => MetricFixture[]>(() => []) +})) + +vi.mock('electron', () => ({ app: { getAppMetrics: appMetricsMock } })) + +const BROWSER_AND_RENDERER: MetricFixture[] = [ + { pid: 10, creationTime: 1, type: 'Browser', memory: { workingSetSize: 1024 * 250 } }, + { + pid: 11, + creationTime: 2, + type: 'Tab', + memory: { workingSetSize: 1024 * 400, peakWorkingSetSize: 1024 * 420, privateBytes: 1024 * 260 } + } +] + +const BROWSER_ONLY: MetricFixture[] = [BROWSER_AND_RENDERER[0]] + +const UNDER_COMMIT_PRESSURE = { + total: 16_000 * 1024, + free: 400 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 200 * 1024 +} + +const AFTER_THE_CORPSE_RELEASED = { + total: 16_000 * 1024, + free: 3_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 2_900 * 1024 +} + +// Commit limit ~= RAM: a disabled or fixed pagefile, which no amount of empty +// disk can grow into. `swapTotal > total` is all this API can say about that. +const FIXED_PAGEFILE_UNDER_PRESSURE = { + total: 16_000 * 1024, + free: 300 * 1024, + swapTotal: 16_100 * 1024, + swapFree: 180 * 1024 +} + +const NO_PAGEFILE_UNDER_PRESSURE = { + ...FIXED_PAGEFILE_UNDER_PRESSURE, + swapTotal: 15_900 * 1024 +} + +const BEFORE_THE_STORM = { + total: 16_000 * 1024, + free: 9_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 30_000 * 1024 +} + +describe('pre-gone host memory', () => { + beforeEach(() => { + resetPreGoneCrashSamplingForTest() + setSystemMemoryInfoReaderForTest(null) + setSwapVolumeFreeSpaceReaderForTest(null) + appMetricsMock.mockClear() + appMetricsMock.mockReturnValue(BROWSER_AND_RENDERER) + }) + + it('carries a pre-gone host reading, not only the post-mortem one', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + // The renderer dies; its ~400 MB returns to the OS, so the gone-time read + // now shows a much healthier machine than the one that refused the alloc. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + appMetricsMock.mockReturnValue(BROWSER_ONLY) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemorySwapFreeMB).toBe(2_900) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(200) + expect(details.systemMemoryPreGoneFreeMB).toBe(400) + expect(details.systemMemoryPreGoneTotalMB).toBe(16_000) + // Why: host memory keeps its own key family, so a `systemMemory` prefix scan sees both reads. + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) + }) + + // Why this decides the cluster: 200 MB available commit is only a REFUSAL when + // the pagefile cannot grow, which is what the volume's free space says. + it('reports swap-volume free space so low commit can be told from refused commit', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(120) + // Which volume was measured: Windows only names the DEFAULT pagefile drive. + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + }) + + it('omits swap-volume free space on Linux, where swap cannot grow into free disk', async () => { + // Linux swap is a fixed partition, a fixed-size swapfile, or zram; reporting + // root-fs free space next to SwapFreeMB 0 would read as headroom that is not there. + setSwapVolumeFreeSpaceReaderForTest(null) + + await expect(readSwapVolumeFreeSpace('linux')).resolves.toBeUndefined() + }) + + it('labels the reading with the pressure verdict the platform can actually give', () => { + // Windows available commit is only a REFUSAL when the pagefile cannot grow, + // which nothing here proves, so no label may read as that verdict. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + const windowsCommit = getSystemMemoryDetails('win32') + expect(windowsCommit.systemMemoryPressureSignal).toBe('available-commit-unqualified') + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32') + .systemMemoryPressureSignal + ).toBe('available-commit-volume-cotimed') + // A volume number from a different moment describes a different machine. + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32', false) + .systemMemoryPressureSignal + ).toBe('available-commit-unqualified') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, free: 400 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('none') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, available: 900 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('mem-available') + + // darwin free/fileBacked/purgeable answer reclaimability, never pressure. + setSystemMemoryInfoReaderForTest(() => ({ + total: 16_000 * 1024, + free: 272 * 1024, + fileBacked: 2_694 * 1024, + purgeable: 0 + })) + expect(getSystemMemoryDetails('darwin').systemMemoryPressureSignal).toBe('none') + }) + + // Why this and not the volume number: the branch's own repro needed a pagefile + // that CANNOT grow to kill anything, and neither the pagefile maximum nor its + // drive is readable here — `swapVolumeAnchor` measures SystemRoot's volume, + // which a relocated pagefile does not live on. + it('never reads free disk as proof the pagefile could have grown', () => { + setSystemMemoryInfoReaderForTest(() => FIXED_PAGEFILE_UNDER_PRESSURE) + const fixedPagefile = withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ) + // 180 MB of commit beside 812 GB of free disk: co-timed, and still not a + // verdict — reading it as "the pagefile had room" is the opposite conclusion. + expect(fixedPagefile.systemMemoryPressureSignal).toBe('available-commit-volume-cotimed') + + // The one decisive win32 case: commit limit at or below RAM means there is + // no pagefile behind it, so the floor cannot heal however empty the disk is. + setSystemMemoryInfoReaderForTest(() => NO_PAGEFILE_UNDER_PRESSURE) + expect( + withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ).systemMemoryPressureSignal + ).toBe('available-commit-hard-capped') + }) + + // Why the verdict and not just the field: a statfs issued on a healthy host at + // t=0 that resolves 20 s into a commit storm prints "200 MB commit, 40 GB of + // pagefile headroom" — which reads as NOT a commit refusal, the opposite + // conclusion, under the branch's most confident label. + it('will not let a statfs that outlived its tick qualify the win32 commit verdict', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // The storm arrives; the in-flight latch makes every tick skip the merge, + // so the pending statfs is as old as the tick that STARTED it. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(20_000) + const stale = buildProcessGoneCrashDetails({}, 'renderer') + expect(stale.systemMemoryPreGoneSwapFreeMB).toBe(200) + // The pre-storm volume number still ships — but carrying its own age, and + // without promoting the verdict the analyst reads. + expect(stale.systemMemoryPreGoneSwapVolumeFreeMB).toBe(40_000) + expect(stale.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(stale.systemMemoryPreGoneSwapVolumeAgeMs).toBe(20_000) + expect(stale.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + + // The next tick's statfs answers on its own tick, so it qualifies again. + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 900, volume: 'C:' })) + await samplePreGoneSystemMemory(30_000) + vi.setSystemTime(30_000) + const fresh = buildProcessGoneCrashDetails({}, 'renderer') + expect(fresh.systemMemoryPreGoneSwapVolumeFreeMB).toBe(900) + expect(fresh.systemMemoryPreGoneSwapVolumeAgeMs).toBe(0) + expect(fresh.systemMemoryPreGonePressureSignal).toBe('available-commit-volume-cotimed') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + // Round 5: sample identity alone could not see these ticks. A host read that + // returns nothing leaves the sample object in place, so `sample === issuedFor` + // still held 25 s and two ticks later and the statfs re-qualified the verdict. + it('will not let ticks with a failed host read pass a stale statfs off as co-timed', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // GlobalMemoryStatusEx starts failing: the sample is neither replaced nor erased. + setSystemMemoryInfoReaderForTest(() => null) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(25_000) + const details = buildProcessGoneCrashDetails({}, 'renderer') + // 25 s of lag: the label must not say co-timed beside that age. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(25_000) + expect(details.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + it("arms the host sampler on its own unref'd 10 s timer, not the metric sweep's", async () => { + vi.useFakeTimers() + vi.setSystemTime(0) + const readHostMemory = vi.fn(() => UNDER_COMMIT_PRESSURE) + setSystemMemoryInfoReaderForTest(readHostMemory) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') + try { + startPreGoneCrashSampling() + + // Literal millisecond values: asserting the constants against themselves + // would let a cadence regression through, and 37 s of staleness is the bug. + expect(setIntervalSpy.mock.calls.map(([, ms]) => ms)).toEqual([60_000, 10_000]) + for (const { value } of setIntervalSpy.mock.results) { + expect((value as NodeJS.Timeout).hasRef()).toBe(false) + } + expect(readHostMemory).toHaveBeenCalledTimes(1) + + readHostMemory.mockReturnValue(AFTER_THE_CORPSE_RELEASED) + await vi.advanceTimersByTimeAsync(10_000) + // One host tick, no extra metric sweep: the two samplers run independently. + expect(readHostMemory).toHaveBeenCalledTimes(2) + expect(appMetricsMock).toHaveBeenCalledTimes(1) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(2_900) + } finally { + setIntervalSpy.mockRestore() + vi.useRealTimers() + } + }) + + it('commits the host reading without waiting on the swap-volume statfs', async () => { + // Why: statfs is slowest during the paging storm this sampler targets, and + // a hung volume must not stall or silently skip host sampling. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + + void samplePreGoneSystemMemory(Date.now() - 5_000) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(200) + + // A second tick still refreshes the reading while that statfs hangs. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + void samplePreGoneSystemMemory(Date.now()) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(2_900) + }) + + it('publishes no pre-gone host keys when every memory field failed to read', async () => { + // Why not "no keys at all": the reading always carries its signal label, so a + // committed empty one would ship an age and a volume number with no memory + // numbers beside them — a disk-free figure standing in for a host reading. + setSystemMemoryInfoReaderForTest(() => ({ total: Number.NaN, free: undefined })) + await samplePreGoneSystemMemory(Date.now()) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) + + it('carries the last volume reading forward, aged, instead of dropping it', async () => { + vi.useFakeTimers() + try { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 42, volume: 'C:' })) + await samplePreGoneSystemMemory(0) + + // The next tick's statfs hangs — during the paging storm this targets, that + // is the normal case — so the tick has no volume reading of its own, and + // the sample that replaces the last one would otherwise drop the field. + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + void samplePreGoneSystemMemory(10_000) + vi.setSystemTime(10_000) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(42) + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + // Carried, not re-read: it ships at its real age, never as a fresh number. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(10_000) + } finally { + vi.useRealTimers() + } + }) + + it('keeps a failed host read from erasing the process-metric sample', async () => { + samplePreGoneProcessMetrics(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(() => { + throw new Error('getSystemMemoryInfo unavailable') + }) + await samplePreGoneSystemMemory(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(null) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.processMetricsPreGoneRendererWorkingSetMB).toBe(400) + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) +}) diff --git a/src/main/crash-reporting/pre-gone-host-memory.ts b/src/main/crash-reporting/pre-gone-host-memory.ts new file mode 100644 index 00000000000..0db56796750 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.ts @@ -0,0 +1,164 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import { readSwapVolumeFreeSpace } from './swap-volume-free-space' +import { + getSystemMemoryDetails, + SYSTEM_MEMORY_KEY_PREFIX, + withSwapVolumeFreeSpace +} from './system-memory-details' + +// ─── Pre-gone host memory sampling ────────────────────────────────── +// Why sample at all: the gone-time host read lands after the corpse released +// its pages, so it reports a healthier machine than the one that refused the +// allocation. +// Why 10 s and not the 60 s process-metrics cadence: at 60 s, four of five +// field OOMs carried a ~37 s old host reading — far too stale to see a +// transient commit refusal. A refusal shorter than the interval stays +// invisible; no cadence fixes that. + +export const PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS = 10_000 + +type CrashReportDetails = Record + +type PreGoneSystemMemorySample = { + details: CrashReportDetails + sampledAtMs: number + /** Tick that ISSUED the statfs now merged in — never the tick it resolved on. */ + swapVolumeSampledAtMs?: number +} + +let preGoneSample: PreGoneSystemMemorySample | null = null +let preGoneTimer: ReturnType | null = null +let swapVolumeReadInFlight = false +let samplingGeneration = 0 +let sampleTick = 0 + +const PRESSURE_SIGNAL_KEY = `${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal` + +/** + * Carries the last volume reading onto the sample that replaces its own. + * + * Why: a statfs slower than one tick would otherwise make the field vanish from + * the reports it exists for — the next tick replaces the sample wholesale, and + * the in-flight latch keeps intervening ticks from merging anything. It ships + * with its own (now larger) age and, not being co-timed, never names the label. + */ +function withCarriedSwapVolume(sample: PreGoneSystemMemorySample): PreGoneSystemMemorySample { + const previous = preGoneSample + if (!previous || previous.swapVolumeSampledAtMs === undefined) { + return sample + } + const freeMB = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`] + const volume = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`] + if (typeof freeMB !== 'number' || typeof volume !== 'string') { + return sample + } + return { + ...sample, + details: withSwapVolumeFreeSpace(sample.details, { freeMB, volume }, process.platform, false), + swapVolumeSampledAtMs: previous.swapVolumeSampledAtMs + } +} + +function commitHostMemorySample(nowMs: number): boolean { + try { + const details = getSystemMemoryDetails() + // Why not `length === 0`: the signal label is appended unconditionally, so a + // reading that resolved no memory field at all still arrives with one key. + if (!Object.keys(details).some((key) => key !== PRESSURE_SIGNAL_KEY)) { + return false + } + preGoneSample = withCarriedSwapVolume({ details, sampledAtMs: nowMs }) + return true + } catch { + // Why: a failed read must not erase the previous good sample. + return false + } +} + +async function mergeSwapVolumeFreeSpace(issuedOnTick: number): Promise { + if (swapVolumeReadInFlight) { + return + } + swapVolumeReadInFlight = true + const generation = samplingGeneration + const issuedFor = preGoneSample + try { + const volume = await readSwapVolumeFreeSpace() + if (volume && preGoneSample && generation === samplingGeneration) { + // Why only its own tick qualifies: a statfs that outlived its tick carries a + // pre-storm volume number, and the latch makes that lag unbounded. It still + // ships beside its age, but it may not decide the verdict. + // Why the tick counter and not sample identity: a tick whose host read fails + // leaves the sample object in place, so identity alone reads as co-timed. + const coTimed = issuedOnTick === sampleTick + preGoneSample = { + ...preGoneSample, + details: withSwapVolumeFreeSpace(preGoneSample.details, volume, process.platform, coTimed), + swapVolumeSampledAtMs: issuedFor?.sampledAtMs + } + } + } catch { + // Why: the memory reading is already committed and stands on its own. + } finally { + swapVolumeReadInFlight = false + } +} + +export async function samplePreGoneSystemMemory(nowMs: number = Date.now()): Promise { + // Why commit before awaiting: the volume read is a statfs, and under the very + // paging storm this targets it is slowest — it must never delay, or (via an + // in-flight latch) skip, the cheap synchronous host reading. + const tick = ++sampleTick + if (!commitHostMemorySample(nowMs)) { + return + } + await mergeSwapVolumeFreeSpace(tick) +} + +export function startPreGoneSystemMemorySampling( + intervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS +): void { + if (preGoneTimer) { + return + } + void samplePreGoneSystemMemory() + preGoneTimer = setInterval(() => void samplePreGoneSystemMemory(), intervalMs) + preGoneTimer.unref?.() +} + +export function resetPreGoneSystemMemorySamplingForTest(): void { + if (preGoneTimer) { + clearInterval(preGoneTimer) + } + preGoneTimer = null + preGoneSample = null + swapVolumeReadInFlight = false + // Why bump: an already-awaited volume read must not repopulate a reset sample. + samplingGeneration += 1 +} + +/** Keyed as `systemMemoryPreGone*` so a scan over the `systemMemory` family sees both reads. */ +export function preGoneSystemMemoryDetails(nowMs: number): CrashReportDetails { + if (!preGoneSample) { + return {} + } + const details: CrashReportDetails = { + [`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSampleAgeMs`]: Math.max( + 0, + nowMs - preGoneSample.sampledAtMs + ) + } + // Why its own age: the volume read resolves out of band, so it can be older + // than the memory reading printed beside it, and that gap must be readable. + if (preGoneSample.swapVolumeSampledAtMs !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSwapVolumeAgeMs`] = Math.max( + 0, + nowMs - preGoneSample.swapVolumeSampledAtMs + ) + } + for (const [key, value] of Object.entries(preGoneSample.details)) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGone${key.slice(SYSTEM_MEMORY_KEY_PREFIX.length)}`] = + value + } + return details +} diff --git a/src/main/crash-reporting/process-gone-diagnostics.test.ts b/src/main/crash-reporting/process-gone-diagnostics.test.ts index e31645865d8..6a6ec410733 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.test.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.test.ts @@ -2,11 +2,11 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { buildProcessGoneCrashDetails, collectProcessGoneMetricDetails, - resetPreGoneProcessMetricsSamplingForTest, + resetPreGoneCrashSamplingForTest, samplePreGoneProcessMetrics, - startPreGoneProcessMetricsSampling + startPreGoneCrashSampling } from './process-gone-diagnostics' -import { setSystemMemoryInfoReaderForTest } from './gone-time-system-memory' +import { setSystemMemoryInfoReaderForTest } from './system-memory-details' type MetricFixture = { pid?: number @@ -27,7 +27,7 @@ vi.mock('electron', () => ({ describe('process gone diagnostics', () => { beforeEach(() => { - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() setSystemMemoryInfoReaderForTest(null) }) @@ -141,8 +141,8 @@ describe('process gone diagnostics', () => { appMetricsMock.mockReturnValue([ { pid: 30, type: 'Tab', memory: { workingSetSize: 1024 * 100 } } ]) - startPreGoneProcessMetricsSampling(1_000) - startPreGoneProcessMetricsSampling(1_000) + startPreGoneCrashSampling(1_000) + startPreGoneCrashSampling(1_000) // A crash inside the first interval already has a sample to draw from. expect(buildProcessGoneCrashDetails({}, 'renderer')).toMatchObject({ @@ -582,12 +582,12 @@ describe('process gone diagnostics', () => { it("arms an unref'd interval so sampling never holds the event loop open", () => { const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') try { - startPreGoneProcessMetricsSampling(60_000) + startPreGoneCrashSampling(60_000) const timer = setIntervalSpy.mock.results[0]?.value as NodeJS.Timeout expect(timer.hasRef()).toBe(false) } finally { setIntervalSpy.mockRestore() - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() } }) @@ -641,7 +641,7 @@ describe('process gone diagnostics', () => { expect(details.systemMemoryTotalMB).toBe(16_384) }) - it('samples system memory at gone time but never into the pre-gone snapshot', () => { + it('samples system memory at gone time but never into the processMetrics family', () => { appMetricsMock.mockReturnValue([{ pid: 1, type: 'Browser', memory: { workingSetSize: 0 } }]) samplePreGoneProcessMetrics() setSystemMemoryInfoReaderForTest(() => ({ @@ -658,7 +658,9 @@ describe('process gone diagnostics', () => { systemMemorySwapTotalMB: 8_192, systemMemorySwapFreeMB: 40 }) - expect(details.processMetricsPreGoneSystemMemoryTotalMB).toBeUndefined() + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) }) it('leaves records unflagged when the crashed bucket is still populated', () => { diff --git a/src/main/crash-reporting/process-gone-diagnostics.ts b/src/main/crash-reporting/process-gone-diagnostics.ts index d0bb380a2b6..bf0d735a2c7 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.ts @@ -3,7 +3,13 @@ import { sanitizeCrashReportDetails, type CrashReportDetailValue } from '../../shared/crash-reporting' -import { getSystemMemoryAtGoneDetails, memoryKBFieldMB } from './gone-time-system-memory' +import { getSystemMemoryDetails, memoryKBFieldMB } from './system-memory-details' +import { + PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS, + preGoneSystemMemoryDetails, + resetPreGoneSystemMemorySamplingForTest, + startPreGoneSystemMemorySampling +} from './pre-gone-host-memory' type ProcessMetricLike = { pid?: unknown @@ -204,8 +210,9 @@ export function samplePreGoneProcessMetrics(nowMs: number = Date.now()): void { } } -export function startPreGoneProcessMetricsSampling( - intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS +export function startPreGoneCrashSampling( + intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS, + systemMemoryIntervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS ): void { if (preGoneSampleTimer) { return @@ -213,14 +220,16 @@ export function startPreGoneProcessMetricsSampling( samplePreGoneProcessMetrics() preGoneSampleTimer = setInterval(() => samplePreGoneProcessMetrics(), intervalMs) preGoneSampleTimer.unref?.() + startPreGoneSystemMemorySampling(systemMemoryIntervalMs) } -export function resetPreGoneProcessMetricsSamplingForTest(): void { +export function resetPreGoneCrashSamplingForTest(): void { if (preGoneSampleTimer) { clearInterval(preGoneSampleTimer) } preGoneSampleTimer = null preGoneSample = null + resetPreGoneSystemMemorySamplingForTest() } const PROCESS_METRICS_KEY_PREFIX = 'processMetrics' @@ -271,7 +280,7 @@ export function buildProcessGoneCrashDetails( const crashDetails: CrashReportDetails = { ...sanitizedDetails, ...liveMetricDetails, - ...getSystemMemoryAtGoneDetails() + ...getSystemMemoryDetails() } // Why: with the crasher gone, Largest names a survivor — flag that so the // live buckets are read as "everyone else", not as the crashed process. @@ -290,8 +299,10 @@ export function buildProcessGoneCrashDetails( if (liveMetricDetails[crashedBucketCountKey] === 0 || sampledSameBucketProcessVanished) { crashDetails.processMetricsCrashedProcessAbsent = true } + const nowMs = Date.now() if (preGoneSample) { - Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, Date.now())) + Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, nowMs)) } + Object.assign(crashDetails, preGoneSystemMemoryDetails(nowMs)) return crashDetails } diff --git a/src/main/crash-reporting/swap-volume-free-space.ts b/src/main/crash-reporting/swap-volume-free-space.ts new file mode 100644 index 00000000000..3ad40b7629b --- /dev/null +++ b/src/main/crash-reporting/swap-volume-free-space.ts @@ -0,0 +1,67 @@ +import { statfs } from 'node:fs/promises' +import path from 'node:path' + +// Why: a system-managed Windows pagefile — and a macOS swapfile — only grows +// into free space on its own volume, so low available commit is a REFUSED +// allocation only when that volume is full too. Linux is excluded on purpose: +// its swap is a fixed partition, a fixed-size swapfile, or zram, none of which +// grow into root-fs free space, so the number would read as headroom that +// cannot exist. The measured volume ships alongside because Windows only names +// the DEFAULT pagefile drive; a relocated pagefile lives elsewhere. + +const BYTES_PER_MB = 1024 * 1024 + +export type SwapVolumeFreeSpace = { + freeMB: number + /** Which volume was measured, separator-trimmed so redaction sees no path. */ + volume: string +} + +type SwapVolumeFreeSpaceReader = ( + platform: NodeJS.Platform +) => Promise + +function swapVolumeAnchor(platform: NodeJS.Platform): string | undefined { + if (platform === 'win32') { + const anchor = process.env.SystemRoot || process.env.SystemDrive + return anchor ? path.parse(anchor).root || anchor : undefined + } + return platform === 'darwin' ? path.sep : undefined +} + +function volumeLabel(root: string): string { + const trimmed = root.replace(/[\\/]+$/, '') + return trimmed.length > 0 ? trimmed : root +} + +async function statfsSwapVolumeFreeSpace( + platform: NodeJS.Platform +): Promise { + const root = swapVolumeAnchor(platform) + if (!root) { + return undefined + } + try { + const stats = await statfs(root) + const bytes = Number(stats.bsize) * Number(stats.bavail) + return Number.isFinite(bytes) + ? { freeMB: Math.round(Math.max(0, bytes) / BYTES_PER_MB), volume: volumeLabel(root) } + : undefined + } catch { + return undefined + } +} + +let swapVolumeFreeSpaceReader: SwapVolumeFreeSpaceReader = statfsSwapVolumeFreeSpace + +export function setSwapVolumeFreeSpaceReaderForTest( + reader: SwapVolumeFreeSpaceReader | null +): void { + swapVolumeFreeSpaceReader = reader ?? statfsSwapVolumeFreeSpace +} + +export function readSwapVolumeFreeSpace( + platform: NodeJS.Platform = process.platform +): Promise { + return swapVolumeFreeSpaceReader(platform) +} diff --git a/src/main/crash-reporting/system-memory-details.ts b/src/main/crash-reporting/system-memory-details.ts new file mode 100644 index 00000000000..1f2cf556faa --- /dev/null +++ b/src/main/crash-reporting/system-memory-details.ts @@ -0,0 +1,161 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import type { SwapVolumeFreeSpace } from './swap-volume-free-space' + +// ─── Host system memory for crash reports ─────────────────────────── +// Why: the system outlives the crashed process, so this IS sampleable at +// process-gone — it separates "renderer grew huge" from "machine out of +// memory/commit", which the per-process buckets alone cannot. The gone-time +// caller reads AFTER the corpse returned its pages, so free/swapFree read +// healthier than at kill time; the pre-gone sampler carries a live reading past +// that. +// Every reading is labelled `systemMemoryPressureSignal` so no report can be +// read as a pressure verdict the platform never gave: +// win32 — swapFree is MEMORYSTATUSEX.ullAvailPageFile, i.e. available +// COMMIT, which pagefile growth can heal (a 127 MB commit floor healed to +// 2029 MB mid-hold on the win-lowspec repro, killing nothing). Free space on +// the swap volume does NOT establish that it could: a fixed-size or disabled +// pagefile grows into no amount of empty disk, its maximum is unreadable +// here (needs a registry read), and the measured volume is only the DEFAULT +// pagefile drive. So a co-timed volume reading is context beside the commit +// number — `available-commit-volume-cotimed` — never a verdict. The one +// decisive win32 case is a commit limit at or below RAM: no pagefile exists +// to grow, so the floor cannot heal (`available-commit-hard-capped`). +// linux — MemAvailable is the real signal; MemFree is not (it excludes page +// cache and other reclaimable memory). +// darwin — none. `free` stays low on healthy machines and +// fileBacked/purgeable are only a reclaimability proxy. The real signal +// needs `memory_pressure -Q`; Orca's reader for it +// (src/main/memory/host-memory.ts) is on-demand, and spawning a subprocess +// on a 10 s app-lifetime timer costs more than the gap it closes. + +type CrashReportDetails = Record + +export const SYSTEM_MEMORY_KEY_PREFIX = 'systemMemory' + +export function memoryKBFieldMB(value: unknown): number | undefined { + const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined + return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) +} + +type SystemMemoryInfoLike = { + total?: unknown + free?: unknown + available?: unknown + swapTotal?: unknown + swapFree?: unknown + fileBacked?: unknown + purgeable?: unknown +} + +type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null + +/** How far this reading may be read as a "was the host under pressure" verdict. */ +export type SystemMemoryPressureSignal = + | 'available-commit-hard-capped' + | 'available-commit-volume-cotimed' + | 'available-commit-unqualified' + | 'mem-available' + | 'none' + +function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { + const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) + .getSystemMemoryInfo + if (typeof read !== 'function') { + return null + } + try { + return read.call(process) + } catch { + return null + } +} + +let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo + +export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { + systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo +} + +function numericDetail(details: CrashReportDetails, suffix: string): number | undefined { + const value = details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] + return typeof value === 'number' ? value : undefined +} + +/** Windows commit limit = RAM + pagefile, so a limit at or below RAM has no pagefile behind it. */ +function pagefileBacksCommit(details: CrashReportDetails): boolean | undefined { + const total = numericDetail(details, 'TotalMB') + const swapTotal = numericDetail(details, 'SwapTotalMB') + return total === undefined || swapTotal === undefined ? undefined : swapTotal > total +} + +function pressureSignal( + platform: NodeJS.Platform, + details: CrashReportDetails, + volumeCoTimed = true +): SystemMemoryPressureSignal { + if (platform === 'win32' && `${SYSTEM_MEMORY_KEY_PREFIX}SwapFreeMB` in details) { + if (pagefileBacksCommit(details) === false) { + return 'available-commit-hard-capped' + } + return volumeCoTimed && `${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB` in details + ? 'available-commit-volume-cotimed' + : 'available-commit-unqualified' + } + if (platform === 'linux' && `${SYSTEM_MEMORY_KEY_PREFIX}AvailableMB` in details) { + return 'mem-available' + } + return 'none' +} + +export function getSystemMemoryDetails( + platform: NodeJS.Platform = process.platform +): CrashReportDetails { + const info = systemMemoryInfoReader() + if (!info) { + return {} + } + const details: CrashReportDetails = {} + const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ + ['total', 'TotalMB'], + ['free', 'FreeMB'], + ['available', 'AvailableMB'], + ['swapTotal', 'SwapTotalMB'], + ['swapFree', 'SwapFreeMB'], + ['fileBacked', 'FileBackedMB'], + ['purgeable', 'PurgeableMB'] + ] + for (const [field, suffix] of fields) { + const mb = memoryKBFieldMB(info[field]) + if (mb !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] = mb + } + } + details[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, details) + return details +} + +/** + * Merges the statfs-derived volume datum, which needs an await and so is only + * reachable from the periodic sampler, and relabels the reading it sits beside. + * + * `coTimed` false means the statfs outlived the tick that issued it, so this + * volume number and the commit number beside it describe different moments — + * during a pagefile-growth storm that is exactly when they diverge, and a + * pre-storm 40 GB printed next to 200 MB of commit reads as "the pagefile had + * room", the opposite conclusion. The datum still ships (with its own age), but + * only a co-timed one is named in the label. + */ +export function withSwapVolumeFreeSpace( + details: CrashReportDetails, + volume: SwapVolumeFreeSpace, + platform: NodeJS.Platform = process.platform, + coTimed = true +): CrashReportDetails { + const merged: CrashReportDetails = { + ...details, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`]: volume.freeMB, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`]: volume.volume + } + merged[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, merged, coTimed) + return merged +} diff --git a/src/main/git/status.test.ts b/src/main/git/status.test.ts index 4675e3a57a2..7b73739cd3f 100644 --- a/src/main/git/status.test.ts +++ b/src/main/git/status.test.ts @@ -77,6 +77,13 @@ describe('getStatus', () => { gitExecFileAsyncMock.mockResolvedValue({ stdout: '' }) }) + /** `access` targets outside the git dir — i.e. working-tree probes, not conflict-marker reads. */ + function conflictFileProbes(): string[] { + return accessMock.mock.calls + .map(([target]) => String(target).replaceAll('\\', '/')) + .filter((target) => !target.includes('/.git/')) + } + it('parses unmerged porcelain v2 entries into unresolved conflict rows', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') accessMock.mockImplementation(async (target: string) => { @@ -104,11 +111,12 @@ describe('getStatus', () => { ]) }) - it('maps deleted conflicts to deleted when the working tree file is absent', async () => { + // The 7th field of a `u` record is the working-tree mode; `000000` is how Git reports an absent path. + it('maps deleted conflicts to deleted from the porcelain working-tree mode', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: - 'u UD N... 100644 100644 000000 100644 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/deleted.ts\n' + 'u UD N... 100644 100644 000000 000000 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/deleted.ts\n' }) const result = await getStatus('/repo') @@ -120,10 +128,12 @@ describe('getStatus', () => { conflictKind: 'deleted_by_them', conflictStatus: 'unresolved' }) + expect(conflictFileProbes()).toEqual([]) }) - it('falls back to modified when the working-tree probe fails for a non-absence reason', async () => { + it('never re-probes the working tree for a conflict row, whatever the filesystem would say', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') + // Every probe fails ENOENT (beforeEach) or EIO — neither may reach the row's status. accessMock.mockRejectedValue(Object.assign(new Error('EIO'), { code: 'EIO' })) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: @@ -134,19 +144,14 @@ describe('getStatus', () => { expect(result.entries[0]?.status).toBe('modified') expect(result.entries[0]?.conflictKind).toBe('added_by_us') + expect(conflictFileProbes()).toEqual([]) }) // Why both cases normalize separators: git reports the worktree in the WSL guest namespace, and // the assertion is about which path is probed, not which separator this host's `path` emits. - it('probes the conflict working tree through the distro spelling on Windows', async () => { + it('resolves a WSL conflict row without crossing the 9p share', async () => { const platformSpy = vi.spyOn(process, 'platform', 'get').mockReturnValue('win32') readFileMock.mockResolvedValue('gitdir: /home/me/repo/.git/worktrees/feature\n') - accessMock.mockImplementation(async (target: string) => { - if (String(target).endsWith('new.ts')) { - return undefined - } - throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) - }) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: 'u DU N... 100644 100644 100644 100644 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/new.ts\n' @@ -156,10 +161,12 @@ describe('getStatus', () => { const result = await getStatus('/home/me/repo/feature', { wslDistro: 'Ubuntu' }) const probed = accessMock.mock.calls.map(([target]) => String(target).replaceAll('\\', '/')) - expect(probed).toContain('//wsl.localhost/Ubuntu/home/me/repo/feature/src/new.ts') + // No `\\wsl.localhost` round trip per conflict row: the porcelain `mW` field already answered. + expect(probed).not.toContain('//wsl.localhost/Ubuntu/home/me/repo/feature/src/new.ts') + expect(conflictFileProbes()).toEqual([]) expect(result.entries[0]?.status).toBe('modified') expect(result.entries[0]?.conflictKind).toBe('deleted_by_us') - // The conflict-marker probes travel the same way. + // The conflict-marker probes still travel through the distro spelling. expect( probed.filter((target) => target.startsWith('//wsl.localhost/Ubuntu/home/me/repo/.git/worktrees/feature/') diff --git a/src/main/host-tree-removal-asar.electron.test.ts b/src/main/host-tree-removal-asar.electron.test.ts new file mode 100644 index 00000000000..a41a55035ac --- /dev/null +++ b/src/main/host-tree-removal-asar.electron.test.ts @@ -0,0 +1,132 @@ +import { spawnSync } from 'node:child_process' +import { + copyFileSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + writeFileSync +} from 'node:fs' +import { createRequire, isBuiltin } from 'node:module' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import { afterAll, describe, expect, it } from 'vitest' +import { removeTreeSync } from '../shared/windows-transient-lock-removal' + +/** + * Why the real binary: Electron patches `fs` so a `*.asar` file reports `isDirectory() === true`, so + * a recursive `rm` descends into the archive, `rmdir`s a real file, and fails the parent with + * ENOTEMPTY. Plain Node has no such shim, so no in-process unit test can reproduce it — and every + * worktree that has run `pnpm install` carries a `default_app.asar`, which is what stranded 267 + * trash entries on the reporting machine. This runs the shipped `removeHostTree` against a real + * archive under the real binary. + */ +const requireFromTest = createRequire(import.meta.url) +const electronBinary = requireFromTest('electron') as string +const electronDist = join(dirname(requireFromTest.resolve('electron/package.json')), 'dist') +const FIXTURE_ASAR = [ + join(electronDist, 'Electron.app/Contents/Resources/default_app.asar'), + join(electronDist, 'resources/default_app.asar') +].find((candidate) => existsSync(candidate)) + +// Mirrors the residue reported on the failing machine, down to the depth of the blocking leaf. +const ENTRY_NAME = 'wt-1700000000000-abcdef01' +const ASAR_PARENT = 'node_modules/.pnpm/electron/node_modules/electron/dist/App/Contents/Resources' + +const roots: string[] = [] + +afterAll(() => { + for (const root of roots) { + try { + removeTreeSync(root) + } catch { + // A fixture the shim strands is exactly what this file is about; never fail teardown on it. + } + } +}) + +type ProbeResult = { failure: string | null; residue: string[] } + +function buildDriver(bundlePath: string, target: string, resultPath: string): string { + return [ + `const fs = require('node:fs')`, + `const { removeHostTree } = require(${JSON.stringify(bundlePath)})`, + // Why noAsar for the read-back: the shim would report the stranded archive as a directory here + // too, so the residue listing has to be taken with real filesystem semantics. + `const withoutAsar = (fn) => { const prev = process.noAsar; process.noAsar = true; try { return fn() } finally { process.noAsar = prev } }`, + `;(async () => {`, + ` let failure = null`, + ` try { await removeHostTree(${JSON.stringify(target)}) } catch (error) { failure = error.code ?? String(error) }`, + ` const residue = withoutAsar(() => fs.existsSync(${JSON.stringify(target)})`, + ` ? fs.readdirSync(${JSON.stringify(target)}, { recursive: true }).map(String)`, + ` : [])`, + ` fs.writeFileSync(${JSON.stringify(resultPath)}, JSON.stringify({ failure, residue }))`, + `})()` + ].join('\n') +} + +async function bundleHostTreeRemoval(outFile: string): Promise { + const { build } = await import('vite') + const result = await build({ + root: process.cwd(), + configFile: false, + logLevel: 'error', + build: { + write: false, + minify: false, + ssr: true, + rollupOptions: { + input: 'src/main/host-tree-removal.ts', + // Why mirror `isExternalMainModule` from electron.vite.config.ts exactly — CJS, and + // `original-fs` deliberately *not* externalized: the shipped bundle does not list it either, + // so if the archive-aware `rm` ever became a static import (or the bundler learned to fold + // `createRequire(...)('original-fs')`) production would silently degrade to the shimmed `fs` + // while a test that pre-externalized it kept passing. + output: { format: 'cjs' }, + external: (id: string) => isBuiltin(id) || id === 'electron' || id.startsWith('electron/') + } + } + }) + const output = (Array.isArray(result) ? result[0] : result) as { output: { code?: string }[] } + const code = output.output[0]?.code + expect(typeof code).toBe('string') + writeFileSync(outFile, code as string, 'utf8') +} + +function buildStrandedTree(root: string): string { + const target = join(root, ENTRY_NAME) + const asarParent = join(target, ...ASAR_PARENT.split('/')) + mkdirSync(asarParent, { recursive: true }) + copyFileSync(FIXTURE_ASAR as string, join(asarParent, 'default_app.asar')) + writeFileSync(join(asarParent, 'plain.txt'), 'x', 'utf8') + return target +} + +describe('removeHostTree against a tree holding an asar archive', () => { + it.runIf(FIXTURE_ASAR)( + 'removes the whole tree under the real Electron binary', + async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-host-tree-asar-')) + roots.push(root) + const bundlePath = join(root, 'host-tree-removal.cjs') + await bundleHostTreeRemoval(bundlePath) + const target = buildStrandedTree(root) + const resultPath = join(root, 'result.json') + const driverPath = join(root, 'driver.cjs') + writeFileSync(driverPath, buildDriver(bundlePath, target, resultPath), 'utf8') + + const run = spawnSync(electronBinary, [driverPath], { + encoding: 'utf8', + env: { ...process.env, ELECTRON_RUN_AS_NODE: '1' }, + timeout: 60_000 + }) + expect(run.status, run.stderr?.slice(-2000)).toBe(0) + + const probe = JSON.parse(readFileSync(resultPath, 'utf8')) as ProbeResult + // Without an asar-transparent `rm` this is `ENOTEMPTY` and the residue stops at the archive, + // on every attempt, forever — it is not a race a retry can win. + expect(probe).toEqual({ failure: null, residue: [] }) + }, + 120_000 + ) +}) diff --git a/src/main/host-tree-removal.ts b/src/main/host-tree-removal.ts index a5d5d447956..f789a9861d0 100644 --- a/src/main/host-tree-removal.ts +++ b/src/main/host-tree-removal.ts @@ -1,10 +1,12 @@ // Why: every recursive host delete Orca performs (worktrees, terminal history, quarantined recovery -// generations) hits the same Windows stickiness — AV/indexers/late handle releases surface transient -// EBUSY/ENOTEMPTY/EPERM on a tree Node just emptied. One helper so no call site forgets the retries. +// generations) hits the same two hazards, so one helper exists so no call site forgets either. +// Windows stickiness — AV/indexers/late handle releases surface transient EBUSY/ENOTEMPTY/EPERM on a +// tree Node just emptied — and Electron's asar shim, which strands any tree holding a `*.asar` +// (see `asar-transparent-fs`). -import { rm } from 'node:fs/promises' import { win32 } from 'node:path' import { setTimeout as delay } from 'node:timers/promises' +import { rm } from './asar-transparent-fs' import { isWindowsAbsolutePathLike } from '../shared/cross-platform-path' import { isWslUncPath } from '../shared/wsl-paths' import { transientLockRemovalOptions } from '../shared/windows-transient-lock-removal' diff --git a/src/main/host/deferred-secret-protection-report.ts b/src/main/host/deferred-secret-protection-report.ts index 8f5be1fb047..7b4ea14367c 100644 --- a/src/main/host/deferred-secret-protection-report.ts +++ b/src/main/host/deferred-secret-protection-report.ts @@ -1,4 +1,4 @@ -import { app, type BrowserWindow } from 'electron' +import { runAfterFirstWindowShown } from '../startup/first-window-deferral' import { reportSecretProtectionGap } from './secret-protection-report' /** @@ -72,21 +72,5 @@ export function scheduleSecretProtectionGapReport({ } } - let ran = false - const run = (): void => { - if (ran) { - return - } - ran = true - clearTimeout(fallback) - // Why setImmediate: keep the blocking keyring probe off the event handler that - // reveals the window, so the reveal paints first. - setImmediate(report) - } - - const fallback = setTimeout(run, REPORT_FALLBACK_MS) - fallback.unref?.() - app.once('browser-window-created', (_event: Electron.Event, window: BrowserWindow) => { - window.once('ready-to-show', run) - }) + runAfterFirstWindowShown(report, REPORT_FALLBACK_MS) } diff --git a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts index 97e0a8f9d65..8d126f557b5 100644 --- a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts +++ b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts @@ -49,6 +49,7 @@ function recordRendererBreadcrumbTrace( const DUPLICATE_TAB_OWNER_BREADCRUMB = 'terminal_tab_id_owned_by_multiple_worktrees' const PARK_VERDICT_CHURN_BREADCRUMB = 'terminal_park_verdict_churn' const REACT_COMMIT_CASCADE_BREADCRUMB = 'react_commit_cascade' +const REPLAY_GUARD_WEDGED_BREADCRUMB = 'terminal_replay_guard_wedged_release' const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ 'renderer_error', 'renderer_unhandled_rejection', @@ -56,6 +57,7 @@ const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ DUPLICATE_TAB_OWNER_BREADCRUMB, PARK_VERDICT_CHURN_BREADCRUMB, REACT_COMMIT_CASCADE_BREADCRUMB, + REPLAY_GUARD_WEDGED_BREADCRUMB, TERMINAL_WEBGL_DIAGNOSTIC_BREADCRUMB ]) const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 @@ -69,6 +71,11 @@ const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 // 30-entry ring to two such bursts. `suppressedSinceLast` keeps the pane count // — the only signal these carry — in one slot. const NAME_ONLY_COALESCED_BREADCRUMB_NAMES = new Set(['terminal_safe_fit_retry_exhausted']) +// Why: the 30-slot ring is the scarce sink; the durable span stream is not. For +// bounded-rate pane telemetry whose multiplicity is the whole signal, spans are the +// only place a burst survives the restart that clears the ring, so coalesce the ring +// but keep every event's span. +const PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES = new Set([REPLAY_GUARD_WEDGED_BREADCRUMB]) function rendererBreadcrumbCoalesceKey( name: string, @@ -77,6 +84,13 @@ function rendererBreadcrumbCoalesceKey( if (NAME_ONLY_COALESCED_BREADCRUMB_NAMES.has(name)) { return name } + // Why presence and not value: `ptyId`/`tabIdHash` are absent on the restore call + // site (layout-serialization restoreScrollbackBuffers) and present on reattach, so + // their presence is the call-site identity a mixed burst would otherwise lose. Four + // slots per storm at most, regardless of pane count. + if (name === REPLAY_GUARD_WEDGED_BREADCRUMB) { + return `${name}:${data?.ptyId ? 'pty' : ''}:${data?.tabIdHash ? 'tab' : ''}` + } // Why trigger and not name alone: `burst` means damping engaged a commit // short of React #185, `window` means slow benign churn. Collapsing them // would drop the near-crash signal into a slow-churn slot. Still bounded — @@ -191,9 +205,13 @@ export function recordRendererBreadcrumbFromRenderer( minIntervalMs: RENDERER_BREADCRUMB_COALESCE_MS, ...(origin ? { origin } : {}) }) - // Why: tracing every suppressed duplicate would preserve the same - // serialization and disk churn that breadcrumb coalescing removes. - if (coalesceResult) { + if (PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES.has(args.name)) { + // Why the raw data: every event already gets its own span, so folding the ring's + // running count in here would double-count in any span-stream total. + recordRendererBreadcrumbTrace(args.name, data) + } else if (coalesceResult) { + // Why gated: tracing every suppressed duplicate would preserve the same + // serialization and disk churn that breadcrumb coalescing removes. recordRendererBreadcrumbTrace( args.name, coalesceResult.suppressedSinceLast > 0 diff --git a/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts new file mode 100644 index 00000000000..e3823f0b313 --- /dev/null +++ b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts @@ -0,0 +1,128 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { + clearCrashBreadcrumbsForTest, + getCrashBreadcrumbSnapshot, + recordCrashBreadcrumb +} from '../crash-reporting/crash-breadcrumb-store' +import { recordRendererBreadcrumbFromRenderer } from './crash-reporting-renderer-breadcrumbs' + +type SpanOptions = { attributes: Record } +const startSpanMock = vi.fn((_name: string, _options: SpanOptions) => ({ end: () => {} })) +vi.mock('../observability/tracer', () => ({ + startSpan: (name: string, options: SpanOptions) => startSpanMock(name, options) +})) + +const WEDGE_BREADCRUMB = 'terminal_replay_guard_wedged_release' + +/** Reattach-path shape: identity-bearing (`tabIdHash`, optionally `ptyId`). */ +function emitReattachWedge(pane: number, withPtyId = false): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { + paneId: pane, + leafIdHash: `leaf${String(pane).padStart(5, '0')}`, + tabIdHash: `tab${String(pane).padStart(6, '0')}`, + worktreeIdHash: 'caa15fa9', + ...(withPtyId ? { ptyId: `…@@pty-${pane}` } : {}) + } + }) +} + +/** Restore-path shape (restoreScrollbackBuffers): no tabIdHash, no ptyId. */ +function emitRestoreWedge(pane: number): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { paneId: pane, leafIdHash: `leaf${String(pane).padStart(5, '0')}` } + }) +} + +function wedgeCrumbs(): ReturnType { + return getCrashBreadcrumbSnapshot().filter((entry) => entry.name === WEDGE_BREADCRUMB) +} + +function wedgeSpanCount(): number { + return startSpanMock.mock.calls.filter( + (call) => call[1].attributes['breadcrumb.name'] === WEDGE_BREADCRUMB + ).length +} + +beforeEach(() => { + startSpanMock.mockClear() +}) + +afterEach(() => { + clearCrashBreadcrumbsForTest() +}) + +// One mount/reveal/wake transition expires every in-flight replay write at once, so +// the burst reaches the 30-slot ring as N distinct entries. Field span streams measure +// bursts of 26 in 0.96s and 62 over 85s. No captured report in the 09-02 corpus shows +// a ring that actually drained — all nine bursts predate their report's ring window — +// so this bounds a demonstrated hazard, not an observed loss, and must not cost the +// durable span evidence that did carry those bursts. +describe('replay-guard wedge burst against the fixed-size breadcrumb ring', () => { + it('costs one ring slot per call site and preserves the pre-crash trail', () => { + for (let index = 0; index < 10; index += 1) { + recordCrashBreadcrumb(`pre_crash_evidence_${index}`, { index }) + } + + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + const snapshot = getCrashBreadcrumbSnapshot() + expect(snapshot.filter((entry) => entry.name.startsWith('pre_crash_evidence_'))).toHaveLength( + 10 + ) + expect(wedgeCrumbs()).toHaveLength(1) + }) + + it('carries the burst multiplicity into the ring as suppressedSinceLast', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + // 26 emissions: one owns the slot, 25 fold into it. + expect(wedgeCrumbs()[0]?.data?.suppressedSinceLast).toBe(25) + }) + + // The 121-event field corpus lives entirely in the renderer.breadcrumb span stream, + // and the ring is cleared by the restart that usually precedes the crash report, so + // ring coalescing must not suppress the per-event spans. + it('still emits one durable span per wedge event', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + expect(wedgeSpanCount()).toBe(26) + // Why no count on the span: one span per event already carries the multiplicity. + expect( + startSpanMock.mock.calls.some((call) => + JSON.stringify(call[1]).includes('suppressedSinceLast') + ) + ).toBe(false) + }) + + // Bundle 8907a508 mixes restore-path (identity-less) and reattach-path crumbs in one + // window; name-only keying would report only the last one's shape. + it('keeps restore-path and reattach-path call sites in separate slots', () => { + emitRestoreWedge(1) + emitRestoreWedge(2) + emitReattachWedge(3) + emitReattachWedge(4, true) + + const crumbs = wedgeCrumbs() + expect(crumbs).toHaveLength(3) + expect(crumbs.map((crumb) => Boolean(crumb.data?.tabIdHash))).toEqual([false, true, true]) + expect(crumbs.map((crumb) => Boolean(crumb.data?.ptyId))).toEqual([false, false, true]) + }) + + it('bounds a many-pane burst to one slot within a call site', () => { + for (let pane = 0; pane < 40; pane += 1) { + emitReattachWedge(pane, pane % 2 === 0) + } + + expect(wedgeCrumbs()).toHaveLength(2) + }) +}) diff --git a/src/main/ipc/deferred-emoji-shortcode-dataset.test.ts b/src/main/ipc/deferred-emoji-shortcode-dataset.test.ts new file mode 100644 index 00000000000..a4d510650d6 --- /dev/null +++ b/src/main/ipc/deferred-emoji-shortcode-dataset.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it, vi } from 'vitest' +import emojiShortcodes from 'emojibase-data/en/shortcodes/emojibase.json' +import { requireEmojiShortcodeDataset } from './deferred-emoji-shortcode-dataset' + +// Lives under src/main (not next to the shared catalog) so the shared tsconfig projects stay +// free of a src/main import — the boundary emoji-shortcode-catalog.lazy.test.ts asserts on. +describe('deferred emoji shortcode dataset', () => { + it('loads the main-side dataset synchronously into an identical catalog', async () => { + vi.resetModules() + const eager = await import('../../shared/emoji-shortcode-catalog.js') + eager.setEmojiShortcodeDatasetLoader(() => emojiShortcodes) + const eagerEntries = eager.getStandardEmojiShortcodeEntries() + const eagerTransform = eager.replaceKnownEmojiWithShortcodes('ship \u{1F389} \u{1F44D}') + + vi.resetModules() + const deferred = await import('../../shared/emoji-shortcode-catalog.js') + deferred.setEmojiShortcodeDatasetLoader(requireEmojiShortcodeDataset) + + // No await between registration and first use: the require path keeps the sync contract. + expect(deferred.getStandardEmojiShortcodeEntries()).toEqual(eagerEntries) + expect(deferred.replaceKnownEmojiWithShortcodes('ship \u{1F389} \u{1F44D}')).toBe( + eagerTransform + ) + }) +}) diff --git a/src/main/ipc/deferred-emoji-shortcode-dataset.ts b/src/main/ipc/deferred-emoji-shortcode-dataset.ts new file mode 100644 index 00000000000..0d499d25750 --- /dev/null +++ b/src/main/ipc/deferred-emoji-shortcode-dataset.ts @@ -0,0 +1,13 @@ +import { createRequire } from 'node:module' +import type { EmojiShortcodeDataset } from '../../shared/emoji-shortcode-catalog' + +// Why createRequire (same reason as linear-sdk.ts): a static import inlines the 166 KB +// shortcode dataset into out/main/index.js and JSON.parses it on every launch, while only +// worktree-name sanitization ever reads it. app.asar ships no node_modules, so this bare require +// resolves out of Resources/node_modules — electron-builder.config.cjs copies exactly this file +// there (the package root is 49 MB of locale data). +const requireFromMain = createRequire(__filename) + +export function requireEmojiShortcodeDataset(): EmojiShortcodeDataset { + return requireFromMain('emojibase-data/en/shortcodes/emojibase.json') as EmojiShortcodeDataset +} diff --git a/src/main/ipc/filesystem-allowed-roots.test.ts b/src/main/ipc/filesystem-allowed-roots.test.ts new file mode 100644 index 00000000000..f94c99c5fdb --- /dev/null +++ b/src/main/ipc/filesystem-allowed-roots.test.ts @@ -0,0 +1,372 @@ +import { mkdir, mkdtemp, realpath, rm, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import type * as RepoWorktrees from '../repo-worktrees' +import { listRepoWorktreeGraph } from '../repo-worktrees' +import type * as ProjectGroupsModule from '../../shared/project-groups' +import { buildProjectGroupChildIndex, getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { isPathInsideOrEqual } from '../../shared/cross-platform-path' +import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import type { FolderWorkspace } from '../../shared/folder-workspace-types' +import type { ProjectGroup } from '../../shared/project-group-types' +import type { Project } from '../../shared/project-types' +import type { Repo } from '../../shared/repo-types' +import { getAllowedRoots } from './filesystem-allowed-roots' +import { authorizeExternalPath, resolveAuthorizedPath } from './filesystem-auth' +import { invalidateAuthorizedRootsCache } from './registered-worktree-roots-cache' +import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' + +vi.mock('../repo-worktrees', async () => { + const actual = await vi.importActual('../repo-worktrees') + return { ...actual, listRepoWorktreeGraph: vi.fn(async () => []) } +}) + +vi.mock('../../shared/project-groups', async () => { + const actual = await vi.importActual('../../shared/project-groups') + return { + ...actual, + buildProjectGroupChildIndex: vi.fn(actual.buildProjectGroupChildIndex), + getProjectGroupSubtreeIds: vi.fn(actual.getProjectGroupSubtreeIds) + } +}) + +type StoreFixture = { + repos: Repo[] + projects: Project[] + projectGroups: ProjectGroup[] + folderWorkspaces: FolderWorkspace[] + workspaceDir?: string +} + +type StoreCallCounts = { + getRepos: number + getProjects: number + getProjectGroups: number + getFolderWorkspaces: number +} + +function makeCountingStore(fixture: StoreFixture): { store: Store; counts: StoreCallCounts } { + const counts: StoreCallCounts = { + getRepos: 0, + getProjects: 0, + getProjectGroups: 0, + getFolderWorkspaces: 0 + } + const store = { + getRepos: () => { + counts.getRepos += 1 + // Match the real store, which rehydrates fresh repo objects on every read. + return fixture.repos.map((repo) => ({ ...repo })) + }, + getProjects: () => { + counts.getProjects += 1 + return fixture.projects.map((project) => ({ ...project })) + }, + getProjectGroups: () => { + counts.getProjectGroups += 1 + return fixture.projectGroups.map((group) => ({ ...group })) + }, + getFolderWorkspaces: () => { + counts.getFolderWorkspaces += 1 + return fixture.folderWorkspaces.map((workspace) => ({ ...workspace })) + }, + getSettings: () => ({ nestWorkspaces: false, workspaceDir: fixture.workspaceDir ?? '' }) + } as unknown as Store + return { store, counts } +} + +/** + * The pre-change `getAllowedRoots` algorithm, kept verbatim so the equivalence test compares the + * new root list against the old one rather than against a hand-written expectation. + */ +function referenceAllowedRoots(store: Store): string[] { + const scopeStore = store as unknown as { + getRepos: () => Repo[] + getProjectGroups?: () => ProjectGroup[] + getFolderWorkspaces?: () => FolderWorkspace[] + getSettings: () => { workspaceDir?: string; nestWorkspaces?: boolean } + } + const localRepos = scopeStore.getRepos().filter((repo) => !repo.connectionId) + const settings = scopeStore.getSettings() + + const scopeRepos = scopeStore.getRepos() + const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const isRemoteOnly = ( + folderPath: string, + projectGroupId: string, + connectionId: string | null | undefined + ): boolean => { + if (connectionId) { + return true + } + const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const candidates = scopeRepos.filter( + (repo) => + (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || + isPathInsideOrEqual(folderPath, repo.path) + ) + return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) + } + const folderScopeRoots: string[] = [] + for (const group of projectGroups) { + if (group.parentPath && !isRemoteOnly(group.parentPath, group.id, group.connectionId)) { + folderScopeRoots.push(resolve(group.parentPath)) + } + } + for (const workspace of scopeStore.getFolderWorkspaces?.() ?? []) { + const connectionId = + workspace.connectionId ?? + projectGroups.find((group) => group.id === workspace.projectGroupId)?.connectionId ?? + null + if (!isRemoteOnly(workspace.folderPath, workspace.projectGroupId, connectionId)) { + folderScopeRoots.push(resolve(workspace.folderPath)) + } + } + + const roots = [...localRepos.map((repo) => resolve(repo.path)), ...folderScopeRoots] + if (settings.workspaceDir) { + if (localRepos.length === 0) { + roots.push(resolve(settings.workspaceDir)) + } else { + for (const repo of localRepos) { + roots.push( + resolve( + computeWorkspaceRoot( + repo.path, + getWorktreePathSettings(repo, settings as never, getWorktreeMirrorDistro(store, repo)) + ) + ) + ) + } + } + } + return roots +} + +function makeRepo(overrides: Partial & Pick): Repo { + return { + displayName: overrides.id, + badgeColor: '#000000', + addedAt: 1, + kind: 'git', + ...overrides + } +} + +function makeGroup(overrides: Partial & Pick): ProjectGroup { + return { + name: overrides.id, + parentPath: null, + parentGroupId: null, + createdFrom: 'folder-scan', + tabOrder: 0, + isCollapsed: false, + color: null, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +function makeWorkspace( + overrides: Partial & Pick +): FolderWorkspace { + return { + projectGroupId: 'group-root', + name: overrides.id, + comment: '', + linkedTask: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 1, + lastActivityAt: 1, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +/** Repos, nested groups, folder workspaces (one not a git worktree), and an SSH repo. */ +function makeMixedFixture(): StoreFixture { + const repos = [ + makeRepo({ id: 'repo-local', path: '/repos/app', projectGroupId: 'group-root' }), + makeRepo({ id: 'repo-nested', path: '/repos/nested', projectGroupId: 'group-child' }), + makeRepo({ id: 'repo-folder', path: '/folders/plain', kind: 'folder' }), + makeRepo({ + id: 'repo-ssh', + path: '/remote/app', + connectionId: 'ssh-1', + projectGroupId: 'group-remote' + }) + ] + const projectGroups = [ + makeGroup({ id: 'group-root', parentPath: '/folders/root' }), + makeGroup({ id: 'group-child', parentGroupId: 'group-root', parentPath: '/folders/child' }), + makeGroup({ id: 'group-grandchild', parentGroupId: 'group-child' }), + makeGroup({ id: 'group-remote', parentPath: '/remote/scope' }), + makeGroup({ id: 'group-connection', parentPath: '/remote/via-group', connectionId: 'ssh-1' }) + ] + const folderWorkspaces = [ + makeWorkspace({ id: 'ws-git', folderPath: '/folders/root/feature' }), + // Not a git worktree: a plain folder workspace under a folder-kind repo. + makeWorkspace({ + id: 'ws-plain', + folderPath: '/folders/plain/scratch', + projectGroupId: 'group-child' + }), + makeWorkspace({ id: 'ws-remote', folderPath: '/remote/ws', projectGroupId: 'group-remote' }), + makeWorkspace({ + id: 'ws-connection', + folderPath: '/remote/direct', + projectGroupId: 'group-connection' + }), + makeWorkspace({ + id: 'ws-unlinked', + folderPath: '/folders/unlinked', + projectGroupId: 'group-orphan' + }) + ] + const projects: Project[] = [ + { + id: 'project-1', + displayName: 'App', + badgeColor: '#000000', + sourceRepoIds: ['repo-local', 'repo-nested'], + createdAt: 1, + updatedAt: 1 + }, + { + id: 'project-2', + displayName: 'Folder', + badgeColor: '#000000', + sourceRepoIds: ['repo-folder'], + createdAt: 1, + updatedAt: 1 + } + ] + return { repos, projects, projectGroups, folderWorkspaces, workspaceDir: '/workspaces' } +} + +beforeEach(() => { + invalidateAuthorizedRootsCache() + vi.mocked(buildProjectGroupChildIndex).mockClear() + vi.mocked(getProjectGroupSubtreeIds).mockClear() +}) + +describe('getAllowedRoots', () => { + it('produces the same roots as the pre-change implementation', () => { + const { store } = makeCountingStore(makeMixedFixture()) + + expect(getAllowedRoots(store)).toEqual(referenceAllowedRoots(store)) + }) + + it('reads the store once and indexes project groups once per build', () => { + const fixture = makeMixedFixture() + const { store, counts } = makeCountingStore(fixture) + + getAllowedRoots(store) + + expect.soft(counts.getRepos).toBe(1) + expect.soft(counts.getProjectGroups).toBe(1) + expect.soft(counts.getFolderWorkspaces).toBe(1) + // Batched runtime resolution scans the project list once, not once per local repo. + expect.soft(counts.getProjects).toBe(1) + // The per-scope subtree walk no longer rebuilds the parent->children index. + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(1) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) +}) + +describe('resolveAuthorizedPath allowed-root reuse', () => { + let repoRoot: string + let outsideRoot: string + let store: Store + let counts: StoreCallCounts + + beforeEach(async () => { + repoRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-allowed-roots-')) + outsideRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-outside-')) + const fixture = makeMixedFixture() + fixture.repos = [makeRepo({ id: 'repo-local', path: repoRoot }), ...fixture.repos] + fixture.projects[0]!.sourceRepoIds = ['repo-local'] + ;({ store, counts } = makeCountingStore(fixture)) + }) + + afterEach(async () => { + await rm(repoRoot, { recursive: true, force: true }) + await rm(outsideRoot, { recursive: true, force: true }) + }) + + it('builds the allowed-root list once per call across repeated reads', async () => { + const dirPath = join(repoRoot, 'src') + await mkdir(dirPath) + await writeFile(join(dirPath, 'index.ts'), 'export {}\n') + const callCount = 5 + + for (let index = 0; index < callCount; index += 1) { + await resolveAuthorizedPath(dirPath, store) + await resolveAuthorizedPath(join(dirPath, 'index.ts'), store) + } + + const buildCount = callCount * 2 + // One build per authorization, not one per raw-path check plus one per realpath check. + expect.soft(counts.getFolderWorkspaces).toBe(buildCount) + expect.soft(counts.getRepos).toBe(buildCount) + expect.soft(counts.getProjects).toBe(buildCount) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(buildCount) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) + + // Why (both symlink cases): creating a symlink on Windows needs elevation or + // Developer Mode, so these would fail EPERM in setup rather than exercise the + // escape check. Every non-symlink case still runs there. + it.skipIf(process.platform === 'win32')( + 'still refuses a symlink that escapes every allowed root', + async () => { + const secret = join(outsideRoot, 'secret.txt') + await writeFile(secret, 'secret\n') + const escape = join(repoRoot, 'escape.txt') + await symlink(secret, escape) + + await expect(resolveAuthorizedPath(escape, store)).rejects.toThrow('Access denied') + expect(vi.mocked(listRepoWorktreeGraph)).toHaveBeenCalled() + } + ) + + it('builds no allowed-root list at all for a granted external path', async () => { + const external = join(outsideRoot, 'external.md') + await writeFile(external, 'notes\n') + authorizeExternalPath(external) + counts.getRepos = 0 + counts.getProjects = 0 + counts.getFolderWorkspaces = 0 + + for (let index = 0; index < 5; index += 1) { + await expect(resolveAuthorizedPath(external, store)).resolves.toBe(external) + } + + // The grant answers on its own; hoisting the snapshot must not turn zero builds into one per read. + expect.soft(counts.getRepos).toBe(0) + expect.soft(counts.getProjects).toBe(0) + expect.soft(counts.getFolderWorkspaces).toBe(0) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).not.toHaveBeenCalled() + }) + + it.skipIf(process.platform === 'win32')( + 'still refuses a directory symlink that escapes every allowed root', + async () => { + const outsideDir = join(outsideRoot, 'nested') + await mkdir(outsideDir) + await writeFile(join(outsideDir, 'file.txt'), 'secret\n') + const escape = join(repoRoot, 'escape-dir') + await symlink(outsideDir, escape) + + await expect(resolveAuthorizedPath(join(escape, 'file.txt'), store)).rejects.toThrow( + 'Access denied' + ) + } + ) +}) diff --git a/src/main/ipc/filesystem-allowed-roots.ts b/src/main/ipc/filesystem-allowed-roots.ts index 3cb7fe4fa55..cef249430c6 100644 --- a/src/main/ipc/filesystem-allowed-roots.ts +++ b/src/main/ipc/filesystem-allowed-roots.ts @@ -1,9 +1,16 @@ import { resolve } from 'node:path' import type { Store } from '../persistence' import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' -import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import { + getWorktreeMirrorDistroForRuntime, + resolveLocalProjectRuntimesForRepos +} from '../project-runtime-git-options' import { isPathInsideOrEqual } from '../../shared/cross-platform-path' -import { getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { + buildProjectGroupChildIndex, + collectProjectGroupSubtreeIds, + type ProjectGroupChildIndex +} from '../../shared/project-groups' import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { ProjectGroup } from '../../shared/project-group-types' import type { Repo } from '../../shared/repo-types' @@ -11,18 +18,22 @@ import type { Repo } from '../../shared/repo-types' type FolderScopeStore = Pick & Partial> +// Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. +function filterLocalRepos(repos: readonly Repo[]): Repo[] { + return repos.filter((repo) => !repo.connectionId) +} + export function getLocalRepos(store: Store) { - // Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. - return store.getRepos().filter((repo) => !repo.connectionId) + return filterLocalRepos(store.getRepos()) } function getFolderScopeCandidateRepos( folderPath: string, projectGroupId: string, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): Repo[] { - const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const groupIds = collectProjectGroupSubtreeIds(childGroupIndex, projectGroupId) return repos.filter( (repo) => (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || @@ -34,13 +45,18 @@ function isRemoteOnlyFolderScope( folderPath: string, projectGroupId: string, connectionId: string | null | undefined, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): boolean { if (connectionId) { return true } - const candidates = getFolderScopeCandidateRepos(folderPath, projectGroupId, projectGroups, repos) + const candidates = getFolderScopeCandidateRepos( + folderPath, + projectGroupId, + childGroupIndex, + repos + ) return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) } @@ -55,16 +71,22 @@ function getFolderWorkspaceConnectionId( ) } -function getLocalFolderScopeRoots(store: Store): string[] { +function getLocalFolderScopeRoots(store: Store, repos: readonly Repo[]): string[] { const scopeStore = store as FolderScopeStore - const repos = scopeStore.getRepos() // Why: many filesystem tests use narrow Store doubles; folder scopes are additive. const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const childGroupIndex = buildProjectGroupChildIndex(projectGroups) const roots: string[] = [] for (const group of projectGroups) { if ( group.parentPath && - !isRemoteOnlyFolderScope(group.parentPath, group.id, group.connectionId, projectGroups, repos) + !isRemoteOnlyFolderScope( + group.parentPath, + group.id, + group.connectionId, + childGroupIndex, + repos + ) ) { roots.push(resolve(group.parentPath)) } @@ -75,7 +97,7 @@ function getLocalFolderScopeRoots(store: Store): string[] { workspace.folderPath, workspace.projectGroupId, getFolderWorkspaceConnectionId(workspace, projectGroups), - projectGroups, + childGroupIndex, repos ) ) { @@ -86,16 +108,19 @@ function getLocalFolderScopeRoots(store: Store): string[] { } export function getAllowedRoots(store: Store): string[] { - const localRepos = getLocalRepos(store) + // Why one read: `getRepos` rehydrates every repo, and this runs twice per filesystem IPC. + const repos = store.getRepos() + const localRepos = filterLocalRepos(repos) const settings = store.getSettings() const roots = [ ...localRepos.map((repo) => resolve(repo.path)), - ...getLocalFolderScopeRoots(store) + ...getLocalFolderScopeRoots(store, repos) ] if (settings.workspaceDir) { if (localRepos.length === 0) { roots.push(resolve(settings.workspaceDir)) } else { + const projectRuntimeByRepoId = resolveLocalProjectRuntimesForRepos(store, localRepos) for (const repo of localRepos) { roots.push( resolve( @@ -104,7 +129,11 @@ export function getAllowedRoots(store: Store): string[] { // Why enriched here too: placement has to agree with the create // flow, or renderer file access is denied for a worktree Orca // just put on the WSL side. - getWorktreePathSettings(repo, settings, getWorktreeMirrorDistro(store, repo)) + getWorktreePathSettings( + repo, + settings, + getWorktreeMirrorDistroForRuntime(projectRuntimeByRepoId.get(repo.id)) + ) ) ) ) diff --git a/src/main/ipc/filesystem-auth.ts b/src/main/ipc/filesystem-auth.ts index 122617845ed..894e39945c1 100644 --- a/src/main/ipc/filesystem-auth.ts +++ b/src/main/ipc/filesystem-auth.ts @@ -43,7 +43,24 @@ export function authorizeExternalPath(targetPath: string): void { } catch {} } -export function isPathAllowed(targetPath: string, store: Store): boolean { +/** + * One allowed-root list shared by every check in a single authorization. + * + * Lazy so a path already covered by an external grant still builds nothing at all, the way it did + * before the list was hoisted out of the individual checks. + */ +type AllowedRootsSnapshot = { get: () => readonly string[] } + +function createAllowedRootsSnapshot(store: Store): AllowedRootsSnapshot { + let roots: readonly string[] | undefined + return { get: () => (roots ??= getAllowedRoots(store)) } +} + +export function isPathAllowed( + targetPath: string, + store: Store, + allowedRoots?: AllowedRootsSnapshot +): boolean { const resolvedTarget = resolve(targetPath) if (authorizedExternalPaths.has(resolvedTarget)) { return true @@ -53,7 +70,9 @@ export function isPathAllowed(targetPath: string, store: Store): boolean { return true } } - return getAllowedRoots(store).some((root) => isDescendantOrEqual(resolvedTarget, root)) + return (allowedRoots?.get() ?? getAllowedRoots(store)).some((root) => + isDescendantOrEqual(resolvedTarget, root) + ) } export type ResolveAuthorizedPathOptions = { @@ -69,7 +88,10 @@ export async function resolveAuthorizedPath( options: ResolveAuthorizedPathOptions = {} ): Promise { const resolvedTarget = resolve(targetPath) - if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store))) { + // Why: the roots depend only on store state, not on the candidate path, so one snapshot serves + // every authorization below; each candidate is still checked against it in full. + const allowedRoots = createAllowedRootsSnapshot(store) + if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store, { allowedRoots }))) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) } @@ -80,14 +102,15 @@ export async function resolveAuthorizedPath( realParent = await realpath(dirname(resolvedTarget)) } catch (error) { if (isENOENT(error)) { - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } throw error } const candidateTarget = resolve(realParent, basename(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -100,7 +123,8 @@ export async function resolveAuthorizedPath( const realTarget = resolve(await realpath(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(realTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -110,11 +134,15 @@ export async function resolveAuthorizedPath( if (!isENOENT(error)) { throw error } - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } } -async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store): Promise { +async function resolveAuthorizedMissingPath( + resolvedTarget: string, + store: Store, + allowedRoots: AllowedRootsSnapshot +): Promise { let existingAncestor = resolvedTarget const missingSegments: string[] = [] @@ -124,7 +152,8 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store const candidateTarget = resolve(realAncestor, ...missingSegments) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -148,9 +177,9 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store async function isPathAllowedIncludingRegisteredWorktrees( targetPath: string, store: Store, - options: { canonicalSourcePath?: string } = {} + options: { canonicalSourcePath?: string; allowedRoots?: AllowedRootsSnapshot } = {} ): Promise { - if (isPathAllowed(targetPath, store)) { + if (isPathAllowed(targetPath, store, options.allowedRoots)) { return true } @@ -158,7 +187,14 @@ async function isPathAllowedIncludingRegisteredWorktrees( return true } - if (await isPathAllowedByCanonicalAllowedRoot(targetPath, options.canonicalSourcePath, store)) { + if ( + await isPathAllowedByCanonicalAllowedRoot( + targetPath, + options.canonicalSourcePath, + store, + options.allowedRoots + ) + ) { return true } @@ -178,12 +214,13 @@ async function isPathAllowedIncludingRegisteredWorktrees( async function isPathAllowedByCanonicalAllowedRoot( targetPath: string, sourcePath: string | undefined, - store: Store + store: Store, + allowedRoots?: AllowedRootsSnapshot ): Promise { if (!sourcePath) { return false } - for (const root of getAllowedRoots(store)) { + for (const root of allowedRoots?.get() ?? getAllowedRoots(store)) { const resolvedRoot = resolve(root) if (!isDescendantOrEqual(sourcePath, resolvedRoot)) { continue diff --git a/src/main/ipc/orca-profile-auth-status-broadcast.ts b/src/main/ipc/orca-profile-auth-status-broadcast.ts new file mode 100644 index 00000000000..b5ac8483943 --- /dev/null +++ b/src/main/ipc/orca-profile-auth-status-broadcast.ts @@ -0,0 +1,15 @@ +import { BrowserWindow } from 'electron' +import { ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL } from '../../shared/orca-profiles' + +export function broadcastOrcaProfileAuthStatusChanged(): void { + for (const window of BrowserWindow.getAllWindows()) { + if (window.isDestroyed()) { + continue + } + try { + window.webContents.send(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL) + } catch { + // A renderer can disappear between isDestroyed() and send(). + } + } +} diff --git a/src/main/ipc/orca-profiles.ts b/src/main/ipc/orca-profiles.ts index bf1bf666270..480c6f350f9 100644 --- a/src/main/ipc/orca-profiles.ts +++ b/src/main/ipc/orca-profiles.ts @@ -45,6 +45,8 @@ import { signOutCurrentOrcaProfile } from '../orca-profiles/profile-cloud-service' import { registerOrcaProfileOrgMemberHandlers } from './orca-profile-org-members-handlers' +import { onOrcaCloudSessionInvalidated } from '../orca-profiles/profile-cloud-session-invalidation' +import { broadcastOrcaProfileAuthStatusChanged } from './orca-profile-auth-status-broadcast' type RegisterOrcaProfileHandlersOptions = { onBeforeRelaunch?: () => void | Promise @@ -178,6 +180,12 @@ export function registerOrcaProfileHandlers( getCurrentOrcaProfileAuthStatus(getProfileUserDataPath()) ) + // Why: a background refresh can revoke the session with no renderer request in + // flight, so push the change instead of waiting for the next pane to ask. + // Why not options.onAuthMutation: that hook drives the relay coordinator, which + // is the caller that just failed the refresh — re-entering it here would be a loop. + onOrcaCloudSessionInvalidated(broadcastOrcaProfileAuthStatusChanged) + ipcMain.handle( 'orcaProfiles:createLocal', (_event, args?: CreateLocalOrcaProfileArgs): CreateLocalOrcaProfileResult => { diff --git a/src/main/ipc/pty/ipc/inspect.ts b/src/main/ipc/pty/ipc/inspect.ts index c13c4242fea..041abaa7854 100644 --- a/src/main/ipc/pty/ipc/inspect.ts +++ b/src/main/ipc/pty/ipc/inspect.ts @@ -170,7 +170,10 @@ export function installPtyInspectIpcHandlers(deps: { ipcMain.handle( 'pty:inspectProcess', - async (_event, args: { id: string; expectedIncarnationId?: string }) => { + async ( + _event, + args: { id: string; expectedIncarnationId?: string; scanChildProcesses?: boolean } + ) => { // Why: same routing hazard as pty:hasPty — an unroutable id must read as client-only unverifiable, not as a local-provider answer or a raised IPC error. if (typeof args?.id !== 'string' || !args.id || args.id.startsWith('remote:')) { return clientOnlyUnverifiableInspection('terminal_gone') @@ -182,10 +185,14 @@ export function installPtyInspectIpcHandlers(deps: { if (!hasPtyProviderForInspection(args.id)) { return clientOnlyUnverifiableInspection('terminal_gone') } - return args.expectedIncarnationId - ? inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id, { - expectedIncarnationId: args.expectedIncarnationId - }) + const options = { + ...(args.expectedIncarnationId + ? { expectedIncarnationId: args.expectedIncarnationId } + : {}), + ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}) + } + return Object.keys(options).length > 0 + ? inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id, options) : inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id) } ) diff --git a/src/main/ipc/pty/runtime/queried-host-kinds.test.ts b/src/main/ipc/pty/runtime/queried-host-kinds.test.ts new file mode 100644 index 00000000000..1f7ce2459d8 --- /dev/null +++ b/src/main/ipc/pty/runtime/queried-host-kinds.test.ts @@ -0,0 +1,60 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { parseExecutionHostId } from '../../../../shared/execution-host' +import { sshProviders } from '../provider/registry' +import { listProcessesWithHostScopeFromRuntimeController } from './inventory-operations' +import type { PtyRuntimeControllerDeps } from './controller-deps' + +/** + * `hostScopeCensusIsComplete` discounts a `runtime:` host in `omittedHostIds` on the strength of + * one fact about this process: it has no paired-runtime PTY provider, so it never queried that + * host and never owed it coverage. This file pins the producer side of that fact. + * + * What it catches: a new branch here that spells a queried host `runtime:`. Every id this + * function emits is built by `toSshExecutionHostId` or is `LOCAL_EXECUTION_HOST_ID`, so a third + * shape is the observable form of "a runtime host can now answer an inventory" — at which point + * the client predicate would start calling a genuine gap complete. + * + * What it does NOT catch, so do not lean on it: a runtime-backed transport registered under an + * SSH connection id still reports as `ssh:` and passes, which is fine — the predicate only + * discounts the `runtime:` spelling. The consolidation moving the SSH path onto orcad is expected + * to look exactly like that. The other route into `queriedHostIds` is separately fenced to + * `kind === 'ssh'` in `orca-runtime-refresh-pty-worktree-records-with-controller-inventory.ts`. + */ +describe('the hosts a PTY inventory can report having queried', () => { + afterEach(() => { + sshProviders.clear() + }) + + it('emits only local and ssh spellings, never a paired-runtime one', async () => { + const listProcesses = vi.fn(async () => []) + sshProviders.set('box-1', { listProcesses } as never) + // A connection id shaped like an environment uuid still has to come back `ssh:`; the spelling + // is what the gate keys on, so a `runtime:` id appearing here is the breakage that matters. + sshProviders.set('a2478221-1d5c-4603-b8bf-b6b728eac9df', { listProcesses } as never) + + const { hostIds } = await listProcessesWithHostScopeFromRuntimeController({ + runtime: null + } as unknown as PtyRuntimeControllerDeps) + + expect(hostIds).toContain('ssh:a2478221-1d5c-4603-b8bf-b6b728eac9df') + expect(new Set(hostIds.map((hostId) => parseExecutionHostId(hostId)?.kind))).toEqual( + new Set(['local', 'ssh']) + ) + }) + + it('drops a provider that threw rather than reporting its host as queried', async () => { + sshProviders.set('box-live', { listProcesses: vi.fn(async () => []) } as never) + sshProviders.set('box-down', { + listProcesses: vi.fn(async () => { + throw new Error('relay unavailable') + }) + } as never) + + const { hostIds } = await listProcessesWithHostScopeFromRuntimeController({ + runtime: { markPtyLivenessUnverifiable: vi.fn() } + } as unknown as PtyRuntimeControllerDeps) + + expect(hostIds).toContain('ssh:box-live') + expect(hostIds).not.toContain('ssh:box-down') + }) +}) diff --git a/src/main/ipc/remote-workspace.test.ts b/src/main/ipc/remote-workspace.test.ts index 56eb4804a82..3b900e175bc 100644 --- a/src/main/ipc/remote-workspace.test.ts +++ b/src/main/ipc/remote-workspace.test.ts @@ -7,18 +7,35 @@ import type { RemoteWorkspaceSnapshot } from '../../shared/remote-workspace-types' import type { SshTarget } from '../../shared/ssh-types' +import type * as WorktreeExecutionHostResolution from '../../shared/worktree-execution-host-resolution' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' const { getActiveMultiplexerMock, getSshConnectionStoreMock, - registerRemoteWorkspaceNotificationHandlerMock + registerRemoteWorkspaceNotificationHandlerMock, + resolveWorktreeExecutionHostCalls } = vi.hoisted(() => ({ getActiveMultiplexerMock: vi.fn(), getSshConnectionStoreMock: vi.fn(), - registerRemoteWorkspaceNotificationHandlerMock: vi.fn(() => vi.fn()) + registerRemoteWorkspaceNotificationHandlerMock: vi.fn(() => vi.fn()), + resolveWorktreeExecutionHostCalls: { count: 0 } })) +// Counts ownership resolutions without changing any of them. +vi.mock('../../shared/worktree-execution-host-resolution', async (importOriginal) => { + const actual = (await importOriginal()) as typeof WorktreeExecutionHostResolution + return { + ...actual, + resolveWorktreeExecutionHost: ( + ...args: Parameters + ) => { + resolveWorktreeExecutionHostCalls.count += 1 + return actual.resolveWorktreeExecutionHost(...args) + } + } +}) + vi.mock('electron', () => ({ ipcMain: { handle: vi.fn(), @@ -154,9 +171,10 @@ describe('remoteWorkspace:setForConnectedTargets', () => { const getRepoMock = vi.fn() const getWorkspaceSessionMock = vi.fn() // Ownership resolution reads the catalog, not one id-keyed row, so the fake has to project one. + const getReposMock = vi.fn(() => [getRepoMock('repo-target-1')].filter(Boolean)) const store = { getRepo: getRepoMock, - getRepos: () => [getRepoMock('repo-target-1')].filter(Boolean), + getRepos: getReposMock, getWorkspaceSession: getWorkspaceSessionMock } as unknown as Store @@ -176,6 +194,7 @@ describe('remoteWorkspace:setForConnectedTargets', () => { getTarget: (targetId: string) => targets.find((target) => target.id === targetId) }) getRepoMock.mockReset() + getReposMock.mockClear() getWorkspaceSessionMock.mockReset() getWorkspaceSessionMock.mockReturnValue(baseSession) getRepoMock.mockImplementation((repoId: string) => @@ -251,6 +270,79 @@ describe('remoteWorkspace:setForConnectedTargets', () => { return observed as RemoteWorkspaceObservedSnapshot } + it('reads the repo catalog once per publish, not once per worktree', async () => { + // `store.getRepos()` re-hydrates every repo row. The export asks "is this worktree mine?" once + // per worktree, so reading the catalog inside that callback multiplied hydration by the + // worktree count — 413 on the session that surfaced this. + const worktrees = Object.fromEntries( + Array.from({ length: 12 }, (_, index) => [`repo-target-1::/remote/repo-${index}`, []]) + ) + getWorkspaceSessionMock.mockReturnValue({ + ...baseSession, + tabsByWorktree: worktrees + } as WorkspaceSessionState) + const observed = await observeTarget('target-1') + getReposMock.mockClear() + + await callSetForConnectedTargets({ + hydratedTargetIds: ['target-1'], + expectedRevisionsByTargetId: { 'target-1': observed.revision }, + expectedHostObservationTokensByTargetId: { + 'target-1': observed.hostObservationToken + } + }) + + expect(getReposMock).toHaveBeenCalledTimes(1) + }) + + it('resolves each worktree ownership once for the whole publish, not once per target', async () => { + // Ownership is a function of the repo catalog alone; only the final `=== targetId` differs, so + // exporting to N targets used to repeat the identical resolution N times per worktree key. + const worktrees = Object.fromEntries( + Array.from({ length: 6 }, (_, index) => [`repo-target-1::/remote/repo-${index}`, []]) + ) + getWorkspaceSessionMock.mockReturnValue({ + ...baseSession, + tabsByWorktree: worktrees + } as WorkspaceSessionState) + const observed = await Promise.all(targets.map((target) => observeTarget(target.id))) + getReposMock.mockClear() + resolveWorktreeExecutionHostCalls.count = 0 + + await callSetForConnectedTargets({ + hydratedTargetIds: targets.map((target) => target.id), + expectedRevisionsByTargetId: Object.fromEntries( + targets.map((target, index) => [target.id, observed[index].revision]) + ), + expectedHostObservationTokensByTargetId: Object.fromEntries( + targets.map((target, index) => [target.id, observed[index].hostObservationToken]) + ) + }) + + expect(getReposMock).toHaveBeenCalledTimes(1) + // 6 worktree keys resolved once each, regardless of how many targets are published to. + expect(resolveWorktreeExecutionHostCalls.count).toBe(6) + }) + + it('skips the session and repo-catalog reads when no hydrated target is connected', async () => { + // A hydrated but disconnected target leaves nothing to project onto, so hoisting the catalog + // read must not make the idle path pay for a full repo hydration it never used before. + getActiveMultiplexerMock.mockReturnValue(undefined) + getReposMock.mockClear() + getWorkspaceSessionMock.mockClear() + + await expect( + callSetForConnectedTargets({ + hydratedTargetIds: ['target-1'], + expectedRevisionsByTargetId: { 'target-1': 7 }, + expectedHostObservationTokensByTargetId: { 'target-1': 'token' } + }) + ).resolves.toEqual([]) + + expect(getReposMock).not.toHaveBeenCalled() + expect(getWorkspaceSessionMock).not.toHaveBeenCalled() + }) + it('does not write without an explicit non-empty hydrated target set', async () => { await expect(callSetForConnectedTargets({ session: baseSession })).resolves.toEqual([]) await expect( diff --git a/src/main/ipc/remote-workspace.ts b/src/main/ipc/remote-workspace.ts index f9479f15ce9..935fd1c9f72 100644 --- a/src/main/ipc/remote-workspace.ts +++ b/src/main/ipc/remote-workspace.ts @@ -1,5 +1,6 @@ import { ipcMain, type BrowserWindow } from 'electron' import type { Store } from '../persistence' +import type { Repo } from '../../shared/repo-types' import { getActiveMultiplexer, getSshConnectionStore } from './ssh' import { exportRemoteWorkspaceSession } from '../../shared/remote-workspace-session-projection' import { @@ -107,7 +108,7 @@ function getExpectedHostObservationTokens( } function targetForWorktree( - store: Store, + repoLookup: ReturnType>, worktreeId: string, executionHostId?: string ): string | null { @@ -115,21 +116,49 @@ function targetForWorktree( // `getRepo(id)?.connectionId`, which is host-blind — the same repo id can name rows on several // hosts, so a session could be published to a machine that never owned the worktree (#11163). // Unresolvable ownership exports to nobody rather than guessing. - const resolution = resolveWorktreeExecutionHost( - createRepoRowExecutionHostLookup(store.getRepos()), - { repoId: getRepoIdFromWorktreeId(worktreeId), hostId: executionHostId ?? null } - ) + const resolution = resolveWorktreeExecutionHost(repoLookup, { + repoId: getRepoIdFromWorktreeId(worktreeId), + hostId: executionHostId ?? null + }) return resolution.kind === 'resolved' ? resolution.connectionId : null } +/** + * Resolve each worktree's owning connection at most once for a whole publish. + * + * Why this is shared and not per target: `targetForWorktree` computes a connection id from the + * repo catalog alone — only the final `=== targetId` differs — so exporting to N targets used to + * repeat the identical resolution N times over every worktree key. `store.getRepos()` also + * re-hydrates every repo row on each call, and the projection asks this question once per key of + * `tabsByWorktree`, `activeTabIdByWorktree`, `lastVisitedAtByWorktreeId` and + * `defaultTerminalTabsAppliedByWorktreeId`. + */ +function createWorktreeTargetResolver( + repoLookup: ReturnType> +): (worktreeId: string, executionHostId?: string) => string | null { + const resolved = new Map() + return (worktreeId, executionHostId) => { + // Host id participates in resolution, so it has to participate in the key. NUL cannot appear + // in either id, so it is a collision-free separator. + const key = `${worktreeId}\u0000${executionHostId ?? ''}` + const cached = resolved.get(key) + if (cached !== undefined) { + return cached + } + const connectionId = targetForWorktree(repoLookup, worktreeId, executionHostId) + resolved.set(key, connectionId) + return connectionId + } +} + function exportSessionForTarget( - store: Store, + resolveWorktreeTarget: (worktreeId: string, executionHostId?: string) => string | null, targetId: string, session: WorkspaceSessionState ): RemoteWorkspaceSession { return exportRemoteWorkspaceSession(session, { isTargetWorktree: (worktreeId, executionHostId) => - targetForWorktree(store, worktreeId, executionHostId) === targetId + resolveWorktreeTarget(worktreeId, executionHostId) === targetId }) } @@ -245,12 +274,21 @@ export function registerRemoteWorkspaceHandlers( (target) => hydratedTargetIds.has(target.id) && getActiveMultiplexer(target.id) ) ?? [] + if (targets.length === 0) { + // Nothing to project onto, so skip the session and repo-catalog reads entirely. + return [] + } + const workspaceSession = args.session ?? store.getWorkspaceSession() + // One repo read, and ownership resolutions shared across targets: neither depends on the target. + const resolveWorktreeTarget = createWorktreeTargetResolver( + createRepoRowExecutionHostLookup(store.getRepos()) + ) const results = await Promise.all( targets.map(async (target) => { // Why: each target has its own revision stream. Keep same-target // writes queued, but do not let one slow relay block others. - const session = exportSessionForTarget(store, target.id, workspaceSession) + const session = exportSessionForTarget(resolveWorktreeTarget, target.id, workspaceSession) const result = await queueRemoteWorkspacePatch(target.id, async () => { const current = getCachedRemoteWorkspaceSnapshot(target.id) ?? (await getRemoteSnapshot(target)) diff --git a/src/main/ipc/worktree-base-directory-marker-poller.ts b/src/main/ipc/worktree-base-directory-marker-poller.ts index dba1c4df03c..5f0acee4ee0 100644 --- a/src/main/ipc/worktree-base-directory-marker-poller.ts +++ b/src/main/ipc/worktree-base-directory-marker-poller.ts @@ -28,7 +28,7 @@ const PENDING_MARKER_MAX_TICKS = 300 // Why: matches the git-common poller's fan-out bound (#17828) — bounded // concurrency turns hundreds of serial round trips into a handful of batches // without dumping every candidate onto libuv's 4-thread pool at once. -const MARKER_PROBE_CONCURRENCY = 8 +export const MARKER_PROBE_CONCURRENCY = 8 function statSignature(s: { mtimeMs: number; ctimeMs: number; ino: number }): string { return `${s.mtimeMs}:${s.ctimeMs}:${s.ino}` @@ -186,7 +186,7 @@ export async function startBasePoller( } const checkPendingMarkers = async (): Promise => { - const events: WorktreeBasePollEvent[] = [] + const dueDirs: string[] = [] for (const [dir, firstSeenTick] of markerProbeStartedAt) { if (firstSeenTick === null) { continue @@ -195,13 +195,19 @@ export async function startBasePoller( markerProbeStartedAt.set(dir, null) continue } + dueDirs.push(dir) + } + const events: WorktreeBasePollEvent[] = [] + // Same bound as the full scan's fan-out: serial probes cost D x latency per tick, + // which a WSL- or network-backed base directory pays for up to `pendingMarkerMaxTicks`. + await forEachWithConcurrency(dueDirs, MARKER_PROBE_CONCURRENCY, async (dir) => { options.onPendingMarkerProbe?.(join(dir, '.git')) if (await hasGitMarker(dir)) { markerProbeStartedAt.delete(dir) snapshot.markers.set(dir, true) events.push({ type: 'create', path: join(dir, '.git') }) } - } + }) if (!disposed && events.length > 0) { onEvents(events) } diff --git a/src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts b/src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts index 023a8390ccd..508a7bf8c41 100644 --- a/src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts +++ b/src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts @@ -3,6 +3,7 @@ import { mkdir, mkdtemp, realpath, rm, writeFile } from 'node:fs/promises' import type * as NodeFsPromises from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' +import { MARKER_PROBE_CONCURRENCY } from './worktree-base-directory-marker-poller' import { startWorktreeBaseDirectoryPoller } from './worktree-base-directory-poller' import type { WorktreeBaseRepoWatchConfig, @@ -12,7 +13,11 @@ import type { // Why: the backstop full scan stats a `.git` marker per candidate dir; an // unbounded fan-out at hundreds of worktrees would queue thousands of `stat` // calls on libuv's 4-thread pool (#17828). -const { concurrency } = vi.hoisted(() => ({ concurrency: { current: 0, peak: 0 } })) +const { concurrency, markerStatGate } = vi.hoisted(() => ({ + concurrency: { current: 0, peak: 0 }, + // Parks `.git` stats so a batch's launched-at-once width is observable without wall clocks. + markerStatGate: { hold: false, parked: [] as (() => void)[] } +})) vi.mock('node:fs/promises', async (importOriginal) => { const actual = await importOriginal() @@ -22,6 +27,9 @@ vi.mock('node:fs/promises', async (importOriginal) => { concurrency.current += 1 concurrency.peak = Math.max(concurrency.peak, concurrency.current) try { + if (markerStatGate.hold && String(args[0]).endsWith('.git')) { + await new Promise((resolve) => markerStatGate.parked.push(resolve)) + } return await actual.stat(...args) } finally { concurrency.current -= 1 @@ -50,12 +58,27 @@ describe('worktree base directory poller marker fan-out (#17828)', () => { beforeEach(() => { concurrency.current = 0 concurrency.peak = 0 + markerStatGate.hold = false + markerStatGate.parked.length = 0 }) afterEach(async () => { + markerStatGate.hold = false + for (const resume of markerStatGate.parked.splice(0)) { + resume() + } await Promise.all(cleanups.splice(0).map((cleanup) => cleanup())) }) + async function waitUntil(predicate: () => boolean): Promise { + for (let attempt = 0; attempt < 2_000 && !predicate(); attempt++) { + await new Promise((resolve) => setTimeout(resolve, 5)) + } + if (!predicate()) { + throw new Error('timed out waiting for the poller') + } + } + it('bounds concurrent `.git`-marker stats regardless of candidate count', async () => { const root = await realpath(await mkdtemp(join(tmpdir(), 'orca-base-poller-fanout-'))) cleanups.push(() => rm(root, { recursive: true, force: true })) @@ -81,4 +104,47 @@ describe('worktree base directory poller marker fan-out (#17828)', () => { expect(concurrency.peak).toBeGreaterThan(1) expect(concurrency.peak).toBeLessThan(20) }) + + it('probes pending `.git` markers in bounded batches instead of one at a time', async () => { + const root = await realpath(await mkdtemp(join(tmpdir(), 'orca-base-poller-pending-'))) + cleanups.push(() => rm(root, { recursive: true, force: true })) + const pendingCount = MARKER_PROBE_CONCURRENCY * 4 + for (let i = 0; i < pendingCount; i++) { + // No `.git`: every dir stays a pending-marker candidate for the whole test. + await mkdir(join(root, `pending-${i}`)) + } + + const probed: string[] = [] + let parkFirstBatch = true + const target = makeTarget(root) + const poller = await startWorktreeBaseDirectoryPoller( + target, + () => target.repos, + () => {}, + { + pollIntervalMs: 1, + onPendingMarkerProbe: (path) => { + probed.push(path) + // Park from the first probe onward, so the count below is the batch width. + markerStatGate.hold = parkFirstBatch + } + } + ) + cleanups.push(() => poller.unsubscribe()) + + await waitUntil(() => markerStatGate.parked.length > 0) + + // Serial probing parks after one; the batch launches exactly the bound at once. + expect(probed.length).toBe(MARKER_PROBE_CONCURRENCY) + + parkFirstBatch = false + markerStatGate.hold = false + for (const resume of markerStatGate.parked.splice(0)) { + resume() + } + await waitUntil(() => probed.length >= pendingCount) + + // The first tick still probes every due dir exactly once. + expect(new Set(probed.slice(0, pendingCount)).size).toBe(pendingCount) + }) }) diff --git a/src/main/ipc/worktree-logic.ts b/src/main/ipc/worktree-logic.ts index 17ad0c49e46..7a8fe175c89 100644 --- a/src/main/ipc/worktree-logic.ts +++ b/src/main/ipc/worktree-logic.ts @@ -4,9 +4,15 @@ import type { Repo } from '../../shared/repo-types' import { isWindowsAbsolutePathLike, resolveRuntimePath } from '../../shared/cross-platform-path' import { isWslUncPath, resolveWslRepoWorktreeBasePath } from '../../shared/wsl-paths' import { splitWorktreeId } from '../../shared/worktree/id' -import { replaceKnownEmojiWithShortcodes } from '../../shared/emoji-shortcode-catalog' +import { + replaceKnownEmojiWithShortcodes, + setEmojiShortcodeDatasetLoader +} from '../../shared/emoji-shortcode-catalog' +import { requireEmojiShortcodeDataset } from './deferred-emoji-shortcode-dataset' import { getWslHome, getWslHomeAsync, parseWslPath } from '../wsl' +setEmojiShortcodeDatasetLoader(requireEmojiShortcodeDataset) + type WorktreePathSettings = Pick & { /** Distro to mirror the workspace root into when the repo itself sits on a * Windows drive but this project's git runs in WSL. Omitted = today's diff --git a/src/main/ipc/worktrees/listing/detected-provider-listing-meta-index.test.ts b/src/main/ipc/worktrees/listing/detected-provider-listing-meta-index.test.ts new file mode 100644 index 00000000000..4bf55e97492 --- /dev/null +++ b/src/main/ipc/worktrees/listing/detected-provider-listing-meta-index.test.ts @@ -0,0 +1,131 @@ +/** + * The SSH worktree-meta index is only ever read via `metaIndex.get(repo.id)` on the disconnected + * fallbacks, so a connected listing must not pay `parseWorktreeId` over the whole host snapshot. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { Store } from '../../../persistence/loading-store/store' +import type * as SshWorktreeFallbackModule from './ssh-worktree-fallback' + +const { getSshGitProviderMock, indexBuildSpy } = vi.hoisted(() => ({ + getSshGitProviderMock: vi.fn(), + indexBuildSpy: vi.fn() +})) + +vi.mock('../../../providers/ssh-git-dispatch', () => ({ + getSshGitProvider: getSshGitProviderMock, + requireSshGitProvider: getSshGitProviderMock, + getSshGitProviderGeneration: () => 1 +})) + +vi.mock('./ssh-worktree-fallback', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + // Both builders are counted: the point is that NO index is built on the connected path. + createSshWorktreeMetaIndex: (...args: Parameters) => { + indexBuildSpy('all-hosts', ...args) + return actual.createSshWorktreeMetaIndex(...args) + }, + createSshWorktreeMetaIndexForRepo: ( + ...args: Parameters + ) => { + indexBuildSpy('repo-scoped', ...args) + return actual.createSshWorktreeMetaIndexForRepo(...args) + } + } +}) + +const { listDetectedWorktreesForCapturedRepo } = await import('./detected-provider-listing') + +const repo = { + id: 'repo-1', + path: '/home/user/repo', + displayName: 'repo', + connectionId: 'conn-1' +} as Repo + +const worktreeId = `${repo.id}::/home/user/feature` + +function createStore(): Store { + const rows: Record = { + [worktreeId]: { instanceId: 'instance-1' }, + // Other repos' rows share the host snapshot; only this repo's bucket is ever read back. + 'repo-2::/home/user/other': { instanceId: 'instance-2' } + } + return { + getRepos: () => [repo], + getRepo: () => repo, + getSettings: () => ({}), + getProjectHostSetups: () => [], + getAllWorktreeLineage: () => ({}), + getAllWorktreeMeta: () => rows, + getWorktreeMeta: (id: string) => rows[id], + setWorktreeMeta: vi.fn() + } as unknown as Store +} + +describe('SSH worktree meta index construction', () => { + beforeEach(() => { + indexBuildSpy.mockClear() + getSshGitProviderMock.mockReset() + }) + + it('does not build the index when the provider answers', async () => { + const provider = { + listWorktrees: vi.fn().mockResolvedValue([ + { path: repo.path, head: 'a', branch: 'main', isBare: false, isMainWorktree: true }, + { + path: '/home/user/feature', + head: 'b', + branch: 'feature', + isBare: false, + isMainWorktree: false + } + ]) + } + + const result = await listDetectedWorktreesForCapturedRepo( + createStore(), + repo, + () => true, + provider as never + ) + + expect(result).toMatchObject({ authoritative: true, source: 'git' }) + expect(indexBuildSpy).not.toHaveBeenCalled() + }) + + it('builds the index once when no provider is available', async () => { + const result = await listDetectedWorktreesForCapturedRepo( + createStore(), + repo, + () => true, + undefined + ) + + expect(result).toMatchObject({ authoritative: false, source: 'metadata-fallback' }) + expect(indexBuildSpy).toHaveBeenCalledTimes(1) + expect(indexBuildSpy).toHaveBeenCalledWith('all-hosts', expect.anything()) + expect( + (result as { worktrees: { id: string }[] }).worktrees.map((worktree) => worktree.id) + ).toEqual([worktreeId]) + }) + + it('builds the index once when the provider listing fails', async () => { + const provider = { listWorktrees: vi.fn().mockRejectedValue(new Error('relay down')) } + + const result = await listDetectedWorktreesForCapturedRepo( + createStore(), + repo, + () => true, + provider as never + ) + + expect(result).toMatchObject({ authoritative: false, source: 'metadata-fallback' }) + expect(indexBuildSpy).toHaveBeenCalledTimes(1) + expect( + (result as { worktrees: { id: string }[] }).worktrees.map((worktree) => worktree.id) + ).toEqual([worktreeId]) + }) +}) diff --git a/src/main/ipc/worktrees/listing/detected-provider-listing.ts b/src/main/ipc/worktrees/listing/detected-provider-listing.ts index 51a0618ba66..388c825530f 100644 --- a/src/main/ipc/worktrees/listing/detected-provider-listing.ts +++ b/src/main/ipc/worktrees/listing/detected-provider-listing.ts @@ -10,7 +10,8 @@ import type { ListDesktopLineageForHostArgs } from '../../../../shared/host-line import { buildDetectedGitWorktrees, createSshWorktreeMetaIndex, - listDisconnectedSshWorktrees + listDisconnectedSshWorktrees, + type SshWorktreeMetaIndex } from './ssh-worktree-fallback' import { buildDisconnectedDetectedWorktrees, @@ -25,8 +26,7 @@ import { type DetectedWorktreeSideEffectToken } from './detected-worktree-scan-cache' import { loggedWorktreeListFailures, warnOnce } from './worktree-listing-diagnostics' -import { readAllWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' -import { getRepoExecutionHostId } from '../../../../shared/execution-host' +import { readAllWorktreeMetaForRepo } from '../../../persistence/host-qualified-worktree-meta' export async function listDetectedWorktreesForCapturedRepo( store: Store, @@ -39,12 +39,12 @@ export async function listDetectedWorktreesForCapturedRepo( providerAbort?.signal.aborted ? ({ providerAbortStatus: providerAbort.status() } as const) : undefined - const allMeta = isFolderRepo(repo) - ? undefined - : readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) - const sshWorktreeMetaIndex = repo.connectionId - ? createSshWorktreeMetaIndex(Object.entries(allMeta ?? {})) - : new Map() + const allMeta = isFolderRepo(repo) ? undefined : readAllWorktreeMetaForRepo(store, repo) + // Why: only the disconnected fallbacks read this, so keep parseWorktreeId over the whole host snapshot + // off the connected path entirely. + let cachedSshWorktreeMetaIndex: SshWorktreeMetaIndex | undefined + const sshWorktreeMetaIndex = (): SshWorktreeMetaIndex => + (cachedSshWorktreeMetaIndex ??= createSshWorktreeMetaIndex(Object.entries(allMeta ?? {}))) try { let gitWorktrees: GitWorktreeInfo[] @@ -86,7 +86,7 @@ export async function listDetectedWorktreesForCapturedRepo( if (!isCurrent()) { return null } - const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex) + const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex()) return { repoId: repo.id, authoritative: false, @@ -158,7 +158,7 @@ export async function listDetectedWorktreesForCapturedRepo( err ) if (repo.connectionId) { - const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex) + const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex()) return { repoId: repo.id, authoritative: false, diff --git a/src/main/ipc/worktrees/listing/detected-worktree-classification.test.ts b/src/main/ipc/worktrees/listing/detected-worktree-classification.test.ts new file mode 100644 index 00000000000..6a25d696c9b --- /dev/null +++ b/src/main/ipc/worktrees/listing/detected-worktree-classification.test.ts @@ -0,0 +1,234 @@ +/** + * Guards the single-classification contract of `buildDetectedGitWorktrees`: every visible worktree + * used to be run through `mergeWorktree` + `toDetectedWorktree` twice per catalog pass. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { Store } from '../../../persistence/loading-store/store' +import type { WorktreeMeta } from '../../../../shared/worktree/meta-types' +import type { GitWorktreeInfo } from '../../../../shared/worktree/types' +import type * as NodeCryptoModule from 'node:crypto' +import type * as OwnershipModule from '../../../../shared/worktree/ownership' + +const { toDetectedWorktreeSpy } = vi.hoisted(() => ({ toDetectedWorktreeSpy: vi.fn() })) + +vi.mock('../../../../shared/worktree/ownership', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + toDetectedWorktree: (args: Parameters[0]) => { + toDetectedWorktreeSpy(args) + return actual.toDetectedWorktree(args) + } + } +}) + +vi.mock('node:crypto', async (importOriginal) => ({ + ...(await importOriginal()), + randomUUID: () => 'fixed-instance-id' +})) + +const { buildDetectedGitWorktrees } = await import('./ssh-worktree-fallback') +const { getProjectHostSetupWorktreeMeta } = + await import('../../../../shared/project-host-setup-lookup') +const { mergeWorktree } = await import('../../worktree-logic') +const { resolveWorktreeMetaWithDiscoveryBackfill } = await import('./worktree-discovery-metadata') +const ownership = await import('../../../../shared/worktree/ownership') +const { projectResolvedWorktreeLineage } = + await import('../../../../shared/resolved-worktree-lineage') +const { createWorktreeVisibilitySourceMatcher, resolveCustomWorktreeVisibilitySources } = + await import('../../../../shared/worktree/visibility-sources') +const { resolveConfiguredWorktreeBasePaths } = + await import('../../../../shared/worktree/configured-worktree-base-path') +const { dedupeWorktreesByPath } = await import('../../worktree-path-comparison') +const { readWorktreeMetaForHost } = + await import('../../../persistence/host-qualified-worktree-meta') +const { getRepoOwnedWorktreeMeta } = await import('../../../worktree-metadata-ownership') +const { getRepoExecutionHostId } = await import('../../../../shared/execution-host') + +const repo: Repo = { + id: 'repo-1', + path: '/workspace/repo', + displayName: 'repo', + badgeColor: '#000', + addedAt: 0 +} as Repo + +const ownershipMeta = getProjectHostSetupWorktreeMeta([], repo) + +function gitWorktree(path: string): GitWorktreeInfo { + return { + path, + head: 'abc123', + branch: 'refs/heads/feature', + isBare: false, + isMainWorktree: false + } +} + +/** Fully settled metadata: discovery backfill has nothing to write, so it hands the same object back. */ +function settledMeta(overrides: Partial = {}): WorktreeMeta { + return { + ...ownershipMeta, + instanceId: 'instance-settled', + orcaCreatedAt: 1, + lastActivityAt: 5, + ...overrides + } as WorktreeMeta +} + +function createStore(meta: Record, repos: Repo[] = [repo]) { + const rows = { ...meta } + return { + getRepos: () => repos, + getSettings: () => ({ workspaceDir: '/workspace', nestWorkspaces: true }), + getProjectHostSetups: () => [], + getAllWorktreeLineage: () => ({}), + getAllWorktreeMeta: () => rows, + getWorktreeMeta: (id: string) => rows[id], + getWorktreeMetaForHost: (id: string, hostId: string) => + rows[id]?.hostId === hostId ? rows[id] : undefined, + getAllWorktreeMetaForHost: () => rows, + setWorktreeMeta: (id: string, patch: Partial) => { + rows[id] = { ...rows[id], ...patch } as WorktreeMeta + return rows[id] + }, + setWorktreeMetaForHost: (id: string, hostId: string, patch: Partial) => { + rows[id] = { ...rows[id], ...patch, hostId } as WorktreeMeta + return rows[id] + } + } as unknown as Store +} + +/** The pre-change implementation, verbatim, as the equivalence oracle. */ +function buildDetectedGitWorktreesTwoPass( + store: Store, + target: Repo, + gitWorktrees: GitWorktreeInfo[], + allMetaOverride?: Record +) { + const settings = store.getSettings() + const knownOrcaLayouts = ownership.buildKnownOrcaWorkspaceLayouts(settings, target) + const isLegacyRepoForVisibility = ownership.isLegacyRepoForExternalWorktreeVisibility(target) + const liveWorktrees = dedupeWorktreesByPath(gitWorktrees.filter((info) => !info.prunable)) + const worktreeVisibilitySourceMatcher = createWorktreeVisibilitySourceMatcher( + [target.path, ...liveWorktrees.map((worktree) => worktree.path)], + resolveCustomWorktreeVisibilitySources(target, settings.worktreeVisibilityDefaults), + resolveConfiguredWorktreeBasePaths(target) + ) + const allMeta = allMetaOverride ?? store.getAllWorktreeMeta?.() + const repoOwnerCount = store.getRepos().filter((candidate) => candidate.id === target.id).length + const detectedRows = liveWorktrees.map((info) => { + const worktreeId = `${target.id}::${info.path}` + const legacyMeta = store.getWorktreeMeta?.(worktreeId) + const metaById = allMeta ?? (legacyMeta ? { [worktreeId]: legacyMeta } : {}) + let meta = + readWorktreeMetaForHost(store, worktreeId, getRepoExecutionHostId(target)) ?? + getRepoOwnedWorktreeMeta(target, worktreeId, metaById, repoOwnerCount) + const worktree = mergeWorktree(target.id, info, meta, target.displayName) + const detected = ownership.toDetectedWorktree({ + repo: target, + worktree, + meta, + settings, + knownOrcaLayouts, + isLegacyRepoForVisibility, + worktreeVisibilitySourceMatcher + }) + if (!detected.visible) { + return detected + } + meta = resolveWorktreeMetaWithDiscoveryBackfill( + store, + target, + worktreeId, + allMeta, + repoOwnerCount + ) + return ownership.toDetectedWorktree({ + repo: target, + worktree: mergeWorktree(target.id, info, meta, target.displayName), + meta, + settings, + knownOrcaLayouts, + isLegacyRepoForVisibility, + worktreeVisibilitySourceMatcher + }) + }) + return projectResolvedWorktreeLineage(detectedRows, store.getAllWorktreeLineage?.() ?? {}) +} + +describe('buildDetectedGitWorktrees classification passes', () => { + beforeEach(() => { + toDetectedWorktreeSpy.mockClear() + // Discovery backfill stamps lastActivityAt from the clock; freeze it so equivalence is deterministic. + vi.spyOn(Date, 'now').mockReturnValue(1_700_000_000_000) + }) + + it('classifies each visible worktree once per catalog pass, not twice', () => { + const paths = ['/workspace/one', '/workspace/two', '/workspace/three'] + const meta = Object.fromEntries( + paths.map((path) => [`${repo.id}::${path}`, settledMeta({ displayName: path })]) + ) + const store = createStore(meta) + + const detected = buildDetectedGitWorktrees(store, repo, paths.map(gitWorktree), meta) + + expect(detected).toHaveLength(3) + expect(detected.every((row) => row.visible)).toBe(true) + expect(toDetectedWorktreeSpy).toHaveBeenCalledTimes(paths.length) + }) + + it('reads the locator-keyed metadata row only when no host snapshot is available', () => { + const worktreeId = `${repo.id}::/workspace/one` + const meta = { [worktreeId]: settledMeta() } + const store = createStore(meta) + const legacyReads = vi.spyOn(store, 'getWorktreeMeta') + + buildDetectedGitWorktrees(store, repo, [gitWorktree('/workspace/one')], meta) + expect(legacyReads).not.toHaveBeenCalled() + + // Partial stores (compatibility shapes) expose no snapshot, so the locator-keyed lookup must still run. + const partialStore = createStore(meta) as Partial + delete partialStore.getAllWorktreeMeta + delete partialStore.getAllWorktreeMetaForHost + delete partialStore.getWorktreeMetaForHost + const partialLegacyReads = vi.spyOn(partialStore as Store, 'getWorktreeMeta') + + const rows = buildDetectedGitWorktrees( + partialStore as Store, + repo, + [gitWorktree('/workspace/one')], + undefined + ) + expect(partialLegacyReads).toHaveBeenCalledWith(worktreeId) + expect(rows[0]).toMatchObject({ id: worktreeId, lastActivityAt: 5 }) + }) + + it.each([ + ['settled metadata', () => settledMeta()], + ['metadata needing discovery backfill', () => ({ orcaCreatedAt: 1 }) as WorktreeMeta], + ['no metadata at all', () => undefined] + ])('emits a catalog deep-equal to the two-pass build for %s', (_label, makeMeta) => { + const worktreeId = `${repo.id}::/workspace/one` + const seed = makeMeta() + const build = (fn: typeof buildDetectedGitWorktrees) => + fn( + createStore(seed ? { [worktreeId]: seed } : {}), + repo, + [gitWorktree('/workspace/one'), gitWorktree('/workspace/hidden-external')], + seed ? { [worktreeId]: seed } : {} + ) + + expect(build(buildDetectedGitWorktrees)).toEqual(build(buildDetectedGitWorktreesTwoPass)) + }) + + it('emits a catalog deep-equal to the two-pass build for a folder-style listing with no host snapshot', () => { + const worktreeId = `${repo.id}::/workspace/one` + const seed = settledMeta() + const build = (fn: typeof buildDetectedGitWorktrees) => + fn(createStore({ [worktreeId]: seed }), repo, [gitWorktree('/workspace/one')], undefined) + + expect(build(buildDetectedGitWorktrees)).toEqual(build(buildDetectedGitWorktreesTwoPass)) + }) +}) diff --git a/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts b/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts index d684461a381..4f3055c63a2 100644 --- a/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts +++ b/src/main/ipc/worktrees/listing/register-worktree-catalog-handlers.ts @@ -23,7 +23,10 @@ import { warnOnce } from './worktree-listing-diagnostics' import type { WorktreeIpcContext } from '../worktree-ipc-context' -import { readAllWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' +import { + readAllWorktreeMetaForHost, + readAllWorktreeMetaForRepo +} from '../../../persistence/host-qualified-worktree-meta' import type { WorktreeMeta } from '../../../../shared/worktree/meta-types' const WORKTREE_LIST_ALL_CONCURRENCY = 8 @@ -174,9 +177,7 @@ export function registerWorktreeCatalogHandlers(context: WorktreeIpcContext): vo if (!repo) { return [] } - const allMeta = repo.connectionId - ? readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) - : undefined + const allMeta = repo.connectionId ? readAllWorktreeMetaForRepo(store, repo) : undefined const sshWorktreeMetaIndex = repo.connectionId ? createSshWorktreeMetaIndex(Object.entries(allMeta ?? {})) : new Map() @@ -226,7 +227,7 @@ export function registerWorktreeCatalogHandlers(context: WorktreeIpcContext): vo }) } loggedWorktreeListFailures.delete(`${repo.id}:${repo.path}`) - const metadata = allMeta ?? readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) + const metadata = allMeta ?? readAllWorktreeMetaForRepo(store, repo) return buildDetectedGitWorktrees(store, repo, gitWorktrees, metadata) .filter((worktree) => worktree.visible) .map((worktree) => stampAndMergeVisibleDetectedWorktree(store, repo, worktree, metadata)) diff --git a/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts b/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts index 7ec2cac1fc9..ecf8ab3abc1 100644 --- a/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts +++ b/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts @@ -9,7 +9,7 @@ import type { GitWorktreeInfo, DetectedWorktree, Worktree } from '../../../../sh import type { Store } from '../../../persistence/loading-store/store' import { getRepoExecutionHostId } from '../../../../shared/execution-host' import { - readWorktreeMetaForHost, + readWorktreeMetaForRepo, writeWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from '../../../worktree-metadata-ownership' @@ -155,10 +155,11 @@ export function buildDetectedGitWorktrees( const repoOwnerCount = store.getRepos().filter((candidate) => candidate.id === repo.id).length const detected = liveWorktrees.map((gitWorktree) => { const worktreeId = `${repo.id}::${gitWorktree.path}` - const legacyMeta = store.getWorktreeMeta?.(worktreeId) + // Why: the locator-keyed row is only a stand-in for a missing host snapshot, so don't read it when we have one. + const legacyMeta = allMeta === undefined ? store.getWorktreeMeta?.(worktreeId) : undefined const metaById = allMeta ?? (legacyMeta ? { [worktreeId]: legacyMeta } : {}) - let meta = - readWorktreeMetaForHost(store, worktreeId, getRepoExecutionHostId(repo)) ?? + const meta = + readWorktreeMetaForRepo(store, worktreeId, repo) ?? getRepoOwnedWorktreeMeta(repo, worktreeId, metaById, repoOwnerCount) const worktree = mergeWorktree(repo.id, gitWorktree, meta, repo.displayName) const detected = toDetectedWorktree({ @@ -174,17 +175,21 @@ export function buildDetectedGitWorktrees( return detected } - meta = resolveWorktreeMetaWithDiscoveryBackfill( + const backfilledMeta = resolveWorktreeMetaWithDiscoveryBackfill( store, repo, worktreeId, allMeta, repoOwnerCount ) + // Why: backfill hands back the same object when it wrote nothing, and both builders are pure over it. + if (backfilledMeta === meta) { + return detected + } return toDetectedWorktree({ repo, - worktree: mergeWorktree(repo.id, gitWorktree, meta, repo.displayName), - meta, + worktree: mergeWorktree(repo.id, gitWorktree, backfilledMeta, repo.displayName), + meta: backfilledMeta, settings, knownOrcaLayouts, isLegacyRepoForVisibility, diff --git a/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts b/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts index 6af4d45c3cd..3edb8efc1d7 100644 --- a/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts +++ b/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts @@ -4,7 +4,7 @@ import type { WorktreeMeta } from '../../../../shared/worktree/meta-types' import { getProjectHostSetupWorktreeMeta } from '../../../../shared/project-host-setup-lookup' import { getRepoExecutionHostId } from '../../../../shared/execution-host' import { - readWorktreeMetaForHost, + readWorktreeMetaForRepo, writeWorktreeMetaForHost } from '../../../persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from '../../../worktree-metadata-ownership' @@ -40,10 +40,11 @@ export function resolveWorktreeMetaWithDiscoveryBackfill( repoOwnerCount = store.getRepos().filter((candidate) => candidate.id === repo.id).length ): WorktreeMeta { const executionHostId = getRepoExecutionHostId(repo) - const legacyMeta = store.getWorktreeMeta?.(worktreeId) const allMeta = allMetaOverride ?? store.getAllWorktreeMeta?.() + // Why: the locator-keyed row is only a stand-in for a missing snapshot, so don't read it when we have one. + const legacyMeta = allMeta === undefined ? store.getWorktreeMeta?.(worktreeId) : undefined const existing = - readWorktreeMetaForHost(store, worktreeId, executionHostId) ?? + readWorktreeMetaForRepo(store, worktreeId, repo) ?? getRepoOwnedWorktreeMeta( repo, worktreeId, diff --git a/src/main/orca-chromium-process-pids.ts b/src/main/orca-chromium-process-pids.ts index f22babc6921..b2613e42b79 100644 --- a/src/main/orca-chromium-process-pids.ts +++ b/src/main/orca-chromium-process-pids.ts @@ -1,4 +1,5 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' +import { recordCoalescedDurableCrashBreadcrumb } from './crash-reporting/durable-crash-breadcrumb' /** * PIDs of Orca's own Chromium processes — browser, renderers, GPU, utilities. @@ -11,6 +12,14 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' * Empty on a Node host and empty on failure: that is "no refusal proven", never * "safe to kill" — callers must keep every other guard they already have. * + * Why failure stays open rather than refusing everything: a refusal is not free. + * `terminateWindowsProcessTree` resolves without killing, and + * `killSourceControlAgentProcess` returns that straight to a caller that then + * releases the managed-home lock, so failing closed would trade one unreadable + * metrics table for every PTY, git, codex and notebook tree in main leaking at + * once. The `own_chromium_pids_unreadable` crumb is the price of that choice: + * without it a throw is byte-identical to "no Chromium on this host". + * * Host coverage: only Electron main installs a Chromium-backed AppEnvironment * (main-process-preflight). The standalone daemon installs none and `orcad` * installs a Node one whose `getAppMetrics()` is `[]`, so this set is empty in @@ -30,7 +39,26 @@ export function readOrcaChromiumProcessPids(): ReadonlySet { .map((metric) => metric.pid) .filter((pid) => Number.isInteger(pid) && pid > 0) return new Set(pids) - } catch { + } catch (error) { + recordUnreadableOwnChromiumMetrics(error) return new Set() } } + +// Why coalesced: the gate reads this set on every tree kill, so a persistently +// broken metrics table would otherwise flood the 30-slot ring it shares. +const UNREADABLE_METRICS_COALESCE_MS = 60_000 + +function recordUnreadableOwnChromiumMetrics(error: unknown): void { + try { + recordCoalescedDurableCrashBreadcrumb({ + name: 'own_chromium_pids_unreadable', + data: { cause: error instanceof Error ? error.message : String(error) }, + coalesceKey: 'own-chromium-pids-unreadable', + minIntervalMs: UNREADABLE_METRICS_COALESCE_MS + }) + } catch { + // Diagnostics must never turn an admitted kill into a thrown one: callers + // read this set outside their own try. + } +} diff --git a/src/main/orca-profiles/profile-cloud-session-invalidation.ts b/src/main/orca-profiles/profile-cloud-session-invalidation.ts new file mode 100644 index 00000000000..a9415e28ac6 --- /dev/null +++ b/src/main/orca-profiles/profile-cloud-session-invalidation.ts @@ -0,0 +1,30 @@ +type OrcaCloudSessionInvalidationListener = () => void + +const listeners = new Set() + +/** + * Fires when an auth failure (revoked or rotated-away refresh token) clears a + * stored cloud session. Never fires for an explicit user sign-out, which already + * hands the fresh auth status back to its caller. + */ +export function onOrcaCloudSessionInvalidated( + listener: OrcaCloudSessionInvalidationListener +): () => void { + listeners.add(listener) + return () => { + listeners.delete(listener) + } +} + +export function emitOrcaCloudSessionInvalidated(): void { + for (const listener of listeners) { + try { + listener() + } catch (error) { + console.warn( + '[orca-profiles] Cloud session invalidation listener failed:', + error instanceof Error ? error.message : String(error) + ) + } + } +} diff --git a/src/main/orca-profiles/profile-cloud-session-refresh.test.ts b/src/main/orca-profiles/profile-cloud-session-refresh.test.ts index 42b665d4e35..3b869beb6de 100644 --- a/src/main/orca-profiles/profile-cloud-session-refresh.test.ts +++ b/src/main/orca-profiles/profile-cloud-session-refresh.test.ts @@ -36,6 +36,8 @@ vi.mock('./profile-cloud-client', async (importOriginal) => { vi.mock('./profile-cloud-index', () => ({ linkOrcaProfileToCloud: linkMock })) import { readFreshOrcaCloudSession } from './profile-cloud-session-refresh' +import { OrcaCloudRequestError } from './profile-cloud-client' +import { onOrcaCloudSessionInvalidated } from './profile-cloud-session-invalidation' const config = {} as OrcaCloudAuthConfig const active = { @@ -110,4 +112,47 @@ describe('profile cloud session refresh', () => { expect(saveIfCurrentMock).toHaveBeenCalledTimes(1) expect(linkMock).toHaveBeenCalledTimes(1) }) + + it('notifies subscribers when an auth failure clears the stored session', async () => { + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(clearMock).toHaveBeenCalledTimes(1) + expect(invalidated).toHaveBeenCalledTimes(1) + unsubscribe() + }) + + it('stays silent when a concurrent rotation already replaced the failed session', async () => { + const invalidated = vi.fn() + const unsubscribe = onOrcaCloudSessionInvalidated(invalidated) + refreshMock.mockRejectedValue(new OrcaCloudRequestError(401)) + readMock.mockReturnValueOnce({ + status: 'found', + session: staleSession, + persistence: 'memory-only' + }) + readMock.mockReturnValueOnce({ + status: 'found', + session: staleSession, + persistence: 'memory-only' + }) + readMock.mockReturnValue({ + status: 'found', + session: { ...staleSession, refreshToken: 'rotated-refresh' }, + persistence: 'memory-only' + }) + + await expect(readFreshOrcaCloudSession(config, active, '/data')).resolves.toEqual({ + status: 'reconnect-required' + }) + + expect(clearMock).not.toHaveBeenCalled() + expect(invalidated).not.toHaveBeenCalled() + unsubscribe() + }) }) diff --git a/src/main/orca-profiles/profile-cloud-session-refresh.ts b/src/main/orca-profiles/profile-cloud-session-refresh.ts index ced48a16e7b..32e2e39f6e0 100644 --- a/src/main/orca-profiles/profile-cloud-session-refresh.ts +++ b/src/main/orca-profiles/profile-cloud-session-refresh.ts @@ -13,6 +13,7 @@ import { cloudSessionIdentity, tombstoneCloudSession } from './profile-cloud-session-mutation' +import { emitOrcaCloudSessionInvalidated } from './profile-cloud-session-invalidation' const CLOUD_SESSION_REFRESH_SKEW_MS = 60_000 @@ -66,6 +67,9 @@ function clearCloudSessionIfUnchanged( ) } clearOrcaCloudSession(profileId, userDataPath) + // Why: the renderer cached auth status at startup; without this it keeps + // showing "Connected" until the app restarts. + emitOrcaCloudSessionInvalidated() } async function refreshStoredCloudSession( diff --git a/src/main/own-chromium-tree-kill-guard.test.ts b/src/main/own-chromium-tree-kill-guard.test.ts index 7e98661aca6..bd3b1674e18 100644 --- a/src/main/own-chromium-tree-kill-guard.test.ts +++ b/src/main/own-chromium-tree-kill-guard.test.ts @@ -143,6 +143,40 @@ describe('refusing to tree-kill our own Chromium processes', () => { ) }) + /** + * Fail-open is the deliberate choice — see `orca-chromium-process-pids.ts` for + * why refusing everything is worse — so the crumb is the only thing that keeps + * an unreadable metrics table distinguishable from a host that has no Chromium. + */ + it('leaves proof, and still admits the kill, when the Chromium metrics cannot be read', () => { + appMetricsMock.mockImplementation(() => { + throw new Error('getAppMetrics unavailable') + }) + + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + // Coalesced: the gate reads this set on every kill, so a broken table must + // not evict the ring it shares with the refusal crumb. + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + expect( + admitSelfInitiatedTreeKill({ + pid: RENDERER_PID, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree' + }) + ).toBe(true) + + expect( + getCrashBreadcrumbSnapshot().filter( + (breadcrumb) => breadcrumb.name === 'own_chromium_pids_unreadable' + ) + ).toEqual([ + expect.objectContaining({ + name: 'own_chromium_pids_unreadable', + data: expect.objectContaining({ cause: 'getAppMetrics unavailable' }) + }) + ]) + }) + it('refuses an own-Chromium pid at the gate the account teardowns share', () => { expect( admitSelfInitiatedTreeKill({ diff --git a/src/main/persistence-loading-store-extraction.test.ts b/src/main/persistence-loading-store-extraction.test.ts index e48971d6c5c..a3c5e248909 100644 --- a/src/main/persistence-loading-store-extraction.test.ts +++ b/src/main/persistence-loading-store-extraction.test.ts @@ -116,6 +116,47 @@ describe('loading Store extraction seams', () => { }) }) + it('timestamps persistence-load-done before resolving its details closure', () => { + const sentinel = 'startup-diagnostics-workspace-session-sentinel-ordering' + vi.stubEnv('ORCA_STARTUP_DIAGNOSTICS', '1') + const state = getDefaultPersistedState(testState.dir) + state.workspaceSession = { ...state.workspaceSession, activeTabId: sentinel } + writeDataFile(state) + + // Fake clock only the details closure advances, so a post-closure timestamp is unambiguous. + let clock = 0 + const nowSpy = vi.spyOn(performance, 'now').mockImplementation(() => clock) + const realStringify = JSON.stringify + const stringifySpy = vi.spyOn(JSON, 'stringify').mockImplementation((( + value: unknown, + ...rest: unknown[] + ) => { + if ( + value && + typeof value === 'object' && + (value as { activeTabId?: unknown }).activeTabId === sentinel + ) { + clock += 1000 + } + return (realStringify as (...args: unknown[]) => string)(value, ...rest) + }) as typeof JSON.stringify) + + try { + const store = createStore() + store.freezeWrites() + } finally { + stringifySpy.mockRestore() + nowSpy.mockRestore() + } + + const loadDoneCall = logStartupDiagnosticMock.mock.calls.find( + ([event]) => event === 'persistence-load-done' + ) + const details = loadDoneCall?.[1] as Record | undefined + expect(details?.workspaceSessionBytes).toEqual(expect.any(Number)) + expect(details?.t).toBe(0) + }) + it('accepts the first JSON-parseable backup even when an older backup has richer state', async () => { mkdirSync(testState.dir, { recursive: true }) writeFileSync(dataFile(), '{{corrupt-primary', 'utf-8') diff --git a/src/main/persistence-pty-binding-leaf-tab-resolution.test.ts b/src/main/persistence-pty-binding-leaf-tab-resolution.test.ts new file mode 100644 index 00000000000..2f0dc41daaf --- /dev/null +++ b/src/main/persistence-pty-binding-leaf-tab-resolution.test.ts @@ -0,0 +1,86 @@ +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest' +import { rmSync, mkdtempSync } from 'node:fs' +import { join } from 'node:path' +import { tmpdir } from 'node:os' +import { getDefaultWorkspaceSession } from '../shared/constants' +import { findTerminalTabIdForLeaf } from './runtime/workspace-session-terminal-membership-authority' +import { testState, createStore, makeTerminalTab } from './persistence-test-harness' +import { TEST_LEAF_1, TEST_LEAF_2 } from './persistence-session-fixtures' + +vi.mock('electron', () => ({ + app: { getPath: () => testState.dir }, + safeStorage: { isEncryptionAvailable: () => false } +})) + +vi.mock('./telemetry/client', () => ({ track: vi.fn() })) +vi.mock('./telemetry/cohort-classifier', () => ({ getCohortAtEmit: () => ({}) })) + +describe('findTerminalTabIdForLeaf after persistPtyBinding grafts a leaf', () => { + beforeEach(() => { + testState.dir = mkdtempSync(join(tmpdir(), 'orca-test-')) + }) + afterEach(() => { + rmSync(testState.dir, { recursive: true, force: true }) + }) + + // `persistPtyBinding` grafts the leaf by assigning `layout.root` on the SAME layout object inside + // the SAME layouts record (pty-binding-persistence.ts), so the resolver has to answer from the + // tree that is there now, not from anything derived on an earlier call. + it('resolves a leaf grafted in place by a split spawn', async () => { + const store = await createStore() + store.setWorkspaceSession({ + ...getDefaultWorkspaceSession(), + tabsByWorktree: { + wt1: [makeTerminalTab({ id: 'tab1', worktreeId: 'wt1', ptyId: 'pty-source' })] + }, + terminalLayoutsByTabId: { + tab1: { + root: { type: 'leaf', leafId: TEST_LEAF_1 }, + activeLeafId: TEST_LEAF_1, + expandedLeafId: null, + ptyIdsByLeafId: { [TEST_LEAF_1]: 'pty-source' } + } + } + }) + // A reader runs first, exactly as the syncWindowGraph lease sweep does. + expect(findTerminalTabIdForLeaf(store.getWorkspaceSession(), TEST_LEAF_1)).toBe('tab1') + + expect( + store.persistPtyBinding({ + worktreeId: 'wt1', + tabId: 'tab1', + leafId: TEST_LEAF_2, + ptyId: 'pty-split' + }) + ).toBe(true) + + expect(findTerminalTabIdForLeaf(store.getWorkspaceSession(), TEST_LEAF_2)).toBe('tab1') + expect(findTerminalTabIdForLeaf(store.getWorkspaceSession(), TEST_LEAF_1)).toBe('tab1') + }) + + // The other in-place graft: an empty persisted layout gets its first durable root. + it('resolves the first leaf grafted onto an empty layout', async () => { + const store = await createStore() + store.setWorkspaceSession({ + ...getDefaultWorkspaceSession(), + tabsByWorktree: { + wt1: [makeTerminalTab({ id: 'tab1', worktreeId: 'wt1', ptyId: null })] + }, + terminalLayoutsByTabId: { + tab1: { root: null, activeLeafId: null, expandedLeafId: null, ptyIdsByLeafId: {} } + } + }) + expect(findTerminalTabIdForLeaf(store.getWorkspaceSession(), TEST_LEAF_1)).toBeUndefined() + + expect( + store.persistPtyBinding({ + worktreeId: 'wt1', + tabId: 'tab1', + leafId: TEST_LEAF_1, + ptyId: 'pty-first' + }) + ).toBe(true) + + expect(findTerminalTabIdForLeaf(store.getWorkspaceSession(), TEST_LEAF_1)).toBe('tab1') + }) +}) diff --git a/src/main/persistence-ssh-lease-reattach-reclaim.test.ts b/src/main/persistence-ssh-lease-reattach-reclaim.test.ts index 9d6c74f710e..b2088fa616b 100644 --- a/src/main/persistence-ssh-lease-reattach-reclaim.test.ts +++ b/src/main/persistence-ssh-lease-reattach-reclaim.test.ts @@ -49,14 +49,16 @@ describe('ssh remote pty lease reclaim after a proven reattach', () => { expect(sshRemotePtyLeaseAllowsReattach(lease)).toBe(true) }) - it('leaves a terminated lease absorbing even when the id appears in a reattach batch', async () => { + it('never lets a reattach batch revive an operator-closed id', async () => { const store = await createStore() store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId: 'pty-1', state: 'attached' }) store.markSshRemotePtyLease('ssh-1', 'pty-1', 'terminated') await store.markSshRemotePtyLeasesAttachedAsync('ssh-1', ['pty-1']) - expect(store.getSshRemotePtyLeases('ssh-1')[0]).toMatchObject({ state: 'terminated' }) + // The unbound tombstone is retired at close, and the batch only ever updates existing rows — + // so the id stays out of the reattach set either way. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) }) it('does not revive an expired lease from an unqualified bulk attach', async () => { diff --git a/src/main/persistence-ssh-lease-tombstone-retention.test.ts b/src/main/persistence-ssh-lease-tombstone-retention.test.ts new file mode 100644 index 00000000000..feafdd766e4 --- /dev/null +++ b/src/main/persistence-ssh-lease-tombstone-retention.test.ts @@ -0,0 +1,147 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { createStore, testState } from './persistence-test-harness' +import { TEST_LEAF_1 } from './persistence-session-fixtures' + +vi.mock('electron', () => ({ + app: { getPath: () => testState.dir }, + safeStorage: { isEncryptionAvailable: () => false } +})) + +vi.mock('./telemetry/client', () => ({ track: vi.fn() })) +vi.mock('./telemetry/cohort-classifier', () => ({ getCohortAtEmit: () => ({}) })) + +describe('operator-closed SSH lease tombstones', () => { + beforeEach(() => { + testState.dir = mkdtempSync(join(tmpdir(), 'orca-test-')) + }) + + afterEach(() => { + rmSync(testState.dir, { recursive: true, force: true }) + }) + + /** A pane whose lease froze `tab-old` before `detachTerminalPaneToTab` moved it to `tab-new`. + * The binding scrub matches tab-qualified, so it cannot reach this row's binding. */ + async function storeWithDetachedPaneBinding(): Promise>> { + const store = await createStore() + store.upsertSshRemotePtyLease({ + targetId: 'ssh-1', + ptyId: 'remote-pty', + worktreeId: 'wt1', + tabId: 'tab-old', + leafId: TEST_LEAF_1, + state: 'attached' + }) + store.setWorkspaceSession({ + activeRepoId: 'r1', + activeWorktreeId: 'wt1', + activeTabId: 'tab-new', + tabsByWorktree: { + wt1: [ + { + id: 'tab-new', + worktreeId: 'wt1', + title: 'Terminal', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 1, + ptyId: null + } + ] + }, + terminalLayoutsByTabId: { + 'tab-new': { + root: { type: 'leaf', leafId: TEST_LEAF_1 }, + activeLeafId: TEST_LEAF_1, + expandedLeafId: null, + ptyIdsByLeafId: { [TEST_LEAF_1]: 'ssh:ssh-1@@remote-pty' } + } + } + }) + return store + } + + it('keeps the tombstone while a binding the scrub could not reach still names the pty', async () => { + const store = await storeWithDetachedPaneBinding() + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') + + // `isRestorablePtyBinding` still consults this row to refuse replaying that binding. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ + expect.objectContaining({ ptyId: 'remote-pty', state: 'terminated' }) + ]) + expect(store.getWorkspaceSession().terminalLayoutsByTabId['tab-new'].ptyIdsByLeafId).toEqual({ + [TEST_LEAF_1]: 'ssh:ssh-1@@remote-pty' + }) + }) + + it('keeps an operator-closed lease that still owes an undelivered stop', async () => { + const store = await createStore() + store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId: 'remote-pty', state: 'attached' }) + store.recordSshRemotePtyKillIntent('ssh-1', 'remote-pty', { + incarnationId: 'inc-1', + requestedAt: 1, + attempts: 0 + }) + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') + + expect(store.getSshRemotePtyKillIntents('ssh-1', 2)).toHaveLength(1) + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ + expect.objectContaining({ ptyId: 'remote-pty', state: 'terminated' }) + ]) + }) + + // `expired` is never evidence the shell died, and `sweepOrphanedRelayPtys` reads these ids as its + // leave-alone list, so dropping one would authorize stopping a process left running on purpose. + it('keeps a superseded expired lease when a sibling pane is closed', async () => { + const store = await createStore() + store.upsertSshRemotePtyLease({ + targetId: 'ssh-1', + ptyId: 'remote-pty-1', + worktreeId: 'wt1', + tabId: 'tab1', + leafId: TEST_LEAF_1, + state: 'attached' + }) + store.upsertSshRemotePtyLease({ + targetId: 'ssh-1', + ptyId: 'remote-pty-2', + worktreeId: 'wt1', + tabId: 'tab1', + leafId: TEST_LEAF_1, + state: 'attached' + }) + store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId: 'remote-pty-3', state: 'attached' }) + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty-3', 'terminated') + + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ + expect.objectContaining({ + ptyId: 'remote-pty-1', + state: 'expired', + supersededBy: 'remote-pty-2' + }), + expect.objectContaining({ ptyId: 'remote-pty-2', state: 'attached' }) + ]) + }) + + it('retires every unreachable tombstone for the target, not only the one just closed', async () => { + const store = await createStore() + for (const ptyId of ['remote-pty-1', 'remote-pty-2', 'remote-pty-3']) { + store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId, state: 'terminated' }) + } + store.upsertSshRemotePtyLease({ targetId: 'ssh-2', ptyId: 'other-pty', state: 'terminated' }) + expect(store.getSshRemotePtyLeases()).toHaveLength(4) + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty-1', 'terminated') + + // Other targets are untouched: the pass is scoped to the one whose bindings were just scrubbed. + expect(store.getSshRemotePtyLeases()).toEqual([ + expect.objectContaining({ targetId: 'ssh-2', ptyId: 'other-pty' }) + ]) + }) +}) diff --git a/src/main/persistence-ssh-remote-pty-leases.test.ts b/src/main/persistence-ssh-remote-pty-leases.test.ts index 464138f69d4..d4cdc79285c 100644 --- a/src/main/persistence-ssh-remote-pty-leases.test.ts +++ b/src/main/persistence-ssh-remote-pty-leases.test.ts @@ -526,12 +526,9 @@ describe('Store', () => { store.markSshRemotePtyLeases('ssh-1', 'terminated') const session = store.getWorkspaceSession() - expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ - expect.objectContaining({ - ptyId: 'remote-pty', - state: 'terminated' - }) - ]) + // The scrub is what retires the row: with no binding left naming the id, the tombstone routes + // nothing and is dropped in the same write. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) expect(session.tabsByWorktree.wt1[0].ptyId).toBeNull() expect(session.terminalLayoutsByTabId.tab1.ptyIdsByLeafId).toEqual({}) }) @@ -622,12 +619,8 @@ describe('Store', () => { store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') - expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ - expect.objectContaining({ - ptyId: 'remote-pty', - state: 'terminated' - }) - ]) + // An unresolved id would have left the lease `attached`; this unbound row is retired instead. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) }) // `expired` never means the shell exited — every writer records that the CLIENT lost its route @@ -658,12 +651,7 @@ describe('Store', () => { store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') const session = store.getWorkspaceSession() - expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ - expect.objectContaining({ - ptyId: 'remote-pty', - state: 'terminated' - }) - ]) + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) expect(session.tabsByWorktree.wt1[0].ptyId).toBeNull() expect(session.terminalLayoutsByTabId.tab1.ptyIdsByLeafId).toEqual({}) }) diff --git a/src/main/persistence-test-harness.ts b/src/main/persistence-test-harness.ts index 7b628b4b678..e2c04495e92 100644 --- a/src/main/persistence-test-harness.ts +++ b/src/main/persistence-test-harness.ts @@ -6,6 +6,8 @@ import type { Repo } from '../shared/repo-types' import type { TerminalTab } from '../shared/terminal-tab-types' import type { WorkspaceLineage, WorktreeLineage } from '../shared/worktree/lineage-types' import { folderWorkspaceKey, worktreeWorkspaceKey } from '../shared/workspace-scope' +import type { PersistedState } from '../shared/persisted-state-types' +import { hydrateWorktreeMetaAliasProjection } from './persistence/loading-store/worktree-meta-alias-projection' import { Store } from './persistence/loading-store/store' import { initDataPath } from './persistence/loading-store/user-data-path' @@ -38,8 +40,16 @@ export function writeDataFile(data: unknown): void { writeFileSync(dataFile(), JSON.stringify(data, null, 2), 'utf-8') } +/** + * The persisted state as a reader gets it, not the raw bytes: the serializer omits any + * `worktreeMetaByIdentity` row the locator row regenerates, and every consumer of this file -- + * including the Store's own load path -- rebuilds those before looking at them. Tests that need + * the literal bytes parse the file themselves (see `worktree-meta-alias-projection.test.ts`). + */ export function readDataFile(): unknown { - return JSON.parse(readFileSync(dataFile(), 'utf-8')) + const parsed = JSON.parse(readFileSync(dataFile(), 'utf-8')) as PersistedState + hydrateWorktreeMetaAliasProjection(parsed) + return parsed } export function symlinkDirectorySync(target: string, linkPath: string): void { diff --git a/src/main/persistence/host-qualified-worktree-meta.ts b/src/main/persistence/host-qualified-worktree-meta.ts index a9d1c0e8fb1..6267311f471 100644 --- a/src/main/persistence/host-qualified-worktree-meta.ts +++ b/src/main/persistence/host-qualified-worktree-meta.ts @@ -1,4 +1,5 @@ -import type { ExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' +import type { Repo } from '../../shared/repo-types' import type { WorktreeMeta } from '../../shared/worktree/meta-types' /** @@ -52,6 +53,26 @@ export function readWorktreeMetaForHost( return store.getWorktreeMetaForHost?.(worktreeId, executionHostId) } +/** + * The same two reads keyed off a repo row, so the resolve-then-read pair lives in one place. Four + * call sites had open-coded it identically, which is the shape that lets one copy drift from the + * rest (F7/F8). + */ +export function readAllWorktreeMetaForRepo( + store: Pick, + repo: Pick +): Record { + return readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) +} + +export function readWorktreeMetaForRepo( + store: Pick, + worktreeId: string, + repo: Pick +): WorktreeMeta | undefined { + return readWorktreeMetaForHost(store, worktreeId, getRepoExecutionHostId(repo)) +} + export function writeWorktreeMetaForHost( store: Pick, worktreeId: string, diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts index f0628cb25e5..f06e5b0ece7 100644 --- a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts @@ -2,6 +2,7 @@ import type { PersistedState } from '../../../shared/persisted-state-types' import type { SshRemotePtyLease } from '../../../shared/ssh-types' import { isTerminalLeafId } from '../../../shared/stable-pane-id' import { invalidateLocalWorktreeMetadataPruneInputs } from '../../local-worktree-metadata-prune-gate' +import { pruneRetiredSshRemotePtyLeaseTombstones } from './ssh-pty-lease-tombstone-retention' import { supersedeSiblingLeasesForPane } from './ssh-pty-pane-supersession' export type SshPtyLeaseOperations = { @@ -145,7 +146,11 @@ function updateSshRemotePtyLeaseStates( const bindingsChanged = shouldClearBindings ? operations.clearBindingsForLeases(targetId, leasesToClear) : false - return changed || bindingsChanged + // Why after the scrub: it is the scrub that makes the tombstones unreachable. + const tombstonesPruned = shouldClearBindings + ? pruneRetiredSshRemotePtyLeaseTombstones(operations, targetId) + : false + return changed || bindingsChanged || tombstonesPruned } export function markSshRemotePtyLeases( @@ -215,10 +220,11 @@ export function markSshRemotePtyLease( } const shouldClearBindings = leaseStateWithdrawsBinding(state) if (lease.state === state) { - if ( - (shouldClearBindings && operations.clearBindingsForLeases(targetId, [lease])) || - recycledChanged - ) { + const bindingsCleared = + shouldClearBindings && operations.clearBindingsForLeases(targetId, [lease]) + const tombstonesPruned = + shouldClearBindings && pruneRetiredSshRemotePtyLeaseTombstones(operations, targetId) + if (bindingsCleared || tombstonesPruned || recycledChanged) { operations.flush() } return @@ -233,6 +239,7 @@ export function markSshRemotePtyLease( } if (shouldClearBindings) { operations.clearBindingsForLeases(targetId, [lease]) + pruneRetiredSshRemotePtyLeaseTombstones(operations, targetId) } operations.flush() } diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-tombstone-retention.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-tombstone-retention.ts new file mode 100644 index 00000000000..261162a3825 --- /dev/null +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-tombstone-retention.ts @@ -0,0 +1,89 @@ +import type { PersistedState } from '../../../shared/persisted-state-types' +import type { SshRemotePtyLease } from '../../../shared/ssh-types' + +export type SshPtyLeaseTombstoneRetentionOperations = { + state: PersistedState + toComparablePtyId: (targetId: string, ptyId: string) => string +} + +/** A routing tombstone with nothing left to route: the operator closed this PTY and no stop is + * still owed for it. `expired` is deliberately not here — it says only that the CLIENT lost its + * route (docs/reference/ssh-execution-boundary.md), and `sweepOrphanedRelayPtys` reads those ids + * as its leave-alone list, so deleting one would authorize stopping a remote shell that + * supersession left running on purpose. */ +function isRetiredRoutingTombstone(lease: SshRemotePtyLease, targetId: string): boolean { + return ( + lease.targetId === targetId && lease.state === 'terminated' && lease.pendingKill === undefined + ) +} + +/** Every stored-form relay pty id some persisted pane binding still names for this target. + * + * Reads all partitions, not only the two `clearSshRemotePtyBindingsForLeases` scrubs: this answer + * authorizes a delete, so a partition left unscanned would be a binding whose tombstone we dropped. + */ +function boundRelayPtyIds( + operations: SshPtyLeaseTombstoneRetentionOperations, + targetId: string +): Set { + const bound = new Set() + const sessions = [ + operations.state.workspaceSession, + ...Object.values(operations.state.workspaceSessionsByHostId ?? {}) + ] + for (const session of sessions) { + if (!session) { + continue + } + for (const tabs of Object.values(session.tabsByWorktree ?? {})) { + for (const tab of tabs) { + if (tab.ptyId) { + bound.add(operations.toComparablePtyId(targetId, tab.ptyId)) + } + } + } + for (const layout of Object.values(session.terminalLayoutsByTabId ?? {})) { + for (const ptyId of Object.values(layout?.ptyIdsByLeafId ?? {})) { + bound.add(operations.toComparablePtyId(targetId, ptyId)) + } + } + } + return bound +} + +/** + * Deletes the `terminated` rows nothing can reach, bounding an array that otherwise only grew. + * + * `terminated` is written with a binding scrub in the same call, so once no persisted binding names + * the id the row answers no question any reader asks. Reattach refuses it + * (`sshRemotePtyLeaseAllowsReattach`), pane recovery matches on `expired` only, the orphan sweep + * already classes it neither routed nor expired, and `ssh:reset` / `ssh:terminateSessions` skip it + * outright — every one of those behaves identically on an absent row. The one reader that can still + * observe it is `isRestorablePtyBinding`, and only through a binding whose pty id matches, which is + * exactly what the reachability test rules out. A `pendingKill` is an undelivered stop, so those + * rows stay until the replay retires them. + * + * The reachability test is not redundant with the scrub: a lease freezes its `tabId`, so a pane + * broken out into a new tab leaves a binding the scrub's tab-qualified match no longer reaches. + * + * Does not re-arm the local-worktree-metadata prune gate: a `terminated` lease no longer counts as + * a persisted workspace owner, so dropping one cannot make any metadata row more removable. + */ +export function pruneRetiredSshRemotePtyLeaseTombstones( + operations: SshPtyLeaseTombstoneRetentionOperations, + targetId: string +): boolean { + const leases = operations.state.sshRemotePtyLeases ?? [] + if (!leases.some((lease) => isRetiredRoutingTombstone(lease, targetId))) { + return false + } + const bound = boundRelayPtyIds(operations, targetId) + const retained = leases.filter( + (lease) => !isRetiredRoutingTombstone(lease, targetId) || bound.has(lease.ptyId) + ) + if (retained.length === leases.length) { + return false + } + operations.state.sshRemotePtyLeases = retained + return true +} diff --git a/src/main/persistence/loading-store/loaded-state-parsing.ts b/src/main/persistence/loading-store/loaded-state-parsing.ts index 5bd6e572ab1..e4034fa2b17 100644 --- a/src/main/persistence/loading-store/loaded-state-parsing.ts +++ b/src/main/persistence/loading-store/loaded-state-parsing.ts @@ -51,8 +51,10 @@ function logPersistenceStartupMilestone( if (!isStartupDiagnosticsEnabled()) { return } + // Why: snapshot `t` before resolving lazy details — otherwise an expensive details closure is billed to the milestone it measures. + const t = Math.round(performance.now()) const resolvedDetails = typeof details === 'function' ? details() : details - logStartupDiagnostic(event, { t: Math.round(performance.now()), ...resolvedDetails }) + logStartupDiagnostic(event, { t, ...resolvedDetails }) } import type { StoreRuntimeState } from './store-runtime-state' diff --git a/src/main/persistence/loading-store/normalize-loaded-profile-state.ts b/src/main/persistence/loading-store/normalize-loaded-profile-state.ts index 39d625c525a..987a6e81a17 100644 --- a/src/main/persistence/loading-store/normalize-loaded-profile-state.ts +++ b/src/main/persistence/loading-store/normalize-loaded-profile-state.ts @@ -25,6 +25,7 @@ import { normalizeLoadedProjectCatalog } from './normalize-loaded-state-collections' import { normalizeRetiredNameRegistryMap } from './retired-name-registry-normalization' +import { hydrateWorktreeMetaAliasProjection } from './worktree-meta-alias-projection' export function normalizeLoadedProfileState( parsed: PersistedState, @@ -53,6 +54,13 @@ export function normalizeLoadedProfileState( folderWorkspaceDiffComments: normalizeFolderWorkspaceDiffComments( parsed.folderWorkspaceDiffComments ), + // Rebuilds the identity rows the serializer left to the locator map, and restores the shared + // object reference JSON.parse splits. Not `markNeedsSave`: this IS the canonical on-disk shape. + // Conditional so a file with no identity map keeps none, rather than gaining an own key whose + // value is `undefined`. + ...(parsed.worktreeMetaByIdentity === undefined + ? {} + : { worktreeMetaByIdentity: hydrateWorktreeMetaAliasProjection(parsed) }), worktreeLineageById: parsed.worktreeLineageById ?? {}, mobileClientTabSelectionsByDeviceId: normalizePersistedMobileClientTabSelections( parsed.mobileClientTabSelectionsByDeviceId diff --git a/src/main/persistence/loading-store/state-serialization-secret-handling.ts b/src/main/persistence/loading-store/state-serialization-secret-handling.ts index 844a0381fce..c8557f846af 100644 --- a/src/main/persistence/loading-store/state-serialization-secret-handling.ts +++ b/src/main/persistence/loading-store/state-serialization-secret-handling.ts @@ -8,6 +8,7 @@ import { } from '../../protected-secret-persistence' import { stripRetiredGlobalSettings } from '../applying-settings/terminal-settings-migrations' import { omitDefaultWorktreeMetaFieldsInMap } from '../../../shared/worktree/meta-persisted-defaults' +import { projectWorktreeMetaByIdentityOntoLocators } from './worktree-meta-alias-projection' import { withoutRedundantPartitionGlobals } from '../../../shared/workspace-session-host-field-ownership' import { @@ -68,16 +69,28 @@ export class StateSerializationSecretHandlingOperations { const encrypted = encryptToSentinel(slot, plaintext ?? '') return encrypted || null } + // Ordered before the default omission on purpose: the two maps hold the SAME row object, so + // the projection settles almost every row on a reference check. Omitting first rebuilds each + // row twice into two distinct objects and forces a deep compare per row instead. Omission is + // a pure function of the value, so a pair equal here is equal after it too -- and it never + // touches `hostId`/`instanceId`, which is what the reader re-derives the omitted key from. + const projectedWorktreeMetaByIdentity = + this.runtime.state.worktreeMetaByIdentity === undefined + ? undefined + : projectWorktreeMetaByIdentityOntoLocators( + this.runtime.state.worktreeMetaByIdentity, + this.runtime.state + ) // Why: clone before encrypting secrets so in-memory this.state stays plaintext. const stateToSave = { ...this.getDurableState(), // Default-valued metadata slots are re-filled at load (normalizeWorktreeLinkedItemMetadata), // so omitting them here is lossless and drops ~12% of the file on a heavy install. worktreeMeta: omitDefaultWorktreeMetaFieldsInMap(this.runtime.state.worktreeMeta), - ...(this.runtime.state.worktreeMetaByIdentity !== undefined + ...(projectedWorktreeMetaByIdentity !== undefined ? { worktreeMetaByIdentity: omitDefaultWorktreeMetaFieldsInMap( - this.runtime.state.worktreeMetaByIdentity + projectedWorktreeMetaByIdentity ) } : {}), diff --git a/src/main/persistence/loading-store/worktree-meta-alias-projection.test.ts b/src/main/persistence/loading-store/worktree-meta-alias-projection.test.ts new file mode 100644 index 00000000000..f324787cdf3 --- /dev/null +++ b/src/main/persistence/loading-store/worktree-meta-alias-projection.test.ts @@ -0,0 +1,421 @@ +/** + * `setWorktreeMetaForHost` puts one object in both `worktreeMeta` and `worktreeMetaByIdentity`, so + * a heavy profile serializes every metadata row twice. On a measured 3.64 MB install 1,347 of + * 1,349 locator rows were byte-identical to their identity twin and cost 540 KB per save. + * + * These tests drive the real Store over a seeded corpus that contains every shape the projection + * has to get right -- identical twins, divergent twins, rows with no identity at all, one locator + * claimed by two hosts, an alias whose locator row was pruned away, a dangling identity key, and + * an ambiguous alias with two instances behind one locator -- and pin the properties that make the + * omission safe: load(save(x)) deep-equals x, an old-serializer file and a new-serializer file load + * to the same state, the locator map is never reduced (which is what makes a downgrade lossless), + * and a build with no rebuild at all recovers every row from the file the new build wrote. + */ +import { mkdtempSync, readFileSync, realpathSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { PersistedState } from '../../../shared/persisted-state-types' +import type { WorktreeMeta } from '../../../shared/worktree/meta-types' +import { canonicalWorktreeIdentity } from '../../../shared/worktree/identity' +import { composeWorktreeHostIdentity } from '../../../shared/worktree/host-qualified-identity' +import { normalizeWorktreeLinkedItemMetadata } from '../tracking-repos/worktree-metadata-normalization' + +vi.mock('electron', () => ({ + app: { + getPath: () => tmpdir(), + getName: () => 'orca-test', + getVersion: () => '0.0.0-test', + isPackaged: false, + on: () => {}, + whenReady: () => Promise.resolve() + }, + safeStorage: { isEncryptionAvailable: () => false }, + ipcMain: { on: () => {}, handle: () => {} }, + BrowserWindow: { getAllWindows: () => [] } +})) + +const { Store } = await import('./store') + +const REPO_ID = 'repo-1' +const LOCAL = 'local' +const REMOTE = 'ssh:user@host' +const TWIN_ROWS = 400 +/** Recent enough that the 30-day stale-metadata GC leaves the fixture alone. */ +const RECENTLY = Date.now() + +/** Seeded so the corpus is the same on every run and a failure is reproducible. */ +function seededRandom(seed: number): () => number { + let state = seed >>> 0 + return () => { + state = (state * 1_664_525 + 1_013_904_223) >>> 0 + return state / 0x1_0000_0000 + } +} + +const stores: InstanceType[] = [] +afterEach(() => { + for (const store of stores.splice(0)) { + store.freezeWrites() + } + vi.restoreAllMocks() +}) + +function openStore(dataFile: string): InstanceType { + const store = new Store({ dataFile }) + stores.push(store) + return store +} + +function tempDataFile(): string { + return join(realpathSync(mkdtempSync(join(tmpdir(), 'orca-alias-projection-'))), 'orca-data.json') +} + +function worktreeId(index: number): string { + return `${REPO_ID}::/tmp/wt-${index}` +} + +/** Every optional slot exercised on a fraction of rows, so a row that must stay written does. */ +function meta(index: number, random: () => number, overrides: Partial = {}) { + const rich = random() < 0.25 + return { + instanceId: `instance-${index}`, + hostId: LOCAL, + displayName: `workspace-${index}`, + comment: rich ? `note ${index}` : '', + linkedIssue: null, + linkedPR: rich ? index : null, + linkedLinearIssue: null, + linkedWorkItem: null, + linkedTaskSourceContext: null, + isArchived: false, + isUnread: random() < 0.3, + isPinned: rich, + sortOrder: RECENTLY + index, + manualOrder: rich ? index : undefined, + lastActivityAt: RECENTLY + index, + createdAt: RECENTLY, + baseRef: rich ? 'main' : undefined, + workspaceStatus: 'none', + ...overrides + } as WorktreeMeta +} + +type Fixture = { + state: PersistedState + /** Identity keys the locator row regenerates on its own, so they must leave the file. */ + omittable: string[] + /** Identity keys no locator row regenerates, so they must stay on disk. */ + irreducible: string[] +} + +/** + * A file in the pre-change shape: every alias' identity row duplicated into `worktreeMeta`, which + * is exactly what the old serializer wrote. + */ +function buildFixture(): Fixture { + const random = seededRandom(20_260_903) + const worktreeMeta: Record = {} + const worktreeMetaByIdentity: Record = {} + const worktreeIdentityAliases: Record = {} + const omittable: string[] = [] + const irreducible: string[] = [] + + const link = (id: string, host: string, row: WorktreeMeta): string => { + const identityKey = canonicalWorktreeIdentity({ + worktreeId: id, + executionHostId: host as never, + instanceId: row.instanceId as string + }) + worktreeMetaByIdentity[identityKey] = row + worktreeIdentityAliases[composeWorktreeHostIdentity(host as never, id)] = [identityKey] + return identityKey + } + + // 1. The common case: the identity row and the locator row are the same value. + for (let index = 0; index < TWIN_ROWS; index++) { + const row = meta(index, random) + worktreeMeta[worktreeId(index)] = { ...row } + omittable.push(link(worktreeId(index), LOCAL, row)) + } + // 2. Divergent twin: the locator row carries a value the identity row does not. + const divergent = worktreeId(TWIN_ROWS) + const divergentRow = meta(TWIN_ROWS, random) + worktreeMeta[divergent] = { ...divergentRow, displayName: 'locator-only-name' } + irreducible.push(link(divergent, LOCAL, divergentRow)) + // 3. No identity twin at all, and no hostId — the shape of Orca's synthetic pseudo-worktrees. + for (const pseudo of ['global-floating-terminal', 'onboarding-setup-terminal']) { + worktreeMeta[pseudo] = meta(0, random, { hostId: undefined, displayName: pseudo }) + } + // 4. One locator claimed by two hosts: nothing on disk records which one owns the projection. + const contested = worktreeId(TWIN_ROWS + 1) + const localClaim = meta(TWIN_ROWS + 1, random) + const remoteClaim = meta(TWIN_ROWS + 1, random, { + hostId: REMOTE as never, + instanceId: `instance-${TWIN_ROWS + 1}-remote`, + lastActivityAt: RECENTLY + 99_999 + }) + worktreeMeta[contested] = { ...localClaim } + // Only the host the locator row names can regenerate a key from it, so the other host's row stays. + omittable.push(link(contested, LOCAL, localClaim)) + irreducible.push(link(contested, REMOTE, remoteClaim)) + // 5. An alias whose locator row a host-scoped prune already removed: rebuilding it would + // resurrect a workspace the user deleted. + const voided = worktreeId(TWIN_ROWS + 2) + irreducible.push(link(voided, REMOTE, meta(TWIN_ROWS + 2, random, { hostId: REMOTE as never }))) + // 6. A dangling identity key: the alias points at a row that is not there. + const dangling = worktreeId(TWIN_ROWS + 3) + worktreeMeta[dangling] = meta(TWIN_ROWS + 3, random) + worktreeIdentityAliases[composeWorktreeHostIdentity(LOCAL, dangling)] = ['wt2:local:missing'] + // 7. An ambiguous alias — two instances behind one locator. `setWorktreeMetaForHost` refuses to + // write one, so it is a repair state and its locator row must stay written in full. + const ambiguous = worktreeId(TWIN_ROWS + 4) + const claimA = meta(TWIN_ROWS + 4, random) + const claimB = meta(TWIN_ROWS + 4, random, { + instanceId: `instance-${TWIN_ROWS + 4}-b`, + displayName: 'second-instance', + lastActivityAt: RECENTLY + 99_999 + }) + worktreeMeta[ambiguous] = { ...claimA } + const ambiguousKey = link(ambiguous, LOCAL, claimA) + irreducible.push(ambiguousKey) + const secondKey = canonicalWorktreeIdentity({ + worktreeId: ambiguous, + executionHostId: LOCAL as never, + instanceId: claimB.instanceId as string + }) + worktreeMetaByIdentity[secondKey] = claimB + worktreeIdentityAliases[composeWorktreeHostIdentity(LOCAL, ambiguous)] = [ambiguousKey, secondKey] + irreducible.push(secondKey) + + return { + state: { + // Registered: the load-time deregistered-repo sweep drops residue rows for unknown repos. + repos: [{ id: REPO_ID, name: REPO_ID, path: '/tmp/repo-1', worktreesPath: '/tmp' }], + projects: [], + worktreeMeta, + worktreeMetaByIdentity, + worktreeIdentityAliases, + worktreeLineageById: {}, + workspaceLineageByChildKey: {} + } as unknown as PersistedState, + omittable, + irreducible + } +} + +function writeFixture(dataFile: string, state: PersistedState): void { + writeFileSync(dataFile, JSON.stringify(state), 'utf-8') +} + +function snapshot(store: InstanceType) { + return { + meta: structuredClone(store.getAllWorktreeMeta()), + local: structuredClone(store.getAllWorktreeMetaForHost(LOCAL)), + remote: structuredClone(store.getAllWorktreeMetaForHost(REMOTE as never)) + } +} + +describe('worktree meta alias projection', () => { + it('round-trips every corpus shape and writes only the identity rows no locator regenerates', () => { + const fixture = buildFixture() + const dataFile = tempDataFile() + writeFixture(dataFile, fixture.state) + + // One load+flush first, so the baseline is not comparing against the one-time settings + // migrations a synthetic fixture triggers (same reason as state-write-round-trip.test.ts). + openStore(dataFile).flush() + + const loaded = openStore(dataFile) + const before = snapshot(loaded) + loaded.flush() + const rewritten = readFileSync(dataFile, 'utf-8') + const onDisk = JSON.parse(rewritten) as PersistedState + + // The counter this change exists for: 401 regenerable identity rows leave the file. + expect(Object.keys(onDisk.worktreeMetaByIdentity ?? {}).sort()).toEqual( + [...fixture.irreducible].sort() + ) + expect(fixture.omittable).toHaveLength(TWIN_ROWS + 1) + // ...and the locator map, which is what regenerates them, is written in full. This is the + // property the downgrade story rests on, so it is asserted as a set, not a count. + expect(Object.keys(onDisk.worktreeMeta).sort()).toEqual(Object.keys(before.meta).sort()) + + // load(save(x)) deep-equals x, for every reader of the metadata maps. + const reloaded = openStore(dataFile) + expect(reloaded.getAllWorktreeMeta()).toEqual(before.meta) + expect(reloaded.getAllWorktreeMetaForHost(LOCAL)).toEqual(before.local) + expect(reloaded.getAllWorktreeMetaForHost(REMOTE as never)).toEqual(before.remote) + // The locator row a host-scoped prune already removed stays removed. + expect(reloaded.getAllWorktreeMeta()).not.toHaveProperty(worktreeId(TWIN_ROWS + 2)) + // The contested locator keeps the host that owned the projection, not the newer claim. + expect(reloaded.getAllWorktreeMeta()[worktreeId(TWIN_ROWS + 1)]?.hostId).toBe(LOCAL) + // The ambiguous locator keeps its own row, not the newer instance behind the same alias. + expect(reloaded.getAllWorktreeMeta()[worktreeId(TWIN_ROWS + 4)]?.displayName).toBe( + `workspace-${TWIN_ROWS + 4}` + ) + + // A quiet app does not rewrite the file with new content on the next flush. + reloaded.flush() + expect(readFileSync(dataFile, 'utf-8')).toBe(rewritten) + }) + + /** + * The risk this projection direction exists to remove. A build without the rebuild -- an older + * one, or any raw reader of the file -- gets a complete `worktreeMeta`; its normalizer drops the + * now-dangling aliases and `migrateLegacyWorktreeMetadata` re-mints the identical identity key + * from the `instanceId` the locator row still carries. Nothing is lost at any step. + */ + it('loses no row on a build that has no rebuild at all', () => { + const fixture = buildFixture() + const dataFile = tempDataFile() + writeFixture(dataFile, fixture.state) + openStore(dataFile).flush() + const upgraded = openStore(dataFile) + const before = snapshot(upgraded) + upgraded.flush() + + // What a build without this change does with that file: parse it, run the metadata normalizer + // it already ships (untouched here), write the result back. + const downgraded = JSON.parse(readFileSync(dataFile, 'utf-8')) as PersistedState + normalizeWorktreeLinkedItemMetadata(downgraded) + expect(Object.keys(downgraded.worktreeMeta).sort()).toEqual(Object.keys(before.meta).sort()) + // It drops the aliases whose identity row is not there; it never touches a locator row. + expect(downgraded.worktreeIdentityAliases).not.toHaveProperty( + composeWorktreeHostIdentity(LOCAL, worktreeId(0)) + ) + writeFileSync(dataFile, JSON.stringify(downgraded), 'utf-8') + + // Every reader is where it started, with no rebuild and without touching a row first. + const rolledBack = openStore(dataFile) + expect(rolledBack.getAllWorktreeMeta()).toEqual(before.meta) + expect(rolledBack.getAllWorktreeMetaForHost(LOCAL)).toEqual(before.local) + expect(rolledBack.getAllWorktreeMetaForHost(REMOTE as never)).toEqual(before.remote) + + // ...and the first touch re-mints the SAME identity key the save omitted, so re-upgrading + // compacts the same row again rather than stranding a second lineage for it. + expect(rolledBack.getWorktreeMetaForHost(worktreeId(0), LOCAL)).toEqual( + before.meta[worktreeId(0)] + ) + rolledBack.flush() + const reminted = JSON.parse(readFileSync(dataFile, 'utf-8')) as PersistedState + expect( + reminted.worktreeIdentityAliases?.[composeWorktreeHostIdentity(LOCAL, worktreeId(0))] + ).toEqual([ + canonicalWorktreeIdentity({ + worktreeId: worktreeId(0), + executionHostId: LOCAL as never, + instanceId: 'instance-0' + }) + ]) + }) + + it('rebuilds the identity rows as the same objects the locator map holds', () => { + const fixture = buildFixture() + const dataFile = tempDataFile() + writeFixture(dataFile, fixture.state) + openStore(dataFile).flush() + + const store = openStore(dataFile) + const rebuilt = store.getAllWorktreeMeta() + // JSON.parse splits the one object the write path shared into two; the rebuild puts it back, + // worth ~0.46 MB of heap on the measured 3.64 MB profile. + let shared = 0 + for (let index = 0; index < TWIN_ROWS; index++) { + if (store.getWorktreeMetaForHost(worktreeId(index), LOCAL) === rebuilt[worktreeId(index)]) { + shared++ + } + } + expect(shared).toBe(TWIN_ROWS) + }) + + it('loads an old-serializer file and a new-serializer file to the same state', () => { + const fixture = buildFixture() + const legacyFile = tempDataFile() + writeFixture(legacyFile, fixture.state) + const fromLegacy = openStore(legacyFile) + // Writing it back produces the new, projected shape in place. + fromLegacy.flush() + + const compactFile = tempDataFile() + writeFileSync(compactFile, readFileSync(legacyFile)) + const fromCompact = openStore(compactFile) + + expect(fromCompact.getAllWorktreeMeta()).toEqual(fromLegacy.getAllWorktreeMeta()) + expect(fromCompact.getAllWorktreeMetaForHost(LOCAL)).toEqual( + fromLegacy.getAllWorktreeMetaForHost(LOCAL) + ) + expect(fromCompact.getAllWorktreeMetaForHost(REMOTE as never)).toEqual( + fromLegacy.getAllWorktreeMetaForHost(REMOTE as never) + ) + }) + + it('keeps every locator row when the alias map is missing or unreadable', () => { + for (const aliases of [undefined, null, [], { 'local|x': 'not-an-array' }]) { + const fixture = buildFixture() + const dataFile = tempDataFile() + writeFixture(dataFile, { + ...fixture.state, + worktreeIdentityAliases: aliases as never + }) + const store = openStore(dataFile) + // Nothing resolvable, so nothing is omitted -- and a garbled alias map costs exactly what it + // costs today, because every row's name/pin/links is still in the locator map. + expect(Object.keys(store.getAllWorktreeMeta()).length).toBe( + Object.keys(fixture.state.worktreeMeta).length + ) + expect(store.getAllWorktreeMeta()[worktreeId(0)]?.displayName).toBe('workspace-0') + store.flush() + const onDisk = JSON.parse(readFileSync(dataFile, 'utf-8')) as PersistedState + expect(Object.keys(onDisk.worktreeMeta).length).toBe( + Object.keys(fixture.state.worktreeMeta).length + ) + // The identity rows a garbled alias map strands are pruned exactly as they are today; the + // projection never adds to that, because it only omits a row an alias can rebuild. + expect(openStore(dataFile).getAllWorktreeMeta()).toEqual(store.getAllWorktreeMeta()) + } + }) + + /** + * A file that never had an identity map must not gain one: the rebuild returns the parsed value + * unchanged for a non-record, so an unconditional spread would give the loaded state an own + * `worktreeMetaByIdentity: undefined` -- a key that outranks the defaults spread and reaches + * every `Object.hasOwn`/`in` reader as present-but-empty. + */ + it('never materializes an identity map a file did not have', () => { + const dataFile = tempDataFile() + writeFileSync( + dataFile, + JSON.stringify({ + repos: [{ id: REPO_ID, name: REPO_ID, path: '/tmp/repo-1', worktreesPath: '/tmp' }], + worktreeMeta: { [worktreeId(0)]: meta(0, seededRandom(1)) }, + worktreeLineageById: { + [`${REPO_ID}::/tmp/child`]: { + parentWorktreeId: `${REPO_ID}::/tmp/parent`, + createdAt: RECENTLY + } + }, + workspaceLineageByChildKey: { + [`worktree:${REPO_ID}::/tmp/child`]: { + parentWorkspaceKey: `worktree:${REPO_ID}::/tmp/parent`, + createdAt: RECENTLY + } + } + }), + 'utf-8' + ) + + const store = openStore(dataFile) + expect(Object.keys(store.getAllWorktreeMeta())).toEqual([worktreeId(0)]) + store.flush() + + const onDisk = JSON.parse(readFileSync(dataFile, 'utf-8')) as PersistedState + expect(Object.hasOwn(onDisk, 'worktreeMetaByIdentity')).toBe(false) + // The locator map and its lineage companions are all still there, untouched by the projection. + expect(Object.keys(onDisk.worktreeMeta)).toEqual([worktreeId(0)]) + expect(Object.keys(onDisk.worktreeLineageById)).toEqual([`${REPO_ID}::/tmp/child`]) + expect(Object.keys(onDisk.workspaceLineageByChildKey)).toEqual([ + `worktree:${REPO_ID}::/tmp/child` + ]) + }) +}) diff --git a/src/main/persistence/loading-store/worktree-meta-alias-projection.ts b/src/main/persistence/loading-store/worktree-meta-alias-projection.ts new file mode 100644 index 00000000000..4789f38d445 --- /dev/null +++ b/src/main/persistence/loading-store/worktree-meta-alias-projection.ts @@ -0,0 +1,149 @@ +/** + * `setWorktreeMetaForHost` stores one object in both `worktreeMeta` and `worktreeMetaByIdentity`, + * so the profile serializes every metadata row twice. On a measured 3.64 MB install, 1,347 of + * 1,349 locator rows were byte-identical to their identity twin: 540 KB of identity rows + * re-serialized on every debounced save and re-parsed on every launch. + * + * The identity row is the copy that is dropped, never the locator row, and only when the locator + * row *regenerates its own key*: `wt2::` is a pure function of two fields the + * locator row still carries. That direction is what makes the change free of a format marker. + * "Alias present, identity row absent, locator row derives the key" is not a state any build ever + * writes deliberately -- `pruneUnreferencedWorktreeIdentityMeta` only drops rows whose alias is + * already gone, and `normalizeWorktreeLinkedItemMetadata` only drops aliases whose row is already + * gone -- and it is a state every build since #16691 already heals, to exactly the row this + * rebuild produces, via `migrateLegacyWorktreeMetadata`. So a downgrade is lossless by + * construction: the old build sees a complete `worktreeMeta`, drops the dangling aliases, and + * re-mints the identical identity key from the row's preserved `instanceId` on first read. + * + * The rebuild also reinstates the shared object reference that `JSON.parse` splits in two. + */ +import { isDeepStrictEqual } from 'node:util' +import type { PersistedState } from '../../../shared/persisted-state-types' +import type { WorktreeMeta } from '../../../shared/worktree/meta-types' +import { canonicalWorktreeIdentity } from '../../../shared/worktree/identity' +import { + getExecutionHostIdFromWorktreeHostIdentity, + getWorktreeIdFromHostIdentity +} from '../../../shared/worktree/host-qualified-identity' + +/** Every slice of a parsed profile file the projection reads; `PersistedState` satisfies it. */ +export type WorktreeMetaAliasProjectionSource = Pick< + PersistedState, + 'worktreeMeta' | 'worktreeIdentityAliases' +> + +function isPlainRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +/** + * The identity key an alias' own locator row regenerates, or undefined when it does not. + * + * The single definition of the omission rule: writer and reader both go through it, so they cannot + * disagree about which key is derivable. `hostId` and `instanceId` are not in + * `WORKTREE_META_PERSISTED_DEFAULTS`, so the answer is the same before and after default omission. + */ +function identityKeyDerivedFromLocatorRow(alias: string, locatorRow: unknown): string | undefined { + if (!isPlainRecord(locatorRow)) { + return undefined + } + const { hostId, instanceId } = locatorRow as WorktreeMeta + if (typeof hostId !== 'string' || !hostId || typeof instanceId !== 'string' || !instanceId) { + return undefined + } + // The alias must name the same host, or the key the reader derives is not the key it replaces. + if (getExecutionHostIdFromWorktreeHostIdentity(alias) !== hostId) { + return undefined + } + return canonicalWorktreeIdentity({ + worktreeId: getWorktreeIdFromHostIdentity(alias), + executionHostId: hostId, + instanceId + }) +} + +/** + * Identity key -> the locator row that regenerates it, for every alias that does so unambiguously. + * + * A key two aliases both derive is left out entirely: which locator row would rebuild it would + * then depend on object key order, which is not a durable contract across a JSON round trip. An + * alias carrying more than one key is left out too -- `setWorktreeMetaForHost` refuses to write + * one, so it is a repair state, and only one of its rows could ever be derivable anyway. + */ +function derivableIdentityRows( + state: WorktreeMetaAliasProjectionSource +): Map { + const derivable = new Map() + const aliases = state.worktreeIdentityAliases + const worktreeMeta = state.worktreeMeta as unknown + if (!isPlainRecord(aliases) || !isPlainRecord(worktreeMeta)) { + return derivable + } + const contested = new Set() + for (const [alias, identityKeys] of Object.entries(aliases)) { + if (!Array.isArray(identityKeys) || identityKeys.length !== 1) { + continue + } + const locatorRow = worktreeMeta[getWorktreeIdFromHostIdentity(alias)] + const derivedKey = identityKeyDerivedFromLocatorRow(alias, locatorRow) + if (derivedKey === undefined || derivedKey !== identityKeys[0]) { + continue + } + if (derivable.has(derivedKey)) { + contested.add(derivedKey) + continue + } + derivable.set(derivedKey, locatorRow as WorktreeMeta) + } + for (const key of contested) { + derivable.delete(key) + } + return derivable +} + +/** + * Serialize side, on the raw in-memory maps: an untouched row is the same object in both, so the + * common case settles on a reference check and never a deep compare. + */ +export function projectWorktreeMetaByIdentityOntoLocators( + worktreeMetaByIdentity: Record, + state: WorktreeMetaAliasProjectionSource +): Record { + let projected: Record | undefined + for (const [identityKey, locatorRow] of derivableIdentityRows(state)) { + const identityRow = worktreeMetaByIdentity[identityKey] + if (!isPlainRecord(identityRow)) { + continue + } + if (identityRow !== locatorRow && !isDeepStrictEqual(identityRow, locatorRow)) { + continue + } + projected ??= { ...worktreeMetaByIdentity } + delete projected[identityKey] + } + return projected ?? worktreeMetaByIdentity +} + +/** + * Load side, in place. Runs before the metadata normalizers, because + * `normalizeWorktreeLinkedItemMetadata` drops an alias whose identity row is not there yet. + * + * Only ever adds a key the locator row already fully describes, so an untouched legacy file is a + * no-op on it (nothing is missing) and a garbled one loses no more than it does today. + */ +export function hydrateWorktreeMetaAliasProjection( + parsed: WorktreeMetaAliasProjectionSource & Pick +): Record | undefined { + const worktreeMetaByIdentity = parsed.worktreeMetaByIdentity + if (!isPlainRecord(worktreeMetaByIdentity)) { + return worktreeMetaByIdentity + } + for (const [identityKey, locatorRow] of derivableIdentityRows(parsed)) { + if (Object.hasOwn(worktreeMetaByIdentity, identityKey)) { + continue + } + // Same reference in both maps, as every in-session write leaves it. + worktreeMetaByIdentity[identityKey] = locatorRow + } + return worktreeMetaByIdentity +} diff --git a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts index cdad14034e2..0588a2f6dae 100644 --- a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts +++ b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts @@ -287,6 +287,64 @@ describe('pruneSessionlessMissingLocalWorktreeMetadataForRepo', () => { } }) + // A route-retired lease is a tombstone, not a claim: counting one pinned its worktree's metadata + // row for good, so the prune could never make progress on it (#17775). + it('does not let route-retired SSH leases pin a metadata row', () => { + const state = makeState() + const liveIds = Array.from({ length: 3 }, (_, i) => `${REPO_ID}::/workspace/live-${i}`) + const terminatedIds = Array.from({ length: 5 }, (_, i) => `${REPO_ID}::/workspace/closed-${i}`) + const supersededIds = Array.from({ length: 4 }, (_, i) => `${REPO_ID}::/workspace/lost-${i}`) + const recycledIds = [`${REPO_ID}::/workspace/recycled`] + const allIds = [...liveIds, ...terminatedIds, ...supersededIds, ...recycledIds] + for (const worktreeId of allIds) { + state.worktreeMeta[worktreeId] = makeMeta(worktreeId) + } + const lease = (worktreeId: string, index: number, extra: object) => ({ + targetId: 'builder', + ptyId: `pty-${index}`, + worktreeId, + createdAt: 1, + updatedAt: 1, + ...extra + }) + state.sshRemotePtyLeases = [ + ...liveIds.map((id, i) => lease(id, i, { state: 'detached' })), + ...terminatedIds.map((id, i) => lease(id, 100 + i, { state: 'terminated' })), + ...supersededIds.map((id, i) => + lease(id, 200 + i, { state: 'expired', supersededBy: 'pty-9' }) + ), + ...recycledIds.map((id, i) => lease(id, 300 + i, { state: 'expired', relayIdRecycled: true })) + ] as never + + const scan = capture(state) + + expect(pruneCaptured(state, scan, allIds).sort()).toEqual( + [...terminatedIds, ...supersededIds, ...recycledIds].sort() + ) + expect(Object.keys(state.worktreeMeta).sort()).toEqual([...liveIds].sort()) + }) + + // A plain `expired` lease says only that the CLIENT lost its route, so its pane is still + // recoverable and its metadata row is still owned (docs/reference/ssh-execution-boundary.md). + it('keeps a metadata row pinned by an unmarked expired lease', () => { + const state = makeState() + const worktreeId = `${REPO_ID}::/workspace/orphaned` + state.worktreeMeta[worktreeId] = makeMeta(worktreeId) + const scan = capture(state) + state.sshRemotePtyLeases = [ + { + targetId: 'builder', + ptyId: 'pty', + worktreeId, + state: 'expired', + createdAt: 1, + updatedAt: 1 + } + ] + + expect(pruneCaptured(state, scan, [worktreeId])).toEqual([]) + }) + it('preserves canonically equivalent session and top-level owners', () => { const candidateId = `${REPO_ID}::/workspace/Café`.normalize('NFC') const ownerId = candidateId.normalize('NFD') diff --git a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts index 0e03a560a17..db24c25bcf4 100644 --- a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts +++ b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts @@ -2,6 +2,7 @@ import { isWindowsAbsolutePathLike } from '../../../shared/cross-platform-path' import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../../../shared/execution-host' import type { PersistedState } from '../../../shared/persisted-state-types' import { getRepoKind } from '../../../shared/repo-kind' +import { sshRemotePtyLeaseAllowsReattach } from '../../../shared/ssh-types' import { worktreeWorkspaceKey } from '../../../shared/workspace-scope' import { FOLDER_WORKSPACE_INSTANCE_SEPARATOR, splitWorktreeId } from '../../../shared/worktree/id' import { isWslUncPath } from '../../../shared/wsl-paths' @@ -40,6 +41,13 @@ function collectPersistedWorkspaceOwners( } } for (const lease of state.sshRemotePtyLeases) { + // A lease that can never be reattached is a routing tombstone, not a claim on a workspace: + // `terminated` is the operator close, and an `expired` row marked `supersededBy` / + // `relayIdRecycled` already lost its pane to a newer lease. Counting them as owners pinned + // their worktree's metadata row permanently, so the prune could never make progress (#17775). + if (!sshRemotePtyLeaseAllowsReattach(lease)) { + continue + } add(lease.worktreeId) } for (const entry of state.migrationUnsupportedPtyEntries) { diff --git a/src/main/project-runtime-git-options.ts b/src/main/project-runtime-git-options.ts index 808d31d5fcf..20aa0e9659a 100644 --- a/src/main/project-runtime-git-options.ts +++ b/src/main/project-runtime-git-options.ts @@ -102,7 +102,12 @@ export function getWorktreeMirrorDistro( store: ProjectRuntimeResolutionStore, repo: Repo ): string | undefined { - const projectRuntime = resolveLocalProjectRuntimeForRepo(store, repo) + return getWorktreeMirrorDistroForRuntime(resolveLocalProjectRuntimeForRepo(store, repo)) +} + +export function getWorktreeMirrorDistroForRuntime( + projectRuntime: ProjectExecutionRuntimeResolution | undefined +): string | undefined { if (!projectRuntime || projectRuntime.status !== 'resolved') { return undefined } diff --git a/src/main/providers/pty-process-inspection.ts b/src/main/providers/pty-process-inspection.ts index 6869899de8b..59d910b2238 100644 --- a/src/main/providers/pty-process-inspection.ts +++ b/src/main/providers/pty-process-inspection.ts @@ -11,14 +11,24 @@ export type PtyProcessInspection = TerminalProcessInspection type CompletionSensitivePtyProvider = IPtyProvider & { inspectProcess?: ( id: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: PtyProcessInspectionOptions ) => Promise } +/** + * `scanChildProcesses` marks a read whose answer decides something once, rather than a poll that + * self-corrects on its next tick. Only hosts where the child answer costs a process-table read + * act on it; everywhere else the answer was already captured. + */ +export type PtyProcessInspectionOptions = { + expectedIncarnationId?: PtyIncarnationId + scanChildProcesses?: boolean +} + export async function inspectPtyProviderProcess( provider: IPtyProvider, ptyId: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: PtyProcessInspectionOptions ): Promise { if (provider.hasPty?.(ptyId) === false) { throw new Error('terminal_gone') @@ -37,7 +47,7 @@ export async function inspectPtyProviderProcess( export async function inspectPtyProviderProcessForRenderer( provider: IPtyProvider, ptyId: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: PtyProcessInspectionOptions ): Promise { try { return await inspectPtyProviderProcess(provider, ptyId, options) diff --git a/src/main/providers/ssh-git-read-provider.ts b/src/main/providers/ssh-git-read-provider.ts index 5cbcbc17b5a..cbf20271003 100644 --- a/src/main/providers/ssh-git-read-provider.ts +++ b/src/main/providers/ssh-git-read-provider.ts @@ -44,7 +44,8 @@ export class SshGitReadProvider { } } - private invalidateGitReads(): void { + /** Overridden by subclasses that own additional read caches (worktree listings). */ + protected invalidateGitReads(): void { this.gitDiffReadDedupe.clear() this.statusReadLeaseOwner.invalidate() this.upstreamStatusReadOwner.invalidate() diff --git a/src/main/providers/ssh-git-worktree-list-dedupe.test.ts b/src/main/providers/ssh-git-worktree-list-dedupe.test.ts new file mode 100644 index 00000000000..4030dbe87e4 --- /dev/null +++ b/src/main/providers/ssh-git-worktree-list-dedupe.test.ts @@ -0,0 +1,156 @@ +/** + * Local repos coalesce concurrent `git worktree list` scans (`shareWorktreeScan`); the SSH path + * branched away from that and paid one relay round trip per independent caller (`worktrees:list`, + * `worktrees:listAll`, the space repo scan, provisioned-root adoption). These are call counters. + */ +import { describe, expect, it } from 'vitest' +import { SshGitProvider } from './ssh-git-provider' +import { createMockMux, type MockMultiplexer } from './ssh-git-provider-test-harness' + +const REPO_PATH = '/home/user/repo' + +const WORKTREES = [ + { path: REPO_PATH, head: 'abc123', branch: 'main', isBare: false, isMainWorktree: true } +] + +type Deferred = { resolve: (value: unknown) => void; reject: (error: unknown) => void } + +/** Holds `git.listWorktrees` open so overlap is deterministic; answers everything else at once. */ +function createPendingListMux(): { mux: MockMultiplexer; listDeferreds: Deferred[] } { + const mux = createMockMux() + const listDeferreds: Deferred[] = [] + mux.request.mockImplementation((method: string) => { + if (method !== 'git.listWorktrees') { + return Promise.resolve(undefined) + } + return new Promise((resolve, reject) => { + listDeferreds.push({ resolve, reject }) + }) + }) + return { mux, listDeferreds } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +function countListRequests(mux: MockMultiplexer): number { + return mux.request.mock.calls.filter((call) => call[0] === 'git.listWorktrees').length +} + +describe('SSH git.listWorktrees in-flight dedupe', () => { + it('collapses concurrent listings of one repo into a single relay request', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = Array.from({ length: 6 }, () => provider.listWorktrees(REPO_PATH)) + await flush() + + expect(countListRequests(mux)).toBe(1) + expect(mux.request).toHaveBeenCalledWith( + 'git.listWorktrees', + { repoPath: REPO_PATH }, + { signal: undefined } + ) + + listDeferreds[0].resolve(WORKTREES) + expect(await Promise.all(listings)).toEqual(Array.from({ length: 6 }, () => WORKTREES)) + }) + + it('does not share across repos or connections', async () => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + void provider.listWorktrees('/home/user/other') + await flush() + expect(countListRequests(mux)).toBe(2) + + const second = createPendingListMux() + void new SshGitProvider('conn-2', second.mux as never).listWorktrees(REPO_PATH) + await flush() + + expect(countListRequests(second.mux)).toBe(1) + expect(countListRequests(mux)).toBe(2) + }) + + it('keeps a signalled listing on its own request', async () => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + await flush() + const controller = new AbortController() + void provider.listWorktrees(REPO_PATH, { signal: controller.signal }) + await flush() + + expect(countListRequests(mux)).toBe(2) + expect(mux.request).toHaveBeenCalledWith( + 'git.listWorktrees', + { repoPath: REPO_PATH }, + { signal: controller.signal } + ) + }) + + it('re-requests after the shared listing settles instead of caching it', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const first = provider.listWorktrees(REPO_PATH) + await flush() + listDeferreds[0].resolve(WORKTREES) + await first + + void provider.listWorktrees(REPO_PATH) + await flush() + + expect(countListRequests(mux)).toBe(2) + }) + + it.each([ + ['addWorktree', (p: SshGitProvider) => p.addWorktree(REPO_PATH, 'feature', '/home/user/feat')], + ['removeWorktree', (p: SshGitProvider) => p.removeWorktree('/home/user/feat')] + ])('invalidates the shared listing after %s', async (_name, mutate) => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(1) + + await mutate(provider) + + // The catalog moved, so a joiner must not inherit the pre-mutation scan. + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(2) + }) + + it('shares a failed listing with its joiners and re-requests afterwards', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)] + await flush() + expect(countListRequests(mux)).toBe(1) + + const failure = new Error('relay request failed') + listDeferreds[0].reject(failure) + await expect(listings[0]).rejects.toBe(failure) + await expect(listings[1]).rejects.toBe(failure) + + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(2) + }) + + it('refuses an unauthoritative relay answer for every joiner (#14004)', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)] + await flush() + listDeferreds[0].resolve([]) + + await expect(listings[0]).rejects.toThrow() + await expect(listings[1]).rejects.toThrow() + }) +}) diff --git a/src/main/providers/ssh-git-worktree-provider.ts b/src/main/providers/ssh-git-worktree-provider.ts index 8dae1413321..17e5b273de7 100644 --- a/src/main/providers/ssh-git-worktree-provider.ts +++ b/src/main/providers/ssh-git-worktree-provider.ts @@ -2,6 +2,7 @@ import type { GitStatusResult } from '../../shared/git-status-types' import type { RemoveWorktreeResult } from '../../shared/worktree/create-types' import type { GitWorktreeInfo } from '../../shared/worktree/types' import { CapabilityProbeCache } from '../../shared/capability-probe-cache' +import { InFlightPromiseDedupe, stableInFlightKey } from '../../shared/in-flight-promise-dedupe' import { assertAuthoritativeWorktreeCatalog } from '../../shared/worktree/worktree-catalog-availability' import { isJsonRpcMethodNotFoundError } from './ssh-git-relay-errors' import { SshGitReviewHeadProvider } from './ssh-git-review-head-provider' @@ -29,16 +30,35 @@ export class SshGitWorktreeProvider extends SshGitReviewHeadProvider { private readonly worktreeIsCleanCapabilityCache = new CapabilityProbeCache< typeof WORKTREE_IS_CLEAN_CAPABILITY >(Number.POSITIVE_INFINITY) + // Scoped to this provider instance, so two SSH hosts never share an entry. + private readonly worktreeListDedupe = new InFlightPromiseDedupe() + protected override invalidateGitReads(): void { + super.invalidateGitReads() + this.worktreeListDedupe.clear() + } + + /** Un-signalled reads of one repo coalesce onto the request already in flight; nothing is cached. */ async listWorktrees( repoPath: string, options?: { signal?: AbortSignal } ): Promise { - const response = await this.mux.request( - 'git.listWorktrees', - { repoPath }, - { signal: options?.signal } + // Why: same rule as shareWorktreeScan — one caller's abort must not cancel the scan its + // joiners are still waiting on, so a signalled read keeps its own request. + if (options?.signal) { + return this.requestWorktreeList(repoPath, options.signal) + } + return this.worktreeListDedupe.run(stableInFlightKey(['listWorktrees', repoPath]), () => + this.requestWorktreeList(repoPath) ) + } + + /** The one real relay round trip a coalesced read's joiners all wait on. */ + private async requestWorktreeList( + repoPath: string, + signal?: AbortSignal + ): Promise { + const response = await this.mux.request('git.listWorktrees', { repoPath }, { signal }) // Why (#14004): relays before this fix answered a failed worktree scan with `[]`. Mixed versions are // normal, so refuse the shape here too — a Git repo always lists its own checkout. return assertAuthoritativeWorktreeCatalog(response, repoPath) diff --git a/src/main/providers/ssh-pty-inspect-observation-identity.test.ts b/src/main/providers/ssh-pty-inspect-observation-identity.test.ts new file mode 100644 index 00000000000..4e0dae1ead1 --- /dev/null +++ b/src/main/providers/ssh-pty-inspect-observation-identity.test.ts @@ -0,0 +1,68 @@ +/** + * Ratchet (#18419): `pty.inspectProcess` must NOT be in-flight coalesced the way the sibling git + * reads in `SshGitReadProvider` are. The host mints one `observationEpoch` per request and the + * pane foreground reader commits that epoch per read, so a shared reply reads as a stale replay to + * the second reader to settle — see the companion renderer proof in + * `src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts`. + * These are request counters, not timings. + */ +import { describe, expect, it, vi } from 'vitest' +import { createSshPtyProviderRpcOperations } from './ssh-pty-provider-rpc-operations' + +const RELAY_PTY_ID = 'pty-1' +const APP_PTY_ID = `ssh:conn-1@@${RELAY_PTY_ID}` +const INCARNATION_ID = 'inc-1' + +/** Answers `pty.inspectProcess` with a fresh host observation per request, held open on demand. */ +function createInspectingOperations(): { + operations: ReturnType + request: ReturnType + resolvers: ((value: unknown) => void)[] +} { + const resolvers: ((value: unknown) => void)[] = [] + const request = vi.fn(() => new Promise((resolve) => resolvers.push(resolve))) + return { + operations: createSshPtyProviderRpcOperations({ + mux: { request } as never, + toRelayPtyId: () => RELAY_PTY_ID + }), + request, + resolvers + } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +describe('SSH pty.inspectProcess observation identity', () => { + it('gives each overlapping probe of one pane+incarnation its own host observation', async () => { + const { operations, request, resolvers } = createInspectingOperations() + + const first = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + const second = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + await flush() + + expect(request).toHaveBeenCalledTimes(2) + resolvers[0]({ foregroundProcess: 'claude', observationEpoch: 1 }) + resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 2 }) + // Each read settles on the observation minted for it, never a neighbour's. + expect(await first).toMatchObject({ observationEpoch: 1 }) + expect(await second).toMatchObject({ observationEpoch: 2 }) + }) + + it('does not share a failed probe with an overlapping one', async () => { + const { operations, request, resolvers } = createInspectingOperations() + + const failing = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + const overlapping = operations.inspectProcess(APP_PTY_ID, { + expectedIncarnationId: INCARNATION_ID + }) + await flush() + + expect(request).toHaveBeenCalledTimes(2) + resolvers[0](Promise.reject(new Error('relay dropped the probe'))) + resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 1 }) + + await expect(failing).rejects.toThrow('relay dropped the probe') + expect(await overlapping).toMatchObject({ observationEpoch: 1 }) + }) +}) diff --git a/src/main/providers/ssh-pty-provider-rpc-operations.ts b/src/main/providers/ssh-pty-provider-rpc-operations.ts index 3f9953b9d1c..2e273239cd4 100644 --- a/src/main/providers/ssh-pty-provider-rpc-operations.ts +++ b/src/main/providers/ssh-pty-provider-rpc-operations.ts @@ -50,15 +50,21 @@ export function createSshPtyProviderRpcOperations({ mux, toRelayPtyId }: SshPtyP const result = await mux.request('pty.getForegroundProcess', { id: toRelayPtyId(id) }) return result as string | null }, + // Do NOT in-flight coalesce this the way the sibling git reads are: the host mints one + // `observationEpoch` per request and the pane foreground reader commits it per read, so a + // shared reply reads as a stale replay and degrades a `live` identity read to `unverifiable`. + // Guarded by ssh-pty-inspect-observation-identity.test.ts; #17525 removes the poll. inspectProcess: async ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise => { return (await mux.request('pty.inspectProcess', { id: toRelayPtyId(id), ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } - : {}) + : {}), + // Additive request member: an older relay ignores it and answers as it always did. + ...(options?.scanChildProcesses === true ? { scanChildProcesses: true } : {}) })) as PtyProcessInspection }, serialize: async (ids: string[]): Promise => { diff --git a/src/main/providers/ssh-pty-provider.ts b/src/main/providers/ssh-pty-provider.ts index 70c01b5a1c7..3d0c46d7a03 100644 --- a/src/main/providers/ssh-pty-provider.ts +++ b/src/main/providers/ssh-pty-provider.ts @@ -63,7 +63,7 @@ export class SshPtyProvider implements IPtyProvider { this.rpcOperations.getForegroundProcess(id) inspectProcess = ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise => this.rpcOperations.inspectProcess(id, options) serialize = (ids: string[]): Promise => this.rpcOperations.serialize(ids) revive = (state: string): Promise => this.rpcOperations.revive(state) diff --git a/src/main/proxy-guarded-fetch-call-site-audit.test.ts b/src/main/proxy-guarded-fetch-call-site-audit.test.ts new file mode 100644 index 00000000000..cf9e1f60609 --- /dev/null +++ b/src/main/proxy-guarded-fetch-call-site-audit.test.ts @@ -0,0 +1,140 @@ +import { readdirSync, readFileSync } from 'node:fs' +import { join, relative, sep } from 'node:path' +import { describe, expect, it } from 'vitest' + +// Startup applies the persisted proxy to `session.defaultSession` only, and +// `installElectronProxyRequestGuard(session.defaultSession)` is what actually holds requests +// until that apply (and every later proxy transition) settles. Two ways a main-process fetcher +// can escape that fence, both audited here: +// 1. a `net.fetch` / `net.request` that names another `session` or `partition` +// 2. a `.fetch(` on a `session.fromPartition(...)` session +// Known pre-existing gap outside this repo's reach: electron-updater runs on its own partition. +// +// Rule 2 entries map a file to its expected number of non-`net` `.fetch(` calls. A count change +// means a call site was added, removed, or moved: re-audit the file and update the count. +const AUDITED_NON_NET_FETCH_CALLS = new Map([ + // Isolated cookie-jar session, proxied by createOpenCodeRequestSession before any request. + ['main/rate-limits/opencode-go-usage-fetcher.ts', 2], + // Isolated cookie-jar session that does NOT apply the proxy — a pre-existing gap, not a + // regression: no proxy has ever reached this partition. Keep it listed so it stays visible. + ['main/rate-limits/minimax-request-context.ts', 2], + // Injected HttpClient, not a session: resolves to net.fetch on defaultSession + // (main/host/electron-http-client.ts) or to the global-fetch-audited Node fallback. + ['main/jira/authenticated-request.ts', 1] +]) + +// `globalThis.fetch` / `global.fetch` belong to global-fetch-call-site-audit.test.ts. +// `\s*` before `(`: the formatter never emits `net.fetch (url)`, but an unformatted call must not +// be a hole in a guard whose whole job is to fail on the call nobody reviewed. +const FETCH_CALL = /\.fetch\s*\(/g +const RECEIVER_IDENTIFIER = /(?:^|[^.\w$])([A-Za-z_$][\w$]*)\s*$/ +const DEFAULT_SESSION_RECEIVERS = new Set(['net', 'globalThis', 'global']) +const NET_REQUEST_CALL = /(? { + const sources = auditedSourceFiles(__dirname) + + it('keeps every net.fetch/net.request on the guarded default session', () => { + const offenders: string[] = [] + for (const { file, content } of sources) { + for (const match of content.matchAll(NET_REQUEST_CALL)) { + const args = callArgumentText(content, match.index + match[0].length) + if (SESSION_SCOPED_OPTION.test(args)) { + offenders.push(`${file}:${content.slice(0, match.index).split('\n').length}`) + } + } + } + expect( + offenders.sort(), + 'This request names its own session/partition, so it is not covered by ' + + 'installElectronProxyRequestGuard(session.defaultSession) and startup never applies the ' + + 'persisted proxy to it. Either drop the option, or apply the proxy to that session ' + + 'yourself (see main/rate-limits/opencode-go-request-session.ts) and allowlist it here.' + ).toEqual([]) + }) + + it('keeps every non-default-session fetcher audited with its expected count', () => { + const found = new Map() + for (const { file, content } of sources) { + const hits = [...content.matchAll(FETCH_CALL)].filter((match) => { + const receiver = RECEIVER_IDENTIFIER.exec(content.slice(0, match.index))?.[1] + // A chained (`session.fromPartition(...).fetch(`) or member (`ctx.session.fetch(`) + // receiver has no bare trailing identifier, and is never the default session. + return receiver === undefined || !DEFAULT_SESSION_RECEIVERS.has(receiver) + }).length + if (hits > 0) { + found.set(file, hits) + } + } + + const drifted = [...found] + .filter(([file, count]) => AUDITED_NON_NET_FETCH_CALLS.get(file) !== count) + .map(([file, count]) => `${file}: found ${count} call(s)`) + .sort() + expect( + drifted, + 'A session.fromPartition(...) session is not covered by ' + + 'installElectronProxyRequestGuard(session.defaultSession), so nothing holds its requests ' + + 'until the proxy lands and startup never applies the proxy to it. Apply the proxy to that ' + + 'session yourself (see main/rate-limits/opencode-go-request-session.ts), then update ' + + 'AUDITED_NON_NET_FETCH_CALLS.' + ).toEqual([]) + + const stale = [...AUDITED_NON_NET_FETCH_CALLS.keys()].filter((file) => !found.has(file)).sort() + expect(stale, 'Remove audited entries whose .fetch( calls are gone.').toEqual([]) + }) +}) diff --git a/src/main/pty/node-pty-self-exit-pseudoconsole-close.test.ts b/src/main/pty/node-pty-self-exit-pseudoconsole-close.test.ts new file mode 100644 index 00000000000..1f9cae275c3 --- /dev/null +++ b/src/main/pty/node-pty-self-exit-pseudoconsole-close.test.ts @@ -0,0 +1,180 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * A shell that exits by itself must still close its pseudoconsole. + * + * `ClosePseudoConsole` is the only thing that reaps a ConPTY's console host — + * Orca's own job-ownership patch says so, because `CreatePseudoConsole` spawns + * that host before the per-pty job exists and it is therefore not a job member. + * Upstream node-pty calls it from exactly one place, `PtyKill`, which begins by + * looking the baton up by id — and the exit watcher in `SetupExitCallback` + * erased the baton the moment the shell died. So on the self-exit path (typing + * `exit`, which is how panes usually close) that lookup missed, `PtyKill` did + * nothing at all, and the pseudoconsole was never closed. + * + * There is a SECOND, independent defect on the same path: the `useConptyDll` + * branch of `WindowsPtyAgent.kill()` disposed the conout worker only from an + * `_outSocket.on('data')` handler, and no more data arrives once the shell has + * gone — so that worker leaked too. The non-DLL branch beside it already + * disposed unconditionally. The desktop always sets `useConptyDll`, so it hit + * both; the relay sets neither and hit only the first. + * + * Measured on Windows 11 / awin, 20 cycles, handles bucketed by NT object type, + * totals before -> after: + * + * self-exit, relay spawn 225 -> 285 becomes 219 -> 219 FLAT + * self-exit, desktop spawn 239 -> 439 becomes 222 -> 222 FLAT + * explicit kill, relay spawn 225 -> 285 becomes 219 -> 219 FLAT + * explicit kill, desktop spawn 235 -> 395 becomes 219 -> 219 FLAT + * + * Neither fix alone is enough on the desktop: the pseudoconsole close is worth + * +1 Process +1 File per terminal, the dispose +2 Thread +4 File. + * + * WHY THIS IS A PATCH-CONTENT PIN AND NOT A BEHAVIOURAL TEST: the defect is + * only observable as a per-NT-type handle count, which needs + * `NtQuerySystemInformation(SystemExtendedHandleInformation)`. Nothing in the + * repo can read that, and the cheaper Windows-observable proxies do not + * discriminate — the console host process is reaped either way (the leak is a + * handle to an already-exited object, not an orphaned process), and the + * `\\.\pipe\conpty-*` entries disappear either way. Both were measured and + * rejected as assertions rather than assumed. So this pins the mechanism + * instead, which is the real risk: a future resync of the vendored patch + * silently dropping the hunk. + */ + +const PATCH = readFileSync(join(__dirname, '../../../config/patches/node-pty@1.1.0.patch'), 'utf8') + +/** + * Just the `PtyKill` hunk. Several markers below also occur in the `PtyConnect` + * hunk above it, and a bare `indexOf` on the whole patch silently matched the + * wrong one — an assertion that then held regardless of what `PtyKill` did. + */ +const ptyKillHunk = (() => { + // Anchored on the hunk header's function context rather than its line + // numbers, which shift whenever anything above it in the patch changes. + const header = /^@@ .* @@ static Napi::Value PtyKill\(.*$/m.exec(PATCH) + if (!header) { + throw new Error('no PtyKill hunk in config/patches/node-pty@1.1.0.patch') + } + const from = header.index + const next = PATCH.indexOf('\n@@ ', from + 1) + return PATCH.slice(from, next === -1 ? undefined : next) +})() + +/** + * `indexOf` that throws instead of returning -1. A missing marker must fail the + * assertion that depends on it, not quietly make a slice or comparison vacuous. + */ +function indexIn(haystack: string, marker: string): number { + const at = haystack.indexOf(marker) + if (at === -1) { + throw new Error(`marker not found in the PtyKill hunk: ${marker}`) + } + return at +} + +describe('node-pty patch: pseudoconsole close on the self-exit path', () => { + it('does not let the exit watcher free the baton while the close is still owed', () => { + // Pinned as one block: the erase must stay INSIDE the consoleClosed guard. + // Upstream ran it unconditionally, which is the line that caused the leak, + // and a resync that re-flattens this is the failure mode worth catching. + expect(PATCH).toContain( + [ + '+ baton->shellExited = true;', + '+ if (baton->consoleClosed) {', + '+ const bool removed = remove_pty_baton(baton->id);', + '+ assert(removed);', + '+ (void)removed;', + '+ }' + ].join('\n') + ) + }) + + it('closes the pseudoconsole from PtyKill even after the shell has exited', () => { + // hpc is copied out under the lock, so the close survives the baton's removal. + expect(PATCH).toContain('+ hpc = handle->hpc;') + expect(PATCH).toContain('+ pfnClosePseudoConsole(hpc);') + }) + + it('resolves the ConPTY DLL before it claims the close', () => { + // LoadConptyDll throws when conpty.dll is missing. Throwing after + // consoleClosed was set would strand the pseudoconsole for good: the retry + // finds the work claimed and does nothing. + // + // Anchored inside PtyKill, not by a bare indexOf: the identical line also + // appears in the PtyConnect hunk, earlier in the file, and matching that one + // made this assertion pass no matter where PtyKill resolved the DLL. + const dllResolve = indexIn( + ptyKillHunk, + '+ HANDLE hLibrary = LoadConptyDll(info, useConptyDll);' + ) + const claim = indexIn(ptyKillHunk, '+ handle->consoleClosed = true;') + expect(dllResolve).toBeLessThan(claim) + }) + + it('reaches hShell only under the null check the watcher can trip', () => { + // Pinned as one block. The watcher nulls hShell on exit, and upstream + // dereferenced it unconditionally; every remaining use — the duplication and + // the failure fallback below it — must stay inside this guard. + const start = indexIn(ptyKillHunk, '+ if (useConptyDll && handle->hShell != nullptr) {') + const end = indexIn(ptyKillHunk, '+ if (handle->shellExited) {') + const guarded = ptyKillHunk.slice(start, end) + expect(guarded).toContain('DuplicateHandle(GetCurrentProcess(), handle->hShell') + expect(guarded).toContain('TerminateProcess(handle->hShell, 1);') + // No ADDED line outside that guard may terminate through hShell. Removed + // (`-`) lines still carry upstream's unguarded call, which is the point. + const strayAdds = PATCH.replace(guarded, '') + .split('\n') + .filter((line) => line.startsWith('+') && line.includes('TerminateProcess(handle->hShell')) + expect(strayAdds).toEqual([]) + }) + + it('frees the baton from PtyKill when the shell has already exited', () => { + // The other half of the two-sided handshake. Without it a self-exit followed + // by kill() — the ordinary pane close — leaks one baton and one entry in the + // vector get_pty_baton scans linearly, forever. + expect(ptyKillHunk).toContain( + [ + '+ if (handle->shellExited) {', + '+ const bool removed = remove_pty_baton(id);', + '+ assert(removed);', + '+ (void)removed;' + ].join('\n') + ) + }) + + it('still kills the shell when DuplicateHandle fails', () => { + // A null hShellDup is indistinguishable from the self-exit case, so a + // swallowed failure would leave the shell running after its pane closed — + // a worse outcome than the leak this patch exists to fix. + expect(PATCH).toContain( + [ + '+ hShellDup = nullptr;', + '+ TerminateProcess(handle->hShell, 1);', + '+ }' + ].join('\n') + ) + }) + + it('keeps the close idempotent so a second kill cannot double-close', () => { + expect(PATCH).toContain('+ if (handle != nullptr && !handle->consoleClosed) {') + expect(PATCH).toContain('+ handle->consoleClosed = true;') + }) +}) + +describe('node-pty patch: conout worker disposal on the self-exit path', () => { + // The desktop's larger half: 8 of its 10 leaked handles per terminal. + it('disposes the conout worker unconditionally in the useConptyDll branch', () => { + expect(PATCH).toContain('+ this._conoutSocketWorker.dispose();') + // The data handler is what never fired once the shell had gone. + expect(PATCH).toContain("- this._outSocket.on('data', function () {") + expect(PATCH).toContain('- _this._conoutSocketWorker.dispose();') + }) + + it('applies the same change to the TypeScript source the patch also carries', () => { + expect(PATCH).toContain('+ this._conoutSocketWorker.dispose();') + expect(PATCH).toContain("- this._outSocket.on('data', () => {") + }) +}) diff --git a/src/main/runtime/expired-ssh-lease-pane-candidacy.test.ts b/src/main/runtime/expired-ssh-lease-pane-candidacy.test.ts index 82986a4b3fd..3d8157c9a28 100644 --- a/src/main/runtime/expired-ssh-lease-pane-candidacy.test.ts +++ b/src/main/runtime/expired-ssh-lease-pane-candidacy.test.ts @@ -12,6 +12,12 @@ const TARGET = 'ssh-target' const TAB_ID = 'tab-candidacy' type LeaseReader = { + workspaceSessionWorktreeHasRuntimeOwnedPtyCandidate: ( + session: { terminalLayoutsByTabId?: Record }, + worktreeId: string, + tabs: { id: string; ptyId: string | null }[] + ) => boolean + collectRecentExpiredSshLeaseTabIds: (worktreeId: string) => ReadonlySet getRecentExpiredSshLease: ( worktreeId: string, tabId: string, @@ -87,4 +93,45 @@ describe('recent expired SSH lease candidacy', () => { ) expect(reader.hasRecentExpiredSshLeasePane(TEST_WORKTREE_ID, pane)).toBe(true) }) + + it('collects the same tabs the per-tab reader reports, in one sweep of the leases', () => { + const leases = [leaseFor('pty-1', { supersededBy: 'pty-2' }), leaseFor('pty-2')] + let sweeps = 0 + const reader = new OrcaRuntimeService({ + ...store, + getSshRemotePtyLeases: () => { + sweeps += 1 + return leases + } + }) as unknown as LeaseReader + const tabs = Array.from({ length: 8 }, (_, index) => ({ + id: index === 7 ? TAB_ID : `tab-${index}`, + ptyId: null + })) + + expect( + reader.workspaceSessionWorktreeHasRuntimeOwnedPtyCandidate( + { terminalLayoutsByTabId: {} }, + TEST_WORKTREE_ID, + tabs + ) + ).toBe(true) + // One sweep answers all eight tabs; the per-tab reader used to sweep once per tab. + expect(sweeps).toBe(1) + expect([...reader.collectRecentExpiredSshLeaseTabIds(TEST_WORKTREE_ID)]).toEqual([TAB_ID]) + }) + + it('reports no candidate when no lease names any of the worktree tabs', () => { + const reader = readerWithLeases([ + leaseFor('pty-1', { tabId: 'somewhere-else', leafId: undefined }) + ]) + + expect( + reader.workspaceSessionWorktreeHasRuntimeOwnedPtyCandidate( + { terminalLayoutsByTabId: {} }, + TEST_WORKTREE_ID, + [{ id: TAB_ID, ptyId: null }] + ) + ).toBe(false) + }) }) diff --git a/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts b/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts index c4687c96d14..a9f377e5a7b 100644 --- a/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts +++ b/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts @@ -82,17 +82,24 @@ export class OrcaRuntimeWithReconcileHeadlessMobileSessionBrowserTabs extends Or worktreeId: string, tabs: WorkspaceSessionState['tabsByWorktree'][string] ): boolean { + // Why resolved lazily and reused: the per-tab question is the same lease sweep with a + // different tabId, so asking it once per worktree answers every tab. Kept lazy so a + // worktree whose first tab already owns a serve/SSH pty never sweeps at all. + let recoverableTabIds: ReadonlySet | undefined return tabs.some((tab) => { if (this.isServeOrSshOwnedPtyId(tab.ptyId)) { return true } const leafPtyIds = session.terminalLayoutsByTabId?.[tab.id]?.ptyIdsByLeafId - return ( - (leafPtyIds && - Object.values(leafPtyIds).some((ptyId) => this.isServeOrSshOwnedPtyId(ptyId))) || - // Why: expiry keeps pane coordinates so paired viewers can request a fresh shell. - this.getRecentExpiredSshLease(worktreeId, tab.id, undefined) !== null - ) + if ( + leafPtyIds && + Object.values(leafPtyIds).some((ptyId) => this.isServeOrSshOwnedPtyId(ptyId)) + ) { + return true + } + // Why: expiry keeps pane coordinates so paired viewers can request a fresh shell. + recoverableTabIds ??= this.collectRecentExpiredSshLeaseTabIds(worktreeId) + return recoverableTabIds.has(tab.id) }) } @@ -128,6 +135,54 @@ export class OrcaRuntimeWithReconcileHeadlessMobileSessionBrowserTabs extends Or ) } + /** + * Why eligibility belongs in the selection, not after it: a pane accumulates leases as it + * re-leases under new relay ids, so `(worktreeId, tabId, leafId)` names several. A superseded or + * relay-id-recycled predecessor is `expired` for a reason that already names its successor, and + * the unqualified callers use this answer to decide a pane is still recoverable — reporting one + * would offer paired viewers a recovery `recoverTerminalPane` then refuses. Picking the first + * ELIGIBLE orphan also keeps a predecessor from shadowing the successor that is genuinely + * reattachable. + */ + private isRecentExpiredSshLeaseForWorktree( + lease: ReturnType>[number], + worktreeId: string, + now: number + ): boolean { + return ( + lease.state === 'expired' && + lease.worktreeId === worktreeId && + sshRemotePtyLeaseAllowsReattach(lease) && + lease.updatedAt <= now && + now - lease.updatedAt <= SSH_PANE_RECOVERY_GRACE_MS + ) + } + + /** + * Leaf is the pane's identity; the frozen tabId is only trustworthy while nothing else can say + * where the leaf actually lives. + */ + private resolveExpiredSshLeaseTabId( + lease: ReturnType>[number] + ): string { + const currentTabId = lease.leafId + ? this.findCurrentTerminalTabIdForLeaf(lease.targetId, lease.leafId) + : undefined + return currentTabId ?? lease.tabId + } + + /** The tabs a recent eligible expired lease still names, resolved in one sweep of the leases. */ + protected collectRecentExpiredSshLeaseTabIds(worktreeId: string): ReadonlySet { + const now = Date.now() + const tabIds = new Set() + for (const lease of this.store?.getSshRemotePtyLeases?.() ?? []) { + if (this.isRecentExpiredSshLeaseForWorktree(lease, worktreeId, now)) { + tabIds.add(this.resolveExpiredSshLeaseTabId(lease)) + } + } + return tabIds + } + protected getRecentExpiredSshLease( worktreeId: string, tabId: string, @@ -137,34 +192,17 @@ export class OrcaRuntimeWithReconcileHeadlessMobileSessionBrowserTabs extends Or const now = Date.now() return ( this.store?.getSshRemotePtyLeases?.().find((lease) => { - if (lease.state !== 'expired' || lease.worktreeId !== worktreeId) { + if (!this.isRecentExpiredSshLeaseForWorktree(lease, worktreeId, now)) { return false } - // Why eligibility belongs in the selection, not after it: a pane accumulates leases as it - // re-leases under new relay ids, so `(worktreeId, tabId, leafId)` names several. A - // superseded or relay-id-recycled predecessor is `expired` for a reason that already names - // its successor, and the unqualified callers use this answer to decide a pane is still - // recoverable — reporting one would offer paired viewers a recovery `recoverTerminalPane` - // then refuses. Picking the first ELIGIBLE orphan also keeps a predecessor from shadowing - // the successor that is genuinely reattachable. - if (!sshRemotePtyLeaseAllowsReattach(lease)) { - return false - } - // Leaf is the pane's identity; the frozen tabId is only trustworthy while nothing else can - // say where the leaf actually lives. - const currentTabId = lease.leafId - ? this.findCurrentTerminalTabIdForLeaf(lease.targetId, lease.leafId) - : undefined return ( - (currentTabId ?? lease.tabId) === tabId && + this.resolveExpiredSshLeaseTabId(lease) === tabId && // Leases store RELAY form (`toStoredPtyId` -> `toRelaySshPtyId`); the runtime hands us // the APP form (`ssh:@@pty-3`). A raw `===` therefore never held for an SSH // pane, which is what kept this reader's only ptyId-qualified caller inert. (ptyId === undefined || lease.ptyId === toComparableRelaySshPtyId(lease.targetId, ptyId)) && - (leafId === undefined || lease.leafId === undefined || lease.leafId === leafId) && - lease.updatedAt <= now && - now - lease.updatedAt <= SSH_PANE_RECOVERY_GRACE_MS + (leafId === undefined || lease.leafId === undefined || lease.leafId === leafId) ) }) ?? null ) diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index 97a6fdec161..5b2c160f2c6 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -153,7 +153,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu async inspectTerminalProcess( terminalSelector: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise { const leaf = this.resolveLiveLeafForHandle(terminalSelector) if (!leaf?.ptyId || !this.ptyController) { diff --git a/src/main/runtime/relay/desktop-relay-service.ts b/src/main/runtime/relay/desktop-relay-service.ts index 55b993bd5a6..def786e7758 100644 --- a/src/main/runtime/relay/desktop-relay-service.ts +++ b/src/main/runtime/relay/desktop-relay-service.ts @@ -6,6 +6,7 @@ import type { PairingGetEndpointsResult, PairingProvisionRelayParams } from '../../../shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import { readRelayAuthContext } from './relay-auth-context' import { RelayAuthCoordinator } from './relay-auth-coordinator' import { RelaySessionBroker, type RelayBrokerStatus } from './relay-session-broker' @@ -112,7 +113,7 @@ export class DesktopRelayService { this.refreshDemand() } - fenceAndCloseNow(): void { + fenceAndCloseNow(hostCloseReason?: RelayHostCloseReason): void { // Why: a fence must be hard — a surviving liveness tick could catch the // window between the pre-sign-out fence and the profile wipe and briefly // resurrect a broker. The next auth mutation re-arms via refreshDemand. @@ -120,7 +121,7 @@ export class DesktopRelayService { clearInterval(this.livenessTimer) this.livenessTimer = null } - this.coordinator.fenceAndCloseNow() + this.coordinator.fenceAndCloseNow(hostCloseReason) } async createPairingRelay( diff --git a/src/main/runtime/relay/relay-auth-coordinator.ts b/src/main/runtime/relay/relay-auth-coordinator.ts index ae439a320d9..a7db0e3a81e 100644 --- a/src/main/runtime/relay/relay-auth-coordinator.ts +++ b/src/main/runtime/relay/relay-auth-coordinator.ts @@ -1,3 +1,7 @@ +import { + RELAY_HOST_CLOSE_REASON, + type RelayHostCloseReason +} from '../../../shared/relay-host-close-reason' import type { RelayBrokerStatus } from './relay-session-broker' import { RelayHttpError, shouldRetryRelayConnectionError } from './relay-http-client' @@ -14,7 +18,7 @@ export type RelayAuthContext = { } export type CoordinatedRelayBroker = { - closeNow(): void + closeNow(hostCloseReason?: RelayHostCloseReason): void isLive?(): boolean } @@ -78,13 +82,16 @@ export class RelayAuthCoordinator { void reconcile } - fenceAndCloseNow(): void { + // hostCloseReason names an auth loss the phone should be told about. Quit, + // relaunch and every other fence pass nothing, so the control socket dies + // abruptly exactly as before and the cell records no cause. + fenceAndCloseNow(hostCloseReason?: RelayHostCloseReason): void { ++this.authEpoch this.cancelLinger() this.cancelRetry() this.retryAttempt = 0 this.invalidatePendingOwnerships() - this.invalidateOwnership() + this.invalidateOwnership(hostCloseReason) this.options.onStatus('offline') } @@ -146,7 +153,11 @@ export class RelayAuthCoordinator { if (!context || !context.relayEntitled) { this.cancelLinger() this.retryAttempt = 0 - this.invalidateOwnership() + // Why only the null case: readContext throws on transient failures and + // returns null solely when the cloud session is gone (absent, or cleared + // by a 401). A present-but-unentitled context is still a signed-in + // desktop, and "sign in to reconnect" would be wrong advice for it. + this.invalidateOwnership(context ? undefined : RELAY_HOST_CLOSE_REASON.SIGNED_OUT) this.options.onStatus('offline') return } @@ -271,12 +282,12 @@ export class RelayAuthCoordinator { return context.accessToken } - private invalidateOwnership(): void { + private invalidateOwnership(hostCloseReason?: RelayHostCloseReason): void { const ownership = this.ownership this.ownership = null if (ownership) { ownership.valid = false - ownership.broker?.closeNow() + ownership.broker?.closeNow(hostCloseReason) } } diff --git a/src/main/runtime/relay/relay-auth-host-close-reason.test.ts b/src/main/runtime/relay/relay-auth-host-close-reason.test.ts new file mode 100644 index 00000000000..7d10e207a33 --- /dev/null +++ b/src/main/runtime/relay/relay-auth-host-close-reason.test.ts @@ -0,0 +1,122 @@ +import { describe, expect, it, vi } from 'vitest' +import { RELAY_HOST_CLOSE_REASON } from '../../../shared/relay-host-close-reason' +import { RelayAuthCoordinator, type RelayAuthContext } from './relay-auth-coordinator' + +const context: RelayAuthContext = { + identity: { userId: 'user-1', profileId: 'profile-1', organizationId: 'org-1' }, + accessToken: 'access-1', + relayEntitled: true +} + +function coordinatorOver(readContext: () => Promise) { + const broker = { closeNow: vi.fn() } + const coordinator = new RelayAuthCoordinator({ + readContext, + openBroker: async () => broker, + onStatus: vi.fn() + }) + return { broker, coordinator } +} + +describe('relay control close reason', () => { + it('names the sign-out when the cloud session is gone', async () => { + let current: RelayAuthContext | null = context + const { broker, coordinator } = coordinatorOver(async () => current) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + current = null + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(broker.closeNow).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('names the sign-out on the explicit pre-sign-out fence', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.fenceAndCloseNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + expect(broker.closeNow).toHaveBeenCalledWith(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + }) + + it('stays silent on quit, which fences without a reason', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.fenceAndCloseNow() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + it('stays silent on stop, which is teardown rather than auth loss', async () => { + const { broker, coordinator } = coordinatorOver(async () => context) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + coordinator.stop() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + // A signed-in desktop that merely lost the entitlement must not tell the + // phone to sign in — the copy would be wrong and the user has nothing to do. + it('stays silent when the session survives but the entitlement is gone', async () => { + let current: RelayAuthContext = context + const { broker, coordinator } = coordinatorOver(async () => current) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + current = { ...context, relayEntitled: false } + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + it('stays silent when demand drops and the broker lingers out', async () => { + let demanded = true + const broker = { closeNow: vi.fn() } + const coordinator = new RelayAuthCoordinator({ + readContext: async () => context, + hasDemand: () => demanded, + openBroker: async () => broker, + onStatus: vi.fn(), + lingerMs: 5 + }) + coordinator.reconcile() + await expect(coordinator.waitForLiveBroker()).resolves.toBe(broker) + + demanded = false + coordinator.reconcile() + await vi.waitFor(() => expect(broker.closeNow).toHaveBeenCalled()) + + expect(broker.closeNow).toHaveBeenCalledWith(undefined) + }) + + // Replacing a stale broker is a reconnect, not a sign-out. + it('stays silent when an identity switch replaces the broker', async () => { + let current = context + const brokers: { closeNow: ReturnType }[] = [] + const coordinator = new RelayAuthCoordinator({ + readContext: async () => current, + openBroker: async () => { + const broker = { closeNow: vi.fn() } + brokers.push(broker) + return broker + }, + onStatus: vi.fn() + }) + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + current = { ...context, identity: { ...context.identity, organizationId: 'org-2' } } + coordinator.reconcile() + await coordinator.waitForLiveBroker() + + expect(brokers[0]?.closeNow).toHaveBeenCalledWith(undefined) + }) +}) diff --git a/src/main/runtime/relay/relay-control-client.ts b/src/main/runtime/relay/relay-control-client.ts index 7e742173f72..968f05795b2 100644 --- a/src/main/runtime/relay/relay-control-client.ts +++ b/src/main/runtime/relay/relay-control-client.ts @@ -1,6 +1,7 @@ import { randomUUID } from 'node:crypto' import WebSocket, { type RawData } from 'ws' import { MOBILE_RELAY_CLOSE_CODE } from '../../../shared/mobile-relay-close-codes' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { E2EEKeypair } from '../e2ee-keypair' import { RelayConnectionOpenMessageSchema, @@ -8,6 +9,7 @@ import { RelayHostChallengeMessageSchema, RelayHostHelloAckMessageSchema, RelayPingMessageSchema, + encodeRelayHostHello, parseRelayControlMessage, type RelayConnectionOpenMessage, type RelayDrainMessage, @@ -21,6 +23,7 @@ import { RELAY_CONTROL_SILENCE_LIMIT_MS, RelayControlSilenceWatchdog } from './relay-control-silence-watchdog' +import { closeRelayControlSocket } from './relay-control-socket-close' import { controlWebSocketUrl } from './relay-control-url' type RelayControlState = 'idle' | 'opening' | 'proving' | 'active' | 'draining' | 'closed' @@ -166,7 +169,7 @@ export class RelayControlClient { return this.requests.confirmResume(reqId, basisConnId, (payload) => this.sendActive(payload)) } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { const wasConnecting = this.state === 'opening' || this.state === 'proving' this.state = 'closed' this.silenceWatchdog.stop() @@ -175,8 +178,9 @@ export class RelayControlClient { this.clearConnectPromise() } this.requests.rejectAll(new Error('relay_control_closed')) - this.socket?.terminate() + const socket = this.socket this.socket = null + closeRelayControlSocket(socket, hostCloseReason) } private sendHostHello(): void { @@ -185,19 +189,9 @@ export class RelayControlClient { } this.state = 'proving' this.socket.send( - JSON.stringify({ - type: 'host-hello', - v: 1, - relayHostId: this.options.relayHostId, - assignmentEpoch: this.options.assignmentEpoch, - hostPublicKeyB64: this.options.keypair.publicKeyB64, - appVersion: this.options.appVersion, - ...(this.options.previousGeneration === undefined - ? {} - : { previousGeneration: this.options.previousGeneration }), - ...(this.options.controlResumeSecret - ? { controlResumeSecret: this.options.controlResumeSecret } - : {}) + encodeRelayHostHello({ + ...this.options, + hostPublicKeyB64: this.options.keypair.publicKeyB64 }) ) } diff --git a/src/main/runtime/relay/relay-control-close-reason.test.ts b/src/main/runtime/relay/relay-control-close-reason.test.ts new file mode 100644 index 00000000000..0d9aba54f88 --- /dev/null +++ b/src/main/runtime/relay/relay-control-close-reason.test.ts @@ -0,0 +1,92 @@ +import { createHash } from 'node:crypto' +import { afterEach, describe, expect, it, vi } from 'vitest' +import nacl from 'tweetnacl' +import { WebSocketServer, type WebSocket } from 'ws' +import { RELAY_HOST_CLOSE_REASON } from '../../../shared/relay-host-close-reason' +import { RelayControlClient } from './relay-control-client' + +type ObservedClose = { code: number; reason: string } + +describe('RelayControlClient close reason', () => { + const servers: WebSocketServer[] = [] + const clients: RelayControlClient[] = [] + + afterEach(async () => { + for (const client of clients.splice(0)) { + client.closeNow() + } + await Promise.all( + servers.splice(0).map( + (server) => + new Promise((resolve) => { + for (const socket of server.clients) { + socket.terminate() + } + server.close(() => resolve()) + }) + ) + ) + }) + + async function connectedClient(): Promise<{ + client: RelayControlClient + closed: Promise + }> { + const server = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) + servers.push(server) + await new Promise((resolve) => server.once('listening', resolve)) + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('expected TCP relay test server') + } + const accepted = new Promise((resolve) => server.once('connection', resolve)) + const keypair = nacl.box.keyPair() + const client = new RelayControlClient({ + cellUrl: `http://127.0.0.1:${address.port}`, + relayJwt: 'scoped-token', + relayHostId: createHash('sha256').update(keypair.publicKey).digest('base64url').slice(0, 16), + assignmentEpoch: 1, + identity: { userId: 'user-1', profileId: 'profile-1', organizationId: 'org-1' }, + keypair: { ...keypair, publicKeyB64: Buffer.from(keypair.publicKey).toString('base64') }, + appVersion: '1.2.3', + onConnectionOpen: vi.fn(), + onDrain: vi.fn(), + onClose: vi.fn() + }) + clients.push(client) + // The handshake never completes here; only the transport close matters. + void client.connect().catch(() => {}) + const socket = await accepted + // A pong proves the client socket left CONNECTING; closeNow can only send a + // close frame from OPEN, and that is the state a real sign-out fences from. + await new Promise((resolve) => { + socket.once('pong', () => resolve()) + socket.ping() + }) + const closed = new Promise((resolve) => { + socket.once('close', (code, reason) => resolve({ code, reason: reason.toString() })) + }) + return { client, closed } + } + + it('delivers the sign-out reason to the cell', async () => { + const { client, closed } = await connectedClient() + + client.closeNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) + + await expect(closed).resolves.toEqual({ + code: 1000, + reason: RELAY_HOST_CLOSE_REASON.SIGNED_OUT + }) + }) + + // Every non-auth close keeps today's abrupt terminate, so a cell can never + // read a quit, a rotation or a sleep as a sign-out. + it('closes abruptly with no reason when none is given', async () => { + const { client, closed } = await connectedClient() + + client.closeNow() + + await expect(closed).resolves.toEqual({ code: 1006, reason: '' }) + }) +}) diff --git a/src/main/runtime/relay/relay-control-origin.ts b/src/main/runtime/relay/relay-control-origin.ts index 4d42da47977..3a33e1617e6 100644 --- a/src/main/runtime/relay/relay-control-origin.ts +++ b/src/main/runtime/relay/relay-control-origin.ts @@ -8,6 +8,7 @@ import type { RelayDrainMessage, RelayHostHelloAckMessage } from './relay-control-protocol' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { RelayIdentity } from './relay-session-broker-contract' import type { RelayAssignment } from './relay-http-client' @@ -144,7 +145,7 @@ export class RelayControlOrigin { } } - async close(): Promise { + async close(hostCloseReason?: RelayHostCloseReason): Promise { if (this.closed) { return } @@ -154,7 +155,7 @@ export class RelayControlOrigin { } this.retiredControlTimers.clear() for (const control of this.controls) { - control.closeNow() + control.closeNow(hostCloseReason) } this.controls.clear() this.activeControl = null @@ -166,8 +167,8 @@ export class RelayControlOrigin { } } - closeNow(): void { - void this.close() + closeNow(hostCloseReason?: RelayHostCloseReason): void { + void this.close(hostCloseReason) } private async openControl(overrides?: { diff --git a/src/main/runtime/relay/relay-control-protocol.ts b/src/main/runtime/relay/relay-control-protocol.ts index c94797f4815..75bf3a390e2 100644 --- a/src/main/runtime/relay/relay-control-protocol.ts +++ b/src/main/runtime/relay/relay-control-protocol.ts @@ -143,3 +143,29 @@ export function parseRelayControlMessage(raw: RawData): Record return null } } + +export type RelayHostHello = { + relayHostId: string + assignmentEpoch: number + hostPublicKeyB64: string + appVersion: string + previousGeneration?: number + controlResumeSecret?: string +} + +// Optional members are omitted rather than sent as undefined: the cell parses +// host-hello strictly and an explicit null is not the same as absent. +export function encodeRelayHostHello(hello: RelayHostHello): string { + return JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hello.relayHostId, + assignmentEpoch: hello.assignmentEpoch, + hostPublicKeyB64: hello.hostPublicKeyB64, + appVersion: hello.appVersion, + ...(hello.previousGeneration === undefined + ? {} + : { previousGeneration: hello.previousGeneration }), + ...(hello.controlResumeSecret ? { controlResumeSecret: hello.controlResumeSecret } : {}) + }) +} diff --git a/src/main/runtime/relay/relay-control-socket-close.ts b/src/main/runtime/relay/relay-control-socket-close.ts new file mode 100644 index 00000000000..2984b075425 --- /dev/null +++ b/src/main/runtime/relay/relay-control-socket-close.ts @@ -0,0 +1,26 @@ +import type WebSocket from 'ws' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' + +const NORMAL_CLOSE_CODE = 1000 +const REASONED_CLOSE_FLUSH_MS = 1_000 + +// hostCloseReason: only auth loss names itself. Every other control close +// (rotation, drain, quit, sleep) stays an abrupt terminate, so the cell learns +// nothing and can never read a restart as a sign-out. A named close has to +// reach the cell as a real close frame, but the fence must still be hard — +// bound the handshake and then terminate. +export function closeRelayControlSocket( + socket: WebSocket | null, + hostCloseReason?: RelayHostCloseReason +): void { + if (!socket) { + return + } + if (hostCloseReason && socket.readyState === socket.OPEN) { + socket.close(NORMAL_CLOSE_CODE, hostCloseReason) + const timer = setTimeout(() => socket.terminate(), REASONED_CLOSE_FLUSH_MS) + timer.unref?.() + return + } + socket.terminate() +} diff --git a/src/main/runtime/relay/relay-origin-pool.ts b/src/main/runtime/relay/relay-origin-pool.ts index a95d8f21340..ca29c41e8a7 100644 --- a/src/main/runtime/relay/relay-origin-pool.ts +++ b/src/main/runtime/relay/relay-origin-pool.ts @@ -4,6 +4,7 @@ import type { MobileSocketWiring } from '../rpc/mobile-socket-wiring' import { RelayControlOrigin } from './relay-control-origin' import type { RelayControlClient } from './relay-control-client' import type { RelayDrainMessage } from './relay-control-protocol' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import { RelayDrainRetrySchedule } from './relay-drain-retry-schedule' import { RelayHttpError, requestRelayAssignment, type RelayAssignment } from './relay-http-client' import type { RelayBrokerStatus, RelayIdentity } from './relay-session-broker-contract' @@ -79,7 +80,7 @@ export class RelayOriginPool { } } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { if (this.closed) { return } @@ -94,7 +95,7 @@ export class RelayOriginPool { } this.drainTimers.clear() for (const origin of this.origins) { - origin.closeNow() + origin.closeNow(hostCloseReason) } this.origins.clear() this.drainingOrigins.clear() diff --git a/src/main/runtime/relay/relay-session-broker.ts b/src/main/runtime/relay/relay-session-broker.ts index e12e09119a0..d8d0ec45047 100644 --- a/src/main/runtime/relay/relay-session-broker.ts +++ b/src/main/runtime/relay/relay-session-broker.ts @@ -6,6 +6,7 @@ import type { MobileRelayEndpoint, PairingProvisionRelayParams } from '../../../shared/mobile-relay-credential-contract' +import type { RelayHostCloseReason } from '../../../shared/relay-host-close-reason' import type { DeviceCredentialInstallAuthorization } from './relay-control-requests' import { deriveRelayHostId, @@ -180,7 +181,7 @@ export class RelaySessionBroker { return result } - closeNow(): void { + closeNow(hostCloseReason?: RelayHostCloseReason): void { if (this.closed) { return } @@ -190,7 +191,7 @@ export class RelaySessionBroker { clearTimeout(this.refreshTimer) this.refreshTimer = null } - this.originPool.closeNow() + this.originPool.closeNow(hostCloseReason) if (publishOffline) { this.options.onStatus('offline') } diff --git a/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts new file mode 100644 index 00000000000..48649bc372c --- /dev/null +++ b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts @@ -0,0 +1,92 @@ +// The host half of the same contract: an RPC schema silently strips keys it does not declare, which +// is exactly what forward compatibility needs and exactly how a caller's option can vanish inside +// one version. `scanChildProcesses` has to be declared here, and only here -- the sibling handle +// methods have no use for it and must keep refusing it. +import { describe, expect, it, vi } from 'vitest' +import type { ZodType } from 'zod' +import { TERMINAL_QUERY_METHODS } from './terminal-query-methods' +import { TerminalHandle, TerminalInspectProcess } from './unary-schemas' + +/** The method as registered, so a schema swap on the definition cannot pass unseen. */ +function inspectProcessMethod() { + const method = TERMINAL_QUERY_METHODS.find((entry) => entry.name === 'terminal.inspectProcess') + if (!method) { + throw new Error('terminal.inspectProcess is not registered') + } + return method +} + +async function callRegisteredHandler( + params: Record +): Promise<{ terminal: string; options: unknown }> { + const method = inspectProcessMethod() + const parsed = (method.params as ZodType).parse(params) + const inspectTerminalProcess = vi.fn(async () => ({ + foregroundProcess: null, + hasChildProcesses: false + })) + await method.handler(parsed, { runtime: { inspectTerminalProcess } } as never, undefined as never) + const [terminal, options] = inspectTerminalProcess.mock.calls[0] as unknown as [string, unknown] + return { terminal, options } +} + +describe('terminal.inspectProcess registration', () => { + // The half the schema test alone cannot see: pointing the method back at the shared handle schema + // compiles, parses, and silently drops the option. This exercises the registered definition. + it('carries scanChildProcesses from the wire into the runtime call', async () => { + await expect( + callRegisteredHandler({ terminal: 'term_1', scanChildProcesses: true }) + ).resolves.toEqual({ terminal: 'term_1', options: { scanChildProcesses: true } }) + }) + + it('carries it alongside the incarnation fence', async () => { + await expect( + callRegisteredHandler({ + terminal: 'term_1', + expectedIncarnationId: 'inc-1', + scanChildProcesses: true + }) + ).resolves.toEqual({ + terminal: 'term_1', + options: { expectedIncarnationId: 'inc-1', scanChildProcesses: true } + }) + }) + + it('keeps the legacy one-argument shape for a bare poll', async () => { + await expect(callRegisteredHandler({ terminal: 'term_1' })).resolves.toEqual({ + terminal: 'term_1', + options: undefined + }) + }) +}) + +describe('terminal.inspectProcess params', () => { + it('preserves scanChildProcesses', () => { + expect(TerminalInspectProcess.parse({ terminal: 'term_1', scanChildProcesses: true })).toEqual({ + terminal: 'term_1', + scanChildProcesses: true + }) + }) + + it('preserves it alongside the incarnation fence', () => { + expect( + TerminalInspectProcess.parse({ + terminal: 'term_1', + expectedIncarnationId: 'inc-1', + scanChildProcesses: true + }) + ).toEqual({ terminal: 'term_1', expectedIncarnationId: 'inc-1', scanChildProcesses: true }) + }) + + it('leaves it absent for a polling caller', () => { + expect(TerminalInspectProcess.parse({ terminal: 'term_1' })).toEqual({ terminal: 'term_1' }) + }) + + // The shape that produced the bug, pinned so nobody "simplifies" the method back onto the shared + // handle schema: TerminalHandle drops the option on the floor without complaining. + it('shows why the shared handle schema could not carry it', () => { + expect(TerminalHandle.parse({ terminal: 'term_1', scanChildProcesses: true })).toEqual({ + terminal: 'term_1' + }) + }) +}) diff --git a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts index a8aec9ff573..bf7b4a5bd87 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts @@ -1,6 +1,7 @@ import { defineMethod, type RpcAnyMethod } from '../../core' import { TerminalHandle, + TerminalInspectProcess, TerminalListParams, TerminalRead, TerminalRecoverPane, @@ -65,15 +66,21 @@ export const TERMINAL_QUERY_METHODS: RpcAnyMethod[] = [ }), defineMethod({ name: 'terminal.inspectProcess', - params: TerminalHandle, - handler: async (params, { runtime }) => ({ - process: await runtime.inspectTerminalProcess( - params.terminal, - params.expectedIncarnationId + params: TerminalInspectProcess, + handler: async (params, { runtime }) => { + const options = { + ...(params.expectedIncarnationId ? { expectedIncarnationId: params.expectedIncarnationId } - : undefined - ) - }) + : {}), + ...(params.scanChildProcesses === true ? { scanChildProcesses: true } : {}) + } + return { + process: await runtime.inspectTerminalProcess( + params.terminal, + Object.keys(options).length > 0 ? options : undefined + ) + } + } }), defineMethod({ name: 'terminal.isRunningAgent', diff --git a/src/main/runtime/rpc/methods/terminal/unary-schemas.ts b/src/main/runtime/rpc/methods/terminal/unary-schemas.ts index 323b34a7544..afe0a3bb486 100644 --- a/src/main/runtime/rpc/methods/terminal/unary-schemas.ts +++ b/src/main/runtime/rpc/methods/terminal/unary-schemas.ts @@ -13,6 +13,17 @@ export const TerminalFocus = TerminalHandle.extend({ navigation: z.enum(['caller', 'host']).optional() }) +/** + * `terminal.inspectProcess` carries one member the sibling handle methods must not: whether the + * caller's answer decides something once, which is what licenses the host to pay for a process-table + * read. Extended rather than added to `TerminalHandle` so `clearBuffer`/`agentStatus`/`isRunningAgent` + * keep refusing an option they have no use for. + */ +export const TerminalInspectProcess = TerminalHandle.extend({ + // Additive request member understood by newer hosts; legacy hosts safely ignore it. + scanChildProcesses: z.boolean().optional() +}) + export const TerminalListParams = z.object({ worktree: OptionalString, limit: OptionalFiniteNumber, diff --git a/src/main/runtime/runtime-managed-worktree-queries.test.ts b/src/main/runtime/runtime-managed-worktree-queries.test.ts index 01df53cbd1d..3fb792fcc7c 100644 --- a/src/main/runtime/runtime-managed-worktree-queries.test.ts +++ b/src/main/runtime/runtime-managed-worktree-queries.test.ts @@ -40,14 +40,18 @@ function metadata(overrides: Partial = {}): WorktreeMeta { } } -function queries(store: RuntimeStore): RuntimeManagedWorktreeQueries { +function queries( + store: RuntimeStore, + overrides: Partial[0]> = {} +): RuntimeManagedWorktreeQueries { return new RuntimeManagedWorktreeQueries({ getStore: () => store, listResolved: async () => [], resolveRepo: async () => store.getRepos()[0]!, selectRepos: () => store.getRepos(), scanRepo: async () => ({ ok: true, worktrees: [] }), - listKnownHostIds: () => [] + listKnownHostIds: () => [], + ...overrides }) } @@ -105,3 +109,115 @@ describe('RuntimeManagedWorktreeQueries.listDetected', () => { expect(legacy.worktrees[0]).not.toHaveProperty('visibilitySource') }) }) + +describe('RuntimeManagedWorktreeQueries.list host scope', () => { + // Measured on hardware before this fix, same runtime and same refusing SSH host in the same + // second: the UNSCOPED listing reported `omittedHostIds: ["local","ssh:ssh-scope-refused"]` with + // `--host` selectors, while the SCOPED listing reported `{"hostIds":[],"omittedHostIds":[]}`. + // A listing that covered nothing, reporting no gaps, is indistinguishable from a repo that + // genuinely has no worktrees -- the thing docs/reference/ssh-execution-boundary.md forbids. + function sshStore(): RuntimeStore { + const repo = folderRepo({ + id: 'repo-ssh', + kind: 'git', + connectionId: 'conn-1', + path: '/home/dev/app' + }) + return { + getRepos: () => [repo], + getRepo: () => repo, + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getSettings: () => settings + } as unknown as RuntimeStore + } + + it('names the scoped repo host as omitted when the listing covered nothing', async () => { + const result = await queries(sshStore()).list('repo-ssh', 50) + + expect(result.totalCount).toBe(0) + expect(result.hostScope).toEqual({ + hostIds: [], + omittedHostIds: ['ssh:conn-1'] + }) + }) + + it('does not report the scoped host as omitted once it contributes rows', async () => { + const store = sshStore() + const result = await queries(store, { + listResolved: async () => + [ + { + id: 'repo-ssh::/home/dev/app', + repoId: 'repo-ssh', + path: '/home/dev/app', + hostId: 'ssh:conn-1' + } + ] as never + }).list('repo-ssh', 50) + + expect(result.hostScope?.hostIds).toEqual(['ssh:conn-1']) + expect(result.hostScope?.omittedHostIds).toEqual([]) + }) + + // The caller scoped the listing, so the hosts they excluded must not come back as gaps. + it('never names a host the caller scoped out', async () => { + const scoped = await queries(sshStore(), { + listKnownHostIds: () => ['local', 'ssh:other', 'runtime:elsewhere'] as never + }).list('repo-ssh', 50) + + expect(scoped.hostScope?.omittedHostIds).toEqual(['ssh:conn-1']) + }) + + // `getRepoExecutionHostId` derives the host from two spellings, and a scoped listing that named + // the wrong one would be worse than naming none. These pin both. + it('names the local host for a scoped local repo', async () => { + const repo = folderRepo({ id: 'repo-local', kind: 'git', path: '/workspace/local' }) + const store = { + getRepos: () => [repo], + getRepo: () => repo, + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getSettings: () => settings + } as unknown as RuntimeStore + + const result = await queries(store).list('repo-local', 50) + + expect(result.hostScope?.omittedHostIds).toEqual(['local']) + }) + + it('prefers executionHostId over connectionId for the scoped host', async () => { + const repo = folderRepo({ + id: 'repo-runtime', + kind: 'git', + connectionId: 'conn-legacy', + executionHostId: 'runtime:env-1', + path: '/workspace/runtime' + }) + const store = { + getRepos: () => [repo], + getRepo: () => repo, + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getSettings: () => settings + } as unknown as RuntimeStore + + const result = await queries(store).list('repo-runtime', 50) + + expect(result.hostScope?.omittedHostIds).toEqual(['runtime:env-1']) + }) + + it('still reports every configured host when the listing is unscoped', async () => { + const unscoped = await queries(sshStore(), { + listKnownHostIds: () => ['local', 'ssh:conn-1'] as never + }).list(undefined, 50) + + expect(unscoped.hostScope?.omittedHostIds).toEqual(['local', 'ssh:conn-1']) + }) +}) diff --git a/src/main/runtime/runtime-managed-worktree-queries.ts b/src/main/runtime/runtime-managed-worktree-queries.ts index 5a5812236af..158ab77cdd0 100644 --- a/src/main/runtime/runtime-managed-worktree-queries.ts +++ b/src/main/runtime/runtime-managed-worktree-queries.ts @@ -2,7 +2,7 @@ import type { DetectedWorktreeListResult, Worktree } from '../../shared/worktree import type { Repo } from '../../shared/repo-types' import type { RuntimeWorktreeListResult } from '../../shared/runtime-types' import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' -import { buildWorktreeListingPage } from './worktree-listing-host-scope' +import { buildWorktreeListingPage, listingKnownHostIds } from './worktree-listing-host-scope' import { readWorktreeMetaForHost } from '../persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from '../worktree-metadata-ownership' import type { WorktreeMeta } from '../../shared/worktree/meta-types' @@ -80,7 +80,7 @@ export class RuntimeManagedWorktreeQueries { throw new Error('invalid_limit') } const resolved = await this.deps.listResolved() - const repoId = repoSelector ? (await this.deps.resolveRepo(repoSelector)).id : null + const scopedRepo = repoSelector ? await this.deps.resolveRepo(repoSelector) : null const pathsByRepo = new Map() for (const worktree of resolved) { const paths = pathsByRepo.get(worktree.repoId) ?? [] @@ -100,12 +100,12 @@ export class RuntimeManagedWorktreeQueries { ) const worktrees = resolved.filter( (worktree) => - (!repoId || worktree.repoId === repoId) && + (!scopedRepo || worktree.repoId === scopedRepo.id) && this.isVisible(worktree, matchers.get(worktree.repoId), sourceDefaultsSupported) ) - // Why: a `--repo` listing was scoped by the caller, so naming every configured host as - // omitted would report a gap the caller deliberately excluded. - return buildWorktreeListingPage(worktrees, limit, repoId ? [] : this.deps.listKnownHostIds()) + // See `listingKnownHostIds`: a scoped listing must still name the host it was asked about. + const knownHostIds = listingKnownHostIds(scopedRepo, () => this.deps.listKnownHostIds()) + return buildWorktreeListingPage(worktrees, limit, knownHostIds) } resolveRepoForConnection(selector: string, connectionId?: string | null): Promise { diff --git a/src/main/runtime/runtime-metadata-ownership-watch.test.ts b/src/main/runtime/runtime-metadata-ownership-watch.test.ts index d8eb4196534..abaf95b6b4b 100644 --- a/src/main/runtime/runtime-metadata-ownership-watch.test.ts +++ b/src/main/runtime/runtime-metadata-ownership-watch.test.ts @@ -1,4 +1,6 @@ import { mkdtempSync, writeFileSync } from 'node:fs' +import type * as NodeFs from 'node:fs' +import type * as NodeFsPromises from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' @@ -10,6 +12,82 @@ import { type RuntimeMetadataOwnershipWatch } from './runtime-metadata-ownership-watch' +// Counts blocking fs calls against orca-runtime.json so the poll tick's I/O stays off the main thread. +const metadataSyncCalls = vi.hoisted(() => { + const state = { recording: false, calls: [] as string[] } + return { + state, + record(fn: string, target: unknown): void { + if (state.recording && typeof target === 'string' && target.endsWith('orca-runtime.json')) { + state.calls.push(fn) + } + } + } +}) + +// Lets a test park the tick's async read so overlapping ticks are observable without wall clocks. +const metadataReadGate = vi.hoisted(() => { + const gate = { + hold: false, + reads: 0, + parked: [] as (() => void)[], + /** Reads handed to the real fs; parked ones are excluded so `whenIdle` stays answerable. */ + active: 0, + idle: [] as (() => void)[], + whenIdle(): Promise { + return gate.active === 0 + ? Promise.resolve() + : new Promise((resolve) => gate.idle.push(resolve)) + } + } + return gate +}) + +vi.mock('node:fs/promises', async () => { + const actual = await vi.importActual('node:fs/promises') + return { + ...actual, + default: actual, + readFile: (async (target: unknown, options: never) => { + const call = (): unknown => + (actual.readFile as (...args: never[]) => unknown)(target as never, options) + if (typeof target !== 'string' || !target.endsWith('orca-runtime.json')) { + return call() + } + metadataReadGate.reads += 1 + if (metadataReadGate.hold) { + await new Promise((resolve) => metadataReadGate.parked.push(resolve)) + } + metadataReadGate.active += 1 + try { + return await call() + } finally { + metadataReadGate.active -= 1 + if (metadataReadGate.active === 0) { + for (const resolve of metadataReadGate.idle.splice(0)) { + resolve() + } + } + } + }) as typeof actual.readFile + } +}) + +vi.mock('node:fs', async () => { + const actual = await vi.importActual('node:fs') + return { + ...actual, + existsSync: (target: NodeFs.PathLike) => { + metadataSyncCalls.record('existsSync', target) + return actual.existsSync(target) + }, + readFileSync: ((target: never, options: never) => { + metadataSyncCalls.record('readFileSync', target) + return actual.readFileSync(target, options) + }) as typeof actual.readFileSync + } +}) + const OWNED_PID = 4242 const OWNED_RUNTIME_ID = 'rt_owner' const FOREIGN_LIVE_PID = 5151 @@ -81,6 +159,14 @@ describe('watchRuntimeMetadataOwnership', () => { const userDataPaths: string[] = [] afterEach(() => { + metadataSyncCalls.state.recording = false + metadataSyncCalls.state.calls.length = 0 + metadataReadGate.hold = false + metadataReadGate.reads = 0 + for (const resume of metadataReadGate.parked.splice(0)) { + resume() + } + metadataReadGate.idle.splice(0) for (const watch of watches.splice(0)) { watch.stop() } @@ -103,13 +189,32 @@ describe('watchRuntimeMetadataOwnership', () => { return watch } + function usePolledTimers(): void { + vi.useFakeTimers({ toFake: ['setInterval', 'clearInterval'] }) + } + + /** Waits out the tick's real read; everything after it resolves as microtasks. */ + async function settleReads(): Promise { + await new Promise((resolve) => setImmediate(resolve)) + await metadataReadGate.whenIdle() + await new Promise((resolve) => setImmediate(resolve)) + } + + /** Fires one interval at a time so each tick's async read settles before the next. */ + async function advancePolls(ms: number, stepMs = 1_000): Promise { + for (let elapsed = 0; elapsed < ms; elapsed += stepMs) { + await vi.advanceTimersByTimeAsync(stepMs) + await settleReads() + } + } + function makeUserDataPath(): string { const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-ownership-')) userDataPaths.push(userDataPath) return userDataPath } - it('republishes after a second instance clobbers the record and exits', () => { + it('republishes after a second instance clobbers the record and exits', async () => { const userDataPath = makeUserDataPath() writeRuntimeMetadata(userDataPath, record()) const watch = armWatch(userDataPath) @@ -118,7 +223,7 @@ describe('watchRuntimeMetadataOwnership', () => { userDataPath, record({ pid: FOREIGN_DEAD_PID, runtimeId: 'rt_second_instance' }) ) - watch.check() + await watch.check() expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: OWNED_PID, @@ -126,28 +231,28 @@ describe('watchRuntimeMetadataOwnership', () => { }) }) - it('republishes a record that was deleted underneath the runtime', () => { + it('republishes a record that was deleted underneath the runtime', async () => { const userDataPath = makeUserDataPath() writeRuntimeMetadata(userDataPath, record()) const watch = armWatch(userDataPath) clearRuntimeMetadata(userDataPath) - watch.check() + await watch.check() expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: OWNED_PID }) }) - it('replaces an unreadable record', () => { + it('replaces an unreadable record', async () => { const userDataPath = makeUserDataPath() const watch = armWatch(userDataPath) writeFileSync(getRuntimeMetadataPath(userDataPath), '{ truncated') - watch.check() + await watch.check() expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: OWNED_PID }) }) - it('leaves a live sibling runtime in place', () => { + it('leaves a live sibling runtime in place', async () => { const userDataPath = makeUserDataPath() const watch = armWatch(userDataPath) writeRuntimeMetadata( @@ -155,13 +260,13 @@ describe('watchRuntimeMetadataOwnership', () => { record({ pid: FOREIGN_LIVE_PID, runtimeId: 'rt_second_instance' }) ) - watch.check() + await watch.check() expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: FOREIGN_LIVE_PID }) }) - it('reclaims on the poll interval without an explicit check', () => { - vi.useFakeTimers() + it('reclaims on the poll interval without an explicit check', async () => { + usePolledTimers() const userDataPath = makeUserDataPath() armWatch(userDataPath, 1_000) writeRuntimeMetadata( @@ -169,13 +274,13 @@ describe('watchRuntimeMetadataOwnership', () => { record({ pid: FOREIGN_DEAD_PID, runtimeId: 'rt_second_instance' }) ) - vi.advanceTimersByTime(1_000) + await advancePolls(1_000) expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: OWNED_PID }) }) - it('stops reclaiming once the watch is stopped', () => { - vi.useFakeTimers() + it('stops reclaiming once the watch is stopped', async () => { + usePolledTimers() const userDataPath = makeUserDataPath() const watch = armWatch(userDataPath, 1_000) @@ -184,13 +289,59 @@ describe('watchRuntimeMetadataOwnership', () => { userDataPath, record({ pid: FOREIGN_DEAD_PID, runtimeId: 'rt_second_instance' }) ) - vi.advanceTimersByTime(5_000) + await advancePolls(5_000) expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: FOREIGN_DEAD_PID }) }) - it('keeps polling after a republish failure', () => { - vi.useFakeTimers() + it('reads the record off-thread, so the poll tick never blocks the main thread', async () => { + const userDataPath = makeUserDataPath() + writeRuntimeMetadata(userDataPath, record()) + const watch = armWatch(userDataPath) + + metadataSyncCalls.state.recording = true + await watch.check() + await watch.check() + metadataSyncCalls.state.recording = false + + expect(metadataSyncCalls.state.calls).toEqual([]) + }) + + it('treats a missing record as reclaimable without a pre-existence check', async () => { + const userDataPath = makeUserDataPath() + const watch = armWatch(userDataPath) + + metadataSyncCalls.state.recording = true + await watch.check() + metadataSyncCalls.state.recording = false + + expect(metadataSyncCalls.state.calls).toEqual([]) + expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: OWNED_PID }) + }) + + it('never runs two overlapping ownership checks', async () => { + usePolledTimers() + const userDataPath = makeUserDataPath() + writeRuntimeMetadata(userDataPath, record()) + armWatch(userDataPath, 1_000) + + metadataReadGate.hold = true + await advancePolls(3_000) + + expect(metadataReadGate.reads).toBe(1) + + metadataReadGate.hold = false + for (const resume of metadataReadGate.parked.splice(0)) { + resume() + } + await settleReads() + await advancePolls(1_000) + + expect(metadataReadGate.reads).toBe(2) + }) + + it('keeps polling after a republish failure', async () => { + usePolledTimers() const userDataPath = makeUserDataPath() const republish = vi .fn() @@ -209,7 +360,7 @@ describe('watchRuntimeMetadataOwnership', () => { }) watches.push(watch) - vi.advanceTimersByTime(2_000) + await advancePolls(2_000) expect(republish).toHaveBeenCalledTimes(2) expect(readRuntimeMetadata(userDataPath)).toMatchObject({ pid: OWNED_PID }) diff --git a/src/main/runtime/runtime-metadata-ownership-watch.ts b/src/main/runtime/runtime-metadata-ownership-watch.ts index ccdb8bf8f29..a8da6e9eb5a 100644 --- a/src/main/runtime/runtime-metadata-ownership-watch.ts +++ b/src/main/runtime/runtime-metadata-ownership-watch.ts @@ -1,5 +1,5 @@ import { getRuntimeMetadataPath, type RuntimeMetadata } from '../../shared/runtime-bootstrap' -import { readRuntimeMetadata } from './runtime-metadata' +import { readRuntimeMetadataAsync } from './runtime-metadata' /** * Why: `orca-runtime.json` is the CLI's only pointer at a live runtime, and it @@ -18,7 +18,7 @@ export const RUNTIME_METADATA_OWNERSHIP_POLL_MS = 10_000 export type RuntimeMetadataOwnershipWatch = { /** Runs one ownership check immediately; exposed for tests and eager repair. */ - check: () => void + check: () => Promise stop: () => void } @@ -56,8 +56,9 @@ export function watchRuntimeMetadataOwnership( options: RuntimeMetadataOwnershipWatchOptions ): RuntimeMetadataOwnershipWatch { const isProcessRunning = options.isProcessRunning ?? isPidRunning - const check = (): void => { - const current = tryReadRuntimeMetadata(options.userDataPath) + let inFlight: Promise | null = null + const runCheck = async (): Promise => { + const current = await tryReadRuntimeMetadata(options.userDataPath) if ( !shouldReclaimRuntimeMetadata( current, @@ -77,8 +78,18 @@ export function watchRuntimeMetadataOwnership( } options.onReclaim?.(current) } + // Why: the read is off-thread now, so a slow volume could otherwise stack ticks on one file. + const check = (): Promise => { + inFlight ??= runCheck().finally(() => { + inFlight = null + }) + return inFlight + } - const timer = setInterval(check, options.pollIntervalMs ?? RUNTIME_METADATA_OWNERSHIP_POLL_MS) + const timer = setInterval( + () => void check(), + options.pollIntervalMs ?? RUNTIME_METADATA_OWNERSHIP_POLL_MS + ) // Why: discovery bookkeeping must never be the reason the process stays alive. timer.unref?.() return { @@ -87,9 +98,9 @@ export function watchRuntimeMetadataOwnership( } } -function tryReadRuntimeMetadata(userDataPath: string): RuntimeMetadata | null { +async function tryReadRuntimeMetadata(userDataPath: string): Promise { try { - return readRuntimeMetadata(userDataPath) + return await readRuntimeMetadataAsync(userDataPath) } catch (error) { // Why: an unparseable record is as useless to the CLI as a missing one, so treat it as reclaimable. console.warn( diff --git a/src/main/runtime/runtime-metadata.ts b/src/main/runtime/runtime-metadata.ts index bd43203252d..a4909d4723c 100644 --- a/src/main/runtime/runtime-metadata.ts +++ b/src/main/runtime/runtime-metadata.ts @@ -1,4 +1,5 @@ import { existsSync, readFileSync, rmSync } from 'node:fs' +import { readFile } from 'node:fs/promises' import { getRuntimeMetadataPath, type RuntimeMetadata } from '../../shared/runtime-bootstrap' import { writeSecureJsonFile } from '../../shared/secure-file' @@ -15,6 +16,22 @@ export function readRuntimeMetadata(userDataPath: string): RuntimeMetadata | nul return JSON.parse(readFileSync(metadataPath, 'utf-8')) as RuntimeMetadata } +/** Off-thread twin of {@link readRuntimeMetadata} for pollers; a missing file is not an error. */ +export async function readRuntimeMetadataAsync( + userDataPath: string +): Promise { + let raw: string + try { + raw = await readFile(getRuntimeMetadataPath(userDataPath), 'utf-8') + } catch (error) { + if ((error as NodeJS.ErrnoException).code === 'ENOENT') { + return null + } + throw error + } + return JSON.parse(raw) as RuntimeMetadata +} + export function clearRuntimeMetadata(userDataPath: string): void { rmSync(getRuntimeMetadataPath(userDataPath), { force: true }) } diff --git a/src/main/runtime/runtime-pty-controller-contract.ts b/src/main/runtime/runtime-pty-controller-contract.ts index 194d0bb665c..665c6fdb609 100644 --- a/src/main/runtime/runtime-pty-controller-contract.ts +++ b/src/main/runtime/runtime-pty-controller-contract.ts @@ -109,7 +109,7 @@ export type RuntimePtyController = { getForegroundProcess(ptyId: string): Promise inspectProcess?( ptyId: string, - options?: { expectedIncarnationId?: PtyIncarnationId } + options?: { expectedIncarnationId?: PtyIncarnationId; scanChildProcesses?: boolean } ): Promise confirmForegroundProcess?(ptyId: string): Promise confirmShellForeground?(ptyId: string): Promise diff --git a/src/main/runtime/runtime-rpc-metadata-lifecycle.test.ts b/src/main/runtime/runtime-rpc-metadata-lifecycle.test.ts index ac213cae535..4735601c6a9 100644 --- a/src/main/runtime/runtime-rpc-metadata-lifecycle.test.ts +++ b/src/main/runtime/runtime-rpc-metadata-lifecycle.test.ts @@ -7,6 +7,7 @@ import * as runtimeMetadataModule from './runtime-metadata' import { readRuntimeMetadata, writeRuntimeMetadata } from './runtime-metadata' import { createRuntimeTransportMetadata, OrcaRuntimeRpcServer } from './runtime-rpc' import type { DeviceRegistry } from './device-registry' +import type { RuntimeMetadata } from '../../shared/runtime-bootstrap' vi.mock('../git/worktree', () => { const worktrees = [ @@ -61,7 +62,7 @@ describe('OrcaRuntimeRpcServer', () => { authToken: 'second-instance-token', startedAt: 1 }) - server.checkRuntimeMetadataOwnership() + await server.checkRuntimeMetadataOwnership() expect(readRuntimeMetadata(userDataPath)).toEqual(published) @@ -86,7 +87,7 @@ describe('OrcaRuntimeRpcServer', () => { authToken: 'sibling-token', startedAt: 1 }) - server.checkRuntimeMetadataOwnership() + await server.checkRuntimeMetadataOwnership() expect(readRuntimeMetadata(userDataPath)).toMatchObject({ runtimeId: 'rt_live_sibling' }) @@ -112,13 +113,46 @@ describe('OrcaRuntimeRpcServer', () => { authToken: 'second-instance-token', startedAt: 1 }) - server.checkRuntimeMetadataOwnership() + await server.checkRuntimeMetadataOwnership() expect(watchStop).toHaveBeenCalledTimes(1) expect(server['metadataOwnershipWatch']).toBeNull() expect(readRuntimeMetadata(userDataPath)).toMatchObject({ runtimeId: 'rt_second_instance' }) }) + it('drops a republish from an ownership read that lands after the server stopped', async () => { + // Why: the read is off-thread now, so a tick can outlive stop(); the cleared + // activeTransports guard — not the interval teardown — is what stops it republishing. + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-rpc-')) + const runtime = new OrcaRuntimeService() + const server = new OrcaRuntimeRpcServer({ runtime, userDataPath, pid: 1001 }) + await server.start() + + let releaseRead: (record: RuntimeMetadata | null) => void = () => {} + vi.spyOn(runtimeMetadataModule, 'readRuntimeMetadataAsync').mockImplementationOnce( + () => + new Promise((resolve) => { + releaseRead = resolve + }) + ) + const heldCheck = server.checkRuntimeMetadataOwnership() + + await server.stop() + writeRuntimeMetadata(userDataPath, { + runtimeId: 'rt_second_instance', + pid: 99999999, + transports: [], + authToken: 'second-instance-token', + startedAt: 1 + }) + // A missing record is the most reclaimable verdict there is, so an unguarded + // resume would rewrite the file the second instance just published. + releaseRead(null) + await heldCheck + + expect(readRuntimeMetadata(userDataPath)).toMatchObject({ runtimeId: 'rt_second_instance' }) + }) + it('flushes a lastSeen refresh scheduled while transports stop', async () => { const server = new OrcaRuntimeRpcServer({ runtime: new OrcaRuntimeService(), diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-shutdown.ts b/src/main/runtime/runtime-rpc/runtime-rpc-shutdown.ts index d6edaf78928..f79834fbcc2 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-shutdown.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-shutdown.ts @@ -2,8 +2,8 @@ import { RuntimeRpcMobilePairing } from './runtime-rpc-mobile-pairing' export class RuntimeRpcShutdown extends RuntimeRpcMobilePairing { /** Why: test-only seam — runs one ownership check instead of waiting out the poll interval. */ - checkRuntimeMetadataOwnership(): void { - this.metadataOwnershipWatch?.check() + checkRuntimeMetadataOwnership(): Promise { + return this.metadataOwnershipWatch?.check() ?? Promise.resolve() } async stop(): Promise { diff --git a/src/main/runtime/terminal-leaf-tab-resolution.test.ts b/src/main/runtime/terminal-leaf-tab-resolution.test.ts new file mode 100644 index 00000000000..2fe8a6d0344 --- /dev/null +++ b/src/main/runtime/terminal-leaf-tab-resolution.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, it } from 'vitest' + +import type { TerminalLayoutSnapshot } from '../../shared/terminal-tab-types' +import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' +import { findTerminalTabIdForLeaf } from './workspace-session-terminal-membership-authority' + +function layout(...leafIds: string[]): TerminalLayoutSnapshot { + let root = { type: 'leaf' as const, leafId: leafIds[0] } + for (const leafId of leafIds.slice(1)) { + root = { + type: 'split', + direction: 'row', + first: root, + second: { type: 'leaf' as const, leafId } + } as never + } + return { root, activeLeafId: leafIds[0], ptyIdsByLeafId: {} } as TerminalLayoutSnapshot +} + +function session(layouts: Record): WorkspaceSessionState { + return { terminalLayoutsByTabId: layouts } as WorkspaceSessionState +} + +describe('findTerminalTabIdForLeaf', () => { + it('resolves every leaf of a split tree to its tab', () => { + const state = session({ + 'tab-a': layout('leaf-1', 'leaf-2', 'leaf-3'), + 'tab-b': layout('leaf-4') + }) + expect(findTerminalTabIdForLeaf(state, 'leaf-2')).toBe('tab-a') + expect(findTerminalTabIdForLeaf(state, 'leaf-3')).toBe('tab-a') + expect(findTerminalTabIdForLeaf(state, 'leaf-4')).toBe('tab-b') + }) + + it('keeps the first tab in record order when two layouts claim one leaf', () => { + const layouts = { 'tab-a': layout('shared'), 'tab-b': layout('shared') } + expect(findTerminalTabIdForLeaf(session(layouts), 'shared')).toBe('tab-a') + }) + + it('answers misses, empty sessions and empty layouts with undefined', () => { + expect(findTerminalTabIdForLeaf(undefined, 'leaf-1')).toBeUndefined() + expect(findTerminalTabIdForLeaf(session({}), 'leaf-1')).toBeUndefined() + expect(findTerminalTabIdForLeaf(session({ 'tab-a': layout('leaf-1') }), 'nope')).toBeUndefined() + }) + + // The guard a leafId -> tabId cache needed and this scan does not: membership is read from the + // tree that is there NOW. `persistPtyBinding` grafts leaves by assigning into a layout already in + // the record, so anything memoized across calls has to be revalidated against every mutation + // shape a writer can produce - including one that leaves the root node's identity untouched. + it('reflects a subtree replaced in place after an earlier read', () => { + const tracked = layout('leaf-1', 'leaf-2') + const state = session({ 'tab-a': tracked }) + expect(findTerminalTabIdForLeaf(state, 'leaf-2')).toBe('tab-a') + expect(findTerminalTabIdForLeaf(state, 'leaf-9')).toBeUndefined() + + const root = tracked.root as { second: unknown } + root.second = { type: 'leaf', leafId: 'leaf-9' } + + expect(findTerminalTabIdForLeaf(state, 'leaf-9')).toBe('tab-a') + expect(findTerminalTabIdForLeaf(state, 'leaf-2')).toBeUndefined() + }) +}) diff --git a/src/main/runtime/workspace-session-terminal-membership-authority.ts b/src/main/runtime/workspace-session-terminal-membership-authority.ts index 4c85f46b0c8..2869e3c813c 100644 --- a/src/main/runtime/workspace-session-terminal-membership-authority.ts +++ b/src/main/runtime/workspace-session-terminal-membership-authority.ts @@ -5,6 +5,7 @@ import type { } from '../../shared/terminal-tab-types' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { getRepoIdFromWorktreeId } from '../../shared/worktree/id' +import { layoutContainsLeafId } from '../persistence/restoring-sessions/terminal-layout-normalization' import { pruneTabGroupLayoutAfterRetirement } from './mobile-session-terminal-retirement' function collectLeafIds(node: TerminalPaneLayoutNode | null, ids: Set): void { @@ -164,15 +165,22 @@ export function advanceTerminalTopologyRevision( * The tab whose live layout holds this leaf. Only the leaf half of a pane key is remint-stable — * `detachTerminalPaneToTab` moves a live pane into a new tab, so a stored tabId names the tab the * pane left. Callers fencing on location must resolve it here rather than trust a frozen tabId. + * + * Stateless on purpose: writers graft leaves by assigning into a layout that is already inside the + * layouts record, so any cache here would need a revalidation key that is itself O(tabs) per read — + * the same cost as this walk, with a staleness invariant to keep. `Object.keys` over a guarded + * `for...in` is deliberate too: the key array is cheaper than a `hasOwn` call per tab (measured). */ export function findTerminalTabIdForLeaf( session: WorkspaceSessionState | undefined, leafId: string ): string | undefined { - for (const [tabId, layout] of Object.entries(session?.terminalLayoutsByTabId ?? {})) { - const leafIds = new Set() - collectLeafIds(layout.root, leafIds) - if (leafIds.has(leafId)) { + const layouts = session?.terminalLayoutsByTabId + if (!layouts) { + return undefined + } + for (const tabId of Object.keys(layouts)) { + if (layoutContainsLeafId(layouts[tabId]?.root ?? null, leafId)) { return tabId } } diff --git a/src/main/runtime/worktree-launch-host-repo.ts b/src/main/runtime/worktree-launch-host-repo.ts index 4decb7acb56..7db9f4dae18 100644 --- a/src/main/runtime/worktree-launch-host-repo.ts +++ b/src/main/runtime/worktree-launch-host-repo.ts @@ -17,7 +17,10 @@ export type WorktreeHostRouting = | { kind: 'resolved'; hostId: ExecutionHostId; repo: T | null } /** No row carries this repo id and the worktree names no host — nothing ever named a host. */ | { kind: 'unowned' } - /** Rival rows disagree about the host; guessing one is the cross-host leak. */ + /** + * No single trustworthy host: rival rows disagree, or the resolved row named one that cannot be + * parsed. Guessing is the cross-host leak in both cases. + */ | { kind: 'ambiguous' } /** @@ -33,7 +36,10 @@ export function resolveWorktreeHostRouting { const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) if (resolution.kind === 'unresolved') { - return resolution.reason === 'ambiguous' ? { kind: 'ambiguous' } : { kind: 'unowned' } + // Only `unknown` — nothing anywhere carries the id — becomes `unowned`, which callers dispose of + // as a plain local folder. `malformed` is a row that declared a host and named an unparseable + // one, so it joins `ambiguous`: guessing is the cross-host leak either way. + return resolution.reason === 'unknown' ? { kind: 'unowned' } : { kind: 'ambiguous' } } return { kind: 'resolved', hostId: resolution.hostId, repo: resolution.owner } } diff --git a/src/main/runtime/worktree-listing-host-scope.ts b/src/main/runtime/worktree-listing-host-scope.ts index 99650882670..e499d7faaf3 100644 --- a/src/main/runtime/worktree-listing-host-scope.ts +++ b/src/main/runtime/worktree-listing-host-scope.ts @@ -1,4 +1,5 @@ -import type { ExecutionHostId } from '../../shared/execution-host' +import type { Repo } from '../../shared/repo-types' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' import { selectHostBalancedPage } from '../../shared/host-balanced-listing-page' import type { RuntimeListingHostScope } from '../../shared/runtime-listing-host-scope' @@ -62,3 +63,28 @@ export function buildWorktreeListingHostScope(args: { } return { hostIds: [...covered].sort(), omittedHostIds: [...omitted].sort() } } + +/** + * Which hosts a listing claims to have been looking at. + * + * A `--repo` listing was scoped by the caller, so naming every configured host would report gaps + * the caller deliberately excluded. Naming NONE — which is what a scoped listing did before — means + * the scope can never report a gap at all, for any host kind, because `covered` and `omitted` are + * both derived from the returned rows plus this list. A scoped listing whose scan did not succeed + * then answers `{hostIds: [], omittedHostIds: []}`: byte-identical to a repo that genuinely has no + * worktrees, which is the one thing docs/reference/ssh-execution-boundary.md forbids a listing from + * implying. + * + * Measured before the fix, on one runtime with one refusing SSH host, in the same second: the + * unscoped listing reported `omittedHostIds: ["local", "ssh:"]` while the scoped listing + * reported `[]`. + * + * Naming the single host the caller asked about costs nothing when rows come back — it lands in + * `covered`, so it is never reported omitted — and is the whole answer when they do not. + */ +export function listingKnownHostIds( + scopedRepo: Repo | null, + listKnownHostIds: () => Iterable +): Iterable { + return scopedRepo ? [getRepoExecutionHostId(scopedRepo)] : listKnownHostIds() +} diff --git a/src/main/ssh/relay-daemon-service-children.ts b/src/main/ssh/relay-daemon-service-children.ts new file mode 100644 index 00000000000..35e755ea518 --- /dev/null +++ b/src/main/ssh/relay-daemon-service-children.ts @@ -0,0 +1,62 @@ +/** + * Telling a relay daemon's own service processes apart from the work it holds. + * + * The reap gate used to ask `pgrep -P | grep -c .` and demand zero. But the daemon + * forks service children of its own — `relay-ai-vault-service.js` is spawned lazily and then + * never exits — so that count is permanently non-zero on any relay that has touched the AI + * Vault, whether or not it holds a single PTY. A superseded, disconnected relay holding + * nothing therefore reported `retained-live-work` forever, its version directory stayed + * pinned against GC by its own live socket, and the population grew without bound (#13614). + * + * The asymmetry below is the whole safety argument, and it follows + * docs/reference/ssh-execution-boundary.md: *subtracting a child we can positively identify + * as relay infrastructure is sound; assuming anything about a child we cannot identify is + * not.* An argv that does not match, an argv `ps` would not print, and a host without + * `pgrep` all count against the relay and keep it unreapable. Losing sight of a child is + * never evidence that it holds nothing. + */ +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' +import { shellEscape } from './ssh-connection-utils' + +/** Shell variable set to the daemon's direct-child count, or `unknown`. */ +export const RELAY_CHILD_COUNT_VAR = 'kids' + +/** Shell variable set to the count of children not identified as relay services, or `unknown`. */ +export const RELAY_UNRECOGNIZED_CHILD_COUNT_VAR = 'unrecognized_kids' + +/** + * `case` patterns matching a service child's argv. Suffix-anchored on purpose: both entries + * are forked with no script arguments, so the argv ends at the filename, and the leading `/` + * requires the absolute path the daemon forks rather than a bare mention of the name. A + * future arg would stop matching and the relay would go back to being retained — the safe + * direction to fail in. + */ +function serviceChildArgvPatterns(): string { + return RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.map( + (filename) => `*${shellEscape(`/${filename}`)}` + ).join('|') +} + +/** + * POSIX shell that censuses the direct children of `$pid`, setting `kids` and + * `unrecognized_kids`. Both stay `unknown` when the host cannot enumerate children at all. + */ +export function relayDaemonChildCensusShell(): string[] { + return [ + `${RELAY_CHILD_COUNT_VAR}=unknown`, + `${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=unknown`, + 'if command -v pgrep >/dev/null 2>&1; then', + ` ${RELAY_CHILD_COUNT_VAR}=0`, + ` ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=0`, + ' for kid in $(pgrep -P "$pid" 2>/dev/null); do', + ` ${RELAY_CHILD_COUNT_VAR}=$((${RELAY_CHILD_COUNT_VAR}+1))`, + ' kid_args=$(ps -o args= -p "$kid" 2>/dev/null | tr -d "\\n")', + ' case "$kid_args" in', + ` ${serviceChildArgvPatterns()}) ;;`, + // An unreadable or unrecognised argv lands here, which is what keeps the relay retained. + ` *) ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=$((${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}+1)) ;;`, + ' esac', + ' done', + 'fi' + ] +} diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index 257eb984caa..e7450478d92 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -743,6 +743,7 @@ function uploadStageNamespaceIfSupported( const NODE_PTY_VERSION = '1.1.0' const NODE_PTY_CONSOLE_LIST_PATCH_FILENAME = 'node-pty-1.1.0-console-list-agent-patch.cjs' +const NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME = 'node-pty-1.1.0-windows-pty-teardown-patch.cjs' const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' const NODE_PTY_CLOEXEC_STATUS_PREFIX = 'ORCA-NPTY-CLOEXEC:' /** @@ -791,7 +792,8 @@ function nativeDepsProbeJs(successToken: string): string { // Why: node-pty's Windows wrapper defers conpty.node until first spawn, so require("node-pty") alone can't prove the binding is healthy. const loadNodePty = 'require("node-pty"); require("node-pty/lib/utils").loadNativeModule(process.platform==="win32"&&Number(require("os").release().split(".")[2])>=18309?"conpty":"pty");' + - `if(process.platform==="win32"){require("./${NODE_PTY_CONSOLE_LIST_PATCH_FILENAME}").assertPatchedNodePtyConsoleListAgent(process.cwd())}` + `if(process.platform==="win32"){require("./${NODE_PTY_CONSOLE_LIST_PATCH_FILENAME}").assertPatchedNodePtyConsoleListAgent(process.cwd());` + + `require("./${NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME}").assertPatchedNodePtyWindowsTeardown(process.cwd())}` return `(()=>{const missing=[];try{${loadNodePty}}catch{missing.push("node-pty")}try{require("@parcel/watcher")}catch{missing.push("@parcel/watcher")}if(missing.length){console.log("${NATIVE_DEPS_MISSING_PREFIX}"+missing.join(","));process.exitCode=1}else{console.log(${JSON.stringify(successToken)})}})()` } @@ -1327,10 +1329,16 @@ async function applyNodePtyMasterCloexecPatch( nodePath: string, signal?: AbortSignal ): Promise { - // Both Unix relay platforms leak, by different bugs: Linux inherits the master through forkpty()'s - // no-O_CLOEXEC path, macOS orphans one throwaway /dev/ptmx fd per spawn in pty_posix_spawn. Only - // Windows, which has no fds, is short-circuited -- and answering 'fixed' from a gate that ran - // nothing is exactly how a leaking darwin tree got published to the shared cache. + // Both Unix relay platforms leak the pty master, by different bugs: Linux inherits it through + // forkpty()'s no-O_CLOEXEC path, macOS orphans one throwaway /dev/ptmx fd per spawn in + // pty_posix_spawn. Windows is short-circuited because it has no fds for a master to leak into -- + // and answering 'fixed' from a gate that ran nothing is exactly how a leaking darwin tree got + // published to the shared cache. + // + // What 'fixed' means here is exactly "this tree does not leak the pty MASTER", which is the only + // thing the shared native-deps cache keys on. It is NOT a statement that a Windows relay leaks + // nothing: it leaked one Windows File handle per terminal until the ConPTY teardown patch above, + // by a mechanism that has nothing to do with fds. Read this gate as scoped to its own question. if (isWindowsRemoteHost(hostPlatform) || isWindowsRelayPlatform(platform)) { return 'fixed' } @@ -1530,8 +1538,11 @@ async function rebuildNativeDeps( } function windowsNodePtyPatchCommand(nodePath: string): string { - // Why: pnpm patches do not cross the SSH boundary; apply the version-checked fallback to the remote npm package. - return `& ${powerShellLiteral(nodePath)} ${powerShellLiteral(NODE_PTY_CONSOLE_LIST_PATCH_FILENAME)}; if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }` + // Why: pnpm patches do not cross the SSH boundary; apply the version-checked fallbacks to the remote npm package. + return [ + `& ${powerShellLiteral(nodePath)} ${powerShellLiteral(NODE_PTY_CONSOLE_LIST_PATCH_FILENAME)}; if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }`, + `& ${powerShellLiteral(nodePath)} ${powerShellLiteral(NODE_PTY_WINDOWS_TEARDOWN_PATCH_FILENAME)}; if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }` + ].join('; ') } async function makeNodePtySpawnHelperExecutable( diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts index 7ece0b6532e..a8975d0520b 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts @@ -4,7 +4,7 @@ * generated scripts through /bin/sh against real unix sockets and real processes. */ import { execFile, spawn, type ChildProcess } from 'node:child_process' -import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterAll, afterEach, beforeAll, describe, expect, it } from 'vitest' @@ -15,21 +15,32 @@ import { type RelayEndpointIncumbent } from './ssh-relay-endpoint-incumbent' import { reapEmptyRelayHuskCommand } from './ssh-relay-endpoint-takeover' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' const posixOnly = process.platform === 'win32' ? describe.skip : describe const FAKE_RELAY_SOURCE = ` const net = require('net') +const path = require('path') const sock = process.argv[process.argv.indexOf('--sock-path') + 1] +function spawnChild(args) { + require('child_process').spawn(process.execPath, args, { stdio: 'ignore' }) +} if (process.argv.includes('--with-child')) { - require('child_process').spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000)'], { - stdio: 'ignore' - }) + spawnChild(['-e', 'setTimeout(() => {}, 60000)']) +} +// Why forked the same way production does: the exclusion is argv-shaped, so a hand-written +// stand-in would test the test rather than the shell that runs on someone's host. +for (const name of process.argv.filter((arg) => arg.startsWith('--service-child='))) { + spawnChild([path.join(__dirname, name.slice('--service-child='.length))]) } net.createServer(() => {}).listen(sock, () => process.stdout.write('READY\\n')) process.on('SIGTERM', () => process.exit(0)) ` +// Self-limiting: these are orphaned when the relay under test is reaped. +const IDLE_SERVICE_SOURCE = 'setTimeout(() => {}, 60000)\n' + function sh(script: string): Promise { return new Promise((resolve, reject) => { execFile('/bin/sh', ['-c', script], { timeout: 20_000 }, (error, stdout) => { @@ -43,14 +54,21 @@ function sh(script: string): Promise { } let workDir: string +let pgreplessBinDir: string let hasLsof = false const running: ChildProcess[] = [] -function startFakeRelay(sockPath: string, withChild = false): Promise { +function startFakeRelay( + sockPath: string, + options: { withChild?: boolean; serviceChildren?: readonly string[] } = {} +): Promise { const args = [join(workDir, 'relay.js'), '--sock-path', sockPath] - if (withChild) { + if (options.withChild) { args.push('--with-child') } + for (const name of options.serviceChildren ?? []) { + args.push(`--service-child=${name}`) + } const child = spawn(process.execPath, args, { stdio: ['ignore', 'pipe', 'ignore'] }) running.push(child) return new Promise((resolve, reject) => { @@ -68,9 +86,31 @@ async function probe(sockPath: string): Promise { return parseRelayEndpointIncumbentProbe(sockPath, output) } +/** The relay forks its children after it starts listening, so the probe can race them. */ +async function waitForChildCount( + sockPath: string, + expected: number +): Promise { + let incumbent = await probe(sockPath) + for (let attempt = 0; attempt < 50 && incumbent.holders[0]?.childCount !== expected; attempt++) { + await new Promise((resolve) => setTimeout(resolve, 100)) + incumbent = await probe(sockPath) + } + return incumbent +} + beforeAll(async () => { workDir = mkdtempSync(join(tmpdir(), 'orca-relay-incumbent-')) writeFileSync(join(workDir, 'relay.js'), FAKE_RELAY_SOURCE) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + writeFileSync(join(workDir, filename), IDLE_SERVICE_SOURCE) + } + writeFileSync(join(workDir, 'looks-like-relay-watcher.js'), IDLE_SERVICE_SOURCE) + pgreplessBinDir = join(workDir, 'pgrepless-bin') + mkdirSync(pgreplessBinDir) + for (const tool of ['ps', 'tr']) { + symlinkSync((await sh(`command -v ${tool}`)).trim(), join(pgreplessBinDir, tool)) + } hasLsof = await sh('command -v lsof >/dev/null 2>&1 && echo yes || echo no').then( (out) => out.trim() === 'yes' ) @@ -105,13 +145,51 @@ posixOnly('relay endpoint probe against a real socket', () => { return } expect(incumbent.holders.map((holder) => holder.pid)).toEqual([relay.pid]) - expect(incumbent.holders[0]).toMatchObject({ matchesRelayArgv: true, childCount: 0 }) + expect(incumbent.holders[0]).toMatchObject({ + matchesRelayArgv: true, + childCount: 0, + unrecognizedChildCount: 0 + }) expect(isReapableRelayHusk(incumbent)).toBe(true) }) + it("counts the daemon's own service children but does not hold them against it", async () => { + const sockPath = join(workDir, 'services.sock') + await startFakeRelay(sockPath, { serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES }) + const incumbent = await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + + expect(incumbent.holders[0].childCount).toBe(RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + expect(incumbent.holders[0].unrecognizedChildCount).toBe(0) + expect(isReapableRelayHusk(incumbent)).toBe(true) + }) + + it('still retains a relay holding work alongside its service children', async () => { + const sockPath = join(workDir, 'services-and-work.sock') + await startFakeRelay(sockPath, { + withChild: true, + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + const incumbent = await waitForChildCount( + sockPath, + RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length + 1 + ) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + + it('does not excuse a child that merely mentions a service entry name', async () => { + const sockPath = join(workDir, 'lookalike.sock') + await startFakeRelay(sockPath, { serviceChildren: ['looks-like-relay-watcher.js'] }) + const incumbent = await waitForChildCount(sockPath, 1) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + it('refuses to call a relay with a live child an empty husk', async () => { const sockPath = join(workDir, 'busy.sock') - await startFakeRelay(sockPath, true) + await startFakeRelay(sockPath, { withChild: true }) const incumbent = await probe(sockPath) expect(incumbent.verdict).toBe('live') @@ -150,12 +228,34 @@ posixOnly('empty relay husk reap against a real process', () => { it('refuses to signal a relay that acquired a child after it was probed', async () => { const sockPath = join(workDir, 'raced.sock') - const relay = await startFakeRelay(sockPath, true) + const relay = await startFakeRelay(sockPath, { withChild: true }) const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) expect(output.trim()).toBe('BUSY') expect(relay.killed).toBe(false) }) + it('terminates a relay whose only children are its own service processes (#13614)', async () => { + const sockPath = join(workDir, 'service-husk.sock') + const relay = await startFakeRelay(sockPath, { + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) + expect(output.trim()).toBe('GONE') + }) + + it('refuses to signal when the host cannot enumerate children at all', async () => { + const sockPath = join(workDir, 'no-pgrep.sock') + const relay = await startFakeRelay(sockPath) + // A PATH carrying every tool the script needs except `pgrep`: the census answers + // `unknown`, which must reach BUSY rather than the zero a missing tool would imply. + const output = await sh( + `PATH=${pgreplessBinDir}\n${reapEmptyRelayHuskCommand(relay.pid!, sockPath)}` + ) + expect(output.trim()).toBe('BUSY') + expect(relay.killed).toBe(false) + }) + it('refuses to signal a pid whose argv is not this relay at this socket', async () => { const sockPath = join(workDir, 'mismatch.sock') await startFakeRelay(sockPath) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts index a65cb33fc57..de4cc28d170 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts @@ -32,17 +32,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('reports live when the socket accepted a connection', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=4242 yes 13']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=4242 yes 13 11' + ]) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('accepted-connection') - expect(incumbent.holders).toEqual([{ pid: 4242, matchesRelayArgv: true, childCount: 13 }]) + expect(incumbent.holders).toEqual([ + { pid: 4242, matchesRelayArgv: true, childCount: 13, unrecognizedChildCount: 11 } + ]) }) it('reports live when a process still holds an inode that refuses connections', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2']) + probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2 2']) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('holder-process') @@ -85,7 +92,12 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('drops holder lines that do not carry a usable pid', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=- no unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=refused', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=- no unknown unknown' + ]) ) expect(incumbent.holders).toEqual([]) expect(incumbent.verdict).toBe('exited') @@ -94,9 +106,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('keeps an unreadable child count as null rather than zero', () => { const [holder] = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=7 yes unknown unknown' + ]) ).holders expect(holder.childCount).toBeNull() + expect(holder.unrecognizedChildCount).toBeNull() + }) + + it('keeps a holder line with no unrecognized-child field unreapable', () => { + const incumbent = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes 0']) + ) + expect(incumbent.holders[0].unrecognizedChildCount).toBeNull() + expect(isReapableRelayHusk(incumbent)).toBe(false) }) }) @@ -172,27 +199,36 @@ describe('mayLaunchOverRelayEndpoint', () => { describe('isReapableRelayHusk', () => { const husk = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0']) + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0 0']) ) - it('accepts a single proven relay holder with zero children', () => { + it('accepts a single proven relay holder with no unaccounted-for children', () => { expect(isReapableRelayHusk(husk)).toBe(true) }) - it('refuses a relay that still holds children', () => { + it('accepts a relay whose only children are its own service processes (#13614)', () => { + const withServices = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 2 0']) + ) + expect(withServices.holders[0].childCount).toBe(2) + expect(isReapableRelayHusk(withServices)).toBe(true) + }) + + it('refuses a relay that still holds children it could not account for', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: 1 }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 3, unrecognizedChildCount: 1 }] }) ).toBe(false) }) - it('refuses a holder whose child count could not be read', () => { + it('refuses a holder whose unrecognized-child count could not be read', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: null }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: null }] }) ).toBe(false) }) @@ -201,7 +237,7 @@ describe('isReapableRelayHusk', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0 }] + holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0, unrecognizedChildCount: 0 }] }) ).toBe(false) }) @@ -211,8 +247,8 @@ describe('isReapableRelayHusk', () => { isReapableRelayHusk({ ...husk, holders: [ - { pid: 500, matchesRelayArgv: true, childCount: 0 }, - { pid: 501, matchesRelayArgv: true, childCount: 0 } + { pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 }, + { pid: 501, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 } ] }) ).toBe(false) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.ts index 2688267f4c7..628a9558793 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.ts @@ -20,6 +20,11 @@ */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_CHILD_COUNT_VAR, + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { isWindowsRemoteHost, type RemoteHostPlatform } from './ssh-remote-platform' @@ -38,6 +43,12 @@ export type RelayEndpointHolder = { matchesRelayArgv: boolean /** Direct children, or null when `pgrep` could not answer. Never guessed. */ childCount: number | null + /** + * Direct children *not* positively identified as the daemon's own service processes, or + * null when the host could not enumerate them. This — not `childCount` — is what says + * whether the relay holds anything; see relay-daemon-service-children.ts. + */ + unrecognizedChildCount: number | null } export type RelayEndpointIncumbent = { @@ -95,11 +106,9 @@ export function relayEndpointIncumbentProbeCommand(nodePath: string, sockPath: s ' args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', ' match=no', ' case "$args" in *relay.js*"$sock"*) match=yes ;; esac', - ' kids=unknown', - ' if command -v pgrep >/dev/null 2>&1; then', - ' kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - ' fi', - ' printf \'HOLDER=%s %s %s\\n\' "$pid" "$match" "$kids"', + ...relayDaemonChildCensusShell().map((line) => ` ${line}`), + ' printf \'HOLDER=%s %s %s %s\\n\' "$pid" "$match" ' + + `"$${RELAY_CHILD_COUNT_VAR}" "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}"`, ' done', 'else', " printf 'HOLDERS_SOURCE=unavailable\\n'", @@ -159,19 +168,25 @@ export function parseRelayEndpointIncumbentProbe( } function parseHolder(value: string): RelayEndpointHolder | null { - const [rawPid, rawMatch, rawKids] = value.split(/\s+/) + const [rawPid, rawMatch, rawKids, rawUnrecognized] = value.split(/\s+/) const pid = Number.parseInt(rawPid ?? '', 10) if (!Number.isInteger(pid) || pid <= 0) { return null } - const childCount = Number.parseInt(rawKids ?? '', 10) return { pid, matchesRelayArgv: rawMatch === 'yes', - childCount: Number.isInteger(childCount) && childCount >= 0 ? childCount : null + childCount: parseChildCount(rawKids), + unrecognizedChildCount: parseChildCount(rawUnrecognized) } } +/** `unknown`, a missing field, and anything unparseable are all "could not tell" — never 0. */ +function parseChildCount(raw: string | undefined): number | null { + const count = Number.parseInt(raw ?? '', 10) + return Number.isInteger(count) && count >= 0 ? count : null +} + function unverifiableEndpoint(sockPath: string): RelayEndpointIncumbent { return { sockPath, @@ -239,8 +254,12 @@ export function mayLaunchOverRelayEndpoint(incumbent: RelayEndpointIncumbent): b /** * A live relay that provably holds nothing: identity confirmed against its argv, exactly one - * holder, and zero children. Reaping it destroys no user work. Anything less is retained — - * killing the wrong pid on someone's remote host is the worst outcome available here. + * holder, and no child the host could not account for as one of the daemon's own service + * processes. Reaping it destroys no user work. Anything less is retained — killing the wrong + * pid on someone's remote host is the worst outcome available here. + * + * Why not `childCount === 0`: the daemon's AI Vault sidecar never exits once spawned, so that + * gate was unreachable for any relay that had ever served a vault request (#13614). */ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean { if (incumbent.verdict !== 'live' || !incumbent.holdersEnumerable) { @@ -250,12 +269,16 @@ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean return false } const [holder] = incumbent.holders - return holder.matchesRelayArgv && holder.childCount === 0 + return holder.matchesRelayArgv && holder.unrecognizedChildCount === 0 } export function describeRelayEndpointIncumbent(incumbent: RelayEndpointIncumbent): string { const holders = incumbent.holders - .map((holder) => `${holder.pid}(children=${holder.childCount ?? 'unknown'})`) + .map( + (holder) => + `${holder.pid}(children=${holder.childCount ?? 'unknown'},` + + `unrecognized=${holder.unrecognizedChildCount ?? 'unknown'})` + ) .join(',') return ( `${incumbent.sockPath} verdict=${incumbent.verdict} evidence=${incumbent.evidence} ` + diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts index 687d633b92b..d23f065f478 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts @@ -14,6 +14,7 @@ import { resolveRelayEndpointBeforeRelaunch } from './ssh-relay-endpoint-takeover' import { RelayVersionMismatchError } from './ssh-relay-version-mismatch-error' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' import type { SshConnection } from './ssh-connection' import { getRemoteHostPlatform } from './ssh-remote-platform' @@ -42,7 +43,7 @@ beforeEach(() => { describe('incumbent alive and refusing', () => { it('refuses to rebind a live relay holding PTYs, and signals nothing', async () => { execCommand.mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) // The whole point of #8585: the incumbent's socket must survive so it is not orphaned. @@ -52,9 +53,9 @@ describe('incumbent alive and refusing', () => { it('names the incumbent pid and the Reset Relay escape hatch in the error', async () => { execCommand.mockResolvedValue( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) - await expect(resolve()).rejects.toThrow(/3669803\(children=13\)/) + await expect(resolve()).rejects.toThrow(/3669803\(children=13,unrecognized=11\)/) await expect(resolve()).rejects.toThrow(/Reset Relay/) }) @@ -70,7 +71,7 @@ describe('incumbent alive and refusing', () => { it('reaps a live relay only when it provably holds nothing, and confirms it is gone', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') await expect(resolve()).resolves.toMatchObject({ verdict: 'live' }) @@ -80,7 +81,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over an empty relay whose death could not be confirmed', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -89,7 +90,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over a relay the host refused to signal on its own re-check', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('BUSY\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -138,9 +139,19 @@ describe('reapEmptyRelayHuskCommand', () => { }) it('aborts without signalling when the host cannot count children', () => { - expect(reapEmptyRelayHuskCommand(4242, SOCK)).toContain( - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }" - ) + const command = reapEmptyRelayHuskCommand(4242, SOCK) + // The census leaves both counters at `unknown` without pgrep, and the gate demands "0". + expect(command).toContain('unrecognized_kids=unknown') + expect(command).toContain('command -v pgrep >/dev/null 2>&1') + expect(command).toContain('[ "$unrecognized_kids" = "0" ] ||') + }) + + it('subtracts only the daemon service children it can name from the reap gate', () => { + const command = reapEmptyRelayHuskCommand(4242, SOCK) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + expect(command).toContain(`*'/${filename}'`) + } + expect(command).toContain('unrecognized_kids=$((unrecognized_kids+1))') }) }) diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.ts b/src/main/ssh/ssh-relay-endpoint-takeover.ts index f104aab5256..8f6130620cb 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.ts @@ -2,13 +2,17 @@ * Deciding whether a relay socket path is ours to take, and acting on the answer. * * The only destructive action available here is a SIGTERM to a relay that has been proven — - * by argv, by socket-holder enumeration, and by a zero child count re-checked on the host - * immediately before the signal — to hold nothing at all. Everything else is left running. + * by argv, by socket-holder enumeration, and by a child census re-run on the host immediately + * before the signal — to hold nothing at all. Everything else is left running. * Per docs/reference/ssh-execution-boundary.md, a relay we merely failed to reach is * `unverifiable`, and `unverifiable` never authorizes a kill or a rebind. */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { describeRelayEndpointIncumbent, @@ -39,9 +43,10 @@ export function reapEmptyRelayHuskCommand(pid: number, sockPath: string): string `sock=${shellEscape(sockPath)}`, 'args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', 'case "$args" in *relay.js*"$sock"*) ;; *) printf \'MISMATCH\\n\'; exit 0 ;; esac', - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }", - 'kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - '[ "$kids" = "0" ] || { printf \'BUSY\\n\'; exit 0; }', + // Why the same census as the probe: `unknown` (no pgrep) and any child this host could + // not account for as a relay service both land on BUSY, so nothing is signalled. + ...relayDaemonChildCensusShell(), + `[ "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}" = "0" ] || { printf 'BUSY\\n'; exit 0; }`, // SIGTERM only: the relay's own handler disposes and unlinks. SIGKILL would leave the // socket inode behind and skip that shutdown path for no gain on an empty daemon. 'kill -TERM "$pid" 2>/dev/null || true', diff --git a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts index 874d9aae3fe..9168f4688bd 100644 --- a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts +++ b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts @@ -67,7 +67,7 @@ describe('classifySupersededRelay', () => { 'PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', - 'HOLDER=3669803 yes 13' + 'HOLDER=3669803 yes 13 11' ]) ) ).toBe('retained-live-work') @@ -76,7 +76,7 @@ describe('classifySupersededRelay', () => { it('nominates only a proven empty relay for reaping', () => { expect( classifySupersededRelay( - incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) ).toBe('reap-candidate') }) @@ -101,7 +101,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) expect(findings).toHaveLength(1) @@ -114,7 +114,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) @@ -126,7 +126,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) diff --git a/src/main/ssh/ssh-remote-powershell.ts b/src/main/ssh/ssh-remote-powershell.ts index 8c94fd3c483..420223ced29 100644 --- a/src/main/ssh/ssh-remote-powershell.ts +++ b/src/main/ssh/ssh-remote-powershell.ts @@ -11,14 +11,24 @@ export { // to leave room for the `/c` wrapper sshd adds before cmd.exe counts the line. const WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS = 8_000 -export function powerShellCommand(script: string): string { - const inline = encodedPowerShellCommand(script) +/** + * `pwsh.exe` is PowerShell 7. It is not present on a stock Windows install, so it is only ever + * chosen after a probe — but where it exists it reads a redirected stdin correctly, which Windows + * PowerShell 5.1 does not (see `system-ssh-file-binary-transfer.ts`). + */ +export type WindowsPowerShellExecutable = 'powershell.exe' | 'pwsh.exe' + +export function powerShellCommand( + script: string, + executable: WindowsPowerShellExecutable = 'powershell.exe' +): string { + const inline = encodedPowerShellCommand(script, executable) if (inline.length <= WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { return inline } // Why: these scripts are repetitive enough that gzip beats the UTF-16LE tax by // ~4x, which is the difference between a line cmd.exe runs and one it refuses. - const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script)) + const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script), executable) if (compressed.length > WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { throw new Error( `Remote Windows command needs ${compressed.length} characters; Orca budgets ${WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS} for a line sshd hands to cmd.exe, which itself refuses more than ${CMD_EXE_COMMAND_LINE_MAX_CHARS}.` @@ -27,8 +37,8 @@ export function powerShellCommand(script: string): string { return compressed } -function encodedPowerShellCommand(script: string): string { - return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` +function encodedPowerShellCommand(script: string, executable: WindowsPowerShellExecutable): string { + return `${executable} -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` } /** Orca-prefixed names so the payload can never shadow the bootstrap's own state. */ diff --git a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts index c104078238e..fa34a8fbda7 100644 --- a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts +++ b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts @@ -4,6 +4,10 @@ import { CMD_EXE_COMMAND_LINE_MAX_CHARS } from '../providers/windows-shell-args' import { getRemoteHostPlatform } from './ssh-remote-platform' import { tryStealInstallLockCommand } from './ssh-relay-install-lock-commands' import { decodeRemotePowerShellScript, powerShellCommand } from './ssh-remote-powershell' +import { + makeWindowsPublishStagedFileCommand, + makeWindowsWriteFileCommand +} from './system-ssh-windows-file-write' import { cleanupOwnedRelayUploadStageCommand, promoteOwnedRelayUploadStageCommand, @@ -38,6 +42,17 @@ describe('Windows remote command line limit', () => { [ 'steal stale install lock', tryStealInstallLockCommand(windows, 'C:\\Users\\orca\\.orca-remote\\relay', 1_200) + ], + // F11 flagged these two as uncovered. They carry one path literal each, so they are the file + // commands whose length a caller can actually move. + ['write file', makeWindowsWriteFileCommand('C:\\Users\\orca\\.orca-remote\\relay.js')], + [ + 'publish staged file', + makeWindowsPublishStagedFileCommand( + 'C:\\Users\\orca\\.orca-remote\\relay.js.orca-partial-0123456789ab', + 'C:\\Users\\orca\\.orca-remote\\relay.js', + 'create' + ) ] ])('keeps the %s command inside what sshd\u2019s cmd.exe accepts', (_name, command) => { expect(command.length).toBeLessThanOrEqual(CMD_EXE_COMMAND_LINE_MAX_CHARS) @@ -76,3 +91,32 @@ describe('Windows remote command line limit', () => { ) }) }) + +/** + * F11 asked whether a pathological path could reach the budget, and what happens if it does. + * Measured: the inline encoding crosses 8000 at roughly 2500 high-entropy path characters — an + * order of magnitude past what Windows itself accepts — and the failure is a throw before any ssh + * is spawned, never a hang. + */ +describe('Windows file command budget headroom', () => { + it('absorbs a path far longer than Windows will accept', () => { + const deep = `C:\\Users\\orca\\${'segment\\'.repeat(30)}relay.js` + + expect(deep.length).toBeGreaterThan(260) + expect(makeWindowsWriteFileCommand(deep).length).toBeLessThanOrEqual( + CMD_EXE_COMMAND_LINE_MAX_CHARS + ) + }) + + it('throws rather than spawning a line cmd.exe would refuse', () => { + // Random segments so gzip cannot rescue it, which is the only way to reach the ceiling at all. + const incompressible = Array.from( + { length: 400 }, + (_unused, index) => `${index}-${Math.random().toString(36).slice(2)}` + ).join('\\') + + expect(() => makeWindowsWriteFileCommand(`C:\\${incompressible}\\f.bin`)).toThrow( + /Orca budgets 8000/ + ) + }) +}) diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index c366e899bf6..b477ad682ef 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -709,9 +709,13 @@ describe('spawnSystemSsh', () => { expect(args[standaloneControlIdx + 1]).toBe('none') }) - it('writes files to Windows system SSH targets with PowerShell stdin bytes', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('sends Windows file writes over sftp, not through a remote PowerShell stdin', async () => { + const spawned: EventedProcess[] = [] + spawnMock.mockImplementation(() => { + const proc = createEventedProcess() + spawned.push(proc) + return closeOnceSpawned(proc) + }) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -722,16 +726,24 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('0.1.0', 'utf-8')) + // #16432, re-measured: Windows PowerShell 5.1 can lose a redirected stdin for good when a read + // finds it momentarily empty, so the bytes must not travel that way at all. + const batch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(batch).toContain('put ') + expect(batch).toContain('/C:/Users/me/.orca-remote/relay/.version.orca-partial-') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + expect(sftpArgs).toContain('-b') + // The rename that publishes it reads the staged file, never a pipe. + const publish = (spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '' + expect(publish).toContain('powershell.exe') + expect(decodePowerShellCommand(publish)).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + expect(publish).not.toContain('/bin/sh') }) - it('writes binary buffers to Windows system SSH targets with CreateNew mode', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('enforces an exclusive Windows buffer write at the rename, where it is atomic', async () => { + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeBufferViaSystemSsh( @@ -742,12 +754,11 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(decodePowerShellCommand(remoteCommand)).toContain('CreateNew') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('png')) + const publish = decodePowerShellCommand((spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '') + // `File::Move` raising on an existing destination is what carries the exclusive contract now; + // a `CreateNew` on the staged file would only refuse a leftover of our own. + expect(publish).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish).not.toContain('[System.IO.File]::Delete($path)') }) it('downloads files from Windows system SSH targets with PowerShell stdout bytes', async () => { @@ -779,8 +790,7 @@ describe('spawnSystemSsh', () => { }) it('forces standalone SSH for Windows file writes when requested', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -791,10 +801,14 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + // sftp's own `-S` names a program to run, so the same request has to be spelled as an option. + expect(sftpArgs).not.toContain('-S') + expect(sftpArgs).toContain('ControlPath=none') + const publishArgs = spawnMock.mock.calls[1][1] as string[] + const standaloneControlIdx = publishArgs.indexOf('-S') expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + expect(publishArgs[standaloneControlIdx + 1]).toBe('none') }) it('uploads a Windows directory as a mkdir batch plus per-file writes, never one blob', async () => { @@ -819,19 +833,20 @@ describe('spawnSystemSsh', () => { rmSync(localDir, { recursive: true, force: true }) } + // #16432: directories first, then the file — but both over sftp now, so the only PowerShell + // left is the rename that publishes the staged file, which reads a file rather than a pipe. + const mkdirBatch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(mkdirBatch).toBe('-mkdir "/C:/Users/me/.orca-remote/relay"\n') + const putBatch = String(spawned[1]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(putBatch).toContain('put ') + expect(putBatch).toContain('/C:/Users/me/.orca-remote/relay/relay.js.orca-partial-') const commands = spawnMock.mock.calls.map((call) => (call[1] as string[]).at(-1) ?? '') - // #16432: directories first (metadata only), then the file bytes on their own stdin. One batch - // meant base64-ing the whole bundle into a single PowerShell string, which the remote never read. - expect(commands).toHaveLength(2) - expect(commands.every((command) => command.includes('powershell.exe'))).toBe(true) expect(commands.every((command) => !command.includes('/bin/sh'))).toBe(true) expect(commands.join('\n')).not.toContain('tar -xzf') - expect(JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string)).toEqual([ - 'C:/Users/me/.orca-remote/relay' - ]) - expect(Buffer.from(spawned[1].stdin.end.mock.calls[0]?.[0] as Buffer).toString('utf-8')).toBe( - 'console.log("relay")' - ) + // Nothing base64s the bundle into one PowerShell string any more, and nothing reads one. + expect( + commands.some((command) => decodePowerShellCommand(command).includes('OpenStandardInput')) + ).toBe(false) }) it('forces standalone SSH for Windows upload packages when requested', async () => { @@ -855,9 +870,9 @@ describe('spawnSystemSsh', () => { } const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') - expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + // The first spawn is the sftp client, whose own `-S` names a program to run. + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') }) it('throws when no system ssh is found', () => { diff --git a/src/main/ssh/system-ssh-file-binary-transfer.ts b/src/main/ssh/system-ssh-file-binary-transfer.ts index b0c5b662ed1..149d10dfbd3 100644 --- a/src/main/ssh/system-ssh-file-binary-transfer.ts +++ b/src/main/ssh/system-ssh-file-binary-transfer.ts @@ -1,5 +1,7 @@ import { constants, createWriteStream } from 'node:fs' -import { lstat, open } from 'node:fs/promises' +import { lstat, mkdtemp, open, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import type { Writable } from 'node:stream' import { pipeline } from 'node:stream/promises' import type { SshTarget } from '../../shared/ssh-types' @@ -16,6 +18,16 @@ import { throwIfAborted, waitForChannelClose } from './system-ssh-operation-lifecycle' +import { + writeWindowsRemoteFile, + type WindowsWriteSource +} from './system-ssh-windows-write-strategy' + +export { + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS +} from './system-ssh-windows-write-strategy' +export { WINDOWS_STAGED_WRITE_SUFFIX } from './system-ssh-windows-file-write' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -74,13 +86,16 @@ export async function writeBufferViaSystemSsh( ): Promise { throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - await writeWindowsBytesViaSystemSsh( + await writeWindowsRemoteFile( target, remotePath, - contents.length, - (offset, maxBytes) => - Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), - options + { + totalBytes: contents.length, + readChunk: (offset, maxBytes) => + Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), + withLocalFile: (send) => withTemporaryLocalFile(contents, send) + }, + options ?? {} ) return } @@ -127,20 +142,19 @@ export async function uploadFileViaSystemSsh( throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - // #16432: a Windows host cannot take a whole file through one stdin, however the local side - // paces it — see WINDOWS_STDIN_WRITE_CHUNK_BYTES. This is the path that carries the large - // files, so it is the one that has to be chunked and bounded. - await writeWindowsBytesViaSystemSsh( - target, - remotePath, - openedStat.size, - async (offset, maxBytes) => { + // This is the path that carries the large files, so it is the one the transport choice is + // made for; see the #16432 note below. + const source: WindowsWriteSource = { + totalBytes: openedStat.size, + readChunk: async (offset, maxBytes) => { const buffer = Buffer.allocUnsafe(Math.min(maxBytes, openedStat.size - offset)) const { bytesRead } = await handle.read(buffer, 0, buffer.length, offset) return buffer.subarray(0, bytesRead) }, - options - ) + // The verified local file is already exactly the payload, so sftp sends it as is. + withLocalFile: (send) => send(localPath) + } + await writeWindowsRemoteFile(target, remotePath, source, options ?? {}) return } @@ -173,158 +187,56 @@ export async function uploadFileViaSystemSsh( } /** - * #16432: Windows PowerShell 5.1 stops draining a redirected stdin over a non-pty ssh exec - * somewhere between 50KB and 1MB, depending on the host's `DefaultShell`, and it hangs rather than - * failing. The reporter measured that on both constructs he tried — `[Console]::In.ReadToEnd()` and - * `new IO.StreamReader([Console]::OpenStandardInput())`, the latter reading incrementally, which is - * why the limit cannot be attributed to materializing the payload. `Stream.CopyTo` reads the same - * `[Console]::OpenStandardInput()` object with the same incremental `Read` loop, so nothing in it - * escapes that limit either: no single write may exceed what one stdin is known to carry. + * #16432, re-measured: the constraint is not a size limit, and it is not cmd.exe's. * - * 32KB is an order of magnitude under the low end of the measured range, and under 50KB, which the - * reporter measured succeeding against a stream reader on the worse of the two `DefaultShell` - * settings. - */ -export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 - -/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ -export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 - -/** Suffix for the path a multi-exec Windows write lands on before it is published by rename. */ -export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' - -/** - * Splits one logical Windows write into stdin-sized execs. + * A read on Windows PowerShell 5.1's redirected-stdin handle over a non-pty ssh exec can die + * permanently when it finds the stream momentarily empty: no further bytes arrive, and no EOF ever + * does. It is probabilistic per such read — not a size threshold, and not certain on the first one. + * Measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2 with `DefaultShell = cmd.exe`, by + * replacing the copy loop with a counting reader: * - * A write that needs more than one exec cannot land on the destination directly: a chunk failing - * mid-file would leave a truncated artifact under the real name with nothing marking it incomplete, - * and the retry would then meet its own leftovers — under `exclusive` the retry's `CreateNew` fails - * on them. Multi-exec creates therefore land on a staging path and are published by a rename, which - * is also where `exclusive` is enforced: once, at the destination, instead of smeared across the - * first chunk. A caller-requested append cannot be staged without reading the remote file back, so - * it keeps writing straight through, as its own protocol already implies. + * - a 1.5s gap before any byte, which forces the first read to find nothing -> 0 bytes, 6 of 6 + * - one byte, a 1.5s gap, then 32767 more -> exactly 1 byte, then nothing + * - 32768, a 1.5s gap, then 32768 more -> exactly 32768, then nothing + * - a continuous 2MB -> 167936 / 270336 / 372736, then nothing + * + * Those three 2MB death points are one payload run three times under the same conditions, which is + * what rules out a threshold: a stream that died at a fixed point would not vary by 2x. Independently reproduced by + * a second harness, where one 1.9MB counted read survived 39 reads to completion and another died + * after 11 — same construct, same payload. + * + * A payload small enough to arrive in one burst usually presents only one read that can find the + * stream empty (the one waiting for EOF), which is why 32KB mostly works: it still failed 15 times + * in 120 with the host under load, and 1 in 40 on a quiet one. Neither rate is survivable across + * the 62 execs a 1.9MB file needs — even 2.5% compounds to roughly four uploads in five failing — + * and no chunk size helps, because the client does not control whether its bytes arrive together. + * + * The same host, same `DefaultShell`, same connection pattern contradicts every size-limit reading: + * `findstr` took 2,016,000 bytes through one exec's stdin, and PowerShell 7 took 2MB. So cmd.exe is + * not the ceiling and neither is ~50KB. Writes now go over sftp, which moves the whole payload + * without any remote process reading a pipe; see `system-ssh-windows-write-strategy.ts` for the + * fallback order. + * + * Successes are never partial. Across every run in both harnesses a failed write hung; not one + * produced a short file, so this defect cannot silently truncate an upload. */ -async function writeWindowsBytesViaSystemSsh( - target: SshTarget, - remotePath: string, - totalBytes: number, - readChunk: (offset: number, maxBytes: number) => Promise, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const staged = !options.append && totalBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES - const writePath = staged ? `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` : remotePath - let offset = 0 - // An empty write still has to run: it is what creates (or truncates) the file. - do { - const chunk = await readChunk(offset, WINDOWS_STDIN_WRITE_CHUNK_BYTES) - if (chunk.length === 0 && offset < totalBytes) { - throw new Error(`Source ran short during upload of ${remotePath}`) - } - await writeWindowsChunkViaSystemSsh( - target, - writePath, - chunk, - { - ...options, - append: staged ? offset > 0 : options.append === true || offset > 0, - exclusive: staged ? false : options.exclusive === true && offset === 0 - }, - offset - ) - offset += chunk.length - } while (offset < totalBytes) - if (staged) { - await publishWindowsStagedWrite(target, writePath, remotePath, options) + +/** A staged write is materialized locally first when the source is a buffer rather than a file. */ +async function withTemporaryLocalFile( + contents: Buffer, + send: (localPath: string) => Promise +): Promise { + const directory = await mkdtemp(join(tmpdir(), 'orca-win-upload-')) + const localPath = join(directory, 'payload.bin') + try { + // 0600: the payload can be repository content, and tmpdir is shared on every platform. + await writeFile(localPath, contents, { mode: 0o600 }) + return await send(localPath) + } finally { + await rm(directory, { recursive: true, force: true }).catch(() => {}) } } -async function writeWindowsChunkViaSystemSsh( - target: SshTarget, - remotePath: string, - chunk: Buffer, - options: SystemSshWriteBufferOptions, - offset: number -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsWriteFileCommand(remotePath, options), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose( - channel, - `write ${remotePath} at offset ${offset}`, - WINDOWS_STDIN_WRITE_TIMEOUT_MS - ) - ) - if (!options.signal?.aborted) { - channel.stdin.end(chunk) - } - await closePromise -} - -async function publishWindowsStagedWrite( - target: SshTarget, - stagingPath: string, - remotePath: string, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand( - target, - makeWindowsPublishStagedFileCommand(stagingPath, remotePath, options.exclusive === true), - { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } - ) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, `publish ${remotePath}`, WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ) - if (!options.signal?.aborted) { - channel.stdin.end() - } - await closePromise -} - -function makeWindowsWriteFileCommand( - remotePath: string, - options?: { append?: boolean; exclusive?: boolean } -): string { - const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$path = ${powerShellLiteral(remotePath)}`, - '$parent = [System.IO.Path]::GetDirectoryName($path)', - 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', - '$inputStream = [Console]::OpenStandardInput()', - `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, - 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' - ].join('; ') - ) -} - -// `File::Move` throws when the destination exists, which is exactly the exclusive contract; the -// non-exclusive caller asked to replace, so it deletes first (a no-op on an absent path). -function makeWindowsPublishStagedFileCommand( - stagingPath: string, - remotePath: string, - exclusive: boolean -): string { - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$staging = ${powerShellLiteral(stagingPath)}`, - `$path = ${powerShellLiteral(remotePath)}`, - ...(exclusive ? [] : ['[System.IO.File]::Delete($path)']), - '[System.IO.File]::Move($staging, $path)' - ].join('; ') - ) -} - function makePosixWriteFileCommand( remotePath: string, options?: { append?: boolean; exclusive?: boolean } diff --git a/src/main/ssh/system-ssh-file-transfer.ts b/src/main/ssh/system-ssh-file-transfer.ts index f728c0eab02..d5757904356 100644 --- a/src/main/ssh/system-ssh-file-transfer.ts +++ b/src/main/ssh/system-ssh-file-transfer.ts @@ -27,6 +27,12 @@ import { WINDOWS_STDIN_WRITE_TIMEOUT_MS, writeBufferViaSystemSsh } from './system-ssh-file-binary-transfer' +import { + isSftpPathUnsupportedError, + isSftpUnavailableError, + makeDirectoriesViaSftp +} from './system-ssh-sftp-transfer' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -161,9 +167,14 @@ async function collectWindowsUploadPlan( return plan } -// Why the JSON envelope survives here: a path list is metadata, so this payload stays in the -// hundreds of bytes even for a deep tree. Batched anyway, so a pathological tree cannot walk back -// into the same stdin size that wedges PowerShell. +/** + * Creates the upload's directories, preferring sftp's own `mkdir`. + * + * The PowerShell fallback keeps the JSON envelope, batched under one stdin's worth: a path list is + * metadata, so it stays in the hundreds of bytes even for a deep tree. It is still a redirected + * stdin read though, so on Windows PowerShell 5.1 it carries the same defect as any other — which + * is why sftp is tried first even for a payload this small. + */ async function createWindowsUploadDirectories( target: SshTarget, directories: readonly string[], @@ -175,23 +186,27 @@ async function createWindowsUploadDirectories( if (batch.length === 0) { return } + const pending = batch const payload = JSON.stringify(batch) batch = [] batchBytes = 0 throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + await getWindowsRemoteWriteCapabilities(target).runWithFallback( + 'sftp-subsystem', + async () => { + try { + await makeDirectoriesViaSftp(target, pending, options) + } catch (error) { + // A directory sftp cannot address is this batch's problem, not the host's verdict. + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await createWindowsUploadDirectoriesViaPowerShell(target, payload, options) + } + }, + () => createWindowsUploadDirectoriesViaPowerShell(target, payload, options), + isSftpUnavailableError ) - if (!options.signal?.aborted) { - channel.stdin.end(payload) - } - await closePromise } for (const directory of directories) { const entryBytes = Buffer.byteLength(directory) + 4 @@ -204,17 +219,44 @@ async function createWindowsUploadDirectories( await flush() } +async function createWindowsUploadDirectoriesViaPowerShell( + target: SshTarget, + payload: string, + options: SystemSshOperationOptions +): Promise { + const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end(payload) + } + await closePromise +} + function makeWindowsCreateDirectoriesCommand(): string { return powerShellCommand( [ '$ErrorActionPreference = "Stop"', - // The reporter measured this reader surviving 50KB where `[Console]::In` wedged at the same - // size (#16432); the batch above stays under that. + // Reached only where the host has no sftp subsystem. Windows PowerShell 5.1 can lose a + // redirected stdin for good when a read finds it empty (#16432); a batch this small usually + // arrives in one piece, and "usually" is exactly why sftp is preferred. '$reader = New-Object System.IO.StreamReader([Console]::OpenStandardInput())', 'try { $json = $reader.ReadToEnd() } finally { $reader.Dispose() }', 'if ([string]::IsNullOrWhiteSpace($json)) { return }', - 'foreach ($path in @($json | ConvertFrom-Json)) {', - ' $null = [System.IO.Directory]::CreateDirectory([string]$path)', + // `[string[]]`, not `@(...)`: ConvertFrom-Json emits the parsed array as a single pipeline + // object, so `@(...)` wraps it in *another* array and the loop variable binds to the whole + // thing. `[string]` of that is the paths joined by spaces, which CreateDirectory rejects with + // "The given path's format is not supported". It only ever worked for a one-element batch, + // where stringifying a single-element array happens to yield the element. Measured on + // WindowsPowerShell 5.1.26100 against a three-directory tree. + 'foreach ($path in [string[]]($json | ConvertFrom-Json)) {', + ' $null = [System.IO.Directory]::CreateDirectory($path)', '}' ].join('; ') ) diff --git a/src/main/ssh/system-ssh-sftp-args.test.ts b/src/main/ssh/system-ssh-sftp-args.test.ts new file mode 100644 index 00000000000..d971390f8dd --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.test.ts @@ -0,0 +1,141 @@ +/** + * `buildSshArgs` is shared with the sftp client, and three of its flags mean something else there. + * Every case below is a silent wrong-target rather than an error if the translation is skipped, + * which is why the fallback is "refuse and use another transport", never "pass it through". + */ +import { describe, expect, it } from 'vitest' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' + +describe('translateSshArgsToSftpArgs', () => { + it('sends the port as an option, since sftp -p preserves mtimes instead', () => { + const args = translateSshArgsToSftpArgs(['-p', '2222', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'Port=2222', '--', 'dev@win.example']) + }) + + it('sends the login name as an option, since sftp has no -l', () => { + // `buildSshArgs` emits `-l` for a config alias no Host block claims. Throwing here would send + // exactly those hosts to the transport this PR exists to stop using, silently. + const args = translateSshArgsToSftpArgs(['-l', 'neil', '--', 'awin']) + + expect(args).toEqual(['-o', 'User=neil', '--', 'awin']) + }) + + it('translates the whole unclaimed-alias shape buildSshArgs emits', () => { + const args = translateSshArgsToSftpArgs([ + '-o', + 'BatchMode=no', + '-T', + '-S', + 'none', + '-o', + 'Hostname=192.168.0.186', + '-p', + '2222', + '-l', + 'neil', + '--', + 'awin' + ]) + + expect(args).toEqual([ + '-o', + 'BatchMode=no', + '-o', + 'ControlPath=none', + '-o', + 'Hostname=192.168.0.186', + '-o', + 'Port=2222', + '-o', + 'User=neil', + '--', + 'awin' + ]) + }) + + it('spells ControlPath=none out, since sftp -S names a program to run', () => { + // `sftp -S none` would try to exec a binary called `none`. + const args = translateSshArgsToSftpArgs(['-S', 'none', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'ControlPath=none', '--', 'dev@win.example']) + }) + + it('refuses any other -S, which would hand sftp an ssh binary Orca did not choose', () => { + expect(() => translateSshArgsToSftpArgs(['-S', '/tmp/ctl.sock'])).toThrow( + SftpArgTranslationError + ) + }) + + it('drops -T, which sftp does not have', () => { + expect(translateSshArgsToSftpArgs(['-T', '--', 'host'])).toEqual(['--', 'host']) + }) + + it('passes through the flags both clients spell the same way', () => { + const args = translateSshArgsToSftpArgs([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + + expect(args).toEqual([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + }) + + it('takes everything after -- as the destination without reinterpreting it', () => { + // A host literally named `-p` is not a flag once `--` has been seen. + expect(translateSshArgsToSftpArgs(['--', '-p'])).toEqual(['--', '-p']) + }) + + it('refuses an unknown flag rather than guessing what sftp would do with it', () => { + // The point of the throw: a flag added to buildSshArgs later must degrade to another + // transport, not reach sftp carrying a different meaning. + expect(() => translateSshArgsToSftpArgs(['-A', '--', 'host'])).toThrow(SftpArgTranslationError) + }) + + it('refuses a value flag with no value', () => { + expect(() => translateSshArgsToSftpArgs(['-o'])).toThrow(SftpArgTranslationError) + }) +}) + +describe('withSftpKeepalive', () => { + it('asks OpenSSH to notice a dead peer, since the transfer itself has no wall-clock bound', () => { + expect(withSftpKeepalive(['--', 'host'])).toEqual([ + '-o', + 'ServerAliveInterval=15', + '-o', + 'ServerAliveCountMax=3', + '--', + 'host' + ]) + }) + + it('leaves a caller-stated keepalive policy alone', () => { + const args = withSftpKeepalive(['-o', 'ServerAliveInterval=60', '--', 'host']) + + expect(args.filter((arg) => arg.startsWith('ServerAliveInterval'))).toEqual([ + 'ServerAliveInterval=60' + ]) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-args.ts b/src/main/ssh/system-ssh-sftp-args.ts new file mode 100644 index 00000000000..17fa37e78bb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.ts @@ -0,0 +1,95 @@ +/** + * Rewrites `buildSshArgs` output for the sftp(1) client. + * + * Three flags ssh and sftp share spell different things: sftp's `-p` is "preserve mtime", its `-S` + * names the ssh binary to run, and it has no `-T` at all. Passing ssh's list through unchanged + * would silently connect to the wrong port and try to exec a program called `none`. + * + * Anything this table does not recognize throws. A flag added to `buildSshArgs` later must degrade + * to the non-sftp transfer path, never reach sftp carrying a different meaning. + */ + +/** `buildSshArgs` emitted a flag with no sftp equivalent; the caller should use another transport. */ +export class SftpArgTranslationError extends Error { + constructor(flag: string) { + super(`No sftp equivalent for system ssh argument ${JSON.stringify(flag)}`) + this.name = 'SftpArgTranslationError' + } +} + +/** Flags whose spelling and meaning are identical in both clients. */ +const PASSTHROUGH_VALUE_FLAGS = new Set(['-F', '-o', '-i', '-J']) + +export function translateSshArgsToSftpArgs(sshArgs: readonly string[]): string[] { + const sftpArgs: string[] = [] + let index = 0 + while (index < sshArgs.length) { + const flag = sshArgs[index]! + if (flag === '--') { + // Everything after `--` is the destination, which both clients spell the same way. + sftpArgs.push(...sshArgs.slice(index)) + return sftpArgs + } + const value = sshArgs[index + 1] + if (PASSTHROUGH_VALUE_FLAGS.has(flag)) { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push(flag, value) + index += 2 + continue + } + if (flag === '-T') { + // sftp never allocates a tty, so ssh's "no tty" request has nothing to translate to. + index += 1 + continue + } + if (flag === '-p') { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `Port=${value}`) + index += 2 + continue + } + if (flag === '-l') { + // sftp has no `-l`; the login name is an option there. `buildSshArgs` emits this for an + // unclaimed config alias, so throwing would route those hosts down the defective path and + // then cache the refusal against them for half an hour. + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `User=${value}`) + index += 2 + continue + } + if (flag === '-S') { + // ssh's `-S none` is ControlPath=none; sftp's `-S` would run a binary called `none`. + if (value !== 'none') { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', 'ControlPath=none') + index += 2 + continue + } + throw new SftpArgTranslationError(flag) + } + return sftpArgs +} + +/** + * A transfer that stalls mid-stream has no per-write bound to catch it, so ask OpenSSH to notice a + * dead peer itself. Only added when the caller has not already stated a keepalive policy. + */ +export function withSftpKeepalive(sftpArgs: readonly string[]): string[] { + const hasOption = (name: string): boolean => + sftpArgs.some((arg, position) => sftpArgs[position - 1] === '-o' && arg.startsWith(`${name}=`)) + const keepalive: string[] = [] + if (!hasOption('ServerAliveInterval')) { + keepalive.push('-o', 'ServerAliveInterval=15') + } + if (!hasOption('ServerAliveCountMax')) { + keepalive.push('-o', 'ServerAliveCountMax=3') + } + return [...keepalive, ...sftpArgs] +} diff --git a/src/main/ssh/system-ssh-sftp-path.test.ts b/src/main/ssh/system-ssh-sftp-path.test.ts new file mode 100644 index 00000000000..e196c2e3e8f --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.test.ts @@ -0,0 +1,59 @@ +/** + * Both functions here guard against the same measured failure: sftp's batch lexer treats `\` as an + * escape, so a Windows path handed over raw is silently mis-targeted *and the client still exits + * 0*. On Windows 11 / OpenSSH 10.0p2, `put src C:\Users\neil\qt\a.bin` created a file literally + * named `C` in the start directory and reported success. + */ +import { describe, expect, it } from 'vitest' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' + +describe('toSftpRemotePath', () => { + it('roots a drive path under /, which is the namespace the Windows sftp-server exposes', () => { + // `pwd` in that session reports `/C:/Users/dev`. + expect(toSftpRemotePath('C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('accepts a path already in that namespace unchanged', () => { + expect(toSftpRemotePath('/C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('converts the separators Orca stores paths with', () => { + expect(toSftpRemotePath('C:\\Users\\dev\\f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('declines a UNC path rather than guessing where it lands', () => { + // A guess here writes real bytes to the wrong place; declining falls back to another transport. + expect(() => toSftpRemotePath('//server/share/f.bin')).toThrow(UnsupportedSftpPathError) + }) + + it('declines a relative path, which would resolve against the session start directory', () => { + expect(() => toSftpRemotePath('Users/dev/f.bin')).toThrow(UnsupportedSftpPathError) + }) +}) + +describe('quoteSftpBatchArgument', () => { + it('escapes the backslashes in a Windows client local path', () => { + // Unescaped, sftp reads this as C:srcf.bin and fails to find the source. + expect(quoteSftpBatchArgument('C:\\src\\f.bin')).toBe('"C:\\\\src\\\\f.bin"') + }) + + it('keeps a path with spaces as one argument', () => { + expect(quoteSftpBatchArgument('/tmp/two words.bin')).toBe('"/tmp/two words.bin"') + }) + + it('escapes an embedded quote, which would otherwise end the argument early', () => { + expect(quoteSftpBatchArgument('/tmp/dq".bin')).toBe('"/tmp/dq\\".bin"') + }) + + it('refuses a line break, which would split one batch command into two', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\nrm -rf b')).toThrow(UnsupportedSftpPathError) + }) + + it('refuses a NUL, which truncates the argument', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\0b')).toThrow(UnsupportedSftpPathError) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-path.ts b/src/main/ssh/system-ssh-sftp-path.ts new file mode 100644 index 00000000000..2b5bfe53f02 --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.ts @@ -0,0 +1,46 @@ +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * A path this transfer cannot express to sftp. Callers treat it as "use another transport", never + * as a transfer failure. + */ +export class UnsupportedSftpPathError extends Error { + constructor(path: string) { + super(`Path cannot be addressed over sftp: ${JSON.stringify(path)}`) + this.name = 'UnsupportedSftpPathError' + } +} + +/** + * Converts a Windows remote path to the namespace OpenSSH's Windows sftp-server exposes, which + * roots every drive under `/`: `C:/Users/dev/f` is `/C:/Users/dev/f`, and `pwd` there reports + * `/C:/Users/dev`. + */ +export function toSftpRemotePath(remotePath: string): string { + const normalized = normalizeWindowsRemotePath(remotePath) + if (/^\/[a-zA-Z]:\//.test(normalized)) { + return normalized + } + if (/^[a-zA-Z]:\//.test(normalized)) { + return `/${normalized}` + } + // UNC (`//server/share`) and relative paths have no settled mapping in this namespace, and a + // guess here writes real bytes to the wrong place. Decline instead. + throw new UnsupportedSftpPathError(remotePath) +} + +/** + * Quotes one argument of an sftp batch line. + * + * Escaping is load-bearing, not cosmetic: sftp's batch lexer treats `\` as an escape even inside + * double quotes, so an unescaped Windows local path `C:\src\f.bin` is read as `C:srcf.bin`, and an + * unescaped destination `C:\Users\dev\f.bin` writes a file literally named `C` in the start + * directory — while sftp still exits 0. Both measured on Windows 11 / OpenSSH 10.0p2. + */ +export function quoteSftpBatchArgument(value: string): string { + if (/[\n\r\0]/.test(value)) { + // A line break would split one batch command into two; NUL truncates the argument. + throw new UnsupportedSftpPathError(value) + } + return `"${value.replace(/([\\"])/g, '\\$1')}"` +} diff --git a/src/main/ssh/system-ssh-sftp-transfer.ts b/src/main/ssh/system-ssh-sftp-transfer.ts new file mode 100644 index 00000000000..c50375fa8eb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-transfer.ts @@ -0,0 +1,191 @@ +import { accessSync, constants, existsSync, statSync } from 'node:fs' +import { posix, win32 } from 'node:path' +import type { SshTarget } from '../../shared/ssh-types' +import { buildSshArgs, type SystemSshBuildArgsOptions } from './system-ssh-args' +import { findSystemSsh } from './system-ssh-binary' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' +import { throwIfAborted } from './system-ssh-operation-lifecycle' +import { runProcess } from '../../shared/child-process/run-process' + +/** The host answered, but not with an sftp subsystem. The caller must fall back, not fail. */ +export class SftpSubsystemUnavailableError extends Error { + constructor(detail: string) { + super(`Remote host has no usable sftp subsystem: ${detail}`) + this.name = 'SftpSubsystemUnavailableError' + } +} + +/** + * True for the errors that mean "this host cannot serve sftp at all". + * + * Host-scoped, and therefore the only errors safe to remember: a capability cache keyed by host + * turns anything it accepts into a verdict about every later write to that host. Deliberately + * narrow — a permission denial or a missing directory is a real failure that must surface, not a + * reason to retry the whole upload down a slower path. + */ +export function isSftpUnavailableError(error: unknown): boolean { + return error instanceof SftpSubsystemUnavailableError || error instanceof SftpArgTranslationError +} + +/** + * True when *this path* cannot be spelled for sftp, which says nothing about the host. + * + * Kept apart from the host verdict on purpose. A UNC destination, or a local file whose name + * contains a newline, is a property of one operation; caching it would degrade every subsequent + * write to that host for the cache's whole retry window on the strength of one odd filename. + */ +export function isSftpPathUnsupportedError(error: unknown): boolean { + return error instanceof UnsupportedSftpPathError +} + +/** Neither kind of refusal moves a byte, so a staged file cannot exist to sweep. */ +export function isSftpRefusalBeforeStaging(error: unknown): boolean { + return isSftpUnavailableError(error) || isSftpPathUnsupportedError(error) +} + +function systemSftpCandidates(sshPath: string | null, platform: NodeJS.Platform): string[] { + const pathApi = platform === 'win32' ? win32 : posix + const executable = platform === 'win32' ? 'sftp.exe' : 'sftp' + const candidates: string[] = [] + // Why the ssh binary's own directory first: a host with two OpenSSH installs must pair the sftp + // client with the ssh that `buildSshArgs` was built for, not whichever one PATH happens to reach. + if (sshPath) { + candidates.push(pathApi.join(pathApi.dirname(sshPath), executable)) + } + if (platform === 'win32') { + const systemRoot = process.env.SystemRoot || process.env.WINDIR + if (systemRoot) { + candidates.push(win32.join(systemRoot, 'System32', 'OpenSSH', executable)) + } + } else { + candidates.push('/usr/bin/sftp', '/usr/local/bin/sftp', '/opt/homebrew/bin/sftp') + } + return candidates +} + +/** Locate the sftp client paired with the system ssh binary. Returns null when there is none. */ +export function findSystemSftp(): string | null { + if (process.env.ORCA_SYSTEM_SFTP_PATH) { + return process.env.ORCA_SYSTEM_SFTP_PATH + } + const sshPath = findSystemSsh() + for (const candidate of systemSftpCandidates(sshPath, process.platform)) { + try { + if (!statSync(candidate).isFile()) { + continue + } + if (process.platform !== 'win32') { + accessSync(candidate, constants.X_OK) + } + return candidate + } catch { + continue + } + } + return findSftpOnPath() +} + +function findSftpOnPath(): string | null { + const pathValue = process.env.PATH + if (!pathValue) { + return null + } + const pathApi = process.platform === 'win32' ? win32 : posix + const executable = process.platform === 'win32' ? 'sftp.exe' : 'sftp' + for (const entry of pathValue.split(pathApi.delimiter)) { + const directory = entry.trim().replace(/^"|"$/g, '') + if (!directory) { + continue + } + const candidate = pathApi.join(directory, executable) + if (existsSync(candidate)) { + return candidate + } + } + return null +} + +/** + * OpenSSH prints this when the server refuses the subsystem — a host with `Subsystem sftp` + * commented out, or an internal-sftp block that does not apply to this user. + */ +const SUBSYSTEM_REFUSED_PATTERN = /subsystem request failed|no such file or directory.*sftp-server/i + +export type SftpBatchOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal } + +/** + * Runs one sftp batch script. + * + * The script goes to the *local* sftp client's stdin, which is the point: no remote process ever + * reads a redirected stdin, so none of this rides the Windows PowerShell stdin defect. + */ +export async function runSftpBatch( + target: SshTarget, + commands: readonly string[], + options?: SftpBatchOptions +): Promise { + throwIfAborted(options?.signal) + const sftpPath = findSystemSftp() + if (!sftpPath) { + throw new SftpSubsystemUnavailableError('no sftp client binary found alongside ssh') + } + const args = withSftpKeepalive(translateSshArgsToSftpArgs(buildSshArgs(target, options))) + let result + try { + result = await runProcess({ + program: sftpPath, + args: ['-b', '-', ...args], + // `-b -` takes the script on stdin, and that stdin is the *local* client's — no remote + // process reads a pipe anywhere in this transfer, which is the whole point of preferring it. + input: `${commands.join('\n')}\n`, + // Why no timeout: a large upload is legitimately slow, and a wall-clock cap would fail a + // healthy transfer on a slow link. A dead peer is caught by the ServerAlive options instead. + timeoutMs: null, + signal: options?.signal + }) + } catch (error) { + // A client that will not start is "this host cannot do sftp" from the caller's side, not a + // transfer failure: the payload never left. Falling back is the only useful answer. + throw new SftpSubsystemUnavailableError( + `sftp client at ${sftpPath} could not be started: ${error instanceof Error ? error.message : String(error)}` + ) + } + if (result.code === 0) { + return + } + throwIfAborted(options?.signal) + const detail = result.stderr.trim() + if (SUBSYSTEM_REFUSED_PATTERN.test(detail)) { + throw new SftpSubsystemUnavailableError(detail) + } + throw new Error(`sftp batch failed (exit ${result.code}): ${detail}`) +} + +/** + * Creates remote directories, parents first. + * + * `-mkdir` keeps sftp going when a directory is already there; batch mode otherwise aborts the + * whole script on the first non-zero status, which for an idempotent tree walk is not a failure. + */ +export function makeDirectoriesViaSftp( + target: SshTarget, + remoteDirectories: readonly string[], + options?: SftpBatchOptions +): Promise { + const commands = remoteDirectories.map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + if (commands.length === 0) { + return Promise.resolve() + } + return runSftpBatch(target, commands, options) +} diff --git a/src/main/ssh/system-ssh-windows-file-write.ts b/src/main/ssh/system-ssh-windows-file-write.ts new file mode 100644 index 00000000000..8b98d4aec50 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-file-write.ts @@ -0,0 +1,138 @@ +import { randomBytes } from 'node:crypto' +import { powerShellCommand, powerShellLiteral } from './ssh-remote-powershell' +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * Suffix marking the path a Windows write lands on before it is published by rename. + * + * The random tail is the fix for a measured harm, not decoration. A write that loses contact with + * the host leaves a remote process that may still hold the staging file open exclusively, and + * `docs/reference/ssh-execution-boundary.md` is explicit that losing contact is not evidence that + * process died — so the retry must not reuse the name it may still own. A fresh name per attempt + * means a retry never meets its predecessor's lock; the abandoned file is cleaned up best-effort + * and never treated as proof of anything. + */ +export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' + +export function makeWindowsStagingPath(remotePath: string): string { + return `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}-${randomBytes(6).toString('hex')}` +} + +export type WindowsPublishMode = 'create' | 'exclusive' | 'append' + +/** + * Publishes a staged upload onto its real name. + * + * Every branch reads the staged *file*, never a redirected stdin, which is what makes this safe on + * a host whose Windows PowerShell 5.1 cannot drain a piped stdin. + * + * The replacing branch must never delete the destination first. Deleting and then moving loses the + * user's existing file outright if the move fails, and exposes a window where a reader sees no file + * at all — a worse outcome than the truncated-partial this staging discipline exists to prevent. + * `File.Replace` is the atomic swap (Win32 `ReplaceFile`), and it requires the destination to + * exist, so an absent one falls back to a plain `Move`. That fallback is raced deliberately: if the + * destination appears in between, `Move` throws, the staged file survives, and the destination is + * left exactly as whoever created it left it. + * + * `File::Move` throwing on an existing destination is also precisely the exclusive contract, which + * is why that branch needs nothing else. + */ +export function makeWindowsPublishStagedFileCommand( + stagingPath: string, + remotePath: string, + mode: WindowsPublishMode +): string { + const preamble = [ + '$ErrorActionPreference = "Stop"', + `$staging = ${powerShellLiteral(stagingPath)}`, + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }' + ] + if (mode === 'append') { + return powerShellCommand( + [ + ...preamble, + // Not atomic, and cannot cheaply be: appending is defined as extending the destination, so + // a failure part-way leaves it longer than it was rather than destroyed. The caller's + // chunked-append protocol already restarts from its own offset. + '$in = [System.IO.File]::OpenRead($staging)', + '$out = [System.IO.File]::Open($path, [System.IO.FileMode]::Append, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)', + 'try { $in.CopyTo($out) } finally { $out.Dispose(); $in.Dispose() }', + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) + } + if (mode === 'exclusive') { + return powerShellCommand([...preamble, '[System.IO.File]::Move($staging, $path)'].join('; ')) + } + return powerShellCommand( + [ + ...preamble, + // `[NullString]::Value`, not `$null`: PowerShell coerces a bare `$null` to an empty string + // when binding a .NET `string` parameter, and `Replace` rejects that with "The path is not + // of a legal form" — so every publish would fail. Measured on WindowsPowerShell 5.1.26100. + 'try { [System.IO.File]::Replace($staging, $path, [NullString]::Value) } catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ].join('; ') + ) +} + +/** Best-effort removal of a staged file whose write was abandoned. Never asserts the writer died. */ +export function makeWindowsDiscardStagedFileCommand(stagingPath: string): string { + return powerShellCommand( + [ + // Deliberately not `Stop`: the previous writer may still hold this file, and that is a + // possibility to tolerate, not an error to report. The unique staging name means a leftover + // blocks nothing; sweeping it is housekeeping. + '$ErrorActionPreference = "SilentlyContinue"', + `$staging = ${powerShellLiteral(stagingPath)}`, + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) +} + +/** + * The ancestor directories of a Windows remote path, drive root first. + * + * sftp's `mkdir` creates one level, so a batch has to name each level itself. The drive root is + * excluded: `-mkdir "/C:/"` is not a directory anyone creates. + */ +export function windowsRemoteAncestorDirectories(remotePath: string): string[] { + const normalized = normalizeWindowsRemotePath(remotePath) + const segments = normalized.split('/') + segments.pop() + const ancestors: string[] = [] + // Start past the drive (`C:`) or the UNC host, which are never created. + for (let depth = 2; depth <= segments.length; depth += 1) { + const directory = segments.slice(0, depth).join('/') + if (directory) { + ancestors.push(directory) + } + } + return ancestors +} + +/** + * `[Console]::OpenStandardInput()` into a `FileStream`, used only by the two stdin fallbacks. + * + * On Windows PowerShell 5.1 this is the defective read; see the strategy comment in + * `system-ssh-file-binary-transfer.ts`. It is correct under PowerShell 7. + */ +export function makeWindowsWriteFileCommand( + remotePath: string, + options?: { append?: boolean; exclusive?: boolean; executable?: 'powershell.exe' | 'pwsh.exe' } +): string { + const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' + return powerShellCommand( + [ + '$ErrorActionPreference = "Stop"', + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', + '$inputStream = [Console]::OpenStandardInput()', + `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, + 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' + ].join('; '), + options?.executable ?? 'powershell.exe' + ) +} diff --git a/src/main/ssh/system-ssh-windows-upload.test.ts b/src/main/ssh/system-ssh-windows-upload.test.ts index 207c3e2df8e..c3a5ff80276 100644 --- a/src/main/ssh/system-ssh-windows-upload.test.ts +++ b/src/main/ssh/system-ssh-windows-upload.test.ts @@ -1,29 +1,40 @@ /** - * #16432: the Windows relay upload pushed the whole bundle into one PowerShell stdin, which - * Windows PowerShell 5.1 cannot drain over a non-pty ssh exec — the remote blocks forever, and - * `waitForChannelClose()` had no timeout, so the UI sat at "Connecting…" with no error. Covered - * here: no write exceeds one stdin's worth on any Windows path (bundle upload *and* single-file - * upload, which is the one that carries large files), a partial write never lands under the real - * name, and a remote that never closes fails instead of hanging. + * #16432. The original fix chunked the payload because the constraint was believed to be a ~50KB + * cmd.exe stdin ceiling. Re-measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2, it is + * not a size limit and not cmd.exe's: a read on Windows PowerShell 5.1's redirected-stdin handle + * over a non-pty ssh exec can die permanently when it finds the stream momentarily empty, taking + * both the remaining data and the EOF with it. It is probabilistic per such read — identical 2MB + * payloads died at 167936, 270336 and 372736 — so a 32KB chunk still failed 15 times in 120 under + * load, while `findstr` took 2,016,000 bytes through one exec on the same host. + * + * So the covering property is no longer "every write is small". It is "the bytes do not cross a + * remote process's stdin at all": sftp first, PowerShell 7 next, and Windows PowerShell 5.1 last, + * bounded and loud. The staging-and-rename discipline is kept on every path, with a unique staging + * name per attempt so a retry never meets a predecessor's lock. */ import { EventEmitter } from 'node:events' import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' -import { rm } from 'node:fs/promises' +import { readFile, rm, stat } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { PassThrough, Writable } from 'node:stream' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as SystemSshOperationLifecycle from './system-ssh-operation-lifecycle' -const { spawnSystemSshCommandMock, waitForChannelCloseSpy } = vi.hoisted(() => ({ +const { spawnSystemSshCommandMock, waitForChannelCloseSpy, runProcessMock } = vi.hoisted(() => ({ spawnSystemSshCommandMock: vi.fn(), - waitForChannelCloseSpy: vi.fn() + waitForChannelCloseSpy: vi.fn(), + runProcessMock: vi.fn() })) vi.mock('./system-ssh-command', () => ({ spawnSystemSshCommand: spawnSystemSshCommandMock })) +vi.mock('../../shared/child-process/run-process', () => ({ + runProcess: runProcessMock +})) + // Delegates to the real implementation; the spy only records whether each wait was given a bound. vi.mock('./system-ssh-operation-lifecycle', async (importActual) => { const actual = (await importActual()) as typeof SystemSshOperationLifecycle @@ -41,6 +52,11 @@ import { } from './system-ssh-file-binary-transfer' import { waitForChannelClose } from './system-ssh-operation-lifecycle' import { getRemoteHostPlatform } from './ssh-remote-platform' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities +} from './system-ssh-windows-write-capabilities' +import { explainWindowsPowerShellStdinFailure } from './system-ssh-windows-write-strategy' import type { SshTarget } from '../../shared/ssh-types' type FakeChannel = EventEmitter & { @@ -50,7 +66,12 @@ type FakeChannel = EventEmitter & { written: Buffer } -const target = { id: 'win-1', host: 'win.example', username: 'dev' } as unknown as SshTarget +const target = { + id: 'win-1', + host: 'win.example', + username: 'dev', + port: 22 +} as unknown as SshTarget const hostPlatform = getRemoteHostPlatform('win32-x64') const remoteRoot = 'C:/Users/dev/.orca-remote' @@ -78,96 +99,238 @@ function createFakeChannel(onEnd: (channel: FakeChannel) => void): FakeChannel { return channel } -type RecordedCommand = { script: string; stdin: Buffer } +type RecordedCommand = { script: string; executable: string; stdin: Buffer } +type RecordedSftpBatch = { args: string[]; script: string } -describe('Windows upload stdin framing', () => { - let localDir: string - const commands: RecordedCommand[] = [] - /** Index of the spawn that should report a non-zero exit, to model a chunk failing mid-file. */ - let failAtSpawn = -1 +const sftpBatches: RecordedSftpBatch[] = [] +const commands: RecordedCommand[] = [] +/** Index of the exec that should report a non-zero exit, to model a chunk failing mid-file. */ +let failAtSpawn = -1 +let localDir: string - const fileWrites = (): RecordedCommand[] => - commands.filter((command) => command.script.includes('FileMode]::')) - const writtenPath = (command: RecordedCommand): string => - /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1].replace(/''/g, "'") ?? '' - const fileMode = (command: RecordedCommand): string | undefined => - /FileMode\]::(\w+)/.exec(command.script)?.[1] +const fileWrites = (): RecordedCommand[] => + commands.filter((command) => command.script.includes('OpenStandardInput')) +const writtenPath = (command: RecordedCommand): string => + /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1]?.replace(/''/g, "'") ?? '' +const fileMode = (command: RecordedCommand): string | undefined => + /FileMode\]::(\w+)/.exec(command.script)?.[1] +const putLines = (): string[] => + sftpBatches.flatMap((batch) => batch.script.split('\n').filter((line) => line.startsWith('put '))) +const putDestination = (line: string): string => /put "(?:[^"]*)" "([^"]*)"/.exec(line)?.[1] ?? '' +const putSource = (line: string): string => /put "([^"]*)"/.exec(line)?.[1] ?? '' - beforeEach(() => { - commands.length = 0 - failAtSpawn = -1 - waitForChannelCloseSpy.mockClear() - localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) - spawnSystemSshCommandMock.mockReset() - spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { - const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 - return createFakeChannel((channel) => { - commands.push({ script: decodePowerShellCommand(command), stdin: channel.written }) - setImmediate(() => - spawnIndex === failAtSpawn - ? channel.emit('close', 1, null) - : channel.emit('close', 0, null) - ) +/** Makes every sftp batch succeed, recording what it was asked to do. */ +function acceptSftp(): void { + runProcessMock.mockImplementation( + async (spec: { args: string[]; input: string; program: string }) => { + const script = spec.input + sftpBatches.push({ args: spec.args, script }) + // Model the real client: `put` copies the local file, so read it while it still exists. + for (const line of script.split('\n').filter((entry) => entry.startsWith('put '))) { + await readFile(putSource(line)) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + } + ) +} + +/** Models a host whose sshd has no `Subsystem sftp` line. */ +function refuseSftp(): void { + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + return { + code: 255, + signal: null, + stdout: '', + stderr: 'subsystem request failed on channel 0\nConnection closed', + timedOut: false + } + }) +} + +/** Models a host with no PowerShell 7, which cmd.exe reports as an unrecognized command. */ +function refusePwsh(): void { + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + const executable = command.split(' ')[0] ?? '' + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable, + stdin: channel.written + }) + setImmediate(() => { + if (executable === 'pwsh.exe') { + channel.stderr.write( + "'pwsh.exe' is not recognized as an internal or external command,\noperable program or batch file." + ) + channel.emit('close', 9009, null) + return + } + channel.emit('close', spawnIndex === failAtSpawn ? 1 : 0, null) }) }) }) +} - afterEach(async () => { - await rm(localDir, { recursive: true, force: true }) +beforeEach(() => { + commands.length = 0 + sftpBatches.length = 0 + failAtSpawn = -1 + clearWindowsRemoteWriteCapabilitiesForTests() + waitForChannelCloseSpy.mockClear() + localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) + process.env.ORCA_SYSTEM_SFTP_PATH = '/usr/bin/sftp' + runProcessMock.mockReset() + acceptSftp() + spawnSystemSshCommandMock.mockReset() + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable: command.split(' ')[0] ?? '', + stdin: channel.written + }) + setImmediate(() => + spawnIndex === failAtSpawn ? channel.emit('close', 1, null) : channel.emit('close', 0, null) + ) + }) }) +}) - it('never pushes a whole artifact bundle into one PowerShell stdin', async () => { - mkdirSync(join(localDir, 'node'), { recursive: true }) - // Comfortably past the ~50KB point at which the reporter measured PowerShell 5.1 wedging. - writeFileSync(join(localDir, 'node', 'relay.js'), Buffer.alloc(600 * 1024, 0x61)) - writeFileSync(join(localDir, 'index.js'), Buffer.alloc(300 * 1024, 0x62)) +afterEach(async () => { + delete process.env.ORCA_SYSTEM_SFTP_PATH + await rm(localDir, { recursive: true, force: true }) +}) - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - - const largest = Math.max(...commands.map((command) => command.stdin.length)) - expect(largest).toBeLessThanOrEqual(WINDOWS_STDIN_WRITE_CHUNK_BYTES) - // The base64 + JSON envelope is gone entirely: nothing reads the bundle as one string. - expect(commands.some((command) => command.script.includes('FromBase64String'))).toBe(false) - // `[Console]::In` wedged at 50KB where the stream reader did not, so the mkdir batch — the one - // payload still read as a string — must use the reader the reporter measured surviving. - expect(commands.some((command) => command.script.includes('[Console]::In.ReadToEnd()'))).toBe( - false - ) - expect( - commands.filter((command) => command.script.includes('StreamReader([Console]::')) - ).toHaveLength(1) - }) - - it('bounds the single-file upload too, which is the path large files take', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) +describe('Windows upload over sftp', () => { + it('moves the payload without any remote process reading a stdin', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 60 + 11, 0x64) const localPath = join(localDir, 'big.node') writeFileSync(localPath, contents) await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/big.node`, { hostPlatform }) - const writes = fileWrites() - expect(writes).toHaveLength(4) - expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( - WINDOWS_STDIN_WRITE_CHUNK_BYTES - ) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. - expect( - waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ).toBe(true) + // The defect is a remote stdin read; the fix is that there is not one. + expect(fileWrites()).toHaveLength(0) + expect(putLines()).toHaveLength(1) + // One transfer, not 61 execs: the whole point of the change. + expect(sftpBatches).toHaveLength(1) }) - it('writes every byte of every artifact across the chunked writes', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 2 + 17, 0x63) - writeFileSync(join(localDir, 'relay.js'), contents) + it('creates the parent chain and sends the payload in one round trip', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/a/b/relay.js`, { + hostPlatform + }) - const writes = fileWrites() - expect(writes).toHaveLength(3) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // Only the first write creates the staging file; the rest must extend it or it is truncated. - expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append']) + expect(sftpBatches).toHaveLength(1) + expect(sftpBatches[0]!.script.split('\n').filter(Boolean)).toEqual([ + '-mkdir "/C:/Users"', + '-mkdir "/C:/Users/dev"', + '-mkdir "/C:/Users/dev/.orca-remote"', + '-mkdir "/C:/Users/dev/.orca-remote/a"', + '-mkdir "/C:/Users/dev/.orca-remote/a/b"', + expect.stringContaining('put ') as unknown as string + ]) + }) + + it('addresses the destination in the drive-rooted namespace sftp exposes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A backslash destination silently writes a file named `C` and still exits 0, so the leading + // slash and forward separators are correctness, not style. + expect(putDestination(putLines()[0]!)).toMatch( + /^\/C:\/Users\/dev\/\.orca-remote\/relay\.js\.orca-partial-[0-9a-f]{12}$/ + ) + }) + + it('never lands a partial under the real name, and publishes by rename', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const remotePath = `${remoteRoot}/relay.js` + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), remotePath, { hostPlatform }) + + const destination = putDestination(putLines()[0]!) + // Assert the positive first: an unmatched regex yields '', which would satisfy the `not.toBe` + // below without this test ever having seen a destination. + expect(destination).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + expect(destination).not.toBe(`/C:${remotePath.slice(2)}`) + const publish = commands.at(-1)! + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // The publish reads the staged file, never a pipe, so it is safe on PowerShell 5.1. + expect(publish.script).not.toContain('OpenStandardInput') + }) + + it('never deletes the destination it is replacing', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const publish = commands.at(-1)! + // Delete-then-move destroys the user's existing file outright if the move then fails, and + // exposes a window where a reader sees no file at all — worse than the truncated partial the + // staging discipline exists to prevent. `File.Replace` is the atomic swap. + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // An absent destination cannot be Replaced, so that case falls back to a plain Move. + expect(publish.script).toContain( + 'catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ) + }) + + it('gives every attempt its own staging name, so a retry cannot meet a predecessor lock', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const [first, second] = putLines().map(putDestination) + expect(first).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + // Losing contact is not evidence the previous writer died, so the name must not be reused. + expect(second).not.toBe(first) + }) + + it('enforces exclusive at the rename, where it is atomic', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + }) + + it('appends by concatenating the staged file, not by piping bytes to the remote', async () => { + await writeBufferViaSystemSsh(target, `${remoteRoot}/log.bin`, Buffer.from('tail'), { + hostPlatform, + append: true + }) + + expect(fileWrites()).toHaveLength(0) + const publish = commands.at(-1)! + expect(publish.script).toContain('FileMode]::Append') + expect(publish.script).toContain('$in.CopyTo($out)') + expect(publish.script).toContain('[System.IO.File]::Delete($staging)') }) it('still creates an empty artifact on the host', async () => { @@ -175,80 +338,310 @@ describe('Windows upload stdin framing', () => { await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - expect(fileWrites().map(writtenPath)).toEqual([`${remoteRoot}/empty.txt`]) - expect(fileWrites()[0].stdin).toHaveLength(0) - expect(fileMode(fileWrites()[0])).toBe('Create') + expect(putLines()).toHaveLength(1) + expect(commands.at(-1)!.script).toContain('[System.IO.File]::Move($staging, $path)') }) - it('lands a multi-chunk write on a staging path and publishes it by rename', async () => { - const remotePath = `${remoteRoot}/relay.js` - writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + it('writes a buffer through a 0600 temp file that does not outlive the transfer', async () => { + const seen: { path: string; contents: Buffer; mode: number }[] = [] + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + for (const line of spec.input.split('\n').filter((entry) => entry.startsWith('put '))) { + const path = putSource(line) + seen.push({ + path, + contents: await readFile(path), + mode: (await stat(path)).mode & 0o777 + }) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + }) + + await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { + hostPlatform + }) + + expect(seen).toHaveLength(1) + expect(seen[0]!.contents.toString()).toBe('1.2.3') + // The payload can be repository content and tmpdir is world-readable on every platform, so the + // window between write and upload must not be group- or world-readable. + expect(seen[0]!.mode).toBe(0o600) + await expect(readFile(seen[0]!.path)).rejects.toThrow() + }) + + it('creates upload directories over sftp rather than a PowerShell stdin batch', async () => { + mkdirSync(join(localDir, 'node'), { recursive: true }) + writeFileSync(join(localDir, 'node', 'relay.js'), 'x') await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - // Nothing touches the real name until every byte is on the host. - expect(fileWrites().map(writtenPath)).toEqual([ - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`, - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` - ]) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).toContain('[System.IO.File]::Delete($path)') + // Anchor on a non-empty observation: `some` is false of an empty list, so this would pass even + // if no command had been recorded at all. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('StreamReader([Console]::'))).toBe( + false + ) + expect(sftpBatches[0]!.script).toContain('-mkdir "/C:/Users/dev/.orca-remote"') + }) + + it('sweeps the staged bytes when the publish is the thing that fails', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + // An exclusive conflict is the ordinary way to get here: the payload is on the host, and the + // rename that would have given it a name refuses. + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const script = decodePowerShellCommand(command) + return createFakeChannel((channel) => { + commands.push({ script, executable: command.split(' ')[0] ?? '', stdin: channel.written }) + const failed = script.includes('::Move($staging, $path)') + setImmediate(() => channel.emit('close', failed ? 1 : 0, null)) + }) + }) + + await expect( + uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + ).rejects.toThrow() + + const sweep = commands.at(-1)! + expect(sweep.script).toContain('[System.IO.File]::Delete($staging)') + // Tolerated, not asserted: the previous writer may still hold the file, and losing contact is + // not evidence it died. + expect(sweep.script).toContain('$ErrorActionPreference = "SilentlyContinue"') + }) + + it('reports a cancelled transfer as an abort, not as a failed one', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const controller = new AbortController() + // runProcess reports the kill as a non-zero exit rather than throwing, so without checking the + // signal first a user pressing cancel is indistinguishable from the transfer genuinely failing. + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + controller.abort() + return { code: 255, signal: 'SIGTERM', stdout: '', stderr: '', timedOut: false } + }) + + let error: Error | undefined + try { + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + signal: controller.signal + }) + } catch (thrown) { + error = thrown as Error + } + + expect(error?.name).toBe('AbortError') + expect(error?.message).not.toContain('sftp batch failed') + // A cancel is also not evidence about the host, so it must not send later writes to the slow + // path, and must not fall through to the defective reader now. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + expect(fileWrites()).toHaveLength(0) + }) + + it('does not let one unaddressable path become a verdict about the host', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + // A UNC destination has no settled mapping in sftp's drive-rooted namespace, so this write + // falls back — but the host still serves sftp perfectly well for every other path. + await uploadFileViaSystemSsh( + target, + join(localDir, 'relay.js'), + '//fileserver/share/relay.js', + { hostPlatform } + ) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(sftpBatches).toHaveLength(0) + // The 30-minute capability cache is keyed by host; caching this would send every later write + // to the same machine down the defective path on the strength of one odd destination. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps using sftp for the next file after one path it could not spell', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), '//fileserver/share/a.js', { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/b.js`, { + hostPlatform + }) + + expect(putLines()).toHaveLength(1) + expect(putDestination(putLines()[0]!)).toContain('/C:/Users/dev/.orca-remote/b.js') + }) + + it('does not let a local filename sftp cannot quote become a verdict either', async () => { + // POSIX clients allow a newline in a filename, and sftp's batch lexer would read it as the end + // of one command and the start of another. + const awkward = join(localDir, 'two\nlines.js') + writeFileSync(awkward, 'x') + + await uploadFileViaSystemSsh(target, awkward, `${remoteRoot}/relay.js`, { hostPlatform }) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('translates the ssh argument list rather than passing it to a client that reads it differently', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + disableControlMaster: true + }) + + const args = sftpBatches[0]!.args + // sftp's `-T` does not exist, its `-p` preserves mtime, and its `-S` names a program to run. + expect(args).not.toContain('-T') + expect(args).not.toContain('-p') + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') + expect(args).toContain('ServerAliveInterval=15') + }) +}) + +describe('Windows upload on a host with no sftp subsystem', () => { + beforeEach(() => { + refuseSftp() + }) + + it('creates a multi-directory tree, which the one-element case never exercised', async () => { + mkdirSync(join(localDir, 'node', 'deep'), { recursive: true }) + writeFileSync(join(localDir, 'index.js'), 'a') + writeFileSync(join(localDir, 'node', 'deep', 'x.js'), 'b') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const mkdir = commands.find((command) => command.script.includes('ConvertFrom-Json'))! + // `@($json | ConvertFrom-Json)` wraps the parsed array in another array, so the loop variable + // binds to the whole thing and `[string]` of it is the paths joined by spaces — which + // CreateDirectory rejects. It only ever worked for a single directory, where stringifying a + // one-element array happens to yield the element, so no batch of one can catch this. + expect(mkdir.script).toContain('[string[]]($json | ConvertFrom-Json)') + expect(mkdir.script).not.toContain('@($json | ConvertFrom-Json)') + const batch = JSON.parse(mkdir.stdin.toString('utf-8')) as string[] + expect(batch.length).toBeGreaterThan(1) + }) + + it('falls back rather than failing the transfer', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 5, 0x61) + writeFileSync(join(localDir, 'relay.js'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(Buffer.concat(fileWrites().map((write) => write.stdin)).equals(contents)).toBe(true) + }) + + it('remembers the refusal, so a multi-file upload probes once', async () => { + writeFileSync(join(localDir, 'a.js'), 'a') + writeFileSync(join(localDir, 'b.js'), 'b') + writeFileSync(join(localDir, 'c.js'), 'c') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + // One refusal is enough; re-probing per file is a wasted round trip on every file. + expect(sftpBatches).toHaveLength(1) + }) + + it('does not spend a sweep round trip when sftp declined before moving any bytes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A refused subsystem staged nothing, so there is nothing to delete — and on a host without + // sftp that sweep would otherwise be paid on every single write. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('Delete($staging)'))).toBe(false) + }) + + it('prefers PowerShell 7, which reads a redirected stdin correctly', async () => { + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(fileWrites().map((write) => write.executable)).toEqual(['pwsh.exe']) + // PowerShell 7 took 2MB through one exec when measured, so chunking it buys nothing. + expect(fileWrites()[0]!.stdin).toHaveLength(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3) + }) + + it('bounds every write when only Windows PowerShell 5.1 is available', async () => { + refusePwsh() + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) + writeFileSync(join(localDir, 'big.node'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + const writes = fileWrites().filter((write) => write.executable === 'powershell.exe') + expect(writes).toHaveLength(4) + expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( + WINDOWS_STDIN_WRITE_CHUNK_BYTES + ) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append', 'Append']) + // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. + // Count first: `every` is true of zero calls, so a wait that moved to a different helper would + // pass this silently. + expect(waitForChannelCloseSpy.mock.calls.length).toBeGreaterThan(0) + expect( + waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ).toBe(true) + }) + + it('remembers that PowerShell 7 is absent instead of re-probing per chunk', async () => { + refusePwsh() + writeFileSync(join(localDir, 'big.node'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + expect(fileWrites().filter((write) => write.executable === 'pwsh.exe')).toHaveLength(1) }) it('leaves no truncated file under the real name when a chunk fails mid-file', async () => { writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) - // Spawns: 0 = mkdir batch, 1..3 = chunk writes. Fail the second chunk. - failAtSpawn = 2 + // Spawn 0 is the pwsh write; fail it and every retry beneath it. + failAtSpawn = 0 await expect( - uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) ).rejects.toThrow() + expect(fileWrites().length).toBeGreaterThan(0) expect(fileWrites().map(writtenPath)).not.toContain(`${remoteRoot}/relay.js`) expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) }) +}) - it('enforces exclusive once at the rename, so a retry is not blocked by its own leftovers', async () => { - const localPath = join(localDir, 'import.bin') - writeFileSync(localPath, Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) +describe('last-resort Windows PowerShell failure reporting', () => { + it('names the host limitation and its remedy, not just the timeout', () => { + const timeout = new Error('write C:/x at offset 0 timed out after 60000ms with no response') - await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/import.bin`, { - hostPlatform, - exclusive: true - }) + const explained = explainWindowsPowerShellStdinFailure(timeout) as Error - // CreateNew on chunk one would fail against a leftover staging file from a failed attempt; - // `File::Move` raising on an existing destination is what carries the exclusive contract. - expect(fileWrites().map(fileMode)).toEqual(['Create', 'Append']) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + // "timed out" alone sends the user to retry a network they cannot fix; the fix is host-side. + expect(explained.message).toContain('Windows PowerShell 5.1') + expect(explained.message).toContain('Subsystem sftp sftp-server.exe') + expect(explained.cause).toBe(timeout) }) - it('keeps a single-chunk write on the destination, with the caller mode intact', async () => { - await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { - hostPlatform, - exclusive: true - }) + it('leaves a real failure alone, so a permission error is not reported as a host limitation', () => { + const denied = new Error('write C:/x at offset 0 failed (exit 1): Access to the path is denied') - expect(fileWrites()).toHaveLength(1) - expect(writtenPath(fileWrites()[0])).toBe(`${remoteRoot}/version`) - expect(fileMode(fileWrites()[0])).toBe('CreateNew') - expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) - }) - - it('appends onto the destination rather than staging, since append cannot be staged', async () => { - const remotePath = `${remoteRoot}/log.bin` - await writeBufferViaSystemSsh( - target, - remotePath, - Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1), - { hostPlatform, append: true } - ) - - expect(fileWrites().map(writtenPath)).toEqual([remotePath, remotePath]) - expect(fileWrites().map(fileMode)).toEqual(['Append', 'Append']) + expect(explainWindowsPowerShellStdinFailure(denied)).toBe(denied) }) }) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.test.ts b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts new file mode 100644 index 00000000000..ad723d592b0 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts @@ -0,0 +1,79 @@ +/** + * Whether a Windows host has an sftp subsystem is a fact about that host, so the cache is keyed by + * the endpoint that executes rather than by Orca's target id — otherwise a hardened host is + * re-probed once per file, and two targets pointing at one machine learn the same fact twice. + */ +import { afterEach, describe, expect, it } from 'vitest' +import type { SshTarget } from '../../shared/ssh-types' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities, + getWindowsRemoteWriteExecutionHostKey +} from './system-ssh-windows-write-capabilities' + +const asTarget = (fields: Partial): SshTarget => fields as SshTarget + +afterEach(() => { + clearWindowsRemoteWriteCapabilitiesForTests() +}) + +describe('getWindowsRemoteWriteExecutionHostKey', () => { + it('gives two targets on one endpoint the same key', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + // A target re-created under a new id has not changed what the host supports. + expect(getWindowsRemoteWriteExecutionHostKey(first)).toBe( + getWindowsRemoteWriteExecutionHostKey(second) + ) + }) + + it('separates hosts, ports and users', () => { + const base = { id: 'a', host: 'win.example', username: 'dev', port: 22 } + const keys = [ + asTarget(base), + asTarget({ ...base, host: 'other.example' }), + asTarget({ ...base, port: 2222 }), + asTarget({ ...base, username: 'ops' }) + ].map(getWindowsRemoteWriteExecutionHostKey) + + expect(new Set(keys).size).toBe(4) + }) + + it('keys a config alias by the alias, since ssh_config decides where it lands', () => { + const alias = asTarget({ id: 'a', host: 'stale.example', configHost: 'winbox' }) + + expect(getWindowsRemoteWriteExecutionHostKey(alias)).toBe('config:winbox') + }) +}) + +describe('getWindowsRemoteWriteCapabilities', () => { + it('shares one cache across targets that reach the same host', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(first).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(second).shouldTry('sftp-subsystem')).toBe(false) + }) + + it('does not let one host answer for another', () => { + const hardened = asTarget({ id: 'a', host: 'hardened.example', username: 'dev', port: 22 }) + const ordinary = asTarget({ id: 'b', host: 'ordinary.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(hardened).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(ordinary).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps the two capabilities independent', () => { + const target = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const capabilities = getWindowsRemoteWriteCapabilities(target) + + capabilities.rememberUnsupported('pwsh') + + // No PowerShell 7 says nothing about whether the host will serve sftp. + expect(capabilities.shouldTry('sftp-subsystem')).toBe(true) + expect(capabilities.shouldTry('pwsh')).toBe(false) + }) +}) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.ts b/src/main/ssh/system-ssh-windows-write-capabilities.ts new file mode 100644 index 00000000000..dcd03f19807 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.ts @@ -0,0 +1,52 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { CapabilityProbeCache } from '../../shared/capability-probe-cache' + +/** + * Whether a Windows host can take a file write over the sftp subsystem, and whether it has a + * PowerShell 7 to fall back to. Both are host facts, so they are cached per execution host rather + * than per transfer — a hardened host with `Subsystem sftp` removed must not be re-probed on every + * file of a multi-file upload. + */ +export type WindowsRemoteWriteCapability = 'sftp-subsystem' | 'pwsh' + +// Why re-probe at all: an admin can enable the subsystem, or install PowerShell 7, without the +// user restarting Orca. Long enough that a hardened host costs one failed probe per half hour. +export const WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS = 30 * 60_000 + +const capabilitiesByExecutionHost = new Map< + string, + CapabilityProbeCache +>() + +/** + * Keyed by the endpoint that executes, not by target id: two Orca targets pointing at one host + * describe the same sshd, and a target re-created under a new id has not changed what that host + * supports. A config alias is its own key because ssh_config, not Orca, resolves where it lands. + */ +export function getWindowsRemoteWriteExecutionHostKey(target: SshTarget): string { + if (target.configHost) { + return `config:${target.configHost}` + } + const port = target.port ?? 22 + return target.username + ? `host:${target.username}@${target.host}:${port}` + : `host:${target.host}:${port}` +} + +export function getWindowsRemoteWriteCapabilities( + target: SshTarget +): CapabilityProbeCache { + const key = getWindowsRemoteWriteExecutionHostKey(target) + let cache = capabilitiesByExecutionHost.get(key) + if (!cache) { + cache = new CapabilityProbeCache( + WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS + ) + capabilitiesByExecutionHost.set(key, cache) + } + return cache +} + +export function clearWindowsRemoteWriteCapabilitiesForTests(): void { + capabilitiesByExecutionHost.clear() +} diff --git a/src/main/ssh/system-ssh-windows-write-strategy.ts b/src/main/ssh/system-ssh-windows-write-strategy.ts new file mode 100644 index 00000000000..f2cdca12516 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-strategy.ts @@ -0,0 +1,329 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { getSystemSshBuildArgsFromOperationOptions } from './system-ssh-args' +import { spawnSystemSshCommand } from './system-ssh-command' +import { + awaitWithSystemSshAbort, + throwIfAborted, + waitForChannelClose +} from './system-ssh-operation-lifecycle' +import { + isSftpPathUnsupportedError, + isSftpRefusalBeforeStaging, + isSftpUnavailableError, + runSftpBatch +} from './system-ssh-sftp-transfer' +import { quoteSftpBatchArgument, toSftpRemotePath } from './system-ssh-sftp-path' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' +import { + makeWindowsDiscardStagedFileCommand, + makeWindowsPublishStagedFileCommand, + makeWindowsStagingPath, + makeWindowsWriteFileCommand, + windowsRemoteAncestorDirectories, + type WindowsPublishMode +} from './system-ssh-windows-file-write' + +/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ +export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 + +/** + * Bound on one stdin write for the last-resort Windows PowerShell 5.1 path. + * + * Measured on Windows 11 26200 / OpenSSH 10.0p2: a 32KB write still hangs 15 times in 120 under + * load, and no smaller value removes the risk. The defect is per blocking read, not per byte, so + * shrinking the chunk trades one risky read for more execs that each carry their own. This is a + * damage bound on a path known to be unreliable, not a safe size. + */ +export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 + +export type WindowsWriteOptions = Parameters< + typeof getSystemSshBuildArgsFromOperationOptions +>[0] & { + signal?: AbortSignal + append?: boolean + exclusive?: boolean +} + +/** Bytes to write, plus a way to present them to sftp, which can only send a local file. */ +export type WindowsWriteSource = { + totalBytes: number + readChunk: (offset: number, maxBytes: number) => Promise + withLocalFile: (send: (localPath: string) => Promise) => Promise +} + +function publishMode(options: WindowsWriteOptions): WindowsPublishMode { + return options.append ? 'append' : options.exclusive === true ? 'exclusive' : 'create' +} + +/** + * Writes one file to a Windows host, preferring transports that do not push bytes through a remote + * PowerShell's stdin. + * + * Order, and why: sftp carries the whole payload in one transfer and never has a remote process + * read a pipe. Measured on Windows 11 / OpenSSH 10.0p2: 1.9MB in a median 315ms over sftp against + * 0 of 6 completions on the chunked path, whose best case was ~62 execs at ~350ms each. PowerShell + * 7 reads a redirected stdin correctly but is not installed by default. Windows PowerShell 5.1 is + * always present and is the defective reader, so it is last and it is bounded. + * + * Every transport stages under a unique name and publishes by rename, so no partial write is ever + * visible under the real name and no retry inherits a predecessor's lock. + */ +export async function writeWindowsRemoteFile( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + throwIfAborted(options.signal) + const capabilities = getWindowsRemoteWriteCapabilities(target) + await capabilities.runWithFallback( + 'sftp-subsystem', + () => writeViaSftp(target, remotePath, source, options), + () => writeViaRemoteStdin(target, remotePath, source, options), + isSftpUnavailableError + ) +} + +/** + * Stages under a name nothing else can own, publishes it, and sweeps the staging file if either + * step fails. + * + * Shared by both transports so the cleanup contract cannot drift between them: a failed publish — + * an exclusive conflict is the ordinary case — leaves bytes on the host that no longer have a + * purpose, and the sweep is what stops them accumulating. + */ +async function stageThenPublish( + target: SshTarget, + remotePath: string, + options: WindowsWriteOptions, + stage: (stagingPath: string) => Promise, + nothingStaged: (error: unknown) => boolean = () => false +): Promise { + const stagingPath = makeWindowsStagingPath(remotePath) + try { + await stage(stagingPath) + await publishStagedWrite(target, stagingPath, remotePath, options) + } catch (error) { + // A transport that declined before it moved any bytes has nothing to sweep, and sweeping + // anyway would spend a round trip on every write to a host that has no sftp subsystem. + if (!nothingStaged(error)) { + await discardStagedWrite(target, stagingPath, options) + } + throw error + } +} + +/** + * A path sftp cannot address falls back for this write alone, without touching the host verdict. + * + * The distinction matters because the capability cache is keyed by host and holds for half an hour: + * routing one UNC destination, or one local filename containing a newline, into + * `rememberUnsupported` would send every later write to that host down the defective path too. + */ +async function writeViaSftp( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + try { + await attemptSftpWrite(target, remotePath, source, options) + } catch (error) { + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await writeViaRemoteStdin(target, remotePath, source, options) + } +} + +function attemptSftpWrite( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const mkdirs = windowsRemoteAncestorDirectories(remotePath).map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + return stageThenPublish( + target, + remotePath, + options, + (stagingPath) => + source.withLocalFile((localPath) => + // One round trip: the parent chain and the payload travel in the same batch. + runSftpBatch( + target, + [ + ...mkdirs, + `put ${quoteSftpBatchArgument(localPath)} ${quoteSftpBatchArgument(toSftpRemotePath(stagingPath))}` + ], + options + ) + ), + isSftpRefusalBeforeStaging + ) +} + +function writeViaRemoteStdin( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const capabilities = getWindowsRemoteWriteCapabilities(target) + return stageThenPublish(target, remotePath, options, (stagingPath) => + capabilities.runWithFallback( + 'pwsh', + () => writeStdinChunks(target, stagingPath, source, options, 'pwsh.exe'), + () => writeStdinChunks(target, stagingPath, source, options, 'powershell.exe'), + isPwshUnavailableError + ) + ) +} + +/** + * PowerShell 7 takes the whole payload in one exec — measured at 2MB — so only the 5.1 path pays + * for chunking, and only because a bounded write is the most that path can be trusted with. + */ +async function writeStdinChunks( + target: SshTarget, + stagingPath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + const chunkBytes = + executable === 'pwsh.exe' ? Math.max(source.totalBytes, 1) : WINDOWS_STDIN_WRITE_CHUNK_BYTES + let offset = 0 + // An empty write still has to run: it is what creates the staged file. + do { + const chunk = await source.readChunk(offset, chunkBytes) + if (chunk.length === 0 && offset < source.totalBytes) { + throw new Error(`Source ran short during upload of ${stagingPath}`) + } + await writeOneStdinChunk( + target, + stagingPath, + chunk, + { ...options, append: offset > 0, exclusive: false }, + offset, + executable + ) + offset += chunk.length + } while (offset < source.totalBytes) +} + +async function writeOneStdinChunk( + target: SshTarget, + stagingPath: string, + chunk: Buffer, + options: WindowsWriteOptions, + offset: number, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand( + target, + makeWindowsWriteFileCommand(stagingPath, { + append: options.append, + exclusive: options.exclusive, + executable + }), + { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } + ) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose( + channel, + `write ${stagingPath} at offset ${offset}`, + WINDOWS_STDIN_WRITE_TIMEOUT_MS + ) + ).catch((error: unknown) => { + throw executable === 'powershell.exe' ? explainWindowsPowerShellStdinFailure(error) : error + }) + if (!options.signal?.aborted) { + channel.stdin.end(chunk) + } + await closePromise +} + +/** + * Names the cause on the one path that can hang, so the failure is not just "timed out". + * + * A user seeing this needs to know it is a host limitation with a host-side remedy, not a network + * fault they should retry into. + */ +export function explainWindowsPowerShellStdinFailure(error: unknown): unknown { + const message = error instanceof Error ? error.message : String(error) + if (!/timed out/i.test(message)) { + return error + } + return new Error( + `${message}\nWindows PowerShell 5.1 can lose a redirected stdin permanently when a read finds it momentarily empty, so this write cannot be made reliable from the client. Enable the sftp subsystem on the host (sshd_config: "Subsystem sftp sftp-server.exe"), or install PowerShell 7, and Orca will use it automatically.`, + { cause: error instanceof Error ? error : undefined } + ) +} + +function isPwshUnavailableError(error: unknown): boolean { + const message = error instanceof Error ? error.message : String(error) + // cmd.exe's "not recognized" and sshd's exit 9009 both mean "no pwsh here". A timeout does not: + // that is the stdin defect, and PowerShell 7 does not have it, so it must not be cached as absent. + return /is not recognized as an internal or external command|9009|CommandNotFoundException/i.test( + message + ) +} + +async function publishStagedWrite( + target: SshTarget, + stagingPath: string, + remotePath: string, + options: WindowsWriteOptions +): Promise { + await runWindowsCommandWithoutStdin( + target, + makeWindowsPublishStagedFileCommand(stagingPath, remotePath, publishMode(options)), + `publish ${remotePath}`, + options + ) +} + +async function discardStagedWrite( + target: SshTarget, + stagingPath: string, + options: WindowsWriteOptions +): Promise { + try { + await runWindowsCommandWithoutStdin( + target, + makeWindowsDiscardStagedFileCommand(stagingPath), + `discard ${stagingPath}`, + { ...options, signal: undefined } + ) + } catch { + // Housekeeping only. The staging name is unique, so a leftover blocks nothing, and a failure + // here says nothing about whether the abandoned writer is still alive. + } +} + +function runWindowsCommandWithoutStdin( + target: SshTarget, + command: string, + label: string, + options: WindowsWriteOptions +): Promise { + const channel = spawnSystemSshCommand(target, command, { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, label, WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end() + } + return closePromise +} diff --git a/src/main/startup/desktop-startup-ordering.test.ts b/src/main/startup/desktop-startup-ordering.test.ts index fc381f15714..5e4d4cfe428 100644 --- a/src/main/startup/desktop-startup-ordering.test.ts +++ b/src/main/startup/desktop-startup-ordering.test.ts @@ -66,7 +66,12 @@ describe('startup ordering', () => { ) expect(desktopStartup).toContain('recordRuntimeRpcStartFailure(') // Why: `void`, not `await` — awaiting the dialog would park the rest of startup behind a modal. - expect(desktopStartup).toMatch(/void showRuntimeRpcStartupFailureDialog\(\s*win,/) + // It chains off the i18n barrier (published before this phase starts) so the translated strings + // it reads are loaded, which is a wait on i18n only, never on the dialog itself. + expect(desktopStartup).toMatch( + /void state\.mainProcessI18nReady\.then\(\(\) =>\s*showRuntimeRpcStartupFailureDialog\(\s*win,/ + ) + expect(desktopStartup).not.toMatch(/await[^\n]*showRuntimeRpcStartupFailureDialog\(/) // Why (#11025): a bare console.error here is exactly what left the CLI dead but the app healthy. expect(desktopStartup).not.toContain( "console.error('[runtime] Failed to start local RPC transport:'" diff --git a/src/main/startup/first-window-deferral.ts b/src/main/startup/first-window-deferral.ts new file mode 100644 index 00000000000..6a144549efe --- /dev/null +++ b/src/main/startup/first-window-deferral.ts @@ -0,0 +1,37 @@ +import { app, type BrowserWindow } from 'electron' + +/** + * Run `task` once the first window can paint, or after `fallbackMs` if it never does. + * + * For startup work nothing on the critical path consumes: a probe or a disk sweep started before the + * window exists competes with window creation for the same main thread and libuv threadpool, and the + * user sees that as the app being slow to open. + * + * Why a fallback as well as the window event: `ready-to-show` can fail to fire at all when the + * GPU/driver cannot present (see main-window-state-lifecycle), and headless serve has no window. + */ +export function runAfterFirstWindowShown(task: () => void, fallbackMs: number): void { + let ran = false + const run = (): void => { + if (ran) { + return + } + ran = true + clearTimeout(fallback) + // Why setImmediate: keep the work off the event handler that reveals the window, so it paints first. + // Why the guard: off whenReady's promise chain a synchronous throw is an uncaughtException, and + // installUncaughtPipeErrorGuard re-throws those fatally — deferred startup chores are never that. + setImmediate(() => { + try { + task() + } catch (error) { + console.warn('[startup] deferred first-window task failed', error) + } + }) + } + const fallback = setTimeout(run, fallbackMs) + fallback.unref?.() + app.once('browser-window-created', (_event: Electron.Event, window: BrowserWindow) => { + window.once('ready-to-show', run) + }) +} diff --git a/src/main/startup/gpu-lifecycle-install-dir-acl-guard.test.ts b/src/main/startup/gpu-lifecycle-install-dir-acl-guard.test.ts new file mode 100644 index 00000000000..e35fe401cdc --- /dev/null +++ b/src/main/startup/gpu-lifecycle-install-dir-acl-guard.test.ts @@ -0,0 +1,442 @@ +import { mkdtempSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' + +// Hoisted with the vi.mock factory below. 'Keep Running' — the prompt firing at all is the signal. +const { showMessageBox, userData } = vi.hoisted(() => ({ + showMessageBox: vi.fn(async () => ({ response: 1 })), + userData: { path: '' } +})) + +// Why the mocks: gpu-lifecycle's import graph reaches electron and the toolkit's +// electron re-export. Everything below this is the real module under test. +vi.mock('electron', () => ({ + app: { + getPath: () => userData.path, + getVersion: () => '1.4.184', + getGPUFeatureStatus: () => ({}), + setAboutPanelOptions: vi.fn(), + commandLine: { appendSwitch: vi.fn() }, + disableHardwareAcceleration: vi.fn(), + isReady: () => true, + exit: vi.fn(), + on: vi.fn(), + name: 'Orca' + }, + dialog: { showMessageBox } +})) +vi.mock('@electron-toolkit/utils', () => ({ + is: { dev: false }, + optimizer: { watchWindowShortcuts: vi.fn() }, + electronApp: { setAppUserModelId: vi.fn() } +})) + +import type { ProcessResult, ProcessSpec } from '../../shared/child-process/run-process' +import { + DEFAULT_GPU_CRASH_FALLBACK_THRESHOLD, + DEFAULT_GPU_CRASH_FALLBACK_WINDOW_MS, + GpuCrashFallbackTracker +} from '../crash-reporting/gpu-crash-fallback-decision' +import { + readGpuFallbackMarker, + writeGpuFallbackMarker, + type GpuFallbackMarker +} from './gpu-fallback-marker' +import { handleGpuChildCrash, presentGpuFallbackRecoveredLaunchPrompt } from './gpu-lifecycle' +import { gpuFallbackEnvironment, mainProcessState as state } from './main-process-state' +import { writeInstallDirAclPoisonMarker } from './windows-install-dir-acl-poison-marker' +import { + isInstallDirAclRepairPending, + noteWindowsInstallDirAclProbePending, + repairKnownPoisonedInstallDirBeforeWindow, + resetWindowsInstallDirAclRecoveryForTest, + startWindowsInstallDirAclRepairIfPoisoned +} from './windows-install-dir-acl-recovery' +import { + resetWindowsInstallDirAclRepairForTest, + WINDOWS_INSTALL_DIR_ACL_REPAIR_MARKER_FILE, + WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION +} from './windows-install-dir-package-acl-repair' + +const INSTALL_DIR = 'C:\\Users\\neil\\AppData\\Local\\Programs\\orca' + +function recoveryOptions(userDataPath?: string): { + platform: 'win32' + installDir: string + appVersion: string + userDataPath: string + recordBreadcrumb: () => void +} { + return { + platform: 'win32', + installDir: INSTALL_DIR, + appVersion: '1.4.184', + userDataPath: userDataPath ?? mkdtempSync(join(tmpdir(), 'orca-acl-gpu-guard-')), + recordBreadcrumb: () => undefined + } +} + +/** icacls hangs until `finishRepair` — the in-flight window is when the GPU children die. */ +function reportProbePoisoned(): { finishRepair: () => Promise } { + let release = (): void => undefined + const walkingTheTree = new Promise((resolve) => { + release = resolve + }) + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: true, wellKnownNameCheckReliable: true }, + { + ...recoveryOptions(), + runProcessFn: (async () => { + await walkingTheTree + return { + code: 0, + signal: null, + stdout: 'Successfully processed 3200 files; Failed processing 0 files', + stderr: '', + timedOut: false + } + }) as unknown as (spec: ProcessSpec) => Promise + } + ) + return { + finishRepair: async () => { + release() + for (let i = 0; i < 200 && isInstallDirAclRepairPending(); i += 1) { + await new Promise((resolve) => setTimeout(resolve, 5)) + } + } + } +} + +/** A repair that settles, so `poison.stage` leaves 'pending' for a terminal verdict. */ +async function reportProbePoisonedWithSettledRepair( + exitCode: number, + userDataPath?: string +): Promise { + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: true, wellKnownNameCheckReliable: true }, + { + ...recoveryOptions(userDataPath), + runProcessFn: (async () => ({ + code: exitCode, + signal: null, + stdout: 'Successfully processed 3200 files; Failed processing 0 files', + stderr: exitCode === 0 ? '' : 'access denied', + timedOut: false + })) as unknown as (spec: ProcessSpec) => Promise + } + ) + for (let i = 0; i < 200 && isInstallDirAclRepairPending(); i += 1) { + await new Promise((resolve) => setTimeout(resolve, 5)) + } +} + +/** + * The pre-window gate meeting a spent repair budget: the tree is still marked poisoned and + * Orca has no repair left to try. icacls must never be reached, so the runner throws. + */ +async function gateFindsRepairBudgetSpent(): Promise { + const options = recoveryOptions() + writeFileSync( + join(options.userDataPath, WINDOWS_INSTALL_DIR_ACL_REPAIR_MARKER_FILE), + JSON.stringify({ + schemeVersion: WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION, + installDir: INSTALL_DIR, + appVersion: options.appVersion, + attemptedAt: Date.now(), + outcome: 'failed', + attempts: 3 + }) + ) + writeInstallDirAclPoisonMarker(options.userDataPath, INSTALL_DIR, options.appVersion) + const mode = await repairKnownPoisonedInstallDirBeforeWindow({ + ...options, + runProcessFn: (() => { + throw new Error('the spent budget must not spawn icacls') + }) as never + }) + expect(mode).toBe('marker-hit') +} + +function reportProbeClean(): void { + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: false }, + recoveryOptions() + ) +} + +/** One short of the fallback threshold, so the caller's next crash is the decisive one. */ +async function crashUpToThreshold(): Promise { + for (let i = 1; i < DEFAULT_GPU_CRASH_FALLBACK_THRESHOLD; i += 1) { + await handleGpuChildCrash('crashed', null, i * 200) + } +} + +/** + * Driven end-to-end against the real tracker rather than asserted against the source: + * a source match is equally happy with the polarity inverted, and the property that + * matters is that a driver burst survives the ACL verdict either way. + */ +describe('handleGpuChildCrash vs the install-dir ACL verdict', () => { + let tracker: GpuCrashFallbackTracker + const realPlatform = process.platform + + beforeAll(() => { + // The whole guard is win32-only, and so is the safe-graphics marker it writes. + Object.defineProperty(process, 'platform', { value: 'win32', configurable: true }) + }) + + afterAll(() => { + Object.defineProperty(process, 'platform', { value: realPlatform, configurable: true }) + }) + + beforeEach(() => { + userData.path = mkdtempSync(join(tmpdir(), 'orca-acl-gpu-userdata-')) + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + showMessageBox.mockClear() + state.isQuitting = false + state.isServeMode = false + state.gpuFallbackActiveThisLaunch = false + tracker = new GpuCrashFallbackTracker({ + windowMs: DEFAULT_GPU_CRASH_FALLBACK_WINDOW_MS, + threshold: DEFAULT_GPU_CRASH_FALLBACK_THRESHOLD + }) + state.gpuCrashFallbackTracker = tracker + }) + + it('engages safe graphics on a driver burst when nothing implicates the install DACL', async () => { + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + expect(showMessageBox).toHaveBeenCalledTimes(1) + }) + + // The regression this guard must never reintroduce: the probe is armed on every + // win32 launch, so a burst landing inside its window is the common driver case. + it('keeps counting crashes that land while the probe verdict is outstanding', async () => { + noteWindowsInstallDirAclProbePending() + await crashUpToThreshold() + const decisive = handleGpuChildCrash('crashed', null, 600) + expect(showMessageBox).not.toHaveBeenCalled() + reportProbeClean() + await decisive + expect(tracker.windowSnapshot()).toHaveLength(DEFAULT_GPU_CRASH_FALLBACK_THRESHOLD) + expect(showMessageBox).toHaveBeenCalledTimes(1) + }) + + it('withholds safe graphics while the install DACL is the suspect, but keeps the evidence', async () => { + reportProbePoisoned() + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + expect(tracker.windowSnapshot()).toHaveLength(DEFAULT_GPU_CRASH_FALLBACK_THRESHOLD) + expect(showMessageBox).not.toHaveBeenCalled() + }) + + it('withholds safe graphics when the outstanding verdict comes back poisoned', async () => { + noteWindowsInstallDirAclProbePending() + await crashUpToThreshold() + const decisive = handleGpuChildCrash('crashed', null, 600) + reportProbePoisoned() + await decisive + expect(showMessageBox).not.toHaveBeenCalled() + }) + + // The gate's 'repaired' is icacls's exit claim, not a reading of the tree, and an icacls + // that silently no-opped exits 0 on a tree it left poisoned. The GPU children die in the + // interval before this launch's probe answers, so a claim that un-suspects the tree there + // engages --in-process-gpu on a tree safe graphics cannot rescue — and a "keep it" answer + // then pins a userConfirmed marker no later repair may clear. + it('withholds safe graphics between a gate repair claim and this launch probe reading', async () => { + writeInstallDirAclPoisonMarker(userData.path, INSTALL_DIR, '1.4.184') + const mode = await repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userData.path), + runProcessFn: (async () => ({ + code: 0, + signal: null, + stdout: 'Successfully processed 3200 files; Failed processing 0 files', + stderr: '', + timedOut: false + })) as unknown as (spec: ProcessSpec) => Promise + }) + expect(mode).toBe('repaired') + noteWindowsInstallDirAclProbePending() + + await crashUpToThreshold() + const decisive = handleGpuChildCrash('crashed', null, 600) + expect(showMessageBox).not.toHaveBeenCalled() + + // The reading lands poisoned: the claim was false, and engagement stays withheld. + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: true, wellKnownNameCheckReliable: true }, + recoveryOptions(userData.path) + ) + await decisive + expect(showMessageBox).not.toHaveBeenCalled() + }) + + // Chromium aborts the browser on the 6th GPU crash, sooner than the probe can answer, + // so the wait must not be the reason a machine comes back hardware-accelerated. + it('holds an unconfirmed safe-graphics marker on disk across the wait', async () => { + noteWindowsInstallDirAclProbePending() + await crashUpToThreshold() + const decisive = handleGpuChildCrash('crashed', null, 600) + expect(readGpuFallbackMarker(userData.path)?.userConfirmed).toBe(false) + reportProbePoisoned() + await decisive + // The verdict dispatched a repair, so the marker stays for the launch that repair rescues. + expect(readGpuFallbackMarker(userData.path)?.userConfirmed).toBe(false) + }) + + it('engages immediately once the probe has already reported the install clean', async () => { + noteWindowsInstallDirAclProbePending() + reportProbeClean() + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + expect(showMessageBox).toHaveBeenCalledTimes(1) + }) + + // Both from the re-run adversarial round. The gate dispatches a repair without arming the + // probe clock, so `waitForInstallDirAclVerdict` returns immediately and the withdrawal used + // to delete the marker inside Chromium's ~1.3s FATAL window — leaving the machine to + // relaunch hardware accelerated into the same 20s gate, forever. + it('keeps the safe-graphics marker on disk while a repair is still in flight', async () => { + reportProbePoisoned() + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + + expect(showMessageBox).not.toHaveBeenCalled() + expect(readGpuFallbackMarker(userData.path)?.userConfirmed).toBe(false) + }) + + it('still withdraws the marker once the verdict is terminal rather than a pending repair', async () => { + await reportProbePoisonedWithSettledRepair(1) + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + + expect(showMessageBox).not.toHaveBeenCalled() + // No repair is in flight to rescue a later launch, so the marker is not held. + expect(readGpuFallbackMarker(userData.path)).toBeNull() + }) + + // The verdict wait can span the probe's whole 15s grace window, and the entry guard was + // read before it. A quit that starts inside the wait must not be answered with a modal. + it('does not prompt when the user quits during the verdict wait', async () => { + noteWindowsInstallDirAclProbePending() + await crashUpToThreshold() + const decisive = handleGpuChildCrash('crashed', null, 600) + state.isQuitting = true + reportProbeClean() + await decisive + expect(showMessageBox).not.toHaveBeenCalled() + }) + + // Withholding is a bounded delay, not a permanent suppression. Once the repair budget is + // spent no repair is coming on this launch or any later one, so pinning the tree as the + // suspect forever denied safe graphics on EVERY launch for the life of that version — and + // deleted the marker each time, so the machine also relaunched hardware accelerated. The + // victims are a standard-user install icacls can never fix and, via the probe's flag-blind + // ACE match, healthy installs whose driver genuinely is broken. + it('offers safe graphics on every launch once the ACL repair budget is spent', async () => { + for (let launch = 1; launch <= 3; launch += 1) { + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + showMessageBox.mockClear() + state.gpuCrashFallbackTracker = new GpuCrashFallbackTracker({ + windowMs: DEFAULT_GPU_CRASH_FALLBACK_WINDOW_MS, + threshold: DEFAULT_GPU_CRASH_FALLBACK_THRESHOLD + }) + await gateFindsRepairBudgetSpent() + + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + expect(showMessageBox).toHaveBeenCalledTimes(1) + } + }) + + // Still withheld while the budget has an attempt left: the repair is the better answer, + // and this is the launch a next one can be rescued on. + it('still withholds while the repair has an attempt left to spend', async () => { + await reportProbePoisonedWithSettledRepair(1) + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + expect(showMessageBox).not.toHaveBeenCalled() + }) + + // recordGpuCrash reports the threshold crossing once and latches. Withholding consumes + // that one report, so without a re-arm the same process could never engage again — a + // machine whose tree is repaired and whose driver is genuinely broken would be stuck + // hardware-accelerated through an unbounded crash loop. + it('can still engage a later burst after a withheld one, once the tree is repaired', async () => { + const repair = reportProbePoisoned() + await crashUpToThreshold() + await handleGpuChildCrash('crashed', null, 600) + expect(showMessageBox).not.toHaveBeenCalled() + + // The repair itself reports 'repaired': the tree is no longer the suspect. + await repair.finishRepair() + expect(isInstallDirAclRepairPending()).toBe(false) + + for (let i = 1; i <= DEFAULT_GPU_CRASH_FALLBACK_THRESHOLD; i += 1) { + await handleGpuChildCrash('crashed', null, 10_000 + i * 200) + } + expect(showMessageBox).toHaveBeenCalledTimes(1) + }) +}) + +// The safe-graphics marker is read before whenReady, and the pre-window ACL gate runs after +// that read. Asking "keep safe graphics?" on a machine Orca has just repaired invites a +// `userConfirmed: true` marker that pins software rendering on healthy hardware. +describe('presentGpuFallbackRecoveredLaunchPrompt vs a marker retired since it was read', () => { + const realPlatform = process.platform + const window = { isDestroyed: () => false } as unknown as Parameters< + typeof presentGpuFallbackRecoveredLaunchPrompt + >[0] + + beforeAll(() => { + Object.defineProperty(process, 'platform', { value: 'win32', configurable: true }) + }) + + afterAll(() => { + Object.defineProperty(process, 'platform', { value: realPlatform, configurable: true }) + }) + + beforeEach(() => { + userData.path = mkdtempSync(join(tmpdir(), 'orca-acl-gpu-recovered-')) + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + showMessageBox.mockClear() + state.isQuitting = false + const info = { engagedAt: Date.now(), crashesInWindow: 3, userConfirmed: false } + writeGpuFallbackMarker(userData.path, info, { + ...gpuFallbackEnvironment(), + platform: 'win32' + }) + state.activeGpuFallbackMarker = readGpuFallbackMarker(userData.path) as GpuFallbackMarker + }) + + it('asks while the marker is still on disk', async () => { + showMessageBox.mockResolvedValueOnce({ response: 0 }) + await presentGpuFallbackRecoveredLaunchPrompt(window) + expect(showMessageBox).toHaveBeenCalledTimes(1) + }) + + it('stays silent once the install-DACL repair has cleared it', async () => { + await reportProbePoisonedWithSettledRepair(0, userData.path) + expect(readGpuFallbackMarker(userData.path)).toBeNull() + + await presentGpuFallbackRecoveredLaunchPrompt(window) + expect(showMessageBox).not.toHaveBeenCalled() + }) + + // The symmetric case to the one above: a FAILED repair leaves the marker on disk and the + // tree a live suspect, the window the prompt lands on is blank, and Keep is both defaultId + // and cancelId — so asking invites a userConfirmed pin no later repair may clear. + it('stays silent while the install DACL is still the suspect', async () => { + await reportProbePoisonedWithSettledRepair(1, userData.path) + expect(readGpuFallbackMarker(userData.path)).not.toBeNull() + + await presentGpuFallbackRecoveredLaunchPrompt(window) + expect(showMessageBox).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/startup/gpu-lifecycle.ts b/src/main/startup/gpu-lifecycle.ts index da852bcb491..89055aa2ba3 100644 --- a/src/main/startup/gpu-lifecycle.ts +++ b/src/main/startup/gpu-lifecycle.ts @@ -16,6 +16,12 @@ import { promptForGpuFallbackRestart } from '../crash-reporting/gpu-fallback-res import { engageGpuFallbackAfterCrashBurst } from '../crash-reporting/gpu-fallback-engagement' import { recordCrashBreadcrumb } from '../crash-reporting/crash-breadcrumb-store' import { recordDurableCrashBreadcrumb } from '../crash-reporting/durable-crash-breadcrumb' +import { + isInstallDirAclRepairExhausted, + isInstallDirAclRepairPending, + isInstallDirAclSuspect, + waitForInstallDirAclVerdict +} from './windows-install-dir-acl-recovery' import { mainProcessState as state, gpuFallbackEnvironment } from './main-process-state' import { createGpuAccelerationAboutPanelOptions } from '../menu/gpu-acceleration-about-panel' @@ -85,6 +91,18 @@ export async function presentGpuFallbackRecoveredLaunchPrompt( // One prompt per process. A failure leaves the on-disk marker unconfirmed so the next launch retries. state.activeGpuFallbackMarker = null const userDataPath = app.getPath('userData') + // The marker was read before whenReady; the pre-window ACL gate can have retired it since. + // Asking then would let a "keep it" answer pin software rendering on a machine Orca just fixed. + if (!readActiveGpuFallbackMarker(userDataPath, gpuFallbackEnvironment())) { + return + } + // The symmetric case: while the tree, not the driver, is on trial (a failed gate leaves it + // a live suspect), a "keep it" answer would pin a userConfirmed marker no later repair may + // clear — on the window the poison keeps blank. Staying silent leaves the marker + // unconfirmed, which a successful repair still retires. + if (isInstallDirAclSuspect()) { + return + } await handleGpuFallbackRecoveredLaunch({ isQuitting: () => state.isQuitting, prompt: () => promptForGpuFallbackRecoveredLaunch(window), @@ -114,6 +132,66 @@ export async function presentGpuFallbackRecoveredLaunchPrompt( }) } +/** + * Why withholding ends with the repair budget: withholding only buys the ACL repair the + * chance to land first. Once its attempts are spent no repair is coming on this launch or + * any later one, so holding safe graphics back forever would deny the only recovery left — + * on a genuinely poisoned tree Orca has already told the user the admin commands, and the + * probe's flag-blind ACE match also over-matches healthy installs whose driver really is + * the fault. It is a bounded delay, not a permanent suppression. + */ +function installDirAclWithholdsGpuFallback(): boolean { + return isInstallDirAclSuspect() && !isInstallDirAclRepairExhausted() +} + +/** + * Why: a poisoned install DACL kills the GPU child exactly like a bad driver, but safe + * graphics does not rescue it and --in-process-gpu removes the GPU child, erasing the + * sibling deaths that identify the real cause. + * + * Why the marker is written before the wait rather than after: Chromium aborts the whole + * browser process on the 6th GPU crash, ~1.3s after the 3rd — less than the probe takes + * to answer — so a machine that dies waiting must still come back software-rendered. + * The withdrawal below, and the repair's own clear of an unconfirmed marker, undo it. + */ +async function installDirAclClearsGpuFallback( + userDataPath: string, + crashesInWindow: number +): Promise { + if (!installDirAclWithholdsGpuFallback()) { + return true + } + const persisted = persistGpuFallbackMarker(userDataPath, { + engagedAt: Date.now(), + crashesInWindow, + userConfirmed: false + }) + await waitForInstallDirAclVerdict() + if (!installDirAclWithholdsGpuFallback()) { + return true + } + // Why the marker survives a pending repair: withdrawing it here left a machine that + // Chromium FATALs mid-repair (crash 6 lands ~1.3s after crash 3, well inside the gate) + // relaunching hardware accelerated into the same 20s gate, spawning the same GPU children, + // FATALing again — with no attempt spent, so the loop never advances. Keeping it costs a + // healthy machine nothing: a successful repair clears an unconfirmed marker itself, and a + // clean probe reading means we never reach here. It is still not *engaged* this launch, so + // --in-process-gpu does not erase the sibling-death evidence on the launch that is running. + const repairPending = isInstallDirAclRepairPending() + if (persisted && !repairPending) { + clearGpuFallbackMarker(userDataPath) + } + // Why re-arm: recordGpuCrash reports the threshold crossing once and latches. Withholding + // consumed that one report, so without this a later burst — including one after the repair + // succeeds and the tree is no longer the suspect — could never engage safe graphics again. + state.gpuCrashFallbackTracker.disengage() + recordDurableCrashBreadcrumb('gpu_fallback_withheld_install_dir_acl', { + crashesInWindow, + markerHeldForPendingRepair: repairPending + }) + return false +} + // Why: a burst of GPU child crashes means HW acceleration is unusable — persist a build-scoped marker and offer software rendering. export async function handleGpuChildCrash( reason: string, @@ -124,12 +202,23 @@ export async function handleGpuChildCrash( if (state.gpuFallbackActiveThisLaunch || state.isQuitting || state.isServeMode) { return } + // Recorded before any install-DACL consideration: the verdict decides whether safe + // graphics is the right answer, never whether the crash happened. Dropping it here + // would erase a real driver burst from the rolling window on healthy machines too. const result = state.gpuCrashFallbackTracker.recordGpuCrash(crashedAt) if (!result.shouldEngageFallback) { return } const fallbackData = { processReason: reason, exitCode, crashesInWindow: result.crashesInWindow } const userDataPath = app.getPath('userData') + if (!(await installDirAclClearsGpuFallback(userDataPath, result.crashesInWindow))) { + return + } + // Re-read after that wait: it can span the probe's whole grace window, and a quit that + // started inside it must not be answered with a modal and a relaunch. + if (state.isQuitting) { + return + } await engageGpuFallbackAfterCrashBurst( { reason, exitCode, crashesInWindow: result.crashesInWindow, engagedAt: Date.now() }, { diff --git a/src/main/startup/main-process-ready-foundation.ts b/src/main/startup/main-process-ready-foundation.ts index e122961b0c9..171aaf50421 100644 --- a/src/main/startup/main-process-ready-foundation.ts +++ b/src/main/startup/main-process-ready-foundation.ts @@ -139,7 +139,22 @@ export async function initializeReadyFoundation(): Promise { }) state.store = store // Why: create pending readiness before the guard can observe the default session. - const initialProxyApplication = applyElectronProxySettings(store.getSettings()) + // Why parked on state instead of awaited here: Dock/Launchpad launches don't inherit shell + // proxy env vars, so the persisted proxy must land before any app-owned network fetcher runs — + // but the guard below already holds every default-session request until this settles, so + // awaiting it inline only delayed window creation. Runtime launch awaits it before the first + // fetcher (the desktop relay / headless serve). + state.initialProxyApplicationReady = applyElectronProxySettings(store.getSettings()).then( + (result) => { + if (result.source === 'invalid-settings') { + // Why (STA-3442): a silent DIRECT fallback made a dead configured proxy undiagnosable. + console.warn('[proxy] persisted proxy settings are invalid; using direct networking') + } + }, + () => { + console.warn('[proxy] Failed to apply network proxy settings') + } + ) installElectronProxyRequestGuard(session.defaultSession) // Why armed here and not at install time: the report remembers what it last said, and // that state lives beside the profile data file, which does not exist until now. @@ -235,16 +250,6 @@ export async function initializeReadyFoundation(): Promise { if (shouldSuppressDevEducation({ isDev: is.dev })) { suppressDevEducationForStore(store) } - try { - // Why: Dock/Launchpad launches don't inherit shell proxy env vars, so apply the persisted proxy before any app-owned network fetchers run. - const proxyApplyResult = await initialProxyApplication - if (proxyApplyResult.source === 'invalid-settings') { - // Why (STA-3442): a silent DIRECT fallback made a dead configured proxy undiagnosable. - console.warn('[proxy] persisted proxy settings are invalid; using direct networking') - } - } catch { - console.warn('[proxy] Failed to apply network proxy settings') - } // Why: the partition installer reads the proxy through this resolver, so register it before sessions materialize. setBrowserNetworkProxySettingsResolver(() => state.store!.getSettings()) // Why: the preview session is protocol-scoped, so the handler must exist before any preview webview attaches. diff --git a/src/main/startup/main-process-ready-phase-ordering.test.ts b/src/main/startup/main-process-ready-phase-ordering.test.ts new file mode 100644 index 00000000000..749eb56cb0f --- /dev/null +++ b/src/main/startup/main-process-ready-phase-ordering.test.ts @@ -0,0 +1,153 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const phaseEvents: string[] = [] +let releaseI18n: (() => void) | null = null + +vi.mock('./main-process-ready-foundation', () => ({ + initializeReadyFoundation: vi.fn(async () => { + phaseEvents.push('foundation') + }) +})) +vi.mock('./main-process-ready-runtime', () => ({ + initializeReadyRuntimeServices: vi.fn(async () => { + phaseEvents.push('runtime-services') + }) +})) +vi.mock('./main-process-i18n-menu', () => ({ + initializeMainProcessI18nAndMenu: vi.fn( + () => + new Promise((resolve) => { + phaseEvents.push('i18n-start') + releaseI18n = () => { + phaseEvents.push('i18n-done') + resolve() + } + }) + ) +})) +vi.mock('./main-process-runtime-launch', () => ({ + initializeMainProcessRuntimeLaunch: vi.fn(async () => { + phaseEvents.push('launch-start') + await Promise.resolve() + phaseEvents.push('window-created') + }) +})) + +const { initializeMainProcessReady } = await import('./main-process-ready') + +describe('ready-phase concurrency', () => { + beforeEach(() => { + phaseEvents.length = 0 + releaseI18n = null + }) + + it('creates the window without waiting for i18n and the native menu', async () => { + const options = { + openMainWindow: vi.fn(), + handleMacAppActivation: vi.fn() + } as unknown as Parameters[0] + + const ready = initializeMainProcessReady(options) + // Drain the launch phase's microtasks while i18n is still pending. + for (let tick = 0; tick < 8; tick += 1) { + await Promise.resolve() + } + + expect(phaseEvents).toEqual([ + 'foundation', + 'runtime-services', + 'i18n-start', + 'launch-start', + 'window-created' + ]) + + releaseI18n?.() + await ready + expect(phaseEvents.at(-1)).toBe('i18n-done') + }) + + it('still resolves only once i18n and the menu have settled', async () => { + const options = { + openMainWindow: vi.fn(), + handleMacAppActivation: vi.fn() + } as unknown as Parameters[0] + + const ready = initializeMainProcessReady(options) + let settled = false + void ready.then(() => { + settled = true + }) + for (let tick = 0; tick < 8; tick += 1) { + await Promise.resolve() + } + + expect(settled).toBe(false) + releaseI18n?.() + await ready + expect(settled).toBe(true) + }) +}) + +describe('initial proxy application ordering', () => { + const readStartupSource = (file: string): string => + readFileSync(join(process.cwd(), 'src/main/startup', file), 'utf8') + + it('parks the default-session proxy apply instead of blocking window creation on it', () => { + const foundation = readStartupSource('main-process-ready-foundation.ts') + + expect(foundation).toContain('state.initialProxyApplicationReady = applyElectronProxySettings(') + // The request guard, not this phase, is what fences fetchers on the proxy; awaiting it here + // only queued openMainWindow behind a ~24 ms setProxy round trip. + expect(foundation).not.toMatch(/await\s+(?:state\.)?initialProxyApplication/) + }) + + it('awaits the proxy after the window opens and before the desktop relay starts', () => { + const launch = readStartupSource('main-process-runtime-launch.ts') + const desktopStart = launch.indexOf('async function launchDesktopMode(') + const desktopEnd = launch.indexOf('\nexport async function initializeMainProcessRuntimeLaunch') + expect(desktopStart).toBeGreaterThanOrEqual(0) + expect(desktopEnd).toBeGreaterThan(desktopStart) + const desktop = launch.slice(desktopStart, desktopEnd) + + const windowIndex = desktop.indexOf('openMainWindow()') + const proxyIndex = desktop.indexOf('await state.initialProxyApplicationReady') + const relayIndex = desktop.indexOf('new DesktopRelayService(') + + expect(windowIndex).toBeGreaterThanOrEqual(0) + expect(proxyIndex).toBeGreaterThan(windowIndex) + expect(relayIndex).toBeGreaterThan(proxyIndex) + }) + + it('waits for i18n before the only launch-phase dialog that reads a translated string', () => { + const ready = readStartupSource('main-process-ready.ts') + const launch = readStartupSource('main-process-runtime-launch.ts') + + // Published before the launch phase starts, or the barrier the dialog awaits is still the + // default resolved promise. + const publishIndex = ready.indexOf('state.mainProcessI18nReady = ') + expect(publishIndex).toBeGreaterThanOrEqual(0) + expect(ready.indexOf('initializeMainProcessRuntimeLaunch(options)')).toBeGreaterThan( + publishIndex + ) + expect(launch).toMatch( + /state\.mainProcessI18nReady\.then\(\(\) =>\s*\n?\s*showRuntimeRpcStartupFailureDialog\(/ + ) + }) + + it('keeps headless serve strictly ordered behind the proxy apply', () => { + const launch = readStartupSource('main-process-runtime-launch.ts') + const serveStart = launch.indexOf('async function launchServeMode(') + const serveEnd = launch.indexOf('\nasync function launchDesktopMode(', serveStart) + expect(serveStart).toBeGreaterThanOrEqual(0) + expect(serveEnd).toBeGreaterThan(serveStart) + const serve = launch.slice(serveStart, serveEnd) + + const proxyIndex = serve.indexOf('await state.initialProxyApplicationReady') + const rpcIndex = serve.indexOf('runtimeRpc.start()') + + expect(proxyIndex).toBeGreaterThanOrEqual(0) + expect(rpcIndex).toBeGreaterThan(proxyIndex) + }) +}) diff --git a/src/main/startup/main-process-ready-runtime.ts b/src/main/startup/main-process-ready-runtime.ts index f89f63186a0..26b920652df 100644 --- a/src/main/startup/main-process-ready-runtime.ts +++ b/src/main/startup/main-process-ready-runtime.ts @@ -9,7 +9,7 @@ import { RpcDispatcher } from '../runtime/rpc/dispatcher' import { browserManager } from '../browser/browser-manager' import { configureBrowserClientPageAutomationRuntime } from '../browser/browser-client-page-automation-runtime' import { BrowserClientPageCommandError } from '../browser/browser-client-page-command-failure' -import { startPreGoneProcessMetricsSampling } from '../crash-reporting/process-gone-diagnostics' +import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics' import { recordProcessGoneCrash } from './main-window-lifecycle-flags' import { handleGpuChildCrash } from './gpu-lifecycle' import { isGpuFallbackCrashCandidate } from '../crash-reporting/gpu-crash-fallback-decision' @@ -32,8 +32,12 @@ import { import { initializeMainProcessAutomations } from './main-process-automations' import { initializeMainProcessPlugins } from './main-process-plugins' import { collectWorktreeTrashSweepRoots, sweepStaleWorktreeTrash } from '../worktree-trash' +import { runAfterFirstWindowShown } from './first-window-deferral' import { logStartupMilestone } from './startup-diagnostics' +// Headless serve never opens a window, so the sweep still has to run off a timer there. +const WORKTREE_TRASH_SWEEP_FALLBACK_MS = 15_000 + export async function initializeReadyRuntimeServices(): Promise { const store = state.store if (!store) { @@ -74,12 +78,16 @@ export async function initializeReadyRuntimeServices(): Promise { state.emulatorBridge = new EmulatorBridge() runtime.setEmulatorBridge(state.emulatorBridge) // Why: worktree deletion renames the checkout aside and deletes it in the background, so a quit or - // crash mid-delete can leave the moved directory on disk. - void sweepStaleWorktreeTrash( - collectWorktreeTrashSweepRoots(store.getRepos(), store.getSettings()) - ).catch((error) => { - console.warn('[worktrees] Failed to sweep leftover worktree directories:', error) - }) + // crash mid-delete can leave the moved directory on disk. Why deferred: the sweep's recursive + // readdir/rm runs on the same libuv threadpool the window's first paint and worktree-catalog + // hydration are reading disk on, and nothing on the startup path consumes its result. + runAfterFirstWindowShown(() => { + void sweepStaleWorktreeTrash( + collectWorktreeTrashSweepRoots(store.getRepos(), store.getSettings()) + ).catch((error) => { + console.warn('[worktrees] Failed to sweep leftover worktree directories:', error) + }) + }, WORKTREE_TRASH_SWEEP_FALLBACK_MS) nativeTheme.themeSource = store.getSettings().theme ?? 'system' // Why (#16441): the real-home grant runs a codex app-server session. It stays // ordered before managed-hook reconciliation — an incapable host must re-arm @@ -122,9 +130,10 @@ export async function initializeReadyRuntimeServices(): Promise { console.warn('[agent-hooks] failed to reconcile managed hooks on startup:', error) ) } - // Why: process-gone metrics only see survivors; retain a recent whole-app - // snapshot for comparison in crash reports. - startPreGoneProcessMetricsSampling() + // Why: process-gone metrics only see survivors, and the gone-time host memory + // read lands after the corpse released its pages; both need a live pre-gone + // sample to compare against in crash reports. + startPreGoneCrashSampling() app.on('child-process-gone', (_event, details) => { recordProcessGoneCrash('child', details.type, details.reason, details.exitCode ?? null, { name: details.name, diff --git a/src/main/startup/main-process-ready.ts b/src/main/startup/main-process-ready.ts index e6d8d782e6e..835c1f1dd13 100644 --- a/src/main/startup/main-process-ready.ts +++ b/src/main/startup/main-process-ready.ts @@ -1,4 +1,5 @@ import { initializeMainProcessI18nAndMenu } from './main-process-i18n-menu' +import { mainProcessState as state } from './main-process-state' import { initializeReadyFoundation } from './main-process-ready-foundation' import { initializeReadyRuntimeServices } from './main-process-ready-runtime' import { @@ -12,6 +13,10 @@ export async function initializeMainProcessReady( ): Promise { await initializeReadyFoundation() await initializeReadyRuntimeServices() - await initializeMainProcessI18nAndMenu() - await initializeMainProcessRuntimeLaunch(options) + // Why concurrent: window creation reads no translated string and no menu item, and both the + // native menu and the tray only become reachable once the window shows — so serializing them + // ahead of openMainWindow only delayed the renderer (8 ms in English, more for a lazy locale). + const i18nAndMenuReady = initializeMainProcessI18nAndMenu() + state.mainProcessI18nReady = i18nAndMenuReady.catch(() => {}) + await Promise.all([i18nAndMenuReady, initializeMainProcessRuntimeLaunch(options)]) } diff --git a/src/main/startup/main-process-runtime-launch.ts b/src/main/startup/main-process-runtime-launch.ts index 9df28ecea8c..5f2691d6f31 100644 --- a/src/main/startup/main-process-runtime-launch.ts +++ b/src/main/startup/main-process-runtime-launch.ts @@ -24,6 +24,7 @@ import { import { prepareCodexRuntimeHomeForLaunch } from './codex-launch-preparation' import { prepareCodexSessionResumeForLaunch } from './codex-session-resume-launch' import { startWindowsDesktopBeforeShellPathReady } from './windows-desktop-shell-path-startup' +import { repairKnownPoisonedInstallDirBeforeWindow } from './windows-install-dir-acl-recovery' import { registerServeSignalHandlers } from './serve-signal-handlers' import { settleServeDesktopActivation } from './serve-desktop-activation' import { @@ -120,6 +121,9 @@ async function launchServeMode( runtimeRpc: OrcaRuntimeRpcServer, serveOptions: NonNullable> ): Promise { + // Why here: headless serve has no window to unblock, so keep the persisted proxy strictly + // ahead of every fetcher this phase can reach (relay, CLI install, RPC clients). + await state.initialProxyApplicationReady // Why: give managed WSL launchers a brief chance to migrate before headless PTYs go live, without slow repairs withholding all RPC readiness. logStartupMilestone('wsl-cli-barrier-start') await state.managedWslCliStartupBarrierReady @@ -226,8 +230,17 @@ async function launchDesktopMode( ) ]) if (!runtimeRpcStartResult.ok) { - void showRuntimeRpcStartupFailureDialog(win, runtimeRpcStartResult.error) + // Why gated: this dialog is the only launch-phase text read through translateMain, and i18n + // now settles alongside this phase — without the wait a non-English user could get the + // English defaultValue fallback. Still off the renderer's path (it is failure-only). + void state.mainProcessI18nReady.then(() => + showRuntimeRpcStartupFailureDialog(win, runtimeRpcStartResult.error) + ) } + // Why after the window and not before it: the default-session request guard already holds every + // fetcher until the persisted proxy lands, so this only has to keep the launch phase itself + // ordered ahead of the relay — it must not gate the renderer. + await state.initialProxyApplicationReady const cloudAuth = getOrcaCloudAuthConfig() if (cloudAuth.configured) { try { @@ -292,6 +305,17 @@ export async function initializeMainProcessRuntimeLaunch( // Why published: the renderer's git-environment barrier must fence on the same // generation the terminal startup services wait for, not a later re-read. state.shellPathReady = shellPathReady + // Why before any window: the poisoned install DACL kills the renderer at init, and + // the probe that detects it cannot finish before createMainWindow. Bounded, and a + // no-op (one absent-file read) unless a previous launch already recorded the verdict. + const aclGate = await repairKnownPoisonedInstallDirBeforeWindow({ + isServeMode: state.isServeMode || serveOptions !== null, + userDataPath: app.getPath('userData'), + appVersion: app.getVersion() + }) + if (aclGate !== 'not-marked' && aclGate !== 'skipped') { + logStartupMilestone('install-dir-acl-repair-blocking-done', { mode: aclGate }) + } let desktopWindow: BrowserWindow | null = null if (process.platform === 'win32' && app.isPackaged && !serveOptions) { const desktopStartup = startWindowsDesktopBeforeShellPathReady({ diff --git a/src/main/startup/main-process-state.ts b/src/main/startup/main-process-state.ts index d48219b461e..c88d5a66c48 100644 --- a/src/main/startup/main-process-state.ts +++ b/src/main/startup/main-process-state.ts @@ -100,6 +100,13 @@ export const mainProcessState = { // Electron with no error. Only the renderer's own pull proves the listener is live. markdownFileOpenListenerReady: false, firstWindowStartupServicesReady: Promise.resolve(), + // Why published: the default-session proxy must be applied before the first app-owned fetcher, + // but window creation has no reason to queue behind it (the request guard already fences it). + initialProxyApplicationReady: Promise.resolve(), + // Why published: i18n/menu init no longer precedes the launch phase, so the one launch-phase + // path that reads a translated string (the runtime-RPC startup failure dialog) waits on this. + // Never rejects: the phase's own failure is surfaced by initializeMainProcessReady. + mainProcessI18nReady: Promise.resolve(), managedWslCliReconciliationReady: Promise.resolve(), managedWslCliStartupBarrierReady: Promise.resolve(), // Why: the serve barrier fails open, so this state tells headless clients a WSL PTY launch may still race an un-migrated registration ('settled' = off-Windows no-op). diff --git a/src/main/startup/main-window-actions.ts b/src/main/startup/main-window-actions.ts index 96751d645db..585fa0a754e 100644 --- a/src/main/startup/main-window-actions.ts +++ b/src/main/startup/main-window-actions.ts @@ -12,7 +12,10 @@ import { ensureAutoUpdaterConfigured } from '../window/attach-main-window-servic import { focusExistingMainWindow, safelyRevealWindow } from '../window/focus-existing-window' import { mainProcessState as state } from './main-process-state' import { loadMainWindow } from '../window/createMainWindow' -import { describeInstallDirAclPoison } from './windows-install-dir-acl-recovery' +import { + describeInstallDirAclPoison, + isBlockingInstallDirAclRepairInFlight +} from './windows-install-dir-acl-recovery' import { presentRendererRecoveryPrompt } from '../window/renderer-recovery-prompt' // The window module injects this callback to avoid a cycle between actions and lifecycle code. @@ -28,6 +31,9 @@ export function focusExistingWindow(): void { app, getWindow: () => state.mainWindow, openWindow, + // Why: a 20s blank launch invites a second double-click, and icacls is rewriting + // the per-file DACLs a fresh renderer would read. The gated launch opens the window. + canOpenWindow: () => !isBlockingInstallDirAclRepairInFlight(), warn: console.warn }) } diff --git a/src/main/startup/main-window-controller.ts b/src/main/startup/main-window-controller.ts index 0d935d5f84a..63d84df763c 100644 --- a/src/main/startup/main-window-controller.ts +++ b/src/main/startup/main-window-controller.ts @@ -10,7 +10,10 @@ import { resolveConsent } from '../telemetry/consent' import { trackAppOpenedOnce } from '../telemetry/client' import { ensureWindowsUserDataAclGrant } from './windows-user-data-acl' import { probeWindowsInstallDirAcl } from './windows-install-dir-acl-probe' -import { startWindowsInstallDirAclRepairIfPoisoned } from './windows-install-dir-acl-recovery' +import { + noteWindowsInstallDirAclProbePending, + startWindowsInstallDirAclRepairIfPoisoned +} from './windows-install-dir-acl-recovery' import { logStartupMilestone } from './startup-diagnostics' import { notifyMainWindowBecameVisible } from '../window/main-window-visibility' import { setTrayAttention } from '../tray/system-tray' @@ -74,7 +77,7 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} }) // Why here: read-only, and the install DACL is the one thing a 0x80000003 // child death cannot tell us about itself. See electron/electron#51761. - probeWindowsInstallDirAcl({ + const probeDispatched = probeWindowsInstallDirAcl({ isServeMode: state.isServeMode, onDone: (data) => startWindowsInstallDirAclRepairIfPoisoned(data, { @@ -83,6 +86,12 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} appVersion: app.getVersion() }) }) + // Why gated on the dispatch: the probe is once-per-process while openMainWindow + // re-runs on every reopen, so arming this again would wait on a verdict that + // already landed — and drop every GPU crash for the grace window. + if (probeDispatched) { + noteWindowsInstallDirAclProbePending() + } } const window = createMainWindow(store, { getIsQuitting: () => state.isQuitting, diff --git a/src/main/startup/main-window-core-services.ts b/src/main/startup/main-window-core-services.ts index 759a4b2b100..d3ece383ee3 100644 --- a/src/main/startup/main-window-core-services.ts +++ b/src/main/startup/main-window-core-services.ts @@ -15,6 +15,7 @@ import { import { prepareCodexRuntimeHomeForLaunch } from './codex-launch-preparation' import { prepareCodexSessionResumeForLaunch } from './codex-session-resume-launch' import { isRecoveryReloadInFlight } from './main-window-lifecycle-flags' +import { RELAY_HOST_CLOSE_REASON } from '../../shared/relay-host-close-reason' export function attachMainWindowCoreServices( window: BrowserWindow, @@ -90,7 +91,10 @@ export function attachMainWindowCoreServices( }) }, onOrcaProfileAuthMutation: () => state.desktopRelayService?.authMutated(), - onBeforeOrcaProfileSignOut: () => state.desktopRelayService?.fenceAndCloseNow() + // Sign-out is the one fence a paired phone can be told about; quit and + // relaunch above stay reasonless so a restart never reads as signed out. + onBeforeOrcaProfileSignOut: () => + state.desktopRelayService?.fenceAndCloseNow(RELAY_HOST_CLOSE_REASON.SIGNED_OUT) }, state.pluginService ?? undefined, state.pluginMarketplaceService && state.pluginMarketplaceInstaller diff --git a/src/main/startup/pre-gone-crash-sampling-wiring.test.ts b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts new file mode 100644 index 00000000000..2a8e008c0b3 --- /dev/null +++ b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts @@ -0,0 +1,49 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * Guards the one line that arms pre-gone crash sampling. + * + * That branch is pure instrumentation, so this line is the whole of its value in + * the shipped app: deleting it left all 691 tests across `src/main/crash-reporting/` + * and `src/main/startup/` green while every crash report silently lost its only + * host reading taken before the dying process returned its pages. + * + * Source-level because that is the property: the sampler is armed once inside the + * ready-phase composition, which has no runtime seam to assert against. + */ +describe('pre-gone crash sampling startup wiring', () => { + // Why normalize: the indent anchors below are `\n`-prefixed, and nothing pins + // src/**/*.ts to LF, so a CRLF Windows checkout would fail them spuriously. + const readSource = (name: string): string => + readFileSync(join(process.cwd(), 'src/main/startup', name), 'utf8').replace(/\r\n/g, '\n') + + const readyRuntimeSource = readSource('main-process-ready-runtime.ts') + const readySource = readSource('main-process-ready.ts') + + const READY_ENTRY = 'export async function initializeReadyRuntimeServices(' + // Why the entry's body and not the file: the call satisfies a whole-file grep + // just as well from a sibling export nothing calls, which arms nothing. + const readyRuntimeEntryBody = readyRuntimeSource + .slice(readyRuntimeSource.indexOf(READY_ENTRY) + READY_ENTRY.length) + .split('\nexport ')[0] + + it('arms the sampler unconditionally inside the function app readiness runs', () => { + expect(readyRuntimeSource).toContain( + "import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics'" + ) + expect(readyRuntimeSource).toContain(READY_ENTRY) + expect(readyRuntimeEntryBody.split('startPreGoneCrashSampling()').length - 1).toBe(1) + // Why pin the indent: the call also matches as the body of an added + // `if (...)` guard, which keeps every other assertion here true while the + // sampler silently stops arming on most startups. + expect(readyRuntimeEntryBody).toContain('\n startPreGoneCrashSampling()') + + // ...and that this really is the function app readiness runs. + expect(readySource).toContain( + "import { initializeReadyRuntimeServices } from './main-process-ready-runtime'" + ) + expect(readySource).toContain('\n await initializeReadyRuntimeServices()') + }) +}) diff --git a/src/main/startup/windows-install-dir-acl-poison-marker.ts b/src/main/startup/windows-install-dir-acl-poison-marker.ts new file mode 100644 index 00000000000..46035e04e74 --- /dev/null +++ b/src/main/startup/windows-install-dir-acl-poison-marker.ts @@ -0,0 +1,77 @@ +import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { join } from 'node:path' + +/** + * "This install directory was found poisoned and has not been proven healthy since." + * + * Why a separate marker from `windows-install-dir-acl-repair.json`: that one is + * written after an attempt finishes, so a launch the poison kills mid-repair + * leaves no state at all and the next launch repeats the whole late-repair dance. + * This one is written the moment the probe's verdict lands, and it is the only + * thing that lets a later launch know it is poisoned *before* it creates a window + * — the probe itself cannot answer that early. Same tiny synchronous-JSON shape + * as `gpu-fallback-marker.ts`, for the same reason. + */ + +export const WINDOWS_INSTALL_DIR_ACL_POISON_MARKER_FILE = 'windows-install-dir-acl-poison.json' +export const WINDOWS_INSTALL_DIR_ACL_POISON_SCHEME_VERSION = 1 + +type PoisonMarker = { + schemeVersion: number + installDir: string + appVersion: string + detectedAt: number +} + +function markerPath(userDataPath: string): string { + return join(userDataPath, WINDOWS_INSTALL_DIR_ACL_POISON_MARKER_FILE) +} + +/** Keyed on both: a reinstall elsewhere or an update ships files with a fresh DACL. */ +export function hasInstallDirAclPoisonMarker( + userDataPath: string, + installDir: string, + appVersion: string +): boolean { + try { + const parsed = JSON.parse(readFileSync(markerPath(userDataPath), 'utf-8')) as + | Partial + | undefined + return ( + parsed?.schemeVersion === WINDOWS_INSTALL_DIR_ACL_POISON_SCHEME_VERSION && + parsed.installDir === installDir && + parsed.appVersion === appVersion + ) + } catch { + return false // missing or corrupt -> treat the install as healthy + } +} + +export function writeInstallDirAclPoisonMarker( + userDataPath: string, + installDir: string, + appVersion: string +): void { + const marker: PoisonMarker = { + schemeVersion: WINDOWS_INSTALL_DIR_ACL_POISON_SCHEME_VERSION, + installDir, + appVersion, + detectedAt: Date.now() + } + try { + if (!existsSync(userDataPath)) { + mkdirSync(userDataPath, { recursive: true }) + } + writeFileSync(markerPath(userDataPath), JSON.stringify(marker)) + } catch { + // Best effort: without it the next launch just falls back to today's late repair. + } +} + +export function clearInstallDirAclPoisonMarker(userDataPath: string): void { + try { + rmSync(markerPath(userDataPath), { force: true }) + } catch { + // Best effort; a stale marker only costs one redundant icacls pass. + } +} diff --git a/src/main/startup/windows-install-dir-acl-probe.ts b/src/main/startup/windows-install-dir-acl-probe.ts index 42db3ee3d10..25d1e581c9b 100644 --- a/src/main/startup/windows-install-dir-acl-probe.ts +++ b/src/main/startup/windows-install-dir-acl-probe.ts @@ -211,13 +211,16 @@ export function resetWindowsInstallDirAclProbeForTest(): void { * Fire-and-forget; returns before any spawn. win32 only — no spawn and no fs I/O * anywhere else. Called from openMainWindow, which runs after initObservability, * so the durable record also emits a span into the diagnostics bundle. + * + * Returns whether THIS call dispatched the probe: openMainWindow re-runs on every + * reopen, and only a dispatch will ever produce an `onDone`. */ -export function probeWindowsInstallDirAcl(options: WindowsInstallDirAclProbeOptions = {}): void { +export function probeWindowsInstallDirAcl(options: WindowsInstallDirAclProbeOptions = {}): boolean { if ((options.platform ?? process.platform) !== 'win32' || options.isServeMode === true) { - return + return false } if (probeStarted) { - return + return false } probeStarted = true // Why the try: this runs inline in openMainWindow, so anything thrown here @@ -229,4 +232,5 @@ export function probeWindowsInstallDirAcl(options: WindowsInstallDirAclProbeOpti } catch { // Nothing left to report to that would not throw again. } + return true } diff --git a/src/main/startup/windows-install-dir-acl-recovery.test.ts b/src/main/startup/windows-install-dir-acl-recovery.test.ts index 2b5d00ff40a..7641cdc0700 100644 --- a/src/main/startup/windows-install-dir-acl-recovery.test.ts +++ b/src/main/startup/windows-install-dir-acl-recovery.test.ts @@ -1,18 +1,34 @@ -import { mkdtempSync } from 'node:fs' +import { mkdtempSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { beforeEach, describe, expect, it } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' import type { ProcessResult, ProcessSpec } from '../../shared/child-process/run-process' +import type { CrashReportBreadcrumbData } from '../../shared/crash-reporting' +import { readActiveGpuFallbackMarker, writeGpuFallbackMarker } from './gpu-fallback-marker' import { probeWindowsInstallDirAcl, resetWindowsInstallDirAclProbeForTest } from './windows-install-dir-acl-probe' +import { + hasInstallDirAclPoisonMarker, + writeInstallDirAclPoisonMarker +} from './windows-install-dir-acl-poison-marker' import { describeInstallDirAclPoison, + isBlockingInstallDirAclRepairInFlight, + isInstallDirAclRepairExhausted, + isInstallDirAclSuspect, + noteWindowsInstallDirAclProbePending, + repairKnownPoisonedInstallDirBeforeWindow, resetWindowsInstallDirAclRecoveryForTest, - startWindowsInstallDirAclRepairIfPoisoned + startWindowsInstallDirAclRepairIfPoisoned, + type WindowsInstallDirAclRecoveryOptions } from './windows-install-dir-acl-recovery' -import { resetWindowsInstallDirAclRepairForTest } from './windows-install-dir-package-acl-repair' +import { + resetWindowsInstallDirAclRepairForTest, + WINDOWS_INSTALL_DIR_ACL_REPAIR_MARKER_FILE, + WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION +} from './windows-install-dir-package-acl-repair' import { ALL_PACKAGES_ACE, fakeIcaclsSpawn, @@ -26,6 +42,8 @@ import { const INSTALL_DIR = 'C:\\Users\\neil\\AppData\\Local\\Programs\\orca' const APP_VERSION = '1.4.184' +type Runner = (spec: ProcessSpec) => Promise + /** * Drives the production path: the real probe hands its verdict to the real gate, * which decides whether icacls ever runs. Only the two process seams are faked. @@ -185,3 +203,778 @@ describe('describeInstallDirAclPoison', () => { expect(describeInstallDirAclPoison()?.detail).toContain('repairing the permissions now') }) }) + +const POISON_VERDICT: CrashReportBreadcrumbData = { + status: 'ok', + matchesPoisonSignature: true, + wellKnownNameCheckReliable: true +} +const GPU_ENV = { appVersion: APP_VERSION, electronVersion: '43.4.1', platform: 'win32' } as const + +function recoveryOptions(userDataPath: string, run: Runner): WindowsInstallDirAclRecoveryOptions { + return { + platform: 'win32', + installDir: INSTALL_DIR, + appVersion: APP_VERSION, + userDataPath, + runProcessFn: run as never, + recordBreadcrumb: () => undefined + } +} + +/** icacls' real success summary, as the repair's parser expects it. */ +const okRun: Runner = async () => ({ + code: 0, + signal: null, + stdout: 'Successfully processed 3200 files; Failed processing 0 files', + stderr: '', + timedOut: false +}) + +describe('install-dir ACL repair vs the GPU safe-graphics marker', () => { + beforeEach(() => { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + }) + + it('clears the sticky safe-graphics marker once the real cause is repaired', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gpu-')) + // The machine is in the reproduced state: the poisoned install DACL killed the + // GPU child three times, so Orca latched safe graphics for this build. + writeGpuFallbackMarker( + userDataPath, + { engagedAt: Date.now(), crashesInWindow: 3, userConfirmed: false }, + GPU_ENV + ) + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)).not.toBeNull() + + await new Promise((resolve) => { + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, { + ...recoveryOptions(userDataPath, okRun), + // Settles after the repair's own setImmediate hop and its two icacls passes. + recordBreadcrumb: () => { + setTimeout(resolve, 0) + return undefined + } + }) + }) + + expect(describeInstallDirAclPoison()?.detail).toContain('repaired the permissions') + // The GPU child deaths were never a driver fault, so safe graphics — and the + // --in-process-gpu launch that hides the next crash's evidence — must not outlive the repair. + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)).toBeNull() + }) + + // "Keep safe graphics" is a durable user choice with its own reasons; the repair + // only retires the latch Orca engaged on its own. + it('leaves a user-confirmed safe-graphics marker alone', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gpu-')) + writeGpuFallbackMarker( + userDataPath, + { engagedAt: Date.now(), crashesInWindow: 3, userConfirmed: true }, + GPU_ENV + ) + + await new Promise((resolve) => { + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, { + ...recoveryOptions(userDataPath, okRun), + recordBreadcrumb: () => { + setTimeout(resolve, 0) + return undefined + } + }) + }) + + expect(describeInstallDirAclPoison()?.detail).toContain('repaired the permissions') + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)?.userConfirmed).toBe(true) + }) +}) + +describe('isInstallDirAclSuspect', () => { + beforeEach(() => { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + }) + + it('is false when nothing has suggested the install DACL is involved', () => { + expect(isInstallDirAclSuspect()).toBe(false) + }) + + // The GPU child dies ~74ms in and the probe answers 0.9-3.0s later, so "no verdict + // yet" is the entire window in which the misdiagnosis happens. + it('holds while the probe verdict is outstanding, and releases on a clean verdict', () => { + noteWindowsInstallDirAclProbePending() + expect(isInstallDirAclSuspect()).toBe(true) + + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: false }, + recoveryOptions(mkdtempSync(join(tmpdir(), 'orca-acl-suspect-')), okRun) + ) + expect(isInstallDirAclSuspect()).toBe(false) + }) + + it('releases once the wait exceeds the grace window, so a silent probe cannot pin it', () => { + noteWindowsInstallDirAclProbePending() + expect(isInstallDirAclSuspect(Date.now() + 14_000)).toBe(true) + expect(isInstallDirAclSuspect(Date.now() + 16_000)).toBe(false) + }) + + it('holds through a repair that failed, and releases once one succeeds', async () => { + const failing: Runner = async () => ({ + code: 5, + signal: null, + stdout: '', + stderr: 'Access is denied.', + timedOut: false + }) + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-suspect-')) + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, failing) + ) + expect(isInstallDirAclSuspect()).toBe(true) + await vi.waitFor(() => expect(describeInstallDirAclPoison()?.detail).toContain('could not')) + // Still suspect: the tree is proven poisoned, and safe graphics does not rescue it. + expect(isInstallDirAclSuspect()).toBe(true) + + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(mkdtempSync(join(tmpdir(), 'orca-acl-suspect-')), okRun) + ) + await vi.waitFor(() => expect(isInstallDirAclSuspect()).toBe(false)) + }) +}) + +describe('repairKnownPoisonedInstallDirBeforeWindow', () => { + beforeEach(() => { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + }) + + it('costs a healthy machine one absent-file read and no icacls', async () => { + const specs: ProcessSpec[] = [] + const run: Runner = async (spec) => { + specs.push(spec) + return okRun(spec) + } + const mode = await repairKnownPoisonedInstallDirBeforeWindow( + recoveryOptions(mkdtempSync(join(tmpdir(), 'orca-acl-gate-')), run) + ) + expect(mode).toBe('not-marked') + expect(specs).toHaveLength(0) + }) + + // The crash this fixes: launch 1 detects the poison but createMainWindow already + // ran, so the renderer is dead before icacls is spawned. Launch 2 must not repeat it. + it('repairs a launch that a previous one recorded as poisoned, before returning', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-')) + // Launch 1: the probe reports poison and the app dies mid-repair. + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + + // Launch 2. + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + writeGpuFallbackMarker( + userDataPath, + { engagedAt: Date.now(), crashesInWindow: 3, userConfirmed: false }, + GPU_ENV + ) + const specs: ProcessSpec[] = [] + const run: Runner = async (spec) => { + specs.push(spec) + return okRun(spec) + } + const mode = await repairKnownPoisonedInstallDirBeforeWindow(recoveryOptions(userDataPath, run)) + expect(mode).toBe('repaired') + // Both passes have already run by the time the window may be created. + expect(specs.map((spec) => spec.args?.[2])).toEqual([ + '*S-1-15-2-2:(OI)(CI)(RX)', + '*S-1-15-2-2:(RX)' + ]) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(false) + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)).toBeNull() + }) + + it('gives up on its budget rather than holding the window open forever', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-')) + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + + const mode = await repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner), + timeoutMs: 20 + }) + expect(mode).toBe('timeout') + }) + + it('is a no-op off win32 and in serve mode', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-')) + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + + expect( + await repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userDataPath, okRun), + platform: 'darwin' + }) + ).toBe('skipped') + expect( + await repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userDataPath, okRun), + isServeMode: true + }) + ).toBe('skipped') + }) + + it('retires the marker when a later probe reports the install clean', () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-')) + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + + resetWindowsInstallDirAclRecoveryForTest() + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: false }, + recoveryOptions(userDataPath, okRun) + ) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(false) + }) + + // An unreadable DACL is not evidence of health; forgetting the verdict there would + // hand the next launch straight back to the crash it already recorded. + it('keeps the marker when the probe could not read the DACL', () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-')) + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + resetWindowsInstallDirAclRecoveryForTest() + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'failed', reason: 'all-targets-unreadable' }, + recoveryOptions(userDataPath, okRun) + ) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + }) + + // The gate-timed-out ordering. The gate's budget is 20s while the tree grant's own cap is + // 120s, so icacls routinely outlives the gate: the window opens, and this launch's probe + // reads the tree POISONED while that repair is still in flight. When the orphaned icacls + // then claims success -- exit 0, and on a localized Windows no parsable failure summary to + // contradict it -- the claim must not outrank a reading taken after it was dispatched. + // Otherwise this launch deletes the poison marker that arms every later gate, un-suspects + // the tree so --in-process-gpu can engage, clears the safe-graphics marker, and tells the + // user their permissions are fixed. + it('does not let a timed-out gate repair outrank a poison reading taken after it', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-timeout-')) + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + writeGpuFallbackMarker( + userDataPath, + { engagedAt: Date.now(), crashesInWindow: 3, userConfirmed: false }, + GPU_ENV + ) + + let releaseIcacls: () => void = () => undefined + const stalled = new Promise((resolve) => { + releaseIcacls = resolve + }) + let repairReported: () => void = () => undefined + const reported = new Promise((resolve) => { + repairReported = resolve + }) + + const mode = await repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userDataPath, async (spec) => { + await stalled + return okRun(spec) + }), + recordBreadcrumb: () => { + setTimeout(repairReported, 0) + return undefined + }, + timeoutMs: 20 + }) + expect(mode).toBe('timeout') + + // The window is open now, and this launch's own probe reads the tree still poisoned. + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, recoveryOptions(userDataPath, okRun)) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + + releaseIcacls() + await reported + + expect(isInstallDirAclSuspect()).toBe(true) + expect(describeInstallDirAclPoison()?.detail).toContain('could not repair them') + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)).not.toBeNull() + }) + + // The other ordering of the same two events: the orphaned icacls exits 0 and clears the + // safe-graphics marker BEFORE the probe reads the tree still poisoned. The disproof must + // give the marker back, or the two orderings disagree about the same launch. + it('restores the safe-graphics marker when the probe disproves a timed-out gate repair', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-timeout-restore-')) + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + writeGpuFallbackMarker( + userDataPath, + { engagedAt: Date.now(), crashesInWindow: 3, userConfirmed: false }, + GPU_ENV + ) + + let releaseIcacls: () => void = () => undefined + const stalled = new Promise((resolve) => { + releaseIcacls = resolve + }) + let repairReported: () => void = () => undefined + const reported = new Promise((resolve) => { + repairReported = resolve + }) + + const mode = await repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userDataPath, async (spec) => { + await stalled + return okRun(spec) + }), + recordBreadcrumb: () => { + setTimeout(repairReported, 0) + return undefined + }, + timeoutMs: 20 + }) + expect(mode).toBe('timeout') + + // The orphan claims success first; the claim is believed and takes the marker with it. + releaseIcacls() + await reported + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)).toBeNull() + + // Then this launch's probe reads the tree still poisoned. + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, recoveryOptions(userDataPath, okRun)) + + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)?.userConfirmed).toBe(false) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + expect(isInstallDirAclSuspect()).toBe(true) + }) +}) + +// The repair marker matches whatever the outcome, so on its own 'marker-hit' cannot tell a +// finished tree from one Orca gave up on. Both callers hold outstanding poison evidence — +// this launch's probe reading, or the persisted marker that armed the gate — so a recorded +// success never stands in for the repair, and 'marker-hit' only ever means budget spent. +describe('a repair marker recording a completed repair', () => { + beforeEach(() => { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + }) + + /** One launch: fresh module latches, then the gate runs against the userData on disk. */ + async function gateLaunch( + userDataPath: string, + run: Runner + ): Promise>> { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + return repairKnownPoisonedInstallDirBeforeWindow(recoveryOptions(userDataPath, run)) + } + + // The three-launch shape the gate exists for, and the one it used to disarm itself on: + // launch 1 repairs; the tree is re-poisoned (an installer, AV, or an icacls run that + // silently no-opped); launch 2's probe records the poison but Chromium FATALs before the + // repair can write its marker. Launch 3's gate then meets a poison marker and a repair + // marker claiming success. Treating that as 'repaired' ran no icacls, deleted the poison + // marker so no later gate ever fires again, un-suspected the tree so --in-process-gpu + // could engage, and told the user their permissions were fixed. + it('re-runs icacls when a poison marker outlives it', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-repaired-hit-')) + + // Launch 1: the gate repairs the tree and retires the poison marker. + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + expect(await gateLaunch(userDataPath, okRun)).toBe('repaired') + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(false) + + // Launch 2: the probe reads the tree as poisoned again; the process dies mid-repair, + // so the repair marker still records launch 1's success. + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + + // Launch 3: the gate must repair, not congratulate itself on launch 1's work. + const spent: ProcessSpec[] = [] + const mode = await gateLaunch(userDataPath, async (spec) => { + spent.push(spec) + return { code: 5, signal: null, stdout: '', stderr: 'Access is denied.', timedOut: false } + }) + expect(mode).toBe('failed') + expect(spent.map((spec) => spec.args?.[2])).toEqual([ + '*S-1-15-2-2:(OI)(CI)(RX)', + '*S-1-15-2-2:(RX)' + ]) + expect(isInstallDirAclSuspect()).toBe(true) + expect(describeInstallDirAclPoison()?.detail).toContain('could not repair them') + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + }) + + // The budget is what stops the retry above running forever; a spent one must still read + // as "Orca could not fix this", never as a repair it never made. + it('does not let the gate report a spent budget as a repair', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gate-budget-')) + writeFileSync( + join(userDataPath, WINDOWS_INSTALL_DIR_ACL_REPAIR_MARKER_FILE), + JSON.stringify({ + schemeVersion: WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION, + installDir: INSTALL_DIR, + appVersion: APP_VERSION, + attemptedAt: Date.now(), + outcome: 'repaired', + attempts: 3 + }) + ) + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + const spent: ProcessSpec[] = [] + const mode = await gateLaunch(userDataPath, async (spec) => { + spent.push(spec) + return okRun(spec) + }) + expect(mode).toBe('marker-hit') + expect(spent).toHaveLength(0) + expect(isInstallDirAclSuspect()).toBe(true) + expect(isInstallDirAclRepairExhausted()).toBe(true) + expect(describeInstallDirAclPoison()?.detail).toContain('could not repair them') + // Still armed: nothing has proven this tree healthy, so a later launch still gates. + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + }) + + // The probe reads the tree AFTER the pre-window gate has finished with it, so a signature + // still matching means the repair never landed however icacls exited. + it('is overruled by a probe that reads the tree poisoned after the gate repaired it', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-noop-icacls-')) + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + expect( + await repairKnownPoisonedInstallDirBeforeWindow(recoveryOptions(userDataPath, okRun)) + ).toBe('repaired') + + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, recoveryOptions(userDataPath, okRun)) + + expect(isInstallDirAclSuspect()).toBe(true) + expect(isInstallDirAclRepairExhausted()).toBe(false) + expect(describeInstallDirAclPoison()?.detail).toContain('could not repair them') + // Re-armed: the next launch gates before it opens a window it cannot render. + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + }) + + // The gate's 'repaired' is icacls's exit claim, not a reading of the tree — and the GPU + // children die 48-1373ms after window creation while the probe answers 0.9-3.0s in. + // Un-suspecting the tree on the claim alone opens exactly that interval to + // --in-process-gpu on a tree safe graphics cannot rescue. + it('keeps a gate-repaired tree suspect until this launch probe has read it', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-provisional-')) + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + expect( + await repairKnownPoisonedInstallDirBeforeWindow(recoveryOptions(userDataPath, okRun)) + ).toBe('repaired') + + // openMainWindow dispatches the probe: the reading is outstanding. + noteWindowsInstallDirAclProbePending() + expect(isInstallDirAclSuspect()).toBe(true) + // A probe that never answers releases at the grace window, like any pending verdict. + expect(isInstallDirAclSuspect(Date.now() + 15_000)).toBe(false) + + // A clean reading corroborates the claim and releases immediately. + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: false }, + recoveryOptions(userDataPath, okRun) + ) + expect(isInstallDirAclSuspect()).toBe(false) + }) + + // A disproved claim owes back everything it took on the false premise — the poison + // marker (above) and the safe-graphics marker, or the machine relaunches hardware + // accelerated into the re-armed gate and FATALs before that gate can finish. + it('restores the safe-graphics marker a disproved repair claim cleared', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-gpu-restore-')) + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + writeGpuFallbackMarker( + userDataPath, + { engagedAt: Date.now(), crashesInWindow: 3, userConfirmed: false }, + GPU_ENV + ) + expect( + await repairKnownPoisonedInstallDirBeforeWindow(recoveryOptions(userDataPath, okRun)) + ).toBe('repaired') + // The claim was believed, so the marker went with it. + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)).toBeNull() + + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, recoveryOptions(userDataPath, okRun)) + + expect(readActiveGpuFallbackMarker(userDataPath, GPU_ENV)?.userConfirmed).toBe(false) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + }) + + // The opposite evidence: the probe has just READ this tree and found it poisoned, so a + // marker claiming success describes a tree that was re-poisoned, or an icacls run that + // silently no-opped. Reporting 'repaired' there runs no icacls, deletes the poison marker + // that arms the next launch's gate, un-suspects the tree so --in-process-gpu can engage, + // and tells the user their permissions are fixed. + it('re-runs icacls when a fresh probe verdict contradicts the repaired marker', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-repoisoned-')) + writeInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION) + expect( + await repairKnownPoisonedInstallDirBeforeWindow(recoveryOptions(userDataPath, okRun)) + ).toBe('repaired') + + // Next launch: the gate is disarmed, and the probe reads the same tree as poisoned. + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + const spent: ProcessSpec[] = [] + const failing: Runner = async (spec) => { + spent.push(spec) + return { code: 5, signal: null, stdout: '', stderr: 'Access is denied.', timedOut: false } + } + await new Promise((resolve) => { + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, { + ...recoveryOptions(userDataPath, failing), + recordBreadcrumb: () => { + setTimeout(resolve, 0) + return undefined + } + }) + }) + + expect(spent.map((spec) => spec.args?.[2])).toEqual([ + '*S-1-15-2-2:(OI)(CI)(RX)', + '*S-1-15-2-2:(RX)' + ]) + expect(isInstallDirAclSuspect()).toBe(true) + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + expect(describeInstallDirAclPoison()?.detail).toContain('could not repair them') + }) + + // The contradiction re-opens the budget, it does not remove it: a tree that has spent + // every attempt must not re-spawn icacls on every launch forever. + it('still stops at the attempt budget when the probe keeps reporting poison', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-repoisoned-budget-')) + writeFileSync( + join(userDataPath, WINDOWS_INSTALL_DIR_ACL_REPAIR_MARKER_FILE), + JSON.stringify({ + schemeVersion: WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION, + installDir: INSTALL_DIR, + appVersion: APP_VERSION, + attemptedAt: Date.now(), + outcome: 'repaired', + attempts: 3 + }) + ) + const spent: ProcessSpec[] = [] + const run: Runner = async (spec) => { + spent.push(spec) + return okRun(spec) + } + await new Promise((resolve) => { + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, { + ...recoveryOptions(userDataPath, run), + recordBreadcrumb: () => { + setTimeout(resolve, 0) + return undefined + } + }) + }) + + expect(spent).toHaveLength(0) + // Nothing was repaired, so the user still gets the commands and the gate stays armed. + expect(isInstallDirAclSuspect()).toBe(true) + expect(describeInstallDirAclPoison()?.detail).toContain('could not repair them') + expect(hasInstallDirAclPoisonMarker(userDataPath, INSTALL_DIR, APP_VERSION)).toBe(true) + }) +}) + +describe('a clean probe verdict', () => { + beforeEach(() => { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + }) + + // The launch this covers: the repair budget is spent, so the gate can only report + // 'marker-hit' — and then the probe reads the tree and finds it healthy. + it('retires a verdict the gate could no longer act on', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-clean-')) + writeFileSync( + join(userDataPath, WINDOWS_INSTALL_DIR_ACL_REPAIR_MARKER_FILE), + JSON.stringify({ + schemeVersion: WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION, + installDir: INSTALL_DIR, + appVersion: APP_VERSION, + attemptedAt: Date.now(), + outcome: 'failed', + attempts: 3 + }) + ) + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + expect( + await repairKnownPoisonedInstallDirBeforeWindow(recoveryOptions(userDataPath, okRun)) + ).toBe('marker-hit') + expect(isInstallDirAclSuspect()).toBe(true) + + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: false }, + recoveryOptions(userDataPath, okRun) + ) + // Neither the driver fallback stays suppressed nor does the dialog accuse a healthy folder. + expect(isInstallDirAclSuspect()).toBe(false) + expect(describeInstallDirAclPoison()).toBeNull() + }) + + // The probe answers while the repair is still walking the tree: 'failed' from a + // repair with nothing left to fix must not re-accuse an install just read clean. + it('outranks a repair verdict that lands after it', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-clean-')) + const failing: Runner = async () => ({ + code: 5, + signal: null, + stdout: '', + stderr: 'Access is denied.', + timedOut: false + }) + let repairSettled = false + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, { + ...recoveryOptions(userDataPath, failing), + recordBreadcrumb: () => { + repairSettled = true + return undefined + } + }) + startWindowsInstallDirAclRepairIfPoisoned( + { status: 'ok', matchesPoisonSignature: false }, + recoveryOptions(userDataPath, okRun) + ) + await vi.waitFor(() => expect(repairSettled).toBe(true)) + expect(isInstallDirAclSuspect()).toBe(false) + expect(describeInstallDirAclPoison()).toBeNull() + }) +}) + +describe('the probe-pending grace window', () => { + beforeEach(() => { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + }) + + // openMainWindow re-runs on every tray/second-instance reopen while the probe is + // once-per-process, so a re-arm would wait 15s on a verdict that already landed + // and drop every GPU child crash in between. + it('is armed by a dispatched probe only, so a reopen cannot re-arm it', async () => { + const probeArgs = { + platform: 'win32' as const, + installDir: INSTALL_DIR, + fileExists: () => false, + spawnFn: fakeIcaclsSpawn((target) => icaclsDacl(target, [RESTRICTED_PACKAGES_ACE])).spawnFn, + recordBreadcrumb: () => undefined + } + let settleVerdict: () => void = () => undefined + const verdict = new Promise((resolve) => (settleVerdict = resolve)) + // Launch, wired exactly as main-window-controller wires it. + const dispatched = probeWindowsInstallDirAcl({ + ...probeArgs, + onDone: (data) => { + startWindowsInstallDirAclRepairIfPoisoned( + data, + recoveryOptions(mkdtempSync(join(tmpdir(), 'orca-acl-rearm-')), okRun) + ) + settleVerdict() + } + }) + if (dispatched) { + noteWindowsInstallDirAclProbePending() + } + expect(dispatched).toBe(true) + expect(isInstallDirAclSuspect()).toBe(true) + await verdict + expect(isInstallDirAclSuspect()).toBe(false) + + // Reopen: the probe declines, so nothing arms the grace window again. + const reopened = probeWindowsInstallDirAcl({ ...probeArgs, onDone: () => undefined }) + if (reopened) { + noteWindowsInstallDirAclProbePending() + } + expect(reopened).toBe(false) + expect(isInstallDirAclSuspect()).toBe(false) + expect(isInstallDirAclSuspect(Date.now() + 14_000)).toBe(false) + }) +}) + +describe('isBlockingInstallDirAclRepairInFlight', () => { + beforeEach(() => { + resetWindowsInstallDirAclProbeForTest() + resetWindowsInstallDirAclRepairForTest() + resetWindowsInstallDirAclRecoveryForTest() + }) + + it('is false on a healthy machine and clears once the gate returns', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-inflight-')) + expect(isBlockingInstallDirAclRepairInFlight()).toBe(false) + + startWindowsInstallDirAclRepairIfPoisoned( + POISON_VERDICT, + recoveryOptions(userDataPath, (() => new Promise(() => undefined)) as Runner) + ) + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + + let inFlightDuringRepair = false + const gate = repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userDataPath, async (spec) => { + inFlightDuringRepair = isBlockingInstallDirAclRepairInFlight() + return okRun(spec) + }), + timeoutMs: 5_000 + }) + expect(await gate).toBe('repaired') + expect(inFlightDuringRepair).toBe(true) + expect(isBlockingInstallDirAclRepairInFlight()).toBe(false) + }) + + // A second entry has no `onDone` coming, so waiting out the 20s budget for it + // would hold the window closed for nothing. + it('returns immediately when the once-per-process repair already ran', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-inflight-')) + startWindowsInstallDirAclRepairIfPoisoned(POISON_VERDICT, recoveryOptions(userDataPath, okRun)) + resetWindowsInstallDirAclRecoveryForTest() + + const mode = await repairKnownPoisonedInstallDirBeforeWindow({ + ...recoveryOptions(userDataPath, okRun), + timeoutMs: 30_000 + }) + expect(mode).toBe('skipped') + expect(isBlockingInstallDirAclRepairInFlight()).toBe(false) + }) +}) diff --git a/src/main/startup/windows-install-dir-acl-recovery.ts b/src/main/startup/windows-install-dir-acl-recovery.ts index 0aa6870192e..c251bda9267 100644 --- a/src/main/startup/windows-install-dir-acl-recovery.ts +++ b/src/main/startup/windows-install-dir-acl-recovery.ts @@ -1,6 +1,17 @@ import { dirname } from 'node:path' import type { CrashReportBreadcrumbData } from '../../shared/crash-reporting' import { logStartupMilestone } from './startup-diagnostics' +import { + clearGpuFallbackMarker, + readGpuFallbackMarker, + writeGpuFallbackMarker, + type GpuFallbackMarker +} from './gpu-fallback-marker' +import { + clearInstallDirAclPoisonMarker, + hasInstallDirAclPoisonMarker, + writeInstallDirAclPoisonMarker +} from './windows-install-dir-acl-poison-marker' import { buildInstallDirAclRepairCommands, isInstallDirAclPoisonVerdict, @@ -25,10 +36,165 @@ export type WindowsInstallDirAclRecoveryOptions = Omit void>() + +function settleVerdictWaiters(): void { + // `wake` deletes only itself, which is safe to do on the entry being visited. + for (const wake of verdictWaiters) { + wake() + } + verdictWaiters.clear() +} export function resetWindowsInstallDirAclRecoveryForTest(): void { poison = null + probePendingSince = null + installDirReadClean = false + installDirReadPoisonedMidRepair = false + gpuMarkerClearedByRepairClaim = null + blockingRepairInFlight = false + settleVerdictWaiters() +} + +/** Call when the install-DACL probe is dispatched: its verdict is not in yet. */ +export function noteWindowsInstallDirAclProbePending(): void { + probePendingSince = Date.now() +} + +/** + * Resolves when the probe's verdict lands, or when its grace window runs out. + * For callers that must not act on a suspicion the probe is about to withdraw. + */ +export function waitForInstallDirAclVerdict(now: number = Date.now()): Promise { + const remainingMs = + probePendingSince === null ? 0 : PROBE_VERDICT_GRACE_MS - (now - probePendingSince) + if (remainingMs <= 0) { + return Promise.resolve() + } + return new Promise((resolve) => { + const wake = (): void => { + clearTimeout(timer) + verdictWaiters.delete(wake) + resolve() + } + const timer = setTimeout(wake, remainingMs) + timer.unref?.() + verdictWaiters.add(wake) + }) +} + +/** + * True while a sandboxed-child death could be the install DACL rather than the + * graphics driver. Safe graphics does not rescue a poisoned tree — it still kills + * the renderer — and it removes the GPU child, erasing the sibling-death evidence + * that is the only way to recognise the shape in a crash report. + */ +export function isInstallDirAclSuspect(now: number = Date.now()): boolean { + if (installDirReadClean) { + return false + } + if (poison && poison.stage !== 'repaired') { + return true + } + // A 'repaired' stage is icacls's exit claim, not a reading of the tree — and the GPU + // children die 48-1373ms after window creation while the probe answers 0.9-3.0s in. So + // the claim stays provisional while this launch's probe is still out: the grace check + // below keeps the suspicion until the reading corroborates it or the window lapses. + return probePendingSince !== null && now - probePendingSince < PROBE_VERDICT_GRACE_MS +} + +/** + * True while a repair for this tree is dispatched and has not reported yet. + * + * Why it is not the same question as `isInstallDirAclSuspect`: a suspect tree we are + * actively repairing is one a *future* launch can still be rescued on, so the safe-graphics + * marker earns its keep there — a launch Chromium FATALs mid-repair comes back software + * rendered, stops spawning the GPU children that trigger the FATAL, and lets the next gate + * run to completion. A terminal verdict has no such next step. + */ +export function isInstallDirAclRepairPending(): boolean { + return poison?.stage === 'pending' +} + +/** + * True once the repair has nothing left to try for this install and version: `marker-hit` + * is reachable only through the spent attempt budget. The suspicion itself stands — the + * dialog still names the cause and the admin commands — but a caller that was *withholding* + * a recovery to give the repair first go has nothing left to wait for. + */ +export function isInstallDirAclRepairExhausted(): boolean { + return poison?.stage === 'marker-hit' +} + +/** True while the pre-window gate is rewriting the very files a new renderer would load. */ +export function isBlockingInstallDirAclRepairInFlight(): boolean { + return blockingRepairInFlight +} + +/** False when the once-per-process repair had already been dispatched, so no `onDone` is coming. */ +function startRepair( + installDir: string, + options: WindowsInstallDirAclRecoveryOptions, + onDone?: (result: WindowsInstallDirAclRepairResult) => void +): boolean { + writeInstallDirAclPoisonMarker(options.userDataPath, installDir, options.appVersion) + const started = repairWindowsInstallDirPackageAcl({ + ...options, + installDir, + // Every caller here holds outstanding poison evidence — this launch's probe reading, or + // the persisted marker that armed the gate — so a marker recording a completed repair + // describes a re-poisoned tree, or an icacls run that silently no-opped. It must not + // stand in for a repair. `marker-hit` therefore only ever means the budget is spent. + poisonEvidenceOutstanding: true, + onDone: (result) => { + // A tree read poisoned AFTER this repair was dispatched disproves its success claim, + // whatever icacls exited: the gate's budget can expire while the child runs on under + // its own, so the probe's reading is the later evidence. A clean reading since then + // retires it — there was nothing left to repair. + const claimDisproved = + result.mode === 'repaired' && installDirReadPoisonedMidRepair && !installDirReadClean + // A clean reading of the tree outranks this: there was nothing left to repair. + if (!installDirReadClean) { + poison = { installDir, stage: claimDisproved ? 'failed' : result.mode } + } + logStartupMilestone('install-dir-acl-repair-done', { mode: result.mode }) + if (result.mode === 'repaired' && !claimDisproved) { + clearInstallDirAclPoisonMarker(options.userDataPath) + // The GPU child deaths were never a driver fault, so safe graphics — and the + // --in-process-gpu launch that hides the next crash's evidence — must not outlive the repair. + // Never a user-confirmed marker: "keep safe graphics" is a choice, not Orca's latch. + const gpuMarker = readGpuFallbackMarker(options.userDataPath) + if (gpuMarker?.userConfirmed === false) { + // Kept: a probe reading that later disproves this claim restores the marker, + // or the next launch relaunches hardware accelerated into the re-armed gate. + gpuMarkerClearedByRepairClaim = gpuMarker + clearGpuFallbackMarker(options.userDataPath) + } + } + if (result.mode === 'failed') { + console.warn('[win32-acl] install dir package ACL repair failed:', result.reason) + } + onDone?.(result) + } + }) + if (started) { + poison = { installDir, stage: 'pending' } + } + return started } /** The probe's `onDone`: no-op unless the machine is in the reproduced state. */ @@ -36,22 +202,112 @@ export function startWindowsInstallDirAclRepairIfPoisoned( data: CrashReportBreadcrumbData, options: WindowsInstallDirAclRecoveryOptions ): void { - if (!isInstallDirAclPoisonVerdict(data)) { - return + // Cleared for every verdict, including an unreadable one that proves nothing: that + // releases a provisional 'repaired' claim early, but holding it would only move the + // same release to the grace-window expiry — an unreadable probe can never corroborate. + probePendingSince = null + try { + applyInstallDirAclProbeVerdict(data, options) + } finally { + // Only after the verdict is applied: a waiter wakes to re-read `isInstallDirAclSuspect()`. + settleVerdictWaiters() } - const installDir = options.installDir ?? dirname(process.execPath) - poison = { installDir, stage: 'pending' } - repairWindowsInstallDirPackageAcl({ - ...options, - installDir, - onDone: (result) => { - poison = { installDir, stage: result.mode } - logStartupMilestone('install-dir-acl-repair-done', { mode: result.mode }) - if (result.mode === 'failed') { - console.warn('[win32-acl] install dir package ACL repair failed:', result.reason) +} + +function applyInstallDirAclProbeVerdict( + data: CrashReportBreadcrumbData, + options: WindowsInstallDirAclRecoveryOptions +): void { + if (!isInstallDirAclPoisonVerdict(data)) { + // Only a positive clean reading retires the verdict; an unreadable DACL proves nothing. + if (data.matchesPoisonSignature === false) { + clearInstallDirAclPoisonMarker(options.userDataPath) + installDirReadClean = true + // The reading corroborates any repair claim, so its marker clear stands. + gpuMarkerClearedByRepairClaim = null + // Keeping 'repaired' costs nothing and is what tells the user to reload; anything + // else would go on suppressing the driver fallback and accusing a healthy folder. + if (poison?.stage !== 'repaired') { + poison = null } } - }) + return + } + // The blocking pre-window gate still owns this launch's repair; restarting it would + // reset the verdict to 'pending' against a repair that can no longer report. The reading + // is kept, not dropped: it is later evidence than the repair's own exit code. + if (poison?.stage === 'pending') { + installDirReadPoisonedMidRepair = true + return + } + // This reading was taken after the gate finished, so it outranks the gate's own verdict: + // a tree that still matches the signature was never repaired, whatever icacls exited. + if (poison?.stage === 'repaired') { + poison = { installDir: poison.installDir, stage: 'failed' } + // The claim also cleared the safe-graphics marker; disproved, it owes that back, or + // the next launch relaunches hardware accelerated and FATALs before its gate can win. + const cleared = gpuMarkerClearedByRepairClaim + gpuMarkerClearedByRepairClaim = null + if (cleared) { + try { + writeGpuFallbackMarker(options.userDataPath, cleared, cleared) + } catch { + // Best effort: the re-armed poison marker below still gates the next launch. + } + } + } + // Re-writes the poison marker — re-arming the next launch's gate — even when the + // once-per-process latch means no icacls can run again this launch. + startRepair(options.installDir ?? dirname(process.execPath), options) +} + +/** + * Pre-window gate for a machine a previous launch already found poisoned. + * + * Why blocking, and why only here: the probe is `setImmediate`-deferred and takes + * 0.9-3.0s on the affected hosts, while the renderer it has to save is spawned + * synchronously by `createMainWindow` and dies at init 48-1373ms in. The + * persisted verdict is what buys that knowledge for free — a healthy machine + * reads one absent file and pays nothing. + */ +export async function repairKnownPoisonedInstallDirBeforeWindow( + options: WindowsInstallDirAclRecoveryOptions & { timeoutMs?: number } +): Promise<'not-marked' | 'skipped' | WindowsInstallDirAclRepairResult['mode'] | 'timeout'> { + if ((options.platform ?? process.platform) !== 'win32' || options.isServeMode === true) { + return 'skipped' + } + const installDir = options.installDir ?? dirname(process.execPath) + if (!hasInstallDirAclPoisonMarker(options.userDataPath, installDir, options.appVersion)) { + return 'not-marked' + } + logStartupMilestone('install-dir-acl-repair-blocking-start') + blockingRepairInFlight = true + try { + return await new Promise((resolve) => { + const timer = setTimeout( + () => resolve('timeout'), + options.timeoutMs ?? BLOCKING_REPAIR_BUDGET_MS + ) + timer.unref?.() + // The marker is an earlier launch's DACL reading that nothing has retired, so a + // repair marker claiming success cannot stand in for the repair this launch owes. + const started = startRepair(installDir, options, (result) => { + clearTimeout(timer) + resolve(result.mode) + }) + // No dispatch means no `onDone`, so waiting out the whole budget would buy nothing. + if (!started) { + clearTimeout(timer) + resolve('skipped') + } + }) + } catch (error) { + // This sits in the critical path ahead of window creation; it must never throw into it. + console.warn('[win32-acl] blocking install dir ACL repair faulted:', error) + return 'skipped' + } finally { + blockingRepairInFlight = false + } } const CAUSE = diff --git a/src/main/startup/windows-install-dir-acl-repair.win32.test.ts b/src/main/startup/windows-install-dir-acl-repair.win32.test.ts new file mode 100644 index 00000000000..9a04aa1edef --- /dev/null +++ b/src/main/startup/windows-install-dir-acl-repair.win32.test.ts @@ -0,0 +1,124 @@ +import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { runProcess } from '../../shared/child-process/run-process' +import { getIcaclsExePath } from '../win32-utils' +import { removeTreeSync } from '../../shared/windows-transient-lock-removal' +import { + probeWindowsInstallDirAcl, + resetWindowsInstallDirAclProbeForTest +} from './windows-install-dir-acl-probe' +import { writeInstallDirAclPoisonMarker } from './windows-install-dir-acl-poison-marker' +import { + repairKnownPoisonedInstallDirBeforeWindow, + resetWindowsInstallDirAclRecoveryForTest +} from './windows-install-dir-acl-recovery' +import { resetWindowsInstallDirAclRepairForTest } from './windows-install-dir-package-acl-repair' + +/** + * The other half of the ACL proof: the unit tests fake icacls, and this one runs + * the real binary against a real poisoned tree on a real Windows box. + * + * Both are needed. `icacls /grant "*S-1-15-2-2:(OI)(CI)(RX)"` exits 0 and + * prints "Failed processing 0 files" while writing no ACE at all — a model of + * icacls cannot catch that, and it is the exact mistake that leaves the app dead. + * + * Runs only on win32; skipped elsewhere. + */ +const describeOnWindows = process.platform === 'win32' ? describe : describe.skip + +/** An unresolvable AppContainer SID, the shape the field hosts carry. */ +const ORPHAN_SID = + '*S-1-15-2-1111111111-2222222222-3333333333-4444444444-5555555555-6666666666-7777777777' +const RESTRICTED_PACKAGES_NAME = /ALL RESTRICTED APPLICATION PACKAGES/i + +async function icacls(...args: string[]): Promise<{ code: number | null; out: string }> { + const result = await runProcess({ program: getIcaclsExePath(), args, timeoutMs: 30_000 }) + return { code: result.code, out: `${result.stdout}\n${result.stderr}` } +} + +/** Explicit DACL, inheritance off: what a shipped module carries, and why a root grant alone is not enough. */ +async function createProtectedFile(path: string): Promise { + writeFileSync(path, 'binary') + await icacls(path, '/inheritance:d') +} + +describeOnWindows('install-dir package ACL repair against the real icacls', () => { + let installDir: string + let userDataPath: string + let moduleFile: string + let trapFile: string + + beforeAll(async () => { + installDir = mkdtempSync(join(tmpdir(), 'orca-acl-live-')) + userDataPath = mkdtempSync(join(tmpdir(), 'orca-acl-live-ud-')) + mkdirSync(join(installDir, 'resources'), { recursive: true }) + moduleFile = join(installDir, 'ffmpeg.dll') + trapFile = join(installDir, 'resources', 'trap.dll') + await createProtectedFile(moduleFile) + await createProtectedFile(trapFile) + // Poison: an orphan package ACE on the tree and on the module, no well-known grant. + await icacls(installDir, '/grant', `${ORPHAN_SID}:(OI)(CI)(RX)`) + await icacls(moduleFile, '/grant', `${ORPHAN_SID}:(RX)`) + await icacls(trapFile, '/grant', `${ORPHAN_SID}:(RX)`) + }) + + afterAll(() => { + // Why removeTreeSync: two icacls.exe children just rewrote DACLs on this tree, so a + // raw rmSync races handles Windows has not released and throws EPERM after the + // assertions already passed. + removeTreeSync(installDir) + removeTreeSync(userDataPath) + }) + + function probeVerdict(): Promise> { + resetWindowsInstallDirAclProbeForTest() + return new Promise((resolve) => { + probeWindowsInstallDirAcl({ + installDir, + recordBreadcrumb: () => undefined, + onDone: (data) => resolve(data as Record) + }) + }) + } + + // The trap, pinned against the real binary: this is the form that looks like it worked. + it('confirms an inheritance-flagged grant silently writes nothing to a file', async () => { + const flagged = await icacls(trapFile, '/grant', '*S-1-15-2-2:(OI)(CI)(RX)') + expect(flagged.code).toBe(0) + expect(flagged.out).toMatch(/Failed processing 0 files?/i) + const after = await icacls(trapFile) + expect(after.out).not.toMatch(RESTRICTED_PACKAGES_NAME) + }) + + it('repairs the tree before the window, and the grant lands on the module file', async () => { + expect((await probeVerdict()).matchesPoisonSignature).toBe(true) + + resetWindowsInstallDirAclRecoveryForTest() + resetWindowsInstallDirAclRepairForTest() + // The state a launch that died mid-repair leaves behind. + writeInstallDirAclPoisonMarker(userDataPath, installDir, '1.4.196') + + const startedAt = Date.now() + const mode = await repairKnownPoisonedInstallDirBeforeWindow({ + installDir, + userDataPath, + appVersion: '1.4.196', + recordBreadcrumb: () => undefined + }) + console.log(`[live-acl] blocking repair ${mode} in ${Date.now() - startedAt}ms`) + expect(mode).toBe('repaired') + + // A directory grant is not enough: the file carries its own DACL. + expect((await icacls(moduleFile)).out).toMatch(RESTRICTED_PACKAGES_NAME) + // The /T pass must also reach a NESTED protected file — the shape app.asar.unpacked + // and node_modules actually have. + expect((await icacls(trapFile)).out).toMatch(RESTRICTED_PACKAGES_NAME) + // And the (OI)(CI) root grant exists so files a later update writes inherit it. + const updateFile = join(installDir, 'resources', 'added-by-update.dll') + writeFileSync(updateFile, 'binary') + expect((await icacls(updateFile)).out).toMatch(RESTRICTED_PACKAGES_NAME) + expect((await probeVerdict()).matchesPoisonSignature).toBe(false) + }) +}) diff --git a/src/main/startup/windows-install-dir-acl-startup-wiring.test.ts b/src/main/startup/windows-install-dir-acl-startup-wiring.test.ts new file mode 100644 index 00000000000..c4175e87a06 --- /dev/null +++ b/src/main/startup/windows-install-dir-acl-startup-wiring.test.ts @@ -0,0 +1,56 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * The three call sites that make the repair real. Each is one line of wiring in a + * module whose import graph makes it untestable in-process; the behaviour each + * line depends on is driven for real in `windows-install-dir-acl-recovery.test.ts`, + * `gpu-lifecycle-install-dir-acl-guard.test.ts` and `focus-existing-window.test.ts`. + */ + +function readSource(relativePath: string): string { + return readFileSync(join(process.cwd(), relativePath), 'utf8') +} + +describe('install-dir ACL repair startup wiring', () => { + // The entire premise: a renderer must never be spawned onto a tree a previous + // launch recorded as poisoned before icacls has had its bounded chance at it. + it('awaits the pre-window gate before any window creation', () => { + const source = readSource('src/main/startup/main-process-runtime-launch.ts') + const launchStart = source.indexOf('export async function initializeMainProcessRuntimeLaunch(') + expect(launchStart).toBeGreaterThanOrEqual(0) + const launch = source.slice(launchStart) + + const gateIndex = launch.indexOf('await repairKnownPoisonedInstallDirBeforeWindow(') + const winEarlyWindowIndex = launch.indexOf('startWindowsDesktopBeforeShellPathReady(') + const desktopLaunchIndex = launch.indexOf('await launchDesktopMode(') + expect(gateIndex).toBeGreaterThanOrEqual(0) + expect(winEarlyWindowIndex).toBeGreaterThan(gateIndex) + expect(desktopLaunchIndex).toBeGreaterThan(gateIndex) + }) + + // A 20s blank launch invites a second double-click, and `focusExistingMainWindow` + // opens a window whenever there is none and the app is ready. + it('holds the second-instance reopen while the gate owns the launch', () => { + const source = readSource('src/main/startup/main-window-actions.ts') + const start = source.indexOf('export function focusExistingWindow(') + const end = source.indexOf('\nexport function showMainWindowFromTray(', start) + expect(start).toBeGreaterThanOrEqual(0) + expect(end).toBeGreaterThan(start) + expect(source.slice(start, end)).toContain( + 'canOpenWindow: () => !isBlockingInstallDirAclRepairInFlight()' + ) + }) + + // openMainWindow re-runs on every reopen while the probe is once-per-process, so + // arming the grace window unconditionally would drop GPU crashes on a healthy machine. + it('arms the probe grace window only for a dispatched probe', () => { + const source = readSource('src/main/startup/main-window-controller.ts') + const dispatchIndex = source.indexOf('const probeDispatched = probeWindowsInstallDirAcl(') + const armIndex = source.indexOf('noteWindowsInstallDirAclProbePending()') + expect(dispatchIndex).toBeGreaterThanOrEqual(0) + expect(armIndex).toBeGreaterThan(dispatchIndex) + expect(source.slice(dispatchIndex, armIndex)).toContain('if (probeDispatched) {') + }) +}) diff --git a/src/main/startup/windows-install-dir-package-acl-repair.test.ts b/src/main/startup/windows-install-dir-package-acl-repair.test.ts index 65d11e9fcde..cc41388b811 100644 --- a/src/main/startup/windows-install-dir-package-acl-repair.test.ts +++ b/src/main/startup/windows-install-dir-package-acl-repair.test.ts @@ -135,7 +135,7 @@ describe('repairWindowsInstallDirPackageAcl', () => { const second = fakeRunner() const { result, data } = await repair({ userDataPath, run: second.run }) expect(second.specs).toHaveLength(0) - expect(result).toEqual({ mode: 'marker-hit' }) + expect(result).toEqual({ mode: 'marker-hit', alreadyRepaired: true }) expect(data.reason).toBe('marker-hit') }) @@ -233,6 +233,42 @@ describe('repairWindowsInstallDirPackageAcl', () => { expect(marker.outcome).toBe('failed') }) + // The bricking mechanism: a marker was written on failure and matched regardless of + // outcome, so one Defender-locked file or one timeout pinned the machine to + // 'marker-hit' — repair permanently skipped — for the life of that version. + it('retries a failed repair on later launches, then stops once the budget is spent', async () => { + const userDataPath = userDataDir() + const failing = fakeRunner(() => ({ code: 5, stderr: 'Access is denied.' })) + for (let attempt = 0; attempt < 3; attempt++) { + resetWindowsInstallDirAclRepairForTest() + expect((await repair({ userDataPath, run: failing.run })).result.mode).toBe('failed') + } + expect(failing.specs).toHaveLength(6) + + resetWindowsInstallDirAclRepairForTest() + const spent = fakeRunner() + const { result } = await repair({ userDataPath, run: spent.run }) + // Not alreadyRepaired: the budget ran out, so the tree is still poisoned. + expect(result).toEqual({ mode: 'marker-hit', alreadyRepaired: false }) + expect(spent.specs).toHaveLength(0) + }) + + it('stops retrying immediately once a repair has succeeded', async () => { + const userDataPath = userDataDir() + resetWindowsInstallDirAclRepairForTest() + await repair({ userDataPath, run: fakeRunner(() => ({ code: 5 })).run }) + resetWindowsInstallDirAclRepairForTest() + expect((await repair({ userDataPath })).result).toEqual({ mode: 'repaired' }) + + resetWindowsInstallDirAclRepairForTest() + const after = fakeRunner() + expect((await repair({ userDataPath, run: after.run })).result).toEqual({ + mode: 'marker-hit', + alreadyRepaired: true + }) + expect(after.specs).toHaveLength(0) + }) + it('is a no-op off win32 and in serve mode', async () => { const off = fakeRunner() repairWindowsInstallDirPackageAcl({ diff --git a/src/main/startup/windows-install-dir-package-acl-repair.ts b/src/main/startup/windows-install-dir-package-acl-repair.ts index 606edae417c..6bd6505a2cb 100644 --- a/src/main/startup/windows-install-dir-package-acl-repair.ts +++ b/src/main/startup/windows-install-dir-package-acl-repair.ts @@ -54,7 +54,8 @@ const TREE_GRANT_TIMEOUT_MS = 120_000 const FAILED_PROCESSING = /Failed processing (\d+) files?/i export type WindowsInstallDirAclRepairResult = - | { mode: 'marker-hit' } + /** `alreadyRepaired`: the marker records a completed repair, not an exhausted retry budget. */ + | { mode: 'marker-hit'; alreadyRepaired: boolean } | { mode: 'repaired' } | { mode: 'failed'; reason: string; failedFileCount: number | null } @@ -62,6 +63,14 @@ export type WindowsInstallDirAclRepairOptions = { installDir?: string platform?: NodeJS.Platform isServeMode?: boolean + /** + * A DACL reading found this tree poisoned and nothing has read it clean since — this + * launch's probe, or a persisted poison marker from an earlier one. A marker claiming a + * completed repair therefore describes a tree that has since been re-poisoned, or an + * icacls run that silently no-opped: it stops outranking the reading. The attempt + * budget still bounds retries. + */ + poisonEvidenceOutstanding?: boolean /** Test seams. */ runProcessFn?: typeof runProcess recordBreadcrumb?: typeof recordDurableCrashBreadcrumb @@ -81,8 +90,17 @@ type RepairMarker = { appVersion: string attemptedAt: number outcome: string + /** Absent on schemeVersion-1 markers written before the retry budget existed. */ + attempts?: number } +// Why bounded rather than one-and-done: the failure modes are not all permanent. +// A Defender-locked file, a timeout or a contended volume fails one launch and +// succeeds the next, and pinning on the first failure leaves the machine blank +// forever for that version. Three is enough to stop a standard-user Program Files +// install — which can never win — from re-spawning icacls on every launch. +const MAX_REPAIR_ATTEMPTS = 3 + /** * The probe's verdict is the only trigger: an orphan package ACE with no * well-known package grant to satisfy it. A localized icacls prints those grants @@ -106,31 +124,46 @@ function markerPath(userDataPath: string): string { return join(userDataPath, WINDOWS_INSTALL_DIR_ACL_REPAIR_MARKER_FILE) } -function hasMarkerFor(args: WindowsInstallDirAclRepairArgs): boolean { +/** The marker for this exact install and version, or null. */ +function readMarkerFor(args: WindowsInstallDirAclRepairArgs): Partial | null { try { const parsed = JSON.parse(readFileSync(markerPath(args.userDataPath), 'utf-8')) as | Partial | undefined - return ( - parsed?.schemeVersion === WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION && - parsed.installDir === args.installDir && - parsed.appVersion === args.appVersion - ) + if ( + parsed?.schemeVersion !== WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION || + parsed.installDir !== args.installDir || + parsed.appVersion !== args.appVersion + ) { + return null + } + return parsed } catch { - return false // missing or corrupt -> attempt again + return null // missing or corrupt -> attempt again } } -// Why write it on failure too: a standard-user Program Files install can never -// win, and re-spawning icacls on every launch forever buys nothing. Reinstall or -// update changes the key and retries. +function markerHitFor(args: WindowsInstallDirAclRepairArgs): { alreadyRepaired: boolean } | null { + const marker = readMarkerFor(args) + if (!marker) { + return null + } + if (marker.outcome === 'repaired' && args.poisonEvidenceOutstanding !== true) { + return { alreadyRepaired: true } + } + return (marker.attempts ?? 0) >= MAX_REPAIR_ATTEMPTS ? { alreadyRepaired: false } : null +} + +// Why write it on failure too: re-spawning icacls on every launch forever buys +// nothing, so failures spend the retry budget. Reinstall or update changes the key. function writeMarker(args: WindowsInstallDirAclRepairArgs, outcome: string): void { const marker: RepairMarker = { schemeVersion: WINDOWS_INSTALL_DIR_ACL_REPAIR_SCHEME_VERSION, installDir: args.installDir ?? '', appVersion: args.appVersion, attemptedAt: Date.now(), - outcome + outcome, + attempts: (readMarkerFor(args)?.attempts ?? 0) + 1 } if (!existsSync(args.userDataPath)) { mkdirSync(args.userDataPath, { recursive: true }) @@ -185,9 +218,10 @@ async function runRepair(args: WindowsInstallDirAclRepairArgs): Promise { let result: WindowsInstallDirAclRepairResult let data: CrashReportBreadcrumbData try { - if (hasMarkerFor(resolved)) { - result = { mode: 'marker-hit' } - data = { status: 'skipped', reason: 'marker-hit' } + const markerHit = markerHitFor(resolved) + if (markerHit) { + result = { mode: 'marker-hit', alreadyRepaired: markerHit.alreadyRepaired } + data = { status: 'skipped', reason: 'marker-hit', alreadyRepaired: markerHit.alreadyRepaired } } else { const runner = args.runProcessFn ?? runProcess const root = await runGrant( @@ -257,13 +291,16 @@ export function resetWindowsInstallDirAclRepairForTest(): void { * Fire-and-forget; returns before any spawn. Call only when the probe reported * `matchesPoisonSignature`. win32 only, exempt in serve mode, and it must never * throw into window creation. + * + * Returns whether THIS call dispatched the repair. A caller that waits on `onDone` + * would otherwise wait forever on the once-per-process latch. */ -export function repairWindowsInstallDirPackageAcl(args: WindowsInstallDirAclRepairArgs): void { +export function repairWindowsInstallDirPackageAcl(args: WindowsInstallDirAclRepairArgs): boolean { if ((args.platform ?? process.platform) !== 'win32' || args.isServeMode === true) { - return + return false } if (repairStarted) { - return + return false } repairStarted = true try { @@ -273,4 +310,5 @@ export function repairWindowsInstallDirPackageAcl(args: WindowsInstallDirAclRepa } catch { // Nothing left to report to that would not throw again. } + return true } diff --git a/src/main/window/focus-existing-window.test.ts b/src/main/window/focus-existing-window.test.ts index 85b422e02a4..babb9a15490 100644 --- a/src/main/window/focus-existing-window.test.ts +++ b/src/main/window/focus-existing-window.test.ts @@ -127,6 +127,41 @@ describe('focusExistingMainWindow', () => { expect(timer.scheduledMs()).toEqual([]) }) + // The blocking install-DACL repair holds the first window for up to 20s of blank + // screen, which is exactly when a user double-clicks the shortcut again. That + // second instance must not spawn a renderer onto a tree icacls is rewriting. + it('drops a reopen while another path must own the first window', () => { + const openWindow = vi.fn() + + const result = focusExistingMainWindow({ + app: makeFakeApp(), + getWindow: () => null, + openWindow, + canOpenWindow: () => false + }) + + expect(result).toBe('pending') + expect(openWindow).not.toHaveBeenCalled() + }) + + it('still focuses a window that already exists while reopening is held', () => { + const window = makeFakeWindow() + const openWindow = vi.fn() + + const result = focusExistingMainWindow({ + app: makeFakeApp(), + getWindow: () => window, + openWindow, + canOpenWindow: () => false, + platform: 'darwin', + setTimeout: makeTimer().setTimeout + }) + + expect(result).toBe('focused') + expect(openWindow).not.toHaveBeenCalled() + expect(window.calls.focus).toHaveBeenCalledTimes(1) + }) + it('waits for normal startup when no window exists before app readiness', () => { const openWindow = vi.fn() diff --git a/src/main/window/focus-existing-window.ts b/src/main/window/focus-existing-window.ts index 8c8ac85ca2e..4cadb49a743 100644 --- a/src/main/window/focus-existing-window.ts +++ b/src/main/window/focus-existing-window.ts @@ -9,6 +9,8 @@ export type FocusExistingMainWindowOptions = { app: Pick getWindow: () => BrowserWindow | null openWindow: () => BrowserWindow + /** False while some other path must own the first window; the reopen is dropped, not queued. */ + canOpenWindow?: () => boolean platform?: NodeJS.Platform setTimeout?: FocusTimer warn?: (message: string, error?: unknown) => void @@ -143,7 +145,7 @@ export function focusExistingMainWindow( let openedWindow = false if (!window || window.isDestroyed()) { - if (!opts.app.isReady()) { + if (!opts.app.isReady() || opts.canOpenWindow?.() === false) { return 'pending' } window = openWindowWithRetry(opts, platform, setTimer, 1) diff --git a/src/main/windows/windows-pty-job.ts b/src/main/windows/windows-pty-job.ts index 193b1d8825a..169db375f3b 100644 --- a/src/main/windows/windows-pty-job.ts +++ b/src/main/windows/windows-pty-job.ts @@ -118,8 +118,10 @@ export function terminatePtyJob(proc: IPty): JobTerminationOutcome { /** * Pids still alive in a PTY's tree, or null when there is no answer. * - * Measured on Windows 11: once the shell exits, node-pty drops its handle - * record and closes the job, so a terminated tree reports **null**, not `[]`. + * Measured on Windows 11: once the shell exits, node-pty closes the job, so a + * terminated tree reports **null**, not `[]`. (Its handle record now outlives + * the shell until `kill()` runs — see config/patches/node-pty@1.1.0.patch — but + * the nulled job handle is what makes the answer null either way.) * Null therefore means "unverifiable" in the sense of * docs/reference/ssh-execution-boundary.md — this build has no job support, * the terminal is not a ConPTY, or it is no longer tracked. It is never diff --git a/src/preload/api/orca-profile-api.ts b/src/preload/api/orca-profile-api.ts index 16c2a575078..80e9f08fc8d 100644 --- a/src/preload/api/orca-profile-api.ts +++ b/src/preload/api/orca-profile-api.ts @@ -28,6 +28,8 @@ import type { export type OrcaProfileApi = { list: () => Promise authStatus: () => Promise + /** Fires when main changed the stored auth status on its own (e.g. a revoked session). */ + onAuthStatusChanged: (callback: () => void) => () => void createLocal: (args?: CreateLocalOrcaProfileArgs) => Promise createCloudLinked: ( args?: CreateCloudLinkedOrcaProfileArgs diff --git a/src/preload/api/orca-profiles-bridge.ts b/src/preload/api/orca-profiles-bridge.ts index 0b2897f8ab1..da58b2d9def 100644 --- a/src/preload/api/orca-profiles-bridge.ts +++ b/src/preload/api/orca-profiles-bridge.ts @@ -1,9 +1,15 @@ import { ipcRenderer } from 'electron' import type { PreloadApi } from '../api-types' +import { ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL } from '../../shared/orca-profiles' export const orcaProfilesApi = { list: () => ipcRenderer.invoke('orcaProfiles:list'), authStatus: () => ipcRenderer.invoke('orcaProfiles:authStatus'), + onAuthStatusChanged: (callback: () => void): (() => void) => { + const listener = (): void => callback() + ipcRenderer.on(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL, listener) + return () => ipcRenderer.removeListener(ORCA_PROFILE_AUTH_STATUS_CHANGED_CHANNEL, listener) + }, createLocal: (args) => ipcRenderer.invoke('orcaProfiles:createLocal', args), createCloudLinked: (args) => ipcRenderer.invoke('orcaProfiles:createCloudLinked', args), switchProfile: (args) => ipcRenderer.invoke('orcaProfiles:switch', args), diff --git a/src/preload/api/pty-api.ts b/src/preload/api/pty-api.ts index bf012306675..a1850398357 100644 --- a/src/preload/api/pty-api.ts +++ b/src/preload/api/pty-api.ts @@ -112,7 +112,7 @@ export type PtyApi = { getForegroundProcess: (id: string) => Promise inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ) => Promise confirmForegroundProcess: (id: string) => Promise getCwd: (id: string) => Promise diff --git a/src/preload/api/pty-bridge-stream-and-serialization.ts b/src/preload/api/pty-bridge-stream-and-serialization.ts index 7e2d7bbe4ff..414a5514bfa 100644 --- a/src/preload/api/pty-bridge-stream-and-serialization.ts +++ b/src/preload/api/pty-bridge-stream-and-serialization.ts @@ -7,7 +7,7 @@ import type { TerminalProcessInspection } from '../../shared/terminal-process-in export const ptyStreamAndSerializationApi = { inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } ): Promise => ipcRenderer.invoke('pty:inspectProcess', { id, ...options }), confirmForegroundProcess: (id: string): Promise => diff --git a/src/relay/git-porcelain-local-parity.test.ts b/src/relay/git-porcelain-local-parity.test.ts index d4969f6fd0e..9fccc85e7b3 100644 --- a/src/relay/git-porcelain-local-parity.test.ts +++ b/src/relay/git-porcelain-local-parity.test.ts @@ -160,7 +160,8 @@ describe('relay/desktop unmerged-entry porcelain parity', () => { const unmergedLines = [ 'u UU N... 100644 100644 100644 100644 aa bb cc plain.ts', 'u UD N... 100644 100644 000000 100644 aa bb cc "present \\303\\251.ts"', - 'u UD N... 100644 100644 000000 100644 aa bb cc "missing \\303\\251.ts"', + // mW=000000: real Git reports an absent working-tree path this way, and the file is not created below. + 'u UD N... 100644 100644 000000 000000 aa bb cc "missing \\303\\251.ts"', 'u DD N... 100644 100644 000000 000000 aa bb cc both-gone.ts' ] const git = vi.fn(async (args) => { diff --git a/src/relay/pty-child-process-inspection.ts b/src/relay/pty-child-process-inspection.ts new file mode 100644 index 00000000000..6d246817053 --- /dev/null +++ b/src/relay/pty-child-process-inspection.ts @@ -0,0 +1,81 @@ +/** + * Whether anything is running under a pane's shell. + * + * Split out of `pty-shell-utils` because it is a distinct question from "what is in front" and + * carries its own platform reasoning, its own cost budget, and the verdict vocabulary from + * docs/reference/ssh-execution-boundary.md. + */ +import { queryWindowsPaneProcessInventory } from '../main/providers/windows-foreground-process-rows' +import { getProcessTableIndex } from '../shared/process-table-index' +import { + getFreshProcessTableSnapshot, + getProcessTableSnapshot +} from '../shared/process-table-snapshot-reader' +import type { PtyChildProcessVerdict } from '../shared/terminal-process-inspection' +import { isProcessAlive } from './pty-shell-utils' + +/** + * Check whether a process has child processes. + * + * Why the shared snapshot and not `pgrep -P`: this answers one field of + * `pty.inspectProcess`, which every tracked pane polls on a 750ms/2000ms + * cadence, and the fork was neither cached nor coalesced. procps-ng opens six + * procfs files per process to resolve a ppid — including a `/proc//ctty` + * that never exists on Linux — so one call cost O(host process count) syscalls, + * ~4k opens per pgrep on a 690-process host, at up to 8 forks/sec (#13537). + * `getForegroundProcessName` in the same RPC already captured the TTL-cached + * `ps` table, whose index carries the parent/child map, so the answer is free. + * + * `fresh` opts out of that TTL. A poll can read a 500ms-old table because its + * next tick corrects it, but a close or cleanup decision acts on the answer + * once and destructively — a child that started inside the TTL would be killed + * with no confirmation. `pgrep` scanned per call, so anything that decides + * has to keep scanning per call. + */ +export async function inspectPtyChildProcesses( + pid: number, + options?: { fresh?: boolean } +): Promise { + if (process.platform === 'win32') { + // Windows has no `ps`, but it does have a process table, and the pane walk over it already + // exists for the foreground reader. Answering `false` from nothing was the older shape: a + // hardcoded negative is indistinguishable from a measurement, and every close guard reads it + // as "nothing is running here". + // + // Deliberately the TTL-cached table even when `fresh` is asked for: on a relay without the + // native binding this falls back to the CIM scan, whose own 1.36s runtime is longer than the + // 500ms TTL a fresh read would be refreshing, so a "fresh" answer is not meaningfully fresher + // while N sequential ones are an N x 1.36s stall. + const inventory = await queryWindowsPaneProcessInventory(pid) + if (inventory) { + return inventory.candidates.length > 0 ? 'children' : 'no-children' + } + // A null inventory is an unreadable table OR a snapshot that never showed the root, and + // neither of those looked at the pane. The one answer available without the table is a root + // the kernel says is gone: nothing runs under a shell that does not exist. + return isProcessAlive(pid) ? 'unverifiable' : 'no-children' + } + try { + const rows = options?.fresh + ? await getFreshProcessTableSnapshot() + : await getProcessTableSnapshot() + return (getProcessTableIndex(rows).childrenByPpid.get(pid)?.length ?? 0) > 0 + ? 'children' + : 'no-children' + } catch { + return 'unverifiable' + } +} + +/** + * The boolean the wire has always carried. `unverifiable` keeps spelling itself `false` here on + * purpose: this value reaches clients too old to know the third answer, and it is read both as + * "busy, do not close" and as "the agent has taken over, safe to type into", so no single mapping + * of `unverifiable` is safe for both. Callers that can act on the distinction read the verdict. + */ +export async function processHasChildren( + pid: number, + options?: { fresh?: boolean } +): Promise { + return (await inspectPtyChildProcesses(pid, options)) === 'children' +} diff --git a/src/relay/pty-handler-spawn-admission.test.ts b/src/relay/pty-handler-spawn-admission.test.ts index 7ba02f78587..07fe8e9d88f 100644 --- a/src/relay/pty-handler-spawn-admission.test.ts +++ b/src/relay/pty-handler-spawn-admission.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' +import * as ptyChildProcessInspection from './pty-child-process-inspection' import * as ptyShellUtils from './pty-shell-utils' import * as processTableSnapshotReader from '../shared/process-table-snapshot-reader' @@ -92,7 +93,7 @@ describe('PtyHandler', () => { }) it('rescans the process table for a close decision but not for a poll', async () => { - const hasChildren = vi.mocked(ptyShellUtils.processHasChildren) + const hasChildren = vi.mocked(ptyChildProcessInspection.processHasChildren) const snapshot = vi .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') .mockResolvedValue({ @@ -127,13 +128,13 @@ describe('PtyHandler', () => { it('does not re-enter the shared capture after the evidence read gave up on it', async () => { // The budget is worthless if the compatibility fields answer by joining the very capture the - // evidence read just abandoned: `processHasChildren` and `getForegroundProcessName` read the - // same TTL-shared table with no budget of their own, so on a slow host this call would still - // block for the whole capture -- once, and then once per managed PTY in the listing. + // evidence read just abandoned: `inspectPtyChildProcesses` and `getForegroundProcessName` + // read the same TTL-shared table with no budget of their own, so on a slow host this call + // would still block for the whole capture -- once, then once per managed PTY in the listing. const snapshot = vi .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') .mockRejectedValue(new Error('process table unreadable: capture_over_budget')) - const hasChildren = vi.spyOn(ptyShellUtils, 'processHasChildren') + const hasChildren = vi.spyOn(ptyChildProcessInspection, 'inspectPtyChildProcesses') const foregroundName = vi.spyOn(ptyShellUtils, 'getForegroundProcessName') const { id } = (await spawnPty({ cols: 80, rows: 24 })) as { id: string } @@ -142,6 +143,7 @@ describe('PtyHandler', () => { const inspection = (await dispatcher.callRequest('pty.inspectProcess', { id })) as { hasChildProcesses: boolean + childProcessEvidence?: string foregroundProcessEvidence?: { verdict: string; reason?: string } } @@ -150,8 +152,9 @@ describe('PtyHandler', () => { // The verdict the gates already handle, reached promptly instead of late. expect(inspection.foregroundProcessEvidence?.verdict).toBe('unverifiable') expect(inspection.foregroundProcessEvidence?.reason).toBe('process_table_unreadable') - // `false` is what the child probe itself answers for an unreadable table; this is that same - // degraded answer without the wait, not a new claim about the pane. + // The honest verdict rather than a fabricated negative, reached without the wait. The + // compatibility boolean still spells `unverifiable` as `false` for older clients. + expect(inspection.childProcessEvidence).toBe('unverifiable') expect(inspection.hasChildProcesses).toBe(false) const listing = (await dispatcher.callRequest('pty.listProcesses', {})) as { diff --git a/src/relay/pty-handler-test-harness.ts b/src/relay/pty-handler-test-harness.ts index e3d99213ea5..fe6e17c5d9c 100644 --- a/src/relay/pty-handler-test-harness.ts +++ b/src/relay/pty-handler-test-harness.ts @@ -1,6 +1,6 @@ import { vi } from 'vitest' import type { Mock } from 'vitest' -import * as ptyShellUtils from './pty-shell-utils' +import * as ptyChildProcessInspection from './pty-child-process-inspection' import { PtyHandler } from './pty-handler' import type { RelayDispatcher } from './dispatcher' @@ -115,7 +115,7 @@ export function beginPtyHandlerTest(mocks: PtyHandlerTestMocks): { notifyOutput: vi.fn(), dispose: vi.fn() }) - vi.spyOn(ptyShellUtils, 'processHasChildren').mockResolvedValue(false) + vi.spyOn(ptyChildProcessInspection, 'processHasChildren').mockResolvedValue(false) mockPtySpawn.mockReturnValue({ ...mockPtyInstance }) diff --git a/src/relay/pty-handler-windows-child-process-evidence.test.ts b/src/relay/pty-handler-windows-child-process-evidence.test.ts new file mode 100644 index 00000000000..6d723824a04 --- /dev/null +++ b/src/relay/pty-handler-windows-child-process-evidence.test.ts @@ -0,0 +1,132 @@ +// Regression guard for the Windows SSH child-process answer. The relay used to return a hardcoded +// `false` here, which every close guard reads as "nothing is running in this pane" -- so a Windows +// SSH pane running a build closed with no prompt. The answer now comes from the process table, and +// the one thing it may never do again is fabricate a negative. +// +// The second contract is cost. `pty.inspectProcess` is the polled path (750ms/2000ms per tracked +// pane) and a relay host has no `@vscode/windows-process-tree`, so its table read falls back to a +// 1.36s CIM scan. Polling that would reinstate the fork storm the shared table exists to prevent, +// so only a caller whose answer decides something asks for the scan. +import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' + +const { mockPtySpawn, mockPtyInstance, mockCreateShellPromptReadinessProbe } = vi.hoisted(() => ({ + mockPtySpawn: vi.fn(), + mockCreateShellPromptReadinessProbe: vi.fn(), + mockPtyInstance: { + pid: process.pid, + process: 'xterm-256color', + onData: vi.fn(), + onExit: vi.fn(), + write: vi.fn(), + resize: vi.fn(), + kill: vi.fn(), + clear: vi.fn(), + pause: vi.fn(), + resume: vi.fn() + } +})) + +vi.mock('node-pty', () => ({ spawn: mockPtySpawn })) + +vi.mock('../main/pty/posix-pty-process-groups', () => ({ + forceKillPosixPtyProcessGroups: vi.fn((_pid: number, fallback: () => void) => fallback()) +})) + +vi.mock('../main/shell-prompt-readiness-probe', () => ({ + createShellPromptReadinessProbe: mockCreateShellPromptReadinessProbe +})) + +import * as ptyChildProcessInspection from './pty-child-process-inspection' +import type { PtyHandler } from './pty-handler' +import { + beginPtyHandlerTest, + createPtyRequestHelpers, + endPtyHandlerTest +} from './pty-handler-test-harness' +import type { MockDispatcher } from './pty-handler-test-harness' + +type Inspection = { + foregroundProcess: string | null + hasChildProcesses: boolean + childProcessEvidence?: string +} + +describe('PtyHandler Windows child-process evidence', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + let inspectChildren: ReturnType + + const { spawnPty } = createPtyRequestHelpers(() => dispatcher) + + /** Spawn under the harness's POSIX platform, then answer as the Windows relay would. */ + async function spawnThenBecomeWindows(): Promise { + const { id } = await spawnPty() + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + return id + } + + async function inspect(params: Record): Promise { + return (await dispatcher.callRequest('pty.inspectProcess', params)) as Inspection + } + + beforeEach(() => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + inspectChildren = vi + .spyOn(ptyChildProcessInspection, 'inspectPtyChildProcesses') + .mockResolvedValue('no-children') + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('publishes what the host observed when the caller pays for the scan', async () => { + const id = await spawnThenBecomeWindows() + inspectChildren.mockResolvedValue('children') + + const result = await inspect({ id, scanChildProcesses: true }) + + expect(inspectChildren).toHaveBeenCalledWith(mockPtyInstance.pid) + expect(result.childProcessEvidence).toBe('children') + expect(result.hasChildProcesses).toBe(true) + }) + + it('reports an observed-empty pane as no-children, not merely false', async () => { + const id = await spawnThenBecomeWindows() + inspectChildren.mockResolvedValue('no-children') + + const result = await inspect({ id, scanChildProcesses: true }) + + // Asserted alongside the value so the case fails if the answer stops coming from a real read. + expect(inspectChildren).toHaveBeenCalledWith(mockPtyInstance.pid) + expect(result.childProcessEvidence).toBe('no-children') + expect(result.hasChildProcesses).toBe(false) + }) + + it('keeps the compatibility boolean false when the host could not observe the pane', async () => { + const id = await spawnThenBecomeWindows() + inspectChildren.mockResolvedValue('unverifiable') + + const result = await inspect({ id, scanChildProcesses: true }) + + expect(result.childProcessEvidence).toBe('unverifiable') + // Clients too old to read the verdict also read `true` as "an agent took the PTY, safe to + // type into it", so `unverifiable` must not be promoted to `true` on the shared boolean. + expect(result.hasChildProcesses).toBe(false) + }) + + it('never reads the process table for a poll, and says so instead of guessing', async () => { + const id = await spawnThenBecomeWindows() + + const result = await inspect({ id }) + + expect(inspectChildren).not.toHaveBeenCalled() + expect(result.childProcessEvidence).toBe('unverifiable') + expect(result.hasChildProcesses).toBe(false) + }) +}) diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 18c79bc7f5d..4b7c6dac2d6 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -10,11 +10,11 @@ import type { RelayDispatcher, RequestContext } from './dispatcher' import { resolveDefaultShell, resolveProcessCwd, - processHasChildren, getForegroundProcessName, isProcessAlive, listShellProfiles } from './pty-shell-utils' +import { inspectPtyChildProcesses, processHasChildren } from './pty-child-process-inspection' import { getRelayShellLaunchConfig, isRelayWslShell } from './pty-shell-launch' import { RetiredPaneSurfaceRegistry } from './retired-pane-surfaces' import { addWslEnvKeys } from '../shared/wsl-env' @@ -55,6 +55,7 @@ import { import { isTuiAgent } from '../shared/tui-agent-config' import type { TuiAgent } from '../shared/tui-agent' import { forceKillPosixPtyProcessGroups } from '../main/pty/posix-pty-process-groups' +import type { PtyChildProcessVerdict } from '../shared/terminal-process-inspection' import { terminatePtyJob } from '../main/windows/windows-pty-job' import { stripInheritedBuildModeEnv } from '../main/pty/build-mode-env' import { stripLegacyTerminalShimEnv } from '../main/pty/legacy-terminal-shim-dir' @@ -2610,6 +2611,7 @@ export class PtyHandler { private async inspectProcess(params: Record): Promise<{ foregroundProcess: string | null hasChildProcesses: boolean + childProcessEvidence?: PtyChildProcessVerdict foregroundProcessEvidence?: RemoteForegroundEvidence }> { pruneRetiredPtyIncarnations(this.retiredIncarnations) @@ -2720,24 +2722,34 @@ export class PtyHandler { evidence?.verdict === 'live' ? (evidence.processName ?? managed.pty.process) || null : managed.pty.process || null + // Derive child liveness from the same capture; do not fork a second process-table probe for + // each field/pane in an event burst. + // + // Why Windows is gated on the caller asking: this is the one field whose Windows answer costs + // a process-table read, and `inspectProcess` is the polled path (750ms/2000ms per tracked + // pane). A relay host has no `@vscode/windows-process-tree`, so the read falls back to the + // 1.36s CIM scan, and polling that would reinstate exactly the fork storm the shared table + // exists to prevent (#15209, #15036). Close and cleanup decisions ask for the scan by name; + // a poll gets the honest `unverifiable` instead of a fabricated negative. + // Why `tableUnavailable` first: it means the budgeted evidence read already gave up. Without + // this arm `inspectPtyChildProcesses` re-enters `getProcessTableSnapshot()` and joins the very + // capture this call just abandoned, blocking for all of it and spending the whole latency the + // budget exists to avoid. The destructive `pty.hasChildProcesses` RPC keeps its fresh probe. + const childProcessEvidence: PtyChildProcessVerdict = rows + ? rows.some((row) => row.ppid === managed.pty.pid) + ? 'children' + : 'no-children' + : tableUnavailable + ? 'unverifiable' + : process.platform === 'win32' && params.scanChildProcesses !== true + ? 'unverifiable' + : await inspectPtyChildProcesses(managed.pty.pid) return { foregroundProcess, - // Derive child liveness from the same capture; do not fork a second - // process-table probe for each field/pane in an event burst. Windows - // has no evidence capture, so preserve the compatibility child probe. - // - // Why the middle branch: `processHasChildren` reads the same shared capture with no budget - // of its own, so on a host slow enough to blow the evidence budget it would join the very - // capture this call just abandoned and block for all of it -- spending the whole latency - // the budget exists to avoid, on a compatibility field. `false` is what that helper already - // answers for an unreadable table, so this is the existing degraded answer reached promptly - // rather than a new one. The destructive gate is the separate `pty.hasChildProcesses` RPC, - // which keeps its unbudgeted fresh probe. - hasChildProcesses: rows - ? rows.some((row) => row.ppid === managed.pty.pid) - : tableUnavailable - ? false - : await processHasChildren(managed.pty.pid), + // `unverifiable` keeps spelling itself `false` on the compatibility field, which is what + // every client too old to read the verdict receives. + hasChildProcesses: childProcessEvidence === 'children', + childProcessEvidence, ...(evidence ? { foregroundProcessEvidence: evidence } : {}) } } diff --git a/src/relay/pty-shell-utils.test.ts b/src/relay/pty-shell-utils.test.ts index 95d6a8e050f..94953aae1bc 100644 --- a/src/relay/pty-shell-utils.test.ts +++ b/src/relay/pty-shell-utils.test.ts @@ -14,10 +14,10 @@ vi.mock('child_process', () => ({ import { resetWindowsProcessRowsSnapshotForTests } from '../main/providers/windows-foreground-process-rows' import { __setWindowsProcessTreeLoaderForTests } from '../main/windows/windows-process-table' import { resetProcessTableSnapshotForTests } from '../shared/process-table-snapshot-reader' +import { inspectPtyChildProcesses, processHasChildren } from './pty-child-process-inspection' import { getForegroundProcessName, isProcessAlive, - processHasChildren, resolveDefaultCwd, resolveWindowsDefaultShell } from './pty-shell-utils' @@ -42,6 +42,14 @@ function mockExecFile( * Feed the native Windows snapshot. A real snapshot always contains the * querying process, and the reader rejects a table without it. */ +/** A native reader that answers, but with no snapshot -- an unreadable table, not an empty one. */ +function mockUnreadableWindowsProcessTable(): void { + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2 }, + getAllProcesses: (cb: (value: undefined) => void) => cb(undefined) + })) +} + function mockWindowsProcessTable( rows: { pid: number; ppid: number; name: string; commandLine?: string }[] ): void { @@ -594,19 +602,79 @@ describe('processHasChildren', () => { }) }) - it('reports no children when the process table is unreadable', async () => { + it('reports an unreadable POSIX table as unverifiable, and still spells it false on the wire', async () => { await withProcessPlatform('linux', async () => { mockExecFile(() => new Error('ps table unavailable')) + await expect(inspectPtyChildProcesses(100)).resolves.toBe('unverifiable') await expect(processHasChildren(100)).resolves.toBe(false) }) }) +}) - it('spawns nothing on Windows, where the answer was always false', async () => { +describe('inspectPtyChildProcesses on Windows', () => { + // Why this describe exists: the relay used to `return false` here unconditionally, and a + // hardcoded negative is indistinguishable from a measurement. Every close guard reads it as + // "nothing is running here", so a Windows SSH pane running a build closed with no prompt. + it('walks the process table rather than answering from nothing', async () => { await withProcessPlatform('win32', async () => { - await expect(processHasChildren(100)).resolves.toBe(false) + mockWindowsProcessTable([ + { pid: 100, ppid: 99, name: 'cmd.exe', commandLine: 'cmd.exe' }, + { pid: 101, ppid: 100, name: 'PING.EXE', commandLine: 'ping -n 40 127.0.0.1' } + ]) - expect(execFileMock).not.toHaveBeenCalled() + await expect(inspectPtyChildProcesses(100)).resolves.toBe('children') + await expect(processHasChildren(100)).resolves.toBe(true) + }) + }) + + it('finds a grandchild the shell backgrounded, not just direct children', async () => { + await withProcessPlatform('win32', async () => { + mockWindowsProcessTable([ + { pid: 100, ppid: 99, name: 'cmd.exe', commandLine: 'cmd.exe' }, + { pid: 101, ppid: 100, name: 'node.exe', commandLine: 'node build.js' }, + { pid: 102, ppid: 101, name: 'tsc.exe', commandLine: 'tsc --watch' } + ]) + + await expect(inspectPtyChildProcesses(102)).resolves.toBe('no-children') + await expect(inspectPtyChildProcesses(101)).resolves.toBe('children') + }) + }) + + it('separates an observed-empty shell from a table it could not read', async () => { + await withProcessPlatform('win32', async () => { + mockWindowsProcessTable([{ pid: 100, ppid: 99, name: 'cmd.exe', commandLine: 'cmd.exe' }]) + await expect(inspectPtyChildProcesses(100)).resolves.toBe('no-children') + + resetWindowsProcessRowsSnapshotForTests() + mockUnreadableWindowsProcessTable() + const alive = vi.spyOn(process, 'kill').mockReturnValue(true as never) + try { + await expect(inspectPtyChildProcesses(100)).resolves.toBe('unverifiable') + // The compatibility boolean keeps spelling unverifiable `false`: it reaches clients that + // cannot read the verdict, and they read `true` as "an agent took the PTY, safe to type". + await expect(processHasChildren(100)).resolves.toBe(false) + } finally { + alive.mockRestore() + } + }) + }) + + it('does not read a missing shell as unverifiable when the kernel says it is gone', async () => { + await withProcessPlatform('win32', async () => { + // The root is absent from the snapshot, which on its own cannot distinguish a filtered + // table from an exited shell. Only ESRCH settles it. + mockWindowsProcessTable([{ pid: 900, ppid: 1, name: 'explorer.exe' }]) + const gone = vi.spyOn(process, 'kill').mockImplementation(() => { + const error = new Error('no such process') as NodeJS.ErrnoException + error.code = 'ESRCH' + throw error + }) + try { + await expect(inspectPtyChildProcesses(100)).resolves.toBe('no-children') + } finally { + gone.mockRestore() + } }) }) }) diff --git a/src/relay/pty-shell-utils.ts b/src/relay/pty-shell-utils.ts index d84faad6ae9..9c8585933a1 100644 --- a/src/relay/pty-shell-utils.ts +++ b/src/relay/pty-shell-utils.ts @@ -11,10 +11,7 @@ import { import { getFirstCommandToken } from '../shared/command-token-scanner' import { getProcessTableIndex, type ProcessTableIndex } from '../shared/process-table-index' import { PS_MAX_BUFFER_BYTES, type ProcessTableRow } from '../shared/process-table-snapshot' -import { - getFreshProcessTableSnapshot, - getProcessTableSnapshot -} from '../shared/process-table-snapshot-reader' +import { getProcessTableSnapshot } from '../shared/process-table-snapshot-reader' import { selectForegroundProcessCandidate } from '../shared/foreground-process-selection' import { resolveOuterWrapperForegroundProcess, @@ -169,43 +166,6 @@ export async function resolveProcessCwd(pid: number, fallbackCwd: string): Promi return fallbackCwd } -/** - * Check whether a process has child processes. - * - * Why the shared snapshot and not `pgrep -P`: this answers one field of - * `pty.inspectProcess`, which every tracked pane polls on a 750ms/2000ms - * cadence, and the fork was neither cached nor coalesced. procps-ng opens six - * procfs files per process to resolve a ppid — including a `/proc//ctty` - * that never exists on Linux — so one call cost O(host process count) syscalls, - * ~4k opens per pgrep on a 690-process host, at up to 8 forks/sec (#13537). - * `getForegroundProcessName` in the same RPC already captured the TTL-cached - * `ps` table, whose index carries the parent/child map, so the answer is free. - * - * `fresh` opts out of that TTL. A poll can read a 500ms-old table because its - * next tick corrects it, but a close or cleanup decision acts on the answer - * once and destructively — a child that started inside the TTL would be killed - * with no confirmation. `pgrep` scanned per call, so anything that decides - * has to keep scanning per call. - */ -export async function processHasChildren( - pid: number, - options?: { fresh?: boolean } -): Promise { - // Windows has no `ps`; the previous `pgrep` fork always failed here too, so - // this keeps the same answer without spawning anything to reach it. - if (process.platform === 'win32') { - return false - } - try { - const rows = options?.fresh - ? await getFreshProcessTableSnapshot() - : await getProcessTableSnapshot() - return (getProcessTableIndex(rows).childrenByPpid.get(pid)?.length ?? 0) > 0 - } catch { - return false - } -} - // Why: signal 0 probes existence without delivering a signal. Only ESRCH ("no // such process") proves the pid is gone; EPERM means it exists but is // unsignalable, so treat every non-ESRCH outcome as alive. Kept conservative so diff --git a/src/renderer/src/assets/main.css b/src/renderer/src/assets/main.css index 8187fe496c7..e3d267cb353 100644 --- a/src/renderer/src/assets/main.css +++ b/src/renderer/src/assets/main.css @@ -392,9 +392,9 @@ z-index: 40 !important; } -/* Keep interruption controls above unrelated updater/onboarding chrome. */ +/* Above the z-40 updater/onboarding chrome, below the floating workspace panel's z-45. */ .native-chat-pane-shell:has([data-native-chat-working='true']) { - z-index: 50; + z-index: 44; } [data-sonner-toaster] [data-sonner-toast][data-styled='true'] { diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.set-location-warm.test.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.set-location-warm.test.tsx new file mode 100644 index 00000000000..fd1a16f7f0a --- /dev/null +++ b/src/renderer/src/components/NewWorkspaceComposerCard.set-location-warm.test.tsx @@ -0,0 +1,123 @@ +// @vitest-environment happy-dom + +import React from 'react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { hostOptions, renderCard } from './NewWorkspaceComposerCard.test-fixture' +import type { ProjectHostSetupOption } from '@/lib/project-host-setup-options' + +// Counts evaluations of the set-location chunk. A dynamic import evaluates a module once, +// so this only moves when the composer actually reaches for the chunk. +const chunk = vi.hoisted(() => ({ loads: 0 })) + +// Renders a marker unconditionally so the "warming did not mount it" assertion below can +// actually fail; a `() => null` stub would make that check vacuous. +vi.mock('@/components/new-workspace/SetProjectLocationDialog', () => { + chunk.loads += 1 + return { + SetProjectLocationDialog: () =>

+ } +}) + +vi.mock('@/store', () => ({ + useAppStore: Object.assign( + (selector: (state: unknown) => unknown) => + selector({ + closeModal: vi.fn(), + openModal: vi.fn(), + openSettingsPage: vi.fn(), + openSettingsTarget: vi.fn(), + setRuntimeEnvironmentStatus: vi.fn(), + setupProjectExistingFolder: vi.fn(), + setupProjectClone: vi.fn(), + activeModal: 'new-workspace-composer', + settings: { defaultTuiAgent: null, disabledTuiAgents: [] }, + updateSettings: vi.fn(), + projects: [], + repos: [] + }), + { getState: () => ({}) } + ) +})) + +vi.mock('@/components/contextual-tours/use-contextual-tour', () => ({ + useContextualTour: vi.fn() +})) + +vi.mock('@/components/ui/tooltip', () => ({ + Tooltip: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipContent: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipTrigger: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('@/components/agent/AgentCombobox', () => ({ + default: () => +})) + +vi.mock('@/components/sidebar/AddRemoteHostDialog', () => ({ + AddRemoteHostDialog: () => null +})) + +vi.mock('@/components/sparse/SparseCheckoutPresetSelect', () => ({ + default: () => null +})) + +vi.mock('@/components/new-workspace/SmartWorkspaceNameField', () => ({ + default: () => +})) + +vi.mock('@/components/new-workspace/ProjectCombobox', () => ({ + default: () =>
+})) + +const readyOnlyHostOptions = hostOptions.filter((option) => option.kind === 'ready') +// A disconnected host is a needs-setup row with no "Set location" action, so it must not warm. +const unavailableHostOptions: ProjectHostSetupOption[] = [ + ...readyOnlyHostOptions, + { + kind: 'needs-setup', + id: 'needs-setup:ssh:offline', + projectId: 'project-group:platform', + hostId: 'ssh:offline', + label: 'Offline box', + detail: 'Not connected', + isAvailable: false, + attention: false, + canSetLocation: false + } +] + +// Declaration order matters here and nowhere else: a module evaluates once, so the +// no-warm cases have to observe the counter before anything warms it. +describe('NewWorkspaceComposerCard set-location chunk warm', () => { + let container: HTMLDivElement | null = null + + afterEach(() => { + container?.remove() + container = null + }) + + it('does not warm the chunk when no host needs its location set', async () => { + container = await renderCard({ projectHostSetupOptions: readyOnlyHostOptions }) + + expect( + [...container.querySelectorAll('button')].some((button) => + button.textContent?.includes('Set project location') + ) + ).toBe(false) + expect(chunk.loads).toBe(0) + }) + + it('does not warm the chunk when the needs-setup host cannot take a location', async () => { + container = await renderCard({ projectHostSetupOptions: unavailableHostOptions }) + + expect(chunk.loads).toBe(0) + }) + + it('warms the chunk on mount for a needs-setup host, before Set project location is clicked', async () => { + container = await renderCard() + + expect(chunk.loads).toBe(1) + // Warming must not mount the dialog; it still waits on an explicit click. + expect(document.body.querySelector('[data-testid="set-project-location-dialog"]')).toBeNull() + }) +}) diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx index 9cdea1d7d54..06718fab91b 100644 --- a/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx +++ b/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx @@ -1,11 +1,8 @@ // @vitest-environment happy-dom import React, { act } from 'react' -import { createRoot } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import NewWorkspaceComposerCard from './NewWorkspaceComposerCard' -import type { NewWorkspaceProjectOption } from '@/lib/new-workspace-project-options' -import type { ProjectHostSetupOption } from '@/lib/project-host-setup-options' +import { renderCard } from './NewWorkspaceComposerCard.test-fixture' const storeMocks = vi.hoisted(() => ({ closeModal: vi.fn(), @@ -99,118 +96,6 @@ vi.mock('@/components/new-workspace/SetProjectLocationDialog', () => ({ ) : null })) -const projectOptions: NewWorkspaceProjectOption[] = [ - { - kind: 'project-group', - id: 'project-group:platform', - projectGroupId: 'platform', - displayName: 'Platform', - badgeColor: 'var(--muted-foreground)', - detail: '/workspace/platform', - parentPath: '/workspace/platform', - connectionId: null - } -] - -const hostOptions: ProjectHostSetupOption[] = [ - { - kind: 'ready', - id: 'setup-local', - projectId: 'project-group:platform', - hostId: 'local', - repoId: 'repo-a', - label: 'Local Mac', - detail: 'Orca', - path: '/Users/alice/orca' - }, - { - kind: 'needs-setup', - id: 'needs-setup:ssh:devbox', - projectId: 'project-group:platform', - hostId: 'ssh:devbox', - label: 'Devbox', - detail: 'Project location not set', - isAvailable: true, - attention: false, - canSetLocation: true - } -] - -function renderCard( - overrides: Partial> = {} -): HTMLDivElement { - const container = document.createElement('div') - document.body.appendChild(container) - const root = createRoot(container) - act(() => { - root.render( - {}} - eligibleRepos={[]} - repoId="repo-a" - projectOptions={projectOptions} - selectedProjectId="project-group:platform" - selectedRepoIsGit - onRepoChange={() => {}} - onProjectChange={() => {}} - primaryActionLabel="Create workspace" - name="" - onNameValueChange={() => {}} - onSmartGitHubItemSelect={() => {}} - onSmartGitLabItemSelect={() => {}} - onSmartBranchSelect={() => {}} - onSmartLinearIssueSelect={() => {}} - smartNameSelection={null} - onClearSmartNameSelection={() => {}} - canReuseSelectedBranch={false} - reuseSelectedBranch={false} - onReuseSelectedBranchChange={() => {}} - forkPushWarning={null} - detectedAgentIds={null} - onOpenAgentSettings={() => {}} - advancedOpen={false} - onToggleAdvanced={() => {}} - parentWorktreeId={null} - onParentWorktreeIdChange={() => {}} - createDisabled={false} - projectError={null} - creating={false} - onCreate={() => {}} - note="" - onNoteChange={() => {}} - setupConfig={null} - requiresExplicitSetupChoice={false} - setupDecision={null} - onSetupDecisionChange={() => {}} - setupAgentStartupPolicy="start-immediately" - onSetupAgentStartupPolicyChange={() => {}} - shouldWaitForSetupCheck={false} - resolvedSetupDecision={null} - createError={null} - selectedRepoConnectionId={null} - selectedRepoSshStatus={null} - selectedRepoRequiresConnection={false} - selectedRepoConnectInProgress={false} - onConnectSelectedRepo={async () => {}} - canUseSparseCheckout={false} - sparsePresets={[]} - sparseSelectedPresetId={null} - onSparseSelectPreset={() => {}} - branchNameOverride={undefined} - onBranchNameOverrideChange={() => {}} - branchesEnabled={false} - setupControlsEnabled={false} - sparseControlsEnabled={false} - projectHostSetupOptions={hostOptions} - selectedProjectHostSetupId="setup-local" - {...overrides} - /> - ) - }) - return container -} - describe('NewWorkspaceComposerCard set location', () => { let container: HTMLDivElement | null = null @@ -225,9 +110,10 @@ describe('NewWorkspaceComposerCard set location', () => { container = null }) - it('opens set-location over the composer without leaving the create dialog', () => { + // Async because the dialog is a lazy chunk: the click mounts Suspense, the chunk resolves next tick. + it('opens set-location over the composer without leaving the create dialog', async () => { const nestedOpenChanges: boolean[] = [] - container = renderCard({ + container = await renderCard({ onNestedDialogOpenChange: (open) => nestedOpenChanges.push(open) }) @@ -239,6 +125,7 @@ describe('NewWorkspaceComposerCard set location', () => { ) expect(setLocation).toBeTruthy() act(() => setLocation?.click()) + await act(async () => {}) const dialog = document.body.querySelector('[data-testid="set-project-location-dialog"]') expect(dialog?.getAttribute('data-host')).toBe('Devbox') @@ -249,10 +136,12 @@ describe('NewWorkspaceComposerCard set location', () => { expect(storeMocks.openSettingsPage).not.toHaveBeenCalled() }) - it('closes the nested dialog before publishing the ready run target', () => { + // Async for the same reason: without the flush this only passes when an earlier + // test in this file already resolved the shared lazy chunk. + it('closes the nested dialog before publishing the ready run target', async () => { const nestedOpenChanges: boolean[] = [] const setupChanges: string[] = [] - container = renderCard({ + container = await renderCard({ onNestedDialogOpenChange: (open) => nestedOpenChanges.push(open), onProjectHostSetupChange: (setupId) => setupChanges.push(setupId) }) @@ -264,6 +153,7 @@ describe('NewWorkspaceComposerCard set location', () => { (button) => button.textContent?.includes('Set project location') ) act(() => setLocation?.click()) + await act(async () => {}) const complete = [...document.body.querySelectorAll('button')].find( (button) => button.textContent === 'Complete location' ) diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.test-fixture.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.test-fixture.tsx new file mode 100644 index 00000000000..c8eb2b56f43 --- /dev/null +++ b/src/renderer/src/components/NewWorkspaceComposerCard.test-fixture.tsx @@ -0,0 +1,120 @@ +import React, { act } from 'react' +import { createRoot } from 'react-dom/client' +import NewWorkspaceComposerCard from './NewWorkspaceComposerCard' +import type { NewWorkspaceProjectOption } from '@/lib/new-workspace-project-options' +import type { ProjectHostSetupOption } from '@/lib/project-host-setup-options' + +export const projectOptions: NewWorkspaceProjectOption[] = [ + { + kind: 'project-group', + id: 'project-group:platform', + projectGroupId: 'platform', + displayName: 'Platform', + badgeColor: 'var(--muted-foreground)', + detail: '/workspace/platform', + parentPath: '/workspace/platform', + connectionId: null + } +] + +export const hostOptions: ProjectHostSetupOption[] = [ + { + kind: 'ready', + id: 'setup-local', + projectId: 'project-group:platform', + hostId: 'local', + repoId: 'repo-a', + label: 'Local Mac', + detail: 'Orca', + path: '/Users/alice/orca' + }, + { + kind: 'needs-setup', + id: 'needs-setup:ssh:devbox', + projectId: 'project-group:platform', + hostId: 'ssh:devbox', + label: 'Devbox', + detail: 'Project location not set', + isAvailable: true, + attention: false, + canSetLocation: true + } +] + +export async function renderCard( + overrides: Partial> = {} +): Promise { + const container = document.createElement('div') + document.body.appendChild(container) + const root = createRoot(container) + act(() => { + root.render( + {}} + eligibleRepos={[]} + repoId="repo-a" + projectOptions={projectOptions} + selectedProjectId="project-group:platform" + selectedRepoIsGit + onRepoChange={() => {}} + onProjectChange={() => {}} + primaryActionLabel="Create workspace" + name="" + onNameValueChange={() => {}} + onSmartGitHubItemSelect={() => {}} + onSmartGitLabItemSelect={() => {}} + onSmartBranchSelect={() => {}} + onSmartLinearIssueSelect={() => {}} + smartNameSelection={null} + onClearSmartNameSelection={() => {}} + canReuseSelectedBranch={false} + reuseSelectedBranch={false} + onReuseSelectedBranchChange={() => {}} + forkPushWarning={null} + detectedAgentIds={null} + onOpenAgentSettings={() => {}} + advancedOpen={false} + onToggleAdvanced={() => {}} + parentWorktreeId={null} + onParentWorktreeIdChange={() => {}} + createDisabled={false} + projectError={null} + creating={false} + onCreate={() => {}} + note="" + onNoteChange={() => {}} + setupConfig={null} + requiresExplicitSetupChoice={false} + setupDecision={null} + onSetupDecisionChange={() => {}} + setupAgentStartupPolicy="start-immediately" + onSetupAgentStartupPolicyChange={() => {}} + shouldWaitForSetupCheck={false} + resolvedSetupDecision={null} + createError={null} + selectedRepoConnectionId={null} + selectedRepoSshStatus={null} + selectedRepoRequiresConnection={false} + selectedRepoConnectInProgress={false} + onConnectSelectedRepo={async () => {}} + canUseSparseCheckout={false} + sparsePresets={[]} + sparseSelectedPresetId={null} + onSparseSelectPreset={() => {}} + branchNameOverride={undefined} + onBranchNameOverrideChange={() => {}} + branchesEnabled={false} + setupControlsEnabled={false} + sparseControlsEnabled={false} + projectHostSetupOptions={hostOptions} + selectedProjectHostSetupId="setup-local" + {...overrides} + /> + ) + }) + // Settle the mount-time chunk warm before the click, so the click's import() is not + // overlapping an in-flight one (vitest's module runner serialises those; a browser does not). + await act(async () => {}) + return container +} diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.tsx index c6cca5192ea..d17bb8ba56d 100644 --- a/src/renderer/src/components/NewWorkspaceComposerCard.tsx +++ b/src/renderer/src/components/NewWorkspaceComposerCard.tsx @@ -11,7 +11,8 @@ import { AddRemoteHostDialog, type AddRemoteHostMode } from '@/components/sidebar/AddRemoteHostDialog' -import { SetProjectLocationDialog } from '@/components/new-workspace/SetProjectLocationDialog' +import { lazyWithRetry } from '@/lib/lazy-with-retry' +import type * as SetProjectLocationDialogModule from '@/components/new-workspace/SetProjectLocationDialog' import { unwrapRuntimeRpcResult } from '@/runtime/runtime-rpc-client' import { withUiConnectTimeout } from '@/ssh/ssh-connect-ui-timeout' import { isSshConnectInFlight, trackSshConnect } from '@/ssh/ssh-connect-in-flight' @@ -37,6 +38,20 @@ import { import { getSshStatusLabel } from './new-workspace/new-workspace-composer-ssh-status' import { useComposerFileDragOver } from './new-workspace/use-composer-file-drag-over' +// Why lazy: this pulls the ~41 KB project-location browser onto the boot graph, and nothing +// reaches it without an explicit "Set location" click. Shared with the warm below so both hit +// the same module-map entry. +const loadSetProjectLocationDialog = (): Promise => + import('@/components/new-workspace/SetProjectLocationDialog') + +const SetProjectLocationDialog = lazyWithRetry( + () => + loadSetProjectLocationDialog().then((module) => ({ + default: module.SetProjectLocationDialog + })), + { reloadKey: 'set-project-location-dialog' } +) + export default function NewWorkspaceComposerCard( props: NewWorkspaceComposerCardProps ): React.JSX.Element { @@ -83,6 +98,9 @@ export default function NewWorkspaceComposerCard( const [setLocationOption, setSetLocationOption] = React.useState( null ) + // Why sticky: the dialog animates itself closed off its own `option` prop, so unmounting it + // when the option clears would cut that animation short. + const [setLocationDialogMounted, setSetLocationDialogMounted] = React.useState(false) const selectedRepo = eligibleRepos.find((candidate) => candidate.id === repoId) const selectedRepoName = selectedRepo?.displayName ?? selectedRepo?.path ?? 'This project' @@ -96,6 +114,16 @@ export default function NewWorkspaceComposerCard( const needsSetupProjectHostSetupOptions = projectHostSetupOptions.filter( (option) => option.kind === 'needs-setup' ) + // Warm on the precursor: the "Set location" row only renders for a needs-setup host that can + // still take one, so the chunk resolves while the picker is being read rather than on the click. + const hasSetLocationOption = needsSetupProjectHostSetupOptions.some( + (option) => option.canSetLocation + ) + React.useEffect(() => { + if (hasSetLocationOption) { + void loadSetProjectLocationDialog().catch(() => {}) + } + }, [hasSetLocationOption]) const shouldShowRunTargetPicker = readyProjectHostSetupOptions.length > 0 || ephemeralVmRecipes.length > 0 || @@ -177,6 +205,7 @@ export default function NewWorkspaceComposerCard( }, [onAddProjectOverride, openModal]) const handleSetLocation = React.useCallback( (option: NeedsProjectHostOption): void => { + setSetLocationDialogMounted(true) setSetLocationOption(option) onNestedDialogOpenChange?.(true) }, @@ -319,14 +348,18 @@ export default function NewWorkspaceComposerCard( submitShortcutModifierLabel={getScreenSubmitModifierLabel()} /> - + {setLocationDialogMounted ? ( + + + + ) : null}
) } diff --git a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx index 52ddbf1ba5d..bb3ffbfd621 100644 --- a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx +++ b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx @@ -24,6 +24,7 @@ export function TerminalWorkspaceDialogs({ saveDialogFile, saveDialogFileId, setWindowCloseDialogOpen, + windowCloseDialogKind, windowCloseDialogOpen } = controller return ( @@ -82,10 +83,15 @@ export function TerminalWorkspaceDialogs({ {translate('auto.components.Terminal.2fa9c69ff3', 'Close Window?')} - {translate( - 'auto.components.Terminal.7958465754', - 'There are local terminals with running processes. Close the window anyway?' - )} + {windowCloseDialogKind === 'unverifiable' + ? translate( + 'auto.components.Terminal.b7c1f0a934', + 'A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?' + ) + : translate( + 'auto.components.Terminal.7958465754', + 'There are terminals with running processes. Close the window anyway?' + )} diff --git a/src/renderer/src/components/cmd-j/palette-live-status.test.tsx b/src/renderer/src/components/cmd-j/palette-live-status.test.tsx index a2656104d87..3cc6a9077dd 100644 --- a/src/renderer/src/components/cmd-j/palette-live-status.test.tsx +++ b/src/renderer/src/components/cmd-j/palette-live-status.test.tsx @@ -451,7 +451,7 @@ describe('palette live status', () => { expect(dotLabels()).toEqual(['Needs permission']) }) - it('cuts the pip out of the dialog surface, and out of accent when selected', async () => { + it('keeps the attention glyph knockout popover-colored when its row is selected', async () => { setAgentState('working') await act(async () => { testRoot.render( @@ -469,18 +469,11 @@ describe('palette live status', () => { ) }) - const pip = testContainer.querySelector('[aria-hidden="true"].rounded-full') + const pip = testContainer.querySelector('[aria-hidden="true"]') expect(pip).not.toBeNull() - // Why popover and not background: the CommandDialog surface is --popover (#171717 dark), while - // --background is the app canvas (#0a0a0a) — the mismatch punched a dark halo through each row. expect(pip?.className).toContain('bg-popover') expect(pip?.className).toContain('ring-popover') - expect(pip?.className).not.toContain('bg-background') - expect(pip?.className).toContain( - 'group-data-[selected=true]:bg-[var(--jump-palette-selection-surface)]' - ) - expect(pip?.className).toContain( - 'group-data-[selected=true]:ring-[var(--jump-palette-selection-surface)]' - ) + expect(pip?.className).toContain('rounded-full') + expect(pip?.className).not.toContain('group-data-[selected=true]') }) }) diff --git a/src/renderer/src/components/cmd-j/palette-live-status.tsx b/src/renderer/src/components/cmd-j/palette-live-status.tsx index 432b6c35210..da59663490f 100644 --- a/src/renderer/src/components/cmd-j/palette-live-status.tsx +++ b/src/renderer/src/components/cmd-j/palette-live-status.tsx @@ -9,7 +9,6 @@ import { buildExplicitEntriesByTabId, type TabPaneInputSources } from '@/components/sidebar/smart-attention' -import { cn } from '@/lib/utils' import { isExplicitAgentStatusFresh } from '@/lib/agent-status' import { getLiveAgentStatusByWorktreeId } from '@/lib/worktree-activity-state' import { @@ -255,15 +254,8 @@ export function PaletteRecentTabStatusDot({ {fallback}