diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 61a73b64dbe..4e1da9fab26 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -111,14 +111,19 @@ describe('incident monitor evaluator', () => { }) }) - it('freezes when postgres retries exceed the recalibrated ceiling', () => { - const sample = healthySample() - sample.sources['relay-logs']!.signals['relay.postgres_retries'] = - signal(INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries + 1) - expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + // Why: the global relay_cells lock made retries a steady-state rate (24 h p99 + // 1320/5min on 2026-09-04); the bar fences only unbounded growth beyond that. + it('tolerates the measured healthy retry rate and freezes above the bar', () => { + const healthy = healthySample() + healthy.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(1504) + expect(evaluateIncidentSample(healthy, startedAt).status).toBe('green') + + const incident = healthySample() + incident.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(2001) + expect(evaluateIncidentSample(incident, startedAt)).toMatchObject({ status: 'freeze', failures: [ - expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 300 }) + expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 2000 }) ] }) }) diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index 868bb86fb93..a121568d918 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -32,11 +32,20 @@ export const INCIDENT_MONITOR_THRESHOLDS = { relayPoolWaiting: 800, relayPoolWaitMs: 2_500, // Why: successful lock retries are the contention machinery working, not harm. - // Healthy 2026-08-26 baseline bursts to 234/5min (26% of windows crossed the old - // bar of 20, set unmeasured at the monitor's 2026-07-28 birth); the 2026-08-23 - // incident ran ~2,200-3,000/5min. 300 clears healthy bursts with ~10x incident - // margin; relayPostgresRetryExhausted below bounds the terminally failed share. - relayPostgresRetries: 300, + // Recalibrated 2026-09-04 from 300, which was set 2026-08-26 when healthy bursts + // reached 234/5min. The global relay_cells FOR UPDATE lock has since become the + // fleet's steady state: measured fleet-wide (director + cells, summed per five + // minutes) 2026-09-03T05Z..2026-09-04T05Z p50 430 / p90 924 / p99 1320 / max + // 1504, with 55% of windows over 300 and only 22% of 15-minute gates clean, so + // the bar blocked the very cell roll that carries the 500 ms lock wait (#18521) + // and the beginProof crash guard to the cells. The 2026-08-23 lock incident on + // this same metric peaked at 1510 in one window and 646 in the next, so it is + // not separable from today's contention by retries alone; it is caught by + // relayPostgresRetryExhausted (467 at the peak vs a 300 bar), director + // concurrency, and the pool bars. 2000 passes every healthy 15-minute window + // measured in the last 24 h and still fences unbounded growth. Re-tighten once + // the fleet is on the 500 ms lock wait and the baseline is re-measured. + relayPostgresRetries: 2000, // Why: 300 per five minutes, recalibrated 2026-09-04 from a bar of zero that no // production window has cleared since #18521 shipped to the director. That // change cut the request-path cell-inventory wait from the 1 s pool lock_timeout @@ -48,7 +57,7 @@ export const INCIDENT_MONITOR_THRESHOLDS = { // quiet hours p50 2 / max 36; pre-#18521 daytime p50 10 / p90 25 / max 87; // post-#18521 p50 42 / p90 147 / max 220. The 2026-08-23 lock incident peaked // at 467. 300 clears every measured healthy window and still sits below the - // incident shape; relayPostgresRetries above stays the ~10x discriminator. + // incident shape; retries above fence only unbounded growth. // User-facing /v1/assign 503 share did not move with #18521 (13.9% old image // vs 12.3% new, same evening), so exhaustion is not a proxy for user harm. relayPostgresRetryExhausted: 300, diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 337d3f1b20f..870c95dd413 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -99,7 +99,7 @@ durably marked consumed before mutation and cannot authorize another run. | Cloud SQL deadlocks | over 0 | | Relay pool waiters | over 800 | | Relay pool wait | over 2,500 ms | -| PostgreSQL retries in five minutes | over 300 | +| PostgreSQL retries in five minutes | over 2,000 | | Exhausted PostgreSQL retries in five minutes | over 300 | | Director instances | outside 5–6 | | Director CPU or memory | over 80% | @@ -138,7 +138,23 @@ heartbeats, and matching live admission. `jsonPayload.event="orca_relay_postgres_transaction_retry"` in production logs: healthy-day bursts reach 234/5min with zero exhausted retries and 26% of five-minute windows over 20, while the 2026-08-23 lock-contention - incident ran roughly 2,200–3,000/5min. + incident ran roughly 2,200–3,000/5min by raw log-line count (the gate's + own `orca_relay_postgres_retries` metric read 1,510 for that window; see the + 2026-09-04 entry). +- Recalibrated the PostgreSQL-retry freeze from 300 to 2,000 per five minutes + (2026-09-04). Basis: the global `relay_cells FOR UPDATE` lock made + successful retries a steady-state rate. Measured fleet-wide (director + + cells, summed per five minutes from the `orca_relay_postgres_retries` + log metric) over 2026-09-03T05Z..2026-09-04T05Z: p50 430 / p90 924 / + p99 1,320 / max 1,504; 55% of windows over 300; only 22% of 15-minute gates + clean at 300 versus 100% at 2,000. Three read-only dry-runs on 2026-09-04 + froze on this bar (runs 33836470590, 33838698725) or on a genuine six-cell + crash storm (33837160275), blocking the same-cap roll that carries #18521 + and the `beginProof` crash guard to the 23 cells. The 2026-08-23 incident + on this metric peaked at 1,510 then 646, so retries alone no longer + separate it from today's baseline; the exhausted-retry bar (incident peak + 467 vs bar 300), director concurrency, and the pool bars carry that role. + Re-tighten after the fleet is on the 500 ms lock wait. - Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters diff --git a/config/scripts/locale-ko-key-overrides.json b/config/scripts/locale-ko-key-overrides.json index f368ecc3cbc..bf5f62d1fa5 100644 --- a/config/scripts/locale-ko-key-overrides.json +++ b/config/scripts/locale-ko-key-overrides.json @@ -492,7 +492,7 @@ "ko": "agent CLI를 찾지 못했습니다. 하나를 설치하거나 설정에서 기본 agent를 선택하세요." }, "auto.components.Terminal.7958465754": { - "ko": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" + "ko": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" }, "auto.components.Terminal.cdc9ac4b2d": { "ko": "편집기" diff --git a/src/main/codex/codex-app-server-process-teardown.test.ts b/src/main/codex/codex-app-server-process-teardown.test.ts index 1ddb672a531..cec8f91d081 100644 --- a/src/main/codex/codex-app-server-process-teardown.test.ts +++ b/src/main/codex/codex-app-server-process-teardown.test.ts @@ -1,7 +1,14 @@ import type { ChildProcess } from 'node:child_process' -import { describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + findSelfInitiatedTreeKills, + resetSelfInitiatedTreeKillLogForTest +} from '../crash-reporting/self-initiated-tree-kill-log' import { terminateCodexAppServerProcessTree } from './codex-app-server-process-teardown' +/** Above pid_max on every supported POSIX host, so the group signal is a real ESRCH. */ +const UNREACHABLE_PGID = 2_147_483_647 + function child() { return { pid: 1234, @@ -10,6 +17,10 @@ function child() { } describe('terminateCodexAppServerProcessTree', () => { + beforeEach(() => { + resetSelfInitiatedTreeKillLogForTest() + }) + it('waits for the Windows tree kill before releasing the wrapper', async () => { const target = child() const release = Promise.withResolvers() @@ -120,6 +131,55 @@ describe('terminateCodexAppServerProcessTree', () => { expect(target.kill).not.toHaveBeenCalled() }) + /** + * `selfInitiatedTreeKillCount` decides whether a `render-process-gone` was + * ours. A group that had already exited was killed by nobody, so crediting it + * puts a suspect in the five-second window that Orca never issued. Exercised + * through the real `process.kill(-pgid)` because the swallow being tested + * lives in the production default, not in an injectable seam. + */ + it('does not claim a snapshot group that was already gone', async () => { + const target = { pid: UNREACHABLE_PGID, kill: vi.fn(() => true) as ChildProcess['kill'] } + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ + rootPgid: UNREACHABLE_PGID, + descendants: [], + capturedAtMs: 1 + }), + terminateDescendants: async () => true + }) + ).resolves.toBe(true) + + expect(target.kill).toHaveBeenLastCalledWith('SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([]) + }) + + it('claims a snapshot group the signal actually reached', async () => { + const target = child() + const signalProcessGroup = vi.fn() + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ rootPgid: 1234, descendants: [], capturedAtMs: 1 }), + terminateDescendants: async () => true, + signalProcessGroup + }) + ).resolves.toBe(true) + + expect(signalProcessGroup).toHaveBeenCalledWith(1234, 'SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([ + expect.objectContaining({ + pid: 1234, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' + }) + ]) + }) + it('tears down 40 dedicated groups without process-table scans or cross-group fanout', async () => { const killMocks = Array.from({ length: 40 }, () => vi.fn(() => true)) const targets = killMocks.map((kill, index) => ({ diff --git a/src/main/codex/codex-app-server-process-teardown.ts b/src/main/codex/codex-app-server-process-teardown.ts index a35ad4d3164..5a9c6e3574b 100644 --- a/src/main/codex/codex-app-server-process-teardown.ts +++ b/src/main/codex/codex-app-server-process-teardown.ts @@ -128,19 +128,24 @@ async function terminatePosixTree( if (descendantsExited && snapshot.rootPgid === rootPid) { const signalGroup = deps.signalProcessGroup ?? - ((pgid: number, signal: NodeJS.Signals) => { - try { - process.kill(-pgid, signal) - } catch { - // Group already exited. - } + ((pgid: number, signal: NodeJS.Signals) => process.kill(-pgid, signal)) + let groupSignalled = false + try { + signalGroup(snapshot.rootPgid, 'SIGKILL') + groupSignalled = true + } catch { + // Already-gone is still the desired outcome, but nothing here killed it, + // and a crumb for a kill we never landed is a false render-process-gone suspect. + } + if (groupSignalled) { + // Outside the try, as in terminateDedicatedPosixGroup: that catch is the + // already-gone contract, not a breadcrumb handler. + recordSelfInitiatedTreeKill({ + pid: snapshot.rootPgid, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' }) - signalGroup(snapshot.rootPgid, 'SIGKILL') - recordSelfInitiatedTreeKill({ - pid: snapshot.rootPgid, - site: 'codex-app-server-teardown', - scope: 'posix-process-group' - }) + } } if (!descendantsExited) { child.kill('SIGCONT') diff --git a/src/main/crash-reporting/gone-time-system-memory.ts b/src/main/crash-reporting/gone-time-system-memory.ts deleted file mode 100644 index 7cca89d2d4b..00000000000 --- a/src/main/crash-reporting/gone-time-system-memory.ts +++ /dev/null @@ -1,77 +0,0 @@ -import type { CrashReportDetailValue } from '../../shared/crash-reporting' - -// ─── System memory at gone time ───────────────────────────────────── -// Why: the system outlives the crashed process, so this IS sampleable at -// process-gone — it separates "renderer grew huge" from "machine out of -// memory/commit", which the per-process buckets alone cannot. -// Timing honesty: this reads AFTER the crashed process's memory returned to -// the OS, so free/swapFree can look healthier than they were at kill time. -// Platform honesty: swap* exist on Windows/Linux only. On Linux `free` is -// /proc/meminfo MemFree and is NOT the pressure signal — it excludes page cache -// and other reclaimable memory; `available` (MemAvailable, Linux-only) is. On -// macOS `free` is near-meaningless (file cache and compression keep it low on -// healthy machines); fileBacked/purgeable are the only reclaimability proxy this -// API gives there, and none of these fields answers "was the machine under -// pressure" on macOS — that needs a signal Electron does not expose. - -type CrashReportDetails = Record - -export function memoryKBFieldMB(value: unknown): number | undefined { - const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined - return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) -} - -type SystemMemoryInfoLike = { - total?: unknown - free?: unknown - available?: unknown - swapTotal?: unknown - swapFree?: unknown - fileBacked?: unknown - purgeable?: unknown -} - -type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null - -function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { - const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) - .getSystemMemoryInfo - if (typeof read !== 'function') { - return null - } - try { - return read.call(process) - } catch { - return null - } -} - -let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo - -export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { - systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo -} - -export function getSystemMemoryAtGoneDetails(): CrashReportDetails { - const info = systemMemoryInfoReader() - if (!info) { - return {} - } - const details: CrashReportDetails = {} - const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ - ['total', 'systemMemoryTotalMB'], - ['free', 'systemMemoryFreeMB'], - ['available', 'systemMemoryAvailableMB'], - ['swapTotal', 'systemMemorySwapTotalMB'], - ['swapFree', 'systemMemorySwapFreeMB'], - ['fileBacked', 'systemMemoryFileBackedMB'], - ['purgeable', 'systemMemoryPurgeableMB'] - ] - for (const [field, key] of fields) { - const mb = memoryKBFieldMB(info[field]) - if (mb !== undefined) { - details[key] = mb - } - } - return details -} diff --git a/src/main/crash-reporting/pre-gone-host-memory.test.ts b/src/main/crash-reporting/pre-gone-host-memory.test.ts new file mode 100644 index 00000000000..0df13d4fee5 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.test.ts @@ -0,0 +1,379 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + getSystemMemoryDetails, + setSystemMemoryInfoReaderForTest, + withSwapVolumeFreeSpace +} from './system-memory-details' +import { + readSwapVolumeFreeSpace, + setSwapVolumeFreeSpaceReaderForTest, + type SwapVolumeFreeSpace +} from './swap-volume-free-space' +import { samplePreGoneSystemMemory } from './pre-gone-host-memory' +import { + buildProcessGoneCrashDetails, + resetPreGoneCrashSamplingForTest, + samplePreGoneProcessMetrics, + startPreGoneCrashSampling +} from './process-gone-diagnostics' + +type MetricFixture = { + pid: number + creationTime: number + type: string + memory: { workingSetSize: number; peakWorkingSetSize?: number; privateBytes?: number } +} + +const { appMetricsMock } = vi.hoisted(() => ({ + appMetricsMock: vi.fn<() => MetricFixture[]>(() => []) +})) + +vi.mock('electron', () => ({ app: { getAppMetrics: appMetricsMock } })) + +const BROWSER_AND_RENDERER: MetricFixture[] = [ + { pid: 10, creationTime: 1, type: 'Browser', memory: { workingSetSize: 1024 * 250 } }, + { + pid: 11, + creationTime: 2, + type: 'Tab', + memory: { workingSetSize: 1024 * 400, peakWorkingSetSize: 1024 * 420, privateBytes: 1024 * 260 } + } +] + +const BROWSER_ONLY: MetricFixture[] = [BROWSER_AND_RENDERER[0]] + +const UNDER_COMMIT_PRESSURE = { + total: 16_000 * 1024, + free: 400 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 200 * 1024 +} + +const AFTER_THE_CORPSE_RELEASED = { + total: 16_000 * 1024, + free: 3_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 2_900 * 1024 +} + +// Commit limit ~= RAM: a disabled or fixed pagefile, which no amount of empty +// disk can grow into. `swapTotal > total` is all this API can say about that. +const FIXED_PAGEFILE_UNDER_PRESSURE = { + total: 16_000 * 1024, + free: 300 * 1024, + swapTotal: 16_100 * 1024, + swapFree: 180 * 1024 +} + +const NO_PAGEFILE_UNDER_PRESSURE = { + ...FIXED_PAGEFILE_UNDER_PRESSURE, + swapTotal: 15_900 * 1024 +} + +const BEFORE_THE_STORM = { + total: 16_000 * 1024, + free: 9_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 30_000 * 1024 +} + +describe('pre-gone host memory', () => { + beforeEach(() => { + resetPreGoneCrashSamplingForTest() + setSystemMemoryInfoReaderForTest(null) + setSwapVolumeFreeSpaceReaderForTest(null) + appMetricsMock.mockClear() + appMetricsMock.mockReturnValue(BROWSER_AND_RENDERER) + }) + + it('carries a pre-gone host reading, not only the post-mortem one', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + // The renderer dies; its ~400 MB returns to the OS, so the gone-time read + // now shows a much healthier machine than the one that refused the alloc. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + appMetricsMock.mockReturnValue(BROWSER_ONLY) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemorySwapFreeMB).toBe(2_900) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(200) + expect(details.systemMemoryPreGoneFreeMB).toBe(400) + expect(details.systemMemoryPreGoneTotalMB).toBe(16_000) + // Why: host memory keeps its own key family, so a `systemMemory` prefix scan sees both reads. + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) + }) + + // Why this decides the cluster: 200 MB available commit is only a REFUSAL when + // the pagefile cannot grow, which is what the volume's free space says. + it('reports swap-volume free space so low commit can be told from refused commit', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(120) + // Which volume was measured: Windows only names the DEFAULT pagefile drive. + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + }) + + it('omits swap-volume free space on Linux, where swap cannot grow into free disk', async () => { + // Linux swap is a fixed partition, a fixed-size swapfile, or zram; reporting + // root-fs free space next to SwapFreeMB 0 would read as headroom that is not there. + setSwapVolumeFreeSpaceReaderForTest(null) + + await expect(readSwapVolumeFreeSpace('linux')).resolves.toBeUndefined() + }) + + it('labels the reading with the pressure verdict the platform can actually give', () => { + // Windows available commit is only a REFUSAL when the pagefile cannot grow, + // which nothing here proves, so no label may read as that verdict. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + const windowsCommit = getSystemMemoryDetails('win32') + expect(windowsCommit.systemMemoryPressureSignal).toBe('available-commit-unqualified') + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32') + .systemMemoryPressureSignal + ).toBe('available-commit-volume-cotimed') + // A volume number from a different moment describes a different machine. + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32', false) + .systemMemoryPressureSignal + ).toBe('available-commit-unqualified') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, free: 400 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('none') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, available: 900 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('mem-available') + + // darwin free/fileBacked/purgeable answer reclaimability, never pressure. + setSystemMemoryInfoReaderForTest(() => ({ + total: 16_000 * 1024, + free: 272 * 1024, + fileBacked: 2_694 * 1024, + purgeable: 0 + })) + expect(getSystemMemoryDetails('darwin').systemMemoryPressureSignal).toBe('none') + }) + + // Why this and not the volume number: the branch's own repro needed a pagefile + // that CANNOT grow to kill anything, and neither the pagefile maximum nor its + // drive is readable here — `swapVolumeAnchor` measures SystemRoot's volume, + // which a relocated pagefile does not live on. + it('never reads free disk as proof the pagefile could have grown', () => { + setSystemMemoryInfoReaderForTest(() => FIXED_PAGEFILE_UNDER_PRESSURE) + const fixedPagefile = withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ) + // 180 MB of commit beside 812 GB of free disk: co-timed, and still not a + // verdict — reading it as "the pagefile had room" is the opposite conclusion. + expect(fixedPagefile.systemMemoryPressureSignal).toBe('available-commit-volume-cotimed') + + // The one decisive win32 case: commit limit at or below RAM means there is + // no pagefile behind it, so the floor cannot heal however empty the disk is. + setSystemMemoryInfoReaderForTest(() => NO_PAGEFILE_UNDER_PRESSURE) + expect( + withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ).systemMemoryPressureSignal + ).toBe('available-commit-hard-capped') + }) + + // Why the verdict and not just the field: a statfs issued on a healthy host at + // t=0 that resolves 20 s into a commit storm prints "200 MB commit, 40 GB of + // pagefile headroom" — which reads as NOT a commit refusal, the opposite + // conclusion, under the branch's most confident label. + it('will not let a statfs that outlived its tick qualify the win32 commit verdict', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // The storm arrives; the in-flight latch makes every tick skip the merge, + // so the pending statfs is as old as the tick that STARTED it. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(20_000) + const stale = buildProcessGoneCrashDetails({}, 'renderer') + expect(stale.systemMemoryPreGoneSwapFreeMB).toBe(200) + // The pre-storm volume number still ships — but carrying its own age, and + // without promoting the verdict the analyst reads. + expect(stale.systemMemoryPreGoneSwapVolumeFreeMB).toBe(40_000) + expect(stale.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(stale.systemMemoryPreGoneSwapVolumeAgeMs).toBe(20_000) + expect(stale.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + + // The next tick's statfs answers on its own tick, so it qualifies again. + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 900, volume: 'C:' })) + await samplePreGoneSystemMemory(30_000) + vi.setSystemTime(30_000) + const fresh = buildProcessGoneCrashDetails({}, 'renderer') + expect(fresh.systemMemoryPreGoneSwapVolumeFreeMB).toBe(900) + expect(fresh.systemMemoryPreGoneSwapVolumeAgeMs).toBe(0) + expect(fresh.systemMemoryPreGonePressureSignal).toBe('available-commit-volume-cotimed') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + // Round 5: sample identity alone could not see these ticks. A host read that + // returns nothing leaves the sample object in place, so `sample === issuedFor` + // still held 25 s and two ticks later and the statfs re-qualified the verdict. + it('will not let ticks with a failed host read pass a stale statfs off as co-timed', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // GlobalMemoryStatusEx starts failing: the sample is neither replaced nor erased. + setSystemMemoryInfoReaderForTest(() => null) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(25_000) + const details = buildProcessGoneCrashDetails({}, 'renderer') + // 25 s of lag: the label must not say co-timed beside that age. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(25_000) + expect(details.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + it("arms the host sampler on its own unref'd 10 s timer, not the metric sweep's", async () => { + vi.useFakeTimers() + vi.setSystemTime(0) + const readHostMemory = vi.fn(() => UNDER_COMMIT_PRESSURE) + setSystemMemoryInfoReaderForTest(readHostMemory) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') + try { + startPreGoneCrashSampling() + + // Literal millisecond values: asserting the constants against themselves + // would let a cadence regression through, and 37 s of staleness is the bug. + expect(setIntervalSpy.mock.calls.map(([, ms]) => ms)).toEqual([60_000, 10_000]) + for (const { value } of setIntervalSpy.mock.results) { + expect((value as NodeJS.Timeout).hasRef()).toBe(false) + } + expect(readHostMemory).toHaveBeenCalledTimes(1) + + readHostMemory.mockReturnValue(AFTER_THE_CORPSE_RELEASED) + await vi.advanceTimersByTimeAsync(10_000) + // One host tick, no extra metric sweep: the two samplers run independently. + expect(readHostMemory).toHaveBeenCalledTimes(2) + expect(appMetricsMock).toHaveBeenCalledTimes(1) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(2_900) + } finally { + setIntervalSpy.mockRestore() + vi.useRealTimers() + } + }) + + it('commits the host reading without waiting on the swap-volume statfs', async () => { + // Why: statfs is slowest during the paging storm this sampler targets, and + // a hung volume must not stall or silently skip host sampling. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + + void samplePreGoneSystemMemory(Date.now() - 5_000) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(200) + + // A second tick still refreshes the reading while that statfs hangs. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + void samplePreGoneSystemMemory(Date.now()) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(2_900) + }) + + it('publishes no pre-gone host keys when every memory field failed to read', async () => { + // Why not "no keys at all": the reading always carries its signal label, so a + // committed empty one would ship an age and a volume number with no memory + // numbers beside them — a disk-free figure standing in for a host reading. + setSystemMemoryInfoReaderForTest(() => ({ total: Number.NaN, free: undefined })) + await samplePreGoneSystemMemory(Date.now()) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) + + it('carries the last volume reading forward, aged, instead of dropping it', async () => { + vi.useFakeTimers() + try { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 42, volume: 'C:' })) + await samplePreGoneSystemMemory(0) + + // The next tick's statfs hangs — during the paging storm this targets, that + // is the normal case — so the tick has no volume reading of its own, and + // the sample that replaces the last one would otherwise drop the field. + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + void samplePreGoneSystemMemory(10_000) + vi.setSystemTime(10_000) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(42) + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + // Carried, not re-read: it ships at its real age, never as a fresh number. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(10_000) + } finally { + vi.useRealTimers() + } + }) + + it('keeps a failed host read from erasing the process-metric sample', async () => { + samplePreGoneProcessMetrics(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(() => { + throw new Error('getSystemMemoryInfo unavailable') + }) + await samplePreGoneSystemMemory(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(null) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.processMetricsPreGoneRendererWorkingSetMB).toBe(400) + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) +}) diff --git a/src/main/crash-reporting/pre-gone-host-memory.ts b/src/main/crash-reporting/pre-gone-host-memory.ts new file mode 100644 index 00000000000..0db56796750 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.ts @@ -0,0 +1,164 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import { readSwapVolumeFreeSpace } from './swap-volume-free-space' +import { + getSystemMemoryDetails, + SYSTEM_MEMORY_KEY_PREFIX, + withSwapVolumeFreeSpace +} from './system-memory-details' + +// ─── Pre-gone host memory sampling ────────────────────────────────── +// Why sample at all: the gone-time host read lands after the corpse released +// its pages, so it reports a healthier machine than the one that refused the +// allocation. +// Why 10 s and not the 60 s process-metrics cadence: at 60 s, four of five +// field OOMs carried a ~37 s old host reading — far too stale to see a +// transient commit refusal. A refusal shorter than the interval stays +// invisible; no cadence fixes that. + +export const PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS = 10_000 + +type CrashReportDetails = Record + +type PreGoneSystemMemorySample = { + details: CrashReportDetails + sampledAtMs: number + /** Tick that ISSUED the statfs now merged in — never the tick it resolved on. */ + swapVolumeSampledAtMs?: number +} + +let preGoneSample: PreGoneSystemMemorySample | null = null +let preGoneTimer: ReturnType | null = null +let swapVolumeReadInFlight = false +let samplingGeneration = 0 +let sampleTick = 0 + +const PRESSURE_SIGNAL_KEY = `${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal` + +/** + * Carries the last volume reading onto the sample that replaces its own. + * + * Why: a statfs slower than one tick would otherwise make the field vanish from + * the reports it exists for — the next tick replaces the sample wholesale, and + * the in-flight latch keeps intervening ticks from merging anything. It ships + * with its own (now larger) age and, not being co-timed, never names the label. + */ +function withCarriedSwapVolume(sample: PreGoneSystemMemorySample): PreGoneSystemMemorySample { + const previous = preGoneSample + if (!previous || previous.swapVolumeSampledAtMs === undefined) { + return sample + } + const freeMB = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`] + const volume = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`] + if (typeof freeMB !== 'number' || typeof volume !== 'string') { + return sample + } + return { + ...sample, + details: withSwapVolumeFreeSpace(sample.details, { freeMB, volume }, process.platform, false), + swapVolumeSampledAtMs: previous.swapVolumeSampledAtMs + } +} + +function commitHostMemorySample(nowMs: number): boolean { + try { + const details = getSystemMemoryDetails() + // Why not `length === 0`: the signal label is appended unconditionally, so a + // reading that resolved no memory field at all still arrives with one key. + if (!Object.keys(details).some((key) => key !== PRESSURE_SIGNAL_KEY)) { + return false + } + preGoneSample = withCarriedSwapVolume({ details, sampledAtMs: nowMs }) + return true + } catch { + // Why: a failed read must not erase the previous good sample. + return false + } +} + +async function mergeSwapVolumeFreeSpace(issuedOnTick: number): Promise { + if (swapVolumeReadInFlight) { + return + } + swapVolumeReadInFlight = true + const generation = samplingGeneration + const issuedFor = preGoneSample + try { + const volume = await readSwapVolumeFreeSpace() + if (volume && preGoneSample && generation === samplingGeneration) { + // Why only its own tick qualifies: a statfs that outlived its tick carries a + // pre-storm volume number, and the latch makes that lag unbounded. It still + // ships beside its age, but it may not decide the verdict. + // Why the tick counter and not sample identity: a tick whose host read fails + // leaves the sample object in place, so identity alone reads as co-timed. + const coTimed = issuedOnTick === sampleTick + preGoneSample = { + ...preGoneSample, + details: withSwapVolumeFreeSpace(preGoneSample.details, volume, process.platform, coTimed), + swapVolumeSampledAtMs: issuedFor?.sampledAtMs + } + } + } catch { + // Why: the memory reading is already committed and stands on its own. + } finally { + swapVolumeReadInFlight = false + } +} + +export async function samplePreGoneSystemMemory(nowMs: number = Date.now()): Promise { + // Why commit before awaiting: the volume read is a statfs, and under the very + // paging storm this targets it is slowest — it must never delay, or (via an + // in-flight latch) skip, the cheap synchronous host reading. + const tick = ++sampleTick + if (!commitHostMemorySample(nowMs)) { + return + } + await mergeSwapVolumeFreeSpace(tick) +} + +export function startPreGoneSystemMemorySampling( + intervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS +): void { + if (preGoneTimer) { + return + } + void samplePreGoneSystemMemory() + preGoneTimer = setInterval(() => void samplePreGoneSystemMemory(), intervalMs) + preGoneTimer.unref?.() +} + +export function resetPreGoneSystemMemorySamplingForTest(): void { + if (preGoneTimer) { + clearInterval(preGoneTimer) + } + preGoneTimer = null + preGoneSample = null + swapVolumeReadInFlight = false + // Why bump: an already-awaited volume read must not repopulate a reset sample. + samplingGeneration += 1 +} + +/** Keyed as `systemMemoryPreGone*` so a scan over the `systemMemory` family sees both reads. */ +export function preGoneSystemMemoryDetails(nowMs: number): CrashReportDetails { + if (!preGoneSample) { + return {} + } + const details: CrashReportDetails = { + [`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSampleAgeMs`]: Math.max( + 0, + nowMs - preGoneSample.sampledAtMs + ) + } + // Why its own age: the volume read resolves out of band, so it can be older + // than the memory reading printed beside it, and that gap must be readable. + if (preGoneSample.swapVolumeSampledAtMs !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSwapVolumeAgeMs`] = Math.max( + 0, + nowMs - preGoneSample.swapVolumeSampledAtMs + ) + } + for (const [key, value] of Object.entries(preGoneSample.details)) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGone${key.slice(SYSTEM_MEMORY_KEY_PREFIX.length)}`] = + value + } + return details +} diff --git a/src/main/crash-reporting/process-gone-diagnostics.test.ts b/src/main/crash-reporting/process-gone-diagnostics.test.ts index e31645865d8..6a6ec410733 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.test.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.test.ts @@ -2,11 +2,11 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { buildProcessGoneCrashDetails, collectProcessGoneMetricDetails, - resetPreGoneProcessMetricsSamplingForTest, + resetPreGoneCrashSamplingForTest, samplePreGoneProcessMetrics, - startPreGoneProcessMetricsSampling + startPreGoneCrashSampling } from './process-gone-diagnostics' -import { setSystemMemoryInfoReaderForTest } from './gone-time-system-memory' +import { setSystemMemoryInfoReaderForTest } from './system-memory-details' type MetricFixture = { pid?: number @@ -27,7 +27,7 @@ vi.mock('electron', () => ({ describe('process gone diagnostics', () => { beforeEach(() => { - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() setSystemMemoryInfoReaderForTest(null) }) @@ -141,8 +141,8 @@ describe('process gone diagnostics', () => { appMetricsMock.mockReturnValue([ { pid: 30, type: 'Tab', memory: { workingSetSize: 1024 * 100 } } ]) - startPreGoneProcessMetricsSampling(1_000) - startPreGoneProcessMetricsSampling(1_000) + startPreGoneCrashSampling(1_000) + startPreGoneCrashSampling(1_000) // A crash inside the first interval already has a sample to draw from. expect(buildProcessGoneCrashDetails({}, 'renderer')).toMatchObject({ @@ -582,12 +582,12 @@ describe('process gone diagnostics', () => { it("arms an unref'd interval so sampling never holds the event loop open", () => { const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') try { - startPreGoneProcessMetricsSampling(60_000) + startPreGoneCrashSampling(60_000) const timer = setIntervalSpy.mock.results[0]?.value as NodeJS.Timeout expect(timer.hasRef()).toBe(false) } finally { setIntervalSpy.mockRestore() - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() } }) @@ -641,7 +641,7 @@ describe('process gone diagnostics', () => { expect(details.systemMemoryTotalMB).toBe(16_384) }) - it('samples system memory at gone time but never into the pre-gone snapshot', () => { + it('samples system memory at gone time but never into the processMetrics family', () => { appMetricsMock.mockReturnValue([{ pid: 1, type: 'Browser', memory: { workingSetSize: 0 } }]) samplePreGoneProcessMetrics() setSystemMemoryInfoReaderForTest(() => ({ @@ -658,7 +658,9 @@ describe('process gone diagnostics', () => { systemMemorySwapTotalMB: 8_192, systemMemorySwapFreeMB: 40 }) - expect(details.processMetricsPreGoneSystemMemoryTotalMB).toBeUndefined() + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) }) it('leaves records unflagged when the crashed bucket is still populated', () => { diff --git a/src/main/crash-reporting/process-gone-diagnostics.ts b/src/main/crash-reporting/process-gone-diagnostics.ts index d0bb380a2b6..bf0d735a2c7 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.ts @@ -3,7 +3,13 @@ import { sanitizeCrashReportDetails, type CrashReportDetailValue } from '../../shared/crash-reporting' -import { getSystemMemoryAtGoneDetails, memoryKBFieldMB } from './gone-time-system-memory' +import { getSystemMemoryDetails, memoryKBFieldMB } from './system-memory-details' +import { + PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS, + preGoneSystemMemoryDetails, + resetPreGoneSystemMemorySamplingForTest, + startPreGoneSystemMemorySampling +} from './pre-gone-host-memory' type ProcessMetricLike = { pid?: unknown @@ -204,8 +210,9 @@ export function samplePreGoneProcessMetrics(nowMs: number = Date.now()): void { } } -export function startPreGoneProcessMetricsSampling( - intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS +export function startPreGoneCrashSampling( + intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS, + systemMemoryIntervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS ): void { if (preGoneSampleTimer) { return @@ -213,14 +220,16 @@ export function startPreGoneProcessMetricsSampling( samplePreGoneProcessMetrics() preGoneSampleTimer = setInterval(() => samplePreGoneProcessMetrics(), intervalMs) preGoneSampleTimer.unref?.() + startPreGoneSystemMemorySampling(systemMemoryIntervalMs) } -export function resetPreGoneProcessMetricsSamplingForTest(): void { +export function resetPreGoneCrashSamplingForTest(): void { if (preGoneSampleTimer) { clearInterval(preGoneSampleTimer) } preGoneSampleTimer = null preGoneSample = null + resetPreGoneSystemMemorySamplingForTest() } const PROCESS_METRICS_KEY_PREFIX = 'processMetrics' @@ -271,7 +280,7 @@ export function buildProcessGoneCrashDetails( const crashDetails: CrashReportDetails = { ...sanitizedDetails, ...liveMetricDetails, - ...getSystemMemoryAtGoneDetails() + ...getSystemMemoryDetails() } // Why: with the crasher gone, Largest names a survivor — flag that so the // live buckets are read as "everyone else", not as the crashed process. @@ -290,8 +299,10 @@ export function buildProcessGoneCrashDetails( if (liveMetricDetails[crashedBucketCountKey] === 0 || sampledSameBucketProcessVanished) { crashDetails.processMetricsCrashedProcessAbsent = true } + const nowMs = Date.now() if (preGoneSample) { - Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, Date.now())) + Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, nowMs)) } + Object.assign(crashDetails, preGoneSystemMemoryDetails(nowMs)) return crashDetails } diff --git a/src/main/crash-reporting/swap-volume-free-space.ts b/src/main/crash-reporting/swap-volume-free-space.ts new file mode 100644 index 00000000000..3ad40b7629b --- /dev/null +++ b/src/main/crash-reporting/swap-volume-free-space.ts @@ -0,0 +1,67 @@ +import { statfs } from 'node:fs/promises' +import path from 'node:path' + +// Why: a system-managed Windows pagefile — and a macOS swapfile — only grows +// into free space on its own volume, so low available commit is a REFUSED +// allocation only when that volume is full too. Linux is excluded on purpose: +// its swap is a fixed partition, a fixed-size swapfile, or zram, none of which +// grow into root-fs free space, so the number would read as headroom that +// cannot exist. The measured volume ships alongside because Windows only names +// the DEFAULT pagefile drive; a relocated pagefile lives elsewhere. + +const BYTES_PER_MB = 1024 * 1024 + +export type SwapVolumeFreeSpace = { + freeMB: number + /** Which volume was measured, separator-trimmed so redaction sees no path. */ + volume: string +} + +type SwapVolumeFreeSpaceReader = ( + platform: NodeJS.Platform +) => Promise + +function swapVolumeAnchor(platform: NodeJS.Platform): string | undefined { + if (platform === 'win32') { + const anchor = process.env.SystemRoot || process.env.SystemDrive + return anchor ? path.parse(anchor).root || anchor : undefined + } + return platform === 'darwin' ? path.sep : undefined +} + +function volumeLabel(root: string): string { + const trimmed = root.replace(/[\\/]+$/, '') + return trimmed.length > 0 ? trimmed : root +} + +async function statfsSwapVolumeFreeSpace( + platform: NodeJS.Platform +): Promise { + const root = swapVolumeAnchor(platform) + if (!root) { + return undefined + } + try { + const stats = await statfs(root) + const bytes = Number(stats.bsize) * Number(stats.bavail) + return Number.isFinite(bytes) + ? { freeMB: Math.round(Math.max(0, bytes) / BYTES_PER_MB), volume: volumeLabel(root) } + : undefined + } catch { + return undefined + } +} + +let swapVolumeFreeSpaceReader: SwapVolumeFreeSpaceReader = statfsSwapVolumeFreeSpace + +export function setSwapVolumeFreeSpaceReaderForTest( + reader: SwapVolumeFreeSpaceReader | null +): void { + swapVolumeFreeSpaceReader = reader ?? statfsSwapVolumeFreeSpace +} + +export function readSwapVolumeFreeSpace( + platform: NodeJS.Platform = process.platform +): Promise { + return swapVolumeFreeSpaceReader(platform) +} diff --git a/src/main/crash-reporting/system-memory-details.ts b/src/main/crash-reporting/system-memory-details.ts new file mode 100644 index 00000000000..1f2cf556faa --- /dev/null +++ b/src/main/crash-reporting/system-memory-details.ts @@ -0,0 +1,161 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import type { SwapVolumeFreeSpace } from './swap-volume-free-space' + +// ─── Host system memory for crash reports ─────────────────────────── +// Why: the system outlives the crashed process, so this IS sampleable at +// process-gone — it separates "renderer grew huge" from "machine out of +// memory/commit", which the per-process buckets alone cannot. The gone-time +// caller reads AFTER the corpse returned its pages, so free/swapFree read +// healthier than at kill time; the pre-gone sampler carries a live reading past +// that. +// Every reading is labelled `systemMemoryPressureSignal` so no report can be +// read as a pressure verdict the platform never gave: +// win32 — swapFree is MEMORYSTATUSEX.ullAvailPageFile, i.e. available +// COMMIT, which pagefile growth can heal (a 127 MB commit floor healed to +// 2029 MB mid-hold on the win-lowspec repro, killing nothing). Free space on +// the swap volume does NOT establish that it could: a fixed-size or disabled +// pagefile grows into no amount of empty disk, its maximum is unreadable +// here (needs a registry read), and the measured volume is only the DEFAULT +// pagefile drive. So a co-timed volume reading is context beside the commit +// number — `available-commit-volume-cotimed` — never a verdict. The one +// decisive win32 case is a commit limit at or below RAM: no pagefile exists +// to grow, so the floor cannot heal (`available-commit-hard-capped`). +// linux — MemAvailable is the real signal; MemFree is not (it excludes page +// cache and other reclaimable memory). +// darwin — none. `free` stays low on healthy machines and +// fileBacked/purgeable are only a reclaimability proxy. The real signal +// needs `memory_pressure -Q`; Orca's reader for it +// (src/main/memory/host-memory.ts) is on-demand, and spawning a subprocess +// on a 10 s app-lifetime timer costs more than the gap it closes. + +type CrashReportDetails = Record + +export const SYSTEM_MEMORY_KEY_PREFIX = 'systemMemory' + +export function memoryKBFieldMB(value: unknown): number | undefined { + const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined + return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) +} + +type SystemMemoryInfoLike = { + total?: unknown + free?: unknown + available?: unknown + swapTotal?: unknown + swapFree?: unknown + fileBacked?: unknown + purgeable?: unknown +} + +type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null + +/** How far this reading may be read as a "was the host under pressure" verdict. */ +export type SystemMemoryPressureSignal = + | 'available-commit-hard-capped' + | 'available-commit-volume-cotimed' + | 'available-commit-unqualified' + | 'mem-available' + | 'none' + +function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { + const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) + .getSystemMemoryInfo + if (typeof read !== 'function') { + return null + } + try { + return read.call(process) + } catch { + return null + } +} + +let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo + +export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { + systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo +} + +function numericDetail(details: CrashReportDetails, suffix: string): number | undefined { + const value = details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] + return typeof value === 'number' ? value : undefined +} + +/** Windows commit limit = RAM + pagefile, so a limit at or below RAM has no pagefile behind it. */ +function pagefileBacksCommit(details: CrashReportDetails): boolean | undefined { + const total = numericDetail(details, 'TotalMB') + const swapTotal = numericDetail(details, 'SwapTotalMB') + return total === undefined || swapTotal === undefined ? undefined : swapTotal > total +} + +function pressureSignal( + platform: NodeJS.Platform, + details: CrashReportDetails, + volumeCoTimed = true +): SystemMemoryPressureSignal { + if (platform === 'win32' && `${SYSTEM_MEMORY_KEY_PREFIX}SwapFreeMB` in details) { + if (pagefileBacksCommit(details) === false) { + return 'available-commit-hard-capped' + } + return volumeCoTimed && `${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB` in details + ? 'available-commit-volume-cotimed' + : 'available-commit-unqualified' + } + if (platform === 'linux' && `${SYSTEM_MEMORY_KEY_PREFIX}AvailableMB` in details) { + return 'mem-available' + } + return 'none' +} + +export function getSystemMemoryDetails( + platform: NodeJS.Platform = process.platform +): CrashReportDetails { + const info = systemMemoryInfoReader() + if (!info) { + return {} + } + const details: CrashReportDetails = {} + const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ + ['total', 'TotalMB'], + ['free', 'FreeMB'], + ['available', 'AvailableMB'], + ['swapTotal', 'SwapTotalMB'], + ['swapFree', 'SwapFreeMB'], + ['fileBacked', 'FileBackedMB'], + ['purgeable', 'PurgeableMB'] + ] + for (const [field, suffix] of fields) { + const mb = memoryKBFieldMB(info[field]) + if (mb !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] = mb + } + } + details[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, details) + return details +} + +/** + * Merges the statfs-derived volume datum, which needs an await and so is only + * reachable from the periodic sampler, and relabels the reading it sits beside. + * + * `coTimed` false means the statfs outlived the tick that issued it, so this + * volume number and the commit number beside it describe different moments — + * during a pagefile-growth storm that is exactly when they diverge, and a + * pre-storm 40 GB printed next to 200 MB of commit reads as "the pagefile had + * room", the opposite conclusion. The datum still ships (with its own age), but + * only a co-timed one is named in the label. + */ +export function withSwapVolumeFreeSpace( + details: CrashReportDetails, + volume: SwapVolumeFreeSpace, + platform: NodeJS.Platform = process.platform, + coTimed = true +): CrashReportDetails { + const merged: CrashReportDetails = { + ...details, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`]: volume.freeMB, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`]: volume.volume + } + merged[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, merged, coTimed) + return merged +} diff --git a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts index 97e0a8f9d65..8d126f557b5 100644 --- a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts +++ b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts @@ -49,6 +49,7 @@ function recordRendererBreadcrumbTrace( const DUPLICATE_TAB_OWNER_BREADCRUMB = 'terminal_tab_id_owned_by_multiple_worktrees' const PARK_VERDICT_CHURN_BREADCRUMB = 'terminal_park_verdict_churn' const REACT_COMMIT_CASCADE_BREADCRUMB = 'react_commit_cascade' +const REPLAY_GUARD_WEDGED_BREADCRUMB = 'terminal_replay_guard_wedged_release' const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ 'renderer_error', 'renderer_unhandled_rejection', @@ -56,6 +57,7 @@ const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ DUPLICATE_TAB_OWNER_BREADCRUMB, PARK_VERDICT_CHURN_BREADCRUMB, REACT_COMMIT_CASCADE_BREADCRUMB, + REPLAY_GUARD_WEDGED_BREADCRUMB, TERMINAL_WEBGL_DIAGNOSTIC_BREADCRUMB ]) const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 @@ -69,6 +71,11 @@ const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 // 30-entry ring to two such bursts. `suppressedSinceLast` keeps the pane count // — the only signal these carry — in one slot. const NAME_ONLY_COALESCED_BREADCRUMB_NAMES = new Set(['terminal_safe_fit_retry_exhausted']) +// Why: the 30-slot ring is the scarce sink; the durable span stream is not. For +// bounded-rate pane telemetry whose multiplicity is the whole signal, spans are the +// only place a burst survives the restart that clears the ring, so coalesce the ring +// but keep every event's span. +const PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES = new Set([REPLAY_GUARD_WEDGED_BREADCRUMB]) function rendererBreadcrumbCoalesceKey( name: string, @@ -77,6 +84,13 @@ function rendererBreadcrumbCoalesceKey( if (NAME_ONLY_COALESCED_BREADCRUMB_NAMES.has(name)) { return name } + // Why presence and not value: `ptyId`/`tabIdHash` are absent on the restore call + // site (layout-serialization restoreScrollbackBuffers) and present on reattach, so + // their presence is the call-site identity a mixed burst would otherwise lose. Four + // slots per storm at most, regardless of pane count. + if (name === REPLAY_GUARD_WEDGED_BREADCRUMB) { + return `${name}:${data?.ptyId ? 'pty' : ''}:${data?.tabIdHash ? 'tab' : ''}` + } // Why trigger and not name alone: `burst` means damping engaged a commit // short of React #185, `window` means slow benign churn. Collapsing them // would drop the near-crash signal into a slow-churn slot. Still bounded — @@ -191,9 +205,13 @@ export function recordRendererBreadcrumbFromRenderer( minIntervalMs: RENDERER_BREADCRUMB_COALESCE_MS, ...(origin ? { origin } : {}) }) - // Why: tracing every suppressed duplicate would preserve the same - // serialization and disk churn that breadcrumb coalescing removes. - if (coalesceResult) { + if (PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES.has(args.name)) { + // Why the raw data: every event already gets its own span, so folding the ring's + // running count in here would double-count in any span-stream total. + recordRendererBreadcrumbTrace(args.name, data) + } else if (coalesceResult) { + // Why gated: tracing every suppressed duplicate would preserve the same + // serialization and disk churn that breadcrumb coalescing removes. recordRendererBreadcrumbTrace( args.name, coalesceResult.suppressedSinceLast > 0 diff --git a/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts new file mode 100644 index 00000000000..e3823f0b313 --- /dev/null +++ b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts @@ -0,0 +1,128 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { + clearCrashBreadcrumbsForTest, + getCrashBreadcrumbSnapshot, + recordCrashBreadcrumb +} from '../crash-reporting/crash-breadcrumb-store' +import { recordRendererBreadcrumbFromRenderer } from './crash-reporting-renderer-breadcrumbs' + +type SpanOptions = { attributes: Record } +const startSpanMock = vi.fn((_name: string, _options: SpanOptions) => ({ end: () => {} })) +vi.mock('../observability/tracer', () => ({ + startSpan: (name: string, options: SpanOptions) => startSpanMock(name, options) +})) + +const WEDGE_BREADCRUMB = 'terminal_replay_guard_wedged_release' + +/** Reattach-path shape: identity-bearing (`tabIdHash`, optionally `ptyId`). */ +function emitReattachWedge(pane: number, withPtyId = false): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { + paneId: pane, + leafIdHash: `leaf${String(pane).padStart(5, '0')}`, + tabIdHash: `tab${String(pane).padStart(6, '0')}`, + worktreeIdHash: 'caa15fa9', + ...(withPtyId ? { ptyId: `…@@pty-${pane}` } : {}) + } + }) +} + +/** Restore-path shape (restoreScrollbackBuffers): no tabIdHash, no ptyId. */ +function emitRestoreWedge(pane: number): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { paneId: pane, leafIdHash: `leaf${String(pane).padStart(5, '0')}` } + }) +} + +function wedgeCrumbs(): ReturnType { + return getCrashBreadcrumbSnapshot().filter((entry) => entry.name === WEDGE_BREADCRUMB) +} + +function wedgeSpanCount(): number { + return startSpanMock.mock.calls.filter( + (call) => call[1].attributes['breadcrumb.name'] === WEDGE_BREADCRUMB + ).length +} + +beforeEach(() => { + startSpanMock.mockClear() +}) + +afterEach(() => { + clearCrashBreadcrumbsForTest() +}) + +// One mount/reveal/wake transition expires every in-flight replay write at once, so +// the burst reaches the 30-slot ring as N distinct entries. Field span streams measure +// bursts of 26 in 0.96s and 62 over 85s. No captured report in the 09-02 corpus shows +// a ring that actually drained — all nine bursts predate their report's ring window — +// so this bounds a demonstrated hazard, not an observed loss, and must not cost the +// durable span evidence that did carry those bursts. +describe('replay-guard wedge burst against the fixed-size breadcrumb ring', () => { + it('costs one ring slot per call site and preserves the pre-crash trail', () => { + for (let index = 0; index < 10; index += 1) { + recordCrashBreadcrumb(`pre_crash_evidence_${index}`, { index }) + } + + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + const snapshot = getCrashBreadcrumbSnapshot() + expect(snapshot.filter((entry) => entry.name.startsWith('pre_crash_evidence_'))).toHaveLength( + 10 + ) + expect(wedgeCrumbs()).toHaveLength(1) + }) + + it('carries the burst multiplicity into the ring as suppressedSinceLast', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + // 26 emissions: one owns the slot, 25 fold into it. + expect(wedgeCrumbs()[0]?.data?.suppressedSinceLast).toBe(25) + }) + + // The 121-event field corpus lives entirely in the renderer.breadcrumb span stream, + // and the ring is cleared by the restart that usually precedes the crash report, so + // ring coalescing must not suppress the per-event spans. + it('still emits one durable span per wedge event', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + expect(wedgeSpanCount()).toBe(26) + // Why no count on the span: one span per event already carries the multiplicity. + expect( + startSpanMock.mock.calls.some((call) => + JSON.stringify(call[1]).includes('suppressedSinceLast') + ) + ).toBe(false) + }) + + // Bundle 8907a508 mixes restore-path (identity-less) and reattach-path crumbs in one + // window; name-only keying would report only the last one's shape. + it('keeps restore-path and reattach-path call sites in separate slots', () => { + emitRestoreWedge(1) + emitRestoreWedge(2) + emitReattachWedge(3) + emitReattachWedge(4, true) + + const crumbs = wedgeCrumbs() + expect(crumbs).toHaveLength(3) + expect(crumbs.map((crumb) => Boolean(crumb.data?.tabIdHash))).toEqual([false, true, true]) + expect(crumbs.map((crumb) => Boolean(crumb.data?.ptyId))).toEqual([false, false, true]) + }) + + it('bounds a many-pane burst to one slot within a call site', () => { + for (let pane = 0; pane < 40; pane += 1) { + emitReattachWedge(pane, pane % 2 === 0) + } + + expect(wedgeCrumbs()).toHaveLength(2) + }) +}) diff --git a/src/main/ipc/filesystem-allowed-roots.test.ts b/src/main/ipc/filesystem-allowed-roots.test.ts new file mode 100644 index 00000000000..f94c99c5fdb --- /dev/null +++ b/src/main/ipc/filesystem-allowed-roots.test.ts @@ -0,0 +1,372 @@ +import { mkdir, mkdtemp, realpath, rm, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import type * as RepoWorktrees from '../repo-worktrees' +import { listRepoWorktreeGraph } from '../repo-worktrees' +import type * as ProjectGroupsModule from '../../shared/project-groups' +import { buildProjectGroupChildIndex, getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { isPathInsideOrEqual } from '../../shared/cross-platform-path' +import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import type { FolderWorkspace } from '../../shared/folder-workspace-types' +import type { ProjectGroup } from '../../shared/project-group-types' +import type { Project } from '../../shared/project-types' +import type { Repo } from '../../shared/repo-types' +import { getAllowedRoots } from './filesystem-allowed-roots' +import { authorizeExternalPath, resolveAuthorizedPath } from './filesystem-auth' +import { invalidateAuthorizedRootsCache } from './registered-worktree-roots-cache' +import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' + +vi.mock('../repo-worktrees', async () => { + const actual = await vi.importActual('../repo-worktrees') + return { ...actual, listRepoWorktreeGraph: vi.fn(async () => []) } +}) + +vi.mock('../../shared/project-groups', async () => { + const actual = await vi.importActual('../../shared/project-groups') + return { + ...actual, + buildProjectGroupChildIndex: vi.fn(actual.buildProjectGroupChildIndex), + getProjectGroupSubtreeIds: vi.fn(actual.getProjectGroupSubtreeIds) + } +}) + +type StoreFixture = { + repos: Repo[] + projects: Project[] + projectGroups: ProjectGroup[] + folderWorkspaces: FolderWorkspace[] + workspaceDir?: string +} + +type StoreCallCounts = { + getRepos: number + getProjects: number + getProjectGroups: number + getFolderWorkspaces: number +} + +function makeCountingStore(fixture: StoreFixture): { store: Store; counts: StoreCallCounts } { + const counts: StoreCallCounts = { + getRepos: 0, + getProjects: 0, + getProjectGroups: 0, + getFolderWorkspaces: 0 + } + const store = { + getRepos: () => { + counts.getRepos += 1 + // Match the real store, which rehydrates fresh repo objects on every read. + return fixture.repos.map((repo) => ({ ...repo })) + }, + getProjects: () => { + counts.getProjects += 1 + return fixture.projects.map((project) => ({ ...project })) + }, + getProjectGroups: () => { + counts.getProjectGroups += 1 + return fixture.projectGroups.map((group) => ({ ...group })) + }, + getFolderWorkspaces: () => { + counts.getFolderWorkspaces += 1 + return fixture.folderWorkspaces.map((workspace) => ({ ...workspace })) + }, + getSettings: () => ({ nestWorkspaces: false, workspaceDir: fixture.workspaceDir ?? '' }) + } as unknown as Store + return { store, counts } +} + +/** + * The pre-change `getAllowedRoots` algorithm, kept verbatim so the equivalence test compares the + * new root list against the old one rather than against a hand-written expectation. + */ +function referenceAllowedRoots(store: Store): string[] { + const scopeStore = store as unknown as { + getRepos: () => Repo[] + getProjectGroups?: () => ProjectGroup[] + getFolderWorkspaces?: () => FolderWorkspace[] + getSettings: () => { workspaceDir?: string; nestWorkspaces?: boolean } + } + const localRepos = scopeStore.getRepos().filter((repo) => !repo.connectionId) + const settings = scopeStore.getSettings() + + const scopeRepos = scopeStore.getRepos() + const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const isRemoteOnly = ( + folderPath: string, + projectGroupId: string, + connectionId: string | null | undefined + ): boolean => { + if (connectionId) { + return true + } + const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const candidates = scopeRepos.filter( + (repo) => + (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || + isPathInsideOrEqual(folderPath, repo.path) + ) + return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) + } + const folderScopeRoots: string[] = [] + for (const group of projectGroups) { + if (group.parentPath && !isRemoteOnly(group.parentPath, group.id, group.connectionId)) { + folderScopeRoots.push(resolve(group.parentPath)) + } + } + for (const workspace of scopeStore.getFolderWorkspaces?.() ?? []) { + const connectionId = + workspace.connectionId ?? + projectGroups.find((group) => group.id === workspace.projectGroupId)?.connectionId ?? + null + if (!isRemoteOnly(workspace.folderPath, workspace.projectGroupId, connectionId)) { + folderScopeRoots.push(resolve(workspace.folderPath)) + } + } + + const roots = [...localRepos.map((repo) => resolve(repo.path)), ...folderScopeRoots] + if (settings.workspaceDir) { + if (localRepos.length === 0) { + roots.push(resolve(settings.workspaceDir)) + } else { + for (const repo of localRepos) { + roots.push( + resolve( + computeWorkspaceRoot( + repo.path, + getWorktreePathSettings(repo, settings as never, getWorktreeMirrorDistro(store, repo)) + ) + ) + ) + } + } + } + return roots +} + +function makeRepo(overrides: Partial & Pick): Repo { + return { + displayName: overrides.id, + badgeColor: '#000000', + addedAt: 1, + kind: 'git', + ...overrides + } +} + +function makeGroup(overrides: Partial & Pick): ProjectGroup { + return { + name: overrides.id, + parentPath: null, + parentGroupId: null, + createdFrom: 'folder-scan', + tabOrder: 0, + isCollapsed: false, + color: null, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +function makeWorkspace( + overrides: Partial & Pick +): FolderWorkspace { + return { + projectGroupId: 'group-root', + name: overrides.id, + comment: '', + linkedTask: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 1, + lastActivityAt: 1, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +/** Repos, nested groups, folder workspaces (one not a git worktree), and an SSH repo. */ +function makeMixedFixture(): StoreFixture { + const repos = [ + makeRepo({ id: 'repo-local', path: '/repos/app', projectGroupId: 'group-root' }), + makeRepo({ id: 'repo-nested', path: '/repos/nested', projectGroupId: 'group-child' }), + makeRepo({ id: 'repo-folder', path: '/folders/plain', kind: 'folder' }), + makeRepo({ + id: 'repo-ssh', + path: '/remote/app', + connectionId: 'ssh-1', + projectGroupId: 'group-remote' + }) + ] + const projectGroups = [ + makeGroup({ id: 'group-root', parentPath: '/folders/root' }), + makeGroup({ id: 'group-child', parentGroupId: 'group-root', parentPath: '/folders/child' }), + makeGroup({ id: 'group-grandchild', parentGroupId: 'group-child' }), + makeGroup({ id: 'group-remote', parentPath: '/remote/scope' }), + makeGroup({ id: 'group-connection', parentPath: '/remote/via-group', connectionId: 'ssh-1' }) + ] + const folderWorkspaces = [ + makeWorkspace({ id: 'ws-git', folderPath: '/folders/root/feature' }), + // Not a git worktree: a plain folder workspace under a folder-kind repo. + makeWorkspace({ + id: 'ws-plain', + folderPath: '/folders/plain/scratch', + projectGroupId: 'group-child' + }), + makeWorkspace({ id: 'ws-remote', folderPath: '/remote/ws', projectGroupId: 'group-remote' }), + makeWorkspace({ + id: 'ws-connection', + folderPath: '/remote/direct', + projectGroupId: 'group-connection' + }), + makeWorkspace({ + id: 'ws-unlinked', + folderPath: '/folders/unlinked', + projectGroupId: 'group-orphan' + }) + ] + const projects: Project[] = [ + { + id: 'project-1', + displayName: 'App', + badgeColor: '#000000', + sourceRepoIds: ['repo-local', 'repo-nested'], + createdAt: 1, + updatedAt: 1 + }, + { + id: 'project-2', + displayName: 'Folder', + badgeColor: '#000000', + sourceRepoIds: ['repo-folder'], + createdAt: 1, + updatedAt: 1 + } + ] + return { repos, projects, projectGroups, folderWorkspaces, workspaceDir: '/workspaces' } +} + +beforeEach(() => { + invalidateAuthorizedRootsCache() + vi.mocked(buildProjectGroupChildIndex).mockClear() + vi.mocked(getProjectGroupSubtreeIds).mockClear() +}) + +describe('getAllowedRoots', () => { + it('produces the same roots as the pre-change implementation', () => { + const { store } = makeCountingStore(makeMixedFixture()) + + expect(getAllowedRoots(store)).toEqual(referenceAllowedRoots(store)) + }) + + it('reads the store once and indexes project groups once per build', () => { + const fixture = makeMixedFixture() + const { store, counts } = makeCountingStore(fixture) + + getAllowedRoots(store) + + expect.soft(counts.getRepos).toBe(1) + expect.soft(counts.getProjectGroups).toBe(1) + expect.soft(counts.getFolderWorkspaces).toBe(1) + // Batched runtime resolution scans the project list once, not once per local repo. + expect.soft(counts.getProjects).toBe(1) + // The per-scope subtree walk no longer rebuilds the parent->children index. + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(1) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) +}) + +describe('resolveAuthorizedPath allowed-root reuse', () => { + let repoRoot: string + let outsideRoot: string + let store: Store + let counts: StoreCallCounts + + beforeEach(async () => { + repoRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-allowed-roots-')) + outsideRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-outside-')) + const fixture = makeMixedFixture() + fixture.repos = [makeRepo({ id: 'repo-local', path: repoRoot }), ...fixture.repos] + fixture.projects[0]!.sourceRepoIds = ['repo-local'] + ;({ store, counts } = makeCountingStore(fixture)) + }) + + afterEach(async () => { + await rm(repoRoot, { recursive: true, force: true }) + await rm(outsideRoot, { recursive: true, force: true }) + }) + + it('builds the allowed-root list once per call across repeated reads', async () => { + const dirPath = join(repoRoot, 'src') + await mkdir(dirPath) + await writeFile(join(dirPath, 'index.ts'), 'export {}\n') + const callCount = 5 + + for (let index = 0; index < callCount; index += 1) { + await resolveAuthorizedPath(dirPath, store) + await resolveAuthorizedPath(join(dirPath, 'index.ts'), store) + } + + const buildCount = callCount * 2 + // One build per authorization, not one per raw-path check plus one per realpath check. + expect.soft(counts.getFolderWorkspaces).toBe(buildCount) + expect.soft(counts.getRepos).toBe(buildCount) + expect.soft(counts.getProjects).toBe(buildCount) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(buildCount) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) + + // Why (both symlink cases): creating a symlink on Windows needs elevation or + // Developer Mode, so these would fail EPERM in setup rather than exercise the + // escape check. Every non-symlink case still runs there. + it.skipIf(process.platform === 'win32')( + 'still refuses a symlink that escapes every allowed root', + async () => { + const secret = join(outsideRoot, 'secret.txt') + await writeFile(secret, 'secret\n') + const escape = join(repoRoot, 'escape.txt') + await symlink(secret, escape) + + await expect(resolveAuthorizedPath(escape, store)).rejects.toThrow('Access denied') + expect(vi.mocked(listRepoWorktreeGraph)).toHaveBeenCalled() + } + ) + + it('builds no allowed-root list at all for a granted external path', async () => { + const external = join(outsideRoot, 'external.md') + await writeFile(external, 'notes\n') + authorizeExternalPath(external) + counts.getRepos = 0 + counts.getProjects = 0 + counts.getFolderWorkspaces = 0 + + for (let index = 0; index < 5; index += 1) { + await expect(resolveAuthorizedPath(external, store)).resolves.toBe(external) + } + + // The grant answers on its own; hoisting the snapshot must not turn zero builds into one per read. + expect.soft(counts.getRepos).toBe(0) + expect.soft(counts.getProjects).toBe(0) + expect.soft(counts.getFolderWorkspaces).toBe(0) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).not.toHaveBeenCalled() + }) + + it.skipIf(process.platform === 'win32')( + 'still refuses a directory symlink that escapes every allowed root', + async () => { + const outsideDir = join(outsideRoot, 'nested') + await mkdir(outsideDir) + await writeFile(join(outsideDir, 'file.txt'), 'secret\n') + const escape = join(repoRoot, 'escape-dir') + await symlink(outsideDir, escape) + + await expect(resolveAuthorizedPath(join(escape, 'file.txt'), store)).rejects.toThrow( + 'Access denied' + ) + } + ) +}) diff --git a/src/main/ipc/filesystem-allowed-roots.ts b/src/main/ipc/filesystem-allowed-roots.ts index 3cb7fe4fa55..cef249430c6 100644 --- a/src/main/ipc/filesystem-allowed-roots.ts +++ b/src/main/ipc/filesystem-allowed-roots.ts @@ -1,9 +1,16 @@ import { resolve } from 'node:path' import type { Store } from '../persistence' import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' -import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import { + getWorktreeMirrorDistroForRuntime, + resolveLocalProjectRuntimesForRepos +} from '../project-runtime-git-options' import { isPathInsideOrEqual } from '../../shared/cross-platform-path' -import { getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { + buildProjectGroupChildIndex, + collectProjectGroupSubtreeIds, + type ProjectGroupChildIndex +} from '../../shared/project-groups' import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { ProjectGroup } from '../../shared/project-group-types' import type { Repo } from '../../shared/repo-types' @@ -11,18 +18,22 @@ import type { Repo } from '../../shared/repo-types' type FolderScopeStore = Pick & Partial> +// Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. +function filterLocalRepos(repos: readonly Repo[]): Repo[] { + return repos.filter((repo) => !repo.connectionId) +} + export function getLocalRepos(store: Store) { - // Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. - return store.getRepos().filter((repo) => !repo.connectionId) + return filterLocalRepos(store.getRepos()) } function getFolderScopeCandidateRepos( folderPath: string, projectGroupId: string, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): Repo[] { - const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const groupIds = collectProjectGroupSubtreeIds(childGroupIndex, projectGroupId) return repos.filter( (repo) => (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || @@ -34,13 +45,18 @@ function isRemoteOnlyFolderScope( folderPath: string, projectGroupId: string, connectionId: string | null | undefined, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): boolean { if (connectionId) { return true } - const candidates = getFolderScopeCandidateRepos(folderPath, projectGroupId, projectGroups, repos) + const candidates = getFolderScopeCandidateRepos( + folderPath, + projectGroupId, + childGroupIndex, + repos + ) return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) } @@ -55,16 +71,22 @@ function getFolderWorkspaceConnectionId( ) } -function getLocalFolderScopeRoots(store: Store): string[] { +function getLocalFolderScopeRoots(store: Store, repos: readonly Repo[]): string[] { const scopeStore = store as FolderScopeStore - const repos = scopeStore.getRepos() // Why: many filesystem tests use narrow Store doubles; folder scopes are additive. const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const childGroupIndex = buildProjectGroupChildIndex(projectGroups) const roots: string[] = [] for (const group of projectGroups) { if ( group.parentPath && - !isRemoteOnlyFolderScope(group.parentPath, group.id, group.connectionId, projectGroups, repos) + !isRemoteOnlyFolderScope( + group.parentPath, + group.id, + group.connectionId, + childGroupIndex, + repos + ) ) { roots.push(resolve(group.parentPath)) } @@ -75,7 +97,7 @@ function getLocalFolderScopeRoots(store: Store): string[] { workspace.folderPath, workspace.projectGroupId, getFolderWorkspaceConnectionId(workspace, projectGroups), - projectGroups, + childGroupIndex, repos ) ) { @@ -86,16 +108,19 @@ function getLocalFolderScopeRoots(store: Store): string[] { } export function getAllowedRoots(store: Store): string[] { - const localRepos = getLocalRepos(store) + // Why one read: `getRepos` rehydrates every repo, and this runs twice per filesystem IPC. + const repos = store.getRepos() + const localRepos = filterLocalRepos(repos) const settings = store.getSettings() const roots = [ ...localRepos.map((repo) => resolve(repo.path)), - ...getLocalFolderScopeRoots(store) + ...getLocalFolderScopeRoots(store, repos) ] if (settings.workspaceDir) { if (localRepos.length === 0) { roots.push(resolve(settings.workspaceDir)) } else { + const projectRuntimeByRepoId = resolveLocalProjectRuntimesForRepos(store, localRepos) for (const repo of localRepos) { roots.push( resolve( @@ -104,7 +129,11 @@ export function getAllowedRoots(store: Store): string[] { // Why enriched here too: placement has to agree with the create // flow, or renderer file access is denied for a worktree Orca // just put on the WSL side. - getWorktreePathSettings(repo, settings, getWorktreeMirrorDistro(store, repo)) + getWorktreePathSettings( + repo, + settings, + getWorktreeMirrorDistroForRuntime(projectRuntimeByRepoId.get(repo.id)) + ) ) ) ) diff --git a/src/main/ipc/filesystem-auth.ts b/src/main/ipc/filesystem-auth.ts index 122617845ed..894e39945c1 100644 --- a/src/main/ipc/filesystem-auth.ts +++ b/src/main/ipc/filesystem-auth.ts @@ -43,7 +43,24 @@ export function authorizeExternalPath(targetPath: string): void { } catch {} } -export function isPathAllowed(targetPath: string, store: Store): boolean { +/** + * One allowed-root list shared by every check in a single authorization. + * + * Lazy so a path already covered by an external grant still builds nothing at all, the way it did + * before the list was hoisted out of the individual checks. + */ +type AllowedRootsSnapshot = { get: () => readonly string[] } + +function createAllowedRootsSnapshot(store: Store): AllowedRootsSnapshot { + let roots: readonly string[] | undefined + return { get: () => (roots ??= getAllowedRoots(store)) } +} + +export function isPathAllowed( + targetPath: string, + store: Store, + allowedRoots?: AllowedRootsSnapshot +): boolean { const resolvedTarget = resolve(targetPath) if (authorizedExternalPaths.has(resolvedTarget)) { return true @@ -53,7 +70,9 @@ export function isPathAllowed(targetPath: string, store: Store): boolean { return true } } - return getAllowedRoots(store).some((root) => isDescendantOrEqual(resolvedTarget, root)) + return (allowedRoots?.get() ?? getAllowedRoots(store)).some((root) => + isDescendantOrEqual(resolvedTarget, root) + ) } export type ResolveAuthorizedPathOptions = { @@ -69,7 +88,10 @@ export async function resolveAuthorizedPath( options: ResolveAuthorizedPathOptions = {} ): Promise { const resolvedTarget = resolve(targetPath) - if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store))) { + // Why: the roots depend only on store state, not on the candidate path, so one snapshot serves + // every authorization below; each candidate is still checked against it in full. + const allowedRoots = createAllowedRootsSnapshot(store) + if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store, { allowedRoots }))) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) } @@ -80,14 +102,15 @@ export async function resolveAuthorizedPath( realParent = await realpath(dirname(resolvedTarget)) } catch (error) { if (isENOENT(error)) { - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } throw error } const candidateTarget = resolve(realParent, basename(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -100,7 +123,8 @@ export async function resolveAuthorizedPath( const realTarget = resolve(await realpath(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(realTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -110,11 +134,15 @@ export async function resolveAuthorizedPath( if (!isENOENT(error)) { throw error } - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } } -async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store): Promise { +async function resolveAuthorizedMissingPath( + resolvedTarget: string, + store: Store, + allowedRoots: AllowedRootsSnapshot +): Promise { let existingAncestor = resolvedTarget const missingSegments: string[] = [] @@ -124,7 +152,8 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store const candidateTarget = resolve(realAncestor, ...missingSegments) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -148,9 +177,9 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store async function isPathAllowedIncludingRegisteredWorktrees( targetPath: string, store: Store, - options: { canonicalSourcePath?: string } = {} + options: { canonicalSourcePath?: string; allowedRoots?: AllowedRootsSnapshot } = {} ): Promise { - if (isPathAllowed(targetPath, store)) { + if (isPathAllowed(targetPath, store, options.allowedRoots)) { return true } @@ -158,7 +187,14 @@ async function isPathAllowedIncludingRegisteredWorktrees( return true } - if (await isPathAllowedByCanonicalAllowedRoot(targetPath, options.canonicalSourcePath, store)) { + if ( + await isPathAllowedByCanonicalAllowedRoot( + targetPath, + options.canonicalSourcePath, + store, + options.allowedRoots + ) + ) { return true } @@ -178,12 +214,13 @@ async function isPathAllowedIncludingRegisteredWorktrees( async function isPathAllowedByCanonicalAllowedRoot( targetPath: string, sourcePath: string | undefined, - store: Store + store: Store, + allowedRoots?: AllowedRootsSnapshot ): Promise { if (!sourcePath) { return false } - for (const root of getAllowedRoots(store)) { + for (const root of allowedRoots?.get() ?? getAllowedRoots(store)) { const resolvedRoot = resolve(root) if (!isDescendantOrEqual(sourcePath, resolvedRoot)) { continue diff --git a/src/main/orca-chromium-process-pids.ts b/src/main/orca-chromium-process-pids.ts index f22babc6921..b2613e42b79 100644 --- a/src/main/orca-chromium-process-pids.ts +++ b/src/main/orca-chromium-process-pids.ts @@ -1,4 +1,5 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' +import { recordCoalescedDurableCrashBreadcrumb } from './crash-reporting/durable-crash-breadcrumb' /** * PIDs of Orca's own Chromium processes — browser, renderers, GPU, utilities. @@ -11,6 +12,14 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' * Empty on a Node host and empty on failure: that is "no refusal proven", never * "safe to kill" — callers must keep every other guard they already have. * + * Why failure stays open rather than refusing everything: a refusal is not free. + * `terminateWindowsProcessTree` resolves without killing, and + * `killSourceControlAgentProcess` returns that straight to a caller that then + * releases the managed-home lock, so failing closed would trade one unreadable + * metrics table for every PTY, git, codex and notebook tree in main leaking at + * once. The `own_chromium_pids_unreadable` crumb is the price of that choice: + * without it a throw is byte-identical to "no Chromium on this host". + * * Host coverage: only Electron main installs a Chromium-backed AppEnvironment * (main-process-preflight). The standalone daemon installs none and `orcad` * installs a Node one whose `getAppMetrics()` is `[]`, so this set is empty in @@ -30,7 +39,26 @@ export function readOrcaChromiumProcessPids(): ReadonlySet { .map((metric) => metric.pid) .filter((pid) => Number.isInteger(pid) && pid > 0) return new Set(pids) - } catch { + } catch (error) { + recordUnreadableOwnChromiumMetrics(error) return new Set() } } + +// Why coalesced: the gate reads this set on every tree kill, so a persistently +// broken metrics table would otherwise flood the 30-slot ring it shares. +const UNREADABLE_METRICS_COALESCE_MS = 60_000 + +function recordUnreadableOwnChromiumMetrics(error: unknown): void { + try { + recordCoalescedDurableCrashBreadcrumb({ + name: 'own_chromium_pids_unreadable', + data: { cause: error instanceof Error ? error.message : String(error) }, + coalesceKey: 'own-chromium-pids-unreadable', + minIntervalMs: UNREADABLE_METRICS_COALESCE_MS + }) + } catch { + // Diagnostics must never turn an admitted kill into a thrown one: callers + // read this set outside their own try. + } +} diff --git a/src/main/own-chromium-tree-kill-guard.test.ts b/src/main/own-chromium-tree-kill-guard.test.ts index 7e98661aca6..bd3b1674e18 100644 --- a/src/main/own-chromium-tree-kill-guard.test.ts +++ b/src/main/own-chromium-tree-kill-guard.test.ts @@ -143,6 +143,40 @@ describe('refusing to tree-kill our own Chromium processes', () => { ) }) + /** + * Fail-open is the deliberate choice — see `orca-chromium-process-pids.ts` for + * why refusing everything is worse — so the crumb is the only thing that keeps + * an unreadable metrics table distinguishable from a host that has no Chromium. + */ + it('leaves proof, and still admits the kill, when the Chromium metrics cannot be read', () => { + appMetricsMock.mockImplementation(() => { + throw new Error('getAppMetrics unavailable') + }) + + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + // Coalesced: the gate reads this set on every kill, so a broken table must + // not evict the ring it shares with the refusal crumb. + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + expect( + admitSelfInitiatedTreeKill({ + pid: RENDERER_PID, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree' + }) + ).toBe(true) + + expect( + getCrashBreadcrumbSnapshot().filter( + (breadcrumb) => breadcrumb.name === 'own_chromium_pids_unreadable' + ) + ).toEqual([ + expect.objectContaining({ + name: 'own_chromium_pids_unreadable', + data: expect.objectContaining({ cause: 'getAppMetrics unavailable' }) + }) + ]) + }) + it('refuses an own-Chromium pid at the gate the account teardowns share', () => { expect( admitSelfInitiatedTreeKill({ diff --git a/src/main/project-runtime-git-options.ts b/src/main/project-runtime-git-options.ts index 808d31d5fcf..20aa0e9659a 100644 --- a/src/main/project-runtime-git-options.ts +++ b/src/main/project-runtime-git-options.ts @@ -102,7 +102,12 @@ export function getWorktreeMirrorDistro( store: ProjectRuntimeResolutionStore, repo: Repo ): string | undefined { - const projectRuntime = resolveLocalProjectRuntimeForRepo(store, repo) + return getWorktreeMirrorDistroForRuntime(resolveLocalProjectRuntimeForRepo(store, repo)) +} + +export function getWorktreeMirrorDistroForRuntime( + projectRuntime: ProjectExecutionRuntimeResolution | undefined +): string | undefined { if (!projectRuntime || projectRuntime.status !== 'resolved') { return undefined } diff --git a/src/main/providers/ssh-git-read-provider.ts b/src/main/providers/ssh-git-read-provider.ts index 5cbcbc17b5a..cbf20271003 100644 --- a/src/main/providers/ssh-git-read-provider.ts +++ b/src/main/providers/ssh-git-read-provider.ts @@ -44,7 +44,8 @@ export class SshGitReadProvider { } } - private invalidateGitReads(): void { + /** Overridden by subclasses that own additional read caches (worktree listings). */ + protected invalidateGitReads(): void { this.gitDiffReadDedupe.clear() this.statusReadLeaseOwner.invalidate() this.upstreamStatusReadOwner.invalidate() diff --git a/src/main/providers/ssh-git-worktree-list-dedupe.test.ts b/src/main/providers/ssh-git-worktree-list-dedupe.test.ts new file mode 100644 index 00000000000..4030dbe87e4 --- /dev/null +++ b/src/main/providers/ssh-git-worktree-list-dedupe.test.ts @@ -0,0 +1,156 @@ +/** + * Local repos coalesce concurrent `git worktree list` scans (`shareWorktreeScan`); the SSH path + * branched away from that and paid one relay round trip per independent caller (`worktrees:list`, + * `worktrees:listAll`, the space repo scan, provisioned-root adoption). These are call counters. + */ +import { describe, expect, it } from 'vitest' +import { SshGitProvider } from './ssh-git-provider' +import { createMockMux, type MockMultiplexer } from './ssh-git-provider-test-harness' + +const REPO_PATH = '/home/user/repo' + +const WORKTREES = [ + { path: REPO_PATH, head: 'abc123', branch: 'main', isBare: false, isMainWorktree: true } +] + +type Deferred = { resolve: (value: unknown) => void; reject: (error: unknown) => void } + +/** Holds `git.listWorktrees` open so overlap is deterministic; answers everything else at once. */ +function createPendingListMux(): { mux: MockMultiplexer; listDeferreds: Deferred[] } { + const mux = createMockMux() + const listDeferreds: Deferred[] = [] + mux.request.mockImplementation((method: string) => { + if (method !== 'git.listWorktrees') { + return Promise.resolve(undefined) + } + return new Promise((resolve, reject) => { + listDeferreds.push({ resolve, reject }) + }) + }) + return { mux, listDeferreds } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +function countListRequests(mux: MockMultiplexer): number { + return mux.request.mock.calls.filter((call) => call[0] === 'git.listWorktrees').length +} + +describe('SSH git.listWorktrees in-flight dedupe', () => { + it('collapses concurrent listings of one repo into a single relay request', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = Array.from({ length: 6 }, () => provider.listWorktrees(REPO_PATH)) + await flush() + + expect(countListRequests(mux)).toBe(1) + expect(mux.request).toHaveBeenCalledWith( + 'git.listWorktrees', + { repoPath: REPO_PATH }, + { signal: undefined } + ) + + listDeferreds[0].resolve(WORKTREES) + expect(await Promise.all(listings)).toEqual(Array.from({ length: 6 }, () => WORKTREES)) + }) + + it('does not share across repos or connections', async () => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + void provider.listWorktrees('/home/user/other') + await flush() + expect(countListRequests(mux)).toBe(2) + + const second = createPendingListMux() + void new SshGitProvider('conn-2', second.mux as never).listWorktrees(REPO_PATH) + await flush() + + expect(countListRequests(second.mux)).toBe(1) + expect(countListRequests(mux)).toBe(2) + }) + + it('keeps a signalled listing on its own request', async () => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + await flush() + const controller = new AbortController() + void provider.listWorktrees(REPO_PATH, { signal: controller.signal }) + await flush() + + expect(countListRequests(mux)).toBe(2) + expect(mux.request).toHaveBeenCalledWith( + 'git.listWorktrees', + { repoPath: REPO_PATH }, + { signal: controller.signal } + ) + }) + + it('re-requests after the shared listing settles instead of caching it', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const first = provider.listWorktrees(REPO_PATH) + await flush() + listDeferreds[0].resolve(WORKTREES) + await first + + void provider.listWorktrees(REPO_PATH) + await flush() + + expect(countListRequests(mux)).toBe(2) + }) + + it.each([ + ['addWorktree', (p: SshGitProvider) => p.addWorktree(REPO_PATH, 'feature', '/home/user/feat')], + ['removeWorktree', (p: SshGitProvider) => p.removeWorktree('/home/user/feat')] + ])('invalidates the shared listing after %s', async (_name, mutate) => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(1) + + await mutate(provider) + + // The catalog moved, so a joiner must not inherit the pre-mutation scan. + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(2) + }) + + it('shares a failed listing with its joiners and re-requests afterwards', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)] + await flush() + expect(countListRequests(mux)).toBe(1) + + const failure = new Error('relay request failed') + listDeferreds[0].reject(failure) + await expect(listings[0]).rejects.toBe(failure) + await expect(listings[1]).rejects.toBe(failure) + + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(2) + }) + + it('refuses an unauthoritative relay answer for every joiner (#14004)', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)] + await flush() + listDeferreds[0].resolve([]) + + await expect(listings[0]).rejects.toThrow() + await expect(listings[1]).rejects.toThrow() + }) +}) diff --git a/src/main/providers/ssh-git-worktree-provider.ts b/src/main/providers/ssh-git-worktree-provider.ts index 8dae1413321..17e5b273de7 100644 --- a/src/main/providers/ssh-git-worktree-provider.ts +++ b/src/main/providers/ssh-git-worktree-provider.ts @@ -2,6 +2,7 @@ import type { GitStatusResult } from '../../shared/git-status-types' import type { RemoveWorktreeResult } from '../../shared/worktree/create-types' import type { GitWorktreeInfo } from '../../shared/worktree/types' import { CapabilityProbeCache } from '../../shared/capability-probe-cache' +import { InFlightPromiseDedupe, stableInFlightKey } from '../../shared/in-flight-promise-dedupe' import { assertAuthoritativeWorktreeCatalog } from '../../shared/worktree/worktree-catalog-availability' import { isJsonRpcMethodNotFoundError } from './ssh-git-relay-errors' import { SshGitReviewHeadProvider } from './ssh-git-review-head-provider' @@ -29,16 +30,35 @@ export class SshGitWorktreeProvider extends SshGitReviewHeadProvider { private readonly worktreeIsCleanCapabilityCache = new CapabilityProbeCache< typeof WORKTREE_IS_CLEAN_CAPABILITY >(Number.POSITIVE_INFINITY) + // Scoped to this provider instance, so two SSH hosts never share an entry. + private readonly worktreeListDedupe = new InFlightPromiseDedupe() + protected override invalidateGitReads(): void { + super.invalidateGitReads() + this.worktreeListDedupe.clear() + } + + /** Un-signalled reads of one repo coalesce onto the request already in flight; nothing is cached. */ async listWorktrees( repoPath: string, options?: { signal?: AbortSignal } ): Promise { - const response = await this.mux.request( - 'git.listWorktrees', - { repoPath }, - { signal: options?.signal } + // Why: same rule as shareWorktreeScan — one caller's abort must not cancel the scan its + // joiners are still waiting on, so a signalled read keeps its own request. + if (options?.signal) { + return this.requestWorktreeList(repoPath, options.signal) + } + return this.worktreeListDedupe.run(stableInFlightKey(['listWorktrees', repoPath]), () => + this.requestWorktreeList(repoPath) ) + } + + /** The one real relay round trip a coalesced read's joiners all wait on. */ + private async requestWorktreeList( + repoPath: string, + signal?: AbortSignal + ): Promise { + const response = await this.mux.request('git.listWorktrees', { repoPath }, { signal }) // Why (#14004): relays before this fix answered a failed worktree scan with `[]`. Mixed versions are // normal, so refuse the shape here too — a Git repo always lists its own checkout. return assertAuthoritativeWorktreeCatalog(response, repoPath) diff --git a/src/main/providers/ssh-pty-inspect-observation-identity.test.ts b/src/main/providers/ssh-pty-inspect-observation-identity.test.ts new file mode 100644 index 00000000000..4e0dae1ead1 --- /dev/null +++ b/src/main/providers/ssh-pty-inspect-observation-identity.test.ts @@ -0,0 +1,68 @@ +/** + * Ratchet (#18419): `pty.inspectProcess` must NOT be in-flight coalesced the way the sibling git + * reads in `SshGitReadProvider` are. The host mints one `observationEpoch` per request and the + * pane foreground reader commits that epoch per read, so a shared reply reads as a stale replay to + * the second reader to settle — see the companion renderer proof in + * `src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts`. + * These are request counters, not timings. + */ +import { describe, expect, it, vi } from 'vitest' +import { createSshPtyProviderRpcOperations } from './ssh-pty-provider-rpc-operations' + +const RELAY_PTY_ID = 'pty-1' +const APP_PTY_ID = `ssh:conn-1@@${RELAY_PTY_ID}` +const INCARNATION_ID = 'inc-1' + +/** Answers `pty.inspectProcess` with a fresh host observation per request, held open on demand. */ +function createInspectingOperations(): { + operations: ReturnType + request: ReturnType + resolvers: ((value: unknown) => void)[] +} { + const resolvers: ((value: unknown) => void)[] = [] + const request = vi.fn(() => new Promise((resolve) => resolvers.push(resolve))) + return { + operations: createSshPtyProviderRpcOperations({ + mux: { request } as never, + toRelayPtyId: () => RELAY_PTY_ID + }), + request, + resolvers + } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +describe('SSH pty.inspectProcess observation identity', () => { + it('gives each overlapping probe of one pane+incarnation its own host observation', async () => { + const { operations, request, resolvers } = createInspectingOperations() + + const first = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + const second = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + await flush() + + expect(request).toHaveBeenCalledTimes(2) + resolvers[0]({ foregroundProcess: 'claude', observationEpoch: 1 }) + resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 2 }) + // Each read settles on the observation minted for it, never a neighbour's. + expect(await first).toMatchObject({ observationEpoch: 1 }) + expect(await second).toMatchObject({ observationEpoch: 2 }) + }) + + it('does not share a failed probe with an overlapping one', async () => { + const { operations, request, resolvers } = createInspectingOperations() + + const failing = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + const overlapping = operations.inspectProcess(APP_PTY_ID, { + expectedIncarnationId: INCARNATION_ID + }) + await flush() + + expect(request).toHaveBeenCalledTimes(2) + resolvers[0](Promise.reject(new Error('relay dropped the probe'))) + resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 1 }) + + await expect(failing).rejects.toThrow('relay dropped the probe') + expect(await overlapping).toMatchObject({ observationEpoch: 1 }) + }) +}) diff --git a/src/main/providers/ssh-pty-provider-rpc-operations.ts b/src/main/providers/ssh-pty-provider-rpc-operations.ts index 3f9953b9d1c..71ddbefce97 100644 --- a/src/main/providers/ssh-pty-provider-rpc-operations.ts +++ b/src/main/providers/ssh-pty-provider-rpc-operations.ts @@ -50,6 +50,10 @@ export function createSshPtyProviderRpcOperations({ mux, toRelayPtyId }: SshPtyP const result = await mux.request('pty.getForegroundProcess', { id: toRelayPtyId(id) }) return result as string | null }, + // Do NOT in-flight coalesce this the way the sibling git reads are: the host mints one + // `observationEpoch` per request and the pane foreground reader commits it per read, so a + // shared reply reads as a stale replay and degrades a `live` identity read to `unverifiable`. + // Guarded by ssh-pty-inspect-observation-identity.test.ts; #17525 removes the poll. inspectProcess: async ( id: string, options?: { expectedIncarnationId?: string } diff --git a/src/main/ssh/relay-daemon-service-children.ts b/src/main/ssh/relay-daemon-service-children.ts new file mode 100644 index 00000000000..35e755ea518 --- /dev/null +++ b/src/main/ssh/relay-daemon-service-children.ts @@ -0,0 +1,62 @@ +/** + * Telling a relay daemon's own service processes apart from the work it holds. + * + * The reap gate used to ask `pgrep -P | grep -c .` and demand zero. But the daemon + * forks service children of its own — `relay-ai-vault-service.js` is spawned lazily and then + * never exits — so that count is permanently non-zero on any relay that has touched the AI + * Vault, whether or not it holds a single PTY. A superseded, disconnected relay holding + * nothing therefore reported `retained-live-work` forever, its version directory stayed + * pinned against GC by its own live socket, and the population grew without bound (#13614). + * + * The asymmetry below is the whole safety argument, and it follows + * docs/reference/ssh-execution-boundary.md: *subtracting a child we can positively identify + * as relay infrastructure is sound; assuming anything about a child we cannot identify is + * not.* An argv that does not match, an argv `ps` would not print, and a host without + * `pgrep` all count against the relay and keep it unreapable. Losing sight of a child is + * never evidence that it holds nothing. + */ +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' +import { shellEscape } from './ssh-connection-utils' + +/** Shell variable set to the daemon's direct-child count, or `unknown`. */ +export const RELAY_CHILD_COUNT_VAR = 'kids' + +/** Shell variable set to the count of children not identified as relay services, or `unknown`. */ +export const RELAY_UNRECOGNIZED_CHILD_COUNT_VAR = 'unrecognized_kids' + +/** + * `case` patterns matching a service child's argv. Suffix-anchored on purpose: both entries + * are forked with no script arguments, so the argv ends at the filename, and the leading `/` + * requires the absolute path the daemon forks rather than a bare mention of the name. A + * future arg would stop matching and the relay would go back to being retained — the safe + * direction to fail in. + */ +function serviceChildArgvPatterns(): string { + return RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.map( + (filename) => `*${shellEscape(`/${filename}`)}` + ).join('|') +} + +/** + * POSIX shell that censuses the direct children of `$pid`, setting `kids` and + * `unrecognized_kids`. Both stay `unknown` when the host cannot enumerate children at all. + */ +export function relayDaemonChildCensusShell(): string[] { + return [ + `${RELAY_CHILD_COUNT_VAR}=unknown`, + `${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=unknown`, + 'if command -v pgrep >/dev/null 2>&1; then', + ` ${RELAY_CHILD_COUNT_VAR}=0`, + ` ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=0`, + ' for kid in $(pgrep -P "$pid" 2>/dev/null); do', + ` ${RELAY_CHILD_COUNT_VAR}=$((${RELAY_CHILD_COUNT_VAR}+1))`, + ' kid_args=$(ps -o args= -p "$kid" 2>/dev/null | tr -d "\\n")', + ' case "$kid_args" in', + ` ${serviceChildArgvPatterns()}) ;;`, + // An unreadable or unrecognised argv lands here, which is what keeps the relay retained. + ` *) ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=$((${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}+1)) ;;`, + ' esac', + ' done', + 'fi' + ] +} diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts index 7ece0b6532e..a8975d0520b 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts @@ -4,7 +4,7 @@ * generated scripts through /bin/sh against real unix sockets and real processes. */ import { execFile, spawn, type ChildProcess } from 'node:child_process' -import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterAll, afterEach, beforeAll, describe, expect, it } from 'vitest' @@ -15,21 +15,32 @@ import { type RelayEndpointIncumbent } from './ssh-relay-endpoint-incumbent' import { reapEmptyRelayHuskCommand } from './ssh-relay-endpoint-takeover' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' const posixOnly = process.platform === 'win32' ? describe.skip : describe const FAKE_RELAY_SOURCE = ` const net = require('net') +const path = require('path') const sock = process.argv[process.argv.indexOf('--sock-path') + 1] +function spawnChild(args) { + require('child_process').spawn(process.execPath, args, { stdio: 'ignore' }) +} if (process.argv.includes('--with-child')) { - require('child_process').spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000)'], { - stdio: 'ignore' - }) + spawnChild(['-e', 'setTimeout(() => {}, 60000)']) +} +// Why forked the same way production does: the exclusion is argv-shaped, so a hand-written +// stand-in would test the test rather than the shell that runs on someone's host. +for (const name of process.argv.filter((arg) => arg.startsWith('--service-child='))) { + spawnChild([path.join(__dirname, name.slice('--service-child='.length))]) } net.createServer(() => {}).listen(sock, () => process.stdout.write('READY\\n')) process.on('SIGTERM', () => process.exit(0)) ` +// Self-limiting: these are orphaned when the relay under test is reaped. +const IDLE_SERVICE_SOURCE = 'setTimeout(() => {}, 60000)\n' + function sh(script: string): Promise { return new Promise((resolve, reject) => { execFile('/bin/sh', ['-c', script], { timeout: 20_000 }, (error, stdout) => { @@ -43,14 +54,21 @@ function sh(script: string): Promise { } let workDir: string +let pgreplessBinDir: string let hasLsof = false const running: ChildProcess[] = [] -function startFakeRelay(sockPath: string, withChild = false): Promise { +function startFakeRelay( + sockPath: string, + options: { withChild?: boolean; serviceChildren?: readonly string[] } = {} +): Promise { const args = [join(workDir, 'relay.js'), '--sock-path', sockPath] - if (withChild) { + if (options.withChild) { args.push('--with-child') } + for (const name of options.serviceChildren ?? []) { + args.push(`--service-child=${name}`) + } const child = spawn(process.execPath, args, { stdio: ['ignore', 'pipe', 'ignore'] }) running.push(child) return new Promise((resolve, reject) => { @@ -68,9 +86,31 @@ async function probe(sockPath: string): Promise { return parseRelayEndpointIncumbentProbe(sockPath, output) } +/** The relay forks its children after it starts listening, so the probe can race them. */ +async function waitForChildCount( + sockPath: string, + expected: number +): Promise { + let incumbent = await probe(sockPath) + for (let attempt = 0; attempt < 50 && incumbent.holders[0]?.childCount !== expected; attempt++) { + await new Promise((resolve) => setTimeout(resolve, 100)) + incumbent = await probe(sockPath) + } + return incumbent +} + beforeAll(async () => { workDir = mkdtempSync(join(tmpdir(), 'orca-relay-incumbent-')) writeFileSync(join(workDir, 'relay.js'), FAKE_RELAY_SOURCE) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + writeFileSync(join(workDir, filename), IDLE_SERVICE_SOURCE) + } + writeFileSync(join(workDir, 'looks-like-relay-watcher.js'), IDLE_SERVICE_SOURCE) + pgreplessBinDir = join(workDir, 'pgrepless-bin') + mkdirSync(pgreplessBinDir) + for (const tool of ['ps', 'tr']) { + symlinkSync((await sh(`command -v ${tool}`)).trim(), join(pgreplessBinDir, tool)) + } hasLsof = await sh('command -v lsof >/dev/null 2>&1 && echo yes || echo no').then( (out) => out.trim() === 'yes' ) @@ -105,13 +145,51 @@ posixOnly('relay endpoint probe against a real socket', () => { return } expect(incumbent.holders.map((holder) => holder.pid)).toEqual([relay.pid]) - expect(incumbent.holders[0]).toMatchObject({ matchesRelayArgv: true, childCount: 0 }) + expect(incumbent.holders[0]).toMatchObject({ + matchesRelayArgv: true, + childCount: 0, + unrecognizedChildCount: 0 + }) expect(isReapableRelayHusk(incumbent)).toBe(true) }) + it("counts the daemon's own service children but does not hold them against it", async () => { + const sockPath = join(workDir, 'services.sock') + await startFakeRelay(sockPath, { serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES }) + const incumbent = await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + + expect(incumbent.holders[0].childCount).toBe(RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + expect(incumbent.holders[0].unrecognizedChildCount).toBe(0) + expect(isReapableRelayHusk(incumbent)).toBe(true) + }) + + it('still retains a relay holding work alongside its service children', async () => { + const sockPath = join(workDir, 'services-and-work.sock') + await startFakeRelay(sockPath, { + withChild: true, + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + const incumbent = await waitForChildCount( + sockPath, + RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length + 1 + ) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + + it('does not excuse a child that merely mentions a service entry name', async () => { + const sockPath = join(workDir, 'lookalike.sock') + await startFakeRelay(sockPath, { serviceChildren: ['looks-like-relay-watcher.js'] }) + const incumbent = await waitForChildCount(sockPath, 1) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + it('refuses to call a relay with a live child an empty husk', async () => { const sockPath = join(workDir, 'busy.sock') - await startFakeRelay(sockPath, true) + await startFakeRelay(sockPath, { withChild: true }) const incumbent = await probe(sockPath) expect(incumbent.verdict).toBe('live') @@ -150,12 +228,34 @@ posixOnly('empty relay husk reap against a real process', () => { it('refuses to signal a relay that acquired a child after it was probed', async () => { const sockPath = join(workDir, 'raced.sock') - const relay = await startFakeRelay(sockPath, true) + const relay = await startFakeRelay(sockPath, { withChild: true }) const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) expect(output.trim()).toBe('BUSY') expect(relay.killed).toBe(false) }) + it('terminates a relay whose only children are its own service processes (#13614)', async () => { + const sockPath = join(workDir, 'service-husk.sock') + const relay = await startFakeRelay(sockPath, { + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) + expect(output.trim()).toBe('GONE') + }) + + it('refuses to signal when the host cannot enumerate children at all', async () => { + const sockPath = join(workDir, 'no-pgrep.sock') + const relay = await startFakeRelay(sockPath) + // A PATH carrying every tool the script needs except `pgrep`: the census answers + // `unknown`, which must reach BUSY rather than the zero a missing tool would imply. + const output = await sh( + `PATH=${pgreplessBinDir}\n${reapEmptyRelayHuskCommand(relay.pid!, sockPath)}` + ) + expect(output.trim()).toBe('BUSY') + expect(relay.killed).toBe(false) + }) + it('refuses to signal a pid whose argv is not this relay at this socket', async () => { const sockPath = join(workDir, 'mismatch.sock') await startFakeRelay(sockPath) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts index a65cb33fc57..de4cc28d170 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts @@ -32,17 +32,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('reports live when the socket accepted a connection', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=4242 yes 13']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=4242 yes 13 11' + ]) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('accepted-connection') - expect(incumbent.holders).toEqual([{ pid: 4242, matchesRelayArgv: true, childCount: 13 }]) + expect(incumbent.holders).toEqual([ + { pid: 4242, matchesRelayArgv: true, childCount: 13, unrecognizedChildCount: 11 } + ]) }) it('reports live when a process still holds an inode that refuses connections', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2']) + probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2 2']) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('holder-process') @@ -85,7 +92,12 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('drops holder lines that do not carry a usable pid', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=- no unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=refused', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=- no unknown unknown' + ]) ) expect(incumbent.holders).toEqual([]) expect(incumbent.verdict).toBe('exited') @@ -94,9 +106,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('keeps an unreadable child count as null rather than zero', () => { const [holder] = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=7 yes unknown unknown' + ]) ).holders expect(holder.childCount).toBeNull() + expect(holder.unrecognizedChildCount).toBeNull() + }) + + it('keeps a holder line with no unrecognized-child field unreapable', () => { + const incumbent = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes 0']) + ) + expect(incumbent.holders[0].unrecognizedChildCount).toBeNull() + expect(isReapableRelayHusk(incumbent)).toBe(false) }) }) @@ -172,27 +199,36 @@ describe('mayLaunchOverRelayEndpoint', () => { describe('isReapableRelayHusk', () => { const husk = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0']) + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0 0']) ) - it('accepts a single proven relay holder with zero children', () => { + it('accepts a single proven relay holder with no unaccounted-for children', () => { expect(isReapableRelayHusk(husk)).toBe(true) }) - it('refuses a relay that still holds children', () => { + it('accepts a relay whose only children are its own service processes (#13614)', () => { + const withServices = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 2 0']) + ) + expect(withServices.holders[0].childCount).toBe(2) + expect(isReapableRelayHusk(withServices)).toBe(true) + }) + + it('refuses a relay that still holds children it could not account for', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: 1 }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 3, unrecognizedChildCount: 1 }] }) ).toBe(false) }) - it('refuses a holder whose child count could not be read', () => { + it('refuses a holder whose unrecognized-child count could not be read', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: null }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: null }] }) ).toBe(false) }) @@ -201,7 +237,7 @@ describe('isReapableRelayHusk', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0 }] + holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0, unrecognizedChildCount: 0 }] }) ).toBe(false) }) @@ -211,8 +247,8 @@ describe('isReapableRelayHusk', () => { isReapableRelayHusk({ ...husk, holders: [ - { pid: 500, matchesRelayArgv: true, childCount: 0 }, - { pid: 501, matchesRelayArgv: true, childCount: 0 } + { pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 }, + { pid: 501, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 } ] }) ).toBe(false) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.ts index 2688267f4c7..628a9558793 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.ts @@ -20,6 +20,11 @@ */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_CHILD_COUNT_VAR, + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { isWindowsRemoteHost, type RemoteHostPlatform } from './ssh-remote-platform' @@ -38,6 +43,12 @@ export type RelayEndpointHolder = { matchesRelayArgv: boolean /** Direct children, or null when `pgrep` could not answer. Never guessed. */ childCount: number | null + /** + * Direct children *not* positively identified as the daemon's own service processes, or + * null when the host could not enumerate them. This — not `childCount` — is what says + * whether the relay holds anything; see relay-daemon-service-children.ts. + */ + unrecognizedChildCount: number | null } export type RelayEndpointIncumbent = { @@ -95,11 +106,9 @@ export function relayEndpointIncumbentProbeCommand(nodePath: string, sockPath: s ' args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', ' match=no', ' case "$args" in *relay.js*"$sock"*) match=yes ;; esac', - ' kids=unknown', - ' if command -v pgrep >/dev/null 2>&1; then', - ' kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - ' fi', - ' printf \'HOLDER=%s %s %s\\n\' "$pid" "$match" "$kids"', + ...relayDaemonChildCensusShell().map((line) => ` ${line}`), + ' printf \'HOLDER=%s %s %s %s\\n\' "$pid" "$match" ' + + `"$${RELAY_CHILD_COUNT_VAR}" "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}"`, ' done', 'else', " printf 'HOLDERS_SOURCE=unavailable\\n'", @@ -159,19 +168,25 @@ export function parseRelayEndpointIncumbentProbe( } function parseHolder(value: string): RelayEndpointHolder | null { - const [rawPid, rawMatch, rawKids] = value.split(/\s+/) + const [rawPid, rawMatch, rawKids, rawUnrecognized] = value.split(/\s+/) const pid = Number.parseInt(rawPid ?? '', 10) if (!Number.isInteger(pid) || pid <= 0) { return null } - const childCount = Number.parseInt(rawKids ?? '', 10) return { pid, matchesRelayArgv: rawMatch === 'yes', - childCount: Number.isInteger(childCount) && childCount >= 0 ? childCount : null + childCount: parseChildCount(rawKids), + unrecognizedChildCount: parseChildCount(rawUnrecognized) } } +/** `unknown`, a missing field, and anything unparseable are all "could not tell" — never 0. */ +function parseChildCount(raw: string | undefined): number | null { + const count = Number.parseInt(raw ?? '', 10) + return Number.isInteger(count) && count >= 0 ? count : null +} + function unverifiableEndpoint(sockPath: string): RelayEndpointIncumbent { return { sockPath, @@ -239,8 +254,12 @@ export function mayLaunchOverRelayEndpoint(incumbent: RelayEndpointIncumbent): b /** * A live relay that provably holds nothing: identity confirmed against its argv, exactly one - * holder, and zero children. Reaping it destroys no user work. Anything less is retained — - * killing the wrong pid on someone's remote host is the worst outcome available here. + * holder, and no child the host could not account for as one of the daemon's own service + * processes. Reaping it destroys no user work. Anything less is retained — killing the wrong + * pid on someone's remote host is the worst outcome available here. + * + * Why not `childCount === 0`: the daemon's AI Vault sidecar never exits once spawned, so that + * gate was unreachable for any relay that had ever served a vault request (#13614). */ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean { if (incumbent.verdict !== 'live' || !incumbent.holdersEnumerable) { @@ -250,12 +269,16 @@ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean return false } const [holder] = incumbent.holders - return holder.matchesRelayArgv && holder.childCount === 0 + return holder.matchesRelayArgv && holder.unrecognizedChildCount === 0 } export function describeRelayEndpointIncumbent(incumbent: RelayEndpointIncumbent): string { const holders = incumbent.holders - .map((holder) => `${holder.pid}(children=${holder.childCount ?? 'unknown'})`) + .map( + (holder) => + `${holder.pid}(children=${holder.childCount ?? 'unknown'},` + + `unrecognized=${holder.unrecognizedChildCount ?? 'unknown'})` + ) .join(',') return ( `${incumbent.sockPath} verdict=${incumbent.verdict} evidence=${incumbent.evidence} ` + diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts index 687d633b92b..d23f065f478 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts @@ -14,6 +14,7 @@ import { resolveRelayEndpointBeforeRelaunch } from './ssh-relay-endpoint-takeover' import { RelayVersionMismatchError } from './ssh-relay-version-mismatch-error' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' import type { SshConnection } from './ssh-connection' import { getRemoteHostPlatform } from './ssh-remote-platform' @@ -42,7 +43,7 @@ beforeEach(() => { describe('incumbent alive and refusing', () => { it('refuses to rebind a live relay holding PTYs, and signals nothing', async () => { execCommand.mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) // The whole point of #8585: the incumbent's socket must survive so it is not orphaned. @@ -52,9 +53,9 @@ describe('incumbent alive and refusing', () => { it('names the incumbent pid and the Reset Relay escape hatch in the error', async () => { execCommand.mockResolvedValue( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) - await expect(resolve()).rejects.toThrow(/3669803\(children=13\)/) + await expect(resolve()).rejects.toThrow(/3669803\(children=13,unrecognized=11\)/) await expect(resolve()).rejects.toThrow(/Reset Relay/) }) @@ -70,7 +71,7 @@ describe('incumbent alive and refusing', () => { it('reaps a live relay only when it provably holds nothing, and confirms it is gone', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') await expect(resolve()).resolves.toMatchObject({ verdict: 'live' }) @@ -80,7 +81,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over an empty relay whose death could not be confirmed', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -89,7 +90,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over a relay the host refused to signal on its own re-check', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('BUSY\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -138,9 +139,19 @@ describe('reapEmptyRelayHuskCommand', () => { }) it('aborts without signalling when the host cannot count children', () => { - expect(reapEmptyRelayHuskCommand(4242, SOCK)).toContain( - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }" - ) + const command = reapEmptyRelayHuskCommand(4242, SOCK) + // The census leaves both counters at `unknown` without pgrep, and the gate demands "0". + expect(command).toContain('unrecognized_kids=unknown') + expect(command).toContain('command -v pgrep >/dev/null 2>&1') + expect(command).toContain('[ "$unrecognized_kids" = "0" ] ||') + }) + + it('subtracts only the daemon service children it can name from the reap gate', () => { + const command = reapEmptyRelayHuskCommand(4242, SOCK) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + expect(command).toContain(`*'/${filename}'`) + } + expect(command).toContain('unrecognized_kids=$((unrecognized_kids+1))') }) }) diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.ts b/src/main/ssh/ssh-relay-endpoint-takeover.ts index f104aab5256..8f6130620cb 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.ts @@ -2,13 +2,17 @@ * Deciding whether a relay socket path is ours to take, and acting on the answer. * * The only destructive action available here is a SIGTERM to a relay that has been proven — - * by argv, by socket-holder enumeration, and by a zero child count re-checked on the host - * immediately before the signal — to hold nothing at all. Everything else is left running. + * by argv, by socket-holder enumeration, and by a child census re-run on the host immediately + * before the signal — to hold nothing at all. Everything else is left running. * Per docs/reference/ssh-execution-boundary.md, a relay we merely failed to reach is * `unverifiable`, and `unverifiable` never authorizes a kill or a rebind. */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { describeRelayEndpointIncumbent, @@ -39,9 +43,10 @@ export function reapEmptyRelayHuskCommand(pid: number, sockPath: string): string `sock=${shellEscape(sockPath)}`, 'args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', 'case "$args" in *relay.js*"$sock"*) ;; *) printf \'MISMATCH\\n\'; exit 0 ;; esac', - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }", - 'kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - '[ "$kids" = "0" ] || { printf \'BUSY\\n\'; exit 0; }', + // Why the same census as the probe: `unknown` (no pgrep) and any child this host could + // not account for as a relay service both land on BUSY, so nothing is signalled. + ...relayDaemonChildCensusShell(), + `[ "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}" = "0" ] || { printf 'BUSY\\n'; exit 0; }`, // SIGTERM only: the relay's own handler disposes and unlinks. SIGKILL would leave the // socket inode behind and skip that shutdown path for no gain on an empty daemon. 'kill -TERM "$pid" 2>/dev/null || true', diff --git a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts index 874d9aae3fe..9168f4688bd 100644 --- a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts +++ b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts @@ -67,7 +67,7 @@ describe('classifySupersededRelay', () => { 'PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', - 'HOLDER=3669803 yes 13' + 'HOLDER=3669803 yes 13 11' ]) ) ).toBe('retained-live-work') @@ -76,7 +76,7 @@ describe('classifySupersededRelay', () => { it('nominates only a proven empty relay for reaping', () => { expect( classifySupersededRelay( - incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) ).toBe('reap-candidate') }) @@ -101,7 +101,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) expect(findings).toHaveLength(1) @@ -114,7 +114,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) @@ -126,7 +126,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) diff --git a/src/main/ssh/ssh-remote-powershell.ts b/src/main/ssh/ssh-remote-powershell.ts index 8c94fd3c483..420223ced29 100644 --- a/src/main/ssh/ssh-remote-powershell.ts +++ b/src/main/ssh/ssh-remote-powershell.ts @@ -11,14 +11,24 @@ export { // to leave room for the `/c` wrapper sshd adds before cmd.exe counts the line. const WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS = 8_000 -export function powerShellCommand(script: string): string { - const inline = encodedPowerShellCommand(script) +/** + * `pwsh.exe` is PowerShell 7. It is not present on a stock Windows install, so it is only ever + * chosen after a probe — but where it exists it reads a redirected stdin correctly, which Windows + * PowerShell 5.1 does not (see `system-ssh-file-binary-transfer.ts`). + */ +export type WindowsPowerShellExecutable = 'powershell.exe' | 'pwsh.exe' + +export function powerShellCommand( + script: string, + executable: WindowsPowerShellExecutable = 'powershell.exe' +): string { + const inline = encodedPowerShellCommand(script, executable) if (inline.length <= WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { return inline } // Why: these scripts are repetitive enough that gzip beats the UTF-16LE tax by // ~4x, which is the difference between a line cmd.exe runs and one it refuses. - const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script)) + const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script), executable) if (compressed.length > WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { throw new Error( `Remote Windows command needs ${compressed.length} characters; Orca budgets ${WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS} for a line sshd hands to cmd.exe, which itself refuses more than ${CMD_EXE_COMMAND_LINE_MAX_CHARS}.` @@ -27,8 +37,8 @@ export function powerShellCommand(script: string): string { return compressed } -function encodedPowerShellCommand(script: string): string { - return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` +function encodedPowerShellCommand(script: string, executable: WindowsPowerShellExecutable): string { + return `${executable} -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` } /** Orca-prefixed names so the payload can never shadow the bootstrap's own state. */ diff --git a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts index c104078238e..fa34a8fbda7 100644 --- a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts +++ b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts @@ -4,6 +4,10 @@ import { CMD_EXE_COMMAND_LINE_MAX_CHARS } from '../providers/windows-shell-args' import { getRemoteHostPlatform } from './ssh-remote-platform' import { tryStealInstallLockCommand } from './ssh-relay-install-lock-commands' import { decodeRemotePowerShellScript, powerShellCommand } from './ssh-remote-powershell' +import { + makeWindowsPublishStagedFileCommand, + makeWindowsWriteFileCommand +} from './system-ssh-windows-file-write' import { cleanupOwnedRelayUploadStageCommand, promoteOwnedRelayUploadStageCommand, @@ -38,6 +42,17 @@ describe('Windows remote command line limit', () => { [ 'steal stale install lock', tryStealInstallLockCommand(windows, 'C:\\Users\\orca\\.orca-remote\\relay', 1_200) + ], + // F11 flagged these two as uncovered. They carry one path literal each, so they are the file + // commands whose length a caller can actually move. + ['write file', makeWindowsWriteFileCommand('C:\\Users\\orca\\.orca-remote\\relay.js')], + [ + 'publish staged file', + makeWindowsPublishStagedFileCommand( + 'C:\\Users\\orca\\.orca-remote\\relay.js.orca-partial-0123456789ab', + 'C:\\Users\\orca\\.orca-remote\\relay.js', + 'create' + ) ] ])('keeps the %s command inside what sshd\u2019s cmd.exe accepts', (_name, command) => { expect(command.length).toBeLessThanOrEqual(CMD_EXE_COMMAND_LINE_MAX_CHARS) @@ -76,3 +91,32 @@ describe('Windows remote command line limit', () => { ) }) }) + +/** + * F11 asked whether a pathological path could reach the budget, and what happens if it does. + * Measured: the inline encoding crosses 8000 at roughly 2500 high-entropy path characters — an + * order of magnitude past what Windows itself accepts — and the failure is a throw before any ssh + * is spawned, never a hang. + */ +describe('Windows file command budget headroom', () => { + it('absorbs a path far longer than Windows will accept', () => { + const deep = `C:\\Users\\orca\\${'segment\\'.repeat(30)}relay.js` + + expect(deep.length).toBeGreaterThan(260) + expect(makeWindowsWriteFileCommand(deep).length).toBeLessThanOrEqual( + CMD_EXE_COMMAND_LINE_MAX_CHARS + ) + }) + + it('throws rather than spawning a line cmd.exe would refuse', () => { + // Random segments so gzip cannot rescue it, which is the only way to reach the ceiling at all. + const incompressible = Array.from( + { length: 400 }, + (_unused, index) => `${index}-${Math.random().toString(36).slice(2)}` + ).join('\\') + + expect(() => makeWindowsWriteFileCommand(`C:\\${incompressible}\\f.bin`)).toThrow( + /Orca budgets 8000/ + ) + }) +}) diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index c366e899bf6..b477ad682ef 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -709,9 +709,13 @@ describe('spawnSystemSsh', () => { expect(args[standaloneControlIdx + 1]).toBe('none') }) - it('writes files to Windows system SSH targets with PowerShell stdin bytes', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('sends Windows file writes over sftp, not through a remote PowerShell stdin', async () => { + const spawned: EventedProcess[] = [] + spawnMock.mockImplementation(() => { + const proc = createEventedProcess() + spawned.push(proc) + return closeOnceSpawned(proc) + }) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -722,16 +726,24 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('0.1.0', 'utf-8')) + // #16432, re-measured: Windows PowerShell 5.1 can lose a redirected stdin for good when a read + // finds it momentarily empty, so the bytes must not travel that way at all. + const batch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(batch).toContain('put ') + expect(batch).toContain('/C:/Users/me/.orca-remote/relay/.version.orca-partial-') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + expect(sftpArgs).toContain('-b') + // The rename that publishes it reads the staged file, never a pipe. + const publish = (spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '' + expect(publish).toContain('powershell.exe') + expect(decodePowerShellCommand(publish)).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + expect(publish).not.toContain('/bin/sh') }) - it('writes binary buffers to Windows system SSH targets with CreateNew mode', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('enforces an exclusive Windows buffer write at the rename, where it is atomic', async () => { + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeBufferViaSystemSsh( @@ -742,12 +754,11 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(decodePowerShellCommand(remoteCommand)).toContain('CreateNew') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('png')) + const publish = decodePowerShellCommand((spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '') + // `File::Move` raising on an existing destination is what carries the exclusive contract now; + // a `CreateNew` on the staged file would only refuse a leftover of our own. + expect(publish).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish).not.toContain('[System.IO.File]::Delete($path)') }) it('downloads files from Windows system SSH targets with PowerShell stdout bytes', async () => { @@ -779,8 +790,7 @@ describe('spawnSystemSsh', () => { }) it('forces standalone SSH for Windows file writes when requested', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -791,10 +801,14 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + // sftp's own `-S` names a program to run, so the same request has to be spelled as an option. + expect(sftpArgs).not.toContain('-S') + expect(sftpArgs).toContain('ControlPath=none') + const publishArgs = spawnMock.mock.calls[1][1] as string[] + const standaloneControlIdx = publishArgs.indexOf('-S') expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + expect(publishArgs[standaloneControlIdx + 1]).toBe('none') }) it('uploads a Windows directory as a mkdir batch plus per-file writes, never one blob', async () => { @@ -819,19 +833,20 @@ describe('spawnSystemSsh', () => { rmSync(localDir, { recursive: true, force: true }) } + // #16432: directories first, then the file — but both over sftp now, so the only PowerShell + // left is the rename that publishes the staged file, which reads a file rather than a pipe. + const mkdirBatch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(mkdirBatch).toBe('-mkdir "/C:/Users/me/.orca-remote/relay"\n') + const putBatch = String(spawned[1]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(putBatch).toContain('put ') + expect(putBatch).toContain('/C:/Users/me/.orca-remote/relay/relay.js.orca-partial-') const commands = spawnMock.mock.calls.map((call) => (call[1] as string[]).at(-1) ?? '') - // #16432: directories first (metadata only), then the file bytes on their own stdin. One batch - // meant base64-ing the whole bundle into a single PowerShell string, which the remote never read. - expect(commands).toHaveLength(2) - expect(commands.every((command) => command.includes('powershell.exe'))).toBe(true) expect(commands.every((command) => !command.includes('/bin/sh'))).toBe(true) expect(commands.join('\n')).not.toContain('tar -xzf') - expect(JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string)).toEqual([ - 'C:/Users/me/.orca-remote/relay' - ]) - expect(Buffer.from(spawned[1].stdin.end.mock.calls[0]?.[0] as Buffer).toString('utf-8')).toBe( - 'console.log("relay")' - ) + // Nothing base64s the bundle into one PowerShell string any more, and nothing reads one. + expect( + commands.some((command) => decodePowerShellCommand(command).includes('OpenStandardInput')) + ).toBe(false) }) it('forces standalone SSH for Windows upload packages when requested', async () => { @@ -855,9 +870,9 @@ describe('spawnSystemSsh', () => { } const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') - expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + // The first spawn is the sftp client, whose own `-S` names a program to run. + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') }) it('throws when no system ssh is found', () => { diff --git a/src/main/ssh/system-ssh-file-binary-transfer.ts b/src/main/ssh/system-ssh-file-binary-transfer.ts index b0c5b662ed1..149d10dfbd3 100644 --- a/src/main/ssh/system-ssh-file-binary-transfer.ts +++ b/src/main/ssh/system-ssh-file-binary-transfer.ts @@ -1,5 +1,7 @@ import { constants, createWriteStream } from 'node:fs' -import { lstat, open } from 'node:fs/promises' +import { lstat, mkdtemp, open, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import type { Writable } from 'node:stream' import { pipeline } from 'node:stream/promises' import type { SshTarget } from '../../shared/ssh-types' @@ -16,6 +18,16 @@ import { throwIfAborted, waitForChannelClose } from './system-ssh-operation-lifecycle' +import { + writeWindowsRemoteFile, + type WindowsWriteSource +} from './system-ssh-windows-write-strategy' + +export { + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS +} from './system-ssh-windows-write-strategy' +export { WINDOWS_STAGED_WRITE_SUFFIX } from './system-ssh-windows-file-write' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -74,13 +86,16 @@ export async function writeBufferViaSystemSsh( ): Promise { throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - await writeWindowsBytesViaSystemSsh( + await writeWindowsRemoteFile( target, remotePath, - contents.length, - (offset, maxBytes) => - Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), - options + { + totalBytes: contents.length, + readChunk: (offset, maxBytes) => + Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), + withLocalFile: (send) => withTemporaryLocalFile(contents, send) + }, + options ?? {} ) return } @@ -127,20 +142,19 @@ export async function uploadFileViaSystemSsh( throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - // #16432: a Windows host cannot take a whole file through one stdin, however the local side - // paces it — see WINDOWS_STDIN_WRITE_CHUNK_BYTES. This is the path that carries the large - // files, so it is the one that has to be chunked and bounded. - await writeWindowsBytesViaSystemSsh( - target, - remotePath, - openedStat.size, - async (offset, maxBytes) => { + // This is the path that carries the large files, so it is the one the transport choice is + // made for; see the #16432 note below. + const source: WindowsWriteSource = { + totalBytes: openedStat.size, + readChunk: async (offset, maxBytes) => { const buffer = Buffer.allocUnsafe(Math.min(maxBytes, openedStat.size - offset)) const { bytesRead } = await handle.read(buffer, 0, buffer.length, offset) return buffer.subarray(0, bytesRead) }, - options - ) + // The verified local file is already exactly the payload, so sftp sends it as is. + withLocalFile: (send) => send(localPath) + } + await writeWindowsRemoteFile(target, remotePath, source, options ?? {}) return } @@ -173,158 +187,56 @@ export async function uploadFileViaSystemSsh( } /** - * #16432: Windows PowerShell 5.1 stops draining a redirected stdin over a non-pty ssh exec - * somewhere between 50KB and 1MB, depending on the host's `DefaultShell`, and it hangs rather than - * failing. The reporter measured that on both constructs he tried — `[Console]::In.ReadToEnd()` and - * `new IO.StreamReader([Console]::OpenStandardInput())`, the latter reading incrementally, which is - * why the limit cannot be attributed to materializing the payload. `Stream.CopyTo` reads the same - * `[Console]::OpenStandardInput()` object with the same incremental `Read` loop, so nothing in it - * escapes that limit either: no single write may exceed what one stdin is known to carry. + * #16432, re-measured: the constraint is not a size limit, and it is not cmd.exe's. * - * 32KB is an order of magnitude under the low end of the measured range, and under 50KB, which the - * reporter measured succeeding against a stream reader on the worse of the two `DefaultShell` - * settings. - */ -export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 - -/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ -export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 - -/** Suffix for the path a multi-exec Windows write lands on before it is published by rename. */ -export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' - -/** - * Splits one logical Windows write into stdin-sized execs. + * A read on Windows PowerShell 5.1's redirected-stdin handle over a non-pty ssh exec can die + * permanently when it finds the stream momentarily empty: no further bytes arrive, and no EOF ever + * does. It is probabilistic per such read — not a size threshold, and not certain on the first one. + * Measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2 with `DefaultShell = cmd.exe`, by + * replacing the copy loop with a counting reader: * - * A write that needs more than one exec cannot land on the destination directly: a chunk failing - * mid-file would leave a truncated artifact under the real name with nothing marking it incomplete, - * and the retry would then meet its own leftovers — under `exclusive` the retry's `CreateNew` fails - * on them. Multi-exec creates therefore land on a staging path and are published by a rename, which - * is also where `exclusive` is enforced: once, at the destination, instead of smeared across the - * first chunk. A caller-requested append cannot be staged without reading the remote file back, so - * it keeps writing straight through, as its own protocol already implies. + * - a 1.5s gap before any byte, which forces the first read to find nothing -> 0 bytes, 6 of 6 + * - one byte, a 1.5s gap, then 32767 more -> exactly 1 byte, then nothing + * - 32768, a 1.5s gap, then 32768 more -> exactly 32768, then nothing + * - a continuous 2MB -> 167936 / 270336 / 372736, then nothing + * + * Those three 2MB death points are one payload run three times under the same conditions, which is + * what rules out a threshold: a stream that died at a fixed point would not vary by 2x. Independently reproduced by + * a second harness, where one 1.9MB counted read survived 39 reads to completion and another died + * after 11 — same construct, same payload. + * + * A payload small enough to arrive in one burst usually presents only one read that can find the + * stream empty (the one waiting for EOF), which is why 32KB mostly works: it still failed 15 times + * in 120 with the host under load, and 1 in 40 on a quiet one. Neither rate is survivable across + * the 62 execs a 1.9MB file needs — even 2.5% compounds to roughly four uploads in five failing — + * and no chunk size helps, because the client does not control whether its bytes arrive together. + * + * The same host, same `DefaultShell`, same connection pattern contradicts every size-limit reading: + * `findstr` took 2,016,000 bytes through one exec's stdin, and PowerShell 7 took 2MB. So cmd.exe is + * not the ceiling and neither is ~50KB. Writes now go over sftp, which moves the whole payload + * without any remote process reading a pipe; see `system-ssh-windows-write-strategy.ts` for the + * fallback order. + * + * Successes are never partial. Across every run in both harnesses a failed write hung; not one + * produced a short file, so this defect cannot silently truncate an upload. */ -async function writeWindowsBytesViaSystemSsh( - target: SshTarget, - remotePath: string, - totalBytes: number, - readChunk: (offset: number, maxBytes: number) => Promise, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const staged = !options.append && totalBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES - const writePath = staged ? `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` : remotePath - let offset = 0 - // An empty write still has to run: it is what creates (or truncates) the file. - do { - const chunk = await readChunk(offset, WINDOWS_STDIN_WRITE_CHUNK_BYTES) - if (chunk.length === 0 && offset < totalBytes) { - throw new Error(`Source ran short during upload of ${remotePath}`) - } - await writeWindowsChunkViaSystemSsh( - target, - writePath, - chunk, - { - ...options, - append: staged ? offset > 0 : options.append === true || offset > 0, - exclusive: staged ? false : options.exclusive === true && offset === 0 - }, - offset - ) - offset += chunk.length - } while (offset < totalBytes) - if (staged) { - await publishWindowsStagedWrite(target, writePath, remotePath, options) + +/** A staged write is materialized locally first when the source is a buffer rather than a file. */ +async function withTemporaryLocalFile( + contents: Buffer, + send: (localPath: string) => Promise +): Promise { + const directory = await mkdtemp(join(tmpdir(), 'orca-win-upload-')) + const localPath = join(directory, 'payload.bin') + try { + // 0600: the payload can be repository content, and tmpdir is shared on every platform. + await writeFile(localPath, contents, { mode: 0o600 }) + return await send(localPath) + } finally { + await rm(directory, { recursive: true, force: true }).catch(() => {}) } } -async function writeWindowsChunkViaSystemSsh( - target: SshTarget, - remotePath: string, - chunk: Buffer, - options: SystemSshWriteBufferOptions, - offset: number -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsWriteFileCommand(remotePath, options), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose( - channel, - `write ${remotePath} at offset ${offset}`, - WINDOWS_STDIN_WRITE_TIMEOUT_MS - ) - ) - if (!options.signal?.aborted) { - channel.stdin.end(chunk) - } - await closePromise -} - -async function publishWindowsStagedWrite( - target: SshTarget, - stagingPath: string, - remotePath: string, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand( - target, - makeWindowsPublishStagedFileCommand(stagingPath, remotePath, options.exclusive === true), - { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } - ) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, `publish ${remotePath}`, WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ) - if (!options.signal?.aborted) { - channel.stdin.end() - } - await closePromise -} - -function makeWindowsWriteFileCommand( - remotePath: string, - options?: { append?: boolean; exclusive?: boolean } -): string { - const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$path = ${powerShellLiteral(remotePath)}`, - '$parent = [System.IO.Path]::GetDirectoryName($path)', - 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', - '$inputStream = [Console]::OpenStandardInput()', - `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, - 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' - ].join('; ') - ) -} - -// `File::Move` throws when the destination exists, which is exactly the exclusive contract; the -// non-exclusive caller asked to replace, so it deletes first (a no-op on an absent path). -function makeWindowsPublishStagedFileCommand( - stagingPath: string, - remotePath: string, - exclusive: boolean -): string { - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$staging = ${powerShellLiteral(stagingPath)}`, - `$path = ${powerShellLiteral(remotePath)}`, - ...(exclusive ? [] : ['[System.IO.File]::Delete($path)']), - '[System.IO.File]::Move($staging, $path)' - ].join('; ') - ) -} - function makePosixWriteFileCommand( remotePath: string, options?: { append?: boolean; exclusive?: boolean } diff --git a/src/main/ssh/system-ssh-file-transfer.ts b/src/main/ssh/system-ssh-file-transfer.ts index f728c0eab02..d5757904356 100644 --- a/src/main/ssh/system-ssh-file-transfer.ts +++ b/src/main/ssh/system-ssh-file-transfer.ts @@ -27,6 +27,12 @@ import { WINDOWS_STDIN_WRITE_TIMEOUT_MS, writeBufferViaSystemSsh } from './system-ssh-file-binary-transfer' +import { + isSftpPathUnsupportedError, + isSftpUnavailableError, + makeDirectoriesViaSftp +} from './system-ssh-sftp-transfer' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -161,9 +167,14 @@ async function collectWindowsUploadPlan( return plan } -// Why the JSON envelope survives here: a path list is metadata, so this payload stays in the -// hundreds of bytes even for a deep tree. Batched anyway, so a pathological tree cannot walk back -// into the same stdin size that wedges PowerShell. +/** + * Creates the upload's directories, preferring sftp's own `mkdir`. + * + * The PowerShell fallback keeps the JSON envelope, batched under one stdin's worth: a path list is + * metadata, so it stays in the hundreds of bytes even for a deep tree. It is still a redirected + * stdin read though, so on Windows PowerShell 5.1 it carries the same defect as any other — which + * is why sftp is tried first even for a payload this small. + */ async function createWindowsUploadDirectories( target: SshTarget, directories: readonly string[], @@ -175,23 +186,27 @@ async function createWindowsUploadDirectories( if (batch.length === 0) { return } + const pending = batch const payload = JSON.stringify(batch) batch = [] batchBytes = 0 throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + await getWindowsRemoteWriteCapabilities(target).runWithFallback( + 'sftp-subsystem', + async () => { + try { + await makeDirectoriesViaSftp(target, pending, options) + } catch (error) { + // A directory sftp cannot address is this batch's problem, not the host's verdict. + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await createWindowsUploadDirectoriesViaPowerShell(target, payload, options) + } + }, + () => createWindowsUploadDirectoriesViaPowerShell(target, payload, options), + isSftpUnavailableError ) - if (!options.signal?.aborted) { - channel.stdin.end(payload) - } - await closePromise } for (const directory of directories) { const entryBytes = Buffer.byteLength(directory) + 4 @@ -204,17 +219,44 @@ async function createWindowsUploadDirectories( await flush() } +async function createWindowsUploadDirectoriesViaPowerShell( + target: SshTarget, + payload: string, + options: SystemSshOperationOptions +): Promise { + const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end(payload) + } + await closePromise +} + function makeWindowsCreateDirectoriesCommand(): string { return powerShellCommand( [ '$ErrorActionPreference = "Stop"', - // The reporter measured this reader surviving 50KB where `[Console]::In` wedged at the same - // size (#16432); the batch above stays under that. + // Reached only where the host has no sftp subsystem. Windows PowerShell 5.1 can lose a + // redirected stdin for good when a read finds it empty (#16432); a batch this small usually + // arrives in one piece, and "usually" is exactly why sftp is preferred. '$reader = New-Object System.IO.StreamReader([Console]::OpenStandardInput())', 'try { $json = $reader.ReadToEnd() } finally { $reader.Dispose() }', 'if ([string]::IsNullOrWhiteSpace($json)) { return }', - 'foreach ($path in @($json | ConvertFrom-Json)) {', - ' $null = [System.IO.Directory]::CreateDirectory([string]$path)', + // `[string[]]`, not `@(...)`: ConvertFrom-Json emits the parsed array as a single pipeline + // object, so `@(...)` wraps it in *another* array and the loop variable binds to the whole + // thing. `[string]` of that is the paths joined by spaces, which CreateDirectory rejects with + // "The given path's format is not supported". It only ever worked for a one-element batch, + // where stringifying a single-element array happens to yield the element. Measured on + // WindowsPowerShell 5.1.26100 against a three-directory tree. + 'foreach ($path in [string[]]($json | ConvertFrom-Json)) {', + ' $null = [System.IO.Directory]::CreateDirectory($path)', '}' ].join('; ') ) diff --git a/src/main/ssh/system-ssh-sftp-args.test.ts b/src/main/ssh/system-ssh-sftp-args.test.ts new file mode 100644 index 00000000000..d971390f8dd --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.test.ts @@ -0,0 +1,141 @@ +/** + * `buildSshArgs` is shared with the sftp client, and three of its flags mean something else there. + * Every case below is a silent wrong-target rather than an error if the translation is skipped, + * which is why the fallback is "refuse and use another transport", never "pass it through". + */ +import { describe, expect, it } from 'vitest' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' + +describe('translateSshArgsToSftpArgs', () => { + it('sends the port as an option, since sftp -p preserves mtimes instead', () => { + const args = translateSshArgsToSftpArgs(['-p', '2222', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'Port=2222', '--', 'dev@win.example']) + }) + + it('sends the login name as an option, since sftp has no -l', () => { + // `buildSshArgs` emits `-l` for a config alias no Host block claims. Throwing here would send + // exactly those hosts to the transport this PR exists to stop using, silently. + const args = translateSshArgsToSftpArgs(['-l', 'neil', '--', 'awin']) + + expect(args).toEqual(['-o', 'User=neil', '--', 'awin']) + }) + + it('translates the whole unclaimed-alias shape buildSshArgs emits', () => { + const args = translateSshArgsToSftpArgs([ + '-o', + 'BatchMode=no', + '-T', + '-S', + 'none', + '-o', + 'Hostname=192.168.0.186', + '-p', + '2222', + '-l', + 'neil', + '--', + 'awin' + ]) + + expect(args).toEqual([ + '-o', + 'BatchMode=no', + '-o', + 'ControlPath=none', + '-o', + 'Hostname=192.168.0.186', + '-o', + 'Port=2222', + '-o', + 'User=neil', + '--', + 'awin' + ]) + }) + + it('spells ControlPath=none out, since sftp -S names a program to run', () => { + // `sftp -S none` would try to exec a binary called `none`. + const args = translateSshArgsToSftpArgs(['-S', 'none', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'ControlPath=none', '--', 'dev@win.example']) + }) + + it('refuses any other -S, which would hand sftp an ssh binary Orca did not choose', () => { + expect(() => translateSshArgsToSftpArgs(['-S', '/tmp/ctl.sock'])).toThrow( + SftpArgTranslationError + ) + }) + + it('drops -T, which sftp does not have', () => { + expect(translateSshArgsToSftpArgs(['-T', '--', 'host'])).toEqual(['--', 'host']) + }) + + it('passes through the flags both clients spell the same way', () => { + const args = translateSshArgsToSftpArgs([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + + expect(args).toEqual([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + }) + + it('takes everything after -- as the destination without reinterpreting it', () => { + // A host literally named `-p` is not a flag once `--` has been seen. + expect(translateSshArgsToSftpArgs(['--', '-p'])).toEqual(['--', '-p']) + }) + + it('refuses an unknown flag rather than guessing what sftp would do with it', () => { + // The point of the throw: a flag added to buildSshArgs later must degrade to another + // transport, not reach sftp carrying a different meaning. + expect(() => translateSshArgsToSftpArgs(['-A', '--', 'host'])).toThrow(SftpArgTranslationError) + }) + + it('refuses a value flag with no value', () => { + expect(() => translateSshArgsToSftpArgs(['-o'])).toThrow(SftpArgTranslationError) + }) +}) + +describe('withSftpKeepalive', () => { + it('asks OpenSSH to notice a dead peer, since the transfer itself has no wall-clock bound', () => { + expect(withSftpKeepalive(['--', 'host'])).toEqual([ + '-o', + 'ServerAliveInterval=15', + '-o', + 'ServerAliveCountMax=3', + '--', + 'host' + ]) + }) + + it('leaves a caller-stated keepalive policy alone', () => { + const args = withSftpKeepalive(['-o', 'ServerAliveInterval=60', '--', 'host']) + + expect(args.filter((arg) => arg.startsWith('ServerAliveInterval'))).toEqual([ + 'ServerAliveInterval=60' + ]) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-args.ts b/src/main/ssh/system-ssh-sftp-args.ts new file mode 100644 index 00000000000..17fa37e78bb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.ts @@ -0,0 +1,95 @@ +/** + * Rewrites `buildSshArgs` output for the sftp(1) client. + * + * Three flags ssh and sftp share spell different things: sftp's `-p` is "preserve mtime", its `-S` + * names the ssh binary to run, and it has no `-T` at all. Passing ssh's list through unchanged + * would silently connect to the wrong port and try to exec a program called `none`. + * + * Anything this table does not recognize throws. A flag added to `buildSshArgs` later must degrade + * to the non-sftp transfer path, never reach sftp carrying a different meaning. + */ + +/** `buildSshArgs` emitted a flag with no sftp equivalent; the caller should use another transport. */ +export class SftpArgTranslationError extends Error { + constructor(flag: string) { + super(`No sftp equivalent for system ssh argument ${JSON.stringify(flag)}`) + this.name = 'SftpArgTranslationError' + } +} + +/** Flags whose spelling and meaning are identical in both clients. */ +const PASSTHROUGH_VALUE_FLAGS = new Set(['-F', '-o', '-i', '-J']) + +export function translateSshArgsToSftpArgs(sshArgs: readonly string[]): string[] { + const sftpArgs: string[] = [] + let index = 0 + while (index < sshArgs.length) { + const flag = sshArgs[index]! + if (flag === '--') { + // Everything after `--` is the destination, which both clients spell the same way. + sftpArgs.push(...sshArgs.slice(index)) + return sftpArgs + } + const value = sshArgs[index + 1] + if (PASSTHROUGH_VALUE_FLAGS.has(flag)) { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push(flag, value) + index += 2 + continue + } + if (flag === '-T') { + // sftp never allocates a tty, so ssh's "no tty" request has nothing to translate to. + index += 1 + continue + } + if (flag === '-p') { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `Port=${value}`) + index += 2 + continue + } + if (flag === '-l') { + // sftp has no `-l`; the login name is an option there. `buildSshArgs` emits this for an + // unclaimed config alias, so throwing would route those hosts down the defective path and + // then cache the refusal against them for half an hour. + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `User=${value}`) + index += 2 + continue + } + if (flag === '-S') { + // ssh's `-S none` is ControlPath=none; sftp's `-S` would run a binary called `none`. + if (value !== 'none') { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', 'ControlPath=none') + index += 2 + continue + } + throw new SftpArgTranslationError(flag) + } + return sftpArgs +} + +/** + * A transfer that stalls mid-stream has no per-write bound to catch it, so ask OpenSSH to notice a + * dead peer itself. Only added when the caller has not already stated a keepalive policy. + */ +export function withSftpKeepalive(sftpArgs: readonly string[]): string[] { + const hasOption = (name: string): boolean => + sftpArgs.some((arg, position) => sftpArgs[position - 1] === '-o' && arg.startsWith(`${name}=`)) + const keepalive: string[] = [] + if (!hasOption('ServerAliveInterval')) { + keepalive.push('-o', 'ServerAliveInterval=15') + } + if (!hasOption('ServerAliveCountMax')) { + keepalive.push('-o', 'ServerAliveCountMax=3') + } + return [...keepalive, ...sftpArgs] +} diff --git a/src/main/ssh/system-ssh-sftp-path.test.ts b/src/main/ssh/system-ssh-sftp-path.test.ts new file mode 100644 index 00000000000..e196c2e3e8f --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.test.ts @@ -0,0 +1,59 @@ +/** + * Both functions here guard against the same measured failure: sftp's batch lexer treats `\` as an + * escape, so a Windows path handed over raw is silently mis-targeted *and the client still exits + * 0*. On Windows 11 / OpenSSH 10.0p2, `put src C:\Users\neil\qt\a.bin` created a file literally + * named `C` in the start directory and reported success. + */ +import { describe, expect, it } from 'vitest' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' + +describe('toSftpRemotePath', () => { + it('roots a drive path under /, which is the namespace the Windows sftp-server exposes', () => { + // `pwd` in that session reports `/C:/Users/dev`. + expect(toSftpRemotePath('C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('accepts a path already in that namespace unchanged', () => { + expect(toSftpRemotePath('/C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('converts the separators Orca stores paths with', () => { + expect(toSftpRemotePath('C:\\Users\\dev\\f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('declines a UNC path rather than guessing where it lands', () => { + // A guess here writes real bytes to the wrong place; declining falls back to another transport. + expect(() => toSftpRemotePath('//server/share/f.bin')).toThrow(UnsupportedSftpPathError) + }) + + it('declines a relative path, which would resolve against the session start directory', () => { + expect(() => toSftpRemotePath('Users/dev/f.bin')).toThrow(UnsupportedSftpPathError) + }) +}) + +describe('quoteSftpBatchArgument', () => { + it('escapes the backslashes in a Windows client local path', () => { + // Unescaped, sftp reads this as C:srcf.bin and fails to find the source. + expect(quoteSftpBatchArgument('C:\\src\\f.bin')).toBe('"C:\\\\src\\\\f.bin"') + }) + + it('keeps a path with spaces as one argument', () => { + expect(quoteSftpBatchArgument('/tmp/two words.bin')).toBe('"/tmp/two words.bin"') + }) + + it('escapes an embedded quote, which would otherwise end the argument early', () => { + expect(quoteSftpBatchArgument('/tmp/dq".bin')).toBe('"/tmp/dq\\".bin"') + }) + + it('refuses a line break, which would split one batch command into two', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\nrm -rf b')).toThrow(UnsupportedSftpPathError) + }) + + it('refuses a NUL, which truncates the argument', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\0b')).toThrow(UnsupportedSftpPathError) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-path.ts b/src/main/ssh/system-ssh-sftp-path.ts new file mode 100644 index 00000000000..2b5bfe53f02 --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.ts @@ -0,0 +1,46 @@ +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * A path this transfer cannot express to sftp. Callers treat it as "use another transport", never + * as a transfer failure. + */ +export class UnsupportedSftpPathError extends Error { + constructor(path: string) { + super(`Path cannot be addressed over sftp: ${JSON.stringify(path)}`) + this.name = 'UnsupportedSftpPathError' + } +} + +/** + * Converts a Windows remote path to the namespace OpenSSH's Windows sftp-server exposes, which + * roots every drive under `/`: `C:/Users/dev/f` is `/C:/Users/dev/f`, and `pwd` there reports + * `/C:/Users/dev`. + */ +export function toSftpRemotePath(remotePath: string): string { + const normalized = normalizeWindowsRemotePath(remotePath) + if (/^\/[a-zA-Z]:\//.test(normalized)) { + return normalized + } + if (/^[a-zA-Z]:\//.test(normalized)) { + return `/${normalized}` + } + // UNC (`//server/share`) and relative paths have no settled mapping in this namespace, and a + // guess here writes real bytes to the wrong place. Decline instead. + throw new UnsupportedSftpPathError(remotePath) +} + +/** + * Quotes one argument of an sftp batch line. + * + * Escaping is load-bearing, not cosmetic: sftp's batch lexer treats `\` as an escape even inside + * double quotes, so an unescaped Windows local path `C:\src\f.bin` is read as `C:srcf.bin`, and an + * unescaped destination `C:\Users\dev\f.bin` writes a file literally named `C` in the start + * directory — while sftp still exits 0. Both measured on Windows 11 / OpenSSH 10.0p2. + */ +export function quoteSftpBatchArgument(value: string): string { + if (/[\n\r\0]/.test(value)) { + // A line break would split one batch command into two; NUL truncates the argument. + throw new UnsupportedSftpPathError(value) + } + return `"${value.replace(/([\\"])/g, '\\$1')}"` +} diff --git a/src/main/ssh/system-ssh-sftp-transfer.ts b/src/main/ssh/system-ssh-sftp-transfer.ts new file mode 100644 index 00000000000..c50375fa8eb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-transfer.ts @@ -0,0 +1,191 @@ +import { accessSync, constants, existsSync, statSync } from 'node:fs' +import { posix, win32 } from 'node:path' +import type { SshTarget } from '../../shared/ssh-types' +import { buildSshArgs, type SystemSshBuildArgsOptions } from './system-ssh-args' +import { findSystemSsh } from './system-ssh-binary' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' +import { throwIfAborted } from './system-ssh-operation-lifecycle' +import { runProcess } from '../../shared/child-process/run-process' + +/** The host answered, but not with an sftp subsystem. The caller must fall back, not fail. */ +export class SftpSubsystemUnavailableError extends Error { + constructor(detail: string) { + super(`Remote host has no usable sftp subsystem: ${detail}`) + this.name = 'SftpSubsystemUnavailableError' + } +} + +/** + * True for the errors that mean "this host cannot serve sftp at all". + * + * Host-scoped, and therefore the only errors safe to remember: a capability cache keyed by host + * turns anything it accepts into a verdict about every later write to that host. Deliberately + * narrow — a permission denial or a missing directory is a real failure that must surface, not a + * reason to retry the whole upload down a slower path. + */ +export function isSftpUnavailableError(error: unknown): boolean { + return error instanceof SftpSubsystemUnavailableError || error instanceof SftpArgTranslationError +} + +/** + * True when *this path* cannot be spelled for sftp, which says nothing about the host. + * + * Kept apart from the host verdict on purpose. A UNC destination, or a local file whose name + * contains a newline, is a property of one operation; caching it would degrade every subsequent + * write to that host for the cache's whole retry window on the strength of one odd filename. + */ +export function isSftpPathUnsupportedError(error: unknown): boolean { + return error instanceof UnsupportedSftpPathError +} + +/** Neither kind of refusal moves a byte, so a staged file cannot exist to sweep. */ +export function isSftpRefusalBeforeStaging(error: unknown): boolean { + return isSftpUnavailableError(error) || isSftpPathUnsupportedError(error) +} + +function systemSftpCandidates(sshPath: string | null, platform: NodeJS.Platform): string[] { + const pathApi = platform === 'win32' ? win32 : posix + const executable = platform === 'win32' ? 'sftp.exe' : 'sftp' + const candidates: string[] = [] + // Why the ssh binary's own directory first: a host with two OpenSSH installs must pair the sftp + // client with the ssh that `buildSshArgs` was built for, not whichever one PATH happens to reach. + if (sshPath) { + candidates.push(pathApi.join(pathApi.dirname(sshPath), executable)) + } + if (platform === 'win32') { + const systemRoot = process.env.SystemRoot || process.env.WINDIR + if (systemRoot) { + candidates.push(win32.join(systemRoot, 'System32', 'OpenSSH', executable)) + } + } else { + candidates.push('/usr/bin/sftp', '/usr/local/bin/sftp', '/opt/homebrew/bin/sftp') + } + return candidates +} + +/** Locate the sftp client paired with the system ssh binary. Returns null when there is none. */ +export function findSystemSftp(): string | null { + if (process.env.ORCA_SYSTEM_SFTP_PATH) { + return process.env.ORCA_SYSTEM_SFTP_PATH + } + const sshPath = findSystemSsh() + for (const candidate of systemSftpCandidates(sshPath, process.platform)) { + try { + if (!statSync(candidate).isFile()) { + continue + } + if (process.platform !== 'win32') { + accessSync(candidate, constants.X_OK) + } + return candidate + } catch { + continue + } + } + return findSftpOnPath() +} + +function findSftpOnPath(): string | null { + const pathValue = process.env.PATH + if (!pathValue) { + return null + } + const pathApi = process.platform === 'win32' ? win32 : posix + const executable = process.platform === 'win32' ? 'sftp.exe' : 'sftp' + for (const entry of pathValue.split(pathApi.delimiter)) { + const directory = entry.trim().replace(/^"|"$/g, '') + if (!directory) { + continue + } + const candidate = pathApi.join(directory, executable) + if (existsSync(candidate)) { + return candidate + } + } + return null +} + +/** + * OpenSSH prints this when the server refuses the subsystem — a host with `Subsystem sftp` + * commented out, or an internal-sftp block that does not apply to this user. + */ +const SUBSYSTEM_REFUSED_PATTERN = /subsystem request failed|no such file or directory.*sftp-server/i + +export type SftpBatchOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal } + +/** + * Runs one sftp batch script. + * + * The script goes to the *local* sftp client's stdin, which is the point: no remote process ever + * reads a redirected stdin, so none of this rides the Windows PowerShell stdin defect. + */ +export async function runSftpBatch( + target: SshTarget, + commands: readonly string[], + options?: SftpBatchOptions +): Promise { + throwIfAborted(options?.signal) + const sftpPath = findSystemSftp() + if (!sftpPath) { + throw new SftpSubsystemUnavailableError('no sftp client binary found alongside ssh') + } + const args = withSftpKeepalive(translateSshArgsToSftpArgs(buildSshArgs(target, options))) + let result + try { + result = await runProcess({ + program: sftpPath, + args: ['-b', '-', ...args], + // `-b -` takes the script on stdin, and that stdin is the *local* client's — no remote + // process reads a pipe anywhere in this transfer, which is the whole point of preferring it. + input: `${commands.join('\n')}\n`, + // Why no timeout: a large upload is legitimately slow, and a wall-clock cap would fail a + // healthy transfer on a slow link. A dead peer is caught by the ServerAlive options instead. + timeoutMs: null, + signal: options?.signal + }) + } catch (error) { + // A client that will not start is "this host cannot do sftp" from the caller's side, not a + // transfer failure: the payload never left. Falling back is the only useful answer. + throw new SftpSubsystemUnavailableError( + `sftp client at ${sftpPath} could not be started: ${error instanceof Error ? error.message : String(error)}` + ) + } + if (result.code === 0) { + return + } + throwIfAborted(options?.signal) + const detail = result.stderr.trim() + if (SUBSYSTEM_REFUSED_PATTERN.test(detail)) { + throw new SftpSubsystemUnavailableError(detail) + } + throw new Error(`sftp batch failed (exit ${result.code}): ${detail}`) +} + +/** + * Creates remote directories, parents first. + * + * `-mkdir` keeps sftp going when a directory is already there; batch mode otherwise aborts the + * whole script on the first non-zero status, which for an idempotent tree walk is not a failure. + */ +export function makeDirectoriesViaSftp( + target: SshTarget, + remoteDirectories: readonly string[], + options?: SftpBatchOptions +): Promise { + const commands = remoteDirectories.map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + if (commands.length === 0) { + return Promise.resolve() + } + return runSftpBatch(target, commands, options) +} diff --git a/src/main/ssh/system-ssh-windows-file-write.ts b/src/main/ssh/system-ssh-windows-file-write.ts new file mode 100644 index 00000000000..8b98d4aec50 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-file-write.ts @@ -0,0 +1,138 @@ +import { randomBytes } from 'node:crypto' +import { powerShellCommand, powerShellLiteral } from './ssh-remote-powershell' +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * Suffix marking the path a Windows write lands on before it is published by rename. + * + * The random tail is the fix for a measured harm, not decoration. A write that loses contact with + * the host leaves a remote process that may still hold the staging file open exclusively, and + * `docs/reference/ssh-execution-boundary.md` is explicit that losing contact is not evidence that + * process died — so the retry must not reuse the name it may still own. A fresh name per attempt + * means a retry never meets its predecessor's lock; the abandoned file is cleaned up best-effort + * and never treated as proof of anything. + */ +export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' + +export function makeWindowsStagingPath(remotePath: string): string { + return `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}-${randomBytes(6).toString('hex')}` +} + +export type WindowsPublishMode = 'create' | 'exclusive' | 'append' + +/** + * Publishes a staged upload onto its real name. + * + * Every branch reads the staged *file*, never a redirected stdin, which is what makes this safe on + * a host whose Windows PowerShell 5.1 cannot drain a piped stdin. + * + * The replacing branch must never delete the destination first. Deleting and then moving loses the + * user's existing file outright if the move fails, and exposes a window where a reader sees no file + * at all — a worse outcome than the truncated-partial this staging discipline exists to prevent. + * `File.Replace` is the atomic swap (Win32 `ReplaceFile`), and it requires the destination to + * exist, so an absent one falls back to a plain `Move`. That fallback is raced deliberately: if the + * destination appears in between, `Move` throws, the staged file survives, and the destination is + * left exactly as whoever created it left it. + * + * `File::Move` throwing on an existing destination is also precisely the exclusive contract, which + * is why that branch needs nothing else. + */ +export function makeWindowsPublishStagedFileCommand( + stagingPath: string, + remotePath: string, + mode: WindowsPublishMode +): string { + const preamble = [ + '$ErrorActionPreference = "Stop"', + `$staging = ${powerShellLiteral(stagingPath)}`, + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }' + ] + if (mode === 'append') { + return powerShellCommand( + [ + ...preamble, + // Not atomic, and cannot cheaply be: appending is defined as extending the destination, so + // a failure part-way leaves it longer than it was rather than destroyed. The caller's + // chunked-append protocol already restarts from its own offset. + '$in = [System.IO.File]::OpenRead($staging)', + '$out = [System.IO.File]::Open($path, [System.IO.FileMode]::Append, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)', + 'try { $in.CopyTo($out) } finally { $out.Dispose(); $in.Dispose() }', + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) + } + if (mode === 'exclusive') { + return powerShellCommand([...preamble, '[System.IO.File]::Move($staging, $path)'].join('; ')) + } + return powerShellCommand( + [ + ...preamble, + // `[NullString]::Value`, not `$null`: PowerShell coerces a bare `$null` to an empty string + // when binding a .NET `string` parameter, and `Replace` rejects that with "The path is not + // of a legal form" — so every publish would fail. Measured on WindowsPowerShell 5.1.26100. + 'try { [System.IO.File]::Replace($staging, $path, [NullString]::Value) } catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ].join('; ') + ) +} + +/** Best-effort removal of a staged file whose write was abandoned. Never asserts the writer died. */ +export function makeWindowsDiscardStagedFileCommand(stagingPath: string): string { + return powerShellCommand( + [ + // Deliberately not `Stop`: the previous writer may still hold this file, and that is a + // possibility to tolerate, not an error to report. The unique staging name means a leftover + // blocks nothing; sweeping it is housekeeping. + '$ErrorActionPreference = "SilentlyContinue"', + `$staging = ${powerShellLiteral(stagingPath)}`, + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) +} + +/** + * The ancestor directories of a Windows remote path, drive root first. + * + * sftp's `mkdir` creates one level, so a batch has to name each level itself. The drive root is + * excluded: `-mkdir "/C:/"` is not a directory anyone creates. + */ +export function windowsRemoteAncestorDirectories(remotePath: string): string[] { + const normalized = normalizeWindowsRemotePath(remotePath) + const segments = normalized.split('/') + segments.pop() + const ancestors: string[] = [] + // Start past the drive (`C:`) or the UNC host, which are never created. + for (let depth = 2; depth <= segments.length; depth += 1) { + const directory = segments.slice(0, depth).join('/') + if (directory) { + ancestors.push(directory) + } + } + return ancestors +} + +/** + * `[Console]::OpenStandardInput()` into a `FileStream`, used only by the two stdin fallbacks. + * + * On Windows PowerShell 5.1 this is the defective read; see the strategy comment in + * `system-ssh-file-binary-transfer.ts`. It is correct under PowerShell 7. + */ +export function makeWindowsWriteFileCommand( + remotePath: string, + options?: { append?: boolean; exclusive?: boolean; executable?: 'powershell.exe' | 'pwsh.exe' } +): string { + const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' + return powerShellCommand( + [ + '$ErrorActionPreference = "Stop"', + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', + '$inputStream = [Console]::OpenStandardInput()', + `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, + 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' + ].join('; '), + options?.executable ?? 'powershell.exe' + ) +} diff --git a/src/main/ssh/system-ssh-windows-upload.test.ts b/src/main/ssh/system-ssh-windows-upload.test.ts index 207c3e2df8e..c3a5ff80276 100644 --- a/src/main/ssh/system-ssh-windows-upload.test.ts +++ b/src/main/ssh/system-ssh-windows-upload.test.ts @@ -1,29 +1,40 @@ /** - * #16432: the Windows relay upload pushed the whole bundle into one PowerShell stdin, which - * Windows PowerShell 5.1 cannot drain over a non-pty ssh exec — the remote blocks forever, and - * `waitForChannelClose()` had no timeout, so the UI sat at "Connecting…" with no error. Covered - * here: no write exceeds one stdin's worth on any Windows path (bundle upload *and* single-file - * upload, which is the one that carries large files), a partial write never lands under the real - * name, and a remote that never closes fails instead of hanging. + * #16432. The original fix chunked the payload because the constraint was believed to be a ~50KB + * cmd.exe stdin ceiling. Re-measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2, it is + * not a size limit and not cmd.exe's: a read on Windows PowerShell 5.1's redirected-stdin handle + * over a non-pty ssh exec can die permanently when it finds the stream momentarily empty, taking + * both the remaining data and the EOF with it. It is probabilistic per such read — identical 2MB + * payloads died at 167936, 270336 and 372736 — so a 32KB chunk still failed 15 times in 120 under + * load, while `findstr` took 2,016,000 bytes through one exec on the same host. + * + * So the covering property is no longer "every write is small". It is "the bytes do not cross a + * remote process's stdin at all": sftp first, PowerShell 7 next, and Windows PowerShell 5.1 last, + * bounded and loud. The staging-and-rename discipline is kept on every path, with a unique staging + * name per attempt so a retry never meets a predecessor's lock. */ import { EventEmitter } from 'node:events' import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' -import { rm } from 'node:fs/promises' +import { readFile, rm, stat } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { PassThrough, Writable } from 'node:stream' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as SystemSshOperationLifecycle from './system-ssh-operation-lifecycle' -const { spawnSystemSshCommandMock, waitForChannelCloseSpy } = vi.hoisted(() => ({ +const { spawnSystemSshCommandMock, waitForChannelCloseSpy, runProcessMock } = vi.hoisted(() => ({ spawnSystemSshCommandMock: vi.fn(), - waitForChannelCloseSpy: vi.fn() + waitForChannelCloseSpy: vi.fn(), + runProcessMock: vi.fn() })) vi.mock('./system-ssh-command', () => ({ spawnSystemSshCommand: spawnSystemSshCommandMock })) +vi.mock('../../shared/child-process/run-process', () => ({ + runProcess: runProcessMock +})) + // Delegates to the real implementation; the spy only records whether each wait was given a bound. vi.mock('./system-ssh-operation-lifecycle', async (importActual) => { const actual = (await importActual()) as typeof SystemSshOperationLifecycle @@ -41,6 +52,11 @@ import { } from './system-ssh-file-binary-transfer' import { waitForChannelClose } from './system-ssh-operation-lifecycle' import { getRemoteHostPlatform } from './ssh-remote-platform' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities +} from './system-ssh-windows-write-capabilities' +import { explainWindowsPowerShellStdinFailure } from './system-ssh-windows-write-strategy' import type { SshTarget } from '../../shared/ssh-types' type FakeChannel = EventEmitter & { @@ -50,7 +66,12 @@ type FakeChannel = EventEmitter & { written: Buffer } -const target = { id: 'win-1', host: 'win.example', username: 'dev' } as unknown as SshTarget +const target = { + id: 'win-1', + host: 'win.example', + username: 'dev', + port: 22 +} as unknown as SshTarget const hostPlatform = getRemoteHostPlatform('win32-x64') const remoteRoot = 'C:/Users/dev/.orca-remote' @@ -78,96 +99,238 @@ function createFakeChannel(onEnd: (channel: FakeChannel) => void): FakeChannel { return channel } -type RecordedCommand = { script: string; stdin: Buffer } +type RecordedCommand = { script: string; executable: string; stdin: Buffer } +type RecordedSftpBatch = { args: string[]; script: string } -describe('Windows upload stdin framing', () => { - let localDir: string - const commands: RecordedCommand[] = [] - /** Index of the spawn that should report a non-zero exit, to model a chunk failing mid-file. */ - let failAtSpawn = -1 +const sftpBatches: RecordedSftpBatch[] = [] +const commands: RecordedCommand[] = [] +/** Index of the exec that should report a non-zero exit, to model a chunk failing mid-file. */ +let failAtSpawn = -1 +let localDir: string - const fileWrites = (): RecordedCommand[] => - commands.filter((command) => command.script.includes('FileMode]::')) - const writtenPath = (command: RecordedCommand): string => - /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1].replace(/''/g, "'") ?? '' - const fileMode = (command: RecordedCommand): string | undefined => - /FileMode\]::(\w+)/.exec(command.script)?.[1] +const fileWrites = (): RecordedCommand[] => + commands.filter((command) => command.script.includes('OpenStandardInput')) +const writtenPath = (command: RecordedCommand): string => + /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1]?.replace(/''/g, "'") ?? '' +const fileMode = (command: RecordedCommand): string | undefined => + /FileMode\]::(\w+)/.exec(command.script)?.[1] +const putLines = (): string[] => + sftpBatches.flatMap((batch) => batch.script.split('\n').filter((line) => line.startsWith('put '))) +const putDestination = (line: string): string => /put "(?:[^"]*)" "([^"]*)"/.exec(line)?.[1] ?? '' +const putSource = (line: string): string => /put "([^"]*)"/.exec(line)?.[1] ?? '' - beforeEach(() => { - commands.length = 0 - failAtSpawn = -1 - waitForChannelCloseSpy.mockClear() - localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) - spawnSystemSshCommandMock.mockReset() - spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { - const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 - return createFakeChannel((channel) => { - commands.push({ script: decodePowerShellCommand(command), stdin: channel.written }) - setImmediate(() => - spawnIndex === failAtSpawn - ? channel.emit('close', 1, null) - : channel.emit('close', 0, null) - ) +/** Makes every sftp batch succeed, recording what it was asked to do. */ +function acceptSftp(): void { + runProcessMock.mockImplementation( + async (spec: { args: string[]; input: string; program: string }) => { + const script = spec.input + sftpBatches.push({ args: spec.args, script }) + // Model the real client: `put` copies the local file, so read it while it still exists. + for (const line of script.split('\n').filter((entry) => entry.startsWith('put '))) { + await readFile(putSource(line)) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + } + ) +} + +/** Models a host whose sshd has no `Subsystem sftp` line. */ +function refuseSftp(): void { + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + return { + code: 255, + signal: null, + stdout: '', + stderr: 'subsystem request failed on channel 0\nConnection closed', + timedOut: false + } + }) +} + +/** Models a host with no PowerShell 7, which cmd.exe reports as an unrecognized command. */ +function refusePwsh(): void { + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + const executable = command.split(' ')[0] ?? '' + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable, + stdin: channel.written + }) + setImmediate(() => { + if (executable === 'pwsh.exe') { + channel.stderr.write( + "'pwsh.exe' is not recognized as an internal or external command,\noperable program or batch file." + ) + channel.emit('close', 9009, null) + return + } + channel.emit('close', spawnIndex === failAtSpawn ? 1 : 0, null) }) }) }) +} - afterEach(async () => { - await rm(localDir, { recursive: true, force: true }) +beforeEach(() => { + commands.length = 0 + sftpBatches.length = 0 + failAtSpawn = -1 + clearWindowsRemoteWriteCapabilitiesForTests() + waitForChannelCloseSpy.mockClear() + localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) + process.env.ORCA_SYSTEM_SFTP_PATH = '/usr/bin/sftp' + runProcessMock.mockReset() + acceptSftp() + spawnSystemSshCommandMock.mockReset() + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable: command.split(' ')[0] ?? '', + stdin: channel.written + }) + setImmediate(() => + spawnIndex === failAtSpawn ? channel.emit('close', 1, null) : channel.emit('close', 0, null) + ) + }) }) +}) - it('never pushes a whole artifact bundle into one PowerShell stdin', async () => { - mkdirSync(join(localDir, 'node'), { recursive: true }) - // Comfortably past the ~50KB point at which the reporter measured PowerShell 5.1 wedging. - writeFileSync(join(localDir, 'node', 'relay.js'), Buffer.alloc(600 * 1024, 0x61)) - writeFileSync(join(localDir, 'index.js'), Buffer.alloc(300 * 1024, 0x62)) +afterEach(async () => { + delete process.env.ORCA_SYSTEM_SFTP_PATH + await rm(localDir, { recursive: true, force: true }) +}) - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - - const largest = Math.max(...commands.map((command) => command.stdin.length)) - expect(largest).toBeLessThanOrEqual(WINDOWS_STDIN_WRITE_CHUNK_BYTES) - // The base64 + JSON envelope is gone entirely: nothing reads the bundle as one string. - expect(commands.some((command) => command.script.includes('FromBase64String'))).toBe(false) - // `[Console]::In` wedged at 50KB where the stream reader did not, so the mkdir batch — the one - // payload still read as a string — must use the reader the reporter measured surviving. - expect(commands.some((command) => command.script.includes('[Console]::In.ReadToEnd()'))).toBe( - false - ) - expect( - commands.filter((command) => command.script.includes('StreamReader([Console]::')) - ).toHaveLength(1) - }) - - it('bounds the single-file upload too, which is the path large files take', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) +describe('Windows upload over sftp', () => { + it('moves the payload without any remote process reading a stdin', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 60 + 11, 0x64) const localPath = join(localDir, 'big.node') writeFileSync(localPath, contents) await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/big.node`, { hostPlatform }) - const writes = fileWrites() - expect(writes).toHaveLength(4) - expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( - WINDOWS_STDIN_WRITE_CHUNK_BYTES - ) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. - expect( - waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ).toBe(true) + // The defect is a remote stdin read; the fix is that there is not one. + expect(fileWrites()).toHaveLength(0) + expect(putLines()).toHaveLength(1) + // One transfer, not 61 execs: the whole point of the change. + expect(sftpBatches).toHaveLength(1) }) - it('writes every byte of every artifact across the chunked writes', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 2 + 17, 0x63) - writeFileSync(join(localDir, 'relay.js'), contents) + it('creates the parent chain and sends the payload in one round trip', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/a/b/relay.js`, { + hostPlatform + }) - const writes = fileWrites() - expect(writes).toHaveLength(3) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // Only the first write creates the staging file; the rest must extend it or it is truncated. - expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append']) + expect(sftpBatches).toHaveLength(1) + expect(sftpBatches[0]!.script.split('\n').filter(Boolean)).toEqual([ + '-mkdir "/C:/Users"', + '-mkdir "/C:/Users/dev"', + '-mkdir "/C:/Users/dev/.orca-remote"', + '-mkdir "/C:/Users/dev/.orca-remote/a"', + '-mkdir "/C:/Users/dev/.orca-remote/a/b"', + expect.stringContaining('put ') as unknown as string + ]) + }) + + it('addresses the destination in the drive-rooted namespace sftp exposes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A backslash destination silently writes a file named `C` and still exits 0, so the leading + // slash and forward separators are correctness, not style. + expect(putDestination(putLines()[0]!)).toMatch( + /^\/C:\/Users\/dev\/\.orca-remote\/relay\.js\.orca-partial-[0-9a-f]{12}$/ + ) + }) + + it('never lands a partial under the real name, and publishes by rename', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const remotePath = `${remoteRoot}/relay.js` + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), remotePath, { hostPlatform }) + + const destination = putDestination(putLines()[0]!) + // Assert the positive first: an unmatched regex yields '', which would satisfy the `not.toBe` + // below without this test ever having seen a destination. + expect(destination).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + expect(destination).not.toBe(`/C:${remotePath.slice(2)}`) + const publish = commands.at(-1)! + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // The publish reads the staged file, never a pipe, so it is safe on PowerShell 5.1. + expect(publish.script).not.toContain('OpenStandardInput') + }) + + it('never deletes the destination it is replacing', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const publish = commands.at(-1)! + // Delete-then-move destroys the user's existing file outright if the move then fails, and + // exposes a window where a reader sees no file at all — worse than the truncated partial the + // staging discipline exists to prevent. `File.Replace` is the atomic swap. + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // An absent destination cannot be Replaced, so that case falls back to a plain Move. + expect(publish.script).toContain( + 'catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ) + }) + + it('gives every attempt its own staging name, so a retry cannot meet a predecessor lock', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const [first, second] = putLines().map(putDestination) + expect(first).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + // Losing contact is not evidence the previous writer died, so the name must not be reused. + expect(second).not.toBe(first) + }) + + it('enforces exclusive at the rename, where it is atomic', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + }) + + it('appends by concatenating the staged file, not by piping bytes to the remote', async () => { + await writeBufferViaSystemSsh(target, `${remoteRoot}/log.bin`, Buffer.from('tail'), { + hostPlatform, + append: true + }) + + expect(fileWrites()).toHaveLength(0) + const publish = commands.at(-1)! + expect(publish.script).toContain('FileMode]::Append') + expect(publish.script).toContain('$in.CopyTo($out)') + expect(publish.script).toContain('[System.IO.File]::Delete($staging)') }) it('still creates an empty artifact on the host', async () => { @@ -175,80 +338,310 @@ describe('Windows upload stdin framing', () => { await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - expect(fileWrites().map(writtenPath)).toEqual([`${remoteRoot}/empty.txt`]) - expect(fileWrites()[0].stdin).toHaveLength(0) - expect(fileMode(fileWrites()[0])).toBe('Create') + expect(putLines()).toHaveLength(1) + expect(commands.at(-1)!.script).toContain('[System.IO.File]::Move($staging, $path)') }) - it('lands a multi-chunk write on a staging path and publishes it by rename', async () => { - const remotePath = `${remoteRoot}/relay.js` - writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + it('writes a buffer through a 0600 temp file that does not outlive the transfer', async () => { + const seen: { path: string; contents: Buffer; mode: number }[] = [] + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + for (const line of spec.input.split('\n').filter((entry) => entry.startsWith('put '))) { + const path = putSource(line) + seen.push({ + path, + contents: await readFile(path), + mode: (await stat(path)).mode & 0o777 + }) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + }) + + await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { + hostPlatform + }) + + expect(seen).toHaveLength(1) + expect(seen[0]!.contents.toString()).toBe('1.2.3') + // The payload can be repository content and tmpdir is world-readable on every platform, so the + // window between write and upload must not be group- or world-readable. + expect(seen[0]!.mode).toBe(0o600) + await expect(readFile(seen[0]!.path)).rejects.toThrow() + }) + + it('creates upload directories over sftp rather than a PowerShell stdin batch', async () => { + mkdirSync(join(localDir, 'node'), { recursive: true }) + writeFileSync(join(localDir, 'node', 'relay.js'), 'x') await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - // Nothing touches the real name until every byte is on the host. - expect(fileWrites().map(writtenPath)).toEqual([ - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`, - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` - ]) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).toContain('[System.IO.File]::Delete($path)') + // Anchor on a non-empty observation: `some` is false of an empty list, so this would pass even + // if no command had been recorded at all. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('StreamReader([Console]::'))).toBe( + false + ) + expect(sftpBatches[0]!.script).toContain('-mkdir "/C:/Users/dev/.orca-remote"') + }) + + it('sweeps the staged bytes when the publish is the thing that fails', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + // An exclusive conflict is the ordinary way to get here: the payload is on the host, and the + // rename that would have given it a name refuses. + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const script = decodePowerShellCommand(command) + return createFakeChannel((channel) => { + commands.push({ script, executable: command.split(' ')[0] ?? '', stdin: channel.written }) + const failed = script.includes('::Move($staging, $path)') + setImmediate(() => channel.emit('close', failed ? 1 : 0, null)) + }) + }) + + await expect( + uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + ).rejects.toThrow() + + const sweep = commands.at(-1)! + expect(sweep.script).toContain('[System.IO.File]::Delete($staging)') + // Tolerated, not asserted: the previous writer may still hold the file, and losing contact is + // not evidence it died. + expect(sweep.script).toContain('$ErrorActionPreference = "SilentlyContinue"') + }) + + it('reports a cancelled transfer as an abort, not as a failed one', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const controller = new AbortController() + // runProcess reports the kill as a non-zero exit rather than throwing, so without checking the + // signal first a user pressing cancel is indistinguishable from the transfer genuinely failing. + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + controller.abort() + return { code: 255, signal: 'SIGTERM', stdout: '', stderr: '', timedOut: false } + }) + + let error: Error | undefined + try { + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + signal: controller.signal + }) + } catch (thrown) { + error = thrown as Error + } + + expect(error?.name).toBe('AbortError') + expect(error?.message).not.toContain('sftp batch failed') + // A cancel is also not evidence about the host, so it must not send later writes to the slow + // path, and must not fall through to the defective reader now. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + expect(fileWrites()).toHaveLength(0) + }) + + it('does not let one unaddressable path become a verdict about the host', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + // A UNC destination has no settled mapping in sftp's drive-rooted namespace, so this write + // falls back — but the host still serves sftp perfectly well for every other path. + await uploadFileViaSystemSsh( + target, + join(localDir, 'relay.js'), + '//fileserver/share/relay.js', + { hostPlatform } + ) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(sftpBatches).toHaveLength(0) + // The 30-minute capability cache is keyed by host; caching this would send every later write + // to the same machine down the defective path on the strength of one odd destination. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps using sftp for the next file after one path it could not spell', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), '//fileserver/share/a.js', { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/b.js`, { + hostPlatform + }) + + expect(putLines()).toHaveLength(1) + expect(putDestination(putLines()[0]!)).toContain('/C:/Users/dev/.orca-remote/b.js') + }) + + it('does not let a local filename sftp cannot quote become a verdict either', async () => { + // POSIX clients allow a newline in a filename, and sftp's batch lexer would read it as the end + // of one command and the start of another. + const awkward = join(localDir, 'two\nlines.js') + writeFileSync(awkward, 'x') + + await uploadFileViaSystemSsh(target, awkward, `${remoteRoot}/relay.js`, { hostPlatform }) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('translates the ssh argument list rather than passing it to a client that reads it differently', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + disableControlMaster: true + }) + + const args = sftpBatches[0]!.args + // sftp's `-T` does not exist, its `-p` preserves mtime, and its `-S` names a program to run. + expect(args).not.toContain('-T') + expect(args).not.toContain('-p') + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') + expect(args).toContain('ServerAliveInterval=15') + }) +}) + +describe('Windows upload on a host with no sftp subsystem', () => { + beforeEach(() => { + refuseSftp() + }) + + it('creates a multi-directory tree, which the one-element case never exercised', async () => { + mkdirSync(join(localDir, 'node', 'deep'), { recursive: true }) + writeFileSync(join(localDir, 'index.js'), 'a') + writeFileSync(join(localDir, 'node', 'deep', 'x.js'), 'b') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const mkdir = commands.find((command) => command.script.includes('ConvertFrom-Json'))! + // `@($json | ConvertFrom-Json)` wraps the parsed array in another array, so the loop variable + // binds to the whole thing and `[string]` of it is the paths joined by spaces — which + // CreateDirectory rejects. It only ever worked for a single directory, where stringifying a + // one-element array happens to yield the element, so no batch of one can catch this. + expect(mkdir.script).toContain('[string[]]($json | ConvertFrom-Json)') + expect(mkdir.script).not.toContain('@($json | ConvertFrom-Json)') + const batch = JSON.parse(mkdir.stdin.toString('utf-8')) as string[] + expect(batch.length).toBeGreaterThan(1) + }) + + it('falls back rather than failing the transfer', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 5, 0x61) + writeFileSync(join(localDir, 'relay.js'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(Buffer.concat(fileWrites().map((write) => write.stdin)).equals(contents)).toBe(true) + }) + + it('remembers the refusal, so a multi-file upload probes once', async () => { + writeFileSync(join(localDir, 'a.js'), 'a') + writeFileSync(join(localDir, 'b.js'), 'b') + writeFileSync(join(localDir, 'c.js'), 'c') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + // One refusal is enough; re-probing per file is a wasted round trip on every file. + expect(sftpBatches).toHaveLength(1) + }) + + it('does not spend a sweep round trip when sftp declined before moving any bytes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A refused subsystem staged nothing, so there is nothing to delete — and on a host without + // sftp that sweep would otherwise be paid on every single write. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('Delete($staging)'))).toBe(false) + }) + + it('prefers PowerShell 7, which reads a redirected stdin correctly', async () => { + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(fileWrites().map((write) => write.executable)).toEqual(['pwsh.exe']) + // PowerShell 7 took 2MB through one exec when measured, so chunking it buys nothing. + expect(fileWrites()[0]!.stdin).toHaveLength(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3) + }) + + it('bounds every write when only Windows PowerShell 5.1 is available', async () => { + refusePwsh() + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) + writeFileSync(join(localDir, 'big.node'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + const writes = fileWrites().filter((write) => write.executable === 'powershell.exe') + expect(writes).toHaveLength(4) + expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( + WINDOWS_STDIN_WRITE_CHUNK_BYTES + ) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append', 'Append']) + // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. + // Count first: `every` is true of zero calls, so a wait that moved to a different helper would + // pass this silently. + expect(waitForChannelCloseSpy.mock.calls.length).toBeGreaterThan(0) + expect( + waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ).toBe(true) + }) + + it('remembers that PowerShell 7 is absent instead of re-probing per chunk', async () => { + refusePwsh() + writeFileSync(join(localDir, 'big.node'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + expect(fileWrites().filter((write) => write.executable === 'pwsh.exe')).toHaveLength(1) }) it('leaves no truncated file under the real name when a chunk fails mid-file', async () => { writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) - // Spawns: 0 = mkdir batch, 1..3 = chunk writes. Fail the second chunk. - failAtSpawn = 2 + // Spawn 0 is the pwsh write; fail it and every retry beneath it. + failAtSpawn = 0 await expect( - uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) ).rejects.toThrow() + expect(fileWrites().length).toBeGreaterThan(0) expect(fileWrites().map(writtenPath)).not.toContain(`${remoteRoot}/relay.js`) expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) }) +}) - it('enforces exclusive once at the rename, so a retry is not blocked by its own leftovers', async () => { - const localPath = join(localDir, 'import.bin') - writeFileSync(localPath, Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) +describe('last-resort Windows PowerShell failure reporting', () => { + it('names the host limitation and its remedy, not just the timeout', () => { + const timeout = new Error('write C:/x at offset 0 timed out after 60000ms with no response') - await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/import.bin`, { - hostPlatform, - exclusive: true - }) + const explained = explainWindowsPowerShellStdinFailure(timeout) as Error - // CreateNew on chunk one would fail against a leftover staging file from a failed attempt; - // `File::Move` raising on an existing destination is what carries the exclusive contract. - expect(fileWrites().map(fileMode)).toEqual(['Create', 'Append']) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + // "timed out" alone sends the user to retry a network they cannot fix; the fix is host-side. + expect(explained.message).toContain('Windows PowerShell 5.1') + expect(explained.message).toContain('Subsystem sftp sftp-server.exe') + expect(explained.cause).toBe(timeout) }) - it('keeps a single-chunk write on the destination, with the caller mode intact', async () => { - await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { - hostPlatform, - exclusive: true - }) + it('leaves a real failure alone, so a permission error is not reported as a host limitation', () => { + const denied = new Error('write C:/x at offset 0 failed (exit 1): Access to the path is denied') - expect(fileWrites()).toHaveLength(1) - expect(writtenPath(fileWrites()[0])).toBe(`${remoteRoot}/version`) - expect(fileMode(fileWrites()[0])).toBe('CreateNew') - expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) - }) - - it('appends onto the destination rather than staging, since append cannot be staged', async () => { - const remotePath = `${remoteRoot}/log.bin` - await writeBufferViaSystemSsh( - target, - remotePath, - Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1), - { hostPlatform, append: true } - ) - - expect(fileWrites().map(writtenPath)).toEqual([remotePath, remotePath]) - expect(fileWrites().map(fileMode)).toEqual(['Append', 'Append']) + expect(explainWindowsPowerShellStdinFailure(denied)).toBe(denied) }) }) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.test.ts b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts new file mode 100644 index 00000000000..ad723d592b0 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts @@ -0,0 +1,79 @@ +/** + * Whether a Windows host has an sftp subsystem is a fact about that host, so the cache is keyed by + * the endpoint that executes rather than by Orca's target id — otherwise a hardened host is + * re-probed once per file, and two targets pointing at one machine learn the same fact twice. + */ +import { afterEach, describe, expect, it } from 'vitest' +import type { SshTarget } from '../../shared/ssh-types' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities, + getWindowsRemoteWriteExecutionHostKey +} from './system-ssh-windows-write-capabilities' + +const asTarget = (fields: Partial): SshTarget => fields as SshTarget + +afterEach(() => { + clearWindowsRemoteWriteCapabilitiesForTests() +}) + +describe('getWindowsRemoteWriteExecutionHostKey', () => { + it('gives two targets on one endpoint the same key', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + // A target re-created under a new id has not changed what the host supports. + expect(getWindowsRemoteWriteExecutionHostKey(first)).toBe( + getWindowsRemoteWriteExecutionHostKey(second) + ) + }) + + it('separates hosts, ports and users', () => { + const base = { id: 'a', host: 'win.example', username: 'dev', port: 22 } + const keys = [ + asTarget(base), + asTarget({ ...base, host: 'other.example' }), + asTarget({ ...base, port: 2222 }), + asTarget({ ...base, username: 'ops' }) + ].map(getWindowsRemoteWriteExecutionHostKey) + + expect(new Set(keys).size).toBe(4) + }) + + it('keys a config alias by the alias, since ssh_config decides where it lands', () => { + const alias = asTarget({ id: 'a', host: 'stale.example', configHost: 'winbox' }) + + expect(getWindowsRemoteWriteExecutionHostKey(alias)).toBe('config:winbox') + }) +}) + +describe('getWindowsRemoteWriteCapabilities', () => { + it('shares one cache across targets that reach the same host', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(first).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(second).shouldTry('sftp-subsystem')).toBe(false) + }) + + it('does not let one host answer for another', () => { + const hardened = asTarget({ id: 'a', host: 'hardened.example', username: 'dev', port: 22 }) + const ordinary = asTarget({ id: 'b', host: 'ordinary.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(hardened).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(ordinary).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps the two capabilities independent', () => { + const target = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const capabilities = getWindowsRemoteWriteCapabilities(target) + + capabilities.rememberUnsupported('pwsh') + + // No PowerShell 7 says nothing about whether the host will serve sftp. + expect(capabilities.shouldTry('sftp-subsystem')).toBe(true) + expect(capabilities.shouldTry('pwsh')).toBe(false) + }) +}) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.ts b/src/main/ssh/system-ssh-windows-write-capabilities.ts new file mode 100644 index 00000000000..dcd03f19807 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.ts @@ -0,0 +1,52 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { CapabilityProbeCache } from '../../shared/capability-probe-cache' + +/** + * Whether a Windows host can take a file write over the sftp subsystem, and whether it has a + * PowerShell 7 to fall back to. Both are host facts, so they are cached per execution host rather + * than per transfer — a hardened host with `Subsystem sftp` removed must not be re-probed on every + * file of a multi-file upload. + */ +export type WindowsRemoteWriteCapability = 'sftp-subsystem' | 'pwsh' + +// Why re-probe at all: an admin can enable the subsystem, or install PowerShell 7, without the +// user restarting Orca. Long enough that a hardened host costs one failed probe per half hour. +export const WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS = 30 * 60_000 + +const capabilitiesByExecutionHost = new Map< + string, + CapabilityProbeCache +>() + +/** + * Keyed by the endpoint that executes, not by target id: two Orca targets pointing at one host + * describe the same sshd, and a target re-created under a new id has not changed what that host + * supports. A config alias is its own key because ssh_config, not Orca, resolves where it lands. + */ +export function getWindowsRemoteWriteExecutionHostKey(target: SshTarget): string { + if (target.configHost) { + return `config:${target.configHost}` + } + const port = target.port ?? 22 + return target.username + ? `host:${target.username}@${target.host}:${port}` + : `host:${target.host}:${port}` +} + +export function getWindowsRemoteWriteCapabilities( + target: SshTarget +): CapabilityProbeCache { + const key = getWindowsRemoteWriteExecutionHostKey(target) + let cache = capabilitiesByExecutionHost.get(key) + if (!cache) { + cache = new CapabilityProbeCache( + WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS + ) + capabilitiesByExecutionHost.set(key, cache) + } + return cache +} + +export function clearWindowsRemoteWriteCapabilitiesForTests(): void { + capabilitiesByExecutionHost.clear() +} diff --git a/src/main/ssh/system-ssh-windows-write-strategy.ts b/src/main/ssh/system-ssh-windows-write-strategy.ts new file mode 100644 index 00000000000..f2cdca12516 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-strategy.ts @@ -0,0 +1,329 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { getSystemSshBuildArgsFromOperationOptions } from './system-ssh-args' +import { spawnSystemSshCommand } from './system-ssh-command' +import { + awaitWithSystemSshAbort, + throwIfAborted, + waitForChannelClose +} from './system-ssh-operation-lifecycle' +import { + isSftpPathUnsupportedError, + isSftpRefusalBeforeStaging, + isSftpUnavailableError, + runSftpBatch +} from './system-ssh-sftp-transfer' +import { quoteSftpBatchArgument, toSftpRemotePath } from './system-ssh-sftp-path' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' +import { + makeWindowsDiscardStagedFileCommand, + makeWindowsPublishStagedFileCommand, + makeWindowsStagingPath, + makeWindowsWriteFileCommand, + windowsRemoteAncestorDirectories, + type WindowsPublishMode +} from './system-ssh-windows-file-write' + +/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ +export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 + +/** + * Bound on one stdin write for the last-resort Windows PowerShell 5.1 path. + * + * Measured on Windows 11 26200 / OpenSSH 10.0p2: a 32KB write still hangs 15 times in 120 under + * load, and no smaller value removes the risk. The defect is per blocking read, not per byte, so + * shrinking the chunk trades one risky read for more execs that each carry their own. This is a + * damage bound on a path known to be unreliable, not a safe size. + */ +export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 + +export type WindowsWriteOptions = Parameters< + typeof getSystemSshBuildArgsFromOperationOptions +>[0] & { + signal?: AbortSignal + append?: boolean + exclusive?: boolean +} + +/** Bytes to write, plus a way to present them to sftp, which can only send a local file. */ +export type WindowsWriteSource = { + totalBytes: number + readChunk: (offset: number, maxBytes: number) => Promise + withLocalFile: (send: (localPath: string) => Promise) => Promise +} + +function publishMode(options: WindowsWriteOptions): WindowsPublishMode { + return options.append ? 'append' : options.exclusive === true ? 'exclusive' : 'create' +} + +/** + * Writes one file to a Windows host, preferring transports that do not push bytes through a remote + * PowerShell's stdin. + * + * Order, and why: sftp carries the whole payload in one transfer and never has a remote process + * read a pipe. Measured on Windows 11 / OpenSSH 10.0p2: 1.9MB in a median 315ms over sftp against + * 0 of 6 completions on the chunked path, whose best case was ~62 execs at ~350ms each. PowerShell + * 7 reads a redirected stdin correctly but is not installed by default. Windows PowerShell 5.1 is + * always present and is the defective reader, so it is last and it is bounded. + * + * Every transport stages under a unique name and publishes by rename, so no partial write is ever + * visible under the real name and no retry inherits a predecessor's lock. + */ +export async function writeWindowsRemoteFile( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + throwIfAborted(options.signal) + const capabilities = getWindowsRemoteWriteCapabilities(target) + await capabilities.runWithFallback( + 'sftp-subsystem', + () => writeViaSftp(target, remotePath, source, options), + () => writeViaRemoteStdin(target, remotePath, source, options), + isSftpUnavailableError + ) +} + +/** + * Stages under a name nothing else can own, publishes it, and sweeps the staging file if either + * step fails. + * + * Shared by both transports so the cleanup contract cannot drift between them: a failed publish — + * an exclusive conflict is the ordinary case — leaves bytes on the host that no longer have a + * purpose, and the sweep is what stops them accumulating. + */ +async function stageThenPublish( + target: SshTarget, + remotePath: string, + options: WindowsWriteOptions, + stage: (stagingPath: string) => Promise, + nothingStaged: (error: unknown) => boolean = () => false +): Promise { + const stagingPath = makeWindowsStagingPath(remotePath) + try { + await stage(stagingPath) + await publishStagedWrite(target, stagingPath, remotePath, options) + } catch (error) { + // A transport that declined before it moved any bytes has nothing to sweep, and sweeping + // anyway would spend a round trip on every write to a host that has no sftp subsystem. + if (!nothingStaged(error)) { + await discardStagedWrite(target, stagingPath, options) + } + throw error + } +} + +/** + * A path sftp cannot address falls back for this write alone, without touching the host verdict. + * + * The distinction matters because the capability cache is keyed by host and holds for half an hour: + * routing one UNC destination, or one local filename containing a newline, into + * `rememberUnsupported` would send every later write to that host down the defective path too. + */ +async function writeViaSftp( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + try { + await attemptSftpWrite(target, remotePath, source, options) + } catch (error) { + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await writeViaRemoteStdin(target, remotePath, source, options) + } +} + +function attemptSftpWrite( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const mkdirs = windowsRemoteAncestorDirectories(remotePath).map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + return stageThenPublish( + target, + remotePath, + options, + (stagingPath) => + source.withLocalFile((localPath) => + // One round trip: the parent chain and the payload travel in the same batch. + runSftpBatch( + target, + [ + ...mkdirs, + `put ${quoteSftpBatchArgument(localPath)} ${quoteSftpBatchArgument(toSftpRemotePath(stagingPath))}` + ], + options + ) + ), + isSftpRefusalBeforeStaging + ) +} + +function writeViaRemoteStdin( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const capabilities = getWindowsRemoteWriteCapabilities(target) + return stageThenPublish(target, remotePath, options, (stagingPath) => + capabilities.runWithFallback( + 'pwsh', + () => writeStdinChunks(target, stagingPath, source, options, 'pwsh.exe'), + () => writeStdinChunks(target, stagingPath, source, options, 'powershell.exe'), + isPwshUnavailableError + ) + ) +} + +/** + * PowerShell 7 takes the whole payload in one exec — measured at 2MB — so only the 5.1 path pays + * for chunking, and only because a bounded write is the most that path can be trusted with. + */ +async function writeStdinChunks( + target: SshTarget, + stagingPath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + const chunkBytes = + executable === 'pwsh.exe' ? Math.max(source.totalBytes, 1) : WINDOWS_STDIN_WRITE_CHUNK_BYTES + let offset = 0 + // An empty write still has to run: it is what creates the staged file. + do { + const chunk = await source.readChunk(offset, chunkBytes) + if (chunk.length === 0 && offset < source.totalBytes) { + throw new Error(`Source ran short during upload of ${stagingPath}`) + } + await writeOneStdinChunk( + target, + stagingPath, + chunk, + { ...options, append: offset > 0, exclusive: false }, + offset, + executable + ) + offset += chunk.length + } while (offset < source.totalBytes) +} + +async function writeOneStdinChunk( + target: SshTarget, + stagingPath: string, + chunk: Buffer, + options: WindowsWriteOptions, + offset: number, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand( + target, + makeWindowsWriteFileCommand(stagingPath, { + append: options.append, + exclusive: options.exclusive, + executable + }), + { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } + ) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose( + channel, + `write ${stagingPath} at offset ${offset}`, + WINDOWS_STDIN_WRITE_TIMEOUT_MS + ) + ).catch((error: unknown) => { + throw executable === 'powershell.exe' ? explainWindowsPowerShellStdinFailure(error) : error + }) + if (!options.signal?.aborted) { + channel.stdin.end(chunk) + } + await closePromise +} + +/** + * Names the cause on the one path that can hang, so the failure is not just "timed out". + * + * A user seeing this needs to know it is a host limitation with a host-side remedy, not a network + * fault they should retry into. + */ +export function explainWindowsPowerShellStdinFailure(error: unknown): unknown { + const message = error instanceof Error ? error.message : String(error) + if (!/timed out/i.test(message)) { + return error + } + return new Error( + `${message}\nWindows PowerShell 5.1 can lose a redirected stdin permanently when a read finds it momentarily empty, so this write cannot be made reliable from the client. Enable the sftp subsystem on the host (sshd_config: "Subsystem sftp sftp-server.exe"), or install PowerShell 7, and Orca will use it automatically.`, + { cause: error instanceof Error ? error : undefined } + ) +} + +function isPwshUnavailableError(error: unknown): boolean { + const message = error instanceof Error ? error.message : String(error) + // cmd.exe's "not recognized" and sshd's exit 9009 both mean "no pwsh here". A timeout does not: + // that is the stdin defect, and PowerShell 7 does not have it, so it must not be cached as absent. + return /is not recognized as an internal or external command|9009|CommandNotFoundException/i.test( + message + ) +} + +async function publishStagedWrite( + target: SshTarget, + stagingPath: string, + remotePath: string, + options: WindowsWriteOptions +): Promise { + await runWindowsCommandWithoutStdin( + target, + makeWindowsPublishStagedFileCommand(stagingPath, remotePath, publishMode(options)), + `publish ${remotePath}`, + options + ) +} + +async function discardStagedWrite( + target: SshTarget, + stagingPath: string, + options: WindowsWriteOptions +): Promise { + try { + await runWindowsCommandWithoutStdin( + target, + makeWindowsDiscardStagedFileCommand(stagingPath), + `discard ${stagingPath}`, + { ...options, signal: undefined } + ) + } catch { + // Housekeeping only. The staging name is unique, so a leftover blocks nothing, and a failure + // here says nothing about whether the abandoned writer is still alive. + } +} + +function runWindowsCommandWithoutStdin( + target: SshTarget, + command: string, + label: string, + options: WindowsWriteOptions +): Promise { + const channel = spawnSystemSshCommand(target, command, { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, label, WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end() + } + return closePromise +} diff --git a/src/main/startup/main-process-ready-runtime.ts b/src/main/startup/main-process-ready-runtime.ts index 75784a4c196..26b920652df 100644 --- a/src/main/startup/main-process-ready-runtime.ts +++ b/src/main/startup/main-process-ready-runtime.ts @@ -9,7 +9,7 @@ import { RpcDispatcher } from '../runtime/rpc/dispatcher' import { browserManager } from '../browser/browser-manager' import { configureBrowserClientPageAutomationRuntime } from '../browser/browser-client-page-automation-runtime' import { BrowserClientPageCommandError } from '../browser/browser-client-page-command-failure' -import { startPreGoneProcessMetricsSampling } from '../crash-reporting/process-gone-diagnostics' +import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics' import { recordProcessGoneCrash } from './main-window-lifecycle-flags' import { handleGpuChildCrash } from './gpu-lifecycle' import { isGpuFallbackCrashCandidate } from '../crash-reporting/gpu-crash-fallback-decision' @@ -130,9 +130,10 @@ export async function initializeReadyRuntimeServices(): Promise { console.warn('[agent-hooks] failed to reconcile managed hooks on startup:', error) ) } - // Why: process-gone metrics only see survivors; retain a recent whole-app - // snapshot for comparison in crash reports. - startPreGoneProcessMetricsSampling() + // Why: process-gone metrics only see survivors, and the gone-time host memory + // read lands after the corpse released its pages; both need a live pre-gone + // sample to compare against in crash reports. + startPreGoneCrashSampling() app.on('child-process-gone', (_event, details) => { recordProcessGoneCrash('child', details.type, details.reason, details.exitCode ?? null, { name: details.name, diff --git a/src/main/startup/pre-gone-crash-sampling-wiring.test.ts b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts new file mode 100644 index 00000000000..2a8e008c0b3 --- /dev/null +++ b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts @@ -0,0 +1,49 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * Guards the one line that arms pre-gone crash sampling. + * + * That branch is pure instrumentation, so this line is the whole of its value in + * the shipped app: deleting it left all 691 tests across `src/main/crash-reporting/` + * and `src/main/startup/` green while every crash report silently lost its only + * host reading taken before the dying process returned its pages. + * + * Source-level because that is the property: the sampler is armed once inside the + * ready-phase composition, which has no runtime seam to assert against. + */ +describe('pre-gone crash sampling startup wiring', () => { + // Why normalize: the indent anchors below are `\n`-prefixed, and nothing pins + // src/**/*.ts to LF, so a CRLF Windows checkout would fail them spuriously. + const readSource = (name: string): string => + readFileSync(join(process.cwd(), 'src/main/startup', name), 'utf8').replace(/\r\n/g, '\n') + + const readyRuntimeSource = readSource('main-process-ready-runtime.ts') + const readySource = readSource('main-process-ready.ts') + + const READY_ENTRY = 'export async function initializeReadyRuntimeServices(' + // Why the entry's body and not the file: the call satisfies a whole-file grep + // just as well from a sibling export nothing calls, which arms nothing. + const readyRuntimeEntryBody = readyRuntimeSource + .slice(readyRuntimeSource.indexOf(READY_ENTRY) + READY_ENTRY.length) + .split('\nexport ')[0] + + it('arms the sampler unconditionally inside the function app readiness runs', () => { + expect(readyRuntimeSource).toContain( + "import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics'" + ) + expect(readyRuntimeSource).toContain(READY_ENTRY) + expect(readyRuntimeEntryBody.split('startPreGoneCrashSampling()').length - 1).toBe(1) + // Why pin the indent: the call also matches as the body of an added + // `if (...)` guard, which keeps every other assertion here true while the + // sampler silently stops arming on most startups. + expect(readyRuntimeEntryBody).toContain('\n startPreGoneCrashSampling()') + + // ...and that this really is the function app readiness runs. + expect(readySource).toContain( + "import { initializeReadyRuntimeServices } from './main-process-ready-runtime'" + ) + expect(readySource).toContain('\n await initializeReadyRuntimeServices()') + }) +}) diff --git a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx index 52ddbf1ba5d..bb3ffbfd621 100644 --- a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx +++ b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx @@ -24,6 +24,7 @@ export function TerminalWorkspaceDialogs({ saveDialogFile, saveDialogFileId, setWindowCloseDialogOpen, + windowCloseDialogKind, windowCloseDialogOpen } = controller return ( @@ -82,10 +83,15 @@ export function TerminalWorkspaceDialogs({ {translate('auto.components.Terminal.2fa9c69ff3', 'Close Window?')} - {translate( - 'auto.components.Terminal.7958465754', - 'There are local terminals with running processes. Close the window anyway?' - )} + {windowCloseDialogKind === 'unverifiable' + ? translate( + 'auto.components.Terminal.b7c1f0a934', + 'A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?' + ) + : translate( + 'auto.components.Terminal.7958465754', + 'There are terminals with running processes. Close the window anyway?' + )} diff --git a/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx b/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx index be1c69cbac7..0ae53fee931 100644 --- a/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx +++ b/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx @@ -493,6 +493,37 @@ describe('WorkspacePortScanner', () => { expect(useAppStore.getState().workspacePortScansByKey['environment:env-3:all']).toBeUndefined() }) + // Why: a manual publish (the ports popover) can resolve after the host-set + // change already pruned its key, re-adding it. Per-key writes never delete, so + // that removed host would otherwise hold its ports and a permanent + // unavailable notice until the next host-set change. + it('drops a stale host re-added after pruning on the next poll', async () => { + await act(async () => { + root?.render() + await flushPromises() + }) + + const staleKey = 'environment:env-removed:all' + act(() => { + const state = useAppStore.getState() + state.replaceWorkspacePortScans( + { + ...state.workspacePortScansByKey, + [staleKey]: { ...emptyScan, unavailableReason: 'gone' } + }, + state.workspacePortScan + ) + }) + expect(useAppStore.getState().workspacePortScansByKey[staleKey]).toBeDefined() + + await act(async () => { + vi.advanceTimersByTime(30_000) + await flushPromises() + }) + + expect(useAppStore.getState().workspacePortScansByKey[staleKey]).toBeUndefined() + }) + it('clears ports immediately when the final worktree is removed', async () => { runtimeEnvironmentCall.mockImplementation(({ method }) => { if (method === 'workspacePorts.scan') { diff --git a/src/renderer/src/components/ports/WorkspacePortScanner.tsx b/src/renderer/src/components/ports/WorkspacePortScanner.tsx index f2805eb88a7..2b12bd86cd8 100644 --- a/src/renderer/src/components/ports/WorkspacePortScanner.tsx +++ b/src/renderer/src/components/ports/WorkspacePortScanner.tsx @@ -4,10 +4,11 @@ import { getHasAnyWorktreesFromState } from '@/store/selectors' import { getActiveRuntimeTarget, type RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { mergeWorkspacePortScans, - runtimeTargetForExecutionHostId, + WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY, scanWorkspacePortsForTarget, workspacePortScanKeyForTarget } from '@/lib/workspace-port-actions' +import { runtimeTargetForExecutionHostId } from '@/runtime/runtime-client-target' import { installWindowVisibilityInterval, isWindowVisible } from '@/lib/window-visibility-interval' import { reconcileTransientPortScanFailures, @@ -41,7 +42,6 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) const setWorkspacePortScanProjection = useAppStore((s) => s.setWorkspacePortScanProjection) const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const inFlightRef = useRef | null>(null) const generationRef = useRef(0) @@ -124,40 +124,40 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): const activeTargetKeys = new Set( allTargets.map((target) => workspacePortScanKeyForTarget(target)) ) + const publishedScans = useAppStore.getState().workspacePortScansByKey const reconciled = reconcileTransientPortScanFailures( results, - useAppStore.getState().workspacePortScansByKey, + publishedScans, portScanDebounceRef.current, WORKSPACE_PORT_SCAN_FAILURE_THRESHOLD, activeTargetKeys ) const scansByKey = Object.fromEntries( - Object.entries(useAppStore.getState().workspacePortScansByKey).filter(([key]) => - activeTargetKeys.has(key) - ) + Object.entries(publishedScans).filter(([key]) => activeTargetKeys.has(key)) ) - let sourceChanged = false + // Why: a manual publish that lands after a host is pruned re-adds its key, + // and per-key writes never delete. Dropping the inactive keys here is what + // stops a removed host from holding a permanent unavailable notice. + let sourceChanged = + Object.keys(scansByKey).length !== Object.keys(publishedScans).length for (const { key, result } of reconciled) { sourceChanged ||= scansByKey[key] !== result scansByKey[key] = result - setWorkspacePortScanForKey(key, result) } const activeScan = scansByKey[scanKey] const merged = mergeWorkspacePortScans(scansByKey) const projectionKey = allTargets.length > 1 - ? 'all-hosts:all' + ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : activeScan ? scanKey : workspacePortScanKeyForTarget(allTargets[0]) if (sourceChanged || useAppStore.getState().workspacePortScan?.key !== projectionKey) { - setWorkspacePortScanProjection( - merged - ? { - key: projectionKey, - result: merged - } - : null + // Why: one store update for the whole poll — a large host set must not + // fan out a notification to every subscriber per host. + replaceWorkspacePortScans( + sourceChanged ? scansByKey : publishedScans, + merged ? { key: projectionKey, result: merged } : null ) } } @@ -177,8 +177,7 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): hasWorktrees, scanKey, setWorkspacePortScan, - setWorkspacePortScanProjection, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ] ) @@ -215,7 +214,7 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): : Object.fromEntries(retainedEntries) const retainedProjection = mergeWorkspacePortScans(retainedScans) const retainedProjectionKey = - targetKeys.size > 1 ? 'all-hosts:all' : Object.keys(retainedScans)[0] + targetKeys.size > 1 ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : Object.keys(retainedScans)[0] // Why: unchanged hosts stay visible while the replacement RPC runs; removed // hosts and the old synthetic aggregate are excluded immediately. const nextProjection = diff --git a/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx b/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx index ea1d85f15b8..1121a05ae64 100644 --- a/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx +++ b/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx @@ -451,25 +451,26 @@ describe('PortsPanel runtime routing', () => { }) it('returns post-stop refresh failures without throwing', async () => { - const setWorkspacePortScan = vi.fn() + const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() localScan.mockRejectedValueOnce(new Error('scan failed')) await expect( refreshWorkspacePortScanAfterStop({ runtimeTarget: { kind: 'local' }, - setWorkspacePortScan: setWorkspacePortScan as never, + replaceWorkspacePortScans: replaceWorkspacePortScans as never, + getWorkspacePortScansByKey: () => ({}), setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never }) ).resolves.toEqual({ ok: false, reason: 'scan failed' }) - expect(setWorkspacePortScan).not.toHaveBeenCalled() + expect(replaceWorkspacePortScans).not.toHaveBeenCalled() expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(1, true) expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(2, false) }) it('ignores settled remote post-stop refresh failures after updating state', async () => { - const setWorkspacePortScan = vi.fn() + const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() const firstScan = { ...emptyScan, scannedAt: 2 } let scanCalls = 0 @@ -500,7 +501,8 @@ describe('PortsPanel runtime routing', () => { await expect( refreshWorkspacePortScanAfterStop({ runtimeTarget: { kind: 'environment', environmentId: 'env-1' }, - setWorkspacePortScan: setWorkspacePortScan as never, + replaceWorkspacePortScans: replaceWorkspacePortScans as never, + getWorkspacePortScansByKey: () => ({}), setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never }) ).resolves.toEqual({ ok: true }) @@ -510,18 +512,20 @@ describe('PortsPanel runtime routing', () => { 'workspacePorts.scan', 'workspacePorts.scan' ]) - expect(setWorkspacePortScan).toHaveBeenCalledTimes(1) - expect(setWorkspacePortScan).toHaveBeenCalledWith({ - key: 'environment:env-1:all', - result: firstScan - }) + expect(replaceWorkspacePortScans).toHaveBeenCalledTimes(1) + expect(replaceWorkspacePortScans).toHaveBeenCalledWith( + { 'environment:env-1:all': firstScan }, + { + key: 'environment:env-1:all', + result: firstScan + } + ) expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(1, true) expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(2, false) }) it('preserves an all-host projection after refreshing one host post-stop', async () => { - const setWorkspacePortScan = vi.fn() - const setWorkspacePortScanForKey = vi.fn() + const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() const localPort: WorkspacePort = { ...workspacePort, id: 'local-port', port: 5173 } const refreshedRemotePort: WorkspacePort = { @@ -571,23 +575,24 @@ describe('PortsPanel runtime routing', () => { await expect( refreshWorkspacePortScanAfterStop({ runtimeTarget: { kind: 'environment', environmentId: 'env-1' }, - setWorkspacePortScan: setWorkspacePortScan as never, - setWorkspacePortScanForKey: setWorkspacePortScanForKey as never, + replaceWorkspacePortScans: replaceWorkspacePortScans as never, getWorkspacePortScansByKey: () => ({ 'local:all': localHostScan }), setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never }) ).resolves.toEqual({ ok: true }) - expect(setWorkspacePortScanForKey).toHaveBeenCalledWith('environment:env-1:all', remoteHostScan) - expect(setWorkspacePortScan).toHaveBeenLastCalledWith({ - key: 'all-hosts:all', - result: expect.objectContaining({ - ports: expect.arrayContaining([ - expect.objectContaining({ port: 5173 }), - expect.objectContaining({ port: 3000 }) - ]) - }) - }) + expect(replaceWorkspacePortScans).toHaveBeenLastCalledWith( + { 'local:all': localHostScan, 'environment:env-1:all': remoteHostScan }, + { + key: 'all-hosts:all', + result: expect.objectContaining({ + ports: expect.arrayContaining([ + expect.objectContaining({ port: 5173 }), + expect.objectContaining({ port: 3000 }) + ]) + }) + } + ) expect(scanCalls).toBe(2) }) diff --git a/src/renderer/src/components/right-sidebar/local-workspace-port-sections.test.ts b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.test.ts new file mode 100644 index 00000000000..68733927ee7 --- /dev/null +++ b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.test.ts @@ -0,0 +1,30 @@ +import { describe, expect, it } from 'vitest' +import { shouldShowLocalWorkspacePortSections } from './local-workspace-port-sections' + +const empty = { activePorts: [], otherWorkspacePorts: [], externalPorts: [] } + +describe('shouldShowLocalWorkspacePortSections', () => { + it('shows the sections whenever the scan succeeded', () => { + expect(shouldShowLocalWorkspacePortSections(null, empty)).toBe(true) + expect(shouldShowLocalWorkspacePortSections({}, empty)).toBe(true) + }) + + // Why: a failed scan keeps the host's last-good ports, and the status bar + // still counts and lists them — hiding the sections here would strip the + // stop and open actions for ports the user can still see elsewhere. + it.each([ + ['activePorts', { ...empty, activePorts: [{}] }], + ['otherWorkspacePorts', { ...empty, otherWorkspacePorts: [{}] }], + ['externalPorts', { ...empty, externalPorts: [{}] }] + ])('keeps the sections when a failed scan retained %s', (_section, sections) => { + expect(shouldShowLocalWorkspacePortSections({ unavailableReason: 'dropped' }, sections)).toBe( + true + ) + }) + + it('lets the notice stand alone when a failed scan has nothing left to list', () => { + expect(shouldShowLocalWorkspacePortSections({ unavailableReason: 'dropped' }, empty)).toBe( + false + ) + }) +}) diff --git a/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts index d968bb85144..2a0eadf380a 100644 --- a/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts +++ b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts @@ -35,6 +35,26 @@ export function getLocalWorkspacePortSections( } } +/** + * Whether the panel still renders its port sections under a failure notice. + * Why: a failed scan retains the host's last-good ports, so hiding every + * section would drop the stop and open actions for ports the status bar still + * counts and lists. + */ +export function shouldShowLocalWorkspacePortSections( + scan: { unavailableReason?: string } | null | undefined, + sections: { activePorts: unknown[]; otherWorkspacePorts: unknown[]; externalPorts: unknown[] } +): boolean { + if (!scan?.unavailableReason) { + return true + } + return ( + sections.activePorts.length > 0 || + sections.otherWorkspacePorts.length > 0 || + sections.externalPorts.length > 0 + ) +} + function workspacePortAsExternal(port: WorkspacePort & { kind: 'workspace' }): WorkspacePort { return { id: port.id, diff --git a/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx b/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx index 0955e89031b..1737e7da487 100644 --- a/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx +++ b/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx @@ -4,11 +4,11 @@ import { toast } from 'sonner' import { useAppStore } from '@/store' import { useActiveWorktree, useRepoById } from '@/store/selectors' import { cn } from '@/lib/utils' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' -import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' import { killWorkspacePortForTarget, openWorkspacePortInBrowser, + publishWorkspacePortScanForHost, refreshWorkspacePortScanAfterStop, resolvePortOpenInOrcaBrowser, scanWorkspacePortsForTarget, @@ -19,10 +19,14 @@ import { Button } from '@/components/ui/button' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import type { WorkspacePort } from '../../../../shared/workspace-ports' import { translate } from '@/i18n/i18n' -import { getLocalWorkspacePortSections } from './local-workspace-port-sections' +import { + getLocalWorkspacePortSections, + shouldShowLocalWorkspacePortSections +} from './local-workspace-port-sections' import { LocalPortSection } from './local-port-section' import { LocalPortDetailsDialog } from './local-port-details-dialog' +/** Right-sidebar Ports panel scoped to the active workspace's owner host. */ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): React.JSX.Element { const activeWorktree = useActiveWorktree() const activeRepo = useRepoById(activeWorktree?.repoId ?? null) @@ -31,8 +35,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle) const scansByKey = useAppStore((s) => s.workspacePortScansByKey) const refreshing = useAppStore((s) => s.workspacePortScanRefreshing) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const [detailsPort, setDetailsPort] = useState(null) const [collapsedSections, setCollapsedSections] = useState>({ @@ -40,26 +43,24 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): external: true }) - const runtimeTarget = useMemo(() => { - const activeRuntimeEnvironmentId = getRuntimeEnvironmentIdForWorktree( - useAppStore.getState(), - activeWorktree?.id - ) - // Why: the Ports panel acts on the active workspace; use that workspace's - // host owner even if the sidebar is focused elsewhere. - return getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId }) - }, [activeWorktree?.id, settings]) - const scanKey = `${workspacePortRuntimeTargetKey(runtimeTarget)}:all` + // Why: the Ports panel acts on the active workspace; use that workspace's + // host owner even if the sidebar is focused elsewhere. + const runtimeTarget = useWorktreeRuntimeTarget(activeWorktree?.id) + const scanKey = runtimeTarget ? `${workspacePortRuntimeTargetKey(runtimeTarget)}:all` : null const refresh = useCallback(() => { - if (!activeRepo) { + if (!activeRepo || !runtimeTarget || !scanKey) { return Promise.resolve() } setWorkspacePortScanRefreshing(true) const promise = scanWorkspacePortsForTarget(runtimeTarget) .then((nextScan) => { - setWorkspacePortScanForKey(scanKey, nextScan) - setWorkspacePortScan({ key: scanKey, result: nextScan }) + publishWorkspacePortScanForHost({ + scanKey, + scan: nextScan, + replaceWorkspacePortScans, + getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey + }) }) .catch((error) => { const message = error instanceof Error ? error.message : String(error) @@ -86,14 +87,13 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): activeRepo, runtimeTarget, scanKey, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ]) // Why: WorkspacePortScanner already owns the 30s all-worktree poll. The // panel scopes that shared result instead of starting a second scan loop. - const displayScan = isVisible ? (scansByKey[scanKey] ?? null) : null + const displayScan = isVisible && scanKey ? (scansByKey[scanKey] ?? null) : null const toggleSection = useCallback((sectionId: string) => { setCollapsedSections((current) => ({ ...current, [sectionId]: !current[sectionId] })) @@ -122,8 +122,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): ) const refreshResult = await refreshWorkspacePortScanAfterStop({ runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey, setWorkspacePortScanRefreshing }) @@ -139,13 +138,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): ) } }, - [ - activeRepo, - runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, - setWorkspacePortScanRefreshing - ] + [activeRepo, runtimeTarget, replaceWorkspacePortScans, setWorkspacePortScanRefreshing] ) const handleOpenPortInBrowser = useCallback( @@ -181,6 +174,12 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): [activeRepo?.id, activeWorktree?.id, displayScan] ) + const showPortSections = shouldShowLocalWorkspacePortSections(displayScan, { + activePorts, + otherWorkspacePorts, + externalPorts + }) + if (!activeRepo) { return (
@@ -209,7 +208,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): size="icon-xs" className="text-muted-foreground hover:text-foreground" onClick={() => void refresh()} - disabled={refreshing} + disabled={refreshing || !runtimeTarget} aria-label={translate( 'auto.components.right.sidebar.PortsPanel.7822e3edc6', 'Refresh Ports' @@ -237,7 +236,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
)} - {!displayScan?.unavailableReason && ( + {showPortSections && ( <> ({ usePromptCacheCountdownStartedAt: vi.fn() @@ -52,7 +52,7 @@ vi.mock('@/store', () => ({ recordFeatureInteraction, remoteBranchConflictByWorktreeId: {}, setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing, settings, sshConnectionStates: new Map(), diff --git a/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx b/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx index e8376a64f08..2c1f10945d3 100644 --- a/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx @@ -13,7 +13,7 @@ import type { WorkspacePortScanResult } from '../../../../shared/workspace-ports const fetchHostedReviewForBranch = vi.fn() const fetchIssue = vi.fn() const fetchLinearIssue = vi.fn() -const setWorkspacePortScan = vi.fn() +const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() const cacheTimerMocks = vi.hoisted(() => ({ usePromptCacheCountdownStartedAt: vi.fn() @@ -43,7 +43,7 @@ vi.mock('@/store', () => ({ recordFeatureInteraction: vi.fn(), remoteBranchConflictByWorktreeId: {}, setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing, settings, sshConnectionStates: new Map(), diff --git a/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx b/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx index 0a8bb4d248e..db23d18dbc2 100644 --- a/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx @@ -8,7 +8,7 @@ vi.mock('@/store', () => ({ selector({ createBrowserTab: vi.fn(), setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan: vi.fn(), + replaceWorkspacePortScans: vi.fn(), setWorkspacePortScanRefreshing: vi.fn(), settings: null }) diff --git a/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx b/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx index 7f7c411dfa2..cc4f567e50a 100644 --- a/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx +++ b/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx @@ -1,11 +1,10 @@ -import React, { useCallback, useMemo } from 'react' +import React, { useCallback } from 'react' import { Plug, Copy, ExternalLink, Trash2 } from 'lucide-react' import { toast } from 'sonner' import { useAppStore } from '@/store' import { Button } from '@/components/ui/button' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' -import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' import { canStopWorkspacePort, getPortOpenBrowserTooltipLabel, @@ -97,21 +96,18 @@ function PortAction({ ) } +/** One port row on a sidebar worktree card, with open/copy/stop actions on its owner host. */ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element { const settings = useAppStore((s) => s.settings) const localhostLabelRoute = useLocalhostLabelRouteForPort(port) - const runtimeEnvironmentId = useAppStore((s) => - getRuntimeEnvironmentIdForWorktree(s, port.kind === 'workspace' ? port.owner.worktreeId : null) - ) + const createBrowserTab = useAppStore((s) => s.createBrowserTab) const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction) - const runtimeTarget = useMemo( - () => getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId: runtimeEnvironmentId }), - [runtimeEnvironmentId, settings] + const runtimeTarget = useWorktreeRuntimeTarget( + port.kind === 'workspace' ? port.owner.worktreeId : null ) const processLabel = port.processName ?? (port.pid ? `PID ${port.pid}` : 'Unknown process') const address = addressForPort(port) @@ -203,8 +199,7 @@ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element { ) const refreshResult = await refreshWorkspacePortScanAfterStop({ runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey, setWorkspacePortScanRefreshing }) @@ -226,8 +221,7 @@ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element { port, recordFeatureInteraction, runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ] ) diff --git a/src/renderer/src/components/status-bar/PortsStatusSegment.host-routing.test.tsx b/src/renderer/src/components/status-bar/PortsStatusSegment.host-routing.test.tsx new file mode 100644 index 00000000000..9b22826d6bd --- /dev/null +++ b/src/renderer/src/components/status-bar/PortsStatusSegment.host-routing.test.tsx @@ -0,0 +1,407 @@ +// @vitest-environment happy-dom + +import React, { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { WorkspacePort, WorkspacePortScanResult } from '../../../../shared/workspace-ports' + +const { popoverHandle, runWorkspacePortScanForTargetMock, storeState } = vi.hoisted(() => { + const storeState = { + settings: { activeRuntimeEnvironmentId: null as string | null }, + activeWorktreeId: 'runtime-repo::/srv/app', + workspacePortScan: null as { key: string; result: WorkspacePortScanResult } | null, + workspacePortScansByKey: {} as Record, + workspacePortScanRefreshing: false, + runtimeEnvironments: [] as { id: string; name: string }[], + recordFeatureInteraction: vi.fn(), + replaceWorkspacePortScans: + vi.fn< + ( + scansByKey: Record, + projection: { key: string; result: WorkspacePortScanResult } | null + ) => void + >() + } + // Why: the real store writes back. A bare spy lets a publish and the notice + // that reads it drift onto different scan keys with every assertion green. + storeState.replaceWorkspacePortScans.mockImplementation((scansByKey, projection) => { + storeState.workspacePortScansByKey = scansByKey + storeState.workspacePortScan = projection + }) + return { + popoverHandle: { onOpenChange: null as ((open: boolean) => void) | null }, + runWorkspacePortScanForTargetMock: vi.fn(), + storeState + } +}) + +vi.mock('@/store', () => { + const useAppStore = Object.assign( + (selector: (state: typeof storeState) => unknown) => selector(storeState), + { getState: () => storeState } + ) + return { useAppStore } +}) + +vi.mock('@/lib/worktree-runtime-owner', () => ({ + getExecutionHostIdForWorktree: (_state: unknown, worktreeId: string | null | undefined) => { + if (worktreeId === 'runtime-repo::/srv/app') { + return 'runtime:env-1' + } + if (worktreeId === 'ssh-repo::/srv/app') { + return 'ssh:server-1' + } + return 'local' + } +})) + +vi.mock('@/runtime/runtime-rpc-client', async () => { + const actual = await import('@/runtime/runtime-client-target') + return { + getActiveRuntimeTarget: actual.getActiveRuntimeTarget, + callRuntimeRpc: vi.fn(), + assertRuntimeEnvironmentCapability: vi.fn(), + RuntimeRpcCallError: class RuntimeRpcCallError extends Error { + code?: string + } + } +}) + +vi.mock('@/lib/workspace-port-scan-client', () => ({ + runWorkspacePortScanForTarget: runWorkspacePortScanForTargetMock +})) + +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorktree: vi.fn() +})) + +vi.mock('@/components/ui/popover', () => ({ + Popover: ({ + children, + onOpenChange + }: { + children: React.ReactNode + onOpenChange: (open: boolean) => void + }) => { + popoverHandle.onOpenChange = onOpenChange + return <>{children} + }, + PopoverContent: ({ children }: { children: React.ReactNode }) => <>{children}, + PopoverTrigger: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('@/components/ui/tooltip', () => ({ + Tooltip: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipContent: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipTrigger: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('@/components/SelectedTextCopyMenu', () => ({ + SelectedTextCopyMenu: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('./ports-status-popover-rows', () => ({ + PortRow: () =>
, + WorkspaceGroupRows: () =>
+})) + +vi.mock('@/i18n/i18n', () => ({ + translate: (_key: string, fallback: string, options?: Record) => + options + ? fallback.replace(/{{(\w+)}}/g, (_match, name: string) => String(options[name] ?? '')) + : fallback +})) + +import { PortsStatusSegment } from './PortsStatusSegment' + +function workspacePort(overrides: Partial & { port: number; id: string }) { + return { + bindHost: '0.0.0.0', + connectHost: '127.0.0.1', + port: overrides.port, + id: overrides.id, + pid: 4321, + processName: 'node', + protocol: 'http' as const, + kind: 'workspace' as const, + owner: { + worktreeId: 'runtime-repo::/srv/app', + repoId: 'runtime-repo', + displayName: 'runtime app', + path: '/srv/app', + confidence: 'cwd' as const + } + } +} + +const localHostScan: WorkspacePortScanResult = { + platform: 'linux', + scannedAt: 10, + ports: [workspacePort({ id: 'local-5173', port: 5173 })] +} + +const remoteHostScan: WorkspacePortScanResult = { + platform: 'linux', + scannedAt: 20, + ports: [workspacePort({ id: 'remote-3000', port: 3000 })] +} + +describe('PortsStatusSegment popover host routing', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + popoverHandle.onOpenChange = null + storeState.settings = { activeRuntimeEnvironmentId: null } + storeState.activeWorktreeId = 'runtime-repo::/srv/app' + storeState.workspacePortScan = null + storeState.workspacePortScansByKey = { 'local:all': localHostScan } + storeState.runtimeEnvironments = [{ id: 'env-1', name: 'linux-box' }] + storeState.recordFeatureInteraction.mockClear() + storeState.replaceWorkspacePortScans.mockClear() + runWorkspacePortScanForTargetMock.mockReset() + runWorkspacePortScanForTargetMock.mockResolvedValue(remoteHostScan) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + act(() => { + root.render() + }) + }) + + afterEach(() => { + act(() => { + root.unmount() + }) + container.remove() + }) + + async function openPopover(): Promise { + await act(async () => { + popoverHandle.onOpenChange?.(true) + await Promise.resolve() + await Promise.resolve() + }) + } + + it("scans the active workspace's host, not the globally focused runtime", async () => { + await openPopover() + + expect(runWorkspacePortScanForTargetMock).toHaveBeenCalledWith( + { kind: 'environment', environmentId: 'env-1' }, + undefined + ) + expect(storeState.replaceWorkspacePortScans).toHaveBeenCalledTimes(1) + expect(storeState.workspacePortScansByKey['environment:env-1:all']).toBe(remoteHostScan) + }) + + it('keeps other hosts in the projection instead of overwriting it with one host', async () => { + await openPopover() + + const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(projection).toEqual({ + key: 'all-hosts:all', + result: expect.objectContaining({ + ports: expect.arrayContaining([ + expect.objectContaining({ port: 5173 }), + expect.objectContaining({ port: 3000 }) + ]) + }) + }) + }) + + it('publishes a failed scan under its own host without dropping other hosts', async () => { + runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed')) + + await openPopover() + + const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(projection.key).toBe('all-hosts:all') + expect(projection.result.ports).toEqual([expect.objectContaining({ port: 5173 })]) + expect(storeState.workspacePortScansByKey['environment:env-1:all']).toEqual( + expect.objectContaining({ unavailableReason: 'remote scan failed' }) + ) + }) + + it('keeps the failed host last-good ports while naming the failure', async () => { + storeState.workspacePortScansByKey = { + 'local:all': localHostScan, + 'environment:env-1:all': remoteHostScan + } + runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed')) + + await openPopover() + + // Why: one dropped scan must not clear the host's ports the way the + // background poll's debounce does not — the notice names the failure + // while the projection keeps serving the last-good rows. + const failed = storeState.workspacePortScansByKey['environment:env-1:all'] + expect(failed.unavailableReason).toBe('remote scan failed') + expect(failed.platform).toBe('linux') + expect(failed.ports).toEqual([expect.objectContaining({ port: 3000 })]) + const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(projection.key).toBe('all-hosts:all') + expect(projection.result.ports.map((port) => port.port).sort()).toEqual([3000, 5173]) + }) + + // Why: separate tests already cover "the failure is stored" and "a stored + // failure renders". Only this one proves both halves name the same scan key. + it('surfaces the host it just failed to scan on the next render', async () => { + runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed')) + + await openPopover() + act(() => { + root.render() + }) + + expect(container.textContent).toContain( + 'Port scan unavailable on linux-box: remote scan failed' + ) + }) + + it('names the host whose scan failed while another host still reports ports', () => { + act(() => { + root.unmount() + }) + storeState.workspacePortScansByKey = { + 'local:all': localHostScan, + 'environment:env-1:all': { + platform: 'linux', + scannedAt: 30, + ports: [], + unavailableReason: 'Remote connection dropped' + } + } + storeState.workspacePortScan = { key: 'all-hosts:all', result: localHostScan } + root = createRoot(container) + act(() => { + root.render() + }) + + expect(container.textContent).toContain( + 'Port scan unavailable on linux-box: Remote connection dropped' + ) + // The notice sits above the list rather than replacing it: a reachable + // host's count still renders. + expect(container.textContent).toContain('1 workspace') + }) + + // Why: a failed scan keeps the host's last-good ports, and the badge and + // header count them. Replacing the list with the notice left the popover + // claiming N ports over an empty body. + it('keeps the list under the notice when a failed scan retained its ports', () => { + act(() => { + root.unmount() + }) + storeState.activeWorktreeId = 'local-repo::/home/dev/app' + const retained: WorkspacePortScanResult = { + ...localHostScan, + unavailableReason: 'lsof is unavailable' + } + storeState.workspacePortScansByKey = { 'local:all': retained } + storeState.workspacePortScan = { key: 'local:all', result: retained } + root = createRoot(container) + act(() => { + root.render() + }) + + expect(container.textContent).toContain('1 workspace · 0 external') + expect(container.querySelectorAll('[data-testid="workspace-group-rows"]')).toHaveLength(1) + expect(container.textContent).toContain('Port scan unavailable on Local Linux') + }) + + // Why: total loss of contact is where naming the host matters most, and the + // merged projection can only offer platform 'unknown' and raw scan keys. + it('names every host when all of them failed with nothing left to list', () => { + act(() => { + root.unmount() + }) + const merged: WorkspacePortScanResult = { + platform: 'unknown', + scannedAt: 30, + ports: [], + unavailableReason: 'local:all: lsof is unavailable; environment:env-1:all: dropped' + } + storeState.workspacePortScansByKey = { + 'local:all': { + platform: 'darwin', + scannedAt: 30, + ports: [], + unavailableReason: 'lsof is unavailable' + }, + 'environment:env-1:all': { + platform: 'linux', + scannedAt: 30, + ports: [], + unavailableReason: 'dropped' + } + } + storeState.workspacePortScan = { key: 'all-hosts:all', result: merged } + root = createRoot(container) + act(() => { + root.render() + }) + + // Local label comes from the scan's own platform, not the renderer's + // userAgent — a paired web client is not the Orca host. + expect(container.textContent).toContain( + 'Port scan unavailable on Local Mac: lsof is unavailable' + ) + expect(container.textContent).toContain('Port scan unavailable on linux-box: dropped') + expect(container.textContent).not.toContain('unavailable on unknown') + expect(container.textContent).not.toContain('environment:env-1:all:') + // The notice takes over the body only when there is nothing left to list. + expect(container.querySelectorAll('[data-testid="workspace-group-rows"]')).toHaveLength(0) + expect(container.textContent).not.toContain('No workspace ports detected') + }) + + it('stays on the local host when the active workspace has no runtime owner', async () => { + act(() => { + root.unmount() + }) + storeState.activeWorktreeId = 'local-repo::/home/dev/app' + root = createRoot(container) + act(() => { + root.render() + }) + + await openPopover() + + expect(runWorkspacePortScanForTargetMock).toHaveBeenCalledWith({ kind: 'local' }, undefined) + const [nextScans, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(nextScans['local:all']).toBe(remoteHostScan) + expect(projection).toEqual({ + key: 'local:all', + result: remoteHostScan + }) + }) + + it('does not substitute the local host for a direct-SSH workspace', async () => { + act(() => { + root.unmount() + }) + storeState.activeWorktreeId = 'ssh-repo::/srv/app' + root = createRoot(container) + act(() => { + root.render() + }) + + await openPopover() + + expect(runWorkspacePortScanForTargetMock).not.toHaveBeenCalled() + expect(storeState.replaceWorkspacePortScans).not.toHaveBeenCalled() + expect(storeState.recordFeatureInteraction).toHaveBeenCalledWith('ports') + }) +}) diff --git a/src/renderer/src/components/status-bar/PortsStatusSegment.tsx b/src/renderer/src/components/status-bar/PortsStatusSegment.tsx index 0626137c010..b56f5432cc8 100644 --- a/src/renderer/src/components/status-bar/PortsStatusSegment.tsx +++ b/src/renderer/src/components/status-bar/PortsStatusSegment.tsx @@ -3,39 +3,83 @@ import { Plug, ChevronDown, ChevronRight, LoaderCircle } from 'lucide-react' import { Popover, PopoverContent, PopoverTrigger } from '@/components/ui/popover' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { useAppStore } from '@/store' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' import { + publishWorkspacePortScanForHost, scanWorkspacePortsForTarget, workspacePortScanKeyForTarget } from '@/lib/workspace-port-actions' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' +import { + getUnavailableWorkspacePortHosts, + type WorkspacePortHostRef +} from '@/lib/workspace-port-host-availability' +import { getLocalExecutionHostLabel } from '../../../../shared/execution-host' import { getExternalWorkspacePorts, getWorkspacePortGroups } from '@/lib/workspace-port-groups' import { SelectedTextCopyMenu } from '@/components/SelectedTextCopyMenu' import { STATUS_BAR_CONTEXT_MENU_EXEMPT_PROPS } from './status-bar-context-menu-policy' import { PortRow, WorkspaceGroupRows } from './ports-status-popover-rows' import { translate } from '@/i18n/i18n' +import type { WorkspacePortScanResult } from '../../../../shared/workspace-ports' type PortsStatusSegmentProps = { compact?: boolean iconOnly: boolean } +/** Status-bar plug icon with the workspace port count and a per-host ports popover. */ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React.JSX.Element { - const settings = useAppStore((s) => s.settings) const scan = useAppStore((s) => s.workspacePortScan?.result ?? null) const refreshing = useAppStore((s) => s.workspacePortScanRefreshing) const activeWorktreeId = useAppStore((s) => s.activeWorktreeId) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) + const scansByKey = useAppStore((s) => s.workspacePortScansByKey) + const runtimeEnvironments = useAppStore((s) => s.runtimeEnvironments) const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction) const [open, setOpen] = useState(false) const [externalOpen, setExternalOpen] = useState(false) - const runtimeTarget = useMemo(() => getActiveRuntimeTarget(settings), [settings]) - const scanKey = workspacePortScanKeyForTarget(runtimeTarget) + const runtimeTarget = useWorktreeRuntimeTarget(activeWorktreeId) + const scanKey = runtimeTarget ? workspacePortScanKeyForTarget(runtimeTarget) : null const workspaceGroups = useMemo(() => getWorkspacePortGroups(scan), [scan]) const externalPorts = useMemo(() => getExternalWorkspacePorts(scan), [scan]) + const unavailableHosts = useMemo(() => getUnavailableWorkspacePortHosts(scansByKey), [scansByKey]) + const hostLabel = useCallback( + (host: WorkspacePortHostRef, hostScanKey: string, platform: NodeJS.Platform | null) => { + if (host.kind === 'local') { + // Why: a paired web client's own userAgent is not the Orca host's + // platform, so name the machine the scan actually ran on. + return getLocalExecutionHostLabel(platform) + } + if (host.kind === 'unknown') { + return hostScanKey + } + return ( + runtimeEnvironments.find((environment) => environment.id === host.environmentId)?.name ?? + host.environmentId + ) + }, + [runtimeEnvironments] + ) const workspacePortCount = workspaceGroups.reduce((count, group) => count + group.ports.length, 0) const totalCount = workspacePortCount + externalPorts.length + const unavailableNotices = useMemo(() => { + if (unavailableHosts.length > 0) { + return unavailableHosts.map((entry) => ({ + id: entry.scanKey, + host: hostLabel(entry.host, entry.scanKey, entry.platform), + reason: entry.reason + })) + } + // Why: a projection published without per-host scans has no host to name. + return scan?.unavailableReason + ? [{ id: 'projection', host: scan.platform, reason: scan.unavailableReason }] + : [] + }, [hostLabel, scan?.platform, scan?.unavailableReason, unavailableHosts]) + // Why: a failed scan keeps the host's last-good ports, and those ports are + // counted in the badge and header — replacing the list with the notice would + // leave the popover claiming N ports over an empty body. Only take over the + // body when there is genuinely nothing left to list. + const noticeReplacesList = Boolean(scan?.unavailableReason) && totalCount === 0 const handleOpenChange = useCallback( (nextOpen: boolean) => { setOpen(nextOpen) @@ -43,33 +87,36 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React return } recordFeatureInteraction('ports') + if (!runtimeTarget || !scanKey) { + return + } // Why: the 30s background poll is intentionally quiet; opening the // popover should still collapse that stale window without flashing icons. - void scanWorkspacePortsForTarget(runtimeTarget) - .then((result) => { - setWorkspacePortScanForKey(scanKey, result) - setWorkspacePortScan({ key: scanKey, result }) + const publish = (result: WorkspacePortScanResult): void => { + publishWorkspacePortScanForHost({ + scanKey, + scan: result, + replaceWorkspacePortScans, + getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey }) + } + void scanWorkspacePortsForTarget(runtimeTarget) + .then(publish) .catch((error) => { const message = error instanceof Error ? error.message : String(error) - setWorkspacePortScan({ - key: scanKey, - result: { - platform: 'unknown', - scannedAt: Date.now(), - ports: [], - unavailableReason: message || 'Workspace port scan failed.' - } + // Why: one dropped scan must not clear the host's last-good ports the + // way the background poll's debounce does not; the failure is still + // recorded so the host is named by the unavailable notice below. + const previous = useAppStore.getState().workspacePortScansByKey[scanKey] + publish({ + platform: previous?.platform ?? 'unknown', + scannedAt: Date.now(), + ports: previous?.ports ?? [], + unavailableReason: message || 'Workspace port scan failed.' }) }) }, - [ - recordFeatureInteraction, - runtimeTarget, - scanKey, - setWorkspacePortScan, - setWorkspacePortScanForKey - ] + [recordFeatureInteraction, runtimeTarget, scanKey, replaceWorkspacePortScans] ) return ( @@ -153,14 +200,18 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React
- {scan?.unavailableReason ? ( -
- {translate( - 'auto.components.status.bar.PortsStatusSegment.95495019ed', - 'Port scan unavailable on {{value0}}: {{value1}}', - { value0: scan.platform, value1: scan.unavailableReason } - )} -
+ {unavailableNotices.length > 0 && !noticeReplacesList && ( + + )} + + {noticeReplacesList ? ( + ) : (
{workspaceGroups.length > 0 ? ( @@ -237,3 +288,27 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React ) } + +type PortScanUnavailableNotice = { id: string; host: string; reason: string } + +function PortScanUnavailableNotices({ + notices, + className +}: { + notices: PortScanUnavailableNotice[] + className: string +}): React.JSX.Element { + return ( +
+ {notices.map((notice) => ( +
+ {translate( + 'auto.components.status.bar.PortsStatusSegment.95495019ed', + 'Port scan unavailable on {{value0}}: {{value1}}', + { value0: notice.host, value1: notice.reason } + )} +
+ ))} +
+ ) +} diff --git a/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx b/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx index 47c86d89c57..ba926909272 100644 --- a/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx +++ b/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx @@ -17,8 +17,7 @@ const { settings: { openLinksInApp: true }, createBrowserTab: vi.fn(), setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan: vi.fn(), - setWorkspacePortScanForKey: vi.fn(), + replaceWorkspacePortScans: vi.fn(), setWorkspacePortScanRefreshing: vi.fn(), recordFeatureInteraction: vi.fn(), workspacePortScansByKey: {} @@ -46,7 +45,7 @@ vi.mock('@/lib/worktree-activation', () => ({ })) vi.mock('@/lib/worktree-runtime-owner', () => ({ - getRuntimeEnvironmentIdForWorktree: () => null + getExecutionHostIdForWorktree: () => 'local' })) vi.mock('@/runtime/runtime-rpc-client', () => ({ diff --git a/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx b/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx index c6d7eb41625..4c07c44ed21 100644 --- a/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx +++ b/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx @@ -1,4 +1,4 @@ -import React, { useCallback, useMemo } from 'react' +import React, { useCallback } from 'react' import { Copy, ExternalLink, FolderOpen, Trash2 } from 'lucide-react' import { toast } from 'sonner' import { Button } from '@/components/ui/button' @@ -15,9 +15,8 @@ import { } from '@/lib/workspace-port-actions' import type { WorkspacePortGroup } from '@/lib/workspace-port-groups' import { useLocalhostLabelRouteForPort } from '@/lib/workspace-port-localhost-label-selector' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' import { useAppStore } from '@/store' -import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' import type { WorkspacePort } from '../../../../shared/workspace-ports' import { translate } from '@/i18n/i18n' @@ -67,6 +66,7 @@ function PortAction({ ) } +/** One port row in the status-bar popover, with open/copy/stop actions on its owner host. */ export function PortRow({ port, activeWorktreeId, @@ -78,21 +78,13 @@ export function PortRow({ }): React.JSX.Element { const settings = useAppStore((s) => s.settings) const localhostLabelRoute = useLocalhostLabelRouteForPort(port) - const runtimeEnvironmentId = useAppStore((s) => - getRuntimeEnvironmentIdForWorktree( - s, - port.kind === 'workspace' ? port.owner.worktreeId : activeWorktreeId - ) - ) const createBrowserTab = useAppStore((s) => s.createBrowserTab) const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction) - const runtimeTarget = useMemo( - () => getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId: runtimeEnvironmentId }), - [runtimeEnvironmentId, settings] + const runtimeTarget = useWorktreeRuntimeTarget( + port.kind === 'workspace' ? port.owner.worktreeId : activeWorktreeId ) const processLabel = port.processName ?? (port.pid ? `PID ${port.pid}` : 'Unknown process') const canStop = canStopWorkspacePort(port) @@ -187,8 +179,7 @@ export function PortRow({ ) const refreshResult = await refreshWorkspacePortScanAfterStop({ runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey, setWorkspacePortScanRefreshing }) @@ -210,8 +201,7 @@ export function PortRow({ port, recordFeatureInteraction, runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ] ) diff --git a/src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts b/src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts new file mode 100644 index 00000000000..71719b11e4f --- /dev/null +++ b/src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts @@ -0,0 +1,95 @@ +/** + * Why `pty.inspectProcess` is not in-flight coalesced (#18419). The host mints one + * `observationEpoch` per request and this reader commits that epoch per read, so a reply shared by + * two overlapping probes reads as a stale replay to the second reader to settle and its would-be + * `live` identity read degrades to `unverifiable`. The pane foreground tracker overlaps its own + * probes on purpose (`cancelPendingRead` bumps the generation but lets the in-flight probe finish, + * then reissues after a 350 ms settle), so that path is reachable. The provider-side ratchet that + * fails if the dedupe returns lives in `src/main/providers/ssh-pty-inspect-observation-identity.test.ts`. + */ +import { describe, expect, it } from 'vitest' +import { createPaneForegroundProcessReader } from './pane-foreground-process-reader' + +const CONNECTION_ID = 'conn-1' +const RELAY_PTY_ID = 'pty-1' +const APP_PTY_ID = `ssh:${CONNECTION_ID}@@${RELAY_PTY_ID}` +const INCARNATION_ID = 'inc-1' + +/** One host scan per request => one epoch per request. */ +const hostObservation = (observationEpoch: number): unknown => ({ + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'live', + processName: 'claude', + ptyId: RELAY_PTY_ID, + ptyIncarnationId: INCARNATION_ID, + authorityGeneration: 'gen-1', + observationEpoch, + capturedAgeMs: 0, + fence: { + platform: 'posix', + shellPid: 100, + shellStartTime: '1000', + tty: '/dev/pts/3', + foregroundPgid: 200 + } + } +}) + +/** Holds every probe open so the tracker's supersede-and-reissue pair really overlaps. */ +function createOverlappingReader(replies: { shared: boolean }): { + readProcess: ReturnType + settle: (index: number) => void +} { + const resolvers: ((value: unknown) => void)[] = [] + return { + // One reader instance per pane, exactly as the foreground tracker holds it. + readProcess: createPaneForegroundProcessReader({ + readForegroundProcess: () => new Promise((resolve) => resolvers.push(resolve)) as never, + isRemotePtyId: () => true, + getExpectedIncarnationId: () => INCARNATION_ID + }), + // `shared` models what an in-flight dedupe would do: every joiner gets one host observation. + settle: (index) => resolvers[index]?.(hostObservation(replies.shared ? 1 : index + 1)) + } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +describe('pane foreground inspect observation identity', () => { + it('keeps a reissued read `live` when it overlaps the probe it superseded', async () => { + const { readProcess, settle } = createOverlappingReader({ shared: false }) + + // The tracker cancels the first read (generation bump) but lets it run to completion, then + // reissues after the settle window — so both are in flight against the same pane. + const superseded = readProcess(APP_PTY_ID, false) + const reissued = readProcess(APP_PTY_ID, false) + await flush() + + // The superseded read's continuation commits its epoch first. + settle(0) + expect((await superseded).remoteEvidenceVerdict).toBe('live') + + settle(1) + const result = await reissued + expect(result.remoteEvidenceVerdict).toBe('live') + expect(result.processName).toBe('claude') + }) + + it('degrades the second overlapping read to `unverifiable` when one observation is shared', async () => { + const { readProcess, settle } = createOverlappingReader({ shared: true }) + + const superseded = readProcess(APP_PTY_ID, false) + const reissued = readProcess(APP_PTY_ID, false) + await flush() + + settle(0) + expect((await superseded).remoteEvidenceVerdict).toBe('live') + + settle(1) + const result = await reissued + expect(result.remoteEvidenceVerdict).toBe('unverifiable') + expect(result.processName).toBeNull() + }) +}) diff --git a/src/renderer/src/components/terminal/pty-running-work-probe.ts b/src/renderer/src/components/terminal/pty-running-work-probe.ts new file mode 100644 index 00000000000..609b71f12ca --- /dev/null +++ b/src/renderer/src/components/terminal/pty-running-work-probe.ts @@ -0,0 +1,88 @@ +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { inspectRuntimeTerminalProcess } from '@/runtime/runtime-terminal-inspection' +import { isRemoteExecutionHostPtyId } from '../../../../shared/remote-execution-host-pty-id' +import { isClientOnlyUnverifiableInspection } from '../../../../shared/terminal-process-inspection' + +/** + * One probe answer in the fixed `live` / `unverifiable` / `exited` vocabulary of + * `docs/reference/ssh-execution-boundary.md`. `exited` is only ever produced by a host that + * answered; every failure to reach the owner — a rejection, a closed transport, or a deadline + * that expired first — stays `unverifiable`, because loss of contact is not evidence of death. + */ +export type PtyRunningWorkVerdict = 'live' | 'unverifiable' | 'exited' + +export type PtyRunningWorkProbe = { + ptyId: string + verdict: PtyRunningWorkVerdict + /** Why the owner could not be observed. Only set for `unverifiable`. */ + reason?: string + /** The deadline expired before this pty's probe answered at all. */ + timedOut: boolean + /** The pty is owned by a remote execution host (relay runtime or app SSH). */ + remote: boolean +} + +type ProbeSettings = Pick | null | undefined + +/** + * Probes every pty for running work and resolves at whichever comes first: every answer, or the + * deadline. Never rejects, and never reports a pty it did not hear back about as idle. + * + * Callers own the policy. This owns only the measurement, so the tab-close guard and the + * window-close guard cannot drift apart on what an unanswered remote host means. + */ +export async function probePtyRunningWork( + settings: ProbeSettings, + ptyIds: readonly string[], + options: { timeoutMs: number } +): Promise { + if (ptyIds.length === 0) { + return [] + } + const probes: PtyRunningWorkProbe[] = ptyIds.map((ptyId) => ({ + ptyId, + verdict: 'unverifiable', + reason: 'probe_deadline', + timedOut: true, + remote: isRemoteExecutionHostPtyId(ptyId) + })) + + const settle = Promise.all( + ptyIds.map(async (ptyId, index) => { + const probe = probes[index] + if (!probe) { + return + } + try { + const inspection = await inspectRuntimeTerminalProcess(settings, ptyId) + probe.timedOut = false + if (isClientOnlyUnverifiableInspection(inspection)) { + probe.verdict = 'unverifiable' + probe.reason = inspection.reason + return + } + probe.verdict = inspection.hasChildProcesses ? 'live' : 'exited' + delete probe.reason + } catch { + // Why: `inspectRuntimeTerminalProcess` already maps every failure it can classify onto a + // reason; an unclassified throw is still a failure to observe, so it stays unverifiable. + probe.timedOut = false + probe.verdict = 'unverifiable' + probe.reason = 'probe_failed' + } + }) + ) + + let deadline: ReturnType | undefined + try { + await Promise.race([ + settle, + new Promise((resolve) => { + deadline = setTimeout(resolve, options.timeoutMs) + }) + ]) + } finally { + clearTimeout(deadline) + } + return probes +} diff --git a/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts b/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts index d63c69df7c9..165119f8b31 100644 --- a/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts +++ b/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts @@ -46,10 +46,12 @@ function visibleRequest() { return useRunningTerminalCloseConfirmStore.getState().runningTerminalCloseConfirm } +// Drains pending microtasks. The probe resolves through several await points (per-pty inspect, +// the batch join, the deadline race), so this flushes generously rather than counting ticks. async function settleProbe(): Promise { - await Promise.resolve() - await Promise.resolve() - await Promise.resolve() + for (let tick = 0; tick < 12; tick += 1) { + await Promise.resolve() + } } describe('shouldConfirmRunningTerminalClose', () => { @@ -329,6 +331,7 @@ describe('guardRunningTerminalClose', () => { vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() + await settleProbe() expect(onClose).not.toHaveBeenCalled() expect(visibleRequest()).toMatchObject({ terminalTabId: 'tab-1', tabLabel: 'npm run dev' }) @@ -351,6 +354,7 @@ describe('guardRunningTerminalClose', () => { guard() vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() + await settleProbe() expect(visibleRequest()?.copyKind).toBe('agent') }) @@ -368,6 +372,7 @@ describe('guardRunningTerminalClose', () => { guard(onClose) vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() + await settleProbe() requestSpy.mockRestore() expect(onClose).toHaveBeenCalledTimes(1) @@ -385,6 +390,7 @@ describe('guardRunningTerminalClose', () => { vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() await settleProbe() + await settleProbe() expect(onClose).not.toHaveBeenCalled() useRunningTerminalCloseConfirmStore.getState().confirmRunningTerminalClose() diff --git a/src/renderer/src/components/terminal/running-terminal-close-guard.ts b/src/renderer/src/components/terminal/running-terminal-close-guard.ts index 881cd262353..6bb9ff5a582 100644 --- a/src/renderer/src/components/terminal/running-terminal-close-guard.ts +++ b/src/renderer/src/components/terminal/running-terminal-close-guard.ts @@ -1,10 +1,9 @@ import { useAppStore } from '@/store' -import { inspectRuntimeTerminalProcess } from '@/runtime/runtime-terminal-inspection' import { useRunningTerminalCloseConfirmStore } from '@/store/running-terminal-close-confirm' import type { TerminalTabCloseReason } from '@/store/slices/terminal-tab-retirement' import type { AppState } from '@/store/types' import { resolveBusyPtyCloseCopyKind } from './terminal-close-copy-kind' -import { isClientOnlyUnverifiableInspection } from '../../../../shared/terminal-process-inspection' +import { probePtyRunningWork } from './pty-running-work-probe' export type RunningTerminalCloseGuardOptions = { force?: boolean @@ -44,7 +43,7 @@ export function shouldConfirmRunningTerminalClose( * the store's own teardown collector unions both for exactly that reason — reading only * the map would let a close slip through the window with no prompt. A stale id costs * nothing: its probe fails and the guard falls open. */ -function collectTabPtyIds( +export function collectTabPtyIds( state: Pick, terminalTabId: string ): string[] { @@ -112,44 +111,33 @@ export function guardRunningTerminalClose(params: { decided = true } - const probeTimeout = setTimeout(() => { - try { + void probePtyRunningWork(settings, ptyIds, { timeoutMs: RUNNING_CLOSE_PROBE_TIMEOUT_MS }) + .then((probes) => { + if (decided) { + return + } // Why: a probe that has not answered yet is unknown, not idle. Ask, treating every pty // as a candidate, so a degraded relay costs a click instead of a killed remote command. - confirmClose(ptyIds) - } catch { - closeNow() - } - }, RUNNING_CLOSE_PROBE_TIMEOUT_MS) - - void Promise.allSettled(ptyIds.map((ptyId) => inspectRuntimeTerminalProcess(settings, ptyId))) - .then((results) => { - clearTimeout(probeTimeout) - if (decided) { + if (probes.some((probe) => probe.timedOut)) { + confirmClose(ptyIds) return } // Why: fail open on an *answered* probe, matching the Cmd+W pane path — a rejection // (wedged relay, legacy provider) or a stale remote handle is not evidence of a live // child, and a close button that silently does nothing is worse than closing a busy tab. - const busyPtyIds = ptyIds.filter((_, index) => { - const result = results[index] - return ( - result?.status === 'fulfilled' && - !isClientOnlyUnverifiableInspection(result.value) && - result.value.hasChildProcesses - ) - }) + const busyPtyIds = probes + .filter((probe) => probe.verdict === 'live') + .map((probe) => probe.ptyId) if (busyPtyIds.length === 0) { closeNow() return } confirmClose(busyPtyIds) }) - // Why: allSettled never rejects, so this only fires when the decision above throws (a + // Why: the probe never rejects, so this only fires when the decision above throws (a // copy-kind lookup, a store subscriber). Without it the tab would silently never close // and the user would get no feedback at all; the pane path it replaced had this catch. .catch(() => { - clearTimeout(probeTimeout) closeNow() }) } diff --git a/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts b/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts index 066c5795ef9..9dfe20ef501 100644 --- a/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts +++ b/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts @@ -87,10 +87,12 @@ function visibleRequest() { return useRunningTerminalCloseConfirmStore.getState().runningTerminalCloseConfirm } +// Drains pending microtasks. The probe resolves through several await points (per-pty inspect, +// the batch join, the deadline race), so this flushes generously rather than counting ticks. async function settleProbe(): Promise { - await Promise.resolve() - await Promise.resolve() - await Promise.resolve() + for (let tick = 0; tick < 12; tick += 1) { + await Promise.resolve() + } } describe('closeTerminalTab running-process confirmation', () => { diff --git a/src/renderer/src/components/terminal/window-close-running-work.test.ts b/src/renderer/src/components/terminal/window-close-running-work.test.ts new file mode 100644 index 00000000000..77dc4ad745b --- /dev/null +++ b/src/renderer/src/components/terminal/window-close-running-work.test.ts @@ -0,0 +1,231 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { getStateMock, inspectRuntimeTerminalProcessMock } = vi.hoisted(() => ({ + getStateMock: vi.fn(), + inspectRuntimeTerminalProcessMock: vi.fn() +})) + +vi.mock('@/store', () => ({ + useAppStore: { getState: getStateMock } +})) + +vi.mock('@/runtime/runtime-terminal-inspection', () => ({ + inspectRuntimeTerminalProcess: inspectRuntimeTerminalProcessMock +})) + +import { + assessWindowCloseRunningWork, + WINDOW_CLOSE_PROBE_TIMEOUT_MS +} from './window-close-running-work' + +const LOCAL_PTY = 'pty-local' +const SSH_PTY = 'ssh:openclaw@@pty-7' +const RUNTIME_PTY = 'remote:env-1@@handle-1' +/** A runtime pty minted without an owner id. Still someone else's machine. */ +const OWNERLESS_RUNTIME_PTY = 'remote:handle-2' + +const BUSY = { + foregroundProcess: 'pnpm build', + hasChildProcesses: true, + foregroundProcessEvidence: {} +} +const IDLE = { foregroundProcess: 'bash', hasChildProcesses: false, foregroundProcessEvidence: {} } +const UNVERIFIABLE = { + foregroundProcess: null, + hasChildProcesses: false, + verdict: 'unverifiable', + reason: 'transport_loss' +} + +/** One worktree, one tab, owning `ptyIds`. */ +function setState(ptyIds: string[]): void { + getStateMock.mockReturnValue({ + settings: { activeRuntimeEnvironmentId: null }, + tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] }, + ptyIdsByTabId: { 'tab-1': ptyIds }, + terminalLayoutsByTabId: {} + }) +} + +/** Answers each pty id from `byPtyId`; anything unlisted never settles. */ +function answerWith(byPtyId: Record): void { + inspectRuntimeTerminalProcessMock.mockImplementation((_settings: unknown, ptyId: string) => + ptyId in byPtyId ? Promise.resolve(byPtyId[ptyId]) : new Promise(() => {}) + ) +} + +beforeEach(() => { + vi.clearAllMocks() +}) + +afterEach(() => { + vi.useRealTimers() +}) + +describe('assessWindowCloseRunningWork', () => { + it('warns about a live process on an SSH host (F15: remote work was filtered out entirely)', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('warns on quit about a live process on an SSH host', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('warns on quit about a live process on a paired runtime host', async () => { + setState([RUNTIME_PTY]) + answerWith({ [RUNTIME_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('counts an owner-less remote pty as remote work', async () => { + setState([OWNERLESS_RUNTIME_PTY]) + answerWith({ [OWNERLESS_RUNTIME_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + // The crux of docs/reference/ssh-execution-boundary.md: an unreachable host is `unverifiable`, + // and quitting on `unverifiable` as though it were `exited` is what orphans live remote work. + it('warns rather than quitting silently when a remote host answers unverifiable', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: UNVERIFIABLE }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'unverifiable' + }) + }) + + it('warns rather than quitting silently when a remote probe throws', async () => { + setState([SSH_PTY]) + inspectRuntimeTerminalProcessMock.mockRejectedValue(new Error('relay wedged')) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'unverifiable' + }) + }) + + it('stops waiting at the budget and warns, so an unreachable host cannot hang the quit', async () => { + setState([SSH_PTY]) + answerWith({}) + vi.useFakeTimers() + + const pending = assessWindowCloseRunningWork({ isQuitting: true }) + await vi.advanceTimersByTimeAsync(WINDOW_CLOSE_PROBE_TIMEOUT_MS) + + await expect(pending).resolves.toEqual({ kind: 'unverifiable' }) + }) + + it('does not resolve before the budget expires', async () => { + setState([SSH_PTY]) + answerWith({}) + vi.useFakeTimers() + const settled = vi.fn() + + void assessWindowCloseRunningWork({ isQuitting: true }).then(settled) + await vi.advanceTimersByTimeAsync(WINDOW_CLOSE_PROBE_TIMEOUT_MS - 1) + + expect(settled).not.toHaveBeenCalled() + }) + + it('does not warn when the owning remote host reports an idle shell', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: IDLE }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'none' + }) + }) + + it('reports a live process even when a sibling remote pane is only unverifiable', async () => { + setState([SSH_PTY, RUNTIME_PTY]) + answerWith({ [SSH_PTY]: UNVERIFIABLE, [RUNTIME_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('still warns about a live local process when closing the window', async () => { + setState([LOCAL_PTY]) + answerWith({ [LOCAL_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'running' + }) + }) + + // A local probe has no transport to lose, so its failure means the pty is gone — unlike a + // remote host going quiet, it is not a reason to hold up the close. + it('does not warn when only a local probe is unverifiable', async () => { + setState([LOCAL_PTY]) + answerWith({ [LOCAL_PTY]: UNVERIFIABLE }) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'none' + }) + }) + + // #524 decided quitting is an unambiguous instruction to end this machine's processes. It is + // not an instruction to end execution on someone else's, which is why remote still warns above. + it('leaves local-only quit unprompted, and never probes for it', async () => { + setState([LOCAL_PTY]) + answerWith({ [LOCAL_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'none' + }) + expect(inspectRuntimeTerminalProcessMock).not.toHaveBeenCalled() + }) + + it('probes a pane the layout has bound before the liveness map caught up', async () => { + getStateMock.mockReturnValue({ + settings: { activeRuntimeEnvironmentId: null }, + tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] }, + ptyIdsByTabId: {}, + terminalLayoutsByTabId: { 'tab-1': { ptyIdsByLeafId: { leaf: SSH_PTY } } } + }) + answerWith({ [SSH_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('probes each pty once when the map and the layout name the same one', async () => { + getStateMock.mockReturnValue({ + settings: { activeRuntimeEnvironmentId: null }, + tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] }, + ptyIdsByTabId: { 'tab-1': [SSH_PTY] }, + terminalLayoutsByTabId: { 'tab-1': { ptyIdsByLeafId: { leaf: SSH_PTY } } } + }) + answerWith({ [SSH_PTY]: IDLE }) + + await assessWindowCloseRunningWork({ isQuitting: true }) + + expect(inspectRuntimeTerminalProcessMock).toHaveBeenCalledTimes(1) + }) + + it('closes without probing when no workspace owns a pty', async () => { + setState([]) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'none' + }) + expect(inspectRuntimeTerminalProcessMock).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/terminal/window-close-running-work.ts b/src/renderer/src/components/terminal/window-close-running-work.ts new file mode 100644 index 00000000000..427ee72dae6 --- /dev/null +++ b/src/renderer/src/components/terminal/window-close-running-work.ts @@ -0,0 +1,70 @@ +import { useAppStore } from '@/store' +import { isRemoteExecutionHostPtyId } from '../../../../shared/remote-execution-host-pty-id' +import { collectTabPtyIds } from './running-terminal-close-guard' +import { probePtyRunningWork } from './pty-running-work-probe' + +/** + * Upper bound on how long closing the window or quitting may wait on the probes. + * + * Shorter than the tab-close guard's 4s because quit is time-sensitive in a way one tab close is + * not: the user has already asked to leave, and a quit that stalls on an unreachable host is its + * own bug. A healthy local inspect answers in single-digit milliseconds and a healthy remote one + * is a single RPC round-trip on an already-open mux channel, so this leaves roughly 3x headroom + * over a slow-but-live transcontinental host while capping the worst case — a host that is simply + * gone — at ~1.5s instead of the 15s RPC timeout the probe would otherwise inherit. + * + * Expiry raises the prompt rather than quitting silently: an unanswered probe is `unverifiable`, + * and `unverifiable` is never evidence that remote work has stopped. + */ +export const WINDOW_CLOSE_PROBE_TIMEOUT_MS = 1_500 + +/** Which warning the close should raise, if any. */ +export type WindowCloseRunningWork = + /** Every pty that mattered answered, and none had children. */ + | { kind: 'none' } + /** An owning host reported a live child process. */ + | { kind: 'running' } + /** A remote execution host could not be observed, so its work may still be live. */ + | { kind: 'unverifiable' } + +/** + * Decides whether a window close or quit should stop and ask. + * + * Two deliberate asymmetries: + * + * - **Quit only considers remote ptys.** Quitting is an unambiguous instruction to end this + * machine's processes (#524), but it is not an instruction to end execution on someone else's: + * the client detaches while the relay keeps running, and a target with a bounded grace period + * then SIGKILLs that work once the countdown expires. + * - **Only a remote `unverifiable` warns.** A local probe has no transport to lose, so its failure + * means the pty is gone. A remote one that cannot be reached is the case + * `docs/reference/ssh-execution-boundary.md` exists to protect: loss of contact is not evidence + * of `exited`, so it must fail toward asking rather than toward a silent quit. + */ +export async function assessWindowCloseRunningWork(params: { + isQuitting: boolean +}): Promise { + const state = useAppStore.getState() + const ptyIds = new Set( + Object.values(state.tabsByWorktree) + .flatMap((worktreeTabs) => worktreeTabs ?? []) + .flatMap((tab) => collectTabPtyIds(state, tab.id)) + ) + const candidatePtyIds = params.isQuitting + ? [...ptyIds].filter(isRemoteExecutionHostPtyId) + : [...ptyIds] + if (candidatePtyIds.length === 0) { + return { kind: 'none' } + } + + const probes = await probePtyRunningWork(state.settings, candidatePtyIds, { + timeoutMs: WINDOW_CLOSE_PROBE_TIMEOUT_MS + }) + if (probes.some((probe) => probe.verdict === 'live')) { + return { kind: 'running' } + } + if (probes.some((probe) => probe.remote && probe.verdict === 'unverifiable')) { + return { kind: 'unverifiable' } + } + return { kind: 'none' } +} diff --git a/src/renderer/src/components/use-terminal-editor-close-foundation.ts b/src/renderer/src/components/use-terminal-editor-close-foundation.ts index 74849e3f5a2..2aceced683d 100644 --- a/src/renderer/src/components/use-terminal-editor-close-foundation.ts +++ b/src/renderer/src/components/use-terminal-editor-close-foundation.ts @@ -1,8 +1,9 @@ import { useCallback, useRef, useState } from 'react' -import { useAppStore } from '../store' -import { getConnectionId } from '../lib/connection-context' -import { isRemoteRuntimePtyId } from '@/runtime/runtime-terminal-inspection' import { CLOSE_DIALOG_DEBOUNCE_MS } from './terminal-workspace-model' +import { + assessWindowCloseRunningWork, + type WindowCloseRunningWork +} from './terminal/window-close-running-work' import type { TerminalWorkspaceProjectionController } from './use-terminal-workspace-projection' import { runWithWindowCloseCheckpointScope } from './window-close-request-coordinator' import { showShutdownCheckpointFailureToast } from '@/lib/shutdown-checkpoint-failure-toast' @@ -27,6 +28,11 @@ export function useTerminalEditorCloseFoundation( closeDialogDebounceTimersRef.current.add(timer) }, []) const [windowCloseDialogOpen, setWindowCloseDialogOpen] = useState(false) + // Why: "running" and "could not reach the host" are different claims, and telling the user + // processes are running when the truth is that a host went quiet is the fabricated certainty + // docs/reference/ssh-execution-boundary.md forbids. + const [windowCloseDialogKind, setWindowCloseDialogKind] = + useState>('running') const windowCloseAfterDirtyRef = useRef<{ isQuitting: boolean } | null>(null) const confirmNativeWindowClose = useCallback(() => { @@ -46,33 +52,21 @@ export function useTerminalEditorCloseFoundation( const proceedToNativeWindowClose = useCallback( (isQuitting: boolean) => { - if (!isQuitting) { - const state = useAppStore.getState() - const localPtyIds = Object.entries(state.tabsByWorktree).flatMap( - ([worktreeId, worktreeTabs]) => { - const connectionId = getConnectionId(worktreeId) - if (connectionId !== null) { - return [] - } - return worktreeTabs - .flatMap((tab) => state.ptyIdsByTabId[tab.id] ?? []) - .filter((ptyId) => !isRemoteRuntimePtyId(ptyId)) + void assessWindowCloseRunningWork({ isQuitting }) + .then((runningWork) => { + if (runningWork.kind === 'none') { + confirmNativeWindowClose() + return } - ) - if (localPtyIds.length > 0) { - void Promise.all(localPtyIds.map((id) => window.api.pty.hasChildProcesses(id))).then( - (results) => { - if (results.some(Boolean)) { - setWindowCloseDialogOpen(true) - } else { - confirmNativeWindowClose() - } - } - ) - return - } - } - confirmNativeWindowClose() + setWindowCloseDialogKind(runningWork.kind) + setWindowCloseDialogOpen(true) + }) + // Why: the assessment must never be able to trap the window. A thrown store read is + // not evidence either way, and a close that silently does nothing is unrecoverable + // without SIGKILL, so fall through to the close the user actually asked for. + .catch(() => { + confirmNativeWindowClose() + }) }, [confirmNativeWindowClose] ) @@ -88,6 +82,7 @@ export function useTerminalEditorCloseFoundation( releaseCloseDialogGuardAfterDebounce, windowCloseDialogOpen, setWindowCloseDialogOpen, + windowCloseDialogKind, windowCloseAfterDirtyRef, confirmNativeWindowClose, proceedToNativeWindowClose diff --git a/src/renderer/src/components/use-terminal-editor-close-foundation.window-close.test.tsx b/src/renderer/src/components/use-terminal-editor-close-foundation.window-close.test.tsx new file mode 100644 index 00000000000..d7f3f69d925 --- /dev/null +++ b/src/renderer/src/components/use-terminal-editor-close-foundation.window-close.test.tsx @@ -0,0 +1,103 @@ +// @vitest-environment happy-dom + +/** + * Wiring for the window-close/quit running-work warning. The policy in + * `terminal/window-close-running-work.ts` is inert unless `proceedToNativeWindowClose` actually + * consults it, so pin that it does — and that a warning stops the native close rather than + * confirming it. + */ +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { assessWindowCloseRunningWorkMock, confirmWindowCloseMock } = vi.hoisted(() => ({ + assessWindowCloseRunningWorkMock: vi.fn(), + confirmWindowCloseMock: vi.fn() +})) + +vi.mock('./terminal/window-close-running-work', () => ({ + assessWindowCloseRunningWork: assessWindowCloseRunningWorkMock +})) +vi.mock('./window-close-request-coordinator', () => ({ + runWithWindowCloseCheckpointScope: (fn: () => unknown) => fn() +})) +vi.mock('@/lib/shutdown-checkpoint-failure-toast', () => ({ + showShutdownCheckpointFailureToast: vi.fn() +})) + +const { useTerminalEditorCloseFoundation } = await import('./use-terminal-editor-close-foundation') + +const controller = { openFiles: [] } as unknown as Parameters< + typeof useTerminalEditorCloseFoundation +>[0] + +function mountFoundation() { + return renderHook(() => useTerminalEditorCloseFoundation(controller)) +} + +beforeEach(() => { + vi.clearAllMocks() + Object.assign(globalThis, { + window: Object.assign(globalThis.window, { + api: { ui: { confirmWindowClose: confirmWindowCloseMock } } + }) + }) +}) + +afterEach(() => { + cleanup() +}) + +describe('proceedToNativeWindowClose', () => { + it('asks the running-work policy about the quit rather than assuming it is safe', async () => { + assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'none' }) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(true) + }) + + expect(assessWindowCloseRunningWorkMock).toHaveBeenCalledWith({ isQuitting: true }) + expect(confirmWindowCloseMock).toHaveBeenCalledTimes(1) + expect(result.current.windowCloseDialogOpen).toBe(false) + }) + + it('raises the dialog and does not close when a host reports live work', async () => { + assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'running' }) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(true) + }) + + expect(result.current.windowCloseDialogOpen).toBe(true) + expect(result.current.windowCloseDialogKind).toBe('running') + expect(confirmWindowCloseMock).not.toHaveBeenCalled() + }) + + it('raises the unverifiable copy when a remote host could not be reached', async () => { + assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'unverifiable' }) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(true) + }) + + expect(result.current.windowCloseDialogOpen).toBe(true) + expect(result.current.windowCloseDialogKind).toBe('unverifiable') + expect(confirmWindowCloseMock).not.toHaveBeenCalled() + }) + + // Why: a thrown assessment is not evidence either way, and a close that silently does nothing + // leaves SIGKILL as the user's only exit. + it('falls through to the close when the assessment throws', async () => { + assessWindowCloseRunningWorkMock.mockRejectedValue(new Error('store blew up')) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(false) + }) + + expect(confirmWindowCloseMock).toHaveBeenCalledTimes(1) + expect(result.current.windowCloseDialogOpen).toBe(false) + }) +}) diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 87fcb50a2a1..f2da58bffb3 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -2251,7 +2251,8 @@ "Terminal": { "73768427cf": "Close", "f82e9f02df": "Cancel", - "7958465754": "There are local terminals with running processes. Close the window anyway?", + "7958465754": "There are terminals with running processes. Close the window anyway?", + "b7c1f0a934": "A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?", "2fa9c69ff3": "Close Window?", "cd51e28d8b": "Save", "0037b21794": "Don't Save", diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 9999634daee..98b07917880 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -1924,7 +1924,7 @@ "Terminal": { "73768427cf": "Cerrar", "f82e9f02df": "Cancelar", - "7958465754": "Hay terminales locales con procesos en ejecución. ¿Cerrar la ventana de todos modos?", + "7958465754": "Hay terminales con procesos en ejecución. ¿Cerrar la ventana de todos modos?", "2fa9c69ff3": "¿Cerrar ventana?", "cd51e28d8b": "Guardar", "0037b21794": "No guardar", diff --git a/src/renderer/src/i18n/locales/fr.json b/src/renderer/src/i18n/locales/fr.json index 8eff250a820..5bf316d36f7 100644 --- a/src/renderer/src/i18n/locales/fr.json +++ b/src/renderer/src/i18n/locales/fr.json @@ -2087,7 +2087,7 @@ "Terminal": { "73768427cf": "Fermer", "f82e9f02df": "Annuler", - "7958465754": "Des terminaux locaux exécutent des processus. Fermer quand même la fenêtre ?", + "7958465754": "Des terminaux exécutent des processus. Fermer quand même la fenêtre ?", "2fa9c69ff3": "Fermer la fenêtre ?", "cd51e28d8b": "Enregistrer", "0037b21794": "Ne pas enregistrer", diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index b3c30da9446..4dc81b9201d 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -1924,7 +1924,7 @@ "Terminal": { "73768427cf": "閉じる", "f82e9f02df": "キャンセル", - "7958465754": "プロセスが実行中のローカルターミナルがあります。このままウィンドウを閉じますか?", + "7958465754": "プロセスが実行中のターミナルがあります。このままウィンドウを閉じますか?", "2fa9c69ff3": "ウィンドウを閉じますか?", "cd51e28d8b": "保存", "0037b21794": "保存しないでください", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index ac75cca2209..d9d822b8fc3 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -1929,7 +1929,7 @@ "Terminal": { "73768427cf": "닫기", "f82e9f02df": "취소", - "7958465754": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?", + "7958465754": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?", "2fa9c69ff3": "창을 닫으시겠습니까?", "cd51e28d8b": "저장", "0037b21794": "저장하지 않음", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 7a3b47c8f2a..46dc592dd2c 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -1927,7 +1927,7 @@ "Terminal": { "73768427cf": "关闭", "f82e9f02df": "取消", - "7958465754": "有正在运行的进程的本地终端。还是关窗吧?", + "7958465754": "有正在运行的进程的终端。还是关窗吧?", "2fa9c69ff3": "关闭窗口?", "cd51e28d8b": "保存", "0037b21794": "不保存", diff --git a/src/renderer/src/lib/workspace-port-actions.ts b/src/renderer/src/lib/workspace-port-actions.ts index cca201234a4..cf745d3e25f 100644 --- a/src/renderer/src/lib/workspace-port-actions.ts +++ b/src/renderer/src/lib/workspace-port-actions.ts @@ -7,7 +7,6 @@ import { type RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' -import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host' import type { WorkspacePort, WorkspacePortKillResult, @@ -22,6 +21,11 @@ import { RUNTIME_BROWSER_UNAVAILABLE_MESSAGE } from './client-creation-action-po export { addressForPort } from './workspace-port-urls' const WORKSPACE_PORT_STOP_SETTLE_MS = 500 +const WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON = + 'Workspace ports are unavailable for this execution host.' + +/** Projection key for the merged multi-host view; never a per-host scan key. */ +export const WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY = 'all-hosts:all' export function canStopWorkspacePort( port: WorkspacePort @@ -33,13 +37,17 @@ type BrowserTabCreator = ReturnType['createBrowserT type RemoteBrowserPageHandleSetter = ReturnType< typeof useAppStore.getState >['setRemoteBrowserPageHandle'] -type WorkspacePortScanSetter = ReturnType['setWorkspacePortScan'] -type WorkspacePortScanByKeySetter = ReturnType< - typeof useAppStore.getState ->['setWorkspacePortScanForKey'] type WorkspacePortScanRefreshingSetter = ReturnType< typeof useAppStore.getState >['setWorkspacePortScanRefreshing'] +type ReplaceWorkspacePortScansSetter = ReturnType< + typeof useAppStore.getState +>['replaceWorkspacePortScans'] + +export type WorkspacePortScanPublisher = { + replaceWorkspacePortScans: ReplaceWorkspacePortScansSetter + getWorkspacePortScansByKey: () => Record +} function delay(ms: number): Promise { return new Promise((resolve) => window.setTimeout(resolve, ms)) @@ -94,12 +102,15 @@ export function goToWorkspacePortOwner(port: WorkspacePort): boolean { export async function openWorkspacePortInBrowser(args: { port: WorkspacePort activeWorktreeId?: string | null - runtimeTarget: RuntimeClientTarget + runtimeTarget: RuntimeClientTarget | null createBrowserTab: BrowserTabCreator setRemoteBrowserPageHandle: RemoteBrowserPageHandleSetter openInOrcaBrowser?: boolean localhostLabelRoute?: LocalhostWorktreeLabelRoute | null }): Promise<{ ok: true } | { ok: false; reason: string }> { + if (!args.runtimeTarget) { + return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON } + } const rawUrl = browserUrlForPort(args.port) let url = rawUrl if (args.runtimeTarget.kind === 'local' && args.localhostLabelRoute) { @@ -166,22 +177,38 @@ export async function openWorkspacePortInBrowser(args: { } } -export async function refreshWorkspacePortScanAfterStop(args: { - runtimeTarget: RuntimeClientTarget - setWorkspacePortScan: WorkspacePortScanSetter - setWorkspacePortScanForKey?: WorkspacePortScanByKeySetter - setWorkspacePortScanRefreshing: WorkspacePortScanRefreshingSetter - getWorkspacePortScansByKey?: () => Record -}): Promise<{ ok: true } | { ok: false; reason: string }> { +/** + * Stores one host's scan and republishes the aggregate the status bar reads. + * Why: a single-host publish used to overwrite that aggregate, so every other + * host's ports vanished from the count until the next background poll. One + * replaceWorkspacePortScans update (not setWorkspacePortScan) keeps the synthetic + * all-hosts key out of workspacePortScansByKey, where re-merging it would + * duplicate rows — and notifies subscribers once instead of twice for one scan. + */ +export function publishWorkspacePortScanForHost( + args: WorkspacePortScanPublisher & { scanKey: string; scan: WorkspacePortScanResult } +): void { + const scansByKey = { ...args.getWorkspacePortScansByKey(), [args.scanKey]: args.scan } + const merged = mergeWorkspacePortScans(scansByKey) + args.replaceWorkspacePortScans(scansByKey, { + key: Object.keys(scansByKey).length > 1 ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : args.scanKey, + result: merged ?? args.scan + }) +} + +/** Re-scans one host after a port stop (immediately, then settled) and republishes the aggregate. */ +export async function refreshWorkspacePortScanAfterStop( + args: WorkspacePortScanPublisher & { + runtimeTarget: RuntimeClientTarget | null + setWorkspacePortScanRefreshing: WorkspacePortScanRefreshingSetter + } +): Promise<{ ok: true } | { ok: false; reason: string }> { + if (!args.runtimeTarget) { + return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON } + } const scanKey = workspacePortScanKeyForTarget(args.runtimeTarget) const publishScan = (scan: WorkspacePortScanResult): void => { - args.setWorkspacePortScanForKey?.(scanKey, scan) - const currentScans = args.getWorkspacePortScansByKey?.() ?? {} - const merged = mergeWorkspacePortScans({ ...currentScans, [scanKey]: scan }) - args.setWorkspacePortScan({ - key: merged && Object.keys(currentScans).length > 0 ? 'all-hosts:all' : scanKey, - result: merged ?? scan - }) + publishWorkspacePortScanForHost({ ...args, scanKey, scan }) } args.setWorkspacePortScanRefreshing(true) try { @@ -216,19 +243,6 @@ export function workspacePortRuntimeTargetKey(target: RuntimeClientTarget): stri return target.kind === 'local' ? 'local' : `environment:${target.environmentId}` } -export function runtimeTargetForExecutionHostId( - hostId: ExecutionHostId -): RuntimeClientTarget | null { - const parsed = parseExecutionHostId(hostId) - if (parsed?.kind === 'local') { - return { kind: 'local' } - } - if (parsed?.kind === 'runtime') { - return { kind: 'environment', environmentId: parsed.environmentId } - } - return null -} - export function workspacePortScanKeyForTarget(target: RuntimeClientTarget): string { return `${workspacePortRuntimeTargetKey(target)}:all` } @@ -295,9 +309,12 @@ export async function scanWorkspacePortsForTarget( } export async function killWorkspacePortForTarget( - target: RuntimeClientTarget, + target: RuntimeClientTarget | null, args: { repoId: string; pid: number; port: number } ): Promise { + if (!target) { + return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON } + } if (target.kind === 'local') { return window.api.workspacePorts.kill(args) } diff --git a/src/renderer/src/lib/workspace-port-host-availability.test.ts b/src/renderer/src/lib/workspace-port-host-availability.test.ts new file mode 100644 index 00000000000..7652c65a21a --- /dev/null +++ b/src/renderer/src/lib/workspace-port-host-availability.test.ts @@ -0,0 +1,148 @@ +import { describe, expect, it } from 'vitest' +import type { WorkspacePortScanResult } from '../../../shared/workspace-ports' +import { + getUnavailableWorkspacePortHosts, + workspacePortHostForScanKey +} from './workspace-port-host-availability' + +function scan(overrides: Partial = {}): WorkspacePortScanResult { + return { platform: 'linux', scannedAt: 1, ports: [], ...overrides } +} + +describe('getUnavailableWorkspacePortHosts', () => { + it('reports the failed host while another host still answers', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan(), + 'environment:env-1:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'environment:env-1:all', + host: { kind: 'environment', environmentId: 'env-1' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + it('reports the local host as a local host ref, not an absent environment id', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ unavailableReason: 'lsof is unavailable' }), + 'environment:env-1:all': scan() + }) + ).toEqual([ + { + scanKey: 'local:all', + host: { kind: 'local' }, + platform: 'linux', + reason: 'lsof is unavailable' + } + ]) + }) + + it('keeps colons inside an environment id when parsing the scan key', () => { + // Why: keys are `${targetKey}:all`, so the id runs to the last `:all` — + // splitting on the first colon would truncate ids that contain colons. + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan(), + 'environment:weird:id:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'environment:weird:id:all', + host: { kind: 'environment', environmentId: 'weird:id' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + // Why: total loss of contact is where naming the host matters most — the merged + // projection joins raw internal scan keys, so it cannot name them itself. + it('names every host when all of them failed', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ unavailableReason: 'lsof is unavailable' }), + 'environment:env-1:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'local:all', + host: { kind: 'local' }, + platform: 'linux', + reason: 'lsof is unavailable' + }, + { + scanKey: 'environment:env-1:all', + host: { kind: 'environment', environmentId: 'env-1' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + it('names a single failed host', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ unavailableReason: 'lsof is unavailable' }) + }) + ).toEqual([ + { + scanKey: 'local:all', + host: { kind: 'local' }, + platform: 'linux', + reason: 'lsof is unavailable' + } + ]) + }) + + // Why: the synthetic all-hosts projection key must never be labelled as the + // local machine — that would blame the wrong host for a remote failure. + it('marks an unrecognised scan key as an unknown host', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'all-hosts:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'all-hosts:all', + host: { kind: 'unknown' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + // Why: a paired web client's userAgent is not the Orca host's platform, so the + // caller labels the local host from the scan's own platform. + it("carries the failed scan's platform, and null when it is unknown", () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ platform: 'win32', unavailableReason: 'netstat failed' }), + 'environment:env-1:all': scan({ platform: 'unknown', unavailableReason: 'dropped' }) + }).map((entry) => entry.platform) + ).toEqual(['win32', null]) + }) + + it('stays silent when nothing failed', () => { + expect(getUnavailableWorkspacePortHosts({ 'local:all': scan() })).toEqual([]) + expect(getUnavailableWorkspacePortHosts({})).toEqual([]) + }) +}) + +describe('workspacePortHostForScanKey', () => { + it.each([ + ['local:all', { kind: 'local' }], + ['environment:env-1:all', { kind: 'environment', environmentId: 'env-1' }], + ['environment:weird:id:all', { kind: 'environment', environmentId: 'weird:id' }], + ['all-hosts:all', { kind: 'unknown' }], + ['environment::all', { kind: 'unknown' }], + ['environment:env-1', { kind: 'unknown' }], + ['local', { kind: 'unknown' }] + ])('maps %s', (scanKey, expected) => { + expect(workspacePortHostForScanKey(scanKey)).toEqual(expected) + }) +}) diff --git a/src/renderer/src/lib/workspace-port-host-availability.ts b/src/renderer/src/lib/workspace-port-host-availability.ts new file mode 100644 index 00000000000..4c643a5a118 --- /dev/null +++ b/src/renderer/src/lib/workspace-port-host-availability.ts @@ -0,0 +1,65 @@ +import type { WorkspacePortScanResult } from '../../../shared/workspace-ports' + +/** + * Host a per-host scan key points at. `unknown` is kept distinct from `local` so + * an unrecognised key (a synthetic projection key that leaked into the per-host + * map, say) is never mislabelled as a local failure. + */ +export type WorkspacePortHostRef = + | { kind: 'local' } + | { kind: 'environment'; environmentId: string } + | { kind: 'unknown' } + +export type UnavailableWorkspacePortHost = { + scanKey: string + host: WorkspacePortHostRef + /** Platform the failed scan last ran on; drives the local host's label. */ + platform: NodeJS.Platform | null + reason: string +} + +// Why: mirrors workspacePortScanKeyForTarget (`${targetKey}:all`, where the +// target key is `local` or `environment:`) without importing the heavier +// workspace-port-actions module into this pure helper. Splitting on the last +// `:all` keeps environment ids that themselves contain colons intact. +const ENVIRONMENT_SCAN_KEY_PREFIX = 'environment:' +const SCAN_KEY_SUFFIX = ':all' +const LOCAL_SCAN_KEY = `local${SCAN_KEY_SUFFIX}` + +/** Host a per-host scan key names; `unknown` for any other key shape. */ +export function workspacePortHostForScanKey(scanKey: string): WorkspacePortHostRef { + if (scanKey === LOCAL_SCAN_KEY) { + return { kind: 'local' } + } + if (!scanKey.endsWith(SCAN_KEY_SUFFIX) || !scanKey.startsWith(ENVIRONMENT_SCAN_KEY_PREFIX)) { + return { kind: 'unknown' } + } + const environmentId = scanKey.slice( + ENVIRONMENT_SCAN_KEY_PREFIX.length, + scanKey.length - SCAN_KEY_SUFFIX.length + ) + return environmentId ? { kind: 'environment', environmentId } : { kind: 'unknown' } +} + +/** + * Every host whose latest scan failed, named by host rather than by scan key. + * Why: on a remote host "none listening" and "could not look" are different + * answers, and the merged projection collapses both the partial case (no reason + * at all) and the total case (reasons joined with raw internal keys). + */ +export function getUnavailableWorkspacePortHosts( + scansByKey: Record +): UnavailableWorkspacePortHost[] { + return Object.entries(scansByKey).flatMap(([scanKey, scan]) => + scan?.unavailableReason + ? [ + { + scanKey, + host: workspacePortHostForScanKey(scanKey), + platform: scan.platform === 'unknown' ? null : scan.platform, + reason: scan.unavailableReason + } + ] + : [] + ) +} diff --git a/src/renderer/src/lib/workspace-port-scan-debounce.test.ts b/src/renderer/src/lib/workspace-port-scan-debounce.test.ts index bacac7ecacf..747de6672aa 100644 --- a/src/renderer/src/lib/workspace-port-scan-debounce.test.ts +++ b/src/renderer/src/lib/workspace-port-scan-debounce.test.ts @@ -25,6 +25,11 @@ function unavailable(): WorkspacePortScanResult { return { platform: 'unknown', scannedAt: 1, ports: [], unavailableReason: 'scan failed' } } +/** What the ports popover publishes when its own scan fails: reason + last-good ports. */ +function unavailableWithRetainedPorts(portIds: string[]): WorkspacePortScanResult { + return { ...good(portIds), unavailableReason: 'scan failed' } +} + const FAILURE_THRESHOLD = 2 function createHarness(): { @@ -125,6 +130,33 @@ describe('reconcileTransientPortScanFailures', () => { expect(state.has('flaky:all')).toBe(false) }) + // Why: the popover publishes reason + last-good ports the moment its own scan + // fails. Counting that as a spent grace period would drop those ports on the + // very next poll, so the retention would never survive one poll interval. + it('still grants the grace period after a failure that retained its ports', () => { + const { apply, publish } = createHarness() + apply([{ key: 'h:all', result: good(['tcp:3000']) }]) + const popoverResult = unavailableWithRetainedPorts(['tcp:3000']) + publish('h:all', popoverResult) + + const next = apply([{ key: 'h:all', result: unavailable() }]) + + expect(next[0].result).toBe(popoverResult) + expect(next[0].result.ports).toHaveLength(1) + }) + + it('still drops retained ports once failures reach the tolerance', () => { + const { apply, publish } = createHarness() + apply([{ key: 'h:all', result: good(['tcp:3000']) }]) + publish('h:all', unavailableWithRetainedPorts(['tcp:3000'])) + apply([{ key: 'h:all', result: unavailable() }]) + + const next = apply([{ key: 'h:all', result: unavailable() }]) + + expect(next[0].result.ports).toHaveLength(0) + expect(next[0].result.unavailableReason).toBe('scan failed') + }) + it('uses a newer manual result instead of resurrecting stale ports', () => { const { apply, publish } = createHarness() apply([{ key: 'h:all', result: good(['tcp:3000']) }]) diff --git a/src/renderer/src/lib/workspace-port-scan-debounce.ts b/src/renderer/src/lib/workspace-port-scan-debounce.ts index a6281f0ed9f..2677cbb0e52 100644 --- a/src/renderer/src/lib/workspace-port-scan-debounce.ts +++ b/src/renderer/src/lib/workspace-port-scan-debounce.ts @@ -34,10 +34,14 @@ export function reconcileTransientPortScanFailures( return { key, result } } const failures = previousFailures + 1 + // Why: a surface that hit the same failure first (the ports popover) republishes + // the host's last-good ports alongside the reason. Treating that as a spent grace + // period would drop those ports on the very next poll, undoing the retention. + const publishedIsRetainable = + Boolean(publishedResult) && + (!publishedResult.unavailableReason || publishedResult.ports.length > 0) const nextResult = - failures < failureThreshold && publishedResult && !publishedResult.unavailableReason - ? publishedResult - : result + failures < failureThreshold && publishedIsRetainable ? publishedResult : result state.set(key, { consecutiveFailures: failures, publishedResult: nextResult }) return { key, result: nextResult } }) diff --git a/src/renderer/src/lib/workspace-port-scan-publish.test.ts b/src/renderer/src/lib/workspace-port-scan-publish.test.ts new file mode 100644 index 00000000000..2f1386855a8 --- /dev/null +++ b/src/renderer/src/lib/workspace-port-scan-publish.test.ts @@ -0,0 +1,153 @@ +// @vitest-environment happy-dom + +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { WorkspacePortScanResult } from '../../../shared/workspace-ports' + +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorktree: vi.fn() +})) + +vi.mock('@/runtime/runtime-rpc-client', () => ({ + getActiveRuntimeTarget: vi.fn(), + callRuntimeRpc: vi.fn(), + assertRuntimeEnvironmentCapability: vi.fn(), + RuntimeRpcCallError: class RuntimeRpcCallError extends Error { + code?: string + } +})) + +vi.mock('./workspace-port-scan-client', () => ({ + runWorkspacePortScanForTarget: vi.fn() +})) + +const { publishWorkspacePortScanForHost, WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY } = + await import('./workspace-port-actions') +type WorkspacePortScanPublisher = Parameters[0] + +function scanWithPort(port: number, scannedAt: number): WorkspacePortScanResult { + return { + platform: 'linux', + scannedAt, + ports: [ + { + id: `tcp:${port}`, + bindHost: '0.0.0.0', + connectHost: '127.0.0.1', + port, + pid: 100 + port, + processName: 'node', + protocol: 'http', + kind: 'external' + } + ] + } +} + +/** Mirrors the store's replaceWorkspacePortScans semantics: one atomic update. */ +function makeStoreHarness(initial: Record = {}): { + scansByKey: Record + projections: { key: string; result: WorkspacePortScanResult }[] + publisher: Omit +} { + let scansByKey: Record = { ...initial } + const projections: { key: string; result: WorkspacePortScanResult }[] = [] + return { + get scansByKey() { + return scansByKey + }, + projections, + publisher: { + replaceWorkspacePortScans: ( + nextScansByKey: Record, + projection: { key: string; result: WorkspacePortScanResult } | null + ) => { + scansByKey = nextScansByKey + if (projection) { + projections.push(projection) + } + }, + getWorkspacePortScansByKey: () => scansByKey + } + } +} + +describe('publishWorkspacePortScanForHost', () => { + let localScan: WorkspacePortScanResult + let remoteScan: WorkspacePortScanResult + + beforeEach(() => { + localScan = scanWithPort(5173, 10) + remoteScan = scanWithPort(3000, 20) + }) + + it('publishes the single tracked host under its own key', () => { + const harness = makeStoreHarness() + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'local:all', + scan: localScan + }) + + expect(harness.projections).toEqual([{ key: 'local:all', result: localScan }]) + expect(Object.keys(harness.scansByKey)).toEqual(['local:all']) + }) + + it('keeps the other host in the projection when one host refreshes', () => { + const harness = makeStoreHarness({ 'local:all': localScan }) + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: remoteScan + }) + + const projection = harness.projections.at(-1) + expect(projection?.key).toBe(WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY) + expect(projection?.result.ports.map((port) => port.port).sort()).toEqual([3000, 5173]) + expect(Object.keys(harness.scansByKey).sort()).toEqual(['environment:env-1:all', 'local:all']) + }) + + it('publishes map and projection in a single store update', () => { + const harness = makeStoreHarness({ 'local:all': localScan }) + const replaceSpy = vi.spyOn(harness.publisher, 'replaceWorkspacePortScans') + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: remoteScan + }) + + // Why: two sequential setter calls notify subscribers twice for one scan; + // one atomic replace keeps map and projection from ever disagreeing. + expect(replaceSpy).toHaveBeenCalledTimes(1) + const [nextScans, projection] = replaceSpy.mock.calls[0] + expect(Object.keys(nextScans).sort()).toEqual(['environment:env-1:all', 'local:all']) + expect(projection?.key).toBe(WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY) + }) + + it('does not accumulate duplicate rows across repeated publishes', () => { + const harness = makeStoreHarness({ 'local:all': localScan }) + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: remoteScan + }) + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: { ...remoteScan, scannedAt: 30 } + }) + + // Why: the aggregate must never land in the per-host map, or the next merge + // folds the merged result back into itself and rows multiply. + expect(harness.scansByKey[WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY]).toBeUndefined() + expect( + harness.projections + .at(-1) + ?.result.ports.map((port) => port.port) + .sort() + ).toEqual([3000, 5173]) + }) +}) diff --git a/src/renderer/src/runtime/runtime-client-target.ts b/src/renderer/src/runtime/runtime-client-target.ts index 1a0b8e4b8a4..fbf9af17374 100644 --- a/src/renderer/src/runtime/runtime-client-target.ts +++ b/src/renderer/src/runtime/runtime-client-target.ts @@ -1,4 +1,5 @@ import type { GlobalSettings } from '../../../shared/global-settings-types' +import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host' export type RuntimeClientTarget = { kind: 'local' } | { kind: 'environment'; environmentId: string } @@ -9,6 +10,20 @@ export function getActiveRuntimeTarget( return environmentId ? { kind: 'environment', environmentId } : { kind: 'local' } } +/** RPC target for a dispatchable host; direct SSH cannot use this client path. */ +export function runtimeTargetForExecutionHostId( + hostId: ExecutionHostId +): RuntimeClientTarget | null { + const parsed = parseExecutionHostId(hostId) + if (parsed?.kind === 'local') { + return { kind: 'local' } + } + if (parsed?.kind === 'runtime') { + return { kind: 'environment', environmentId: parsed.environmentId } + } + return null +} + export function settingsForRuntimeOwner( settings: Pick | null | undefined, runtimeEnvironmentId: string | null | undefined diff --git a/src/renderer/src/runtime/use-worktree-runtime-target.ts b/src/renderer/src/runtime/use-worktree-runtime-target.ts new file mode 100644 index 00000000000..25bd9099e90 --- /dev/null +++ b/src/renderer/src/runtime/use-worktree-runtime-target.ts @@ -0,0 +1,16 @@ +import { useAppStore } from '@/store' +import { getExecutionHostIdForWorktree } from '@/lib/worktree-runtime-owner' +import { runtimeTargetForExecutionHostId, type RuntimeClientTarget } from './runtime-client-target' + +/** + * Runtime target that owns `worktreeId`, which is not always the globally + * focused runtime — acting on the focused one scans the wrong host and reports + * that workspace as having no ports. Direct-SSH owners return null. + */ +export function useWorktreeRuntimeTarget( + worktreeId: string | null | undefined +): RuntimeClientTarget | null { + return useAppStore((state) => + runtimeTargetForExecutionHostId(getExecutionHostIdForWorktree(state, worktreeId)) + ) +} diff --git a/src/renderer/src/store/terminals/terminal-layout-state.ts b/src/renderer/src/store/terminals/terminal-layout-state.ts index d95439cf5a8..9b5cf8323c2 100644 --- a/src/renderer/src/store/terminals/terminal-layout-state.ts +++ b/src/renderer/src/store/terminals/terminal-layout-state.ts @@ -43,15 +43,20 @@ export function createTerminalLayoutActions( } }) }, + // Why: pane mount/unmount re-asserts the same booleans; bailing like setTabLayout keeps map subscribers asleep. setTabPaneExpanded: (tabId, expanded) => { - set((s) => ({ - expandedPaneByTabId: { ...s.expandedPaneByTabId, [tabId]: expanded } - })) + set((s) => + s.expandedPaneByTabId[tabId] === expanded + ? s + : { expandedPaneByTabId: { ...s.expandedPaneByTabId, [tabId]: expanded } } + ) }, setTabCanExpandPane: (tabId, canExpand) => { - set((s) => ({ - canExpandPaneByTabId: { ...s.canExpandPaneByTabId, [tabId]: canExpand } - })) + set((s) => + s.canExpandPaneByTabId[tabId] === canExpand + ? s + : { canExpandPaneByTabId: { ...s.canExpandPaneByTabId, [tabId]: canExpand } } + ) }, setTabLayout: (tabId, layout) => { let ownershipTransfers: ReturnType = [] diff --git a/src/renderer/src/store/terminals/terminal-pane-expansion-write-bailout.test.tsx b/src/renderer/src/store/terminals/terminal-pane-expansion-write-bailout.test.tsx new file mode 100644 index 00000000000..d0d32f7742f --- /dev/null +++ b/src/renderer/src/store/terminals/terminal-pane-expansion-write-bailout.test.tsx @@ -0,0 +1,152 @@ +// @vitest-environment happy-dom + +import { Profiler } from 'react' +import { act, cleanup, render } from '@testing-library/react' +import { afterEach, describe, expect, it } from 'vitest' +import { createTestStore } from '../slices/store-test-helpers' + +afterEach(cleanup) + +type TestStore = ReturnType + +const TAB_ID = 'tab-1' +const NO_OP_WRITES = 25 + +function recordPublishedMapKeys(store: TestStore): string[] { + const published: string[] = [] + store.subscribe((next, previous) => { + if (next.expandedPaneByTabId !== previous.expandedPaneByTabId) { + published.push('expandedPaneByTabId') + } + if (next.canExpandPaneByTabId !== previous.canExpandPaneByTabId) { + published.push('canExpandPaneByTabId') + } + }) + return published +} + +// Mirrors use-terminal-workspace-store-bindings.ts:17, which subscribes to the raw map. +function ExpandedPaneSubscriber({ store }: { store: TestStore }): React.JSX.Element { + const expandedPaneByTabId = store((s) => s.expandedPaneByTabId) + return {String(expandedPaneByTabId[TAB_ID] === true)} +} + +function CanExpandPaneSubscriber({ store }: { store: TestStore }): React.JSX.Element { + const canExpandPaneByTabId = store((s) => s.canExpandPaneByTabId) + return {String(canExpandPaneByTabId[TAB_ID] === true)} +} + +function renderCommitCounter(subscriber: React.JSX.Element): () => number { + let commits = 0 + render( + { + commits += 1 + }} + > + {subscriber} + + ) + const mountCommits = commits + return () => commits - mountCommits +} + +describe('setTabPaneExpanded', () => { + it('publishes the first write for an unseen tab and a real toggle', () => { + const store = createTestStore() + const published = recordPublishedMapKeys(store) + + store.getState().setTabPaneExpanded(TAB_ID, false) + expect(published).toEqual(['expandedPaneByTabId']) + expect(store.getState().expandedPaneByTabId[TAB_ID]).toBe(false) + + store.getState().setTabPaneExpanded(TAB_ID, true) + expect(published).toEqual(['expandedPaneByTabId', 'expandedPaneByTabId']) + expect(store.getState().expandedPaneByTabId[TAB_ID]).toBe(true) + }) + + it('bails out when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabPaneExpanded(TAB_ID, false) + const before = store.getState().expandedPaneByTabId + // Root identity too: returning `{}` keeps the map but allocates a new root, so zustand still walks every listener. + const rootBefore = store.getState() + const published = recordPublishedMapKeys(store) + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + store.getState().setTabPaneExpanded(TAB_ID, false) + } + + expect(published).toEqual([]) + expect(store.getState().expandedPaneByTabId).toBe(before) + expect(store.getState()).toBe(rootBefore) + }) + + it('costs no React commit in a map subscriber when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabPaneExpanded(TAB_ID, false) + const commitsSinceMount = renderCommitCounter() + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + act(() => { + store.getState().setTabPaneExpanded(TAB_ID, false) + }) + } + expect(commitsSinceMount()).toBe(0) + + act(() => { + store.getState().setTabPaneExpanded(TAB_ID, true) + }) + expect(commitsSinceMount()).toBe(1) + }) +}) + +describe('setTabCanExpandPane', () => { + it('publishes the first write for an unseen tab and a real toggle', () => { + const store = createTestStore() + const published = recordPublishedMapKeys(store) + + store.getState().setTabCanExpandPane(TAB_ID, false) + expect(published).toEqual(['canExpandPaneByTabId']) + expect(store.getState().canExpandPaneByTabId[TAB_ID]).toBe(false) + + store.getState().setTabCanExpandPane(TAB_ID, true) + expect(published).toEqual(['canExpandPaneByTabId', 'canExpandPaneByTabId']) + expect(store.getState().canExpandPaneByTabId[TAB_ID]).toBe(true) + }) + + it('bails out when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabCanExpandPane(TAB_ID, false) + const before = store.getState().canExpandPaneByTabId + const rootBefore = store.getState() + const published = recordPublishedMapKeys(store) + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + store.getState().setTabCanExpandPane(TAB_ID, false) + } + + expect(published).toEqual([]) + expect(store.getState().canExpandPaneByTabId).toBe(before) + expect(store.getState()).toBe(rootBefore) + }) + + it('costs no React commit in a map subscriber when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabCanExpandPane(TAB_ID, false) + const commitsSinceMount = renderCommitCounter() + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + act(() => { + store.getState().setTabCanExpandPane(TAB_ID, false) + }) + } + expect(commitsSinceMount()).toBe(0) + + act(() => { + store.getState().setTabCanExpandPane(TAB_ID, true) + }) + expect(commitsSinceMount()).toBe(1) + }) +}) diff --git a/src/shared/project-groups.ts b/src/shared/project-groups.ts index 67c2897ead6..c4fe8badb47 100644 --- a/src/shared/project-groups.ts +++ b/src/shared/project-groups.ts @@ -109,10 +109,12 @@ export function clearMissingProjectGroupMemberships(repos: Repo[], groups: Proje ) } -export function getProjectGroupSubtreeIds( - groups: readonly Pick[], - rootGroupId: string -): Set { +export type ProjectGroupChildIndex = ReadonlyMap + +/** Build once and reuse when collecting subtrees for more than one root. */ +export function buildProjectGroupChildIndex( + groups: readonly Pick[] +): ProjectGroupChildIndex { const childGroupsByParentId = new Map() for (const group of groups) { if (!group.parentGroupId) { @@ -122,7 +124,20 @@ export function getProjectGroupSubtreeIds( children.push(group.id) childGroupsByParentId.set(group.parentGroupId, children) } + return childGroupsByParentId +} +export function getProjectGroupSubtreeIds( + groups: readonly Pick[], + rootGroupId: string +): Set { + return collectProjectGroupSubtreeIds(buildProjectGroupChildIndex(groups), rootGroupId) +} + +export function collectProjectGroupSubtreeIds( + childGroupsByParentId: ProjectGroupChildIndex, + rootGroupId: string +): Set { const subtreeIds = new Set() const pending = [rootGroupId] while (pending.length > 0) { diff --git a/src/shared/relay-artifacts.ts b/src/shared/relay-artifacts.ts index 273f6e059b8..2f9e563f839 100644 --- a/src/shared/relay-artifacts.ts +++ b/src/shared/relay-artifacts.ts @@ -37,6 +37,12 @@ export type RelayArtifact = { * optional one would loop forever redeploying a relay that is already correct. */ optional?: boolean + /** + * Forked by the relay daemon as a long-lived child of its own. These are relay + * infrastructure, never user work, and the reap gate subtracts them from a daemon's + * child census; see src/main/ssh/relay-daemon-service-children.ts. + */ + daemonServiceChild?: boolean } /** The bare Windows process-table addon; see docs/reference/windows-process-enumeration.md. */ @@ -44,8 +50,8 @@ export const RELAY_WINDOWS_PROCESS_TREE_FILENAME = 'windows-process-tree.node' export const RELAY_ARTIFACTS: readonly RelayArtifact[] = [ { filename: 'relay.js' }, - { filename: 'relay-watcher.js' }, - { filename: 'relay-ai-vault-service.js' }, + { filename: 'relay-watcher.js', daemonServiceChild: true }, + { filename: 'relay-ai-vault-service.js', daemonServiceChild: true }, { filename: 'managed-hook-runtime.js' }, // Forked by the AI Vault title reader; without it a relay answers every WSL // title request with no title and no error. @@ -62,6 +68,14 @@ export const RELAY_ARTIFACTS: readonly RelayArtifact[] = [ { filename: RELAY_WINDOWS_PROCESS_TREE_FILENAME, windowsOnly: true, optional: true } ] +/** + * The daemon's own service children, by entry filename. Anything else under a relay pid is + * either user work or unidentified, and both keep the relay unreapable. + */ +export const RELAY_DAEMON_SERVICE_ENTRY_FILENAMES: readonly string[] = RELAY_ARTIFACTS.filter( + (artifact) => artifact.daemonServiceChild +).map((artifact) => artifact.filename) + /** Written after the artifacts, so it is never an input to its own hash. */ export const RELAY_VERSION_FILENAME = '.version' diff --git a/src/shared/remote-execution-host-pty-id.ts b/src/shared/remote-execution-host-pty-id.ts new file mode 100644 index 00000000000..c76ada39e03 --- /dev/null +++ b/src/shared/remote-execution-host-pty-id.ts @@ -0,0 +1,14 @@ +import { parseRemoteRuntimePtyId } from './remote-runtime-pty-id' +import { parseAppSshPtyId } from './ssh-pty-id' + +/** + * Whether the process behind this pty runs on an execution host other than this machine — + * a paired runtime environment or an app SSH target. + * + * Deliberately broader than the inspection module's private remote check, which only counts a + * `remote:` id that carries an owner environment id. An owner-less `remote:` still runs + * somewhere else, and treating it as local is how remote work becomes invisible to a guard. + */ +export function isRemoteExecutionHostPtyId(ptyId: string): boolean { + return parseRemoteRuntimePtyId(ptyId) !== null || parseAppSshPtyId(ptyId) !== null +}