From 8854b5ded52bbc45b695dc88bae476de850f4bb9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 22:03:05 -0700 Subject: [PATCH 01/12] perf(ssh): coalesce concurrent git.listWorktrees reads (#18419) * perf(ssh): coalesce concurrent pty.inspectProcess and git.listWorktrees reads Both were the only reads in their provider class with no in-flight dedupe while their siblings already had it. Route them through the existing InFlightPromiseDedupe, keyed per (relay pty id, incarnation) and per repoPath, scoped to the provider instance so two hosts never share an entry. The worktree listing clears from invalidateGitReads(), and a signalled read keeps its own request so one caller's abort cannot cancel its joiners' scan. In-flight only, no TTL: the relay does answer inspectProcess from a 500ms TTL-cached process table, but a client TTL would compound with it rather than match it, so it is not a free win. * perf(ssh): drop the inspectProcess half, ratchet per-read host observations The `git.listWorktrees` dedupe ships unchanged. The `pty.inspectProcess` dedupe is reverted: the host mints one `observationEpoch` per request and the pane foreground reader commits it per read, so two overlapping probes sharing one reply make the second read a stale replay and `admitRemoteForegroundEvidence` rejects it -- a would-be `live` identity read becomes `unverifiable`. The pane foreground tracker overlaps its own probes by design (cancel-and-reissue after a 350 ms settle), so that path is reachable. Adds a ratchet that fails when the dedupe returns, driving the real reader through the real provider operations. * fix(ssh): move the inspect ratchet to the provider it guards The ratchet lived under src/renderer and imported src/main/providers/ssh-pty-provider-rpc-operations, dragging the whole main-process graph into config/tsconfig.tc.web.json (TS6307). Split it: the request-counter ratchet moves next to the provider it pins, and the renderer file keeps the why -- a shared host observation degrades the second overlapping read to unverifiable -- against the real reader with no cross-project import. * docs(ssh): document the worktree-list coalescing contract --- src/main/providers/ssh-git-read-provider.ts | 3 +- .../ssh-git-worktree-list-dedupe.test.ts | 156 ++++++++++++++++++ .../providers/ssh-git-worktree-provider.ts | 28 +++- ...h-pty-inspect-observation-identity.test.ts | 68 ++++++++ .../ssh-pty-provider-rpc-operations.ts | 4 + ...round-inspect-observation-identity.test.ts | 95 +++++++++++ 6 files changed, 349 insertions(+), 5 deletions(-) create mode 100644 src/main/providers/ssh-git-worktree-list-dedupe.test.ts create mode 100644 src/main/providers/ssh-pty-inspect-observation-identity.test.ts create mode 100644 src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts diff --git a/src/main/providers/ssh-git-read-provider.ts b/src/main/providers/ssh-git-read-provider.ts index 5cbcbc17b5a..cbf20271003 100644 --- a/src/main/providers/ssh-git-read-provider.ts +++ b/src/main/providers/ssh-git-read-provider.ts @@ -44,7 +44,8 @@ export class SshGitReadProvider { } } - private invalidateGitReads(): void { + /** Overridden by subclasses that own additional read caches (worktree listings). */ + protected invalidateGitReads(): void { this.gitDiffReadDedupe.clear() this.statusReadLeaseOwner.invalidate() this.upstreamStatusReadOwner.invalidate() diff --git a/src/main/providers/ssh-git-worktree-list-dedupe.test.ts b/src/main/providers/ssh-git-worktree-list-dedupe.test.ts new file mode 100644 index 00000000000..4030dbe87e4 --- /dev/null +++ b/src/main/providers/ssh-git-worktree-list-dedupe.test.ts @@ -0,0 +1,156 @@ +/** + * Local repos coalesce concurrent `git worktree list` scans (`shareWorktreeScan`); the SSH path + * branched away from that and paid one relay round trip per independent caller (`worktrees:list`, + * `worktrees:listAll`, the space repo scan, provisioned-root adoption). These are call counters. + */ +import { describe, expect, it } from 'vitest' +import { SshGitProvider } from './ssh-git-provider' +import { createMockMux, type MockMultiplexer } from './ssh-git-provider-test-harness' + +const REPO_PATH = '/home/user/repo' + +const WORKTREES = [ + { path: REPO_PATH, head: 'abc123', branch: 'main', isBare: false, isMainWorktree: true } +] + +type Deferred = { resolve: (value: unknown) => void; reject: (error: unknown) => void } + +/** Holds `git.listWorktrees` open so overlap is deterministic; answers everything else at once. */ +function createPendingListMux(): { mux: MockMultiplexer; listDeferreds: Deferred[] } { + const mux = createMockMux() + const listDeferreds: Deferred[] = [] + mux.request.mockImplementation((method: string) => { + if (method !== 'git.listWorktrees') { + return Promise.resolve(undefined) + } + return new Promise((resolve, reject) => { + listDeferreds.push({ resolve, reject }) + }) + }) + return { mux, listDeferreds } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +function countListRequests(mux: MockMultiplexer): number { + return mux.request.mock.calls.filter((call) => call[0] === 'git.listWorktrees').length +} + +describe('SSH git.listWorktrees in-flight dedupe', () => { + it('collapses concurrent listings of one repo into a single relay request', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = Array.from({ length: 6 }, () => provider.listWorktrees(REPO_PATH)) + await flush() + + expect(countListRequests(mux)).toBe(1) + expect(mux.request).toHaveBeenCalledWith( + 'git.listWorktrees', + { repoPath: REPO_PATH }, + { signal: undefined } + ) + + listDeferreds[0].resolve(WORKTREES) + expect(await Promise.all(listings)).toEqual(Array.from({ length: 6 }, () => WORKTREES)) + }) + + it('does not share across repos or connections', async () => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + void provider.listWorktrees('/home/user/other') + await flush() + expect(countListRequests(mux)).toBe(2) + + const second = createPendingListMux() + void new SshGitProvider('conn-2', second.mux as never).listWorktrees(REPO_PATH) + await flush() + + expect(countListRequests(second.mux)).toBe(1) + expect(countListRequests(mux)).toBe(2) + }) + + it('keeps a signalled listing on its own request', async () => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + await flush() + const controller = new AbortController() + void provider.listWorktrees(REPO_PATH, { signal: controller.signal }) + await flush() + + expect(countListRequests(mux)).toBe(2) + expect(mux.request).toHaveBeenCalledWith( + 'git.listWorktrees', + { repoPath: REPO_PATH }, + { signal: controller.signal } + ) + }) + + it('re-requests after the shared listing settles instead of caching it', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const first = provider.listWorktrees(REPO_PATH) + await flush() + listDeferreds[0].resolve(WORKTREES) + await first + + void provider.listWorktrees(REPO_PATH) + await flush() + + expect(countListRequests(mux)).toBe(2) + }) + + it.each([ + ['addWorktree', (p: SshGitProvider) => p.addWorktree(REPO_PATH, 'feature', '/home/user/feat')], + ['removeWorktree', (p: SshGitProvider) => p.removeWorktree('/home/user/feat')] + ])('invalidates the shared listing after %s', async (_name, mutate) => { + const { mux } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(1) + + await mutate(provider) + + // The catalog moved, so a joiner must not inherit the pre-mutation scan. + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(2) + }) + + it('shares a failed listing with its joiners and re-requests afterwards', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)] + await flush() + expect(countListRequests(mux)).toBe(1) + + const failure = new Error('relay request failed') + listDeferreds[0].reject(failure) + await expect(listings[0]).rejects.toBe(failure) + await expect(listings[1]).rejects.toBe(failure) + + void provider.listWorktrees(REPO_PATH) + await flush() + expect(countListRequests(mux)).toBe(2) + }) + + it('refuses an unauthoritative relay answer for every joiner (#14004)', async () => { + const { mux, listDeferreds } = createPendingListMux() + const provider = new SshGitProvider('conn-1', mux as never) + + const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)] + await flush() + listDeferreds[0].resolve([]) + + await expect(listings[0]).rejects.toThrow() + await expect(listings[1]).rejects.toThrow() + }) +}) diff --git a/src/main/providers/ssh-git-worktree-provider.ts b/src/main/providers/ssh-git-worktree-provider.ts index 8dae1413321..17e5b273de7 100644 --- a/src/main/providers/ssh-git-worktree-provider.ts +++ b/src/main/providers/ssh-git-worktree-provider.ts @@ -2,6 +2,7 @@ import type { GitStatusResult } from '../../shared/git-status-types' import type { RemoveWorktreeResult } from '../../shared/worktree/create-types' import type { GitWorktreeInfo } from '../../shared/worktree/types' import { CapabilityProbeCache } from '../../shared/capability-probe-cache' +import { InFlightPromiseDedupe, stableInFlightKey } from '../../shared/in-flight-promise-dedupe' import { assertAuthoritativeWorktreeCatalog } from '../../shared/worktree/worktree-catalog-availability' import { isJsonRpcMethodNotFoundError } from './ssh-git-relay-errors' import { SshGitReviewHeadProvider } from './ssh-git-review-head-provider' @@ -29,16 +30,35 @@ export class SshGitWorktreeProvider extends SshGitReviewHeadProvider { private readonly worktreeIsCleanCapabilityCache = new CapabilityProbeCache< typeof WORKTREE_IS_CLEAN_CAPABILITY >(Number.POSITIVE_INFINITY) + // Scoped to this provider instance, so two SSH hosts never share an entry. + private readonly worktreeListDedupe = new InFlightPromiseDedupe() + protected override invalidateGitReads(): void { + super.invalidateGitReads() + this.worktreeListDedupe.clear() + } + + /** Un-signalled reads of one repo coalesce onto the request already in flight; nothing is cached. */ async listWorktrees( repoPath: string, options?: { signal?: AbortSignal } ): Promise { - const response = await this.mux.request( - 'git.listWorktrees', - { repoPath }, - { signal: options?.signal } + // Why: same rule as shareWorktreeScan — one caller's abort must not cancel the scan its + // joiners are still waiting on, so a signalled read keeps its own request. + if (options?.signal) { + return this.requestWorktreeList(repoPath, options.signal) + } + return this.worktreeListDedupe.run(stableInFlightKey(['listWorktrees', repoPath]), () => + this.requestWorktreeList(repoPath) ) + } + + /** The one real relay round trip a coalesced read's joiners all wait on. */ + private async requestWorktreeList( + repoPath: string, + signal?: AbortSignal + ): Promise { + const response = await this.mux.request('git.listWorktrees', { repoPath }, { signal }) // Why (#14004): relays before this fix answered a failed worktree scan with `[]`. Mixed versions are // normal, so refuse the shape here too — a Git repo always lists its own checkout. return assertAuthoritativeWorktreeCatalog(response, repoPath) diff --git a/src/main/providers/ssh-pty-inspect-observation-identity.test.ts b/src/main/providers/ssh-pty-inspect-observation-identity.test.ts new file mode 100644 index 00000000000..4e0dae1ead1 --- /dev/null +++ b/src/main/providers/ssh-pty-inspect-observation-identity.test.ts @@ -0,0 +1,68 @@ +/** + * Ratchet (#18419): `pty.inspectProcess` must NOT be in-flight coalesced the way the sibling git + * reads in `SshGitReadProvider` are. The host mints one `observationEpoch` per request and the + * pane foreground reader commits that epoch per read, so a shared reply reads as a stale replay to + * the second reader to settle — see the companion renderer proof in + * `src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts`. + * These are request counters, not timings. + */ +import { describe, expect, it, vi } from 'vitest' +import { createSshPtyProviderRpcOperations } from './ssh-pty-provider-rpc-operations' + +const RELAY_PTY_ID = 'pty-1' +const APP_PTY_ID = `ssh:conn-1@@${RELAY_PTY_ID}` +const INCARNATION_ID = 'inc-1' + +/** Answers `pty.inspectProcess` with a fresh host observation per request, held open on demand. */ +function createInspectingOperations(): { + operations: ReturnType + request: ReturnType + resolvers: ((value: unknown) => void)[] +} { + const resolvers: ((value: unknown) => void)[] = [] + const request = vi.fn(() => new Promise((resolve) => resolvers.push(resolve))) + return { + operations: createSshPtyProviderRpcOperations({ + mux: { request } as never, + toRelayPtyId: () => RELAY_PTY_ID + }), + request, + resolvers + } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +describe('SSH pty.inspectProcess observation identity', () => { + it('gives each overlapping probe of one pane+incarnation its own host observation', async () => { + const { operations, request, resolvers } = createInspectingOperations() + + const first = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + const second = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + await flush() + + expect(request).toHaveBeenCalledTimes(2) + resolvers[0]({ foregroundProcess: 'claude', observationEpoch: 1 }) + resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 2 }) + // Each read settles on the observation minted for it, never a neighbour's. + expect(await first).toMatchObject({ observationEpoch: 1 }) + expect(await second).toMatchObject({ observationEpoch: 2 }) + }) + + it('does not share a failed probe with an overlapping one', async () => { + const { operations, request, resolvers } = createInspectingOperations() + + const failing = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID }) + const overlapping = operations.inspectProcess(APP_PTY_ID, { + expectedIncarnationId: INCARNATION_ID + }) + await flush() + + expect(request).toHaveBeenCalledTimes(2) + resolvers[0](Promise.reject(new Error('relay dropped the probe'))) + resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 1 }) + + await expect(failing).rejects.toThrow('relay dropped the probe') + expect(await overlapping).toMatchObject({ observationEpoch: 1 }) + }) +}) diff --git a/src/main/providers/ssh-pty-provider-rpc-operations.ts b/src/main/providers/ssh-pty-provider-rpc-operations.ts index 3f9953b9d1c..71ddbefce97 100644 --- a/src/main/providers/ssh-pty-provider-rpc-operations.ts +++ b/src/main/providers/ssh-pty-provider-rpc-operations.ts @@ -50,6 +50,10 @@ export function createSshPtyProviderRpcOperations({ mux, toRelayPtyId }: SshPtyP const result = await mux.request('pty.getForegroundProcess', { id: toRelayPtyId(id) }) return result as string | null }, + // Do NOT in-flight coalesce this the way the sibling git reads are: the host mints one + // `observationEpoch` per request and the pane foreground reader commits it per read, so a + // shared reply reads as a stale replay and degrades a `live` identity read to `unverifiable`. + // Guarded by ssh-pty-inspect-observation-identity.test.ts; #17525 removes the poll. inspectProcess: async ( id: string, options?: { expectedIncarnationId?: string } diff --git a/src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts b/src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts new file mode 100644 index 00000000000..71719b11e4f --- /dev/null +++ b/src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts @@ -0,0 +1,95 @@ +/** + * Why `pty.inspectProcess` is not in-flight coalesced (#18419). The host mints one + * `observationEpoch` per request and this reader commits that epoch per read, so a reply shared by + * two overlapping probes reads as a stale replay to the second reader to settle and its would-be + * `live` identity read degrades to `unverifiable`. The pane foreground tracker overlaps its own + * probes on purpose (`cancelPendingRead` bumps the generation but lets the in-flight probe finish, + * then reissues after a 350 ms settle), so that path is reachable. The provider-side ratchet that + * fails if the dedupe returns lives in `src/main/providers/ssh-pty-inspect-observation-identity.test.ts`. + */ +import { describe, expect, it } from 'vitest' +import { createPaneForegroundProcessReader } from './pane-foreground-process-reader' + +const CONNECTION_ID = 'conn-1' +const RELAY_PTY_ID = 'pty-1' +const APP_PTY_ID = `ssh:${CONNECTION_ID}@@${RELAY_PTY_ID}` +const INCARNATION_ID = 'inc-1' + +/** One host scan per request => one epoch per request. */ +const hostObservation = (observationEpoch: number): unknown => ({ + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'live', + processName: 'claude', + ptyId: RELAY_PTY_ID, + ptyIncarnationId: INCARNATION_ID, + authorityGeneration: 'gen-1', + observationEpoch, + capturedAgeMs: 0, + fence: { + platform: 'posix', + shellPid: 100, + shellStartTime: '1000', + tty: '/dev/pts/3', + foregroundPgid: 200 + } + } +}) + +/** Holds every probe open so the tracker's supersede-and-reissue pair really overlaps. */ +function createOverlappingReader(replies: { shared: boolean }): { + readProcess: ReturnType + settle: (index: number) => void +} { + const resolvers: ((value: unknown) => void)[] = [] + return { + // One reader instance per pane, exactly as the foreground tracker holds it. + readProcess: createPaneForegroundProcessReader({ + readForegroundProcess: () => new Promise((resolve) => resolvers.push(resolve)) as never, + isRemotePtyId: () => true, + getExpectedIncarnationId: () => INCARNATION_ID + }), + // `shared` models what an in-flight dedupe would do: every joiner gets one host observation. + settle: (index) => resolvers[index]?.(hostObservation(replies.shared ? 1 : index + 1)) + } +} + +const flush = (): Promise => new Promise((resolve) => setTimeout(resolve, 0)) + +describe('pane foreground inspect observation identity', () => { + it('keeps a reissued read `live` when it overlaps the probe it superseded', async () => { + const { readProcess, settle } = createOverlappingReader({ shared: false }) + + // The tracker cancels the first read (generation bump) but lets it run to completion, then + // reissues after the settle window — so both are in flight against the same pane. + const superseded = readProcess(APP_PTY_ID, false) + const reissued = readProcess(APP_PTY_ID, false) + await flush() + + // The superseded read's continuation commits its epoch first. + settle(0) + expect((await superseded).remoteEvidenceVerdict).toBe('live') + + settle(1) + const result = await reissued + expect(result.remoteEvidenceVerdict).toBe('live') + expect(result.processName).toBe('claude') + }) + + it('degrades the second overlapping read to `unverifiable` when one observation is shared', async () => { + const { readProcess, settle } = createOverlappingReader({ shared: true }) + + const superseded = readProcess(APP_PTY_ID, false) + const reissued = readProcess(APP_PTY_ID, false) + await flush() + + settle(0) + expect((await superseded).remoteEvidenceVerdict).toBe('live') + + settle(1) + const result = await reissued + expect(result.remoteEvidenceVerdict).toBe('unverifiable') + expect(result.processName).toBeNull() + }) +}) From 7574ee8403b30dbe3142c114e6633228986ac0a6 Mon Sep 17 00:00:00 2001 From: iverJisty Date: Fri, 4 Sep 2026 13:19:40 +0800 Subject: [PATCH 02/12] fix(ports): route the status-bar popover scan to the workspace's host (#17048) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ports): route the status-bar popover scan to the workspace's host - PortsStatusSegment resolved its runtime target from the global active runtime, so opening the popover on a paired-remote workspace scanned the client OS and reported zero workspace ports - Resolve the target from the active worktree's owner host, matching PortsPanel, PortRow, and WorktreeCardPorts - Add publishWorkspacePortScanForHost: store the host's scan under its own key, then republish the aggregate through setWorkspacePortScanProjection so a single-host refresh no longer drops every other host's ports - Publish through the projection setter instead of setWorkspacePortScan, which wrote the synthetic all-hosts key back into workspacePortScansByKey and made the next merge fold the aggregate into itself (duplicate rows) - Reuse the helper for the manual panel refresh and the post-stop refresh, and share the aggregate key constant with WorkspacePortScanner * test(ports): cover popover host routing and aggregate preservation - PortsStatusSegment.host-routing: popover scans the active workspace's owner host, keeps other hosts in the projection, publishes a failed scan under its own host, and stays local when the workspace has no owner - workspace-port-scan-publish: single-host key vs all-hosts projection, and repeated publishes never accumulate duplicate rows * fix(ports): surface a host whose port scan failed instead of dropping it - The merged projection only carries unavailableReason when every host failed, so one unreachable server read as "this workspace has no ports" - Add getUnavailableWorkspacePortHosts: hosts that failed while another host still answered, with the local host distinguished by a null environment id - Show one notice per failed host in the popover, named by its runtime environment or the local host label, above the surviving hosts' ports - Reuse the existing scan-unavailable string so no catalog entry is added * test(ports): prove the popover's own failed scan reaches the host notice - Make the mocked store setters write back, so a publish and the notice that reads it can no longer name different scan keys with every assertion green - Cover open popover -> remote scan rejects -> notice names the host, the seam the store-write and render-only tests each stopped short of - Drop an assertion comment that claimed to prove port preservation when it only exercised the render path * fix(ports): review nits — single-write publish, colon-safe host keys, failure port retention, docstrings * fix(ports): keep the popover count and body in agreement, name every failed host - A failed scan retains the host's last-good ports, and the badge/header count them; the notice now sits above the list instead of replacing it, so the popover no longer claims N ports over an empty body. - getUnavailableWorkspacePortHosts reports all-hosts-failed too, so total loss of contact names each host instead of printing raw scan keys under platform 'unknown'. - Scan keys parse to a discriminated host ref, so an unrecognised key is 'unknown' rather than silently blamed on the local machine. - Extract useWorktreeRuntimeTarget for the four ports surfaces that hand-rolled the same owner-settings spread. * fix(ports): label the local host from the failed scan's platform, not the renderer's userAgent A paired web client's browser is not the Orca host, so deriving 'Local Mac' from navigator.userAgent mislabels a Linux host. Carry each failed scan's own platform through the unavailable-host list instead. * fix(ports): keep the Ports panel list under its failure notice too The retained-ports change gave a failed scan both ports and an unavailableReason, and the right-sidebar panel hid every section behind the notice — stripping the stop and open actions for ports the status bar still counts. Gate the sections on whether anything is left to list, matching the popover, behind a testable predicate. * fix(ports): let a retained-port failure keep its debounce grace period The popover publishes the host's last-good ports alongside the failure reason the moment its own scan fails. reconcileTransientPortScanFailures treated any published result carrying unavailableReason as a spent grace period, so the very next background poll replaced those ports with an empty unavailable scan — the retention never survived one poll interval. Keep the grace while the published result still has ports; the tolerance still clears them on schedule. * fix(ports): prune stale hosts in the poll's single map write A manual publish (the ports popover) can resolve after the host-set change already pruned its key, re-adding it; the poll's per-key writes only ever added, so a removed host kept its ports in the count and held a permanent unavailable notice until the next host-set change. Publish the poll's already-pruned map in one replaceWorkspacePortScans instead, which also collapses N per-host notifications into one and drops any synthetic all-hosts key that leaked in. * fix(ports): fail closed for direct SSH workspaces --------- Co-authored-by: Neil <4138956+nwparker@users.noreply.github.com> --- .../ports/WorkspacePortScanner.test.tsx | 31 ++ .../components/ports/WorkspacePortScanner.tsx | 37 +- .../right-sidebar/PortsPanel.test.tsx | 53 +-- .../local-workspace-port-sections.test.ts | 30 ++ .../local-workspace-port-sections.ts | 20 + .../local-workspace-ports-panel.tsx | 63 ++- .../WorktreeCard.compact-hover.test.tsx | 4 +- ....compact-ports-hover-independence.test.tsx | 4 +- .../sidebar/WorktreeCardPorts.test.tsx | 2 +- .../components/sidebar/WorktreeCardPorts.tsx | 24 +- .../PortsStatusSegment.host-routing.test.tsx | 407 ++++++++++++++++++ .../status-bar/PortsStatusSegment.tsx | 141 ++++-- .../ports-status-popover-rows.test.tsx | 5 +- .../status-bar/ports-status-popover-rows.tsx | 26 +- .../src/lib/workspace-port-actions.ts | 85 ++-- .../workspace-port-host-availability.test.ts | 148 +++++++ .../lib/workspace-port-host-availability.ts | 65 +++ .../lib/workspace-port-scan-debounce.test.ts | 32 ++ .../src/lib/workspace-port-scan-debounce.ts | 10 +- .../lib/workspace-port-scan-publish.test.ts | 153 +++++++ .../src/runtime/runtime-client-target.ts | 15 + .../runtime/use-worktree-runtime-target.ts | 16 + 22 files changed, 1185 insertions(+), 186 deletions(-) create mode 100644 src/renderer/src/components/right-sidebar/local-workspace-port-sections.test.ts create mode 100644 src/renderer/src/components/status-bar/PortsStatusSegment.host-routing.test.tsx create mode 100644 src/renderer/src/lib/workspace-port-host-availability.test.ts create mode 100644 src/renderer/src/lib/workspace-port-host-availability.ts create mode 100644 src/renderer/src/lib/workspace-port-scan-publish.test.ts create mode 100644 src/renderer/src/runtime/use-worktree-runtime-target.ts diff --git a/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx b/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx index be1c69cbac7..0ae53fee931 100644 --- a/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx +++ b/src/renderer/src/components/ports/WorkspacePortScanner.test.tsx @@ -493,6 +493,37 @@ describe('WorkspacePortScanner', () => { expect(useAppStore.getState().workspacePortScansByKey['environment:env-3:all']).toBeUndefined() }) + // Why: a manual publish (the ports popover) can resolve after the host-set + // change already pruned its key, re-adding it. Per-key writes never delete, so + // that removed host would otherwise hold its ports and a permanent + // unavailable notice until the next host-set change. + it('drops a stale host re-added after pruning on the next poll', async () => { + await act(async () => { + root?.render() + await flushPromises() + }) + + const staleKey = 'environment:env-removed:all' + act(() => { + const state = useAppStore.getState() + state.replaceWorkspacePortScans( + { + ...state.workspacePortScansByKey, + [staleKey]: { ...emptyScan, unavailableReason: 'gone' } + }, + state.workspacePortScan + ) + }) + expect(useAppStore.getState().workspacePortScansByKey[staleKey]).toBeDefined() + + await act(async () => { + vi.advanceTimersByTime(30_000) + await flushPromises() + }) + + expect(useAppStore.getState().workspacePortScansByKey[staleKey]).toBeUndefined() + }) + it('clears ports immediately when the final worktree is removed', async () => { runtimeEnvironmentCall.mockImplementation(({ method }) => { if (method === 'workspacePorts.scan') { diff --git a/src/renderer/src/components/ports/WorkspacePortScanner.tsx b/src/renderer/src/components/ports/WorkspacePortScanner.tsx index f2805eb88a7..2b12bd86cd8 100644 --- a/src/renderer/src/components/ports/WorkspacePortScanner.tsx +++ b/src/renderer/src/components/ports/WorkspacePortScanner.tsx @@ -4,10 +4,11 @@ import { getHasAnyWorktreesFromState } from '@/store/selectors' import { getActiveRuntimeTarget, type RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { mergeWorkspacePortScans, - runtimeTargetForExecutionHostId, + WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY, scanWorkspacePortsForTarget, workspacePortScanKeyForTarget } from '@/lib/workspace-port-actions' +import { runtimeTargetForExecutionHostId } from '@/runtime/runtime-client-target' import { installWindowVisibilityInterval, isWindowVisible } from '@/lib/window-visibility-interval' import { reconcileTransientPortScanFailures, @@ -41,7 +42,6 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) const setWorkspacePortScanProjection = useAppStore((s) => s.setWorkspacePortScanProjection) const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const inFlightRef = useRef | null>(null) const generationRef = useRef(0) @@ -124,40 +124,40 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): const activeTargetKeys = new Set( allTargets.map((target) => workspacePortScanKeyForTarget(target)) ) + const publishedScans = useAppStore.getState().workspacePortScansByKey const reconciled = reconcileTransientPortScanFailures( results, - useAppStore.getState().workspacePortScansByKey, + publishedScans, portScanDebounceRef.current, WORKSPACE_PORT_SCAN_FAILURE_THRESHOLD, activeTargetKeys ) const scansByKey = Object.fromEntries( - Object.entries(useAppStore.getState().workspacePortScansByKey).filter(([key]) => - activeTargetKeys.has(key) - ) + Object.entries(publishedScans).filter(([key]) => activeTargetKeys.has(key)) ) - let sourceChanged = false + // Why: a manual publish that lands after a host is pruned re-adds its key, + // and per-key writes never delete. Dropping the inactive keys here is what + // stops a removed host from holding a permanent unavailable notice. + let sourceChanged = + Object.keys(scansByKey).length !== Object.keys(publishedScans).length for (const { key, result } of reconciled) { sourceChanged ||= scansByKey[key] !== result scansByKey[key] = result - setWorkspacePortScanForKey(key, result) } const activeScan = scansByKey[scanKey] const merged = mergeWorkspacePortScans(scansByKey) const projectionKey = allTargets.length > 1 - ? 'all-hosts:all' + ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : activeScan ? scanKey : workspacePortScanKeyForTarget(allTargets[0]) if (sourceChanged || useAppStore.getState().workspacePortScan?.key !== projectionKey) { - setWorkspacePortScanProjection( - merged - ? { - key: projectionKey, - result: merged - } - : null + // Why: one store update for the whole poll — a large host set must not + // fan out a notification to every subscriber per host. + replaceWorkspacePortScans( + sourceChanged ? scansByKey : publishedScans, + merged ? { key: projectionKey, result: merged } : null ) } } @@ -177,8 +177,7 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): hasWorktrees, scanKey, setWorkspacePortScan, - setWorkspacePortScanProjection, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ] ) @@ -215,7 +214,7 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }): : Object.fromEntries(retainedEntries) const retainedProjection = mergeWorkspacePortScans(retainedScans) const retainedProjectionKey = - targetKeys.size > 1 ? 'all-hosts:all' : Object.keys(retainedScans)[0] + targetKeys.size > 1 ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : Object.keys(retainedScans)[0] // Why: unchanged hosts stay visible while the replacement RPC runs; removed // hosts and the old synthetic aggregate are excluded immediately. const nextProjection = diff --git a/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx b/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx index ea1d85f15b8..1121a05ae64 100644 --- a/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx +++ b/src/renderer/src/components/right-sidebar/PortsPanel.test.tsx @@ -451,25 +451,26 @@ describe('PortsPanel runtime routing', () => { }) it('returns post-stop refresh failures without throwing', async () => { - const setWorkspacePortScan = vi.fn() + const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() localScan.mockRejectedValueOnce(new Error('scan failed')) await expect( refreshWorkspacePortScanAfterStop({ runtimeTarget: { kind: 'local' }, - setWorkspacePortScan: setWorkspacePortScan as never, + replaceWorkspacePortScans: replaceWorkspacePortScans as never, + getWorkspacePortScansByKey: () => ({}), setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never }) ).resolves.toEqual({ ok: false, reason: 'scan failed' }) - expect(setWorkspacePortScan).not.toHaveBeenCalled() + expect(replaceWorkspacePortScans).not.toHaveBeenCalled() expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(1, true) expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(2, false) }) it('ignores settled remote post-stop refresh failures after updating state', async () => { - const setWorkspacePortScan = vi.fn() + const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() const firstScan = { ...emptyScan, scannedAt: 2 } let scanCalls = 0 @@ -500,7 +501,8 @@ describe('PortsPanel runtime routing', () => { await expect( refreshWorkspacePortScanAfterStop({ runtimeTarget: { kind: 'environment', environmentId: 'env-1' }, - setWorkspacePortScan: setWorkspacePortScan as never, + replaceWorkspacePortScans: replaceWorkspacePortScans as never, + getWorkspacePortScansByKey: () => ({}), setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never }) ).resolves.toEqual({ ok: true }) @@ -510,18 +512,20 @@ describe('PortsPanel runtime routing', () => { 'workspacePorts.scan', 'workspacePorts.scan' ]) - expect(setWorkspacePortScan).toHaveBeenCalledTimes(1) - expect(setWorkspacePortScan).toHaveBeenCalledWith({ - key: 'environment:env-1:all', - result: firstScan - }) + expect(replaceWorkspacePortScans).toHaveBeenCalledTimes(1) + expect(replaceWorkspacePortScans).toHaveBeenCalledWith( + { 'environment:env-1:all': firstScan }, + { + key: 'environment:env-1:all', + result: firstScan + } + ) expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(1, true) expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(2, false) }) it('preserves an all-host projection after refreshing one host post-stop', async () => { - const setWorkspacePortScan = vi.fn() - const setWorkspacePortScanForKey = vi.fn() + const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() const localPort: WorkspacePort = { ...workspacePort, id: 'local-port', port: 5173 } const refreshedRemotePort: WorkspacePort = { @@ -571,23 +575,24 @@ describe('PortsPanel runtime routing', () => { await expect( refreshWorkspacePortScanAfterStop({ runtimeTarget: { kind: 'environment', environmentId: 'env-1' }, - setWorkspacePortScan: setWorkspacePortScan as never, - setWorkspacePortScanForKey: setWorkspacePortScanForKey as never, + replaceWorkspacePortScans: replaceWorkspacePortScans as never, getWorkspacePortScansByKey: () => ({ 'local:all': localHostScan }), setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never }) ).resolves.toEqual({ ok: true }) - expect(setWorkspacePortScanForKey).toHaveBeenCalledWith('environment:env-1:all', remoteHostScan) - expect(setWorkspacePortScan).toHaveBeenLastCalledWith({ - key: 'all-hosts:all', - result: expect.objectContaining({ - ports: expect.arrayContaining([ - expect.objectContaining({ port: 5173 }), - expect.objectContaining({ port: 3000 }) - ]) - }) - }) + expect(replaceWorkspacePortScans).toHaveBeenLastCalledWith( + { 'local:all': localHostScan, 'environment:env-1:all': remoteHostScan }, + { + key: 'all-hosts:all', + result: expect.objectContaining({ + ports: expect.arrayContaining([ + expect.objectContaining({ port: 5173 }), + expect.objectContaining({ port: 3000 }) + ]) + }) + } + ) expect(scanCalls).toBe(2) }) diff --git a/src/renderer/src/components/right-sidebar/local-workspace-port-sections.test.ts b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.test.ts new file mode 100644 index 00000000000..68733927ee7 --- /dev/null +++ b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.test.ts @@ -0,0 +1,30 @@ +import { describe, expect, it } from 'vitest' +import { shouldShowLocalWorkspacePortSections } from './local-workspace-port-sections' + +const empty = { activePorts: [], otherWorkspacePorts: [], externalPorts: [] } + +describe('shouldShowLocalWorkspacePortSections', () => { + it('shows the sections whenever the scan succeeded', () => { + expect(shouldShowLocalWorkspacePortSections(null, empty)).toBe(true) + expect(shouldShowLocalWorkspacePortSections({}, empty)).toBe(true) + }) + + // Why: a failed scan keeps the host's last-good ports, and the status bar + // still counts and lists them — hiding the sections here would strip the + // stop and open actions for ports the user can still see elsewhere. + it.each([ + ['activePorts', { ...empty, activePorts: [{}] }], + ['otherWorkspacePorts', { ...empty, otherWorkspacePorts: [{}] }], + ['externalPorts', { ...empty, externalPorts: [{}] }] + ])('keeps the sections when a failed scan retained %s', (_section, sections) => { + expect(shouldShowLocalWorkspacePortSections({ unavailableReason: 'dropped' }, sections)).toBe( + true + ) + }) + + it('lets the notice stand alone when a failed scan has nothing left to list', () => { + expect(shouldShowLocalWorkspacePortSections({ unavailableReason: 'dropped' }, empty)).toBe( + false + ) + }) +}) diff --git a/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts index d968bb85144..2a0eadf380a 100644 --- a/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts +++ b/src/renderer/src/components/right-sidebar/local-workspace-port-sections.ts @@ -35,6 +35,26 @@ export function getLocalWorkspacePortSections( } } +/** + * Whether the panel still renders its port sections under a failure notice. + * Why: a failed scan retains the host's last-good ports, so hiding every + * section would drop the stop and open actions for ports the status bar still + * counts and lists. + */ +export function shouldShowLocalWorkspacePortSections( + scan: { unavailableReason?: string } | null | undefined, + sections: { activePorts: unknown[]; otherWorkspacePorts: unknown[]; externalPorts: unknown[] } +): boolean { + if (!scan?.unavailableReason) { + return true + } + return ( + sections.activePorts.length > 0 || + sections.otherWorkspacePorts.length > 0 || + sections.externalPorts.length > 0 + ) +} + function workspacePortAsExternal(port: WorkspacePort & { kind: 'workspace' }): WorkspacePort { return { id: port.id, diff --git a/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx b/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx index 0955e89031b..1737e7da487 100644 --- a/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx +++ b/src/renderer/src/components/right-sidebar/local-workspace-ports-panel.tsx @@ -4,11 +4,11 @@ import { toast } from 'sonner' import { useAppStore } from '@/store' import { useActiveWorktree, useRepoById } from '@/store/selectors' import { cn } from '@/lib/utils' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' -import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' import { killWorkspacePortForTarget, openWorkspacePortInBrowser, + publishWorkspacePortScanForHost, refreshWorkspacePortScanAfterStop, resolvePortOpenInOrcaBrowser, scanWorkspacePortsForTarget, @@ -19,10 +19,14 @@ import { Button } from '@/components/ui/button' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import type { WorkspacePort } from '../../../../shared/workspace-ports' import { translate } from '@/i18n/i18n' -import { getLocalWorkspacePortSections } from './local-workspace-port-sections' +import { + getLocalWorkspacePortSections, + shouldShowLocalWorkspacePortSections +} from './local-workspace-port-sections' import { LocalPortSection } from './local-port-section' import { LocalPortDetailsDialog } from './local-port-details-dialog' +/** Right-sidebar Ports panel scoped to the active workspace's owner host. */ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): React.JSX.Element { const activeWorktree = useActiveWorktree() const activeRepo = useRepoById(activeWorktree?.repoId ?? null) @@ -31,8 +35,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle) const scansByKey = useAppStore((s) => s.workspacePortScansByKey) const refreshing = useAppStore((s) => s.workspacePortScanRefreshing) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const [detailsPort, setDetailsPort] = useState(null) const [collapsedSections, setCollapsedSections] = useState>({ @@ -40,26 +43,24 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): external: true }) - const runtimeTarget = useMemo(() => { - const activeRuntimeEnvironmentId = getRuntimeEnvironmentIdForWorktree( - useAppStore.getState(), - activeWorktree?.id - ) - // Why: the Ports panel acts on the active workspace; use that workspace's - // host owner even if the sidebar is focused elsewhere. - return getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId }) - }, [activeWorktree?.id, settings]) - const scanKey = `${workspacePortRuntimeTargetKey(runtimeTarget)}:all` + // Why: the Ports panel acts on the active workspace; use that workspace's + // host owner even if the sidebar is focused elsewhere. + const runtimeTarget = useWorktreeRuntimeTarget(activeWorktree?.id) + const scanKey = runtimeTarget ? `${workspacePortRuntimeTargetKey(runtimeTarget)}:all` : null const refresh = useCallback(() => { - if (!activeRepo) { + if (!activeRepo || !runtimeTarget || !scanKey) { return Promise.resolve() } setWorkspacePortScanRefreshing(true) const promise = scanWorkspacePortsForTarget(runtimeTarget) .then((nextScan) => { - setWorkspacePortScanForKey(scanKey, nextScan) - setWorkspacePortScan({ key: scanKey, result: nextScan }) + publishWorkspacePortScanForHost({ + scanKey, + scan: nextScan, + replaceWorkspacePortScans, + getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey + }) }) .catch((error) => { const message = error instanceof Error ? error.message : String(error) @@ -86,14 +87,13 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): activeRepo, runtimeTarget, scanKey, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ]) // Why: WorkspacePortScanner already owns the 30s all-worktree poll. The // panel scopes that shared result instead of starting a second scan loop. - const displayScan = isVisible ? (scansByKey[scanKey] ?? null) : null + const displayScan = isVisible && scanKey ? (scansByKey[scanKey] ?? null) : null const toggleSection = useCallback((sectionId: string) => { setCollapsedSections((current) => ({ ...current, [sectionId]: !current[sectionId] })) @@ -122,8 +122,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): ) const refreshResult = await refreshWorkspacePortScanAfterStop({ runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey, setWorkspacePortScanRefreshing }) @@ -139,13 +138,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): ) } }, - [ - activeRepo, - runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, - setWorkspacePortScanRefreshing - ] + [activeRepo, runtimeTarget, replaceWorkspacePortScans, setWorkspacePortScanRefreshing] ) const handleOpenPortInBrowser = useCallback( @@ -181,6 +174,12 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): [activeRepo?.id, activeWorktree?.id, displayScan] ) + const showPortSections = shouldShowLocalWorkspacePortSections(displayScan, { + activePorts, + otherWorkspacePorts, + externalPorts + }) + if (!activeRepo) { return (
@@ -209,7 +208,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): size="icon-xs" className="text-muted-foreground hover:text-foreground" onClick={() => void refresh()} - disabled={refreshing} + disabled={refreshing || !runtimeTarget} aria-label={translate( 'auto.components.right.sidebar.PortsPanel.7822e3edc6', 'Refresh Ports' @@ -237,7 +236,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
)} - {!displayScan?.unavailableReason && ( + {showPortSections && ( <> ({ usePromptCacheCountdownStartedAt: vi.fn() @@ -52,7 +52,7 @@ vi.mock('@/store', () => ({ recordFeatureInteraction, remoteBranchConflictByWorktreeId: {}, setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing, settings, sshConnectionStates: new Map(), diff --git a/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx b/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx index e8376a64f08..2c1f10945d3 100644 --- a/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeCard.compact-ports-hover-independence.test.tsx @@ -13,7 +13,7 @@ import type { WorkspacePortScanResult } from '../../../../shared/workspace-ports const fetchHostedReviewForBranch = vi.fn() const fetchIssue = vi.fn() const fetchLinearIssue = vi.fn() -const setWorkspacePortScan = vi.fn() +const replaceWorkspacePortScans = vi.fn() const setWorkspacePortScanRefreshing = vi.fn() const cacheTimerMocks = vi.hoisted(() => ({ usePromptCacheCountdownStartedAt: vi.fn() @@ -43,7 +43,7 @@ vi.mock('@/store', () => ({ recordFeatureInteraction: vi.fn(), remoteBranchConflictByWorktreeId: {}, setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing, settings, sshConnectionStates: new Map(), diff --git a/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx b/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx index 0a8bb4d248e..db23d18dbc2 100644 --- a/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx +++ b/src/renderer/src/components/sidebar/WorktreeCardPorts.test.tsx @@ -8,7 +8,7 @@ vi.mock('@/store', () => ({ selector({ createBrowserTab: vi.fn(), setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan: vi.fn(), + replaceWorkspacePortScans: vi.fn(), setWorkspacePortScanRefreshing: vi.fn(), settings: null }) diff --git a/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx b/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx index 7f7c411dfa2..cc4f567e50a 100644 --- a/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx +++ b/src/renderer/src/components/sidebar/WorktreeCardPorts.tsx @@ -1,11 +1,10 @@ -import React, { useCallback, useMemo } from 'react' +import React, { useCallback } from 'react' import { Plug, Copy, ExternalLink, Trash2 } from 'lucide-react' import { toast } from 'sonner' import { useAppStore } from '@/store' import { Button } from '@/components/ui/button' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' -import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' import { canStopWorkspacePort, getPortOpenBrowserTooltipLabel, @@ -97,21 +96,18 @@ function PortAction({ ) } +/** One port row on a sidebar worktree card, with open/copy/stop actions on its owner host. */ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element { const settings = useAppStore((s) => s.settings) const localhostLabelRoute = useLocalhostLabelRouteForPort(port) - const runtimeEnvironmentId = useAppStore((s) => - getRuntimeEnvironmentIdForWorktree(s, port.kind === 'workspace' ? port.owner.worktreeId : null) - ) + const createBrowserTab = useAppStore((s) => s.createBrowserTab) const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction) - const runtimeTarget = useMemo( - () => getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId: runtimeEnvironmentId }), - [runtimeEnvironmentId, settings] + const runtimeTarget = useWorktreeRuntimeTarget( + port.kind === 'workspace' ? port.owner.worktreeId : null ) const processLabel = port.processName ?? (port.pid ? `PID ${port.pid}` : 'Unknown process') const address = addressForPort(port) @@ -203,8 +199,7 @@ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element { ) const refreshResult = await refreshWorkspacePortScanAfterStop({ runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey, setWorkspacePortScanRefreshing }) @@ -226,8 +221,7 @@ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element { port, recordFeatureInteraction, runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ] ) diff --git a/src/renderer/src/components/status-bar/PortsStatusSegment.host-routing.test.tsx b/src/renderer/src/components/status-bar/PortsStatusSegment.host-routing.test.tsx new file mode 100644 index 00000000000..9b22826d6bd --- /dev/null +++ b/src/renderer/src/components/status-bar/PortsStatusSegment.host-routing.test.tsx @@ -0,0 +1,407 @@ +// @vitest-environment happy-dom + +import React, { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { WorkspacePort, WorkspacePortScanResult } from '../../../../shared/workspace-ports' + +const { popoverHandle, runWorkspacePortScanForTargetMock, storeState } = vi.hoisted(() => { + const storeState = { + settings: { activeRuntimeEnvironmentId: null as string | null }, + activeWorktreeId: 'runtime-repo::/srv/app', + workspacePortScan: null as { key: string; result: WorkspacePortScanResult } | null, + workspacePortScansByKey: {} as Record, + workspacePortScanRefreshing: false, + runtimeEnvironments: [] as { id: string; name: string }[], + recordFeatureInteraction: vi.fn(), + replaceWorkspacePortScans: + vi.fn< + ( + scansByKey: Record, + projection: { key: string; result: WorkspacePortScanResult } | null + ) => void + >() + } + // Why: the real store writes back. A bare spy lets a publish and the notice + // that reads it drift onto different scan keys with every assertion green. + storeState.replaceWorkspacePortScans.mockImplementation((scansByKey, projection) => { + storeState.workspacePortScansByKey = scansByKey + storeState.workspacePortScan = projection + }) + return { + popoverHandle: { onOpenChange: null as ((open: boolean) => void) | null }, + runWorkspacePortScanForTargetMock: vi.fn(), + storeState + } +}) + +vi.mock('@/store', () => { + const useAppStore = Object.assign( + (selector: (state: typeof storeState) => unknown) => selector(storeState), + { getState: () => storeState } + ) + return { useAppStore } +}) + +vi.mock('@/lib/worktree-runtime-owner', () => ({ + getExecutionHostIdForWorktree: (_state: unknown, worktreeId: string | null | undefined) => { + if (worktreeId === 'runtime-repo::/srv/app') { + return 'runtime:env-1' + } + if (worktreeId === 'ssh-repo::/srv/app') { + return 'ssh:server-1' + } + return 'local' + } +})) + +vi.mock('@/runtime/runtime-rpc-client', async () => { + const actual = await import('@/runtime/runtime-client-target') + return { + getActiveRuntimeTarget: actual.getActiveRuntimeTarget, + callRuntimeRpc: vi.fn(), + assertRuntimeEnvironmentCapability: vi.fn(), + RuntimeRpcCallError: class RuntimeRpcCallError extends Error { + code?: string + } + } +}) + +vi.mock('@/lib/workspace-port-scan-client', () => ({ + runWorkspacePortScanForTarget: runWorkspacePortScanForTargetMock +})) + +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorktree: vi.fn() +})) + +vi.mock('@/components/ui/popover', () => ({ + Popover: ({ + children, + onOpenChange + }: { + children: React.ReactNode + onOpenChange: (open: boolean) => void + }) => { + popoverHandle.onOpenChange = onOpenChange + return <>{children} + }, + PopoverContent: ({ children }: { children: React.ReactNode }) => <>{children}, + PopoverTrigger: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('@/components/ui/tooltip', () => ({ + Tooltip: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipContent: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipTrigger: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('@/components/SelectedTextCopyMenu', () => ({ + SelectedTextCopyMenu: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('./ports-status-popover-rows', () => ({ + PortRow: () =>
, + WorkspaceGroupRows: () =>
+})) + +vi.mock('@/i18n/i18n', () => ({ + translate: (_key: string, fallback: string, options?: Record) => + options + ? fallback.replace(/{{(\w+)}}/g, (_match, name: string) => String(options[name] ?? '')) + : fallback +})) + +import { PortsStatusSegment } from './PortsStatusSegment' + +function workspacePort(overrides: Partial & { port: number; id: string }) { + return { + bindHost: '0.0.0.0', + connectHost: '127.0.0.1', + port: overrides.port, + id: overrides.id, + pid: 4321, + processName: 'node', + protocol: 'http' as const, + kind: 'workspace' as const, + owner: { + worktreeId: 'runtime-repo::/srv/app', + repoId: 'runtime-repo', + displayName: 'runtime app', + path: '/srv/app', + confidence: 'cwd' as const + } + } +} + +const localHostScan: WorkspacePortScanResult = { + platform: 'linux', + scannedAt: 10, + ports: [workspacePort({ id: 'local-5173', port: 5173 })] +} + +const remoteHostScan: WorkspacePortScanResult = { + platform: 'linux', + scannedAt: 20, + ports: [workspacePort({ id: 'remote-3000', port: 3000 })] +} + +describe('PortsStatusSegment popover host routing', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + popoverHandle.onOpenChange = null + storeState.settings = { activeRuntimeEnvironmentId: null } + storeState.activeWorktreeId = 'runtime-repo::/srv/app' + storeState.workspacePortScan = null + storeState.workspacePortScansByKey = { 'local:all': localHostScan } + storeState.runtimeEnvironments = [{ id: 'env-1', name: 'linux-box' }] + storeState.recordFeatureInteraction.mockClear() + storeState.replaceWorkspacePortScans.mockClear() + runWorkspacePortScanForTargetMock.mockReset() + runWorkspacePortScanForTargetMock.mockResolvedValue(remoteHostScan) + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + act(() => { + root.render() + }) + }) + + afterEach(() => { + act(() => { + root.unmount() + }) + container.remove() + }) + + async function openPopover(): Promise { + await act(async () => { + popoverHandle.onOpenChange?.(true) + await Promise.resolve() + await Promise.resolve() + }) + } + + it("scans the active workspace's host, not the globally focused runtime", async () => { + await openPopover() + + expect(runWorkspacePortScanForTargetMock).toHaveBeenCalledWith( + { kind: 'environment', environmentId: 'env-1' }, + undefined + ) + expect(storeState.replaceWorkspacePortScans).toHaveBeenCalledTimes(1) + expect(storeState.workspacePortScansByKey['environment:env-1:all']).toBe(remoteHostScan) + }) + + it('keeps other hosts in the projection instead of overwriting it with one host', async () => { + await openPopover() + + const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(projection).toEqual({ + key: 'all-hosts:all', + result: expect.objectContaining({ + ports: expect.arrayContaining([ + expect.objectContaining({ port: 5173 }), + expect.objectContaining({ port: 3000 }) + ]) + }) + }) + }) + + it('publishes a failed scan under its own host without dropping other hosts', async () => { + runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed')) + + await openPopover() + + const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(projection.key).toBe('all-hosts:all') + expect(projection.result.ports).toEqual([expect.objectContaining({ port: 5173 })]) + expect(storeState.workspacePortScansByKey['environment:env-1:all']).toEqual( + expect.objectContaining({ unavailableReason: 'remote scan failed' }) + ) + }) + + it('keeps the failed host last-good ports while naming the failure', async () => { + storeState.workspacePortScansByKey = { + 'local:all': localHostScan, + 'environment:env-1:all': remoteHostScan + } + runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed')) + + await openPopover() + + // Why: one dropped scan must not clear the host's ports the way the + // background poll's debounce does not — the notice names the failure + // while the projection keeps serving the last-good rows. + const failed = storeState.workspacePortScansByKey['environment:env-1:all'] + expect(failed.unavailableReason).toBe('remote scan failed') + expect(failed.platform).toBe('linux') + expect(failed.ports).toEqual([expect.objectContaining({ port: 3000 })]) + const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(projection.key).toBe('all-hosts:all') + expect(projection.result.ports.map((port) => port.port).sort()).toEqual([3000, 5173]) + }) + + // Why: separate tests already cover "the failure is stored" and "a stored + // failure renders". Only this one proves both halves name the same scan key. + it('surfaces the host it just failed to scan on the next render', async () => { + runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed')) + + await openPopover() + act(() => { + root.render() + }) + + expect(container.textContent).toContain( + 'Port scan unavailable on linux-box: remote scan failed' + ) + }) + + it('names the host whose scan failed while another host still reports ports', () => { + act(() => { + root.unmount() + }) + storeState.workspacePortScansByKey = { + 'local:all': localHostScan, + 'environment:env-1:all': { + platform: 'linux', + scannedAt: 30, + ports: [], + unavailableReason: 'Remote connection dropped' + } + } + storeState.workspacePortScan = { key: 'all-hosts:all', result: localHostScan } + root = createRoot(container) + act(() => { + root.render() + }) + + expect(container.textContent).toContain( + 'Port scan unavailable on linux-box: Remote connection dropped' + ) + // The notice sits above the list rather than replacing it: a reachable + // host's count still renders. + expect(container.textContent).toContain('1 workspace') + }) + + // Why: a failed scan keeps the host's last-good ports, and the badge and + // header count them. Replacing the list with the notice left the popover + // claiming N ports over an empty body. + it('keeps the list under the notice when a failed scan retained its ports', () => { + act(() => { + root.unmount() + }) + storeState.activeWorktreeId = 'local-repo::/home/dev/app' + const retained: WorkspacePortScanResult = { + ...localHostScan, + unavailableReason: 'lsof is unavailable' + } + storeState.workspacePortScansByKey = { 'local:all': retained } + storeState.workspacePortScan = { key: 'local:all', result: retained } + root = createRoot(container) + act(() => { + root.render() + }) + + expect(container.textContent).toContain('1 workspace · 0 external') + expect(container.querySelectorAll('[data-testid="workspace-group-rows"]')).toHaveLength(1) + expect(container.textContent).toContain('Port scan unavailable on Local Linux') + }) + + // Why: total loss of contact is where naming the host matters most, and the + // merged projection can only offer platform 'unknown' and raw scan keys. + it('names every host when all of them failed with nothing left to list', () => { + act(() => { + root.unmount() + }) + const merged: WorkspacePortScanResult = { + platform: 'unknown', + scannedAt: 30, + ports: [], + unavailableReason: 'local:all: lsof is unavailable; environment:env-1:all: dropped' + } + storeState.workspacePortScansByKey = { + 'local:all': { + platform: 'darwin', + scannedAt: 30, + ports: [], + unavailableReason: 'lsof is unavailable' + }, + 'environment:env-1:all': { + platform: 'linux', + scannedAt: 30, + ports: [], + unavailableReason: 'dropped' + } + } + storeState.workspacePortScan = { key: 'all-hosts:all', result: merged } + root = createRoot(container) + act(() => { + root.render() + }) + + // Local label comes from the scan's own platform, not the renderer's + // userAgent — a paired web client is not the Orca host. + expect(container.textContent).toContain( + 'Port scan unavailable on Local Mac: lsof is unavailable' + ) + expect(container.textContent).toContain('Port scan unavailable on linux-box: dropped') + expect(container.textContent).not.toContain('unavailable on unknown') + expect(container.textContent).not.toContain('environment:env-1:all:') + // The notice takes over the body only when there is nothing left to list. + expect(container.querySelectorAll('[data-testid="workspace-group-rows"]')).toHaveLength(0) + expect(container.textContent).not.toContain('No workspace ports detected') + }) + + it('stays on the local host when the active workspace has no runtime owner', async () => { + act(() => { + root.unmount() + }) + storeState.activeWorktreeId = 'local-repo::/home/dev/app' + root = createRoot(container) + act(() => { + root.render() + }) + + await openPopover() + + expect(runWorkspacePortScanForTargetMock).toHaveBeenCalledWith({ kind: 'local' }, undefined) + const [nextScans, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [ + Record, + { key: string; result: WorkspacePortScanResult } + ] + expect(nextScans['local:all']).toBe(remoteHostScan) + expect(projection).toEqual({ + key: 'local:all', + result: remoteHostScan + }) + }) + + it('does not substitute the local host for a direct-SSH workspace', async () => { + act(() => { + root.unmount() + }) + storeState.activeWorktreeId = 'ssh-repo::/srv/app' + root = createRoot(container) + act(() => { + root.render() + }) + + await openPopover() + + expect(runWorkspacePortScanForTargetMock).not.toHaveBeenCalled() + expect(storeState.replaceWorkspacePortScans).not.toHaveBeenCalled() + expect(storeState.recordFeatureInteraction).toHaveBeenCalledWith('ports') + }) +}) diff --git a/src/renderer/src/components/status-bar/PortsStatusSegment.tsx b/src/renderer/src/components/status-bar/PortsStatusSegment.tsx index 0626137c010..b56f5432cc8 100644 --- a/src/renderer/src/components/status-bar/PortsStatusSegment.tsx +++ b/src/renderer/src/components/status-bar/PortsStatusSegment.tsx @@ -3,39 +3,83 @@ import { Plug, ChevronDown, ChevronRight, LoaderCircle } from 'lucide-react' import { Popover, PopoverContent, PopoverTrigger } from '@/components/ui/popover' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { useAppStore } from '@/store' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' import { + publishWorkspacePortScanForHost, scanWorkspacePortsForTarget, workspacePortScanKeyForTarget } from '@/lib/workspace-port-actions' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' +import { + getUnavailableWorkspacePortHosts, + type WorkspacePortHostRef +} from '@/lib/workspace-port-host-availability' +import { getLocalExecutionHostLabel } from '../../../../shared/execution-host' import { getExternalWorkspacePorts, getWorkspacePortGroups } from '@/lib/workspace-port-groups' import { SelectedTextCopyMenu } from '@/components/SelectedTextCopyMenu' import { STATUS_BAR_CONTEXT_MENU_EXEMPT_PROPS } from './status-bar-context-menu-policy' import { PortRow, WorkspaceGroupRows } from './ports-status-popover-rows' import { translate } from '@/i18n/i18n' +import type { WorkspacePortScanResult } from '../../../../shared/workspace-ports' type PortsStatusSegmentProps = { compact?: boolean iconOnly: boolean } +/** Status-bar plug icon with the workspace port count and a per-host ports popover. */ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React.JSX.Element { - const settings = useAppStore((s) => s.settings) const scan = useAppStore((s) => s.workspacePortScan?.result ?? null) const refreshing = useAppStore((s) => s.workspacePortScanRefreshing) const activeWorktreeId = useAppStore((s) => s.activeWorktreeId) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) + const scansByKey = useAppStore((s) => s.workspacePortScansByKey) + const runtimeEnvironments = useAppStore((s) => s.runtimeEnvironments) const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction) const [open, setOpen] = useState(false) const [externalOpen, setExternalOpen] = useState(false) - const runtimeTarget = useMemo(() => getActiveRuntimeTarget(settings), [settings]) - const scanKey = workspacePortScanKeyForTarget(runtimeTarget) + const runtimeTarget = useWorktreeRuntimeTarget(activeWorktreeId) + const scanKey = runtimeTarget ? workspacePortScanKeyForTarget(runtimeTarget) : null const workspaceGroups = useMemo(() => getWorkspacePortGroups(scan), [scan]) const externalPorts = useMemo(() => getExternalWorkspacePorts(scan), [scan]) + const unavailableHosts = useMemo(() => getUnavailableWorkspacePortHosts(scansByKey), [scansByKey]) + const hostLabel = useCallback( + (host: WorkspacePortHostRef, hostScanKey: string, platform: NodeJS.Platform | null) => { + if (host.kind === 'local') { + // Why: a paired web client's own userAgent is not the Orca host's + // platform, so name the machine the scan actually ran on. + return getLocalExecutionHostLabel(platform) + } + if (host.kind === 'unknown') { + return hostScanKey + } + return ( + runtimeEnvironments.find((environment) => environment.id === host.environmentId)?.name ?? + host.environmentId + ) + }, + [runtimeEnvironments] + ) const workspacePortCount = workspaceGroups.reduce((count, group) => count + group.ports.length, 0) const totalCount = workspacePortCount + externalPorts.length + const unavailableNotices = useMemo(() => { + if (unavailableHosts.length > 0) { + return unavailableHosts.map((entry) => ({ + id: entry.scanKey, + host: hostLabel(entry.host, entry.scanKey, entry.platform), + reason: entry.reason + })) + } + // Why: a projection published without per-host scans has no host to name. + return scan?.unavailableReason + ? [{ id: 'projection', host: scan.platform, reason: scan.unavailableReason }] + : [] + }, [hostLabel, scan?.platform, scan?.unavailableReason, unavailableHosts]) + // Why: a failed scan keeps the host's last-good ports, and those ports are + // counted in the badge and header — replacing the list with the notice would + // leave the popover claiming N ports over an empty body. Only take over the + // body when there is genuinely nothing left to list. + const noticeReplacesList = Boolean(scan?.unavailableReason) && totalCount === 0 const handleOpenChange = useCallback( (nextOpen: boolean) => { setOpen(nextOpen) @@ -43,33 +87,36 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React return } recordFeatureInteraction('ports') + if (!runtimeTarget || !scanKey) { + return + } // Why: the 30s background poll is intentionally quiet; opening the // popover should still collapse that stale window without flashing icons. - void scanWorkspacePortsForTarget(runtimeTarget) - .then((result) => { - setWorkspacePortScanForKey(scanKey, result) - setWorkspacePortScan({ key: scanKey, result }) + const publish = (result: WorkspacePortScanResult): void => { + publishWorkspacePortScanForHost({ + scanKey, + scan: result, + replaceWorkspacePortScans, + getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey }) + } + void scanWorkspacePortsForTarget(runtimeTarget) + .then(publish) .catch((error) => { const message = error instanceof Error ? error.message : String(error) - setWorkspacePortScan({ - key: scanKey, - result: { - platform: 'unknown', - scannedAt: Date.now(), - ports: [], - unavailableReason: message || 'Workspace port scan failed.' - } + // Why: one dropped scan must not clear the host's last-good ports the + // way the background poll's debounce does not; the failure is still + // recorded so the host is named by the unavailable notice below. + const previous = useAppStore.getState().workspacePortScansByKey[scanKey] + publish({ + platform: previous?.platform ?? 'unknown', + scannedAt: Date.now(), + ports: previous?.ports ?? [], + unavailableReason: message || 'Workspace port scan failed.' }) }) }, - [ - recordFeatureInteraction, - runtimeTarget, - scanKey, - setWorkspacePortScan, - setWorkspacePortScanForKey - ] + [recordFeatureInteraction, runtimeTarget, scanKey, replaceWorkspacePortScans] ) return ( @@ -153,14 +200,18 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React
- {scan?.unavailableReason ? ( -
- {translate( - 'auto.components.status.bar.PortsStatusSegment.95495019ed', - 'Port scan unavailable on {{value0}}: {{value1}}', - { value0: scan.platform, value1: scan.unavailableReason } - )} -
+ {unavailableNotices.length > 0 && !noticeReplacesList && ( + + )} + + {noticeReplacesList ? ( + ) : (
{workspaceGroups.length > 0 ? ( @@ -237,3 +288,27 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React ) } + +type PortScanUnavailableNotice = { id: string; host: string; reason: string } + +function PortScanUnavailableNotices({ + notices, + className +}: { + notices: PortScanUnavailableNotice[] + className: string +}): React.JSX.Element { + return ( +
+ {notices.map((notice) => ( +
+ {translate( + 'auto.components.status.bar.PortsStatusSegment.95495019ed', + 'Port scan unavailable on {{value0}}: {{value1}}', + { value0: notice.host, value1: notice.reason } + )} +
+ ))} +
+ ) +} diff --git a/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx b/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx index 47c86d89c57..ba926909272 100644 --- a/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx +++ b/src/renderer/src/components/status-bar/ports-status-popover-rows.test.tsx @@ -17,8 +17,7 @@ const { settings: { openLinksInApp: true }, createBrowserTab: vi.fn(), setRemoteBrowserPageHandle: vi.fn(), - setWorkspacePortScan: vi.fn(), - setWorkspacePortScanForKey: vi.fn(), + replaceWorkspacePortScans: vi.fn(), setWorkspacePortScanRefreshing: vi.fn(), recordFeatureInteraction: vi.fn(), workspacePortScansByKey: {} @@ -46,7 +45,7 @@ vi.mock('@/lib/worktree-activation', () => ({ })) vi.mock('@/lib/worktree-runtime-owner', () => ({ - getRuntimeEnvironmentIdForWorktree: () => null + getExecutionHostIdForWorktree: () => 'local' })) vi.mock('@/runtime/runtime-rpc-client', () => ({ diff --git a/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx b/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx index c6d7eb41625..4c07c44ed21 100644 --- a/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx +++ b/src/renderer/src/components/status-bar/ports-status-popover-rows.tsx @@ -1,4 +1,4 @@ -import React, { useCallback, useMemo } from 'react' +import React, { useCallback } from 'react' import { Copy, ExternalLink, FolderOpen, Trash2 } from 'lucide-react' import { toast } from 'sonner' import { Button } from '@/components/ui/button' @@ -15,9 +15,8 @@ import { } from '@/lib/workspace-port-actions' import type { WorkspacePortGroup } from '@/lib/workspace-port-groups' import { useLocalhostLabelRouteForPort } from '@/lib/workspace-port-localhost-label-selector' -import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' +import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target' import { useAppStore } from '@/store' -import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' import type { WorkspacePort } from '../../../../shared/workspace-ports' import { translate } from '@/i18n/i18n' @@ -67,6 +66,7 @@ function PortAction({ ) } +/** One port row in the status-bar popover, with open/copy/stop actions on its owner host. */ export function PortRow({ port, activeWorktreeId, @@ -78,21 +78,13 @@ export function PortRow({ }): React.JSX.Element { const settings = useAppStore((s) => s.settings) const localhostLabelRoute = useLocalhostLabelRouteForPort(port) - const runtimeEnvironmentId = useAppStore((s) => - getRuntimeEnvironmentIdForWorktree( - s, - port.kind === 'workspace' ? port.owner.worktreeId : activeWorktreeId - ) - ) const createBrowserTab = useAppStore((s) => s.createBrowserTab) const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle) - const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan) - const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey) + const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans) const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing) const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction) - const runtimeTarget = useMemo( - () => getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId: runtimeEnvironmentId }), - [runtimeEnvironmentId, settings] + const runtimeTarget = useWorktreeRuntimeTarget( + port.kind === 'workspace' ? port.owner.worktreeId : activeWorktreeId ) const processLabel = port.processName ?? (port.pid ? `PID ${port.pid}` : 'Unknown process') const canStop = canStopWorkspacePort(port) @@ -187,8 +179,7 @@ export function PortRow({ ) const refreshResult = await refreshWorkspacePortScanAfterStop({ runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey, setWorkspacePortScanRefreshing }) @@ -210,8 +201,7 @@ export function PortRow({ port, recordFeatureInteraction, runtimeTarget, - setWorkspacePortScan, - setWorkspacePortScanForKey, + replaceWorkspacePortScans, setWorkspacePortScanRefreshing ] ) diff --git a/src/renderer/src/lib/workspace-port-actions.ts b/src/renderer/src/lib/workspace-port-actions.ts index cca201234a4..cf745d3e25f 100644 --- a/src/renderer/src/lib/workspace-port-actions.ts +++ b/src/renderer/src/lib/workspace-port-actions.ts @@ -7,7 +7,6 @@ import { type RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector' -import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host' import type { WorkspacePort, WorkspacePortKillResult, @@ -22,6 +21,11 @@ import { RUNTIME_BROWSER_UNAVAILABLE_MESSAGE } from './client-creation-action-po export { addressForPort } from './workspace-port-urls' const WORKSPACE_PORT_STOP_SETTLE_MS = 500 +const WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON = + 'Workspace ports are unavailable for this execution host.' + +/** Projection key for the merged multi-host view; never a per-host scan key. */ +export const WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY = 'all-hosts:all' export function canStopWorkspacePort( port: WorkspacePort @@ -33,13 +37,17 @@ type BrowserTabCreator = ReturnType['createBrowserT type RemoteBrowserPageHandleSetter = ReturnType< typeof useAppStore.getState >['setRemoteBrowserPageHandle'] -type WorkspacePortScanSetter = ReturnType['setWorkspacePortScan'] -type WorkspacePortScanByKeySetter = ReturnType< - typeof useAppStore.getState ->['setWorkspacePortScanForKey'] type WorkspacePortScanRefreshingSetter = ReturnType< typeof useAppStore.getState >['setWorkspacePortScanRefreshing'] +type ReplaceWorkspacePortScansSetter = ReturnType< + typeof useAppStore.getState +>['replaceWorkspacePortScans'] + +export type WorkspacePortScanPublisher = { + replaceWorkspacePortScans: ReplaceWorkspacePortScansSetter + getWorkspacePortScansByKey: () => Record +} function delay(ms: number): Promise { return new Promise((resolve) => window.setTimeout(resolve, ms)) @@ -94,12 +102,15 @@ export function goToWorkspacePortOwner(port: WorkspacePort): boolean { export async function openWorkspacePortInBrowser(args: { port: WorkspacePort activeWorktreeId?: string | null - runtimeTarget: RuntimeClientTarget + runtimeTarget: RuntimeClientTarget | null createBrowserTab: BrowserTabCreator setRemoteBrowserPageHandle: RemoteBrowserPageHandleSetter openInOrcaBrowser?: boolean localhostLabelRoute?: LocalhostWorktreeLabelRoute | null }): Promise<{ ok: true } | { ok: false; reason: string }> { + if (!args.runtimeTarget) { + return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON } + } const rawUrl = browserUrlForPort(args.port) let url = rawUrl if (args.runtimeTarget.kind === 'local' && args.localhostLabelRoute) { @@ -166,22 +177,38 @@ export async function openWorkspacePortInBrowser(args: { } } -export async function refreshWorkspacePortScanAfterStop(args: { - runtimeTarget: RuntimeClientTarget - setWorkspacePortScan: WorkspacePortScanSetter - setWorkspacePortScanForKey?: WorkspacePortScanByKeySetter - setWorkspacePortScanRefreshing: WorkspacePortScanRefreshingSetter - getWorkspacePortScansByKey?: () => Record -}): Promise<{ ok: true } | { ok: false; reason: string }> { +/** + * Stores one host's scan and republishes the aggregate the status bar reads. + * Why: a single-host publish used to overwrite that aggregate, so every other + * host's ports vanished from the count until the next background poll. One + * replaceWorkspacePortScans update (not setWorkspacePortScan) keeps the synthetic + * all-hosts key out of workspacePortScansByKey, where re-merging it would + * duplicate rows — and notifies subscribers once instead of twice for one scan. + */ +export function publishWorkspacePortScanForHost( + args: WorkspacePortScanPublisher & { scanKey: string; scan: WorkspacePortScanResult } +): void { + const scansByKey = { ...args.getWorkspacePortScansByKey(), [args.scanKey]: args.scan } + const merged = mergeWorkspacePortScans(scansByKey) + args.replaceWorkspacePortScans(scansByKey, { + key: Object.keys(scansByKey).length > 1 ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : args.scanKey, + result: merged ?? args.scan + }) +} + +/** Re-scans one host after a port stop (immediately, then settled) and republishes the aggregate. */ +export async function refreshWorkspacePortScanAfterStop( + args: WorkspacePortScanPublisher & { + runtimeTarget: RuntimeClientTarget | null + setWorkspacePortScanRefreshing: WorkspacePortScanRefreshingSetter + } +): Promise<{ ok: true } | { ok: false; reason: string }> { + if (!args.runtimeTarget) { + return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON } + } const scanKey = workspacePortScanKeyForTarget(args.runtimeTarget) const publishScan = (scan: WorkspacePortScanResult): void => { - args.setWorkspacePortScanForKey?.(scanKey, scan) - const currentScans = args.getWorkspacePortScansByKey?.() ?? {} - const merged = mergeWorkspacePortScans({ ...currentScans, [scanKey]: scan }) - args.setWorkspacePortScan({ - key: merged && Object.keys(currentScans).length > 0 ? 'all-hosts:all' : scanKey, - result: merged ?? scan - }) + publishWorkspacePortScanForHost({ ...args, scanKey, scan }) } args.setWorkspacePortScanRefreshing(true) try { @@ -216,19 +243,6 @@ export function workspacePortRuntimeTargetKey(target: RuntimeClientTarget): stri return target.kind === 'local' ? 'local' : `environment:${target.environmentId}` } -export function runtimeTargetForExecutionHostId( - hostId: ExecutionHostId -): RuntimeClientTarget | null { - const parsed = parseExecutionHostId(hostId) - if (parsed?.kind === 'local') { - return { kind: 'local' } - } - if (parsed?.kind === 'runtime') { - return { kind: 'environment', environmentId: parsed.environmentId } - } - return null -} - export function workspacePortScanKeyForTarget(target: RuntimeClientTarget): string { return `${workspacePortRuntimeTargetKey(target)}:all` } @@ -295,9 +309,12 @@ export async function scanWorkspacePortsForTarget( } export async function killWorkspacePortForTarget( - target: RuntimeClientTarget, + target: RuntimeClientTarget | null, args: { repoId: string; pid: number; port: number } ): Promise { + if (!target) { + return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON } + } if (target.kind === 'local') { return window.api.workspacePorts.kill(args) } diff --git a/src/renderer/src/lib/workspace-port-host-availability.test.ts b/src/renderer/src/lib/workspace-port-host-availability.test.ts new file mode 100644 index 00000000000..7652c65a21a --- /dev/null +++ b/src/renderer/src/lib/workspace-port-host-availability.test.ts @@ -0,0 +1,148 @@ +import { describe, expect, it } from 'vitest' +import type { WorkspacePortScanResult } from '../../../shared/workspace-ports' +import { + getUnavailableWorkspacePortHosts, + workspacePortHostForScanKey +} from './workspace-port-host-availability' + +function scan(overrides: Partial = {}): WorkspacePortScanResult { + return { platform: 'linux', scannedAt: 1, ports: [], ...overrides } +} + +describe('getUnavailableWorkspacePortHosts', () => { + it('reports the failed host while another host still answers', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan(), + 'environment:env-1:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'environment:env-1:all', + host: { kind: 'environment', environmentId: 'env-1' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + it('reports the local host as a local host ref, not an absent environment id', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ unavailableReason: 'lsof is unavailable' }), + 'environment:env-1:all': scan() + }) + ).toEqual([ + { + scanKey: 'local:all', + host: { kind: 'local' }, + platform: 'linux', + reason: 'lsof is unavailable' + } + ]) + }) + + it('keeps colons inside an environment id when parsing the scan key', () => { + // Why: keys are `${targetKey}:all`, so the id runs to the last `:all` — + // splitting on the first colon would truncate ids that contain colons. + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan(), + 'environment:weird:id:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'environment:weird:id:all', + host: { kind: 'environment', environmentId: 'weird:id' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + // Why: total loss of contact is where naming the host matters most — the merged + // projection joins raw internal scan keys, so it cannot name them itself. + it('names every host when all of them failed', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ unavailableReason: 'lsof is unavailable' }), + 'environment:env-1:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'local:all', + host: { kind: 'local' }, + platform: 'linux', + reason: 'lsof is unavailable' + }, + { + scanKey: 'environment:env-1:all', + host: { kind: 'environment', environmentId: 'env-1' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + it('names a single failed host', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ unavailableReason: 'lsof is unavailable' }) + }) + ).toEqual([ + { + scanKey: 'local:all', + host: { kind: 'local' }, + platform: 'linux', + reason: 'lsof is unavailable' + } + ]) + }) + + // Why: the synthetic all-hosts projection key must never be labelled as the + // local machine — that would blame the wrong host for a remote failure. + it('marks an unrecognised scan key as an unknown host', () => { + expect( + getUnavailableWorkspacePortHosts({ + 'all-hosts:all': scan({ unavailableReason: 'Remote connection dropped' }) + }) + ).toEqual([ + { + scanKey: 'all-hosts:all', + host: { kind: 'unknown' }, + platform: 'linux', + reason: 'Remote connection dropped' + } + ]) + }) + + // Why: a paired web client's userAgent is not the Orca host's platform, so the + // caller labels the local host from the scan's own platform. + it("carries the failed scan's platform, and null when it is unknown", () => { + expect( + getUnavailableWorkspacePortHosts({ + 'local:all': scan({ platform: 'win32', unavailableReason: 'netstat failed' }), + 'environment:env-1:all': scan({ platform: 'unknown', unavailableReason: 'dropped' }) + }).map((entry) => entry.platform) + ).toEqual(['win32', null]) + }) + + it('stays silent when nothing failed', () => { + expect(getUnavailableWorkspacePortHosts({ 'local:all': scan() })).toEqual([]) + expect(getUnavailableWorkspacePortHosts({})).toEqual([]) + }) +}) + +describe('workspacePortHostForScanKey', () => { + it.each([ + ['local:all', { kind: 'local' }], + ['environment:env-1:all', { kind: 'environment', environmentId: 'env-1' }], + ['environment:weird:id:all', { kind: 'environment', environmentId: 'weird:id' }], + ['all-hosts:all', { kind: 'unknown' }], + ['environment::all', { kind: 'unknown' }], + ['environment:env-1', { kind: 'unknown' }], + ['local', { kind: 'unknown' }] + ])('maps %s', (scanKey, expected) => { + expect(workspacePortHostForScanKey(scanKey)).toEqual(expected) + }) +}) diff --git a/src/renderer/src/lib/workspace-port-host-availability.ts b/src/renderer/src/lib/workspace-port-host-availability.ts new file mode 100644 index 00000000000..4c643a5a118 --- /dev/null +++ b/src/renderer/src/lib/workspace-port-host-availability.ts @@ -0,0 +1,65 @@ +import type { WorkspacePortScanResult } from '../../../shared/workspace-ports' + +/** + * Host a per-host scan key points at. `unknown` is kept distinct from `local` so + * an unrecognised key (a synthetic projection key that leaked into the per-host + * map, say) is never mislabelled as a local failure. + */ +export type WorkspacePortHostRef = + | { kind: 'local' } + | { kind: 'environment'; environmentId: string } + | { kind: 'unknown' } + +export type UnavailableWorkspacePortHost = { + scanKey: string + host: WorkspacePortHostRef + /** Platform the failed scan last ran on; drives the local host's label. */ + platform: NodeJS.Platform | null + reason: string +} + +// Why: mirrors workspacePortScanKeyForTarget (`${targetKey}:all`, where the +// target key is `local` or `environment:`) without importing the heavier +// workspace-port-actions module into this pure helper. Splitting on the last +// `:all` keeps environment ids that themselves contain colons intact. +const ENVIRONMENT_SCAN_KEY_PREFIX = 'environment:' +const SCAN_KEY_SUFFIX = ':all' +const LOCAL_SCAN_KEY = `local${SCAN_KEY_SUFFIX}` + +/** Host a per-host scan key names; `unknown` for any other key shape. */ +export function workspacePortHostForScanKey(scanKey: string): WorkspacePortHostRef { + if (scanKey === LOCAL_SCAN_KEY) { + return { kind: 'local' } + } + if (!scanKey.endsWith(SCAN_KEY_SUFFIX) || !scanKey.startsWith(ENVIRONMENT_SCAN_KEY_PREFIX)) { + return { kind: 'unknown' } + } + const environmentId = scanKey.slice( + ENVIRONMENT_SCAN_KEY_PREFIX.length, + scanKey.length - SCAN_KEY_SUFFIX.length + ) + return environmentId ? { kind: 'environment', environmentId } : { kind: 'unknown' } +} + +/** + * Every host whose latest scan failed, named by host rather than by scan key. + * Why: on a remote host "none listening" and "could not look" are different + * answers, and the merged projection collapses both the partial case (no reason + * at all) and the total case (reasons joined with raw internal keys). + */ +export function getUnavailableWorkspacePortHosts( + scansByKey: Record +): UnavailableWorkspacePortHost[] { + return Object.entries(scansByKey).flatMap(([scanKey, scan]) => + scan?.unavailableReason + ? [ + { + scanKey, + host: workspacePortHostForScanKey(scanKey), + platform: scan.platform === 'unknown' ? null : scan.platform, + reason: scan.unavailableReason + } + ] + : [] + ) +} diff --git a/src/renderer/src/lib/workspace-port-scan-debounce.test.ts b/src/renderer/src/lib/workspace-port-scan-debounce.test.ts index bacac7ecacf..747de6672aa 100644 --- a/src/renderer/src/lib/workspace-port-scan-debounce.test.ts +++ b/src/renderer/src/lib/workspace-port-scan-debounce.test.ts @@ -25,6 +25,11 @@ function unavailable(): WorkspacePortScanResult { return { platform: 'unknown', scannedAt: 1, ports: [], unavailableReason: 'scan failed' } } +/** What the ports popover publishes when its own scan fails: reason + last-good ports. */ +function unavailableWithRetainedPorts(portIds: string[]): WorkspacePortScanResult { + return { ...good(portIds), unavailableReason: 'scan failed' } +} + const FAILURE_THRESHOLD = 2 function createHarness(): { @@ -125,6 +130,33 @@ describe('reconcileTransientPortScanFailures', () => { expect(state.has('flaky:all')).toBe(false) }) + // Why: the popover publishes reason + last-good ports the moment its own scan + // fails. Counting that as a spent grace period would drop those ports on the + // very next poll, so the retention would never survive one poll interval. + it('still grants the grace period after a failure that retained its ports', () => { + const { apply, publish } = createHarness() + apply([{ key: 'h:all', result: good(['tcp:3000']) }]) + const popoverResult = unavailableWithRetainedPorts(['tcp:3000']) + publish('h:all', popoverResult) + + const next = apply([{ key: 'h:all', result: unavailable() }]) + + expect(next[0].result).toBe(popoverResult) + expect(next[0].result.ports).toHaveLength(1) + }) + + it('still drops retained ports once failures reach the tolerance', () => { + const { apply, publish } = createHarness() + apply([{ key: 'h:all', result: good(['tcp:3000']) }]) + publish('h:all', unavailableWithRetainedPorts(['tcp:3000'])) + apply([{ key: 'h:all', result: unavailable() }]) + + const next = apply([{ key: 'h:all', result: unavailable() }]) + + expect(next[0].result.ports).toHaveLength(0) + expect(next[0].result.unavailableReason).toBe('scan failed') + }) + it('uses a newer manual result instead of resurrecting stale ports', () => { const { apply, publish } = createHarness() apply([{ key: 'h:all', result: good(['tcp:3000']) }]) diff --git a/src/renderer/src/lib/workspace-port-scan-debounce.ts b/src/renderer/src/lib/workspace-port-scan-debounce.ts index a6281f0ed9f..2677cbb0e52 100644 --- a/src/renderer/src/lib/workspace-port-scan-debounce.ts +++ b/src/renderer/src/lib/workspace-port-scan-debounce.ts @@ -34,10 +34,14 @@ export function reconcileTransientPortScanFailures( return { key, result } } const failures = previousFailures + 1 + // Why: a surface that hit the same failure first (the ports popover) republishes + // the host's last-good ports alongside the reason. Treating that as a spent grace + // period would drop those ports on the very next poll, undoing the retention. + const publishedIsRetainable = + Boolean(publishedResult) && + (!publishedResult.unavailableReason || publishedResult.ports.length > 0) const nextResult = - failures < failureThreshold && publishedResult && !publishedResult.unavailableReason - ? publishedResult - : result + failures < failureThreshold && publishedIsRetainable ? publishedResult : result state.set(key, { consecutiveFailures: failures, publishedResult: nextResult }) return { key, result: nextResult } }) diff --git a/src/renderer/src/lib/workspace-port-scan-publish.test.ts b/src/renderer/src/lib/workspace-port-scan-publish.test.ts new file mode 100644 index 00000000000..2f1386855a8 --- /dev/null +++ b/src/renderer/src/lib/workspace-port-scan-publish.test.ts @@ -0,0 +1,153 @@ +// @vitest-environment happy-dom + +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { WorkspacePortScanResult } from '../../../shared/workspace-ports' + +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorktree: vi.fn() +})) + +vi.mock('@/runtime/runtime-rpc-client', () => ({ + getActiveRuntimeTarget: vi.fn(), + callRuntimeRpc: vi.fn(), + assertRuntimeEnvironmentCapability: vi.fn(), + RuntimeRpcCallError: class RuntimeRpcCallError extends Error { + code?: string + } +})) + +vi.mock('./workspace-port-scan-client', () => ({ + runWorkspacePortScanForTarget: vi.fn() +})) + +const { publishWorkspacePortScanForHost, WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY } = + await import('./workspace-port-actions') +type WorkspacePortScanPublisher = Parameters[0] + +function scanWithPort(port: number, scannedAt: number): WorkspacePortScanResult { + return { + platform: 'linux', + scannedAt, + ports: [ + { + id: `tcp:${port}`, + bindHost: '0.0.0.0', + connectHost: '127.0.0.1', + port, + pid: 100 + port, + processName: 'node', + protocol: 'http', + kind: 'external' + } + ] + } +} + +/** Mirrors the store's replaceWorkspacePortScans semantics: one atomic update. */ +function makeStoreHarness(initial: Record = {}): { + scansByKey: Record + projections: { key: string; result: WorkspacePortScanResult }[] + publisher: Omit +} { + let scansByKey: Record = { ...initial } + const projections: { key: string; result: WorkspacePortScanResult }[] = [] + return { + get scansByKey() { + return scansByKey + }, + projections, + publisher: { + replaceWorkspacePortScans: ( + nextScansByKey: Record, + projection: { key: string; result: WorkspacePortScanResult } | null + ) => { + scansByKey = nextScansByKey + if (projection) { + projections.push(projection) + } + }, + getWorkspacePortScansByKey: () => scansByKey + } + } +} + +describe('publishWorkspacePortScanForHost', () => { + let localScan: WorkspacePortScanResult + let remoteScan: WorkspacePortScanResult + + beforeEach(() => { + localScan = scanWithPort(5173, 10) + remoteScan = scanWithPort(3000, 20) + }) + + it('publishes the single tracked host under its own key', () => { + const harness = makeStoreHarness() + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'local:all', + scan: localScan + }) + + expect(harness.projections).toEqual([{ key: 'local:all', result: localScan }]) + expect(Object.keys(harness.scansByKey)).toEqual(['local:all']) + }) + + it('keeps the other host in the projection when one host refreshes', () => { + const harness = makeStoreHarness({ 'local:all': localScan }) + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: remoteScan + }) + + const projection = harness.projections.at(-1) + expect(projection?.key).toBe(WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY) + expect(projection?.result.ports.map((port) => port.port).sort()).toEqual([3000, 5173]) + expect(Object.keys(harness.scansByKey).sort()).toEqual(['environment:env-1:all', 'local:all']) + }) + + it('publishes map and projection in a single store update', () => { + const harness = makeStoreHarness({ 'local:all': localScan }) + const replaceSpy = vi.spyOn(harness.publisher, 'replaceWorkspacePortScans') + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: remoteScan + }) + + // Why: two sequential setter calls notify subscribers twice for one scan; + // one atomic replace keeps map and projection from ever disagreeing. + expect(replaceSpy).toHaveBeenCalledTimes(1) + const [nextScans, projection] = replaceSpy.mock.calls[0] + expect(Object.keys(nextScans).sort()).toEqual(['environment:env-1:all', 'local:all']) + expect(projection?.key).toBe(WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY) + }) + + it('does not accumulate duplicate rows across repeated publishes', () => { + const harness = makeStoreHarness({ 'local:all': localScan }) + + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: remoteScan + }) + publishWorkspacePortScanForHost({ + ...harness.publisher, + scanKey: 'environment:env-1:all', + scan: { ...remoteScan, scannedAt: 30 } + }) + + // Why: the aggregate must never land in the per-host map, or the next merge + // folds the merged result back into itself and rows multiply. + expect(harness.scansByKey[WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY]).toBeUndefined() + expect( + harness.projections + .at(-1) + ?.result.ports.map((port) => port.port) + .sort() + ).toEqual([3000, 5173]) + }) +}) diff --git a/src/renderer/src/runtime/runtime-client-target.ts b/src/renderer/src/runtime/runtime-client-target.ts index 1a0b8e4b8a4..fbf9af17374 100644 --- a/src/renderer/src/runtime/runtime-client-target.ts +++ b/src/renderer/src/runtime/runtime-client-target.ts @@ -1,4 +1,5 @@ import type { GlobalSettings } from '../../../shared/global-settings-types' +import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host' export type RuntimeClientTarget = { kind: 'local' } | { kind: 'environment'; environmentId: string } @@ -9,6 +10,20 @@ export function getActiveRuntimeTarget( return environmentId ? { kind: 'environment', environmentId } : { kind: 'local' } } +/** RPC target for a dispatchable host; direct SSH cannot use this client path. */ +export function runtimeTargetForExecutionHostId( + hostId: ExecutionHostId +): RuntimeClientTarget | null { + const parsed = parseExecutionHostId(hostId) + if (parsed?.kind === 'local') { + return { kind: 'local' } + } + if (parsed?.kind === 'runtime') { + return { kind: 'environment', environmentId: parsed.environmentId } + } + return null +} + export function settingsForRuntimeOwner( settings: Pick | null | undefined, runtimeEnvironmentId: string | null | undefined diff --git a/src/renderer/src/runtime/use-worktree-runtime-target.ts b/src/renderer/src/runtime/use-worktree-runtime-target.ts new file mode 100644 index 00000000000..25bd9099e90 --- /dev/null +++ b/src/renderer/src/runtime/use-worktree-runtime-target.ts @@ -0,0 +1,16 @@ +import { useAppStore } from '@/store' +import { getExecutionHostIdForWorktree } from '@/lib/worktree-runtime-owner' +import { runtimeTargetForExecutionHostId, type RuntimeClientTarget } from './runtime-client-target' + +/** + * Runtime target that owns `worktreeId`, which is not always the globally + * focused runtime — acting on the focused one scans the wrong host and reports + * that workspace as having no ports. Direct-SSH owners return null. + */ +export function useWorktreeRuntimeTarget( + worktreeId: string | null | undefined +): RuntimeClientTarget | null { + return useAppStore((state) => + runtimeTargetForExecutionHostId(getExecutionHostIdForWorktree(state, worktreeId)) + ) +} From 3941edd4b6d474bf1c170cfbb7a0c80798d97bdc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 22:20:35 -0700 Subject: [PATCH 03/12] perf(ipc): build the filesystem allowed-root list once per authorization (#18423) * perf(ipc): build the filesystem allowed-root list once per authorization * perf(ipc): keep the allowed-root snapshot lazy so granted external paths build nothing Hoisting getAllowedRoots to the top of resolveAuthorizedPath made every read of a path covered by an external grant build the full root list, where main built none (the grant answered before isPathAllowed reached the roots). Build on first use instead: still one build per authorization, zero when a grant already answers. * test(ipc): skip the allowed-root symlink escapes on Windows Unprivileged Windows cannot create symlinks (EPERM), so both cases failed in setup instead of exercising the escape check. --- src/main/ipc/filesystem-allowed-roots.test.ts | 372 ++++++++++++++++++ src/main/ipc/filesystem-allowed-roots.ts | 59 ++- src/main/ipc/filesystem-auth.ts | 65 ++- src/main/project-runtime-git-options.ts | 7 +- src/shared/project-groups.ts | 23 +- 5 files changed, 492 insertions(+), 34 deletions(-) create mode 100644 src/main/ipc/filesystem-allowed-roots.test.ts diff --git a/src/main/ipc/filesystem-allowed-roots.test.ts b/src/main/ipc/filesystem-allowed-roots.test.ts new file mode 100644 index 00000000000..f94c99c5fdb --- /dev/null +++ b/src/main/ipc/filesystem-allowed-roots.test.ts @@ -0,0 +1,372 @@ +import { mkdir, mkdtemp, realpath, rm, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import type * as RepoWorktrees from '../repo-worktrees' +import { listRepoWorktreeGraph } from '../repo-worktrees' +import type * as ProjectGroupsModule from '../../shared/project-groups' +import { buildProjectGroupChildIndex, getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { isPathInsideOrEqual } from '../../shared/cross-platform-path' +import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import type { FolderWorkspace } from '../../shared/folder-workspace-types' +import type { ProjectGroup } from '../../shared/project-group-types' +import type { Project } from '../../shared/project-types' +import type { Repo } from '../../shared/repo-types' +import { getAllowedRoots } from './filesystem-allowed-roots' +import { authorizeExternalPath, resolveAuthorizedPath } from './filesystem-auth' +import { invalidateAuthorizedRootsCache } from './registered-worktree-roots-cache' +import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' + +vi.mock('../repo-worktrees', async () => { + const actual = await vi.importActual('../repo-worktrees') + return { ...actual, listRepoWorktreeGraph: vi.fn(async () => []) } +}) + +vi.mock('../../shared/project-groups', async () => { + const actual = await vi.importActual('../../shared/project-groups') + return { + ...actual, + buildProjectGroupChildIndex: vi.fn(actual.buildProjectGroupChildIndex), + getProjectGroupSubtreeIds: vi.fn(actual.getProjectGroupSubtreeIds) + } +}) + +type StoreFixture = { + repos: Repo[] + projects: Project[] + projectGroups: ProjectGroup[] + folderWorkspaces: FolderWorkspace[] + workspaceDir?: string +} + +type StoreCallCounts = { + getRepos: number + getProjects: number + getProjectGroups: number + getFolderWorkspaces: number +} + +function makeCountingStore(fixture: StoreFixture): { store: Store; counts: StoreCallCounts } { + const counts: StoreCallCounts = { + getRepos: 0, + getProjects: 0, + getProjectGroups: 0, + getFolderWorkspaces: 0 + } + const store = { + getRepos: () => { + counts.getRepos += 1 + // Match the real store, which rehydrates fresh repo objects on every read. + return fixture.repos.map((repo) => ({ ...repo })) + }, + getProjects: () => { + counts.getProjects += 1 + return fixture.projects.map((project) => ({ ...project })) + }, + getProjectGroups: () => { + counts.getProjectGroups += 1 + return fixture.projectGroups.map((group) => ({ ...group })) + }, + getFolderWorkspaces: () => { + counts.getFolderWorkspaces += 1 + return fixture.folderWorkspaces.map((workspace) => ({ ...workspace })) + }, + getSettings: () => ({ nestWorkspaces: false, workspaceDir: fixture.workspaceDir ?? '' }) + } as unknown as Store + return { store, counts } +} + +/** + * The pre-change `getAllowedRoots` algorithm, kept verbatim so the equivalence test compares the + * new root list against the old one rather than against a hand-written expectation. + */ +function referenceAllowedRoots(store: Store): string[] { + const scopeStore = store as unknown as { + getRepos: () => Repo[] + getProjectGroups?: () => ProjectGroup[] + getFolderWorkspaces?: () => FolderWorkspace[] + getSettings: () => { workspaceDir?: string; nestWorkspaces?: boolean } + } + const localRepos = scopeStore.getRepos().filter((repo) => !repo.connectionId) + const settings = scopeStore.getSettings() + + const scopeRepos = scopeStore.getRepos() + const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const isRemoteOnly = ( + folderPath: string, + projectGroupId: string, + connectionId: string | null | undefined + ): boolean => { + if (connectionId) { + return true + } + const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const candidates = scopeRepos.filter( + (repo) => + (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || + isPathInsideOrEqual(folderPath, repo.path) + ) + return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) + } + const folderScopeRoots: string[] = [] + for (const group of projectGroups) { + if (group.parentPath && !isRemoteOnly(group.parentPath, group.id, group.connectionId)) { + folderScopeRoots.push(resolve(group.parentPath)) + } + } + for (const workspace of scopeStore.getFolderWorkspaces?.() ?? []) { + const connectionId = + workspace.connectionId ?? + projectGroups.find((group) => group.id === workspace.projectGroupId)?.connectionId ?? + null + if (!isRemoteOnly(workspace.folderPath, workspace.projectGroupId, connectionId)) { + folderScopeRoots.push(resolve(workspace.folderPath)) + } + } + + const roots = [...localRepos.map((repo) => resolve(repo.path)), ...folderScopeRoots] + if (settings.workspaceDir) { + if (localRepos.length === 0) { + roots.push(resolve(settings.workspaceDir)) + } else { + for (const repo of localRepos) { + roots.push( + resolve( + computeWorkspaceRoot( + repo.path, + getWorktreePathSettings(repo, settings as never, getWorktreeMirrorDistro(store, repo)) + ) + ) + ) + } + } + } + return roots +} + +function makeRepo(overrides: Partial & Pick): Repo { + return { + displayName: overrides.id, + badgeColor: '#000000', + addedAt: 1, + kind: 'git', + ...overrides + } +} + +function makeGroup(overrides: Partial & Pick): ProjectGroup { + return { + name: overrides.id, + parentPath: null, + parentGroupId: null, + createdFrom: 'folder-scan', + tabOrder: 0, + isCollapsed: false, + color: null, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +function makeWorkspace( + overrides: Partial & Pick +): FolderWorkspace { + return { + projectGroupId: 'group-root', + name: overrides.id, + comment: '', + linkedTask: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 1, + lastActivityAt: 1, + createdAt: 1, + updatedAt: 1, + ...overrides + } +} + +/** Repos, nested groups, folder workspaces (one not a git worktree), and an SSH repo. */ +function makeMixedFixture(): StoreFixture { + const repos = [ + makeRepo({ id: 'repo-local', path: '/repos/app', projectGroupId: 'group-root' }), + makeRepo({ id: 'repo-nested', path: '/repos/nested', projectGroupId: 'group-child' }), + makeRepo({ id: 'repo-folder', path: '/folders/plain', kind: 'folder' }), + makeRepo({ + id: 'repo-ssh', + path: '/remote/app', + connectionId: 'ssh-1', + projectGroupId: 'group-remote' + }) + ] + const projectGroups = [ + makeGroup({ id: 'group-root', parentPath: '/folders/root' }), + makeGroup({ id: 'group-child', parentGroupId: 'group-root', parentPath: '/folders/child' }), + makeGroup({ id: 'group-grandchild', parentGroupId: 'group-child' }), + makeGroup({ id: 'group-remote', parentPath: '/remote/scope' }), + makeGroup({ id: 'group-connection', parentPath: '/remote/via-group', connectionId: 'ssh-1' }) + ] + const folderWorkspaces = [ + makeWorkspace({ id: 'ws-git', folderPath: '/folders/root/feature' }), + // Not a git worktree: a plain folder workspace under a folder-kind repo. + makeWorkspace({ + id: 'ws-plain', + folderPath: '/folders/plain/scratch', + projectGroupId: 'group-child' + }), + makeWorkspace({ id: 'ws-remote', folderPath: '/remote/ws', projectGroupId: 'group-remote' }), + makeWorkspace({ + id: 'ws-connection', + folderPath: '/remote/direct', + projectGroupId: 'group-connection' + }), + makeWorkspace({ + id: 'ws-unlinked', + folderPath: '/folders/unlinked', + projectGroupId: 'group-orphan' + }) + ] + const projects: Project[] = [ + { + id: 'project-1', + displayName: 'App', + badgeColor: '#000000', + sourceRepoIds: ['repo-local', 'repo-nested'], + createdAt: 1, + updatedAt: 1 + }, + { + id: 'project-2', + displayName: 'Folder', + badgeColor: '#000000', + sourceRepoIds: ['repo-folder'], + createdAt: 1, + updatedAt: 1 + } + ] + return { repos, projects, projectGroups, folderWorkspaces, workspaceDir: '/workspaces' } +} + +beforeEach(() => { + invalidateAuthorizedRootsCache() + vi.mocked(buildProjectGroupChildIndex).mockClear() + vi.mocked(getProjectGroupSubtreeIds).mockClear() +}) + +describe('getAllowedRoots', () => { + it('produces the same roots as the pre-change implementation', () => { + const { store } = makeCountingStore(makeMixedFixture()) + + expect(getAllowedRoots(store)).toEqual(referenceAllowedRoots(store)) + }) + + it('reads the store once and indexes project groups once per build', () => { + const fixture = makeMixedFixture() + const { store, counts } = makeCountingStore(fixture) + + getAllowedRoots(store) + + expect.soft(counts.getRepos).toBe(1) + expect.soft(counts.getProjectGroups).toBe(1) + expect.soft(counts.getFolderWorkspaces).toBe(1) + // Batched runtime resolution scans the project list once, not once per local repo. + expect.soft(counts.getProjects).toBe(1) + // The per-scope subtree walk no longer rebuilds the parent->children index. + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(1) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) +}) + +describe('resolveAuthorizedPath allowed-root reuse', () => { + let repoRoot: string + let outsideRoot: string + let store: Store + let counts: StoreCallCounts + + beforeEach(async () => { + repoRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-allowed-roots-')) + outsideRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-outside-')) + const fixture = makeMixedFixture() + fixture.repos = [makeRepo({ id: 'repo-local', path: repoRoot }), ...fixture.repos] + fixture.projects[0]!.sourceRepoIds = ['repo-local'] + ;({ store, counts } = makeCountingStore(fixture)) + }) + + afterEach(async () => { + await rm(repoRoot, { recursive: true, force: true }) + await rm(outsideRoot, { recursive: true, force: true }) + }) + + it('builds the allowed-root list once per call across repeated reads', async () => { + const dirPath = join(repoRoot, 'src') + await mkdir(dirPath) + await writeFile(join(dirPath, 'index.ts'), 'export {}\n') + const callCount = 5 + + for (let index = 0; index < callCount; index += 1) { + await resolveAuthorizedPath(dirPath, store) + await resolveAuthorizedPath(join(dirPath, 'index.ts'), store) + } + + const buildCount = callCount * 2 + // One build per authorization, not one per raw-path check plus one per realpath check. + expect.soft(counts.getFolderWorkspaces).toBe(buildCount) + expect.soft(counts.getRepos).toBe(buildCount) + expect.soft(counts.getProjects).toBe(buildCount) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(buildCount) + expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled() + }) + + // Why (both symlink cases): creating a symlink on Windows needs elevation or + // Developer Mode, so these would fail EPERM in setup rather than exercise the + // escape check. Every non-symlink case still runs there. + it.skipIf(process.platform === 'win32')( + 'still refuses a symlink that escapes every allowed root', + async () => { + const secret = join(outsideRoot, 'secret.txt') + await writeFile(secret, 'secret\n') + const escape = join(repoRoot, 'escape.txt') + await symlink(secret, escape) + + await expect(resolveAuthorizedPath(escape, store)).rejects.toThrow('Access denied') + expect(vi.mocked(listRepoWorktreeGraph)).toHaveBeenCalled() + } + ) + + it('builds no allowed-root list at all for a granted external path', async () => { + const external = join(outsideRoot, 'external.md') + await writeFile(external, 'notes\n') + authorizeExternalPath(external) + counts.getRepos = 0 + counts.getProjects = 0 + counts.getFolderWorkspaces = 0 + + for (let index = 0; index < 5; index += 1) { + await expect(resolveAuthorizedPath(external, store)).resolves.toBe(external) + } + + // The grant answers on its own; hoisting the snapshot must not turn zero builds into one per read. + expect.soft(counts.getRepos).toBe(0) + expect.soft(counts.getProjects).toBe(0) + expect.soft(counts.getFolderWorkspaces).toBe(0) + expect.soft(vi.mocked(buildProjectGroupChildIndex)).not.toHaveBeenCalled() + }) + + it.skipIf(process.platform === 'win32')( + 'still refuses a directory symlink that escapes every allowed root', + async () => { + const outsideDir = join(outsideRoot, 'nested') + await mkdir(outsideDir) + await writeFile(join(outsideDir, 'file.txt'), 'secret\n') + const escape = join(repoRoot, 'escape-dir') + await symlink(outsideDir, escape) + + await expect(resolveAuthorizedPath(join(escape, 'file.txt'), store)).rejects.toThrow( + 'Access denied' + ) + } + ) +}) diff --git a/src/main/ipc/filesystem-allowed-roots.ts b/src/main/ipc/filesystem-allowed-roots.ts index 3cb7fe4fa55..cef249430c6 100644 --- a/src/main/ipc/filesystem-allowed-roots.ts +++ b/src/main/ipc/filesystem-allowed-roots.ts @@ -1,9 +1,16 @@ import { resolve } from 'node:path' import type { Store } from '../persistence' import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic' -import { getWorktreeMirrorDistro } from '../project-runtime-git-options' +import { + getWorktreeMirrorDistroForRuntime, + resolveLocalProjectRuntimesForRepos +} from '../project-runtime-git-options' import { isPathInsideOrEqual } from '../../shared/cross-platform-path' -import { getProjectGroupSubtreeIds } from '../../shared/project-groups' +import { + buildProjectGroupChildIndex, + collectProjectGroupSubtreeIds, + type ProjectGroupChildIndex +} from '../../shared/project-groups' import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { ProjectGroup } from '../../shared/project-group-types' import type { Repo } from '../../shared/repo-types' @@ -11,18 +18,22 @@ import type { Repo } from '../../shared/repo-types' type FolderScopeStore = Pick & Partial> +// Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. +function filterLocalRepos(repos: readonly Repo[]): Repo[] { + return repos.filter((repo) => !repo.connectionId) +} + export function getLocalRepos(store: Store) { - // Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths. - return store.getRepos().filter((repo) => !repo.connectionId) + return filterLocalRepos(store.getRepos()) } function getFolderScopeCandidateRepos( folderPath: string, projectGroupId: string, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): Repo[] { - const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId) + const groupIds = collectProjectGroupSubtreeIds(childGroupIndex, projectGroupId) return repos.filter( (repo) => (typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) || @@ -34,13 +45,18 @@ function isRemoteOnlyFolderScope( folderPath: string, projectGroupId: string, connectionId: string | null | undefined, - projectGroups: readonly ProjectGroup[], + childGroupIndex: ProjectGroupChildIndex, repos: readonly Repo[] ): boolean { if (connectionId) { return true } - const candidates = getFolderScopeCandidateRepos(folderPath, projectGroupId, projectGroups, repos) + const candidates = getFolderScopeCandidateRepos( + folderPath, + projectGroupId, + childGroupIndex, + repos + ) return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId)) } @@ -55,16 +71,22 @@ function getFolderWorkspaceConnectionId( ) } -function getLocalFolderScopeRoots(store: Store): string[] { +function getLocalFolderScopeRoots(store: Store, repos: readonly Repo[]): string[] { const scopeStore = store as FolderScopeStore - const repos = scopeStore.getRepos() // Why: many filesystem tests use narrow Store doubles; folder scopes are additive. const projectGroups = scopeStore.getProjectGroups?.() ?? [] + const childGroupIndex = buildProjectGroupChildIndex(projectGroups) const roots: string[] = [] for (const group of projectGroups) { if ( group.parentPath && - !isRemoteOnlyFolderScope(group.parentPath, group.id, group.connectionId, projectGroups, repos) + !isRemoteOnlyFolderScope( + group.parentPath, + group.id, + group.connectionId, + childGroupIndex, + repos + ) ) { roots.push(resolve(group.parentPath)) } @@ -75,7 +97,7 @@ function getLocalFolderScopeRoots(store: Store): string[] { workspace.folderPath, workspace.projectGroupId, getFolderWorkspaceConnectionId(workspace, projectGroups), - projectGroups, + childGroupIndex, repos ) ) { @@ -86,16 +108,19 @@ function getLocalFolderScopeRoots(store: Store): string[] { } export function getAllowedRoots(store: Store): string[] { - const localRepos = getLocalRepos(store) + // Why one read: `getRepos` rehydrates every repo, and this runs twice per filesystem IPC. + const repos = store.getRepos() + const localRepos = filterLocalRepos(repos) const settings = store.getSettings() const roots = [ ...localRepos.map((repo) => resolve(repo.path)), - ...getLocalFolderScopeRoots(store) + ...getLocalFolderScopeRoots(store, repos) ] if (settings.workspaceDir) { if (localRepos.length === 0) { roots.push(resolve(settings.workspaceDir)) } else { + const projectRuntimeByRepoId = resolveLocalProjectRuntimesForRepos(store, localRepos) for (const repo of localRepos) { roots.push( resolve( @@ -104,7 +129,11 @@ export function getAllowedRoots(store: Store): string[] { // Why enriched here too: placement has to agree with the create // flow, or renderer file access is denied for a worktree Orca // just put on the WSL side. - getWorktreePathSettings(repo, settings, getWorktreeMirrorDistro(store, repo)) + getWorktreePathSettings( + repo, + settings, + getWorktreeMirrorDistroForRuntime(projectRuntimeByRepoId.get(repo.id)) + ) ) ) ) diff --git a/src/main/ipc/filesystem-auth.ts b/src/main/ipc/filesystem-auth.ts index 122617845ed..894e39945c1 100644 --- a/src/main/ipc/filesystem-auth.ts +++ b/src/main/ipc/filesystem-auth.ts @@ -43,7 +43,24 @@ export function authorizeExternalPath(targetPath: string): void { } catch {} } -export function isPathAllowed(targetPath: string, store: Store): boolean { +/** + * One allowed-root list shared by every check in a single authorization. + * + * Lazy so a path already covered by an external grant still builds nothing at all, the way it did + * before the list was hoisted out of the individual checks. + */ +type AllowedRootsSnapshot = { get: () => readonly string[] } + +function createAllowedRootsSnapshot(store: Store): AllowedRootsSnapshot { + let roots: readonly string[] | undefined + return { get: () => (roots ??= getAllowedRoots(store)) } +} + +export function isPathAllowed( + targetPath: string, + store: Store, + allowedRoots?: AllowedRootsSnapshot +): boolean { const resolvedTarget = resolve(targetPath) if (authorizedExternalPaths.has(resolvedTarget)) { return true @@ -53,7 +70,9 @@ export function isPathAllowed(targetPath: string, store: Store): boolean { return true } } - return getAllowedRoots(store).some((root) => isDescendantOrEqual(resolvedTarget, root)) + return (allowedRoots?.get() ?? getAllowedRoots(store)).some((root) => + isDescendantOrEqual(resolvedTarget, root) + ) } export type ResolveAuthorizedPathOptions = { @@ -69,7 +88,10 @@ export async function resolveAuthorizedPath( options: ResolveAuthorizedPathOptions = {} ): Promise { const resolvedTarget = resolve(targetPath) - if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store))) { + // Why: the roots depend only on store state, not on the candidate path, so one snapshot serves + // every authorization below; each candidate is still checked against it in full. + const allowedRoots = createAllowedRootsSnapshot(store) + if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store, { allowedRoots }))) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) } @@ -80,14 +102,15 @@ export async function resolveAuthorizedPath( realParent = await realpath(dirname(resolvedTarget)) } catch (error) { if (isENOENT(error)) { - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } throw error } const candidateTarget = resolve(realParent, basename(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -100,7 +123,8 @@ export async function resolveAuthorizedPath( const realTarget = resolve(await realpath(resolvedTarget)) if ( !(await isPathAllowedIncludingRegisteredWorktrees(realTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -110,11 +134,15 @@ export async function resolveAuthorizedPath( if (!isENOENT(error)) { throw error } - return resolveAuthorizedMissingPath(resolvedTarget, store) + return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots) } } -async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store): Promise { +async function resolveAuthorizedMissingPath( + resolvedTarget: string, + store: Store, + allowedRoots: AllowedRootsSnapshot +): Promise { let existingAncestor = resolvedTarget const missingSegments: string[] = [] @@ -124,7 +152,8 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store const candidateTarget = resolve(realAncestor, ...missingSegments) if ( !(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, { - canonicalSourcePath: resolvedTarget + canonicalSourcePath: resolvedTarget, + allowedRoots })) ) { throw new Error(PATH_ACCESS_DENIED_MESSAGE) @@ -148,9 +177,9 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store async function isPathAllowedIncludingRegisteredWorktrees( targetPath: string, store: Store, - options: { canonicalSourcePath?: string } = {} + options: { canonicalSourcePath?: string; allowedRoots?: AllowedRootsSnapshot } = {} ): Promise { - if (isPathAllowed(targetPath, store)) { + if (isPathAllowed(targetPath, store, options.allowedRoots)) { return true } @@ -158,7 +187,14 @@ async function isPathAllowedIncludingRegisteredWorktrees( return true } - if (await isPathAllowedByCanonicalAllowedRoot(targetPath, options.canonicalSourcePath, store)) { + if ( + await isPathAllowedByCanonicalAllowedRoot( + targetPath, + options.canonicalSourcePath, + store, + options.allowedRoots + ) + ) { return true } @@ -178,12 +214,13 @@ async function isPathAllowedIncludingRegisteredWorktrees( async function isPathAllowedByCanonicalAllowedRoot( targetPath: string, sourcePath: string | undefined, - store: Store + store: Store, + allowedRoots?: AllowedRootsSnapshot ): Promise { if (!sourcePath) { return false } - for (const root of getAllowedRoots(store)) { + for (const root of allowedRoots?.get() ?? getAllowedRoots(store)) { const resolvedRoot = resolve(root) if (!isDescendantOrEqual(sourcePath, resolvedRoot)) { continue diff --git a/src/main/project-runtime-git-options.ts b/src/main/project-runtime-git-options.ts index 808d31d5fcf..20aa0e9659a 100644 --- a/src/main/project-runtime-git-options.ts +++ b/src/main/project-runtime-git-options.ts @@ -102,7 +102,12 @@ export function getWorktreeMirrorDistro( store: ProjectRuntimeResolutionStore, repo: Repo ): string | undefined { - const projectRuntime = resolveLocalProjectRuntimeForRepo(store, repo) + return getWorktreeMirrorDistroForRuntime(resolveLocalProjectRuntimeForRepo(store, repo)) +} + +export function getWorktreeMirrorDistroForRuntime( + projectRuntime: ProjectExecutionRuntimeResolution | undefined +): string | undefined { if (!projectRuntime || projectRuntime.status !== 'resolved') { return undefined } diff --git a/src/shared/project-groups.ts b/src/shared/project-groups.ts index 67c2897ead6..c4fe8badb47 100644 --- a/src/shared/project-groups.ts +++ b/src/shared/project-groups.ts @@ -109,10 +109,12 @@ export function clearMissingProjectGroupMemberships(repos: Repo[], groups: Proje ) } -export function getProjectGroupSubtreeIds( - groups: readonly Pick[], - rootGroupId: string -): Set { +export type ProjectGroupChildIndex = ReadonlyMap + +/** Build once and reuse when collecting subtrees for more than one root. */ +export function buildProjectGroupChildIndex( + groups: readonly Pick[] +): ProjectGroupChildIndex { const childGroupsByParentId = new Map() for (const group of groups) { if (!group.parentGroupId) { @@ -122,7 +124,20 @@ export function getProjectGroupSubtreeIds( children.push(group.id) childGroupsByParentId.set(group.parentGroupId, children) } + return childGroupsByParentId +} +export function getProjectGroupSubtreeIds( + groups: readonly Pick[], + rootGroupId: string +): Set { + return collectProjectGroupSubtreeIds(buildProjectGroupChildIndex(groups), rootGroupId) +} + +export function collectProjectGroupSubtreeIds( + childGroupsByParentId: ProjectGroupChildIndex, + rootGroupId: string +): Set { const subtreeIds = new Set() const pending = [rootGroupId] while (pending.length > 0) { From 79d5fb469a31f0fce51fe360bb894ac2f9d7b121 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 4 Sep 2026 01:23:54 -0400 Subject: [PATCH 04/12] fix(cloud): recalibrate the relay monitor's postgres-retry freeze to a measured bar (#18580) The global relay_cells FOR UPDATE lock made successful retries a steady-state rate: fleet-wide p50 430 / p90 924 / p99 1320 / max 1504 per five minutes over the last 24 h, 55% of windows over the 300 bar, only 22% of 15-minute gates clean. Three read-only dry-runs on 2026-09-04 froze on it, blocking the same-cap roll that carries #18521 and the beginProof crash guard to the 23 cells. 2000 clears every measured healthy gate; the exhausted-retry, director concurrency, and pool bars keep the incident discriminator role. --- .../relay-ops/src/incident-monitor.test.ts | 17 +++++++++------ cloud/apps/relay-ops/src/incident-monitor.ts | 21 +++++++++++++------ cloud/docs/relay-incident-monitor.md | 16 +++++++++++++- 3 files changed, 41 insertions(+), 13 deletions(-) diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 61a73b64dbe..4e1da9fab26 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -111,14 +111,19 @@ describe('incident monitor evaluator', () => { }) }) - it('freezes when postgres retries exceed the recalibrated ceiling', () => { - const sample = healthySample() - sample.sources['relay-logs']!.signals['relay.postgres_retries'] = - signal(INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries + 1) - expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + // Why: the global relay_cells lock made retries a steady-state rate (24 h p99 + // 1320/5min on 2026-09-04); the bar fences only unbounded growth beyond that. + it('tolerates the measured healthy retry rate and freezes above the bar', () => { + const healthy = healthySample() + healthy.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(1504) + expect(evaluateIncidentSample(healthy, startedAt).status).toBe('green') + + const incident = healthySample() + incident.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(2001) + expect(evaluateIncidentSample(incident, startedAt)).toMatchObject({ status: 'freeze', failures: [ - expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 300 }) + expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 2000 }) ] }) }) diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index 868bb86fb93..a121568d918 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -32,11 +32,20 @@ export const INCIDENT_MONITOR_THRESHOLDS = { relayPoolWaiting: 800, relayPoolWaitMs: 2_500, // Why: successful lock retries are the contention machinery working, not harm. - // Healthy 2026-08-26 baseline bursts to 234/5min (26% of windows crossed the old - // bar of 20, set unmeasured at the monitor's 2026-07-28 birth); the 2026-08-23 - // incident ran ~2,200-3,000/5min. 300 clears healthy bursts with ~10x incident - // margin; relayPostgresRetryExhausted below bounds the terminally failed share. - relayPostgresRetries: 300, + // Recalibrated 2026-09-04 from 300, which was set 2026-08-26 when healthy bursts + // reached 234/5min. The global relay_cells FOR UPDATE lock has since become the + // fleet's steady state: measured fleet-wide (director + cells, summed per five + // minutes) 2026-09-03T05Z..2026-09-04T05Z p50 430 / p90 924 / p99 1320 / max + // 1504, with 55% of windows over 300 and only 22% of 15-minute gates clean, so + // the bar blocked the very cell roll that carries the 500 ms lock wait (#18521) + // and the beginProof crash guard to the cells. The 2026-08-23 lock incident on + // this same metric peaked at 1510 in one window and 646 in the next, so it is + // not separable from today's contention by retries alone; it is caught by + // relayPostgresRetryExhausted (467 at the peak vs a 300 bar), director + // concurrency, and the pool bars. 2000 passes every healthy 15-minute window + // measured in the last 24 h and still fences unbounded growth. Re-tighten once + // the fleet is on the 500 ms lock wait and the baseline is re-measured. + relayPostgresRetries: 2000, // Why: 300 per five minutes, recalibrated 2026-09-04 from a bar of zero that no // production window has cleared since #18521 shipped to the director. That // change cut the request-path cell-inventory wait from the 1 s pool lock_timeout @@ -48,7 +57,7 @@ export const INCIDENT_MONITOR_THRESHOLDS = { // quiet hours p50 2 / max 36; pre-#18521 daytime p50 10 / p90 25 / max 87; // post-#18521 p50 42 / p90 147 / max 220. The 2026-08-23 lock incident peaked // at 467. 300 clears every measured healthy window and still sits below the - // incident shape; relayPostgresRetries above stays the ~10x discriminator. + // incident shape; retries above fence only unbounded growth. // User-facing /v1/assign 503 share did not move with #18521 (13.9% old image // vs 12.3% new, same evening), so exhaustion is not a proxy for user harm. relayPostgresRetryExhausted: 300, diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 337d3f1b20f..5c563f6a2e2 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -99,7 +99,7 @@ durably marked consumed before mutation and cannot authorize another run. | Cloud SQL deadlocks | over 0 | | Relay pool waiters | over 800 | | Relay pool wait | over 2,500 ms | -| PostgreSQL retries in five minutes | over 300 | +| PostgreSQL retries in five minutes | over 2,000 | | Exhausted PostgreSQL retries in five minutes | over 300 | | Director instances | outside 5–6 | | Director CPU or memory | over 80% | @@ -139,6 +139,20 @@ heartbeats, and matching live admission. logs: healthy-day bursts reach 234/5min with zero exhausted retries and 26% of five-minute windows over 20, while the 2026-08-23 lock-contention incident ran roughly 2,200–3,000/5min. +- Recalibrated the PostgreSQL-retry freeze from 300 to 2,000 per five minutes + (2026-09-04). Basis: the global `relay_cells FOR UPDATE` lock made + successful retries a steady-state rate. Measured fleet-wide (director + + cells, summed per five minutes from the `orca_relay_postgres_retries` + log metric) over 2026-09-03T05Z..2026-09-04T05Z: p50 430 / p90 924 / + p99 1,320 / max 1,504; 55% of windows over 300; only 22% of 15-minute gates + clean at 300 versus 100% at 2,000. Three read-only dry-runs on 2026-09-04 + froze on this bar (runs 33836470590, 33838698725) or on a genuine six-cell + crash storm (33837160275), blocking the same-cap roll that carries #18521 + and the `beginProof` crash guard to the 23 cells. The 2026-08-23 incident + on this metric peaked at 1,510 then 646, so retries alone no longer + separate it from today's baseline; the exhausted-retry bar (incident peak + 467 vs bar 300), director concurrency, and the pool bars carry that role. + Re-tighten after the fleet is on the 500 ms lock wait. - Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters From b378101901d8062765ec81faa88addf6ad037d94 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Fri, 4 Sep 2026 01:40:55 -0400 Subject: [PATCH 05/12] docs(cloud): reconcile the 2026-08-23 retry figure with the gate metric (#18581) --- cloud/docs/relay-incident-monitor.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 5c563f6a2e2..870c95dd413 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -138,7 +138,9 @@ heartbeats, and matching live admission. `jsonPayload.event="orca_relay_postgres_transaction_retry"` in production logs: healthy-day bursts reach 234/5min with zero exhausted retries and 26% of five-minute windows over 20, while the 2026-08-23 lock-contention - incident ran roughly 2,200–3,000/5min. + incident ran roughly 2,200–3,000/5min by raw log-line count (the gate's + own `orca_relay_postgres_retries` metric read 1,510 for that window; see the + 2026-09-04 entry). - Recalibrated the PostgreSQL-retry freeze from 300 to 2,000 per five minutes (2026-09-04). Basis: the global `relay_cells FOR UPDATE` lock made successful retries a steady-state rate. Measured fleet-wide (director + From 561a94038c3ab550be22bbf680f15e06b6afd7dc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 00:06:09 -0700 Subject: [PATCH 06/12] fix(ssh): stop the daemon's own services from blocking the superseded-relay reap (#18586) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `isReapableRelayHusk` required `childCount === 0`, where `childCount` came from `pgrep -P | grep -c .`. But the relay forks service children of its own, and `relay-ai-vault-service.js` never exits once spawned. Any relay that had served a single AI Vault request therefore reported a non-zero child count forever, so the sweep answered `retained-live-work` for a superseded, disconnected relay holding no user work at all — and its version directory stayed pinned against GC by its own live socket. The probe now censuses each direct child instead of counting them, and the reap gate reads the count of children it could *not* positively identify as relay infrastructure. The asymmetry is the safety argument (docs/reference/ssh-execution-boundary.md): subtracting a child we can name is positive knowledge, assuming about one we cannot is not. An unrecognised argv, an argv `ps` would not print, and a host without `pgrep` all keep the relay unreapable. `reapEmptyRelayHuskCommand` re-runs the same census on the host immediately before signalling. Fixes #13614 --- src/main/ssh/relay-daemon-service-children.ts | 62 +++++++++ ...dpoint-incumbent-shell.integration.test.ts | 118 ++++++++++++++++-- .../ssh/ssh-relay-endpoint-incumbent.test.ts | 64 +++++++--- src/main/ssh/ssh-relay-endpoint-incumbent.ts | 47 +++++-- .../ssh/ssh-relay-endpoint-takeover.test.ts | 29 +++-- src/main/ssh/ssh-relay-endpoint-takeover.ts | 15 ++- .../ssh-relay-superseded-endpoints.test.ts | 10 +- src/shared/relay-artifacts.ts | 18 ++- 8 files changed, 307 insertions(+), 56 deletions(-) create mode 100644 src/main/ssh/relay-daemon-service-children.ts diff --git a/src/main/ssh/relay-daemon-service-children.ts b/src/main/ssh/relay-daemon-service-children.ts new file mode 100644 index 00000000000..35e755ea518 --- /dev/null +++ b/src/main/ssh/relay-daemon-service-children.ts @@ -0,0 +1,62 @@ +/** + * Telling a relay daemon's own service processes apart from the work it holds. + * + * The reap gate used to ask `pgrep -P | grep -c .` and demand zero. But the daemon + * forks service children of its own — `relay-ai-vault-service.js` is spawned lazily and then + * never exits — so that count is permanently non-zero on any relay that has touched the AI + * Vault, whether or not it holds a single PTY. A superseded, disconnected relay holding + * nothing therefore reported `retained-live-work` forever, its version directory stayed + * pinned against GC by its own live socket, and the population grew without bound (#13614). + * + * The asymmetry below is the whole safety argument, and it follows + * docs/reference/ssh-execution-boundary.md: *subtracting a child we can positively identify + * as relay infrastructure is sound; assuming anything about a child we cannot identify is + * not.* An argv that does not match, an argv `ps` would not print, and a host without + * `pgrep` all count against the relay and keep it unreapable. Losing sight of a child is + * never evidence that it holds nothing. + */ +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' +import { shellEscape } from './ssh-connection-utils' + +/** Shell variable set to the daemon's direct-child count, or `unknown`. */ +export const RELAY_CHILD_COUNT_VAR = 'kids' + +/** Shell variable set to the count of children not identified as relay services, or `unknown`. */ +export const RELAY_UNRECOGNIZED_CHILD_COUNT_VAR = 'unrecognized_kids' + +/** + * `case` patterns matching a service child's argv. Suffix-anchored on purpose: both entries + * are forked with no script arguments, so the argv ends at the filename, and the leading `/` + * requires the absolute path the daemon forks rather than a bare mention of the name. A + * future arg would stop matching and the relay would go back to being retained — the safe + * direction to fail in. + */ +function serviceChildArgvPatterns(): string { + return RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.map( + (filename) => `*${shellEscape(`/${filename}`)}` + ).join('|') +} + +/** + * POSIX shell that censuses the direct children of `$pid`, setting `kids` and + * `unrecognized_kids`. Both stay `unknown` when the host cannot enumerate children at all. + */ +export function relayDaemonChildCensusShell(): string[] { + return [ + `${RELAY_CHILD_COUNT_VAR}=unknown`, + `${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=unknown`, + 'if command -v pgrep >/dev/null 2>&1; then', + ` ${RELAY_CHILD_COUNT_VAR}=0`, + ` ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=0`, + ' for kid in $(pgrep -P "$pid" 2>/dev/null); do', + ` ${RELAY_CHILD_COUNT_VAR}=$((${RELAY_CHILD_COUNT_VAR}+1))`, + ' kid_args=$(ps -o args= -p "$kid" 2>/dev/null | tr -d "\\n")', + ' case "$kid_args" in', + ` ${serviceChildArgvPatterns()}) ;;`, + // An unreadable or unrecognised argv lands here, which is what keeps the relay retained. + ` *) ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=$((${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}+1)) ;;`, + ' esac', + ' done', + 'fi' + ] +} diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts index 7ece0b6532e..a8975d0520b 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent-shell.integration.test.ts @@ -4,7 +4,7 @@ * generated scripts through /bin/sh against real unix sockets and real processes. */ import { execFile, spawn, type ChildProcess } from 'node:child_process' -import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterAll, afterEach, beforeAll, describe, expect, it } from 'vitest' @@ -15,21 +15,32 @@ import { type RelayEndpointIncumbent } from './ssh-relay-endpoint-incumbent' import { reapEmptyRelayHuskCommand } from './ssh-relay-endpoint-takeover' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' const posixOnly = process.platform === 'win32' ? describe.skip : describe const FAKE_RELAY_SOURCE = ` const net = require('net') +const path = require('path') const sock = process.argv[process.argv.indexOf('--sock-path') + 1] +function spawnChild(args) { + require('child_process').spawn(process.execPath, args, { stdio: 'ignore' }) +} if (process.argv.includes('--with-child')) { - require('child_process').spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000)'], { - stdio: 'ignore' - }) + spawnChild(['-e', 'setTimeout(() => {}, 60000)']) +} +// Why forked the same way production does: the exclusion is argv-shaped, so a hand-written +// stand-in would test the test rather than the shell that runs on someone's host. +for (const name of process.argv.filter((arg) => arg.startsWith('--service-child='))) { + spawnChild([path.join(__dirname, name.slice('--service-child='.length))]) } net.createServer(() => {}).listen(sock, () => process.stdout.write('READY\\n')) process.on('SIGTERM', () => process.exit(0)) ` +// Self-limiting: these are orphaned when the relay under test is reaped. +const IDLE_SERVICE_SOURCE = 'setTimeout(() => {}, 60000)\n' + function sh(script: string): Promise { return new Promise((resolve, reject) => { execFile('/bin/sh', ['-c', script], { timeout: 20_000 }, (error, stdout) => { @@ -43,14 +54,21 @@ function sh(script: string): Promise { } let workDir: string +let pgreplessBinDir: string let hasLsof = false const running: ChildProcess[] = [] -function startFakeRelay(sockPath: string, withChild = false): Promise { +function startFakeRelay( + sockPath: string, + options: { withChild?: boolean; serviceChildren?: readonly string[] } = {} +): Promise { const args = [join(workDir, 'relay.js'), '--sock-path', sockPath] - if (withChild) { + if (options.withChild) { args.push('--with-child') } + for (const name of options.serviceChildren ?? []) { + args.push(`--service-child=${name}`) + } const child = spawn(process.execPath, args, { stdio: ['ignore', 'pipe', 'ignore'] }) running.push(child) return new Promise((resolve, reject) => { @@ -68,9 +86,31 @@ async function probe(sockPath: string): Promise { return parseRelayEndpointIncumbentProbe(sockPath, output) } +/** The relay forks its children after it starts listening, so the probe can race them. */ +async function waitForChildCount( + sockPath: string, + expected: number +): Promise { + let incumbent = await probe(sockPath) + for (let attempt = 0; attempt < 50 && incumbent.holders[0]?.childCount !== expected; attempt++) { + await new Promise((resolve) => setTimeout(resolve, 100)) + incumbent = await probe(sockPath) + } + return incumbent +} + beforeAll(async () => { workDir = mkdtempSync(join(tmpdir(), 'orca-relay-incumbent-')) writeFileSync(join(workDir, 'relay.js'), FAKE_RELAY_SOURCE) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + writeFileSync(join(workDir, filename), IDLE_SERVICE_SOURCE) + } + writeFileSync(join(workDir, 'looks-like-relay-watcher.js'), IDLE_SERVICE_SOURCE) + pgreplessBinDir = join(workDir, 'pgrepless-bin') + mkdirSync(pgreplessBinDir) + for (const tool of ['ps', 'tr']) { + symlinkSync((await sh(`command -v ${tool}`)).trim(), join(pgreplessBinDir, tool)) + } hasLsof = await sh('command -v lsof >/dev/null 2>&1 && echo yes || echo no').then( (out) => out.trim() === 'yes' ) @@ -105,13 +145,51 @@ posixOnly('relay endpoint probe against a real socket', () => { return } expect(incumbent.holders.map((holder) => holder.pid)).toEqual([relay.pid]) - expect(incumbent.holders[0]).toMatchObject({ matchesRelayArgv: true, childCount: 0 }) + expect(incumbent.holders[0]).toMatchObject({ + matchesRelayArgv: true, + childCount: 0, + unrecognizedChildCount: 0 + }) expect(isReapableRelayHusk(incumbent)).toBe(true) }) + it("counts the daemon's own service children but does not hold them against it", async () => { + const sockPath = join(workDir, 'services.sock') + await startFakeRelay(sockPath, { serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES }) + const incumbent = await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + + expect(incumbent.holders[0].childCount).toBe(RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + expect(incumbent.holders[0].unrecognizedChildCount).toBe(0) + expect(isReapableRelayHusk(incumbent)).toBe(true) + }) + + it('still retains a relay holding work alongside its service children', async () => { + const sockPath = join(workDir, 'services-and-work.sock') + await startFakeRelay(sockPath, { + withChild: true, + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + const incumbent = await waitForChildCount( + sockPath, + RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length + 1 + ) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + + it('does not excuse a child that merely mentions a service entry name', async () => { + const sockPath = join(workDir, 'lookalike.sock') + await startFakeRelay(sockPath, { serviceChildren: ['looks-like-relay-watcher.js'] }) + const incumbent = await waitForChildCount(sockPath, 1) + + expect(incumbent.holders[0].unrecognizedChildCount).toBe(1) + expect(isReapableRelayHusk(incumbent)).toBe(false) + }) + it('refuses to call a relay with a live child an empty husk', async () => { const sockPath = join(workDir, 'busy.sock') - await startFakeRelay(sockPath, true) + await startFakeRelay(sockPath, { withChild: true }) const incumbent = await probe(sockPath) expect(incumbent.verdict).toBe('live') @@ -150,12 +228,34 @@ posixOnly('empty relay husk reap against a real process', () => { it('refuses to signal a relay that acquired a child after it was probed', async () => { const sockPath = join(workDir, 'raced.sock') - const relay = await startFakeRelay(sockPath, true) + const relay = await startFakeRelay(sockPath, { withChild: true }) const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) expect(output.trim()).toBe('BUSY') expect(relay.killed).toBe(false) }) + it('terminates a relay whose only children are its own service processes (#13614)', async () => { + const sockPath = join(workDir, 'service-husk.sock') + const relay = await startFakeRelay(sockPath, { + serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES + }) + await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length) + const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath)) + expect(output.trim()).toBe('GONE') + }) + + it('refuses to signal when the host cannot enumerate children at all', async () => { + const sockPath = join(workDir, 'no-pgrep.sock') + const relay = await startFakeRelay(sockPath) + // A PATH carrying every tool the script needs except `pgrep`: the census answers + // `unknown`, which must reach BUSY rather than the zero a missing tool would imply. + const output = await sh( + `PATH=${pgreplessBinDir}\n${reapEmptyRelayHuskCommand(relay.pid!, sockPath)}` + ) + expect(output.trim()).toBe('BUSY') + expect(relay.killed).toBe(false) + }) + it('refuses to signal a pid whose argv is not this relay at this socket', async () => { const sockPath = join(workDir, 'mismatch.sock') await startFakeRelay(sockPath) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts index a65cb33fc57..de4cc28d170 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.test.ts @@ -32,17 +32,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('reports live when the socket accepted a connection', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=4242 yes 13']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=4242 yes 13 11' + ]) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('accepted-connection') - expect(incumbent.holders).toEqual([{ pid: 4242, matchesRelayArgv: true, childCount: 13 }]) + expect(incumbent.holders).toEqual([ + { pid: 4242, matchesRelayArgv: true, childCount: 13, unrecognizedChildCount: 11 } + ]) }) it('reports live when a process still holds an inode that refuses connections', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2']) + probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2 2']) ) expect(incumbent.verdict).toBe('live') expect(incumbent.evidence).toBe('holder-process') @@ -85,7 +92,12 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('drops holder lines that do not carry a usable pid', () => { const incumbent = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=- no unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=refused', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=- no unknown unknown' + ]) ) expect(incumbent.holders).toEqual([]) expect(incumbent.verdict).toBe('exited') @@ -94,9 +106,24 @@ describe('parseRelayEndpointIncumbentProbe', () => { it('keeps an unreadable child count as null rather than zero', () => { const [holder] = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes unknown']) + probeOutput([ + 'PRESENT=yes', + 'LISTEN=accepted', + 'HOLDERS_SOURCE=lsof', + 'HOLDER=7 yes unknown unknown' + ]) ).holders expect(holder.childCount).toBeNull() + expect(holder.unrecognizedChildCount).toBeNull() + }) + + it('keeps a holder line with no unrecognized-child field unreapable', () => { + const incumbent = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes 0']) + ) + expect(incumbent.holders[0].unrecognizedChildCount).toBeNull() + expect(isReapableRelayHusk(incumbent)).toBe(false) }) }) @@ -172,27 +199,36 @@ describe('mayLaunchOverRelayEndpoint', () => { describe('isReapableRelayHusk', () => { const husk = parseRelayEndpointIncumbentProbe( SOCK, - probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0']) + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0 0']) ) - it('accepts a single proven relay holder with zero children', () => { + it('accepts a single proven relay holder with no unaccounted-for children', () => { expect(isReapableRelayHusk(husk)).toBe(true) }) - it('refuses a relay that still holds children', () => { + it('accepts a relay whose only children are its own service processes (#13614)', () => { + const withServices = parseRelayEndpointIncumbentProbe( + SOCK, + probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 2 0']) + ) + expect(withServices.holders[0].childCount).toBe(2) + expect(isReapableRelayHusk(withServices)).toBe(true) + }) + + it('refuses a relay that still holds children it could not account for', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: 1 }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 3, unrecognizedChildCount: 1 }] }) ).toBe(false) }) - it('refuses a holder whose child count could not be read', () => { + it('refuses a holder whose unrecognized-child count could not be read', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: true, childCount: null }] + holders: [{ pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: null }] }) ).toBe(false) }) @@ -201,7 +237,7 @@ describe('isReapableRelayHusk', () => { expect( isReapableRelayHusk({ ...husk, - holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0 }] + holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0, unrecognizedChildCount: 0 }] }) ).toBe(false) }) @@ -211,8 +247,8 @@ describe('isReapableRelayHusk', () => { isReapableRelayHusk({ ...husk, holders: [ - { pid: 500, matchesRelayArgv: true, childCount: 0 }, - { pid: 501, matchesRelayArgv: true, childCount: 0 } + { pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 }, + { pid: 501, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 } ] }) ).toBe(false) diff --git a/src/main/ssh/ssh-relay-endpoint-incumbent.ts b/src/main/ssh/ssh-relay-endpoint-incumbent.ts index 2688267f4c7..628a9558793 100644 --- a/src/main/ssh/ssh-relay-endpoint-incumbent.ts +++ b/src/main/ssh/ssh-relay-endpoint-incumbent.ts @@ -20,6 +20,11 @@ */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_CHILD_COUNT_VAR, + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { isWindowsRemoteHost, type RemoteHostPlatform } from './ssh-remote-platform' @@ -38,6 +43,12 @@ export type RelayEndpointHolder = { matchesRelayArgv: boolean /** Direct children, or null when `pgrep` could not answer. Never guessed. */ childCount: number | null + /** + * Direct children *not* positively identified as the daemon's own service processes, or + * null when the host could not enumerate them. This — not `childCount` — is what says + * whether the relay holds anything; see relay-daemon-service-children.ts. + */ + unrecognizedChildCount: number | null } export type RelayEndpointIncumbent = { @@ -95,11 +106,9 @@ export function relayEndpointIncumbentProbeCommand(nodePath: string, sockPath: s ' args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', ' match=no', ' case "$args" in *relay.js*"$sock"*) match=yes ;; esac', - ' kids=unknown', - ' if command -v pgrep >/dev/null 2>&1; then', - ' kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - ' fi', - ' printf \'HOLDER=%s %s %s\\n\' "$pid" "$match" "$kids"', + ...relayDaemonChildCensusShell().map((line) => ` ${line}`), + ' printf \'HOLDER=%s %s %s %s\\n\' "$pid" "$match" ' + + `"$${RELAY_CHILD_COUNT_VAR}" "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}"`, ' done', 'else', " printf 'HOLDERS_SOURCE=unavailable\\n'", @@ -159,19 +168,25 @@ export function parseRelayEndpointIncumbentProbe( } function parseHolder(value: string): RelayEndpointHolder | null { - const [rawPid, rawMatch, rawKids] = value.split(/\s+/) + const [rawPid, rawMatch, rawKids, rawUnrecognized] = value.split(/\s+/) const pid = Number.parseInt(rawPid ?? '', 10) if (!Number.isInteger(pid) || pid <= 0) { return null } - const childCount = Number.parseInt(rawKids ?? '', 10) return { pid, matchesRelayArgv: rawMatch === 'yes', - childCount: Number.isInteger(childCount) && childCount >= 0 ? childCount : null + childCount: parseChildCount(rawKids), + unrecognizedChildCount: parseChildCount(rawUnrecognized) } } +/** `unknown`, a missing field, and anything unparseable are all "could not tell" — never 0. */ +function parseChildCount(raw: string | undefined): number | null { + const count = Number.parseInt(raw ?? '', 10) + return Number.isInteger(count) && count >= 0 ? count : null +} + function unverifiableEndpoint(sockPath: string): RelayEndpointIncumbent { return { sockPath, @@ -239,8 +254,12 @@ export function mayLaunchOverRelayEndpoint(incumbent: RelayEndpointIncumbent): b /** * A live relay that provably holds nothing: identity confirmed against its argv, exactly one - * holder, and zero children. Reaping it destroys no user work. Anything less is retained — - * killing the wrong pid on someone's remote host is the worst outcome available here. + * holder, and no child the host could not account for as one of the daemon's own service + * processes. Reaping it destroys no user work. Anything less is retained — killing the wrong + * pid on someone's remote host is the worst outcome available here. + * + * Why not `childCount === 0`: the daemon's AI Vault sidecar never exits once spawned, so that + * gate was unreachable for any relay that had ever served a vault request (#13614). */ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean { if (incumbent.verdict !== 'live' || !incumbent.holdersEnumerable) { @@ -250,12 +269,16 @@ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean return false } const [holder] = incumbent.holders - return holder.matchesRelayArgv && holder.childCount === 0 + return holder.matchesRelayArgv && holder.unrecognizedChildCount === 0 } export function describeRelayEndpointIncumbent(incumbent: RelayEndpointIncumbent): string { const holders = incumbent.holders - .map((holder) => `${holder.pid}(children=${holder.childCount ?? 'unknown'})`) + .map( + (holder) => + `${holder.pid}(children=${holder.childCount ?? 'unknown'},` + + `unrecognized=${holder.unrecognizedChildCount ?? 'unknown'})` + ) .join(',') return ( `${incumbent.sockPath} verdict=${incumbent.verdict} evidence=${incumbent.evidence} ` + diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts index 687d633b92b..d23f065f478 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.test.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.test.ts @@ -14,6 +14,7 @@ import { resolveRelayEndpointBeforeRelaunch } from './ssh-relay-endpoint-takeover' import { RelayVersionMismatchError } from './ssh-relay-version-mismatch-error' +import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts' import type { SshConnection } from './ssh-connection' import { getRemoteHostPlatform } from './ssh-remote-platform' @@ -42,7 +43,7 @@ beforeEach(() => { describe('incumbent alive and refusing', () => { it('refuses to rebind a live relay holding PTYs, and signals nothing', async () => { execCommand.mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) // The whole point of #8585: the incumbent's socket must survive so it is not orphaned. @@ -52,9 +53,9 @@ describe('incumbent alive and refusing', () => { it('names the incumbent pid and the Reset Relay escape hatch in the error', async () => { execCommand.mockResolvedValue( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) - await expect(resolve()).rejects.toThrow(/3669803\(children=13\)/) + await expect(resolve()).rejects.toThrow(/3669803\(children=13,unrecognized=11\)/) await expect(resolve()).rejects.toThrow(/Reset Relay/) }) @@ -70,7 +71,7 @@ describe('incumbent alive and refusing', () => { it('reaps a live relay only when it provably holds nothing, and confirms it is gone', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') await expect(resolve()).resolves.toMatchObject({ verdict: 'live' }) @@ -80,7 +81,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over an empty relay whose death could not be confirmed', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -89,7 +90,7 @@ describe('incumbent alive and refusing', () => { it('does not launch over a relay the host refused to signal on its own re-check', async () => { execCommand .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('BUSY\n') await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError) @@ -138,9 +139,19 @@ describe('reapEmptyRelayHuskCommand', () => { }) it('aborts without signalling when the host cannot count children', () => { - expect(reapEmptyRelayHuskCommand(4242, SOCK)).toContain( - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }" - ) + const command = reapEmptyRelayHuskCommand(4242, SOCK) + // The census leaves both counters at `unknown` without pgrep, and the gate demands "0". + expect(command).toContain('unrecognized_kids=unknown') + expect(command).toContain('command -v pgrep >/dev/null 2>&1') + expect(command).toContain('[ "$unrecognized_kids" = "0" ] ||') + }) + + it('subtracts only the daemon service children it can name from the reap gate', () => { + const command = reapEmptyRelayHuskCommand(4242, SOCK) + for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) { + expect(command).toContain(`*'/${filename}'`) + } + expect(command).toContain('unrecognized_kids=$((unrecognized_kids+1))') }) }) diff --git a/src/main/ssh/ssh-relay-endpoint-takeover.ts b/src/main/ssh/ssh-relay-endpoint-takeover.ts index f104aab5256..8f6130620cb 100644 --- a/src/main/ssh/ssh-relay-endpoint-takeover.ts +++ b/src/main/ssh/ssh-relay-endpoint-takeover.ts @@ -2,13 +2,17 @@ * Deciding whether a relay socket path is ours to take, and acting on the answer. * * The only destructive action available here is a SIGTERM to a relay that has been proven — - * by argv, by socket-holder enumeration, and by a zero child count re-checked on the host - * immediately before the signal — to hold nothing at all. Everything else is left running. + * by argv, by socket-holder enumeration, and by a child census re-run on the host immediately + * before the signal — to hold nothing at all. Everything else is left running. * Per docs/reference/ssh-execution-boundary.md, a relay we merely failed to reach is * `unverifiable`, and `unverifiable` never authorizes a kill or a rebind. */ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' +import { + RELAY_UNRECOGNIZED_CHILD_COUNT_VAR, + relayDaemonChildCensusShell +} from './relay-daemon-service-children' import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers' import { describeRelayEndpointIncumbent, @@ -39,9 +43,10 @@ export function reapEmptyRelayHuskCommand(pid: number, sockPath: string): string `sock=${shellEscape(sockPath)}`, 'args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")', 'case "$args" in *relay.js*"$sock"*) ;; *) printf \'MISMATCH\\n\'; exit 0 ;; esac', - "command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }", - 'kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)', - '[ "$kids" = "0" ] || { printf \'BUSY\\n\'; exit 0; }', + // Why the same census as the probe: `unknown` (no pgrep) and any child this host could + // not account for as a relay service both land on BUSY, so nothing is signalled. + ...relayDaemonChildCensusShell(), + `[ "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}" = "0" ] || { printf 'BUSY\\n'; exit 0; }`, // SIGTERM only: the relay's own handler disposes and unlinks. SIGKILL would leave the // socket inode behind and skip that shutdown path for no gain on an empty daemon. 'kill -TERM "$pid" 2>/dev/null || true', diff --git a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts index 874d9aae3fe..9168f4688bd 100644 --- a/src/main/ssh/ssh-relay-superseded-endpoints.test.ts +++ b/src/main/ssh/ssh-relay-superseded-endpoints.test.ts @@ -67,7 +67,7 @@ describe('classifySupersededRelay', () => { 'PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', - 'HOLDER=3669803 yes 13' + 'HOLDER=3669803 yes 13 11' ]) ) ).toBe('retained-live-work') @@ -76,7 +76,7 @@ describe('classifySupersededRelay', () => { it('nominates only a proven empty relay for reaping', () => { expect( classifySupersededRelay( - incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) ).toBe('reap-candidate') }) @@ -101,7 +101,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11']) ) const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) expect(findings).toHaveLength(1) @@ -114,7 +114,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('GONE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) @@ -126,7 +126,7 @@ describe('sweepSupersededRelayEndpoints', () => { execCommand .mockResolvedValueOnce(`${OLD_SOCK}\n`) .mockResolvedValueOnce( - probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0']) + probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0']) ) .mockResolvedValueOnce('LIVE\n') const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP) diff --git a/src/shared/relay-artifacts.ts b/src/shared/relay-artifacts.ts index 273f6e059b8..2f9e563f839 100644 --- a/src/shared/relay-artifacts.ts +++ b/src/shared/relay-artifacts.ts @@ -37,6 +37,12 @@ export type RelayArtifact = { * optional one would loop forever redeploying a relay that is already correct. */ optional?: boolean + /** + * Forked by the relay daemon as a long-lived child of its own. These are relay + * infrastructure, never user work, and the reap gate subtracts them from a daemon's + * child census; see src/main/ssh/relay-daemon-service-children.ts. + */ + daemonServiceChild?: boolean } /** The bare Windows process-table addon; see docs/reference/windows-process-enumeration.md. */ @@ -44,8 +50,8 @@ export const RELAY_WINDOWS_PROCESS_TREE_FILENAME = 'windows-process-tree.node' export const RELAY_ARTIFACTS: readonly RelayArtifact[] = [ { filename: 'relay.js' }, - { filename: 'relay-watcher.js' }, - { filename: 'relay-ai-vault-service.js' }, + { filename: 'relay-watcher.js', daemonServiceChild: true }, + { filename: 'relay-ai-vault-service.js', daemonServiceChild: true }, { filename: 'managed-hook-runtime.js' }, // Forked by the AI Vault title reader; without it a relay answers every WSL // title request with no title and no error. @@ -62,6 +68,14 @@ export const RELAY_ARTIFACTS: readonly RelayArtifact[] = [ { filename: RELAY_WINDOWS_PROCESS_TREE_FILENAME, windowsOnly: true, optional: true } ] +/** + * The daemon's own service children, by entry filename. Anything else under a relay pid is + * either user work or unidentified, and both keep the relay unreapable. + */ +export const RELAY_DAEMON_SERVICE_ENTRY_FILENAMES: readonly string[] = RELAY_ARTIFACTS.filter( + (artifact) => artifact.daemonServiceChild +).map((artifact) => artifact.filename) + /** Written after the artifacts, so it is never an input to its own hash. */ export const RELAY_VERSION_FILENAME = '.version' From b85510f3a9a3751b7320f18f6c753720c265ed86 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 00:51:21 -0700 Subject: [PATCH 07/12] fix(terminal): warn about remote work when closing the window or quitting (#18593) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The native window-close warning was built from a local-only pty set: any worktree with a connectionId was dropped whole, and any remote runtime pty was filtered out. A build, test run, or agent on an SSH or Orca Remote host was therefore structurally invisible to it, on every platform. The quit path skipped the check entirely (#524), so remote work got no prompt at all. Route both paths through the same probe the tab-close guard uses, so the two cannot drift, and keep the verdict vocabulary of the SSH execution boundary: only a host that answers "no children" suppresses the warning. An unreachable host is `unverifiable`, never `exited`, so it warns rather than quitting silently — with its own copy, because "could not reach the host" is a different claim than "processes are running". Quit still ignores local ptys, preserving #524: quitting is an unambiguous instruction to end this machine's processes, but not to end execution on someone else's, which a bounded relay grace period will SIGKILL once the countdown expires. The probe budget is 1.5s (vs the tab guard's 4s) because quit is time sensitive; expiry raises the prompt, so an unreachable host costs a click rather than the 15s RPC timeout or a silently orphaned build. --- config/scripts/locale-ko-key-overrides.json | 2 +- .../components/TerminalWorkspaceDialogs.tsx | 14 +- .../terminal/pty-running-work-probe.ts | 88 +++++++ .../running-terminal-close-guard.test.ts | 12 +- .../terminal/running-terminal-close-guard.ts | 38 +-- ...terminal-tab-close-running-confirm.test.ts | 8 +- .../window-close-running-work.test.ts | 231 ++++++++++++++++++ .../terminal/window-close-running-work.ts | 70 ++++++ .../use-terminal-editor-close-foundation.ts | 53 ++-- ...tor-close-foundation.window-close.test.tsx | 103 ++++++++ src/renderer/src/i18n/locales/en.json | 3 +- src/renderer/src/i18n/locales/es.json | 2 +- src/renderer/src/i18n/locales/fr.json | 2 +- src/renderer/src/i18n/locales/ja.json | 2 +- src/renderer/src/i18n/locales/ko.json | 2 +- src/renderer/src/i18n/locales/zh.json | 2 +- src/shared/remote-execution-host-pty-id.ts | 14 ++ 17 files changed, 575 insertions(+), 71 deletions(-) create mode 100644 src/renderer/src/components/terminal/pty-running-work-probe.ts create mode 100644 src/renderer/src/components/terminal/window-close-running-work.test.ts create mode 100644 src/renderer/src/components/terminal/window-close-running-work.ts create mode 100644 src/renderer/src/components/use-terminal-editor-close-foundation.window-close.test.tsx create mode 100644 src/shared/remote-execution-host-pty-id.ts diff --git a/config/scripts/locale-ko-key-overrides.json b/config/scripts/locale-ko-key-overrides.json index f368ecc3cbc..bf5f62d1fa5 100644 --- a/config/scripts/locale-ko-key-overrides.json +++ b/config/scripts/locale-ko-key-overrides.json @@ -492,7 +492,7 @@ "ko": "agent CLI를 찾지 못했습니다. 하나를 설치하거나 설정에서 기본 agent를 선택하세요." }, "auto.components.Terminal.7958465754": { - "ko": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" + "ko": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?" }, "auto.components.Terminal.cdc9ac4b2d": { "ko": "편집기" diff --git a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx index 52ddbf1ba5d..bb3ffbfd621 100644 --- a/src/renderer/src/components/TerminalWorkspaceDialogs.tsx +++ b/src/renderer/src/components/TerminalWorkspaceDialogs.tsx @@ -24,6 +24,7 @@ export function TerminalWorkspaceDialogs({ saveDialogFile, saveDialogFileId, setWindowCloseDialogOpen, + windowCloseDialogKind, windowCloseDialogOpen } = controller return ( @@ -82,10 +83,15 @@ export function TerminalWorkspaceDialogs({ {translate('auto.components.Terminal.2fa9c69ff3', 'Close Window?')} - {translate( - 'auto.components.Terminal.7958465754', - 'There are local terminals with running processes. Close the window anyway?' - )} + {windowCloseDialogKind === 'unverifiable' + ? translate( + 'auto.components.Terminal.b7c1f0a934', + 'A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?' + ) + : translate( + 'auto.components.Terminal.7958465754', + 'There are terminals with running processes. Close the window anyway?' + )} diff --git a/src/renderer/src/components/terminal/pty-running-work-probe.ts b/src/renderer/src/components/terminal/pty-running-work-probe.ts new file mode 100644 index 00000000000..609b71f12ca --- /dev/null +++ b/src/renderer/src/components/terminal/pty-running-work-probe.ts @@ -0,0 +1,88 @@ +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { inspectRuntimeTerminalProcess } from '@/runtime/runtime-terminal-inspection' +import { isRemoteExecutionHostPtyId } from '../../../../shared/remote-execution-host-pty-id' +import { isClientOnlyUnverifiableInspection } from '../../../../shared/terminal-process-inspection' + +/** + * One probe answer in the fixed `live` / `unverifiable` / `exited` vocabulary of + * `docs/reference/ssh-execution-boundary.md`. `exited` is only ever produced by a host that + * answered; every failure to reach the owner — a rejection, a closed transport, or a deadline + * that expired first — stays `unverifiable`, because loss of contact is not evidence of death. + */ +export type PtyRunningWorkVerdict = 'live' | 'unverifiable' | 'exited' + +export type PtyRunningWorkProbe = { + ptyId: string + verdict: PtyRunningWorkVerdict + /** Why the owner could not be observed. Only set for `unverifiable`. */ + reason?: string + /** The deadline expired before this pty's probe answered at all. */ + timedOut: boolean + /** The pty is owned by a remote execution host (relay runtime or app SSH). */ + remote: boolean +} + +type ProbeSettings = Pick | null | undefined + +/** + * Probes every pty for running work and resolves at whichever comes first: every answer, or the + * deadline. Never rejects, and never reports a pty it did not hear back about as idle. + * + * Callers own the policy. This owns only the measurement, so the tab-close guard and the + * window-close guard cannot drift apart on what an unanswered remote host means. + */ +export async function probePtyRunningWork( + settings: ProbeSettings, + ptyIds: readonly string[], + options: { timeoutMs: number } +): Promise { + if (ptyIds.length === 0) { + return [] + } + const probes: PtyRunningWorkProbe[] = ptyIds.map((ptyId) => ({ + ptyId, + verdict: 'unverifiable', + reason: 'probe_deadline', + timedOut: true, + remote: isRemoteExecutionHostPtyId(ptyId) + })) + + const settle = Promise.all( + ptyIds.map(async (ptyId, index) => { + const probe = probes[index] + if (!probe) { + return + } + try { + const inspection = await inspectRuntimeTerminalProcess(settings, ptyId) + probe.timedOut = false + if (isClientOnlyUnverifiableInspection(inspection)) { + probe.verdict = 'unverifiable' + probe.reason = inspection.reason + return + } + probe.verdict = inspection.hasChildProcesses ? 'live' : 'exited' + delete probe.reason + } catch { + // Why: `inspectRuntimeTerminalProcess` already maps every failure it can classify onto a + // reason; an unclassified throw is still a failure to observe, so it stays unverifiable. + probe.timedOut = false + probe.verdict = 'unverifiable' + probe.reason = 'probe_failed' + } + }) + ) + + let deadline: ReturnType | undefined + try { + await Promise.race([ + settle, + new Promise((resolve) => { + deadline = setTimeout(resolve, options.timeoutMs) + }) + ]) + } finally { + clearTimeout(deadline) + } + return probes +} diff --git a/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts b/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts index d63c69df7c9..165119f8b31 100644 --- a/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts +++ b/src/renderer/src/components/terminal/running-terminal-close-guard.test.ts @@ -46,10 +46,12 @@ function visibleRequest() { return useRunningTerminalCloseConfirmStore.getState().runningTerminalCloseConfirm } +// Drains pending microtasks. The probe resolves through several await points (per-pty inspect, +// the batch join, the deadline race), so this flushes generously rather than counting ticks. async function settleProbe(): Promise { - await Promise.resolve() - await Promise.resolve() - await Promise.resolve() + for (let tick = 0; tick < 12; tick += 1) { + await Promise.resolve() + } } describe('shouldConfirmRunningTerminalClose', () => { @@ -329,6 +331,7 @@ describe('guardRunningTerminalClose', () => { vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() + await settleProbe() expect(onClose).not.toHaveBeenCalled() expect(visibleRequest()).toMatchObject({ terminalTabId: 'tab-1', tabLabel: 'npm run dev' }) @@ -351,6 +354,7 @@ describe('guardRunningTerminalClose', () => { guard() vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() + await settleProbe() expect(visibleRequest()?.copyKind).toBe('agent') }) @@ -368,6 +372,7 @@ describe('guardRunningTerminalClose', () => { guard(onClose) vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() + await settleProbe() requestSpy.mockRestore() expect(onClose).toHaveBeenCalledTimes(1) @@ -385,6 +390,7 @@ describe('guardRunningTerminalClose', () => { vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS) vi.useRealTimers() await settleProbe() + await settleProbe() expect(onClose).not.toHaveBeenCalled() useRunningTerminalCloseConfirmStore.getState().confirmRunningTerminalClose() diff --git a/src/renderer/src/components/terminal/running-terminal-close-guard.ts b/src/renderer/src/components/terminal/running-terminal-close-guard.ts index 881cd262353..6bb9ff5a582 100644 --- a/src/renderer/src/components/terminal/running-terminal-close-guard.ts +++ b/src/renderer/src/components/terminal/running-terminal-close-guard.ts @@ -1,10 +1,9 @@ import { useAppStore } from '@/store' -import { inspectRuntimeTerminalProcess } from '@/runtime/runtime-terminal-inspection' import { useRunningTerminalCloseConfirmStore } from '@/store/running-terminal-close-confirm' import type { TerminalTabCloseReason } from '@/store/slices/terminal-tab-retirement' import type { AppState } from '@/store/types' import { resolveBusyPtyCloseCopyKind } from './terminal-close-copy-kind' -import { isClientOnlyUnverifiableInspection } from '../../../../shared/terminal-process-inspection' +import { probePtyRunningWork } from './pty-running-work-probe' export type RunningTerminalCloseGuardOptions = { force?: boolean @@ -44,7 +43,7 @@ export function shouldConfirmRunningTerminalClose( * the store's own teardown collector unions both for exactly that reason — reading only * the map would let a close slip through the window with no prompt. A stale id costs * nothing: its probe fails and the guard falls open. */ -function collectTabPtyIds( +export function collectTabPtyIds( state: Pick, terminalTabId: string ): string[] { @@ -112,44 +111,33 @@ export function guardRunningTerminalClose(params: { decided = true } - const probeTimeout = setTimeout(() => { - try { + void probePtyRunningWork(settings, ptyIds, { timeoutMs: RUNNING_CLOSE_PROBE_TIMEOUT_MS }) + .then((probes) => { + if (decided) { + return + } // Why: a probe that has not answered yet is unknown, not idle. Ask, treating every pty // as a candidate, so a degraded relay costs a click instead of a killed remote command. - confirmClose(ptyIds) - } catch { - closeNow() - } - }, RUNNING_CLOSE_PROBE_TIMEOUT_MS) - - void Promise.allSettled(ptyIds.map((ptyId) => inspectRuntimeTerminalProcess(settings, ptyId))) - .then((results) => { - clearTimeout(probeTimeout) - if (decided) { + if (probes.some((probe) => probe.timedOut)) { + confirmClose(ptyIds) return } // Why: fail open on an *answered* probe, matching the Cmd+W pane path — a rejection // (wedged relay, legacy provider) or a stale remote handle is not evidence of a live // child, and a close button that silently does nothing is worse than closing a busy tab. - const busyPtyIds = ptyIds.filter((_, index) => { - const result = results[index] - return ( - result?.status === 'fulfilled' && - !isClientOnlyUnverifiableInspection(result.value) && - result.value.hasChildProcesses - ) - }) + const busyPtyIds = probes + .filter((probe) => probe.verdict === 'live') + .map((probe) => probe.ptyId) if (busyPtyIds.length === 0) { closeNow() return } confirmClose(busyPtyIds) }) - // Why: allSettled never rejects, so this only fires when the decision above throws (a + // Why: the probe never rejects, so this only fires when the decision above throws (a // copy-kind lookup, a store subscriber). Without it the tab would silently never close // and the user would get no feedback at all; the pane path it replaced had this catch. .catch(() => { - clearTimeout(probeTimeout) closeNow() }) } diff --git a/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts b/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts index 066c5795ef9..9dfe20ef501 100644 --- a/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts +++ b/src/renderer/src/components/terminal/terminal-tab-close-running-confirm.test.ts @@ -87,10 +87,12 @@ function visibleRequest() { return useRunningTerminalCloseConfirmStore.getState().runningTerminalCloseConfirm } +// Drains pending microtasks. The probe resolves through several await points (per-pty inspect, +// the batch join, the deadline race), so this flushes generously rather than counting ticks. async function settleProbe(): Promise { - await Promise.resolve() - await Promise.resolve() - await Promise.resolve() + for (let tick = 0; tick < 12; tick += 1) { + await Promise.resolve() + } } describe('closeTerminalTab running-process confirmation', () => { diff --git a/src/renderer/src/components/terminal/window-close-running-work.test.ts b/src/renderer/src/components/terminal/window-close-running-work.test.ts new file mode 100644 index 00000000000..77dc4ad745b --- /dev/null +++ b/src/renderer/src/components/terminal/window-close-running-work.test.ts @@ -0,0 +1,231 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { getStateMock, inspectRuntimeTerminalProcessMock } = vi.hoisted(() => ({ + getStateMock: vi.fn(), + inspectRuntimeTerminalProcessMock: vi.fn() +})) + +vi.mock('@/store', () => ({ + useAppStore: { getState: getStateMock } +})) + +vi.mock('@/runtime/runtime-terminal-inspection', () => ({ + inspectRuntimeTerminalProcess: inspectRuntimeTerminalProcessMock +})) + +import { + assessWindowCloseRunningWork, + WINDOW_CLOSE_PROBE_TIMEOUT_MS +} from './window-close-running-work' + +const LOCAL_PTY = 'pty-local' +const SSH_PTY = 'ssh:openclaw@@pty-7' +const RUNTIME_PTY = 'remote:env-1@@handle-1' +/** A runtime pty minted without an owner id. Still someone else's machine. */ +const OWNERLESS_RUNTIME_PTY = 'remote:handle-2' + +const BUSY = { + foregroundProcess: 'pnpm build', + hasChildProcesses: true, + foregroundProcessEvidence: {} +} +const IDLE = { foregroundProcess: 'bash', hasChildProcesses: false, foregroundProcessEvidence: {} } +const UNVERIFIABLE = { + foregroundProcess: null, + hasChildProcesses: false, + verdict: 'unverifiable', + reason: 'transport_loss' +} + +/** One worktree, one tab, owning `ptyIds`. */ +function setState(ptyIds: string[]): void { + getStateMock.mockReturnValue({ + settings: { activeRuntimeEnvironmentId: null }, + tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] }, + ptyIdsByTabId: { 'tab-1': ptyIds }, + terminalLayoutsByTabId: {} + }) +} + +/** Answers each pty id from `byPtyId`; anything unlisted never settles. */ +function answerWith(byPtyId: Record): void { + inspectRuntimeTerminalProcessMock.mockImplementation((_settings: unknown, ptyId: string) => + ptyId in byPtyId ? Promise.resolve(byPtyId[ptyId]) : new Promise(() => {}) + ) +} + +beforeEach(() => { + vi.clearAllMocks() +}) + +afterEach(() => { + vi.useRealTimers() +}) + +describe('assessWindowCloseRunningWork', () => { + it('warns about a live process on an SSH host (F15: remote work was filtered out entirely)', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('warns on quit about a live process on an SSH host', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('warns on quit about a live process on a paired runtime host', async () => { + setState([RUNTIME_PTY]) + answerWith({ [RUNTIME_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('counts an owner-less remote pty as remote work', async () => { + setState([OWNERLESS_RUNTIME_PTY]) + answerWith({ [OWNERLESS_RUNTIME_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + // The crux of docs/reference/ssh-execution-boundary.md: an unreachable host is `unverifiable`, + // and quitting on `unverifiable` as though it were `exited` is what orphans live remote work. + it('warns rather than quitting silently when a remote host answers unverifiable', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: UNVERIFIABLE }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'unverifiable' + }) + }) + + it('warns rather than quitting silently when a remote probe throws', async () => { + setState([SSH_PTY]) + inspectRuntimeTerminalProcessMock.mockRejectedValue(new Error('relay wedged')) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'unverifiable' + }) + }) + + it('stops waiting at the budget and warns, so an unreachable host cannot hang the quit', async () => { + setState([SSH_PTY]) + answerWith({}) + vi.useFakeTimers() + + const pending = assessWindowCloseRunningWork({ isQuitting: true }) + await vi.advanceTimersByTimeAsync(WINDOW_CLOSE_PROBE_TIMEOUT_MS) + + await expect(pending).resolves.toEqual({ kind: 'unverifiable' }) + }) + + it('does not resolve before the budget expires', async () => { + setState([SSH_PTY]) + answerWith({}) + vi.useFakeTimers() + const settled = vi.fn() + + void assessWindowCloseRunningWork({ isQuitting: true }).then(settled) + await vi.advanceTimersByTimeAsync(WINDOW_CLOSE_PROBE_TIMEOUT_MS - 1) + + expect(settled).not.toHaveBeenCalled() + }) + + it('does not warn when the owning remote host reports an idle shell', async () => { + setState([SSH_PTY]) + answerWith({ [SSH_PTY]: IDLE }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'none' + }) + }) + + it('reports a live process even when a sibling remote pane is only unverifiable', async () => { + setState([SSH_PTY, RUNTIME_PTY]) + answerWith({ [SSH_PTY]: UNVERIFIABLE, [RUNTIME_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('still warns about a live local process when closing the window', async () => { + setState([LOCAL_PTY]) + answerWith({ [LOCAL_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'running' + }) + }) + + // A local probe has no transport to lose, so its failure means the pty is gone — unlike a + // remote host going quiet, it is not a reason to hold up the close. + it('does not warn when only a local probe is unverifiable', async () => { + setState([LOCAL_PTY]) + answerWith({ [LOCAL_PTY]: UNVERIFIABLE }) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'none' + }) + }) + + // #524 decided quitting is an unambiguous instruction to end this machine's processes. It is + // not an instruction to end execution on someone else's, which is why remote still warns above. + it('leaves local-only quit unprompted, and never probes for it', async () => { + setState([LOCAL_PTY]) + answerWith({ [LOCAL_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'none' + }) + expect(inspectRuntimeTerminalProcessMock).not.toHaveBeenCalled() + }) + + it('probes a pane the layout has bound before the liveness map caught up', async () => { + getStateMock.mockReturnValue({ + settings: { activeRuntimeEnvironmentId: null }, + tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] }, + ptyIdsByTabId: {}, + terminalLayoutsByTabId: { 'tab-1': { ptyIdsByLeafId: { leaf: SSH_PTY } } } + }) + answerWith({ [SSH_PTY]: BUSY }) + + await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({ + kind: 'running' + }) + }) + + it('probes each pty once when the map and the layout name the same one', async () => { + getStateMock.mockReturnValue({ + settings: { activeRuntimeEnvironmentId: null }, + tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] }, + ptyIdsByTabId: { 'tab-1': [SSH_PTY] }, + terminalLayoutsByTabId: { 'tab-1': { ptyIdsByLeafId: { leaf: SSH_PTY } } } + }) + answerWith({ [SSH_PTY]: IDLE }) + + await assessWindowCloseRunningWork({ isQuitting: true }) + + expect(inspectRuntimeTerminalProcessMock).toHaveBeenCalledTimes(1) + }) + + it('closes without probing when no workspace owns a pty', async () => { + setState([]) + + await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({ + kind: 'none' + }) + expect(inspectRuntimeTerminalProcessMock).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/terminal/window-close-running-work.ts b/src/renderer/src/components/terminal/window-close-running-work.ts new file mode 100644 index 00000000000..427ee72dae6 --- /dev/null +++ b/src/renderer/src/components/terminal/window-close-running-work.ts @@ -0,0 +1,70 @@ +import { useAppStore } from '@/store' +import { isRemoteExecutionHostPtyId } from '../../../../shared/remote-execution-host-pty-id' +import { collectTabPtyIds } from './running-terminal-close-guard' +import { probePtyRunningWork } from './pty-running-work-probe' + +/** + * Upper bound on how long closing the window or quitting may wait on the probes. + * + * Shorter than the tab-close guard's 4s because quit is time-sensitive in a way one tab close is + * not: the user has already asked to leave, and a quit that stalls on an unreachable host is its + * own bug. A healthy local inspect answers in single-digit milliseconds and a healthy remote one + * is a single RPC round-trip on an already-open mux channel, so this leaves roughly 3x headroom + * over a slow-but-live transcontinental host while capping the worst case — a host that is simply + * gone — at ~1.5s instead of the 15s RPC timeout the probe would otherwise inherit. + * + * Expiry raises the prompt rather than quitting silently: an unanswered probe is `unverifiable`, + * and `unverifiable` is never evidence that remote work has stopped. + */ +export const WINDOW_CLOSE_PROBE_TIMEOUT_MS = 1_500 + +/** Which warning the close should raise, if any. */ +export type WindowCloseRunningWork = + /** Every pty that mattered answered, and none had children. */ + | { kind: 'none' } + /** An owning host reported a live child process. */ + | { kind: 'running' } + /** A remote execution host could not be observed, so its work may still be live. */ + | { kind: 'unverifiable' } + +/** + * Decides whether a window close or quit should stop and ask. + * + * Two deliberate asymmetries: + * + * - **Quit only considers remote ptys.** Quitting is an unambiguous instruction to end this + * machine's processes (#524), but it is not an instruction to end execution on someone else's: + * the client detaches while the relay keeps running, and a target with a bounded grace period + * then SIGKILLs that work once the countdown expires. + * - **Only a remote `unverifiable` warns.** A local probe has no transport to lose, so its failure + * means the pty is gone. A remote one that cannot be reached is the case + * `docs/reference/ssh-execution-boundary.md` exists to protect: loss of contact is not evidence + * of `exited`, so it must fail toward asking rather than toward a silent quit. + */ +export async function assessWindowCloseRunningWork(params: { + isQuitting: boolean +}): Promise { + const state = useAppStore.getState() + const ptyIds = new Set( + Object.values(state.tabsByWorktree) + .flatMap((worktreeTabs) => worktreeTabs ?? []) + .flatMap((tab) => collectTabPtyIds(state, tab.id)) + ) + const candidatePtyIds = params.isQuitting + ? [...ptyIds].filter(isRemoteExecutionHostPtyId) + : [...ptyIds] + if (candidatePtyIds.length === 0) { + return { kind: 'none' } + } + + const probes = await probePtyRunningWork(state.settings, candidatePtyIds, { + timeoutMs: WINDOW_CLOSE_PROBE_TIMEOUT_MS + }) + if (probes.some((probe) => probe.verdict === 'live')) { + return { kind: 'running' } + } + if (probes.some((probe) => probe.remote && probe.verdict === 'unverifiable')) { + return { kind: 'unverifiable' } + } + return { kind: 'none' } +} diff --git a/src/renderer/src/components/use-terminal-editor-close-foundation.ts b/src/renderer/src/components/use-terminal-editor-close-foundation.ts index 74849e3f5a2..2aceced683d 100644 --- a/src/renderer/src/components/use-terminal-editor-close-foundation.ts +++ b/src/renderer/src/components/use-terminal-editor-close-foundation.ts @@ -1,8 +1,9 @@ import { useCallback, useRef, useState } from 'react' -import { useAppStore } from '../store' -import { getConnectionId } from '../lib/connection-context' -import { isRemoteRuntimePtyId } from '@/runtime/runtime-terminal-inspection' import { CLOSE_DIALOG_DEBOUNCE_MS } from './terminal-workspace-model' +import { + assessWindowCloseRunningWork, + type WindowCloseRunningWork +} from './terminal/window-close-running-work' import type { TerminalWorkspaceProjectionController } from './use-terminal-workspace-projection' import { runWithWindowCloseCheckpointScope } from './window-close-request-coordinator' import { showShutdownCheckpointFailureToast } from '@/lib/shutdown-checkpoint-failure-toast' @@ -27,6 +28,11 @@ export function useTerminalEditorCloseFoundation( closeDialogDebounceTimersRef.current.add(timer) }, []) const [windowCloseDialogOpen, setWindowCloseDialogOpen] = useState(false) + // Why: "running" and "could not reach the host" are different claims, and telling the user + // processes are running when the truth is that a host went quiet is the fabricated certainty + // docs/reference/ssh-execution-boundary.md forbids. + const [windowCloseDialogKind, setWindowCloseDialogKind] = + useState>('running') const windowCloseAfterDirtyRef = useRef<{ isQuitting: boolean } | null>(null) const confirmNativeWindowClose = useCallback(() => { @@ -46,33 +52,21 @@ export function useTerminalEditorCloseFoundation( const proceedToNativeWindowClose = useCallback( (isQuitting: boolean) => { - if (!isQuitting) { - const state = useAppStore.getState() - const localPtyIds = Object.entries(state.tabsByWorktree).flatMap( - ([worktreeId, worktreeTabs]) => { - const connectionId = getConnectionId(worktreeId) - if (connectionId !== null) { - return [] - } - return worktreeTabs - .flatMap((tab) => state.ptyIdsByTabId[tab.id] ?? []) - .filter((ptyId) => !isRemoteRuntimePtyId(ptyId)) + void assessWindowCloseRunningWork({ isQuitting }) + .then((runningWork) => { + if (runningWork.kind === 'none') { + confirmNativeWindowClose() + return } - ) - if (localPtyIds.length > 0) { - void Promise.all(localPtyIds.map((id) => window.api.pty.hasChildProcesses(id))).then( - (results) => { - if (results.some(Boolean)) { - setWindowCloseDialogOpen(true) - } else { - confirmNativeWindowClose() - } - } - ) - return - } - } - confirmNativeWindowClose() + setWindowCloseDialogKind(runningWork.kind) + setWindowCloseDialogOpen(true) + }) + // Why: the assessment must never be able to trap the window. A thrown store read is + // not evidence either way, and a close that silently does nothing is unrecoverable + // without SIGKILL, so fall through to the close the user actually asked for. + .catch(() => { + confirmNativeWindowClose() + }) }, [confirmNativeWindowClose] ) @@ -88,6 +82,7 @@ export function useTerminalEditorCloseFoundation( releaseCloseDialogGuardAfterDebounce, windowCloseDialogOpen, setWindowCloseDialogOpen, + windowCloseDialogKind, windowCloseAfterDirtyRef, confirmNativeWindowClose, proceedToNativeWindowClose diff --git a/src/renderer/src/components/use-terminal-editor-close-foundation.window-close.test.tsx b/src/renderer/src/components/use-terminal-editor-close-foundation.window-close.test.tsx new file mode 100644 index 00000000000..d7f3f69d925 --- /dev/null +++ b/src/renderer/src/components/use-terminal-editor-close-foundation.window-close.test.tsx @@ -0,0 +1,103 @@ +// @vitest-environment happy-dom + +/** + * Wiring for the window-close/quit running-work warning. The policy in + * `terminal/window-close-running-work.ts` is inert unless `proceedToNativeWindowClose` actually + * consults it, so pin that it does — and that a warning stops the native close rather than + * confirming it. + */ +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { assessWindowCloseRunningWorkMock, confirmWindowCloseMock } = vi.hoisted(() => ({ + assessWindowCloseRunningWorkMock: vi.fn(), + confirmWindowCloseMock: vi.fn() +})) + +vi.mock('./terminal/window-close-running-work', () => ({ + assessWindowCloseRunningWork: assessWindowCloseRunningWorkMock +})) +vi.mock('./window-close-request-coordinator', () => ({ + runWithWindowCloseCheckpointScope: (fn: () => unknown) => fn() +})) +vi.mock('@/lib/shutdown-checkpoint-failure-toast', () => ({ + showShutdownCheckpointFailureToast: vi.fn() +})) + +const { useTerminalEditorCloseFoundation } = await import('./use-terminal-editor-close-foundation') + +const controller = { openFiles: [] } as unknown as Parameters< + typeof useTerminalEditorCloseFoundation +>[0] + +function mountFoundation() { + return renderHook(() => useTerminalEditorCloseFoundation(controller)) +} + +beforeEach(() => { + vi.clearAllMocks() + Object.assign(globalThis, { + window: Object.assign(globalThis.window, { + api: { ui: { confirmWindowClose: confirmWindowCloseMock } } + }) + }) +}) + +afterEach(() => { + cleanup() +}) + +describe('proceedToNativeWindowClose', () => { + it('asks the running-work policy about the quit rather than assuming it is safe', async () => { + assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'none' }) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(true) + }) + + expect(assessWindowCloseRunningWorkMock).toHaveBeenCalledWith({ isQuitting: true }) + expect(confirmWindowCloseMock).toHaveBeenCalledTimes(1) + expect(result.current.windowCloseDialogOpen).toBe(false) + }) + + it('raises the dialog and does not close when a host reports live work', async () => { + assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'running' }) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(true) + }) + + expect(result.current.windowCloseDialogOpen).toBe(true) + expect(result.current.windowCloseDialogKind).toBe('running') + expect(confirmWindowCloseMock).not.toHaveBeenCalled() + }) + + it('raises the unverifiable copy when a remote host could not be reached', async () => { + assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'unverifiable' }) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(true) + }) + + expect(result.current.windowCloseDialogOpen).toBe(true) + expect(result.current.windowCloseDialogKind).toBe('unverifiable') + expect(confirmWindowCloseMock).not.toHaveBeenCalled() + }) + + // Why: a thrown assessment is not evidence either way, and a close that silently does nothing + // leaves SIGKILL as the user's only exit. + it('falls through to the close when the assessment throws', async () => { + assessWindowCloseRunningWorkMock.mockRejectedValue(new Error('store blew up')) + const { result } = mountFoundation() + + await act(async () => { + result.current.proceedToNativeWindowClose(false) + }) + + expect(confirmWindowCloseMock).toHaveBeenCalledTimes(1) + expect(result.current.windowCloseDialogOpen).toBe(false) + }) +}) diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 87fcb50a2a1..f2da58bffb3 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -2251,7 +2251,8 @@ "Terminal": { "73768427cf": "Close", "f82e9f02df": "Cancel", - "7958465754": "There are local terminals with running processes. Close the window anyway?", + "7958465754": "There are terminals with running processes. Close the window anyway?", + "b7c1f0a934": "A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?", "2fa9c69ff3": "Close Window?", "cd51e28d8b": "Save", "0037b21794": "Don't Save", diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 9999634daee..98b07917880 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -1924,7 +1924,7 @@ "Terminal": { "73768427cf": "Cerrar", "f82e9f02df": "Cancelar", - "7958465754": "Hay terminales locales con procesos en ejecución. ¿Cerrar la ventana de todos modos?", + "7958465754": "Hay terminales con procesos en ejecución. ¿Cerrar la ventana de todos modos?", "2fa9c69ff3": "¿Cerrar ventana?", "cd51e28d8b": "Guardar", "0037b21794": "No guardar", diff --git a/src/renderer/src/i18n/locales/fr.json b/src/renderer/src/i18n/locales/fr.json index 8eff250a820..5bf316d36f7 100644 --- a/src/renderer/src/i18n/locales/fr.json +++ b/src/renderer/src/i18n/locales/fr.json @@ -2087,7 +2087,7 @@ "Terminal": { "73768427cf": "Fermer", "f82e9f02df": "Annuler", - "7958465754": "Des terminaux locaux exécutent des processus. Fermer quand même la fenêtre ?", + "7958465754": "Des terminaux exécutent des processus. Fermer quand même la fenêtre ?", "2fa9c69ff3": "Fermer la fenêtre ?", "cd51e28d8b": "Enregistrer", "0037b21794": "Ne pas enregistrer", diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index b3c30da9446..4dc81b9201d 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -1924,7 +1924,7 @@ "Terminal": { "73768427cf": "閉じる", "f82e9f02df": "キャンセル", - "7958465754": "プロセスが実行中のローカルターミナルがあります。このままウィンドウを閉じますか?", + "7958465754": "プロセスが実行中のターミナルがあります。このままウィンドウを閉じますか?", "2fa9c69ff3": "ウィンドウを閉じますか?", "cd51e28d8b": "保存", "0037b21794": "保存しないでください", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index ac75cca2209..d9d822b8fc3 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -1929,7 +1929,7 @@ "Terminal": { "73768427cf": "닫기", "f82e9f02df": "취소", - "7958465754": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?", + "7958465754": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?", "2fa9c69ff3": "창을 닫으시겠습니까?", "cd51e28d8b": "저장", "0037b21794": "저장하지 않음", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 7a3b47c8f2a..46dc592dd2c 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -1927,7 +1927,7 @@ "Terminal": { "73768427cf": "关闭", "f82e9f02df": "取消", - "7958465754": "有正在运行的进程的本地终端。还是关窗吧?", + "7958465754": "有正在运行的进程的终端。还是关窗吧?", "2fa9c69ff3": "关闭窗口?", "cd51e28d8b": "保存", "0037b21794": "不保存", diff --git a/src/shared/remote-execution-host-pty-id.ts b/src/shared/remote-execution-host-pty-id.ts new file mode 100644 index 00000000000..c76ada39e03 --- /dev/null +++ b/src/shared/remote-execution-host-pty-id.ts @@ -0,0 +1,14 @@ +import { parseRemoteRuntimePtyId } from './remote-runtime-pty-id' +import { parseAppSshPtyId } from './ssh-pty-id' + +/** + * Whether the process behind this pty runs on an execution host other than this machine — + * a paired runtime environment or an app SSH target. + * + * Deliberately broader than the inspection module's private remote check, which only counts a + * `remote:` id that carries an owner environment id. An owner-less `remote:` still runs + * somewhere else, and treating it as local is how remote work becomes invisible to a guard. + */ +export function isRemoteExecutionHostPtyId(ptyId: string): boolean { + return parseRemoteRuntimePtyId(ptyId) !== null || parseAppSshPtyId(ptyId) !== null +} From 11e459e9330b0712988b9c4afb063c71ee0b1507 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 00:53:24 -0700 Subject: [PATCH 08/12] fix(crash-reporting): bound replay-guard wedge bursts in the ring without losing their spans (#18441) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `terminal_replay_guard_wedged_release` was not in COALESCED_RENDERER_BREADCRUMB_NAMES, and its per-pane hashes give every entry a unique ring identity. One mount/reveal/wake transition expires every in-flight replay write at once, so a burst arrives as N distinct entries against a 30-slot FIFO ring. Measured, from the 09-02 corpus (121 `renderer.breadcrumb` spans across 9 of 55 diagnostic bundles): - bundle 26461769: 26 events in 0.96s - murlock1000: 62 events over 85s - 8907a508 mixes two call sites in one window (2 crumbs carry `tabIdHash`, 2 do not) Not measured: no captured report's ring actually lost slots to this crumb. All 57 reports have zero wedge crumbs in `Recent activity:`, and in all 9 bundles the burst predates the report's ring window — for 26461769 the burst ran 13:36:03.784Z-13:36:04.742Z while the ring-owning main process started at 13:43:44.380Z, 7m40s later. So this bounds a demonstrated hazard, not an observed loss. An earlier draft of this commit asserted "26 of 30 slots / 87% of the pre-crash trail" as a measurement; that was a model, and it is removed. The burst evidence lives entirely in the durable span stream, and suppressed repeats normally emit no span (see the 1000-emissions/1-span case in crash-reporting-renderer-breadcrumbs.test.ts), so coalescing alone would have cut that 121-event corpus to 13 with the multiplicity recorded nowhere. Instead: - the ring coalesces: one slot per call site, plus `suppressedSinceLast` - every wedge event still emits its own `renderer.breadcrumb` span, via PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES. Span volume is unchanged at 121, and the span deliberately carries no count so a span-stream total cannot double-count what the ring already claims - the coalesce key is `ptyId`/`tabIdHash` *presence*, not name alone: those fields are absent on the restore call site (restoreScrollbackBuffers) and present on reattach, so name-only keying would collapse 8907a508's two call sites into whichever crumb landed last. Bounded at 4 slots per storm, matching the webgl `kind` and duplicate-tab `resolvedToActiveWorktree` precedents in the same file. Replaying the corpus timestamps: 121 events -> 14 ring writes. This is a diagnostics fix, not a crash fix. It does not stop panes wedging, and it does not explain the "can't type" reports in this round. --- .../crash-reporting-renderer-breadcrumbs.ts | 24 +++- ...reporting-replay-guard-wedge-burst.test.ts | 128 ++++++++++++++++++ 2 files changed, 149 insertions(+), 3 deletions(-) create mode 100644 src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts diff --git a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts index 97e0a8f9d65..8d126f557b5 100644 --- a/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts +++ b/src/main/ipc/crash-reporting-renderer-breadcrumbs.ts @@ -49,6 +49,7 @@ function recordRendererBreadcrumbTrace( const DUPLICATE_TAB_OWNER_BREADCRUMB = 'terminal_tab_id_owned_by_multiple_worktrees' const PARK_VERDICT_CHURN_BREADCRUMB = 'terminal_park_verdict_churn' const REACT_COMMIT_CASCADE_BREADCRUMB = 'react_commit_cascade' +const REPLAY_GUARD_WEDGED_BREADCRUMB = 'terminal_replay_guard_wedged_release' const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ 'renderer_error', 'renderer_unhandled_rejection', @@ -56,6 +57,7 @@ const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([ DUPLICATE_TAB_OWNER_BREADCRUMB, PARK_VERDICT_CHURN_BREADCRUMB, REACT_COMMIT_CASCADE_BREADCRUMB, + REPLAY_GUARD_WEDGED_BREADCRUMB, TERMINAL_WEBGL_DIAGNOSTIC_BREADCRUMB ]) const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 @@ -69,6 +71,11 @@ const RENDERER_BREADCRUMB_COALESCE_MS = 30_000 // 30-entry ring to two such bursts. `suppressedSinceLast` keeps the pane count // — the only signal these carry — in one slot. const NAME_ONLY_COALESCED_BREADCRUMB_NAMES = new Set(['terminal_safe_fit_retry_exhausted']) +// Why: the 30-slot ring is the scarce sink; the durable span stream is not. For +// bounded-rate pane telemetry whose multiplicity is the whole signal, spans are the +// only place a burst survives the restart that clears the ring, so coalesce the ring +// but keep every event's span. +const PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES = new Set([REPLAY_GUARD_WEDGED_BREADCRUMB]) function rendererBreadcrumbCoalesceKey( name: string, @@ -77,6 +84,13 @@ function rendererBreadcrumbCoalesceKey( if (NAME_ONLY_COALESCED_BREADCRUMB_NAMES.has(name)) { return name } + // Why presence and not value: `ptyId`/`tabIdHash` are absent on the restore call + // site (layout-serialization restoreScrollbackBuffers) and present on reattach, so + // their presence is the call-site identity a mixed burst would otherwise lose. Four + // slots per storm at most, regardless of pane count. + if (name === REPLAY_GUARD_WEDGED_BREADCRUMB) { + return `${name}:${data?.ptyId ? 'pty' : ''}:${data?.tabIdHash ? 'tab' : ''}` + } // Why trigger and not name alone: `burst` means damping engaged a commit // short of React #185, `window` means slow benign churn. Collapsing them // would drop the near-crash signal into a slow-churn slot. Still bounded — @@ -191,9 +205,13 @@ export function recordRendererBreadcrumbFromRenderer( minIntervalMs: RENDERER_BREADCRUMB_COALESCE_MS, ...(origin ? { origin } : {}) }) - // Why: tracing every suppressed duplicate would preserve the same - // serialization and disk churn that breadcrumb coalescing removes. - if (coalesceResult) { + if (PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES.has(args.name)) { + // Why the raw data: every event already gets its own span, so folding the ring's + // running count in here would double-count in any span-stream total. + recordRendererBreadcrumbTrace(args.name, data) + } else if (coalesceResult) { + // Why gated: tracing every suppressed duplicate would preserve the same + // serialization and disk churn that breadcrumb coalescing removes. recordRendererBreadcrumbTrace( args.name, coalesceResult.suppressedSinceLast > 0 diff --git a/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts new file mode 100644 index 00000000000..e3823f0b313 --- /dev/null +++ b/src/main/ipc/crash-reporting-replay-guard-wedge-burst.test.ts @@ -0,0 +1,128 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { + clearCrashBreadcrumbsForTest, + getCrashBreadcrumbSnapshot, + recordCrashBreadcrumb +} from '../crash-reporting/crash-breadcrumb-store' +import { recordRendererBreadcrumbFromRenderer } from './crash-reporting-renderer-breadcrumbs' + +type SpanOptions = { attributes: Record } +const startSpanMock = vi.fn((_name: string, _options: SpanOptions) => ({ end: () => {} })) +vi.mock('../observability/tracer', () => ({ + startSpan: (name: string, options: SpanOptions) => startSpanMock(name, options) +})) + +const WEDGE_BREADCRUMB = 'terminal_replay_guard_wedged_release' + +/** Reattach-path shape: identity-bearing (`tabIdHash`, optionally `ptyId`). */ +function emitReattachWedge(pane: number, withPtyId = false): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { + paneId: pane, + leafIdHash: `leaf${String(pane).padStart(5, '0')}`, + tabIdHash: `tab${String(pane).padStart(6, '0')}`, + worktreeIdHash: 'caa15fa9', + ...(withPtyId ? { ptyId: `…@@pty-${pane}` } : {}) + } + }) +} + +/** Restore-path shape (restoreScrollbackBuffers): no tabIdHash, no ptyId. */ +function emitRestoreWedge(pane: number): void { + recordRendererBreadcrumbFromRenderer({ + name: WEDGE_BREADCRUMB, + data: { paneId: pane, leafIdHash: `leaf${String(pane).padStart(5, '0')}` } + }) +} + +function wedgeCrumbs(): ReturnType { + return getCrashBreadcrumbSnapshot().filter((entry) => entry.name === WEDGE_BREADCRUMB) +} + +function wedgeSpanCount(): number { + return startSpanMock.mock.calls.filter( + (call) => call[1].attributes['breadcrumb.name'] === WEDGE_BREADCRUMB + ).length +} + +beforeEach(() => { + startSpanMock.mockClear() +}) + +afterEach(() => { + clearCrashBreadcrumbsForTest() +}) + +// One mount/reveal/wake transition expires every in-flight replay write at once, so +// the burst reaches the 30-slot ring as N distinct entries. Field span streams measure +// bursts of 26 in 0.96s and 62 over 85s. No captured report in the 09-02 corpus shows +// a ring that actually drained — all nine bursts predate their report's ring window — +// so this bounds a demonstrated hazard, not an observed loss, and must not cost the +// durable span evidence that did carry those bursts. +describe('replay-guard wedge burst against the fixed-size breadcrumb ring', () => { + it('costs one ring slot per call site and preserves the pre-crash trail', () => { + for (let index = 0; index < 10; index += 1) { + recordCrashBreadcrumb(`pre_crash_evidence_${index}`, { index }) + } + + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + const snapshot = getCrashBreadcrumbSnapshot() + expect(snapshot.filter((entry) => entry.name.startsWith('pre_crash_evidence_'))).toHaveLength( + 10 + ) + expect(wedgeCrumbs()).toHaveLength(1) + }) + + it('carries the burst multiplicity into the ring as suppressedSinceLast', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + // 26 emissions: one owns the slot, 25 fold into it. + expect(wedgeCrumbs()[0]?.data?.suppressedSinceLast).toBe(25) + }) + + // The 121-event field corpus lives entirely in the renderer.breadcrumb span stream, + // and the ring is cleared by the restart that usually precedes the crash report, so + // ring coalescing must not suppress the per-event spans. + it('still emits one durable span per wedge event', () => { + for (let pane = 0; pane < 26; pane += 1) { + emitReattachWedge(pane) + } + + expect(wedgeSpanCount()).toBe(26) + // Why no count on the span: one span per event already carries the multiplicity. + expect( + startSpanMock.mock.calls.some((call) => + JSON.stringify(call[1]).includes('suppressedSinceLast') + ) + ).toBe(false) + }) + + // Bundle 8907a508 mixes restore-path (identity-less) and reattach-path crumbs in one + // window; name-only keying would report only the last one's shape. + it('keeps restore-path and reattach-path call sites in separate slots', () => { + emitRestoreWedge(1) + emitRestoreWedge(2) + emitReattachWedge(3) + emitReattachWedge(4, true) + + const crumbs = wedgeCrumbs() + expect(crumbs).toHaveLength(3) + expect(crumbs.map((crumb) => Boolean(crumb.data?.tabIdHash))).toEqual([false, true, true]) + expect(crumbs.map((crumb) => Boolean(crumb.data?.ptyId))).toEqual([false, false, true]) + }) + + it('bounds a many-pane burst to one slot within a call site', () => { + for (let pane = 0; pane < 40; pane += 1) { + emitReattachWedge(pane, pane % 2 === 0) + } + + expect(wedgeCrumbs()).toHaveLength(2) + }) +}) From 9acfba401a91f6be5419950fe920447f3bf047f6 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 00:53:31 -0700 Subject: [PATCH 09/12] fix(crash-reporting): stop claiming kills that never landed, and leave proof when the own-Chromium pid set is unreadable (#18578) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(crash-reporting): stop the codex POSIX teardown claiming a group that was already gone terminatePosixTree's default group signal swallowed every process.kill error and then recorded a self_tree_kill unconditionally, so an ESRCH — proof the group was already gone and this teardown killed nothing — still put a suspect in the five-second render-process-gone attribution window. Every sibling group-kill in the tree already records only on a proven signal: terminateDedicatedPosixGroup in this same file, forceKillPosixPtyProcessGroups, and the claude account-login teardown. This makes the outlier match them. * fix(crash-reporting): leave proof when the own-Chromium pid set cannot be read `readOrcaChromiumProcessPids` returns an empty set when `getAppMetrics()` throws, which is the right decision — refusing every kill would orphan every PTY, git, codex and notebook tree main tears down, and on main a refusal from `killSourceControlAgentProcess` releases the managed-home lock with the agent still alive. But the empty set was byte-identical to "no Chromium on this host", so the fail-open was invisible in a field bundle. Keeps the decision, adds a coalesced durable `own_chromium_pids_unreadable` crumb so the two cases are distinguishable. Coalesced because the gate reads this set on every tree kill. * style(crash-reporting): tighten the group-signal comments to the WHY --- .../codex-app-server-process-teardown.test.ts | 62 ++++++++++++++++++- .../codex-app-server-process-teardown.ts | 29 +++++---- src/main/orca-chromium-process-pids.ts | 30 ++++++++- src/main/own-chromium-tree-kill-guard.test.ts | 34 ++++++++++ 4 files changed, 141 insertions(+), 14 deletions(-) diff --git a/src/main/codex/codex-app-server-process-teardown.test.ts b/src/main/codex/codex-app-server-process-teardown.test.ts index 1ddb672a531..cec8f91d081 100644 --- a/src/main/codex/codex-app-server-process-teardown.test.ts +++ b/src/main/codex/codex-app-server-process-teardown.test.ts @@ -1,7 +1,14 @@ import type { ChildProcess } from 'node:child_process' -import { describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + findSelfInitiatedTreeKills, + resetSelfInitiatedTreeKillLogForTest +} from '../crash-reporting/self-initiated-tree-kill-log' import { terminateCodexAppServerProcessTree } from './codex-app-server-process-teardown' +/** Above pid_max on every supported POSIX host, so the group signal is a real ESRCH. */ +const UNREACHABLE_PGID = 2_147_483_647 + function child() { return { pid: 1234, @@ -10,6 +17,10 @@ function child() { } describe('terminateCodexAppServerProcessTree', () => { + beforeEach(() => { + resetSelfInitiatedTreeKillLogForTest() + }) + it('waits for the Windows tree kill before releasing the wrapper', async () => { const target = child() const release = Promise.withResolvers() @@ -120,6 +131,55 @@ describe('terminateCodexAppServerProcessTree', () => { expect(target.kill).not.toHaveBeenCalled() }) + /** + * `selfInitiatedTreeKillCount` decides whether a `render-process-gone` was + * ours. A group that had already exited was killed by nobody, so crediting it + * puts a suspect in the five-second window that Orca never issued. Exercised + * through the real `process.kill(-pgid)` because the swallow being tested + * lives in the production default, not in an injectable seam. + */ + it('does not claim a snapshot group that was already gone', async () => { + const target = { pid: UNREACHABLE_PGID, kill: vi.fn(() => true) as ChildProcess['kill'] } + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ + rootPgid: UNREACHABLE_PGID, + descendants: [], + capturedAtMs: 1 + }), + terminateDescendants: async () => true + }) + ).resolves.toBe(true) + + expect(target.kill).toHaveBeenLastCalledWith('SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([]) + }) + + it('claims a snapshot group the signal actually reached', async () => { + const target = child() + const signalProcessGroup = vi.fn() + + await expect( + terminateCodexAppServerProcessTree(target, undefined, { + platform: 'darwin', + captureDescendants: async () => ({ rootPgid: 1234, descendants: [], capturedAtMs: 1 }), + terminateDescendants: async () => true, + signalProcessGroup + }) + ).resolves.toBe(true) + + expect(signalProcessGroup).toHaveBeenCalledWith(1234, 'SIGKILL') + expect(findSelfInitiatedTreeKills(Date.now())).toEqual([ + expect.objectContaining({ + pid: 1234, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' + }) + ]) + }) + it('tears down 40 dedicated groups without process-table scans or cross-group fanout', async () => { const killMocks = Array.from({ length: 40 }, () => vi.fn(() => true)) const targets = killMocks.map((kill, index) => ({ diff --git a/src/main/codex/codex-app-server-process-teardown.ts b/src/main/codex/codex-app-server-process-teardown.ts index a35ad4d3164..5a9c6e3574b 100644 --- a/src/main/codex/codex-app-server-process-teardown.ts +++ b/src/main/codex/codex-app-server-process-teardown.ts @@ -128,19 +128,24 @@ async function terminatePosixTree( if (descendantsExited && snapshot.rootPgid === rootPid) { const signalGroup = deps.signalProcessGroup ?? - ((pgid: number, signal: NodeJS.Signals) => { - try { - process.kill(-pgid, signal) - } catch { - // Group already exited. - } + ((pgid: number, signal: NodeJS.Signals) => process.kill(-pgid, signal)) + let groupSignalled = false + try { + signalGroup(snapshot.rootPgid, 'SIGKILL') + groupSignalled = true + } catch { + // Already-gone is still the desired outcome, but nothing here killed it, + // and a crumb for a kill we never landed is a false render-process-gone suspect. + } + if (groupSignalled) { + // Outside the try, as in terminateDedicatedPosixGroup: that catch is the + // already-gone contract, not a breadcrumb handler. + recordSelfInitiatedTreeKill({ + pid: snapshot.rootPgid, + site: 'codex-app-server-teardown', + scope: 'posix-process-group' }) - signalGroup(snapshot.rootPgid, 'SIGKILL') - recordSelfInitiatedTreeKill({ - pid: snapshot.rootPgid, - site: 'codex-app-server-teardown', - scope: 'posix-process-group' - }) + } } if (!descendantsExited) { child.kill('SIGCONT') diff --git a/src/main/orca-chromium-process-pids.ts b/src/main/orca-chromium-process-pids.ts index f22babc6921..b2613e42b79 100644 --- a/src/main/orca-chromium-process-pids.ts +++ b/src/main/orca-chromium-process-pids.ts @@ -1,4 +1,5 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' +import { recordCoalescedDurableCrashBreadcrumb } from './crash-reporting/durable-crash-breadcrumb' /** * PIDs of Orca's own Chromium processes — browser, renderers, GPU, utilities. @@ -11,6 +12,14 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment' * Empty on a Node host and empty on failure: that is "no refusal proven", never * "safe to kill" — callers must keep every other guard they already have. * + * Why failure stays open rather than refusing everything: a refusal is not free. + * `terminateWindowsProcessTree` resolves without killing, and + * `killSourceControlAgentProcess` returns that straight to a caller that then + * releases the managed-home lock, so failing closed would trade one unreadable + * metrics table for every PTY, git, codex and notebook tree in main leaking at + * once. The `own_chromium_pids_unreadable` crumb is the price of that choice: + * without it a throw is byte-identical to "no Chromium on this host". + * * Host coverage: only Electron main installs a Chromium-backed AppEnvironment * (main-process-preflight). The standalone daemon installs none and `orcad` * installs a Node one whose `getAppMetrics()` is `[]`, so this set is empty in @@ -30,7 +39,26 @@ export function readOrcaChromiumProcessPids(): ReadonlySet { .map((metric) => metric.pid) .filter((pid) => Number.isInteger(pid) && pid > 0) return new Set(pids) - } catch { + } catch (error) { + recordUnreadableOwnChromiumMetrics(error) return new Set() } } + +// Why coalesced: the gate reads this set on every tree kill, so a persistently +// broken metrics table would otherwise flood the 30-slot ring it shares. +const UNREADABLE_METRICS_COALESCE_MS = 60_000 + +function recordUnreadableOwnChromiumMetrics(error: unknown): void { + try { + recordCoalescedDurableCrashBreadcrumb({ + name: 'own_chromium_pids_unreadable', + data: { cause: error instanceof Error ? error.message : String(error) }, + coalesceKey: 'own-chromium-pids-unreadable', + minIntervalMs: UNREADABLE_METRICS_COALESCE_MS + }) + } catch { + // Diagnostics must never turn an admitted kill into a thrown one: callers + // read this set outside their own try. + } +} diff --git a/src/main/own-chromium-tree-kill-guard.test.ts b/src/main/own-chromium-tree-kill-guard.test.ts index 7e98661aca6..bd3b1674e18 100644 --- a/src/main/own-chromium-tree-kill-guard.test.ts +++ b/src/main/own-chromium-tree-kill-guard.test.ts @@ -143,6 +143,40 @@ describe('refusing to tree-kill our own Chromium processes', () => { ) }) + /** + * Fail-open is the deliberate choice — see `orca-chromium-process-pids.ts` for + * why refusing everything is worse — so the crumb is the only thing that keeps + * an unreadable metrics table distinguishable from a host that has no Chromium. + */ + it('leaves proof, and still admits the kill, when the Chromium metrics cannot be read', () => { + appMetricsMock.mockImplementation(() => { + throw new Error('getAppMetrics unavailable') + }) + + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + // Coalesced: the gate reads this set on every kill, so a broken table must + // not evict the ring it shares with the refusal crumb. + expect([...readOrcaChromiumProcessPids()]).toEqual([]) + expect( + admitSelfInitiatedTreeKill({ + pid: RENDERER_PID, + site: 'pty-descendant-sweep', + scope: 'win-taskkill-tree' + }) + ).toBe(true) + + expect( + getCrashBreadcrumbSnapshot().filter( + (breadcrumb) => breadcrumb.name === 'own_chromium_pids_unreadable' + ) + ).toEqual([ + expect.objectContaining({ + name: 'own_chromium_pids_unreadable', + data: expect.objectContaining({ cause: 'getAppMetrics unavailable' }) + }) + ]) + }) + it('refuses an own-Chromium pid at the gate the account teardowns share', () => { expect( admitSelfInitiatedTreeKill({ From 7a714d1bd2750789cd26ad213f9ae2f129807eef Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 00:53:34 -0700 Subject: [PATCH 10/12] fix(terminals): add equality bailouts to the tab pane-expansion actions (#18332) * fix(terminals): bail out of no-op pane-expansion store writes * test(terminals): lock the root-state identity of the bailout A `return {}` bailout keeps the map reference but still allocates a new root state, so zustand walks every listener. Assert root identity too. --- .../store/terminals/terminal-layout-state.ts | 17 +- ...inal-pane-expansion-write-bailout.test.tsx | 152 ++++++++++++++++++ 2 files changed, 163 insertions(+), 6 deletions(-) create mode 100644 src/renderer/src/store/terminals/terminal-pane-expansion-write-bailout.test.tsx diff --git a/src/renderer/src/store/terminals/terminal-layout-state.ts b/src/renderer/src/store/terminals/terminal-layout-state.ts index d95439cf5a8..9b5cf8323c2 100644 --- a/src/renderer/src/store/terminals/terminal-layout-state.ts +++ b/src/renderer/src/store/terminals/terminal-layout-state.ts @@ -43,15 +43,20 @@ export function createTerminalLayoutActions( } }) }, + // Why: pane mount/unmount re-asserts the same booleans; bailing like setTabLayout keeps map subscribers asleep. setTabPaneExpanded: (tabId, expanded) => { - set((s) => ({ - expandedPaneByTabId: { ...s.expandedPaneByTabId, [tabId]: expanded } - })) + set((s) => + s.expandedPaneByTabId[tabId] === expanded + ? s + : { expandedPaneByTabId: { ...s.expandedPaneByTabId, [tabId]: expanded } } + ) }, setTabCanExpandPane: (tabId, canExpand) => { - set((s) => ({ - canExpandPaneByTabId: { ...s.canExpandPaneByTabId, [tabId]: canExpand } - })) + set((s) => + s.canExpandPaneByTabId[tabId] === canExpand + ? s + : { canExpandPaneByTabId: { ...s.canExpandPaneByTabId, [tabId]: canExpand } } + ) }, setTabLayout: (tabId, layout) => { let ownershipTransfers: ReturnType = [] diff --git a/src/renderer/src/store/terminals/terminal-pane-expansion-write-bailout.test.tsx b/src/renderer/src/store/terminals/terminal-pane-expansion-write-bailout.test.tsx new file mode 100644 index 00000000000..d0d32f7742f --- /dev/null +++ b/src/renderer/src/store/terminals/terminal-pane-expansion-write-bailout.test.tsx @@ -0,0 +1,152 @@ +// @vitest-environment happy-dom + +import { Profiler } from 'react' +import { act, cleanup, render } from '@testing-library/react' +import { afterEach, describe, expect, it } from 'vitest' +import { createTestStore } from '../slices/store-test-helpers' + +afterEach(cleanup) + +type TestStore = ReturnType + +const TAB_ID = 'tab-1' +const NO_OP_WRITES = 25 + +function recordPublishedMapKeys(store: TestStore): string[] { + const published: string[] = [] + store.subscribe((next, previous) => { + if (next.expandedPaneByTabId !== previous.expandedPaneByTabId) { + published.push('expandedPaneByTabId') + } + if (next.canExpandPaneByTabId !== previous.canExpandPaneByTabId) { + published.push('canExpandPaneByTabId') + } + }) + return published +} + +// Mirrors use-terminal-workspace-store-bindings.ts:17, which subscribes to the raw map. +function ExpandedPaneSubscriber({ store }: { store: TestStore }): React.JSX.Element { + const expandedPaneByTabId = store((s) => s.expandedPaneByTabId) + return {String(expandedPaneByTabId[TAB_ID] === true)} +} + +function CanExpandPaneSubscriber({ store }: { store: TestStore }): React.JSX.Element { + const canExpandPaneByTabId = store((s) => s.canExpandPaneByTabId) + return {String(canExpandPaneByTabId[TAB_ID] === true)} +} + +function renderCommitCounter(subscriber: React.JSX.Element): () => number { + let commits = 0 + render( + { + commits += 1 + }} + > + {subscriber} + + ) + const mountCommits = commits + return () => commits - mountCommits +} + +describe('setTabPaneExpanded', () => { + it('publishes the first write for an unseen tab and a real toggle', () => { + const store = createTestStore() + const published = recordPublishedMapKeys(store) + + store.getState().setTabPaneExpanded(TAB_ID, false) + expect(published).toEqual(['expandedPaneByTabId']) + expect(store.getState().expandedPaneByTabId[TAB_ID]).toBe(false) + + store.getState().setTabPaneExpanded(TAB_ID, true) + expect(published).toEqual(['expandedPaneByTabId', 'expandedPaneByTabId']) + expect(store.getState().expandedPaneByTabId[TAB_ID]).toBe(true) + }) + + it('bails out when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabPaneExpanded(TAB_ID, false) + const before = store.getState().expandedPaneByTabId + // Root identity too: returning `{}` keeps the map but allocates a new root, so zustand still walks every listener. + const rootBefore = store.getState() + const published = recordPublishedMapKeys(store) + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + store.getState().setTabPaneExpanded(TAB_ID, false) + } + + expect(published).toEqual([]) + expect(store.getState().expandedPaneByTabId).toBe(before) + expect(store.getState()).toBe(rootBefore) + }) + + it('costs no React commit in a map subscriber when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabPaneExpanded(TAB_ID, false) + const commitsSinceMount = renderCommitCounter() + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + act(() => { + store.getState().setTabPaneExpanded(TAB_ID, false) + }) + } + expect(commitsSinceMount()).toBe(0) + + act(() => { + store.getState().setTabPaneExpanded(TAB_ID, true) + }) + expect(commitsSinceMount()).toBe(1) + }) +}) + +describe('setTabCanExpandPane', () => { + it('publishes the first write for an unseen tab and a real toggle', () => { + const store = createTestStore() + const published = recordPublishedMapKeys(store) + + store.getState().setTabCanExpandPane(TAB_ID, false) + expect(published).toEqual(['canExpandPaneByTabId']) + expect(store.getState().canExpandPaneByTabId[TAB_ID]).toBe(false) + + store.getState().setTabCanExpandPane(TAB_ID, true) + expect(published).toEqual(['canExpandPaneByTabId', 'canExpandPaneByTabId']) + expect(store.getState().canExpandPaneByTabId[TAB_ID]).toBe(true) + }) + + it('bails out when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabCanExpandPane(TAB_ID, false) + const before = store.getState().canExpandPaneByTabId + const rootBefore = store.getState() + const published = recordPublishedMapKeys(store) + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + store.getState().setTabCanExpandPane(TAB_ID, false) + } + + expect(published).toEqual([]) + expect(store.getState().canExpandPaneByTabId).toBe(before) + expect(store.getState()).toBe(rootBefore) + }) + + it('costs no React commit in a map subscriber when the value is unchanged', () => { + const store = createTestStore() + store.getState().setTabCanExpandPane(TAB_ID, false) + const commitsSinceMount = renderCommitCounter() + + for (let i = 0; i < NO_OP_WRITES; i += 1) { + act(() => { + store.getState().setTabCanExpandPane(TAB_ID, false) + }) + } + expect(commitsSinceMount()).toBe(0) + + act(() => { + store.getState().setTabCanExpandPane(TAB_ID, true) + }) + expect(commitsSinceMount()).toBe(1) + }) +}) From cc9e9ed65fc3d4f69d79224eb66c437fb819528d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 00:53:37 -0700 Subject: [PATCH 11/12] fix(crash-reporting): sample system memory before the process is gone (#18356) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(crash-reporting): sample system memory before the renderer dies * fix(crash-reporting): make the pre-gone host sample decisive, not just present Round-1 review said the shipped field set could not decide G4-oom. Fixed. Decisive field (blocking #1). The investigation's own win-lowspec repro falsified "low available commit kills": at a 127 MB commit floor Windows grew the pagefile to 2029 MB and nothing died, and it named the missing datum — pagefile-growth headroom / system-drive free space. `getSystemMemoryInfo()` gives neither. Added `swap-volume-free-space.ts`: one `fs.statfs` on the volume backing the pagefile (SystemRoot on Windows, the root fs elsewhere, resolved via `path.parse().root`), published as `systemMemoryPreGoneSwapVolumeFreeMB`. Together with the already-emitted commit limit that separates "commit was low" from "commit was refused". Pagefile *max* size needs a registry read; skipped deliberately — per-operation interpreter spawning is exactly what docs/reference/windows-edr-posture.md says not to add for telemetry. Darwin honesty (blocking #2). Every reading now carries `systemMemoryPressureSignal`: `available-commit` on Windows (swapFree is ullAvailPageFile), `mem-available` on Linux when MemAvailable is present, `none` otherwise — which is always on darwin. A future analyst cannot now table `freeMB: 272` from a healthy Mac as evidence of exhaustion, because the same record says the platform gave no pressure verdict. Partial rebuttal on the suggested reuse: `host-memory.ts:86` was considered and rejected as a periodic source. It spawns `/usr/bin/memory_pressure` per call, and the sampler this PR needs runs every 10 s for the app's lifetime; a subprocess at that cadence is worse than the gap it closes. The reviewer conceded this tradeoff is arguable — what was not acceptable was shipping the darwin gap silently, so it is now in the data, not only in a comment. Staleness (blocking #3). Confirmed the measurement: four of five G4 reports carried a ~37 s-old sample (4872/36796/37332/38017/39715 ms). Host memory no longer rides the 60 s process-metrics sweep; `pre-gone-host-memory.ts` samples it on its own 10 s timer with its own `systemMemoryPreGoneSampleAgeMs`. One GlobalMemoryStatusEx-class call plus one statfs is cheap enough at that rate. A refusal shorter than the interval stays invisible and the module comment says so — no polling cadence fixes that. Non-blocking, all taken: renamed `gone-time-system-memory.ts` -> `system-memory-details.ts` with the now-false "reads AFTER the crash" framing scoped to the gone-time caller; pre-gone host keys moved out of the `processMetrics` namespace to `systemMemoryPreGone*`, so the string-surgery `preGoneDetailKey` helper is gone and a `systemMemory` prefix scan sees both reads; the bare catch no longer spans both halves of the sample, and a test pins that a throwing host read leaves the process-metric sample intact; the inert second test is replaced by three that go red without this change (verified: swap-volume, pressure-signal and cadence assertions all fail when the production hunks are reverted). Rebuttal, non-blocking #5 (duplicated electron mock across two test files): declined. `vi.mock` is hoisted per file, so the mock cannot be shared without a setup module, and this directory already has 26 focused test files that each re-declare it. Splitting by concern is the local convention. `startPreGoneProcessMetricsSampling` is renamed `startPreGoneCrashSampling` since it now starts two samplers. * fix(crash-reporting): test the arming, gate the swap volume, unblock the host read Round-2 review blocked on four items. All four addressed. WHAT THIS BRANCH ACTUALLY DOES, AT HEAD (blocking #4). The commit-1 message ("13 lines, 1 production file, no new module, new optional numeric fields only", `processMetricsPreGoneSystemMemory*` keys, a `preGoneDetailKey` helper, a `pre-gone-system-memory.test.ts`) describes a superseded revision; every one of those claims is false now, so it must not be used as the PR description. The change against origin/main is: 3 new production modules (`pre-gone-host-memory.ts`, `system-memory-details.ts`, `swap-volume-free-space.ts`), 1 deleted (`gone-time-system-memory.ts`), plus edits to `process-gone-diagnostics.ts` and `main-process-ready-runtime.ts` and 2 test files. It adds a second main-process interval timer that runs for the life of the app: every 10 s one synchronous GlobalMemoryStatusEx-class read, and on win32/darwin one `fs.statfs` on the swap-backing volume. Details are `systemMemoryPreGone*`, and two of them are STRINGS, not numbers: `systemMemoryPreGonePressureSignal` (enum) and `systemMemoryPreGoneSwapVolume` (a drive label, separator-trimmed so it is not a path). Both are assigned after `sanitizeCrashReportDetails`; neither carries user content. Arming is now tested (blocking #1). The reviewer deleted `startPreGoneSystemMemorySampling(...)` from `startPreGoneCrashSampling` and all 264 tests stayed green — confirmed and fixed. `pre-gone-host-memory.test.ts` now calls `startPreGoneCrashSampling()` with production defaults and asserts both `setInterval` calls, their literal periods `[60_000, 10_000]`, that both timers are unref'd, and that advancing 10 s takes a fresh host sample that reaches `buildProcessGoneCrashDetails` with `SampleAgeMs: 0`. Verified red on revert: deleting the arming line -> 1 failure; changing the interval constant to 30_000 -> 1 failure (the old assertion compared the constant to itself and caught neither). The tautological `10_000 < 60_000 / 2` test is gone, superseded by this one. Swap volume is win32/darwin only (blocking #2). On Linux swap is a fixed partition, a fixed-size swapfile, or zram; none grow into root-fs free space, so `SwapVolumeFreeMB: 380000` beside `SwapFreeMB: 0` would have invited exactly the wrong verdict on the two Linux cluster members. `swapVolumeAnchor` returns undefined off win32/darwin, so no field and no statfs at all. The comment claiming "elsewhere swap is on the root fs" was wrong and is gone. The Windows anchor is still the DEFAULT pagefile volume, so the measured volume now ships with the number (`systemMemoryPreGoneSwapVolume: 'C:'`) instead of being implied. The honesty label covers it: win32 reads `available-commit` only when the volume datum is present, and `available-commit-unqualified` otherwise — which also fixes non-blocking #5, where the synchronous gone-time read claimed a verdict its own fields could not support. Host read no longer waits on statfs (blocking #3). `samplePreGoneSystemMemory` now commits the synchronous memory reading first and merges volume free space in afterwards, so the cadence is 10 s regardless of disk-metadata latency and a hung volume can no longer stop host sampling — precisely the paging-storm case this exists for. The in-flight latch now guards only the statfs. Verified red on revert to the serialized shape (2 failures). A stale-but-slow-moving volume value merging into a newer memory sample is deliberate and commented. Non-blocking #3 (reset does not invalidate an in-flight sample): fixed with a generation counter bumped by `resetPreGoneSystemMemorySamplingForTest`, so a late statfs cannot repopulate a reset sample. Separately, the volume read now only runs after a host sample committed, which removes the real `statfs('/')` side effect from `process-gone-diagnostics.test.ts` entirely. REBUTTAL, darwin `memory_pressure` reuse (non-blocking #2): declined, with evidence. `readDarwinAvailableMemory` at src/main/memory/host-memory.ts:87 is reached only via `collectHostMemory` <- `runSnapshot` <- `collectMemorySnapshot`, whose only callers are the `memory:getSnapshot` IPC handler and orca-runtime-pty-foreground-process-reads.ts:170 — both on demand. There is no periodic snapshot, so there is no cached reading to reuse for free; adopting it means spawning `/usr/bin/memory_pressure` on a main-process timer for the life of the app, and its module-global `darwinAvailabilitySupported` latch is shared with the memory UI. The gap is not hidden: darwin ships `PressureSignal: 'none'` in the data, and the module comment now cites the existing reader and why it is not used here rather than claiming Orca lacks one. Verified: `vitest src/main/crash-reporting src/main/startup src/main/memory` = 769 passed / 6 skipped (crash-reporting re-run 5x, no flake); `tsc --noEmit -p config/tsconfig.node.json` 0; `oxlint` 0; `oxfmt --check` 0. * fix(crash-reporting): stop a stale statfs qualifying the commit verdict Round-3 adversarial review, 2 blocking. Both fixed with mutation-verified tests. 1. `mergeSwapVolumeFreeSpace` merged the volume reading into whatever sample was current at RESOLUTION time, and `pressureSignal` then upgraded win32 from `available-commit-unqualified` to the decisive `available-commit` on the strength of it. The `swapVolumeReadInFlight` latch makes every intervening tick skip the merge, so the lag is as old as the last STARTED statfs, not the last tick — and no age field exposed it, because `systemMemoryPreGoneSampleAgeMs` describes only the synchronous memory read. Reviewer's executed scenario: a statfs issued at t=0 on a healthy host (40 GB free) resolving at t=20 s of commit pressure emitted `SwapFreeMB: 200` beside `SwapVolumeFreeMB: 40000`, labelled `available-commit`, with `SampleAgeMs: 0`. That reads as "the pagefile had room, so this was not a commit refusal" — the opposite conclusion, wearing the branch's highest-confidence label, on exactly the win32 G4-oom reports this exists to decide. The datum still ships (it is the only pagefile-expandability signal there is), but now: - the sample carries `swapVolumeSampledAtMs` — the tick that ISSUED the statfs, never the one it resolved on — surfaced as `systemMemoryPreGoneSwapVolumeAgeMs`; - only a statfs that answers on its own tick may qualify the verdict. `withSwapVolumeFreeSpace` takes `coTimed`; false keeps `available-commit-unqualified`. The next tick issues a fresh statfs, so the verdict recovers on its own. 2. The branch's sole production entry point — `startPreGoneCrashSampling()` at main-process-ready-runtime.ts:128 — was untested. Deleting it left 691 tests across crash-reporting/ and startup/ green, while a comment in the new test file claimed that gap was why the test was written. This is pure instrumentation, so that one line is the whole of its value in the shipped app. Added a source-level wiring test (the pattern this repo already uses for arm-once ready-phase lines) that pins the import, exactly one call, the call at statement indent, and that `main-process-ready.ts` awaits the function it lives in. The misleading comment is gone. Mutation-verified — each goes red alone: coTimed -> always true 1 failed (verdict) drop swapVolumeSampledAtMs age 1 failed (verdict test) delete startPreGoneCrashSampling() 1 failed (wiring) wrap it in `if (!is.dev) { ... }` 1 failed (wiring) Verified: tsc -p config/tsconfig.node.json exit 0; oxlint src/main/crash-reporting src/main/startup exit 0; 268 tests in crash-reporting/ pass. Across crash-reporting/ + startup/ + memory/: 769 passed, 2 failed — both environment-dependent and failing identically on the unmodified tree (Xvfb rebind, and a whole-repo glob census that times out). * fix(crash-reporting): stop free disk standing in for pagefile growability The win32 reading was promoted to the decisive `available-commit` whenever a co-timed volume number merely existed, which the data cannot support: a fixed or disabled pagefile grows into no amount of empty disk, its maximum is unreadable here, and the measured volume is only the DEFAULT pagefile drive. A host with 180 MB of available commit, a commit limit at RAM and 812 GB free read as "the pagefile had room" — the opposite conclusion, under the branch's most confident label. The volume datum is now named for what it is (`available-commit-volume-cotimed`, context beside the commit number), and the one decisive win32 case — a commit limit at or below RAM, i.e. no pagefile behind it — gets its own label. Also: carry the last volume reading onto the sample that replaces it, aged and non-qualifying, so a statfs slower than one tick no longer makes the field vanish from the reports it exists for; don't commit a reading whose every memory field failed, which shipped an age and a disk-free number with no host memory beside them; and move the startup wiring test beside the file it pins, scoped to the ready-phase entry's own body so the call cannot satisfy it from a sibling export nothing calls. * fix(crash-reporting): co-time the statfs by tick, not sample identity A tick whose host read fails leaves the pre-gone sample object in place, so the identity check still read a 25 s-late statfs as co-timed. --- .../gone-time-system-memory.ts | 77 ---- .../pre-gone-host-memory.test.ts | 379 ++++++++++++++++++ .../crash-reporting/pre-gone-host-memory.ts | 164 ++++++++ .../process-gone-diagnostics.test.ts | 22 +- .../process-gone-diagnostics.ts | 23 +- .../crash-reporting/swap-volume-free-space.ts | 67 ++++ .../crash-reporting/system-memory-details.ts | 161 ++++++++ .../startup/main-process-ready-runtime.ts | 9 +- .../pre-gone-crash-sampling-wiring.test.ts | 49 +++ 9 files changed, 854 insertions(+), 97 deletions(-) delete mode 100644 src/main/crash-reporting/gone-time-system-memory.ts create mode 100644 src/main/crash-reporting/pre-gone-host-memory.test.ts create mode 100644 src/main/crash-reporting/pre-gone-host-memory.ts create mode 100644 src/main/crash-reporting/swap-volume-free-space.ts create mode 100644 src/main/crash-reporting/system-memory-details.ts create mode 100644 src/main/startup/pre-gone-crash-sampling-wiring.test.ts diff --git a/src/main/crash-reporting/gone-time-system-memory.ts b/src/main/crash-reporting/gone-time-system-memory.ts deleted file mode 100644 index 7cca89d2d4b..00000000000 --- a/src/main/crash-reporting/gone-time-system-memory.ts +++ /dev/null @@ -1,77 +0,0 @@ -import type { CrashReportDetailValue } from '../../shared/crash-reporting' - -// ─── System memory at gone time ───────────────────────────────────── -// Why: the system outlives the crashed process, so this IS sampleable at -// process-gone — it separates "renderer grew huge" from "machine out of -// memory/commit", which the per-process buckets alone cannot. -// Timing honesty: this reads AFTER the crashed process's memory returned to -// the OS, so free/swapFree can look healthier than they were at kill time. -// Platform honesty: swap* exist on Windows/Linux only. On Linux `free` is -// /proc/meminfo MemFree and is NOT the pressure signal — it excludes page cache -// and other reclaimable memory; `available` (MemAvailable, Linux-only) is. On -// macOS `free` is near-meaningless (file cache and compression keep it low on -// healthy machines); fileBacked/purgeable are the only reclaimability proxy this -// API gives there, and none of these fields answers "was the machine under -// pressure" on macOS — that needs a signal Electron does not expose. - -type CrashReportDetails = Record - -export function memoryKBFieldMB(value: unknown): number | undefined { - const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined - return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) -} - -type SystemMemoryInfoLike = { - total?: unknown - free?: unknown - available?: unknown - swapTotal?: unknown - swapFree?: unknown - fileBacked?: unknown - purgeable?: unknown -} - -type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null - -function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { - const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) - .getSystemMemoryInfo - if (typeof read !== 'function') { - return null - } - try { - return read.call(process) - } catch { - return null - } -} - -let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo - -export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { - systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo -} - -export function getSystemMemoryAtGoneDetails(): CrashReportDetails { - const info = systemMemoryInfoReader() - if (!info) { - return {} - } - const details: CrashReportDetails = {} - const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ - ['total', 'systemMemoryTotalMB'], - ['free', 'systemMemoryFreeMB'], - ['available', 'systemMemoryAvailableMB'], - ['swapTotal', 'systemMemorySwapTotalMB'], - ['swapFree', 'systemMemorySwapFreeMB'], - ['fileBacked', 'systemMemoryFileBackedMB'], - ['purgeable', 'systemMemoryPurgeableMB'] - ] - for (const [field, key] of fields) { - const mb = memoryKBFieldMB(info[field]) - if (mb !== undefined) { - details[key] = mb - } - } - return details -} diff --git a/src/main/crash-reporting/pre-gone-host-memory.test.ts b/src/main/crash-reporting/pre-gone-host-memory.test.ts new file mode 100644 index 00000000000..0df13d4fee5 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.test.ts @@ -0,0 +1,379 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + getSystemMemoryDetails, + setSystemMemoryInfoReaderForTest, + withSwapVolumeFreeSpace +} from './system-memory-details' +import { + readSwapVolumeFreeSpace, + setSwapVolumeFreeSpaceReaderForTest, + type SwapVolumeFreeSpace +} from './swap-volume-free-space' +import { samplePreGoneSystemMemory } from './pre-gone-host-memory' +import { + buildProcessGoneCrashDetails, + resetPreGoneCrashSamplingForTest, + samplePreGoneProcessMetrics, + startPreGoneCrashSampling +} from './process-gone-diagnostics' + +type MetricFixture = { + pid: number + creationTime: number + type: string + memory: { workingSetSize: number; peakWorkingSetSize?: number; privateBytes?: number } +} + +const { appMetricsMock } = vi.hoisted(() => ({ + appMetricsMock: vi.fn<() => MetricFixture[]>(() => []) +})) + +vi.mock('electron', () => ({ app: { getAppMetrics: appMetricsMock } })) + +const BROWSER_AND_RENDERER: MetricFixture[] = [ + { pid: 10, creationTime: 1, type: 'Browser', memory: { workingSetSize: 1024 * 250 } }, + { + pid: 11, + creationTime: 2, + type: 'Tab', + memory: { workingSetSize: 1024 * 400, peakWorkingSetSize: 1024 * 420, privateBytes: 1024 * 260 } + } +] + +const BROWSER_ONLY: MetricFixture[] = [BROWSER_AND_RENDERER[0]] + +const UNDER_COMMIT_PRESSURE = { + total: 16_000 * 1024, + free: 400 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 200 * 1024 +} + +const AFTER_THE_CORPSE_RELEASED = { + total: 16_000 * 1024, + free: 3_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 2_900 * 1024 +} + +// Commit limit ~= RAM: a disabled or fixed pagefile, which no amount of empty +// disk can grow into. `swapTotal > total` is all this API can say about that. +const FIXED_PAGEFILE_UNDER_PRESSURE = { + total: 16_000 * 1024, + free: 300 * 1024, + swapTotal: 16_100 * 1024, + swapFree: 180 * 1024 +} + +const NO_PAGEFILE_UNDER_PRESSURE = { + ...FIXED_PAGEFILE_UNDER_PRESSURE, + swapTotal: 15_900 * 1024 +} + +const BEFORE_THE_STORM = { + total: 16_000 * 1024, + free: 9_000 * 1024, + swapTotal: 48_000 * 1024, + swapFree: 30_000 * 1024 +} + +describe('pre-gone host memory', () => { + beforeEach(() => { + resetPreGoneCrashSamplingForTest() + setSystemMemoryInfoReaderForTest(null) + setSwapVolumeFreeSpaceReaderForTest(null) + appMetricsMock.mockClear() + appMetricsMock.mockReturnValue(BROWSER_AND_RENDERER) + }) + + it('carries a pre-gone host reading, not only the post-mortem one', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + // The renderer dies; its ~400 MB returns to the OS, so the gone-time read + // now shows a much healthier machine than the one that refused the alloc. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + appMetricsMock.mockReturnValue(BROWSER_ONLY) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemorySwapFreeMB).toBe(2_900) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(200) + expect(details.systemMemoryPreGoneFreeMB).toBe(400) + expect(details.systemMemoryPreGoneTotalMB).toBe(16_000) + // Why: host memory keeps its own key family, so a `systemMemory` prefix scan sees both reads. + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) + }) + + // Why this decides the cluster: 200 MB available commit is only a REFUSAL when + // the pagefile cannot grow, which is what the volume's free space says. + it('reports swap-volume free space so low commit can be told from refused commit', async () => { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + await samplePreGoneSystemMemory(Date.now() - 5_000) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(120) + // Which volume was measured: Windows only names the DEFAULT pagefile drive. + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + }) + + it('omits swap-volume free space on Linux, where swap cannot grow into free disk', async () => { + // Linux swap is a fixed partition, a fixed-size swapfile, or zram; reporting + // root-fs free space next to SwapFreeMB 0 would read as headroom that is not there. + setSwapVolumeFreeSpaceReaderForTest(null) + + await expect(readSwapVolumeFreeSpace('linux')).resolves.toBeUndefined() + }) + + it('labels the reading with the pressure verdict the platform can actually give', () => { + // Windows available commit is only a REFUSAL when the pagefile cannot grow, + // which nothing here proves, so no label may read as that verdict. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + const windowsCommit = getSystemMemoryDetails('win32') + expect(windowsCommit.systemMemoryPressureSignal).toBe('available-commit-unqualified') + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32') + .systemMemoryPressureSignal + ).toBe('available-commit-volume-cotimed') + // A volume number from a different moment describes a different machine. + expect( + withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32', false) + .systemMemoryPressureSignal + ).toBe('available-commit-unqualified') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, free: 400 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('none') + + setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, available: 900 * 1024 })) + expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('mem-available') + + // darwin free/fileBacked/purgeable answer reclaimability, never pressure. + setSystemMemoryInfoReaderForTest(() => ({ + total: 16_000 * 1024, + free: 272 * 1024, + fileBacked: 2_694 * 1024, + purgeable: 0 + })) + expect(getSystemMemoryDetails('darwin').systemMemoryPressureSignal).toBe('none') + }) + + // Why this and not the volume number: the branch's own repro needed a pagefile + // that CANNOT grow to kill anything, and neither the pagefile maximum nor its + // drive is readable here — `swapVolumeAnchor` measures SystemRoot's volume, + // which a relocated pagefile does not live on. + it('never reads free disk as proof the pagefile could have grown', () => { + setSystemMemoryInfoReaderForTest(() => FIXED_PAGEFILE_UNDER_PRESSURE) + const fixedPagefile = withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ) + // 180 MB of commit beside 812 GB of free disk: co-timed, and still not a + // verdict — reading it as "the pagefile had room" is the opposite conclusion. + expect(fixedPagefile.systemMemoryPressureSignal).toBe('available-commit-volume-cotimed') + + // The one decisive win32 case: commit limit at or below RAM means there is + // no pagefile behind it, so the floor cannot heal however empty the disk is. + setSystemMemoryInfoReaderForTest(() => NO_PAGEFILE_UNDER_PRESSURE) + expect( + withSwapVolumeFreeSpace( + getSystemMemoryDetails('win32'), + { freeMB: 812_000, volume: 'C:' }, + 'win32' + ).systemMemoryPressureSignal + ).toBe('available-commit-hard-capped') + }) + + // Why the verdict and not just the field: a statfs issued on a healthy host at + // t=0 that resolves 20 s into a commit storm prints "200 MB commit, 40 GB of + // pagefile headroom" — which reads as NOT a commit refusal, the opposite + // conclusion, under the branch's most confident label. + it('will not let a statfs that outlived its tick qualify the win32 commit verdict', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // The storm arrives; the in-flight latch makes every tick skip the merge, + // so the pending statfs is as old as the tick that STARTED it. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(20_000) + const stale = buildProcessGoneCrashDetails({}, 'renderer') + expect(stale.systemMemoryPreGoneSwapFreeMB).toBe(200) + // The pre-storm volume number still ships — but carrying its own age, and + // without promoting the verdict the analyst reads. + expect(stale.systemMemoryPreGoneSwapVolumeFreeMB).toBe(40_000) + expect(stale.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(stale.systemMemoryPreGoneSwapVolumeAgeMs).toBe(20_000) + expect(stale.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + + // The next tick's statfs answers on its own tick, so it qualifies again. + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 900, volume: 'C:' })) + await samplePreGoneSystemMemory(30_000) + vi.setSystemTime(30_000) + const fresh = buildProcessGoneCrashDetails({}, 'renderer') + expect(fresh.systemMemoryPreGoneSwapVolumeFreeMB).toBe(900) + expect(fresh.systemMemoryPreGoneSwapVolumeAgeMs).toBe(0) + expect(fresh.systemMemoryPreGonePressureSignal).toBe('available-commit-volume-cotimed') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + // Round 5: sample identity alone could not see these ticks. A host read that + // returns nothing leaves the sample object in place, so `sample === issuedFor` + // still held 25 s and two ticks later and the statfs re-qualified the verdict. + it('will not let ticks with a failed host read pass a stale statfs off as co-timed', async () => { + const platform = Object.getOwnPropertyDescriptor(process, 'platform')! + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + vi.useFakeTimers() + let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {} + try { + setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM) + setSwapVolumeFreeSpaceReaderForTest( + () => + new Promise((resolve) => { + resolveVolume = resolve + }) + ) + void samplePreGoneSystemMemory(0) + + // GlobalMemoryStatusEx starts failing: the sample is neither replaced nor erased. + setSystemMemoryInfoReaderForTest(() => null) + await samplePreGoneSystemMemory(10_000) + await samplePreGoneSystemMemory(20_000) + + resolveVolume({ freeMB: 40_000, volume: 'C:' }) + await vi.advanceTimersByTimeAsync(0) + + vi.setSystemTime(25_000) + const details = buildProcessGoneCrashDetails({}, 'renderer') + // 25 s of lag: the label must not say co-timed beside that age. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(25_000) + expect(details.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified') + } finally { + vi.useRealTimers() + Object.defineProperty(process, 'platform', platform) + } + }) + + it("arms the host sampler on its own unref'd 10 s timer, not the metric sweep's", async () => { + vi.useFakeTimers() + vi.setSystemTime(0) + const readHostMemory = vi.fn(() => UNDER_COMMIT_PRESSURE) + setSystemMemoryInfoReaderForTest(readHostMemory) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' })) + const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') + try { + startPreGoneCrashSampling() + + // Literal millisecond values: asserting the constants against themselves + // would let a cadence regression through, and 37 s of staleness is the bug. + expect(setIntervalSpy.mock.calls.map(([, ms]) => ms)).toEqual([60_000, 10_000]) + for (const { value } of setIntervalSpy.mock.results) { + expect((value as NodeJS.Timeout).hasRef()).toBe(false) + } + expect(readHostMemory).toHaveBeenCalledTimes(1) + + readHostMemory.mockReturnValue(AFTER_THE_CORPSE_RELEASED) + await vi.advanceTimersByTimeAsync(10_000) + // One host tick, no extra metric sweep: the two samplers run independently. + expect(readHostMemory).toHaveBeenCalledTimes(2) + expect(appMetricsMock).toHaveBeenCalledTimes(1) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + expect(details.systemMemoryPreGoneSwapFreeMB).toBe(2_900) + } finally { + setIntervalSpy.mockRestore() + vi.useRealTimers() + } + }) + + it('commits the host reading without waiting on the swap-volume statfs', async () => { + // Why: statfs is slowest during the paging storm this sampler targets, and + // a hung volume must not stall or silently skip host sampling. + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + + void samplePreGoneSystemMemory(Date.now() - 5_000) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(200) + + // A second tick still refreshes the reading while that statfs hangs. + setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED) + void samplePreGoneSystemMemory(Date.now()) + expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(2_900) + }) + + it('publishes no pre-gone host keys when every memory field failed to read', async () => { + // Why not "no keys at all": the reading always carries its signal label, so a + // committed empty one would ship an age and a volume number with no memory + // numbers beside them — a disk-free figure standing in for a host reading. + setSystemMemoryInfoReaderForTest(() => ({ total: Number.NaN, free: undefined })) + await samplePreGoneSystemMemory(Date.now()) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) + + it('carries the last volume reading forward, aged, instead of dropping it', async () => { + vi.useFakeTimers() + try { + setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE) + setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 42, volume: 'C:' })) + await samplePreGoneSystemMemory(0) + + // The next tick's statfs hangs — during the paging storm this targets, that + // is the normal case — so the tick has no volume reading of its own, and + // the sample that replaces the last one would otherwise drop the field. + setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {})) + void samplePreGoneSystemMemory(10_000) + vi.setSystemTime(10_000) + + const details = buildProcessGoneCrashDetails({}, 'renderer') + expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(42) + expect(details.systemMemoryPreGoneSwapVolume).toBe('C:') + expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0) + // Carried, not re-read: it ships at its real age, never as a fresh number. + expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(10_000) + } finally { + vi.useRealTimers() + } + }) + + it('keeps a failed host read from erasing the process-metric sample', async () => { + samplePreGoneProcessMetrics(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(() => { + throw new Error('getSystemMemoryInfo unavailable') + }) + await samplePreGoneSystemMemory(Date.now() - 5_000) + setSystemMemoryInfoReaderForTest(null) + + const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer') + + expect(details.processMetricsPreGoneRendererWorkingSetMB).toBe(400) + expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([]) + }) +}) diff --git a/src/main/crash-reporting/pre-gone-host-memory.ts b/src/main/crash-reporting/pre-gone-host-memory.ts new file mode 100644 index 00000000000..0db56796750 --- /dev/null +++ b/src/main/crash-reporting/pre-gone-host-memory.ts @@ -0,0 +1,164 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import { readSwapVolumeFreeSpace } from './swap-volume-free-space' +import { + getSystemMemoryDetails, + SYSTEM_MEMORY_KEY_PREFIX, + withSwapVolumeFreeSpace +} from './system-memory-details' + +// ─── Pre-gone host memory sampling ────────────────────────────────── +// Why sample at all: the gone-time host read lands after the corpse released +// its pages, so it reports a healthier machine than the one that refused the +// allocation. +// Why 10 s and not the 60 s process-metrics cadence: at 60 s, four of five +// field OOMs carried a ~37 s old host reading — far too stale to see a +// transient commit refusal. A refusal shorter than the interval stays +// invisible; no cadence fixes that. + +export const PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS = 10_000 + +type CrashReportDetails = Record + +type PreGoneSystemMemorySample = { + details: CrashReportDetails + sampledAtMs: number + /** Tick that ISSUED the statfs now merged in — never the tick it resolved on. */ + swapVolumeSampledAtMs?: number +} + +let preGoneSample: PreGoneSystemMemorySample | null = null +let preGoneTimer: ReturnType | null = null +let swapVolumeReadInFlight = false +let samplingGeneration = 0 +let sampleTick = 0 + +const PRESSURE_SIGNAL_KEY = `${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal` + +/** + * Carries the last volume reading onto the sample that replaces its own. + * + * Why: a statfs slower than one tick would otherwise make the field vanish from + * the reports it exists for — the next tick replaces the sample wholesale, and + * the in-flight latch keeps intervening ticks from merging anything. It ships + * with its own (now larger) age and, not being co-timed, never names the label. + */ +function withCarriedSwapVolume(sample: PreGoneSystemMemorySample): PreGoneSystemMemorySample { + const previous = preGoneSample + if (!previous || previous.swapVolumeSampledAtMs === undefined) { + return sample + } + const freeMB = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`] + const volume = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`] + if (typeof freeMB !== 'number' || typeof volume !== 'string') { + return sample + } + return { + ...sample, + details: withSwapVolumeFreeSpace(sample.details, { freeMB, volume }, process.platform, false), + swapVolumeSampledAtMs: previous.swapVolumeSampledAtMs + } +} + +function commitHostMemorySample(nowMs: number): boolean { + try { + const details = getSystemMemoryDetails() + // Why not `length === 0`: the signal label is appended unconditionally, so a + // reading that resolved no memory field at all still arrives with one key. + if (!Object.keys(details).some((key) => key !== PRESSURE_SIGNAL_KEY)) { + return false + } + preGoneSample = withCarriedSwapVolume({ details, sampledAtMs: nowMs }) + return true + } catch { + // Why: a failed read must not erase the previous good sample. + return false + } +} + +async function mergeSwapVolumeFreeSpace(issuedOnTick: number): Promise { + if (swapVolumeReadInFlight) { + return + } + swapVolumeReadInFlight = true + const generation = samplingGeneration + const issuedFor = preGoneSample + try { + const volume = await readSwapVolumeFreeSpace() + if (volume && preGoneSample && generation === samplingGeneration) { + // Why only its own tick qualifies: a statfs that outlived its tick carries a + // pre-storm volume number, and the latch makes that lag unbounded. It still + // ships beside its age, but it may not decide the verdict. + // Why the tick counter and not sample identity: a tick whose host read fails + // leaves the sample object in place, so identity alone reads as co-timed. + const coTimed = issuedOnTick === sampleTick + preGoneSample = { + ...preGoneSample, + details: withSwapVolumeFreeSpace(preGoneSample.details, volume, process.platform, coTimed), + swapVolumeSampledAtMs: issuedFor?.sampledAtMs + } + } + } catch { + // Why: the memory reading is already committed and stands on its own. + } finally { + swapVolumeReadInFlight = false + } +} + +export async function samplePreGoneSystemMemory(nowMs: number = Date.now()): Promise { + // Why commit before awaiting: the volume read is a statfs, and under the very + // paging storm this targets it is slowest — it must never delay, or (via an + // in-flight latch) skip, the cheap synchronous host reading. + const tick = ++sampleTick + if (!commitHostMemorySample(nowMs)) { + return + } + await mergeSwapVolumeFreeSpace(tick) +} + +export function startPreGoneSystemMemorySampling( + intervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS +): void { + if (preGoneTimer) { + return + } + void samplePreGoneSystemMemory() + preGoneTimer = setInterval(() => void samplePreGoneSystemMemory(), intervalMs) + preGoneTimer.unref?.() +} + +export function resetPreGoneSystemMemorySamplingForTest(): void { + if (preGoneTimer) { + clearInterval(preGoneTimer) + } + preGoneTimer = null + preGoneSample = null + swapVolumeReadInFlight = false + // Why bump: an already-awaited volume read must not repopulate a reset sample. + samplingGeneration += 1 +} + +/** Keyed as `systemMemoryPreGone*` so a scan over the `systemMemory` family sees both reads. */ +export function preGoneSystemMemoryDetails(nowMs: number): CrashReportDetails { + if (!preGoneSample) { + return {} + } + const details: CrashReportDetails = { + [`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSampleAgeMs`]: Math.max( + 0, + nowMs - preGoneSample.sampledAtMs + ) + } + // Why its own age: the volume read resolves out of band, so it can be older + // than the memory reading printed beside it, and that gap must be readable. + if (preGoneSample.swapVolumeSampledAtMs !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSwapVolumeAgeMs`] = Math.max( + 0, + nowMs - preGoneSample.swapVolumeSampledAtMs + ) + } + for (const [key, value] of Object.entries(preGoneSample.details)) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGone${key.slice(SYSTEM_MEMORY_KEY_PREFIX.length)}`] = + value + } + return details +} diff --git a/src/main/crash-reporting/process-gone-diagnostics.test.ts b/src/main/crash-reporting/process-gone-diagnostics.test.ts index e31645865d8..6a6ec410733 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.test.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.test.ts @@ -2,11 +2,11 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { buildProcessGoneCrashDetails, collectProcessGoneMetricDetails, - resetPreGoneProcessMetricsSamplingForTest, + resetPreGoneCrashSamplingForTest, samplePreGoneProcessMetrics, - startPreGoneProcessMetricsSampling + startPreGoneCrashSampling } from './process-gone-diagnostics' -import { setSystemMemoryInfoReaderForTest } from './gone-time-system-memory' +import { setSystemMemoryInfoReaderForTest } from './system-memory-details' type MetricFixture = { pid?: number @@ -27,7 +27,7 @@ vi.mock('electron', () => ({ describe('process gone diagnostics', () => { beforeEach(() => { - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() setSystemMemoryInfoReaderForTest(null) }) @@ -141,8 +141,8 @@ describe('process gone diagnostics', () => { appMetricsMock.mockReturnValue([ { pid: 30, type: 'Tab', memory: { workingSetSize: 1024 * 100 } } ]) - startPreGoneProcessMetricsSampling(1_000) - startPreGoneProcessMetricsSampling(1_000) + startPreGoneCrashSampling(1_000) + startPreGoneCrashSampling(1_000) // A crash inside the first interval already has a sample to draw from. expect(buildProcessGoneCrashDetails({}, 'renderer')).toMatchObject({ @@ -582,12 +582,12 @@ describe('process gone diagnostics', () => { it("arms an unref'd interval so sampling never holds the event loop open", () => { const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') try { - startPreGoneProcessMetricsSampling(60_000) + startPreGoneCrashSampling(60_000) const timer = setIntervalSpy.mock.results[0]?.value as NodeJS.Timeout expect(timer.hasRef()).toBe(false) } finally { setIntervalSpy.mockRestore() - resetPreGoneProcessMetricsSamplingForTest() + resetPreGoneCrashSamplingForTest() } }) @@ -641,7 +641,7 @@ describe('process gone diagnostics', () => { expect(details.systemMemoryTotalMB).toBe(16_384) }) - it('samples system memory at gone time but never into the pre-gone snapshot', () => { + it('samples system memory at gone time but never into the processMetrics family', () => { appMetricsMock.mockReturnValue([{ pid: 1, type: 'Browser', memory: { workingSetSize: 0 } }]) samplePreGoneProcessMetrics() setSystemMemoryInfoReaderForTest(() => ({ @@ -658,7 +658,9 @@ describe('process gone diagnostics', () => { systemMemorySwapTotalMB: 8_192, systemMemorySwapFreeMB: 40 }) - expect(details.processMetricsPreGoneSystemMemoryTotalMB).toBeUndefined() + expect( + Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem')) + ).toEqual([]) }) it('leaves records unflagged when the crashed bucket is still populated', () => { diff --git a/src/main/crash-reporting/process-gone-diagnostics.ts b/src/main/crash-reporting/process-gone-diagnostics.ts index d0bb380a2b6..bf0d735a2c7 100644 --- a/src/main/crash-reporting/process-gone-diagnostics.ts +++ b/src/main/crash-reporting/process-gone-diagnostics.ts @@ -3,7 +3,13 @@ import { sanitizeCrashReportDetails, type CrashReportDetailValue } from '../../shared/crash-reporting' -import { getSystemMemoryAtGoneDetails, memoryKBFieldMB } from './gone-time-system-memory' +import { getSystemMemoryDetails, memoryKBFieldMB } from './system-memory-details' +import { + PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS, + preGoneSystemMemoryDetails, + resetPreGoneSystemMemorySamplingForTest, + startPreGoneSystemMemorySampling +} from './pre-gone-host-memory' type ProcessMetricLike = { pid?: unknown @@ -204,8 +210,9 @@ export function samplePreGoneProcessMetrics(nowMs: number = Date.now()): void { } } -export function startPreGoneProcessMetricsSampling( - intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS +export function startPreGoneCrashSampling( + intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS, + systemMemoryIntervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS ): void { if (preGoneSampleTimer) { return @@ -213,14 +220,16 @@ export function startPreGoneProcessMetricsSampling( samplePreGoneProcessMetrics() preGoneSampleTimer = setInterval(() => samplePreGoneProcessMetrics(), intervalMs) preGoneSampleTimer.unref?.() + startPreGoneSystemMemorySampling(systemMemoryIntervalMs) } -export function resetPreGoneProcessMetricsSamplingForTest(): void { +export function resetPreGoneCrashSamplingForTest(): void { if (preGoneSampleTimer) { clearInterval(preGoneSampleTimer) } preGoneSampleTimer = null preGoneSample = null + resetPreGoneSystemMemorySamplingForTest() } const PROCESS_METRICS_KEY_PREFIX = 'processMetrics' @@ -271,7 +280,7 @@ export function buildProcessGoneCrashDetails( const crashDetails: CrashReportDetails = { ...sanitizedDetails, ...liveMetricDetails, - ...getSystemMemoryAtGoneDetails() + ...getSystemMemoryDetails() } // Why: with the crasher gone, Largest names a survivor — flag that so the // live buckets are read as "everyone else", not as the crashed process. @@ -290,8 +299,10 @@ export function buildProcessGoneCrashDetails( if (liveMetricDetails[crashedBucketCountKey] === 0 || sampledSameBucketProcessVanished) { crashDetails.processMetricsCrashedProcessAbsent = true } + const nowMs = Date.now() if (preGoneSample) { - Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, Date.now())) + Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, nowMs)) } + Object.assign(crashDetails, preGoneSystemMemoryDetails(nowMs)) return crashDetails } diff --git a/src/main/crash-reporting/swap-volume-free-space.ts b/src/main/crash-reporting/swap-volume-free-space.ts new file mode 100644 index 00000000000..3ad40b7629b --- /dev/null +++ b/src/main/crash-reporting/swap-volume-free-space.ts @@ -0,0 +1,67 @@ +import { statfs } from 'node:fs/promises' +import path from 'node:path' + +// Why: a system-managed Windows pagefile — and a macOS swapfile — only grows +// into free space on its own volume, so low available commit is a REFUSED +// allocation only when that volume is full too. Linux is excluded on purpose: +// its swap is a fixed partition, a fixed-size swapfile, or zram, none of which +// grow into root-fs free space, so the number would read as headroom that +// cannot exist. The measured volume ships alongside because Windows only names +// the DEFAULT pagefile drive; a relocated pagefile lives elsewhere. + +const BYTES_PER_MB = 1024 * 1024 + +export type SwapVolumeFreeSpace = { + freeMB: number + /** Which volume was measured, separator-trimmed so redaction sees no path. */ + volume: string +} + +type SwapVolumeFreeSpaceReader = ( + platform: NodeJS.Platform +) => Promise + +function swapVolumeAnchor(platform: NodeJS.Platform): string | undefined { + if (platform === 'win32') { + const anchor = process.env.SystemRoot || process.env.SystemDrive + return anchor ? path.parse(anchor).root || anchor : undefined + } + return platform === 'darwin' ? path.sep : undefined +} + +function volumeLabel(root: string): string { + const trimmed = root.replace(/[\\/]+$/, '') + return trimmed.length > 0 ? trimmed : root +} + +async function statfsSwapVolumeFreeSpace( + platform: NodeJS.Platform +): Promise { + const root = swapVolumeAnchor(platform) + if (!root) { + return undefined + } + try { + const stats = await statfs(root) + const bytes = Number(stats.bsize) * Number(stats.bavail) + return Number.isFinite(bytes) + ? { freeMB: Math.round(Math.max(0, bytes) / BYTES_PER_MB), volume: volumeLabel(root) } + : undefined + } catch { + return undefined + } +} + +let swapVolumeFreeSpaceReader: SwapVolumeFreeSpaceReader = statfsSwapVolumeFreeSpace + +export function setSwapVolumeFreeSpaceReaderForTest( + reader: SwapVolumeFreeSpaceReader | null +): void { + swapVolumeFreeSpaceReader = reader ?? statfsSwapVolumeFreeSpace +} + +export function readSwapVolumeFreeSpace( + platform: NodeJS.Platform = process.platform +): Promise { + return swapVolumeFreeSpaceReader(platform) +} diff --git a/src/main/crash-reporting/system-memory-details.ts b/src/main/crash-reporting/system-memory-details.ts new file mode 100644 index 00000000000..1f2cf556faa --- /dev/null +++ b/src/main/crash-reporting/system-memory-details.ts @@ -0,0 +1,161 @@ +import type { CrashReportDetailValue } from '../../shared/crash-reporting' +import type { SwapVolumeFreeSpace } from './swap-volume-free-space' + +// ─── Host system memory for crash reports ─────────────────────────── +// Why: the system outlives the crashed process, so this IS sampleable at +// process-gone — it separates "renderer grew huge" from "machine out of +// memory/commit", which the per-process buckets alone cannot. The gone-time +// caller reads AFTER the corpse returned its pages, so free/swapFree read +// healthier than at kill time; the pre-gone sampler carries a live reading past +// that. +// Every reading is labelled `systemMemoryPressureSignal` so no report can be +// read as a pressure verdict the platform never gave: +// win32 — swapFree is MEMORYSTATUSEX.ullAvailPageFile, i.e. available +// COMMIT, which pagefile growth can heal (a 127 MB commit floor healed to +// 2029 MB mid-hold on the win-lowspec repro, killing nothing). Free space on +// the swap volume does NOT establish that it could: a fixed-size or disabled +// pagefile grows into no amount of empty disk, its maximum is unreadable +// here (needs a registry read), and the measured volume is only the DEFAULT +// pagefile drive. So a co-timed volume reading is context beside the commit +// number — `available-commit-volume-cotimed` — never a verdict. The one +// decisive win32 case is a commit limit at or below RAM: no pagefile exists +// to grow, so the floor cannot heal (`available-commit-hard-capped`). +// linux — MemAvailable is the real signal; MemFree is not (it excludes page +// cache and other reclaimable memory). +// darwin — none. `free` stays low on healthy machines and +// fileBacked/purgeable are only a reclaimability proxy. The real signal +// needs `memory_pressure -Q`; Orca's reader for it +// (src/main/memory/host-memory.ts) is on-demand, and spawning a subprocess +// on a 10 s app-lifetime timer costs more than the gap it closes. + +type CrashReportDetails = Record + +export const SYSTEM_MEMORY_KEY_PREFIX = 'systemMemory' + +export function memoryKBFieldMB(value: unknown): number | undefined { + const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined + return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024) +} + +type SystemMemoryInfoLike = { + total?: unknown + free?: unknown + available?: unknown + swapTotal?: unknown + swapFree?: unknown + fileBacked?: unknown + purgeable?: unknown +} + +type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null + +/** How far this reading may be read as a "was the host under pressure" verdict. */ +export type SystemMemoryPressureSignal = + | 'available-commit-hard-capped' + | 'available-commit-volume-cotimed' + | 'available-commit-unqualified' + | 'mem-available' + | 'none' + +function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null { + const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike }) + .getSystemMemoryInfo + if (typeof read !== 'function') { + return null + } + try { + return read.call(process) + } catch { + return null + } +} + +let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo + +export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void { + systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo +} + +function numericDetail(details: CrashReportDetails, suffix: string): number | undefined { + const value = details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] + return typeof value === 'number' ? value : undefined +} + +/** Windows commit limit = RAM + pagefile, so a limit at or below RAM has no pagefile behind it. */ +function pagefileBacksCommit(details: CrashReportDetails): boolean | undefined { + const total = numericDetail(details, 'TotalMB') + const swapTotal = numericDetail(details, 'SwapTotalMB') + return total === undefined || swapTotal === undefined ? undefined : swapTotal > total +} + +function pressureSignal( + platform: NodeJS.Platform, + details: CrashReportDetails, + volumeCoTimed = true +): SystemMemoryPressureSignal { + if (platform === 'win32' && `${SYSTEM_MEMORY_KEY_PREFIX}SwapFreeMB` in details) { + if (pagefileBacksCommit(details) === false) { + return 'available-commit-hard-capped' + } + return volumeCoTimed && `${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB` in details + ? 'available-commit-volume-cotimed' + : 'available-commit-unqualified' + } + if (platform === 'linux' && `${SYSTEM_MEMORY_KEY_PREFIX}AvailableMB` in details) { + return 'mem-available' + } + return 'none' +} + +export function getSystemMemoryDetails( + platform: NodeJS.Platform = process.platform +): CrashReportDetails { + const info = systemMemoryInfoReader() + if (!info) { + return {} + } + const details: CrashReportDetails = {} + const fields: readonly [keyof SystemMemoryInfoLike, string][] = [ + ['total', 'TotalMB'], + ['free', 'FreeMB'], + ['available', 'AvailableMB'], + ['swapTotal', 'SwapTotalMB'], + ['swapFree', 'SwapFreeMB'], + ['fileBacked', 'FileBackedMB'], + ['purgeable', 'PurgeableMB'] + ] + for (const [field, suffix] of fields) { + const mb = memoryKBFieldMB(info[field]) + if (mb !== undefined) { + details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] = mb + } + } + details[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, details) + return details +} + +/** + * Merges the statfs-derived volume datum, which needs an await and so is only + * reachable from the periodic sampler, and relabels the reading it sits beside. + * + * `coTimed` false means the statfs outlived the tick that issued it, so this + * volume number and the commit number beside it describe different moments — + * during a pagefile-growth storm that is exactly when they diverge, and a + * pre-storm 40 GB printed next to 200 MB of commit reads as "the pagefile had + * room", the opposite conclusion. The datum still ships (with its own age), but + * only a co-timed one is named in the label. + */ +export function withSwapVolumeFreeSpace( + details: CrashReportDetails, + volume: SwapVolumeFreeSpace, + platform: NodeJS.Platform = process.platform, + coTimed = true +): CrashReportDetails { + const merged: CrashReportDetails = { + ...details, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`]: volume.freeMB, + [`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`]: volume.volume + } + merged[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, merged, coTimed) + return merged +} diff --git a/src/main/startup/main-process-ready-runtime.ts b/src/main/startup/main-process-ready-runtime.ts index 75784a4c196..26b920652df 100644 --- a/src/main/startup/main-process-ready-runtime.ts +++ b/src/main/startup/main-process-ready-runtime.ts @@ -9,7 +9,7 @@ import { RpcDispatcher } from '../runtime/rpc/dispatcher' import { browserManager } from '../browser/browser-manager' import { configureBrowserClientPageAutomationRuntime } from '../browser/browser-client-page-automation-runtime' import { BrowserClientPageCommandError } from '../browser/browser-client-page-command-failure' -import { startPreGoneProcessMetricsSampling } from '../crash-reporting/process-gone-diagnostics' +import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics' import { recordProcessGoneCrash } from './main-window-lifecycle-flags' import { handleGpuChildCrash } from './gpu-lifecycle' import { isGpuFallbackCrashCandidate } from '../crash-reporting/gpu-crash-fallback-decision' @@ -130,9 +130,10 @@ export async function initializeReadyRuntimeServices(): Promise { console.warn('[agent-hooks] failed to reconcile managed hooks on startup:', error) ) } - // Why: process-gone metrics only see survivors; retain a recent whole-app - // snapshot for comparison in crash reports. - startPreGoneProcessMetricsSampling() + // Why: process-gone metrics only see survivors, and the gone-time host memory + // read lands after the corpse released its pages; both need a live pre-gone + // sample to compare against in crash reports. + startPreGoneCrashSampling() app.on('child-process-gone', (_event, details) => { recordProcessGoneCrash('child', details.type, details.reason, details.exitCode ?? null, { name: details.name, diff --git a/src/main/startup/pre-gone-crash-sampling-wiring.test.ts b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts new file mode 100644 index 00000000000..2a8e008c0b3 --- /dev/null +++ b/src/main/startup/pre-gone-crash-sampling-wiring.test.ts @@ -0,0 +1,49 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * Guards the one line that arms pre-gone crash sampling. + * + * That branch is pure instrumentation, so this line is the whole of its value in + * the shipped app: deleting it left all 691 tests across `src/main/crash-reporting/` + * and `src/main/startup/` green while every crash report silently lost its only + * host reading taken before the dying process returned its pages. + * + * Source-level because that is the property: the sampler is armed once inside the + * ready-phase composition, which has no runtime seam to assert against. + */ +describe('pre-gone crash sampling startup wiring', () => { + // Why normalize: the indent anchors below are `\n`-prefixed, and nothing pins + // src/**/*.ts to LF, so a CRLF Windows checkout would fail them spuriously. + const readSource = (name: string): string => + readFileSync(join(process.cwd(), 'src/main/startup', name), 'utf8').replace(/\r\n/g, '\n') + + const readyRuntimeSource = readSource('main-process-ready-runtime.ts') + const readySource = readSource('main-process-ready.ts') + + const READY_ENTRY = 'export async function initializeReadyRuntimeServices(' + // Why the entry's body and not the file: the call satisfies a whole-file grep + // just as well from a sibling export nothing calls, which arms nothing. + const readyRuntimeEntryBody = readyRuntimeSource + .slice(readyRuntimeSource.indexOf(READY_ENTRY) + READY_ENTRY.length) + .split('\nexport ')[0] + + it('arms the sampler unconditionally inside the function app readiness runs', () => { + expect(readyRuntimeSource).toContain( + "import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics'" + ) + expect(readyRuntimeSource).toContain(READY_ENTRY) + expect(readyRuntimeEntryBody.split('startPreGoneCrashSampling()').length - 1).toBe(1) + // Why pin the indent: the call also matches as the body of an added + // `if (...)` guard, which keeps every other assertion here true while the + // sampler silently stops arming on most startups. + expect(readyRuntimeEntryBody).toContain('\n startPreGoneCrashSampling()') + + // ...and that this really is the function app readiness runs. + expect(readySource).toContain( + "import { initializeReadyRuntimeServices } from './main-process-ready-runtime'" + ) + expect(readySource).toContain('\n await initializeReadyRuntimeServices()') + }) +}) From 2ee507d744b8f8abc61f91bd0c77563ee3bb5c79 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 01:22:09 -0700 Subject: [PATCH 12/12] fix(ssh): move Windows file writes off PowerShell 5.1 stdin onto sftp (#18596) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): move Windows file writes off PowerShell 5.1 stdin onto sftp #16432 was fixed by chunking writes to 32KB, on the belief that a `DefaultShell=cmd.exe` host caps one stdin at roughly 50KB. Re-measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2, that premise is wrong in both directions, and the chunking does not fix the hang. The real constraint: a read on Windows PowerShell 5.1's redirected-stdin handle over a non-pty ssh exec can die permanently when it finds the stream momentarily empty, taking both the remaining data and the EOF with it. It is probabilistic per such read — not a size threshold, and not certain on the first one. Measured by swapping the copy loop for a counting reader: a 1.5s gap before any byte -> 0 bytes received, 6 of 6 1 byte, 1.5s gap, then 32767 -> exactly 1 byte 32768, 1.5s gap, then 32768 -> exactly 32768 a continuous 2MB -> 167936 / 270336 / 372736 Those three 2MB figures are one payload run three times under the same conditions, which is what rules out a threshold. Independently reproduced by a second harness where one 1.9MB counted read completed through 39 reads and another died after 11. A payload that fits one burst usually presents only one read that can find the stream empty, which is why 32KB mostly works — and it still failed 15 times in 120 under load, and 1 in 40 on a quiet host. Neither rate survives the 62 execs a 1.9MB file needs: even 2.5% compounds to about four uploads in five failing. No chunk size helps, because the defect is per blocking read, not per byte. Three controls on the same host, same DefaultShell, rule out both a size limit and cmd.exe: `findstr` took 2,016,000 bytes through one exec's stdin, sftp moved 1.9MB 5/5, and PowerShell 7 took 2MB in one exec. Windows writes now go over the sftp subsystem, whose batch script is read by the *local* client, so no remote process reads a pipe at all. PowerShell 7 is the fallback where sftp is unavailable, and Windows PowerShell 5.1 is last, still bounded, and now reports the host limitation and its remedy instead of a bare timeout. Measured on the same host, through this code: 1.9MB x20 all succeeded, hash-verified, median 315ms, against 0/6 before. 32KB x120 zero hangs, against 15/120. Also: - Stage under a unique name per attempt. An abandoned write leaves a remote process that may still hold the staging file, and losing contact is not evidence it died (docs/reference/ssh-execution-boundary.md), so a retry must not reuse a name its predecessor may own. Sweep is best-effort and never treated as proof of anything. - Create upload directories over sftp too; the JSON mkdir batch rode the same defective read. - Cover makeWindowsWriteFileCommand and the publish command against the 8000-char budget, which F11 flagged as untested. * fix(ssh): replace the staged Windows write atomically, and translate ssh -l Three review findings, all on the failure path that the success-path measurements say nothing about. CodeRabbit, Critical: the publish deleted the destination before moving the staged file onto it, so a failed move destroyed the user's existing file and left a window where a reader saw no file at all. That is worse than the truncated partial the staging discipline exists to prevent. Now File.Replace (Win32 ReplaceFile, atomic), falling back to a plain Move only when the destination is absent — and that race is safe, because a destination appearing in between makes Move throw with the staged file preserved. The exclusive branch already had it right: Move throwing on an existing destination is the exclusive contract. Append stays non-atomic and now says why. buildSshArgs can emit '-l ' for a config alias no Host block claims, and the translator threw on it. isSftpUnavailableError read that throw as 'this host cannot do sftp', so those hosts fell back to the defective PowerShell 5.1 path and had the refusal cached against them for 30 minutes, silently. '-l' now maps to '-o User=', with a test for the exact argument shape buildSshArgs produces in that case. CodeRabbit, minor: two assertions passed on an absent observation — an unmatched regex yields '' and every() is true of an empty list. Both now assert the positive form first, and the same audit was applied to the three other some()/every() assertions in the file. The temp-file test now asserts mode 0600 rather than only that the file is cleaned up. * fix(ssh): keep a path sftp cannot spell from becoming a verdict about the host Audit of isSftpUnavailableError, prompted by the '-l' gap having the same shape: a per-operation condition being written into a per-host cache that holds for 30 minutes. It had a second instance, and this one was mine. UnsupportedSftpPathError was classified as 'this host cannot do sftp', but it is thrown for a UNC or relative destination and for any path sftp's batch lexer cannot quote -- including a *local* filename containing a newline, which POSIX clients allow. One such file would have routed every later Windows write to that host down the defective PowerShell 5.1 path for the rest of the cache window. The host verdict is now only the errors that really are host-scoped: a refused subsystem, a client that will not start, and an untranslatable argument list. A path refusal falls back for that one write and leaves the cache alone, in both the file-write and directory-creation paths. Revert-tested. Removing the operation-scoped catch fails all three new tests, whether or not the predicate is also widened. Widening the predicate alone does not fail them, correctly: with the catch in place the predicate no longer gates that path, so keeping it narrow is defence-in-depth rather than the live mechanism. Flag audit at the same time: -F, -o, -T, -S, -p, -i, -J, -l and -- are now the complete set buildSshArgs can emit, and all are handled. * fix(ssh): make the atomic publish actually run, and unroll the mkdir batch Two runtime defects that only a real host could surface. Both were invisible to unit tests that assert the shape of the generated command string, because both are PowerShell rejecting an argument at execution time. File.Replace was passed a bare $null for destinationBackupFileName. PowerShell coerces $null to an empty string when binding a .NET string parameter, and Replace rejects that with 'The path is not of a legal form' -- so every create-mode publish failed. The Critical fix was inert as shipped. Now [NullString]::Value, which is the construct that exists for this. Measured on awin, same staging-file lock, opposite outcomes: old publish rc=1 destination MISSING <- prior contents destroyed new publish rc=1 destination PRESENT, sha 7f06b7e0... unchanged control, destination present, no lock rc=0 replaced exactly control, destination absent, no lock rc=0 Move fallback created it End-to-end through the real uploader afterwards: 1.9MB x15 all hashes exact, median 303ms; overwrite of an existing destination exact both times. Separately, the PowerShell mkdir fallback could not create a tree of more than one directory. '@($json | ConvertFrom-Json)' wraps the parsed array in another array, so the loop variable binds to the whole thing and [string] of it is the paths joined by spaces. It only ever worked for a one-element batch, where stringifying a single-element array happens to yield the element -- which is why no existing test caught it. Pre-existing on main; fixed here because this PR puts that command on the fallback tier and claims the ladder works. Both tiers now verified live against a three-directory tree. --- src/main/ssh/ssh-remote-powershell.ts | 20 +- ...-remote-windows-command-line-limit.test.ts | 44 ++ src/main/ssh/ssh-system-fallback.test.ts | 85 ++- .../ssh/system-ssh-file-binary-transfer.ts | 236 ++----- src/main/ssh/system-ssh-file-transfer.ts | 80 ++- src/main/ssh/system-ssh-sftp-args.test.ts | 141 ++++ src/main/ssh/system-ssh-sftp-args.ts | 95 +++ src/main/ssh/system-ssh-sftp-path.test.ts | 59 ++ src/main/ssh/system-ssh-sftp-path.ts | 46 ++ src/main/ssh/system-ssh-sftp-transfer.ts | 191 +++++ src/main/ssh/system-ssh-windows-file-write.ts | 138 ++++ .../ssh/system-ssh-windows-upload.test.ts | 659 ++++++++++++++---- ...tem-ssh-windows-write-capabilities.test.ts | 79 +++ .../system-ssh-windows-write-capabilities.ts | 52 ++ .../ssh/system-ssh-windows-write-strategy.ts | 329 +++++++++ 15 files changed, 1900 insertions(+), 354 deletions(-) create mode 100644 src/main/ssh/system-ssh-sftp-args.test.ts create mode 100644 src/main/ssh/system-ssh-sftp-args.ts create mode 100644 src/main/ssh/system-ssh-sftp-path.test.ts create mode 100644 src/main/ssh/system-ssh-sftp-path.ts create mode 100644 src/main/ssh/system-ssh-sftp-transfer.ts create mode 100644 src/main/ssh/system-ssh-windows-file-write.ts create mode 100644 src/main/ssh/system-ssh-windows-write-capabilities.test.ts create mode 100644 src/main/ssh/system-ssh-windows-write-capabilities.ts create mode 100644 src/main/ssh/system-ssh-windows-write-strategy.ts diff --git a/src/main/ssh/ssh-remote-powershell.ts b/src/main/ssh/ssh-remote-powershell.ts index 8c94fd3c483..420223ced29 100644 --- a/src/main/ssh/ssh-remote-powershell.ts +++ b/src/main/ssh/ssh-remote-powershell.ts @@ -11,14 +11,24 @@ export { // to leave room for the `/c` wrapper sshd adds before cmd.exe counts the line. const WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS = 8_000 -export function powerShellCommand(script: string): string { - const inline = encodedPowerShellCommand(script) +/** + * `pwsh.exe` is PowerShell 7. It is not present on a stock Windows install, so it is only ever + * chosen after a probe — but where it exists it reads a redirected stdin correctly, which Windows + * PowerShell 5.1 does not (see `system-ssh-file-binary-transfer.ts`). + */ +export type WindowsPowerShellExecutable = 'powershell.exe' | 'pwsh.exe' + +export function powerShellCommand( + script: string, + executable: WindowsPowerShellExecutable = 'powershell.exe' +): string { + const inline = encodedPowerShellCommand(script, executable) if (inline.length <= WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { return inline } // Why: these scripts are repetitive enough that gzip beats the UTF-16LE tax by // ~4x, which is the difference between a line cmd.exe runs and one it refuses. - const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script)) + const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script), executable) if (compressed.length > WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { throw new Error( `Remote Windows command needs ${compressed.length} characters; Orca budgets ${WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS} for a line sshd hands to cmd.exe, which itself refuses more than ${CMD_EXE_COMMAND_LINE_MAX_CHARS}.` @@ -27,8 +37,8 @@ export function powerShellCommand(script: string): string { return compressed } -function encodedPowerShellCommand(script: string): string { - return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` +function encodedPowerShellCommand(script: string, executable: WindowsPowerShellExecutable): string { + return `${executable} -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` } /** Orca-prefixed names so the payload can never shadow the bootstrap's own state. */ diff --git a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts index c104078238e..fa34a8fbda7 100644 --- a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts +++ b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts @@ -4,6 +4,10 @@ import { CMD_EXE_COMMAND_LINE_MAX_CHARS } from '../providers/windows-shell-args' import { getRemoteHostPlatform } from './ssh-remote-platform' import { tryStealInstallLockCommand } from './ssh-relay-install-lock-commands' import { decodeRemotePowerShellScript, powerShellCommand } from './ssh-remote-powershell' +import { + makeWindowsPublishStagedFileCommand, + makeWindowsWriteFileCommand +} from './system-ssh-windows-file-write' import { cleanupOwnedRelayUploadStageCommand, promoteOwnedRelayUploadStageCommand, @@ -38,6 +42,17 @@ describe('Windows remote command line limit', () => { [ 'steal stale install lock', tryStealInstallLockCommand(windows, 'C:\\Users\\orca\\.orca-remote\\relay', 1_200) + ], + // F11 flagged these two as uncovered. They carry one path literal each, so they are the file + // commands whose length a caller can actually move. + ['write file', makeWindowsWriteFileCommand('C:\\Users\\orca\\.orca-remote\\relay.js')], + [ + 'publish staged file', + makeWindowsPublishStagedFileCommand( + 'C:\\Users\\orca\\.orca-remote\\relay.js.orca-partial-0123456789ab', + 'C:\\Users\\orca\\.orca-remote\\relay.js', + 'create' + ) ] ])('keeps the %s command inside what sshd\u2019s cmd.exe accepts', (_name, command) => { expect(command.length).toBeLessThanOrEqual(CMD_EXE_COMMAND_LINE_MAX_CHARS) @@ -76,3 +91,32 @@ describe('Windows remote command line limit', () => { ) }) }) + +/** + * F11 asked whether a pathological path could reach the budget, and what happens if it does. + * Measured: the inline encoding crosses 8000 at roughly 2500 high-entropy path characters — an + * order of magnitude past what Windows itself accepts — and the failure is a throw before any ssh + * is spawned, never a hang. + */ +describe('Windows file command budget headroom', () => { + it('absorbs a path far longer than Windows will accept', () => { + const deep = `C:\\Users\\orca\\${'segment\\'.repeat(30)}relay.js` + + expect(deep.length).toBeGreaterThan(260) + expect(makeWindowsWriteFileCommand(deep).length).toBeLessThanOrEqual( + CMD_EXE_COMMAND_LINE_MAX_CHARS + ) + }) + + it('throws rather than spawning a line cmd.exe would refuse', () => { + // Random segments so gzip cannot rescue it, which is the only way to reach the ceiling at all. + const incompressible = Array.from( + { length: 400 }, + (_unused, index) => `${index}-${Math.random().toString(36).slice(2)}` + ).join('\\') + + expect(() => makeWindowsWriteFileCommand(`C:\\${incompressible}\\f.bin`)).toThrow( + /Orca budgets 8000/ + ) + }) +}) diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index c366e899bf6..b477ad682ef 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -709,9 +709,13 @@ describe('spawnSystemSsh', () => { expect(args[standaloneControlIdx + 1]).toBe('none') }) - it('writes files to Windows system SSH targets with PowerShell stdin bytes', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('sends Windows file writes over sftp, not through a remote PowerShell stdin', async () => { + const spawned: EventedProcess[] = [] + spawnMock.mockImplementation(() => { + const proc = createEventedProcess() + spawned.push(proc) + return closeOnceSpawned(proc) + }) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -722,16 +726,24 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('0.1.0', 'utf-8')) + // #16432, re-measured: Windows PowerShell 5.1 can lose a redirected stdin for good when a read + // finds it momentarily empty, so the bytes must not travel that way at all. + const batch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(batch).toContain('put ') + expect(batch).toContain('/C:/Users/me/.orca-remote/relay/.version.orca-partial-') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + expect(sftpArgs).toContain('-b') + // The rename that publishes it reads the staged file, never a pipe. + const publish = (spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '' + expect(publish).toContain('powershell.exe') + expect(decodePowerShellCommand(publish)).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + expect(publish).not.toContain('/bin/sh') }) - it('writes binary buffers to Windows system SSH targets with CreateNew mode', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + it('enforces an exclusive Windows buffer write at the rename, where it is atomic', async () => { + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeBufferViaSystemSsh( @@ -742,12 +754,11 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const remoteCommand = args.at(-1) ?? '' - expect(remoteCommand).toContain('powershell.exe') - expect(decodePowerShellCommand(remoteCommand)).toContain('CreateNew') - expect(remoteCommand).not.toContain('/bin/sh') - expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('png')) + const publish = decodePowerShellCommand((spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '') + // `File::Move` raising on an existing destination is what carries the exclusive contract now; + // a `CreateNew` on the staged file would only refuse a leftover of our own. + expect(publish).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish).not.toContain('[System.IO.File]::Delete($path)') }) it('downloads files from Windows system SSH targets with PowerShell stdout bytes', async () => { @@ -779,8 +790,7 @@ describe('spawnSystemSsh', () => { }) it('forces standalone SSH for Windows file writes when requested', async () => { - const proc = createEventedProcess() - spawnMock.mockImplementation(() => closeOnceSpawned(proc)) + spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess())) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -791,10 +801,14 @@ describe('spawnSystemSsh', () => { ) await expect(promise).resolves.toBeUndefined() - const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') + const sftpArgs = spawnMock.mock.calls[0][1] as string[] + // sftp's own `-S` names a program to run, so the same request has to be spelled as an option. + expect(sftpArgs).not.toContain('-S') + expect(sftpArgs).toContain('ControlPath=none') + const publishArgs = spawnMock.mock.calls[1][1] as string[] + const standaloneControlIdx = publishArgs.indexOf('-S') expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + expect(publishArgs[standaloneControlIdx + 1]).toBe('none') }) it('uploads a Windows directory as a mkdir batch plus per-file writes, never one blob', async () => { @@ -819,19 +833,20 @@ describe('spawnSystemSsh', () => { rmSync(localDir, { recursive: true, force: true }) } + // #16432: directories first, then the file — but both over sftp now, so the only PowerShell + // left is the rename that publishes the staged file, which reads a file rather than a pipe. + const mkdirBatch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(mkdirBatch).toBe('-mkdir "/C:/Users/me/.orca-remote/relay"\n') + const putBatch = String(spawned[1]!.stdin.end.mock.calls[0]?.[0] ?? '') + expect(putBatch).toContain('put ') + expect(putBatch).toContain('/C:/Users/me/.orca-remote/relay/relay.js.orca-partial-') const commands = spawnMock.mock.calls.map((call) => (call[1] as string[]).at(-1) ?? '') - // #16432: directories first (metadata only), then the file bytes on their own stdin. One batch - // meant base64-ing the whole bundle into a single PowerShell string, which the remote never read. - expect(commands).toHaveLength(2) - expect(commands.every((command) => command.includes('powershell.exe'))).toBe(true) expect(commands.every((command) => !command.includes('/bin/sh'))).toBe(true) expect(commands.join('\n')).not.toContain('tar -xzf') - expect(JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string)).toEqual([ - 'C:/Users/me/.orca-remote/relay' - ]) - expect(Buffer.from(spawned[1].stdin.end.mock.calls[0]?.[0] as Buffer).toString('utf-8')).toBe( - 'console.log("relay")' - ) + // Nothing base64s the bundle into one PowerShell string any more, and nothing reads one. + expect( + commands.some((command) => decodePowerShellCommand(command).includes('OpenStandardInput')) + ).toBe(false) }) it('forces standalone SSH for Windows upload packages when requested', async () => { @@ -855,9 +870,9 @@ describe('spawnSystemSsh', () => { } const args = spawnMock.mock.calls[0][1] as string[] - const standaloneControlIdx = args.indexOf('-S') - expect(standaloneControlIdx).toBeGreaterThan(-1) - expect(args[standaloneControlIdx + 1]).toBe('none') + // The first spawn is the sftp client, whose own `-S` names a program to run. + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') }) it('throws when no system ssh is found', () => { diff --git a/src/main/ssh/system-ssh-file-binary-transfer.ts b/src/main/ssh/system-ssh-file-binary-transfer.ts index b0c5b662ed1..149d10dfbd3 100644 --- a/src/main/ssh/system-ssh-file-binary-transfer.ts +++ b/src/main/ssh/system-ssh-file-binary-transfer.ts @@ -1,5 +1,7 @@ import { constants, createWriteStream } from 'node:fs' -import { lstat, open } from 'node:fs/promises' +import { lstat, mkdtemp, open, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import type { Writable } from 'node:stream' import { pipeline } from 'node:stream/promises' import type { SshTarget } from '../../shared/ssh-types' @@ -16,6 +18,16 @@ import { throwIfAborted, waitForChannelClose } from './system-ssh-operation-lifecycle' +import { + writeWindowsRemoteFile, + type WindowsWriteSource +} from './system-ssh-windows-write-strategy' + +export { + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS +} from './system-ssh-windows-write-strategy' +export { WINDOWS_STAGED_WRITE_SUFFIX } from './system-ssh-windows-file-write' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -74,13 +86,16 @@ export async function writeBufferViaSystemSsh( ): Promise { throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - await writeWindowsBytesViaSystemSsh( + await writeWindowsRemoteFile( target, remotePath, - contents.length, - (offset, maxBytes) => - Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), - options + { + totalBytes: contents.length, + readChunk: (offset, maxBytes) => + Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), + withLocalFile: (send) => withTemporaryLocalFile(contents, send) + }, + options ?? {} ) return } @@ -127,20 +142,19 @@ export async function uploadFileViaSystemSsh( throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - // #16432: a Windows host cannot take a whole file through one stdin, however the local side - // paces it — see WINDOWS_STDIN_WRITE_CHUNK_BYTES. This is the path that carries the large - // files, so it is the one that has to be chunked and bounded. - await writeWindowsBytesViaSystemSsh( - target, - remotePath, - openedStat.size, - async (offset, maxBytes) => { + // This is the path that carries the large files, so it is the one the transport choice is + // made for; see the #16432 note below. + const source: WindowsWriteSource = { + totalBytes: openedStat.size, + readChunk: async (offset, maxBytes) => { const buffer = Buffer.allocUnsafe(Math.min(maxBytes, openedStat.size - offset)) const { bytesRead } = await handle.read(buffer, 0, buffer.length, offset) return buffer.subarray(0, bytesRead) }, - options - ) + // The verified local file is already exactly the payload, so sftp sends it as is. + withLocalFile: (send) => send(localPath) + } + await writeWindowsRemoteFile(target, remotePath, source, options ?? {}) return } @@ -173,158 +187,56 @@ export async function uploadFileViaSystemSsh( } /** - * #16432: Windows PowerShell 5.1 stops draining a redirected stdin over a non-pty ssh exec - * somewhere between 50KB and 1MB, depending on the host's `DefaultShell`, and it hangs rather than - * failing. The reporter measured that on both constructs he tried — `[Console]::In.ReadToEnd()` and - * `new IO.StreamReader([Console]::OpenStandardInput())`, the latter reading incrementally, which is - * why the limit cannot be attributed to materializing the payload. `Stream.CopyTo` reads the same - * `[Console]::OpenStandardInput()` object with the same incremental `Read` loop, so nothing in it - * escapes that limit either: no single write may exceed what one stdin is known to carry. + * #16432, re-measured: the constraint is not a size limit, and it is not cmd.exe's. * - * 32KB is an order of magnitude under the low end of the measured range, and under 50KB, which the - * reporter measured succeeding against a stream reader on the worse of the two `DefaultShell` - * settings. - */ -export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 - -/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ -export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 - -/** Suffix for the path a multi-exec Windows write lands on before it is published by rename. */ -export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' - -/** - * Splits one logical Windows write into stdin-sized execs. + * A read on Windows PowerShell 5.1's redirected-stdin handle over a non-pty ssh exec can die + * permanently when it finds the stream momentarily empty: no further bytes arrive, and no EOF ever + * does. It is probabilistic per such read — not a size threshold, and not certain on the first one. + * Measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2 with `DefaultShell = cmd.exe`, by + * replacing the copy loop with a counting reader: * - * A write that needs more than one exec cannot land on the destination directly: a chunk failing - * mid-file would leave a truncated artifact under the real name with nothing marking it incomplete, - * and the retry would then meet its own leftovers — under `exclusive` the retry's `CreateNew` fails - * on them. Multi-exec creates therefore land on a staging path and are published by a rename, which - * is also where `exclusive` is enforced: once, at the destination, instead of smeared across the - * first chunk. A caller-requested append cannot be staged without reading the remote file back, so - * it keeps writing straight through, as its own protocol already implies. + * - a 1.5s gap before any byte, which forces the first read to find nothing -> 0 bytes, 6 of 6 + * - one byte, a 1.5s gap, then 32767 more -> exactly 1 byte, then nothing + * - 32768, a 1.5s gap, then 32768 more -> exactly 32768, then nothing + * - a continuous 2MB -> 167936 / 270336 / 372736, then nothing + * + * Those three 2MB death points are one payload run three times under the same conditions, which is + * what rules out a threshold: a stream that died at a fixed point would not vary by 2x. Independently reproduced by + * a second harness, where one 1.9MB counted read survived 39 reads to completion and another died + * after 11 — same construct, same payload. + * + * A payload small enough to arrive in one burst usually presents only one read that can find the + * stream empty (the one waiting for EOF), which is why 32KB mostly works: it still failed 15 times + * in 120 with the host under load, and 1 in 40 on a quiet one. Neither rate is survivable across + * the 62 execs a 1.9MB file needs — even 2.5% compounds to roughly four uploads in five failing — + * and no chunk size helps, because the client does not control whether its bytes arrive together. + * + * The same host, same `DefaultShell`, same connection pattern contradicts every size-limit reading: + * `findstr` took 2,016,000 bytes through one exec's stdin, and PowerShell 7 took 2MB. So cmd.exe is + * not the ceiling and neither is ~50KB. Writes now go over sftp, which moves the whole payload + * without any remote process reading a pipe; see `system-ssh-windows-write-strategy.ts` for the + * fallback order. + * + * Successes are never partial. Across every run in both harnesses a failed write hung; not one + * produced a short file, so this defect cannot silently truncate an upload. */ -async function writeWindowsBytesViaSystemSsh( - target: SshTarget, - remotePath: string, - totalBytes: number, - readChunk: (offset: number, maxBytes: number) => Promise, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const staged = !options.append && totalBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES - const writePath = staged ? `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` : remotePath - let offset = 0 - // An empty write still has to run: it is what creates (or truncates) the file. - do { - const chunk = await readChunk(offset, WINDOWS_STDIN_WRITE_CHUNK_BYTES) - if (chunk.length === 0 && offset < totalBytes) { - throw new Error(`Source ran short during upload of ${remotePath}`) - } - await writeWindowsChunkViaSystemSsh( - target, - writePath, - chunk, - { - ...options, - append: staged ? offset > 0 : options.append === true || offset > 0, - exclusive: staged ? false : options.exclusive === true && offset === 0 - }, - offset - ) - offset += chunk.length - } while (offset < totalBytes) - if (staged) { - await publishWindowsStagedWrite(target, writePath, remotePath, options) + +/** A staged write is materialized locally first when the source is a buffer rather than a file. */ +async function withTemporaryLocalFile( + contents: Buffer, + send: (localPath: string) => Promise +): Promise { + const directory = await mkdtemp(join(tmpdir(), 'orca-win-upload-')) + const localPath = join(directory, 'payload.bin') + try { + // 0600: the payload can be repository content, and tmpdir is shared on every platform. + await writeFile(localPath, contents, { mode: 0o600 }) + return await send(localPath) + } finally { + await rm(directory, { recursive: true, force: true }).catch(() => {}) } } -async function writeWindowsChunkViaSystemSsh( - target: SshTarget, - remotePath: string, - chunk: Buffer, - options: SystemSshWriteBufferOptions, - offset: number -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsWriteFileCommand(remotePath, options), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose( - channel, - `write ${remotePath} at offset ${offset}`, - WINDOWS_STDIN_WRITE_TIMEOUT_MS - ) - ) - if (!options.signal?.aborted) { - channel.stdin.end(chunk) - } - await closePromise -} - -async function publishWindowsStagedWrite( - target: SshTarget, - stagingPath: string, - remotePath: string, - options: SystemSshWriteBufferOptions -): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand( - target, - makeWindowsPublishStagedFileCommand(stagingPath, remotePath, options.exclusive === true), - { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } - ) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, `publish ${remotePath}`, WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ) - if (!options.signal?.aborted) { - channel.stdin.end() - } - await closePromise -} - -function makeWindowsWriteFileCommand( - remotePath: string, - options?: { append?: boolean; exclusive?: boolean } -): string { - const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$path = ${powerShellLiteral(remotePath)}`, - '$parent = [System.IO.Path]::GetDirectoryName($path)', - 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', - '$inputStream = [Console]::OpenStandardInput()', - `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, - 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' - ].join('; ') - ) -} - -// `File::Move` throws when the destination exists, which is exactly the exclusive contract; the -// non-exclusive caller asked to replace, so it deletes first (a no-op on an absent path). -function makeWindowsPublishStagedFileCommand( - stagingPath: string, - remotePath: string, - exclusive: boolean -): string { - return powerShellCommand( - [ - '$ErrorActionPreference = "Stop"', - `$staging = ${powerShellLiteral(stagingPath)}`, - `$path = ${powerShellLiteral(remotePath)}`, - ...(exclusive ? [] : ['[System.IO.File]::Delete($path)']), - '[System.IO.File]::Move($staging, $path)' - ].join('; ') - ) -} - function makePosixWriteFileCommand( remotePath: string, options?: { append?: boolean; exclusive?: boolean } diff --git a/src/main/ssh/system-ssh-file-transfer.ts b/src/main/ssh/system-ssh-file-transfer.ts index f728c0eab02..d5757904356 100644 --- a/src/main/ssh/system-ssh-file-transfer.ts +++ b/src/main/ssh/system-ssh-file-transfer.ts @@ -27,6 +27,12 @@ import { WINDOWS_STDIN_WRITE_TIMEOUT_MS, writeBufferViaSystemSsh } from './system-ssh-file-binary-transfer' +import { + isSftpPathUnsupportedError, + isSftpUnavailableError, + makeDirectoriesViaSftp +} from './system-ssh-sftp-transfer' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -161,9 +167,14 @@ async function collectWindowsUploadPlan( return plan } -// Why the JSON envelope survives here: a path list is metadata, so this payload stays in the -// hundreds of bytes even for a deep tree. Batched anyway, so a pathological tree cannot walk back -// into the same stdin size that wedges PowerShell. +/** + * Creates the upload's directories, preferring sftp's own `mkdir`. + * + * The PowerShell fallback keeps the JSON envelope, batched under one stdin's worth: a path list is + * metadata, so it stays in the hundreds of bytes even for a deep tree. It is still a redirected + * stdin read though, so on Windows PowerShell 5.1 it carries the same defect as any other — which + * is why sftp is tried first even for a payload this small. + */ async function createWindowsUploadDirectories( target: SshTarget, directories: readonly string[], @@ -175,23 +186,27 @@ async function createWindowsUploadDirectories( if (batch.length === 0) { return } + const pending = batch const payload = JSON.stringify(batch) batch = [] batchBytes = 0 throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + await getWindowsRemoteWriteCapabilities(target).runWithFallback( + 'sftp-subsystem', + async () => { + try { + await makeDirectoriesViaSftp(target, pending, options) + } catch (error) { + // A directory sftp cannot address is this batch's problem, not the host's verdict. + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await createWindowsUploadDirectoriesViaPowerShell(target, payload, options) + } + }, + () => createWindowsUploadDirectoriesViaPowerShell(target, payload, options), + isSftpUnavailableError ) - if (!options.signal?.aborted) { - channel.stdin.end(payload) - } - await closePromise } for (const directory of directories) { const entryBytes = Buffer.byteLength(directory) + 4 @@ -204,17 +219,44 @@ async function createWindowsUploadDirectories( await flush() } +async function createWindowsUploadDirectoriesViaPowerShell( + target: SshTarget, + payload: string, + options: SystemSshOperationOptions +): Promise { + const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end(payload) + } + await closePromise +} + function makeWindowsCreateDirectoriesCommand(): string { return powerShellCommand( [ '$ErrorActionPreference = "Stop"', - // The reporter measured this reader surviving 50KB where `[Console]::In` wedged at the same - // size (#16432); the batch above stays under that. + // Reached only where the host has no sftp subsystem. Windows PowerShell 5.1 can lose a + // redirected stdin for good when a read finds it empty (#16432); a batch this small usually + // arrives in one piece, and "usually" is exactly why sftp is preferred. '$reader = New-Object System.IO.StreamReader([Console]::OpenStandardInput())', 'try { $json = $reader.ReadToEnd() } finally { $reader.Dispose() }', 'if ([string]::IsNullOrWhiteSpace($json)) { return }', - 'foreach ($path in @($json | ConvertFrom-Json)) {', - ' $null = [System.IO.Directory]::CreateDirectory([string]$path)', + // `[string[]]`, not `@(...)`: ConvertFrom-Json emits the parsed array as a single pipeline + // object, so `@(...)` wraps it in *another* array and the loop variable binds to the whole + // thing. `[string]` of that is the paths joined by spaces, which CreateDirectory rejects with + // "The given path's format is not supported". It only ever worked for a one-element batch, + // where stringifying a single-element array happens to yield the element. Measured on + // WindowsPowerShell 5.1.26100 against a three-directory tree. + 'foreach ($path in [string[]]($json | ConvertFrom-Json)) {', + ' $null = [System.IO.Directory]::CreateDirectory($path)', '}' ].join('; ') ) diff --git a/src/main/ssh/system-ssh-sftp-args.test.ts b/src/main/ssh/system-ssh-sftp-args.test.ts new file mode 100644 index 00000000000..d971390f8dd --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.test.ts @@ -0,0 +1,141 @@ +/** + * `buildSshArgs` is shared with the sftp client, and three of its flags mean something else there. + * Every case below is a silent wrong-target rather than an error if the translation is skipped, + * which is why the fallback is "refuse and use another transport", never "pass it through". + */ +import { describe, expect, it } from 'vitest' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' + +describe('translateSshArgsToSftpArgs', () => { + it('sends the port as an option, since sftp -p preserves mtimes instead', () => { + const args = translateSshArgsToSftpArgs(['-p', '2222', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'Port=2222', '--', 'dev@win.example']) + }) + + it('sends the login name as an option, since sftp has no -l', () => { + // `buildSshArgs` emits `-l` for a config alias no Host block claims. Throwing here would send + // exactly those hosts to the transport this PR exists to stop using, silently. + const args = translateSshArgsToSftpArgs(['-l', 'neil', '--', 'awin']) + + expect(args).toEqual(['-o', 'User=neil', '--', 'awin']) + }) + + it('translates the whole unclaimed-alias shape buildSshArgs emits', () => { + const args = translateSshArgsToSftpArgs([ + '-o', + 'BatchMode=no', + '-T', + '-S', + 'none', + '-o', + 'Hostname=192.168.0.186', + '-p', + '2222', + '-l', + 'neil', + '--', + 'awin' + ]) + + expect(args).toEqual([ + '-o', + 'BatchMode=no', + '-o', + 'ControlPath=none', + '-o', + 'Hostname=192.168.0.186', + '-o', + 'Port=2222', + '-o', + 'User=neil', + '--', + 'awin' + ]) + }) + + it('spells ControlPath=none out, since sftp -S names a program to run', () => { + // `sftp -S none` would try to exec a binary called `none`. + const args = translateSshArgsToSftpArgs(['-S', 'none', '--', 'dev@win.example']) + + expect(args).toEqual(['-o', 'ControlPath=none', '--', 'dev@win.example']) + }) + + it('refuses any other -S, which would hand sftp an ssh binary Orca did not choose', () => { + expect(() => translateSshArgsToSftpArgs(['-S', '/tmp/ctl.sock'])).toThrow( + SftpArgTranslationError + ) + }) + + it('drops -T, which sftp does not have', () => { + expect(translateSshArgsToSftpArgs(['-T', '--', 'host'])).toEqual(['--', 'host']) + }) + + it('passes through the flags both clients spell the same way', () => { + const args = translateSshArgsToSftpArgs([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + + expect(args).toEqual([ + '-F', + '/tmp/config', + '-o', + 'BatchMode=yes', + '-i', + '/tmp/key', + '-J', + 'jump.example', + '--', + 'dev@win.example' + ]) + }) + + it('takes everything after -- as the destination without reinterpreting it', () => { + // A host literally named `-p` is not a flag once `--` has been seen. + expect(translateSshArgsToSftpArgs(['--', '-p'])).toEqual(['--', '-p']) + }) + + it('refuses an unknown flag rather than guessing what sftp would do with it', () => { + // The point of the throw: a flag added to buildSshArgs later must degrade to another + // transport, not reach sftp carrying a different meaning. + expect(() => translateSshArgsToSftpArgs(['-A', '--', 'host'])).toThrow(SftpArgTranslationError) + }) + + it('refuses a value flag with no value', () => { + expect(() => translateSshArgsToSftpArgs(['-o'])).toThrow(SftpArgTranslationError) + }) +}) + +describe('withSftpKeepalive', () => { + it('asks OpenSSH to notice a dead peer, since the transfer itself has no wall-clock bound', () => { + expect(withSftpKeepalive(['--', 'host'])).toEqual([ + '-o', + 'ServerAliveInterval=15', + '-o', + 'ServerAliveCountMax=3', + '--', + 'host' + ]) + }) + + it('leaves a caller-stated keepalive policy alone', () => { + const args = withSftpKeepalive(['-o', 'ServerAliveInterval=60', '--', 'host']) + + expect(args.filter((arg) => arg.startsWith('ServerAliveInterval'))).toEqual([ + 'ServerAliveInterval=60' + ]) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-args.ts b/src/main/ssh/system-ssh-sftp-args.ts new file mode 100644 index 00000000000..17fa37e78bb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-args.ts @@ -0,0 +1,95 @@ +/** + * Rewrites `buildSshArgs` output for the sftp(1) client. + * + * Three flags ssh and sftp share spell different things: sftp's `-p` is "preserve mtime", its `-S` + * names the ssh binary to run, and it has no `-T` at all. Passing ssh's list through unchanged + * would silently connect to the wrong port and try to exec a program called `none`. + * + * Anything this table does not recognize throws. A flag added to `buildSshArgs` later must degrade + * to the non-sftp transfer path, never reach sftp carrying a different meaning. + */ + +/** `buildSshArgs` emitted a flag with no sftp equivalent; the caller should use another transport. */ +export class SftpArgTranslationError extends Error { + constructor(flag: string) { + super(`No sftp equivalent for system ssh argument ${JSON.stringify(flag)}`) + this.name = 'SftpArgTranslationError' + } +} + +/** Flags whose spelling and meaning are identical in both clients. */ +const PASSTHROUGH_VALUE_FLAGS = new Set(['-F', '-o', '-i', '-J']) + +export function translateSshArgsToSftpArgs(sshArgs: readonly string[]): string[] { + const sftpArgs: string[] = [] + let index = 0 + while (index < sshArgs.length) { + const flag = sshArgs[index]! + if (flag === '--') { + // Everything after `--` is the destination, which both clients spell the same way. + sftpArgs.push(...sshArgs.slice(index)) + return sftpArgs + } + const value = sshArgs[index + 1] + if (PASSTHROUGH_VALUE_FLAGS.has(flag)) { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push(flag, value) + index += 2 + continue + } + if (flag === '-T') { + // sftp never allocates a tty, so ssh's "no tty" request has nothing to translate to. + index += 1 + continue + } + if (flag === '-p') { + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `Port=${value}`) + index += 2 + continue + } + if (flag === '-l') { + // sftp has no `-l`; the login name is an option there. `buildSshArgs` emits this for an + // unclaimed config alias, so throwing would route those hosts down the defective path and + // then cache the refusal against them for half an hour. + if (value === undefined) { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', `User=${value}`) + index += 2 + continue + } + if (flag === '-S') { + // ssh's `-S none` is ControlPath=none; sftp's `-S` would run a binary called `none`. + if (value !== 'none') { + throw new SftpArgTranslationError(flag) + } + sftpArgs.push('-o', 'ControlPath=none') + index += 2 + continue + } + throw new SftpArgTranslationError(flag) + } + return sftpArgs +} + +/** + * A transfer that stalls mid-stream has no per-write bound to catch it, so ask OpenSSH to notice a + * dead peer itself. Only added when the caller has not already stated a keepalive policy. + */ +export function withSftpKeepalive(sftpArgs: readonly string[]): string[] { + const hasOption = (name: string): boolean => + sftpArgs.some((arg, position) => sftpArgs[position - 1] === '-o' && arg.startsWith(`${name}=`)) + const keepalive: string[] = [] + if (!hasOption('ServerAliveInterval')) { + keepalive.push('-o', 'ServerAliveInterval=15') + } + if (!hasOption('ServerAliveCountMax')) { + keepalive.push('-o', 'ServerAliveCountMax=3') + } + return [...keepalive, ...sftpArgs] +} diff --git a/src/main/ssh/system-ssh-sftp-path.test.ts b/src/main/ssh/system-ssh-sftp-path.test.ts new file mode 100644 index 00000000000..e196c2e3e8f --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.test.ts @@ -0,0 +1,59 @@ +/** + * Both functions here guard against the same measured failure: sftp's batch lexer treats `\` as an + * escape, so a Windows path handed over raw is silently mis-targeted *and the client still exits + * 0*. On Windows 11 / OpenSSH 10.0p2, `put src C:\Users\neil\qt\a.bin` created a file literally + * named `C` in the start directory and reported success. + */ +import { describe, expect, it } from 'vitest' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' + +describe('toSftpRemotePath', () => { + it('roots a drive path under /, which is the namespace the Windows sftp-server exposes', () => { + // `pwd` in that session reports `/C:/Users/dev`. + expect(toSftpRemotePath('C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('accepts a path already in that namespace unchanged', () => { + expect(toSftpRemotePath('/C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('converts the separators Orca stores paths with', () => { + expect(toSftpRemotePath('C:\\Users\\dev\\f.bin')).toBe('/C:/Users/dev/f.bin') + }) + + it('declines a UNC path rather than guessing where it lands', () => { + // A guess here writes real bytes to the wrong place; declining falls back to another transport. + expect(() => toSftpRemotePath('//server/share/f.bin')).toThrow(UnsupportedSftpPathError) + }) + + it('declines a relative path, which would resolve against the session start directory', () => { + expect(() => toSftpRemotePath('Users/dev/f.bin')).toThrow(UnsupportedSftpPathError) + }) +}) + +describe('quoteSftpBatchArgument', () => { + it('escapes the backslashes in a Windows client local path', () => { + // Unescaped, sftp reads this as C:srcf.bin and fails to find the source. + expect(quoteSftpBatchArgument('C:\\src\\f.bin')).toBe('"C:\\\\src\\\\f.bin"') + }) + + it('keeps a path with spaces as one argument', () => { + expect(quoteSftpBatchArgument('/tmp/two words.bin')).toBe('"/tmp/two words.bin"') + }) + + it('escapes an embedded quote, which would otherwise end the argument early', () => { + expect(quoteSftpBatchArgument('/tmp/dq".bin')).toBe('"/tmp/dq\\".bin"') + }) + + it('refuses a line break, which would split one batch command into two', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\nrm -rf b')).toThrow(UnsupportedSftpPathError) + }) + + it('refuses a NUL, which truncates the argument', () => { + expect(() => quoteSftpBatchArgument('/tmp/a\0b')).toThrow(UnsupportedSftpPathError) + }) +}) diff --git a/src/main/ssh/system-ssh-sftp-path.ts b/src/main/ssh/system-ssh-sftp-path.ts new file mode 100644 index 00000000000..2b5bfe53f02 --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-path.ts @@ -0,0 +1,46 @@ +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * A path this transfer cannot express to sftp. Callers treat it as "use another transport", never + * as a transfer failure. + */ +export class UnsupportedSftpPathError extends Error { + constructor(path: string) { + super(`Path cannot be addressed over sftp: ${JSON.stringify(path)}`) + this.name = 'UnsupportedSftpPathError' + } +} + +/** + * Converts a Windows remote path to the namespace OpenSSH's Windows sftp-server exposes, which + * roots every drive under `/`: `C:/Users/dev/f` is `/C:/Users/dev/f`, and `pwd` there reports + * `/C:/Users/dev`. + */ +export function toSftpRemotePath(remotePath: string): string { + const normalized = normalizeWindowsRemotePath(remotePath) + if (/^\/[a-zA-Z]:\//.test(normalized)) { + return normalized + } + if (/^[a-zA-Z]:\//.test(normalized)) { + return `/${normalized}` + } + // UNC (`//server/share`) and relative paths have no settled mapping in this namespace, and a + // guess here writes real bytes to the wrong place. Decline instead. + throw new UnsupportedSftpPathError(remotePath) +} + +/** + * Quotes one argument of an sftp batch line. + * + * Escaping is load-bearing, not cosmetic: sftp's batch lexer treats `\` as an escape even inside + * double quotes, so an unescaped Windows local path `C:\src\f.bin` is read as `C:srcf.bin`, and an + * unescaped destination `C:\Users\dev\f.bin` writes a file literally named `C` in the start + * directory — while sftp still exits 0. Both measured on Windows 11 / OpenSSH 10.0p2. + */ +export function quoteSftpBatchArgument(value: string): string { + if (/[\n\r\0]/.test(value)) { + // A line break would split one batch command into two; NUL truncates the argument. + throw new UnsupportedSftpPathError(value) + } + return `"${value.replace(/([\\"])/g, '\\$1')}"` +} diff --git a/src/main/ssh/system-ssh-sftp-transfer.ts b/src/main/ssh/system-ssh-sftp-transfer.ts new file mode 100644 index 00000000000..c50375fa8eb --- /dev/null +++ b/src/main/ssh/system-ssh-sftp-transfer.ts @@ -0,0 +1,191 @@ +import { accessSync, constants, existsSync, statSync } from 'node:fs' +import { posix, win32 } from 'node:path' +import type { SshTarget } from '../../shared/ssh-types' +import { buildSshArgs, type SystemSshBuildArgsOptions } from './system-ssh-args' +import { findSystemSsh } from './system-ssh-binary' +import { + SftpArgTranslationError, + translateSshArgsToSftpArgs, + withSftpKeepalive +} from './system-ssh-sftp-args' +import { + quoteSftpBatchArgument, + toSftpRemotePath, + UnsupportedSftpPathError +} from './system-ssh-sftp-path' +import { throwIfAborted } from './system-ssh-operation-lifecycle' +import { runProcess } from '../../shared/child-process/run-process' + +/** The host answered, but not with an sftp subsystem. The caller must fall back, not fail. */ +export class SftpSubsystemUnavailableError extends Error { + constructor(detail: string) { + super(`Remote host has no usable sftp subsystem: ${detail}`) + this.name = 'SftpSubsystemUnavailableError' + } +} + +/** + * True for the errors that mean "this host cannot serve sftp at all". + * + * Host-scoped, and therefore the only errors safe to remember: a capability cache keyed by host + * turns anything it accepts into a verdict about every later write to that host. Deliberately + * narrow — a permission denial or a missing directory is a real failure that must surface, not a + * reason to retry the whole upload down a slower path. + */ +export function isSftpUnavailableError(error: unknown): boolean { + return error instanceof SftpSubsystemUnavailableError || error instanceof SftpArgTranslationError +} + +/** + * True when *this path* cannot be spelled for sftp, which says nothing about the host. + * + * Kept apart from the host verdict on purpose. A UNC destination, or a local file whose name + * contains a newline, is a property of one operation; caching it would degrade every subsequent + * write to that host for the cache's whole retry window on the strength of one odd filename. + */ +export function isSftpPathUnsupportedError(error: unknown): boolean { + return error instanceof UnsupportedSftpPathError +} + +/** Neither kind of refusal moves a byte, so a staged file cannot exist to sweep. */ +export function isSftpRefusalBeforeStaging(error: unknown): boolean { + return isSftpUnavailableError(error) || isSftpPathUnsupportedError(error) +} + +function systemSftpCandidates(sshPath: string | null, platform: NodeJS.Platform): string[] { + const pathApi = platform === 'win32' ? win32 : posix + const executable = platform === 'win32' ? 'sftp.exe' : 'sftp' + const candidates: string[] = [] + // Why the ssh binary's own directory first: a host with two OpenSSH installs must pair the sftp + // client with the ssh that `buildSshArgs` was built for, not whichever one PATH happens to reach. + if (sshPath) { + candidates.push(pathApi.join(pathApi.dirname(sshPath), executable)) + } + if (platform === 'win32') { + const systemRoot = process.env.SystemRoot || process.env.WINDIR + if (systemRoot) { + candidates.push(win32.join(systemRoot, 'System32', 'OpenSSH', executable)) + } + } else { + candidates.push('/usr/bin/sftp', '/usr/local/bin/sftp', '/opt/homebrew/bin/sftp') + } + return candidates +} + +/** Locate the sftp client paired with the system ssh binary. Returns null when there is none. */ +export function findSystemSftp(): string | null { + if (process.env.ORCA_SYSTEM_SFTP_PATH) { + return process.env.ORCA_SYSTEM_SFTP_PATH + } + const sshPath = findSystemSsh() + for (const candidate of systemSftpCandidates(sshPath, process.platform)) { + try { + if (!statSync(candidate).isFile()) { + continue + } + if (process.platform !== 'win32') { + accessSync(candidate, constants.X_OK) + } + return candidate + } catch { + continue + } + } + return findSftpOnPath() +} + +function findSftpOnPath(): string | null { + const pathValue = process.env.PATH + if (!pathValue) { + return null + } + const pathApi = process.platform === 'win32' ? win32 : posix + const executable = process.platform === 'win32' ? 'sftp.exe' : 'sftp' + for (const entry of pathValue.split(pathApi.delimiter)) { + const directory = entry.trim().replace(/^"|"$/g, '') + if (!directory) { + continue + } + const candidate = pathApi.join(directory, executable) + if (existsSync(candidate)) { + return candidate + } + } + return null +} + +/** + * OpenSSH prints this when the server refuses the subsystem — a host with `Subsystem sftp` + * commented out, or an internal-sftp block that does not apply to this user. + */ +const SUBSYSTEM_REFUSED_PATTERN = /subsystem request failed|no such file or directory.*sftp-server/i + +export type SftpBatchOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal } + +/** + * Runs one sftp batch script. + * + * The script goes to the *local* sftp client's stdin, which is the point: no remote process ever + * reads a redirected stdin, so none of this rides the Windows PowerShell stdin defect. + */ +export async function runSftpBatch( + target: SshTarget, + commands: readonly string[], + options?: SftpBatchOptions +): Promise { + throwIfAborted(options?.signal) + const sftpPath = findSystemSftp() + if (!sftpPath) { + throw new SftpSubsystemUnavailableError('no sftp client binary found alongside ssh') + } + const args = withSftpKeepalive(translateSshArgsToSftpArgs(buildSshArgs(target, options))) + let result + try { + result = await runProcess({ + program: sftpPath, + args: ['-b', '-', ...args], + // `-b -` takes the script on stdin, and that stdin is the *local* client's — no remote + // process reads a pipe anywhere in this transfer, which is the whole point of preferring it. + input: `${commands.join('\n')}\n`, + // Why no timeout: a large upload is legitimately slow, and a wall-clock cap would fail a + // healthy transfer on a slow link. A dead peer is caught by the ServerAlive options instead. + timeoutMs: null, + signal: options?.signal + }) + } catch (error) { + // A client that will not start is "this host cannot do sftp" from the caller's side, not a + // transfer failure: the payload never left. Falling back is the only useful answer. + throw new SftpSubsystemUnavailableError( + `sftp client at ${sftpPath} could not be started: ${error instanceof Error ? error.message : String(error)}` + ) + } + if (result.code === 0) { + return + } + throwIfAborted(options?.signal) + const detail = result.stderr.trim() + if (SUBSYSTEM_REFUSED_PATTERN.test(detail)) { + throw new SftpSubsystemUnavailableError(detail) + } + throw new Error(`sftp batch failed (exit ${result.code}): ${detail}`) +} + +/** + * Creates remote directories, parents first. + * + * `-mkdir` keeps sftp going when a directory is already there; batch mode otherwise aborts the + * whole script on the first non-zero status, which for an idempotent tree walk is not a failure. + */ +export function makeDirectoriesViaSftp( + target: SshTarget, + remoteDirectories: readonly string[], + options?: SftpBatchOptions +): Promise { + const commands = remoteDirectories.map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + if (commands.length === 0) { + return Promise.resolve() + } + return runSftpBatch(target, commands, options) +} diff --git a/src/main/ssh/system-ssh-windows-file-write.ts b/src/main/ssh/system-ssh-windows-file-write.ts new file mode 100644 index 00000000000..8b98d4aec50 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-file-write.ts @@ -0,0 +1,138 @@ +import { randomBytes } from 'node:crypto' +import { powerShellCommand, powerShellLiteral } from './ssh-remote-powershell' +import { normalizeWindowsRemotePath } from './ssh-remote-platform' + +/** + * Suffix marking the path a Windows write lands on before it is published by rename. + * + * The random tail is the fix for a measured harm, not decoration. A write that loses contact with + * the host leaves a remote process that may still hold the staging file open exclusively, and + * `docs/reference/ssh-execution-boundary.md` is explicit that losing contact is not evidence that + * process died — so the retry must not reuse the name it may still own. A fresh name per attempt + * means a retry never meets its predecessor's lock; the abandoned file is cleaned up best-effort + * and never treated as proof of anything. + */ +export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' + +export function makeWindowsStagingPath(remotePath: string): string { + return `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}-${randomBytes(6).toString('hex')}` +} + +export type WindowsPublishMode = 'create' | 'exclusive' | 'append' + +/** + * Publishes a staged upload onto its real name. + * + * Every branch reads the staged *file*, never a redirected stdin, which is what makes this safe on + * a host whose Windows PowerShell 5.1 cannot drain a piped stdin. + * + * The replacing branch must never delete the destination first. Deleting and then moving loses the + * user's existing file outright if the move fails, and exposes a window where a reader sees no file + * at all — a worse outcome than the truncated-partial this staging discipline exists to prevent. + * `File.Replace` is the atomic swap (Win32 `ReplaceFile`), and it requires the destination to + * exist, so an absent one falls back to a plain `Move`. That fallback is raced deliberately: if the + * destination appears in between, `Move` throws, the staged file survives, and the destination is + * left exactly as whoever created it left it. + * + * `File::Move` throwing on an existing destination is also precisely the exclusive contract, which + * is why that branch needs nothing else. + */ +export function makeWindowsPublishStagedFileCommand( + stagingPath: string, + remotePath: string, + mode: WindowsPublishMode +): string { + const preamble = [ + '$ErrorActionPreference = "Stop"', + `$staging = ${powerShellLiteral(stagingPath)}`, + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }' + ] + if (mode === 'append') { + return powerShellCommand( + [ + ...preamble, + // Not atomic, and cannot cheaply be: appending is defined as extending the destination, so + // a failure part-way leaves it longer than it was rather than destroyed. The caller's + // chunked-append protocol already restarts from its own offset. + '$in = [System.IO.File]::OpenRead($staging)', + '$out = [System.IO.File]::Open($path, [System.IO.FileMode]::Append, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)', + 'try { $in.CopyTo($out) } finally { $out.Dispose(); $in.Dispose() }', + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) + } + if (mode === 'exclusive') { + return powerShellCommand([...preamble, '[System.IO.File]::Move($staging, $path)'].join('; ')) + } + return powerShellCommand( + [ + ...preamble, + // `[NullString]::Value`, not `$null`: PowerShell coerces a bare `$null` to an empty string + // when binding a .NET `string` parameter, and `Replace` rejects that with "The path is not + // of a legal form" — so every publish would fail. Measured on WindowsPowerShell 5.1.26100. + 'try { [System.IO.File]::Replace($staging, $path, [NullString]::Value) } catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ].join('; ') + ) +} + +/** Best-effort removal of a staged file whose write was abandoned. Never asserts the writer died. */ +export function makeWindowsDiscardStagedFileCommand(stagingPath: string): string { + return powerShellCommand( + [ + // Deliberately not `Stop`: the previous writer may still hold this file, and that is a + // possibility to tolerate, not an error to report. The unique staging name means a leftover + // blocks nothing; sweeping it is housekeeping. + '$ErrorActionPreference = "SilentlyContinue"', + `$staging = ${powerShellLiteral(stagingPath)}`, + '[System.IO.File]::Delete($staging)' + ].join('; ') + ) +} + +/** + * The ancestor directories of a Windows remote path, drive root first. + * + * sftp's `mkdir` creates one level, so a batch has to name each level itself. The drive root is + * excluded: `-mkdir "/C:/"` is not a directory anyone creates. + */ +export function windowsRemoteAncestorDirectories(remotePath: string): string[] { + const normalized = normalizeWindowsRemotePath(remotePath) + const segments = normalized.split('/') + segments.pop() + const ancestors: string[] = [] + // Start past the drive (`C:`) or the UNC host, which are never created. + for (let depth = 2; depth <= segments.length; depth += 1) { + const directory = segments.slice(0, depth).join('/') + if (directory) { + ancestors.push(directory) + } + } + return ancestors +} + +/** + * `[Console]::OpenStandardInput()` into a `FileStream`, used only by the two stdin fallbacks. + * + * On Windows PowerShell 5.1 this is the defective read; see the strategy comment in + * `system-ssh-file-binary-transfer.ts`. It is correct under PowerShell 7. + */ +export function makeWindowsWriteFileCommand( + remotePath: string, + options?: { append?: boolean; exclusive?: boolean; executable?: 'powershell.exe' | 'pwsh.exe' } +): string { + const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create' + return powerShellCommand( + [ + '$ErrorActionPreference = "Stop"', + `$path = ${powerShellLiteral(remotePath)}`, + '$parent = [System.IO.Path]::GetDirectoryName($path)', + 'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', + '$inputStream = [Console]::OpenStandardInput()', + `$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`, + 'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }' + ].join('; '), + options?.executable ?? 'powershell.exe' + ) +} diff --git a/src/main/ssh/system-ssh-windows-upload.test.ts b/src/main/ssh/system-ssh-windows-upload.test.ts index 207c3e2df8e..c3a5ff80276 100644 --- a/src/main/ssh/system-ssh-windows-upload.test.ts +++ b/src/main/ssh/system-ssh-windows-upload.test.ts @@ -1,29 +1,40 @@ /** - * #16432: the Windows relay upload pushed the whole bundle into one PowerShell stdin, which - * Windows PowerShell 5.1 cannot drain over a non-pty ssh exec — the remote blocks forever, and - * `waitForChannelClose()` had no timeout, so the UI sat at "Connecting…" with no error. Covered - * here: no write exceeds one stdin's worth on any Windows path (bundle upload *and* single-file - * upload, which is the one that carries large files), a partial write never lands under the real - * name, and a remote that never closes fails instead of hanging. + * #16432. The original fix chunked the payload because the constraint was believed to be a ~50KB + * cmd.exe stdin ceiling. Re-measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2, it is + * not a size limit and not cmd.exe's: a read on Windows PowerShell 5.1's redirected-stdin handle + * over a non-pty ssh exec can die permanently when it finds the stream momentarily empty, taking + * both the remaining data and the EOF with it. It is probabilistic per such read — identical 2MB + * payloads died at 167936, 270336 and 372736 — so a 32KB chunk still failed 15 times in 120 under + * load, while `findstr` took 2,016,000 bytes through one exec on the same host. + * + * So the covering property is no longer "every write is small". It is "the bytes do not cross a + * remote process's stdin at all": sftp first, PowerShell 7 next, and Windows PowerShell 5.1 last, + * bounded and loud. The staging-and-rename discipline is kept on every path, with a unique staging + * name per attempt so a retry never meets a predecessor's lock. */ import { EventEmitter } from 'node:events' import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' -import { rm } from 'node:fs/promises' +import { readFile, rm, stat } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { PassThrough, Writable } from 'node:stream' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as SystemSshOperationLifecycle from './system-ssh-operation-lifecycle' -const { spawnSystemSshCommandMock, waitForChannelCloseSpy } = vi.hoisted(() => ({ +const { spawnSystemSshCommandMock, waitForChannelCloseSpy, runProcessMock } = vi.hoisted(() => ({ spawnSystemSshCommandMock: vi.fn(), - waitForChannelCloseSpy: vi.fn() + waitForChannelCloseSpy: vi.fn(), + runProcessMock: vi.fn() })) vi.mock('./system-ssh-command', () => ({ spawnSystemSshCommand: spawnSystemSshCommandMock })) +vi.mock('../../shared/child-process/run-process', () => ({ + runProcess: runProcessMock +})) + // Delegates to the real implementation; the spy only records whether each wait was given a bound. vi.mock('./system-ssh-operation-lifecycle', async (importActual) => { const actual = (await importActual()) as typeof SystemSshOperationLifecycle @@ -41,6 +52,11 @@ import { } from './system-ssh-file-binary-transfer' import { waitForChannelClose } from './system-ssh-operation-lifecycle' import { getRemoteHostPlatform } from './ssh-remote-platform' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities +} from './system-ssh-windows-write-capabilities' +import { explainWindowsPowerShellStdinFailure } from './system-ssh-windows-write-strategy' import type { SshTarget } from '../../shared/ssh-types' type FakeChannel = EventEmitter & { @@ -50,7 +66,12 @@ type FakeChannel = EventEmitter & { written: Buffer } -const target = { id: 'win-1', host: 'win.example', username: 'dev' } as unknown as SshTarget +const target = { + id: 'win-1', + host: 'win.example', + username: 'dev', + port: 22 +} as unknown as SshTarget const hostPlatform = getRemoteHostPlatform('win32-x64') const remoteRoot = 'C:/Users/dev/.orca-remote' @@ -78,96 +99,238 @@ function createFakeChannel(onEnd: (channel: FakeChannel) => void): FakeChannel { return channel } -type RecordedCommand = { script: string; stdin: Buffer } +type RecordedCommand = { script: string; executable: string; stdin: Buffer } +type RecordedSftpBatch = { args: string[]; script: string } -describe('Windows upload stdin framing', () => { - let localDir: string - const commands: RecordedCommand[] = [] - /** Index of the spawn that should report a non-zero exit, to model a chunk failing mid-file. */ - let failAtSpawn = -1 +const sftpBatches: RecordedSftpBatch[] = [] +const commands: RecordedCommand[] = [] +/** Index of the exec that should report a non-zero exit, to model a chunk failing mid-file. */ +let failAtSpawn = -1 +let localDir: string - const fileWrites = (): RecordedCommand[] => - commands.filter((command) => command.script.includes('FileMode]::')) - const writtenPath = (command: RecordedCommand): string => - /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1].replace(/''/g, "'") ?? '' - const fileMode = (command: RecordedCommand): string | undefined => - /FileMode\]::(\w+)/.exec(command.script)?.[1] +const fileWrites = (): RecordedCommand[] => + commands.filter((command) => command.script.includes('OpenStandardInput')) +const writtenPath = (command: RecordedCommand): string => + /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1]?.replace(/''/g, "'") ?? '' +const fileMode = (command: RecordedCommand): string | undefined => + /FileMode\]::(\w+)/.exec(command.script)?.[1] +const putLines = (): string[] => + sftpBatches.flatMap((batch) => batch.script.split('\n').filter((line) => line.startsWith('put '))) +const putDestination = (line: string): string => /put "(?:[^"]*)" "([^"]*)"/.exec(line)?.[1] ?? '' +const putSource = (line: string): string => /put "([^"]*)"/.exec(line)?.[1] ?? '' - beforeEach(() => { - commands.length = 0 - failAtSpawn = -1 - waitForChannelCloseSpy.mockClear() - localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) - spawnSystemSshCommandMock.mockReset() - spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { - const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 - return createFakeChannel((channel) => { - commands.push({ script: decodePowerShellCommand(command), stdin: channel.written }) - setImmediate(() => - spawnIndex === failAtSpawn - ? channel.emit('close', 1, null) - : channel.emit('close', 0, null) - ) +/** Makes every sftp batch succeed, recording what it was asked to do. */ +function acceptSftp(): void { + runProcessMock.mockImplementation( + async (spec: { args: string[]; input: string; program: string }) => { + const script = spec.input + sftpBatches.push({ args: spec.args, script }) + // Model the real client: `put` copies the local file, so read it while it still exists. + for (const line of script.split('\n').filter((entry) => entry.startsWith('put '))) { + await readFile(putSource(line)) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + } + ) +} + +/** Models a host whose sshd has no `Subsystem sftp` line. */ +function refuseSftp(): void { + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + return { + code: 255, + signal: null, + stdout: '', + stderr: 'subsystem request failed on channel 0\nConnection closed', + timedOut: false + } + }) +} + +/** Models a host with no PowerShell 7, which cmd.exe reports as an unrecognized command. */ +function refusePwsh(): void { + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + const executable = command.split(' ')[0] ?? '' + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable, + stdin: channel.written + }) + setImmediate(() => { + if (executable === 'pwsh.exe') { + channel.stderr.write( + "'pwsh.exe' is not recognized as an internal or external command,\noperable program or batch file." + ) + channel.emit('close', 9009, null) + return + } + channel.emit('close', spawnIndex === failAtSpawn ? 1 : 0, null) }) }) }) +} - afterEach(async () => { - await rm(localDir, { recursive: true, force: true }) +beforeEach(() => { + commands.length = 0 + sftpBatches.length = 0 + failAtSpawn = -1 + clearWindowsRemoteWriteCapabilitiesForTests() + waitForChannelCloseSpy.mockClear() + localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) + process.env.ORCA_SYSTEM_SFTP_PATH = '/usr/bin/sftp' + runProcessMock.mockReset() + acceptSftp() + spawnSystemSshCommandMock.mockReset() + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + return createFakeChannel((channel) => { + commands.push({ + script: decodePowerShellCommand(command), + executable: command.split(' ')[0] ?? '', + stdin: channel.written + }) + setImmediate(() => + spawnIndex === failAtSpawn ? channel.emit('close', 1, null) : channel.emit('close', 0, null) + ) + }) }) +}) - it('never pushes a whole artifact bundle into one PowerShell stdin', async () => { - mkdirSync(join(localDir, 'node'), { recursive: true }) - // Comfortably past the ~50KB point at which the reporter measured PowerShell 5.1 wedging. - writeFileSync(join(localDir, 'node', 'relay.js'), Buffer.alloc(600 * 1024, 0x61)) - writeFileSync(join(localDir, 'index.js'), Buffer.alloc(300 * 1024, 0x62)) +afterEach(async () => { + delete process.env.ORCA_SYSTEM_SFTP_PATH + await rm(localDir, { recursive: true, force: true }) +}) - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - - const largest = Math.max(...commands.map((command) => command.stdin.length)) - expect(largest).toBeLessThanOrEqual(WINDOWS_STDIN_WRITE_CHUNK_BYTES) - // The base64 + JSON envelope is gone entirely: nothing reads the bundle as one string. - expect(commands.some((command) => command.script.includes('FromBase64String'))).toBe(false) - // `[Console]::In` wedged at 50KB where the stream reader did not, so the mkdir batch — the one - // payload still read as a string — must use the reader the reporter measured surviving. - expect(commands.some((command) => command.script.includes('[Console]::In.ReadToEnd()'))).toBe( - false - ) - expect( - commands.filter((command) => command.script.includes('StreamReader([Console]::')) - ).toHaveLength(1) - }) - - it('bounds the single-file upload too, which is the path large files take', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) +describe('Windows upload over sftp', () => { + it('moves the payload without any remote process reading a stdin', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 60 + 11, 0x64) const localPath = join(localDir, 'big.node') writeFileSync(localPath, contents) await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/big.node`, { hostPlatform }) - const writes = fileWrites() - expect(writes).toHaveLength(4) - expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( - WINDOWS_STDIN_WRITE_CHUNK_BYTES - ) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. - expect( - waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) - ).toBe(true) + // The defect is a remote stdin read; the fix is that there is not one. + expect(fileWrites()).toHaveLength(0) + expect(putLines()).toHaveLength(1) + // One transfer, not 61 execs: the whole point of the change. + expect(sftpBatches).toHaveLength(1) }) - it('writes every byte of every artifact across the chunked writes', async () => { - const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 2 + 17, 0x63) - writeFileSync(join(localDir, 'relay.js'), contents) + it('creates the parent chain and sends the payload in one round trip', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') - await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/a/b/relay.js`, { + hostPlatform + }) - const writes = fileWrites() - expect(writes).toHaveLength(3) - expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) - // Only the first write creates the staging file; the rest must extend it or it is truncated. - expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append']) + expect(sftpBatches).toHaveLength(1) + expect(sftpBatches[0]!.script.split('\n').filter(Boolean)).toEqual([ + '-mkdir "/C:/Users"', + '-mkdir "/C:/Users/dev"', + '-mkdir "/C:/Users/dev/.orca-remote"', + '-mkdir "/C:/Users/dev/.orca-remote/a"', + '-mkdir "/C:/Users/dev/.orca-remote/a/b"', + expect.stringContaining('put ') as unknown as string + ]) + }) + + it('addresses the destination in the drive-rooted namespace sftp exposes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A backslash destination silently writes a file named `C` and still exits 0, so the leading + // slash and forward separators are correctness, not style. + expect(putDestination(putLines()[0]!)).toMatch( + /^\/C:\/Users\/dev\/\.orca-remote\/relay\.js\.orca-partial-[0-9a-f]{12}$/ + ) + }) + + it('never lands a partial under the real name, and publishes by rename', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const remotePath = `${remoteRoot}/relay.js` + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), remotePath, { hostPlatform }) + + const destination = putDestination(putLines()[0]!) + // Assert the positive first: an unmatched regex yields '', which would satisfy the `not.toBe` + // below without this test ever having seen a destination. + expect(destination).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + expect(destination).not.toBe(`/C:${remotePath.slice(2)}`) + const publish = commands.at(-1)! + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // The publish reads the staged file, never a pipe, so it is safe on PowerShell 5.1. + expect(publish.script).not.toContain('OpenStandardInput') + }) + + it('never deletes the destination it is replacing', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const publish = commands.at(-1)! + // Delete-then-move destroys the user's existing file outright if the move then fails, and + // exposes a window where a reader sees no file at all — worse than the truncated partial the + // staging discipline exists to prevent. `File.Replace` is the atomic swap. + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + expect(publish.script).toContain( + '[System.IO.File]::Replace($staging, $path, [NullString]::Value)' + ) + // An absent destination cannot be Replaced, so that case falls back to a plain Move. + expect(publish.script).toContain( + 'catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }' + ) + }) + + it('gives every attempt its own staging name, so a retry cannot meet a predecessor lock', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + const [first, second] = putLines().map(putDestination) + expect(first).toContain(WINDOWS_STAGED_WRITE_SUFFIX) + // Losing contact is not evidence the previous writer died, so the name must not be reused. + expect(second).not.toBe(first) + }) + + it('enforces exclusive at the rename, where it is atomic', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + }) + + it('appends by concatenating the staged file, not by piping bytes to the remote', async () => { + await writeBufferViaSystemSsh(target, `${remoteRoot}/log.bin`, Buffer.from('tail'), { + hostPlatform, + append: true + }) + + expect(fileWrites()).toHaveLength(0) + const publish = commands.at(-1)! + expect(publish.script).toContain('FileMode]::Append') + expect(publish.script).toContain('$in.CopyTo($out)') + expect(publish.script).toContain('[System.IO.File]::Delete($staging)') }) it('still creates an empty artifact on the host', async () => { @@ -175,80 +338,310 @@ describe('Windows upload stdin framing', () => { await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - expect(fileWrites().map(writtenPath)).toEqual([`${remoteRoot}/empty.txt`]) - expect(fileWrites()[0].stdin).toHaveLength(0) - expect(fileMode(fileWrites()[0])).toBe('Create') + expect(putLines()).toHaveLength(1) + expect(commands.at(-1)!.script).toContain('[System.IO.File]::Move($staging, $path)') }) - it('lands a multi-chunk write on a staging path and publishes it by rename', async () => { - const remotePath = `${remoteRoot}/relay.js` - writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + it('writes a buffer through a 0600 temp file that does not outlive the transfer', async () => { + const seen: { path: string; contents: Buffer; mode: number }[] = [] + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + for (const line of spec.input.split('\n').filter((entry) => entry.startsWith('put '))) { + const path = putSource(line) + seen.push({ + path, + contents: await readFile(path), + mode: (await stat(path)).mode & 0o777 + }) + } + return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false } + }) + + await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { + hostPlatform + }) + + expect(seen).toHaveLength(1) + expect(seen[0]!.contents.toString()).toBe('1.2.3') + // The payload can be repository content and tmpdir is world-readable on every platform, so the + // window between write and upload must not be group- or world-readable. + expect(seen[0]!.mode).toBe(0o600) + await expect(readFile(seen[0]!.path)).rejects.toThrow() + }) + + it('creates upload directories over sftp rather than a PowerShell stdin batch', async () => { + mkdirSync(join(localDir, 'node'), { recursive: true }) + writeFileSync(join(localDir, 'node', 'relay.js'), 'x') await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) - // Nothing touches the real name until every byte is on the host. - expect(fileWrites().map(writtenPath)).toEqual([ - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`, - `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` - ]) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).toContain('[System.IO.File]::Delete($path)') + // Anchor on a non-empty observation: `some` is false of an empty list, so this would pass even + // if no command had been recorded at all. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('StreamReader([Console]::'))).toBe( + false + ) + expect(sftpBatches[0]!.script).toContain('-mkdir "/C:/Users/dev/.orca-remote"') + }) + + it('sweeps the staged bytes when the publish is the thing that fails', async () => { + writeFileSync(join(localDir, 'import.bin'), 'x') + // An exclusive conflict is the ordinary way to get here: the payload is on the host, and the + // rename that would have given it a name refuses. + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const script = decodePowerShellCommand(command) + return createFakeChannel((channel) => { + commands.push({ script, executable: command.split(' ')[0] ?? '', stdin: channel.written }) + const failed = script.includes('::Move($staging, $path)') + setImmediate(() => channel.emit('close', failed ? 1 : 0, null)) + }) + }) + + await expect( + uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + ).rejects.toThrow() + + const sweep = commands.at(-1)! + expect(sweep.script).toContain('[System.IO.File]::Delete($staging)') + // Tolerated, not asserted: the previous writer may still hold the file, and losing contact is + // not evidence it died. + expect(sweep.script).toContain('$ErrorActionPreference = "SilentlyContinue"') + }) + + it('reports a cancelled transfer as an abort, not as a failed one', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + const controller = new AbortController() + // runProcess reports the kill as a non-zero exit rather than throwing, so without checking the + // signal first a user pressing cancel is indistinguishable from the transfer genuinely failing. + runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => { + sftpBatches.push({ args: spec.args, script: spec.input }) + controller.abort() + return { code: 255, signal: 'SIGTERM', stdout: '', stderr: '', timedOut: false } + }) + + let error: Error | undefined + try { + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + signal: controller.signal + }) + } catch (thrown) { + error = thrown as Error + } + + expect(error?.name).toBe('AbortError') + expect(error?.message).not.toContain('sftp batch failed') + // A cancel is also not evidence about the host, so it must not send later writes to the slow + // path, and must not fall through to the defective reader now. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + expect(fileWrites()).toHaveLength(0) + }) + + it('does not let one unaddressable path become a verdict about the host', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + // A UNC destination has no settled mapping in sftp's drive-rooted namespace, so this write + // falls back — but the host still serves sftp perfectly well for every other path. + await uploadFileViaSystemSsh( + target, + join(localDir, 'relay.js'), + '//fileserver/share/relay.js', + { hostPlatform } + ) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(sftpBatches).toHaveLength(0) + // The 30-minute capability cache is keyed by host; caching this would send every later write + // to the same machine down the defective path on the strength of one odd destination. + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps using sftp for the next file after one path it could not spell', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), '//fileserver/share/a.js', { + hostPlatform + }) + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/b.js`, { + hostPlatform + }) + + expect(putLines()).toHaveLength(1) + expect(putDestination(putLines()[0]!)).toContain('/C:/Users/dev/.orca-remote/b.js') + }) + + it('does not let a local filename sftp cannot quote become a verdict either', async () => { + // POSIX clients allow a newline in a filename, and sftp's batch lexer would read it as the end + // of one command and the start of another. + const awkward = join(localDir, 'two\nlines.js') + writeFileSync(awkward, 'x') + + await uploadFileViaSystemSsh(target, awkward, `${remoteRoot}/relay.js`, { hostPlatform }) + + expect(fileWrites().length).toBeGreaterThan(0) + expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('translates the ssh argument list rather than passing it to a client that reads it differently', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform, + disableControlMaster: true + }) + + const args = sftpBatches[0]!.args + // sftp's `-T` does not exist, its `-p` preserves mtime, and its `-S` names a program to run. + expect(args).not.toContain('-T') + expect(args).not.toContain('-p') + expect(args).not.toContain('-S') + expect(args).toContain('ControlPath=none') + expect(args).toContain('ServerAliveInterval=15') + }) +}) + +describe('Windows upload on a host with no sftp subsystem', () => { + beforeEach(() => { + refuseSftp() + }) + + it('creates a multi-directory tree, which the one-element case never exercised', async () => { + mkdirSync(join(localDir, 'node', 'deep'), { recursive: true }) + writeFileSync(join(localDir, 'index.js'), 'a') + writeFileSync(join(localDir, 'node', 'deep', 'x.js'), 'b') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const mkdir = commands.find((command) => command.script.includes('ConvertFrom-Json'))! + // `@($json | ConvertFrom-Json)` wraps the parsed array in another array, so the loop variable + // binds to the whole thing and `[string]` of it is the paths joined by spaces — which + // CreateDirectory rejects. It only ever worked for a single directory, where stringifying a + // one-element array happens to yield the element, so no batch of one can catch this. + expect(mkdir.script).toContain('[string[]]($json | ConvertFrom-Json)') + expect(mkdir.script).not.toContain('@($json | ConvertFrom-Json)') + const batch = JSON.parse(mkdir.stdin.toString('utf-8')) as string[] + expect(batch.length).toBeGreaterThan(1) + }) + + it('falls back rather than failing the transfer', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 5, 0x61) + writeFileSync(join(localDir, 'relay.js'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(Buffer.concat(fileWrites().map((write) => write.stdin)).equals(contents)).toBe(true) + }) + + it('remembers the refusal, so a multi-file upload probes once', async () => { + writeFileSync(join(localDir, 'a.js'), 'a') + writeFileSync(join(localDir, 'b.js'), 'b') + writeFileSync(join(localDir, 'c.js'), 'c') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + // One refusal is enough; re-probing per file is a wasted round trip on every file. + expect(sftpBatches).toHaveLength(1) + }) + + it('does not spend a sweep round trip when sftp declined before moving any bytes', async () => { + writeFileSync(join(localDir, 'relay.js'), 'x') + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + // A refused subsystem staged nothing, so there is nothing to delete — and on a host without + // sftp that sweep would otherwise be paid on every single write. + expect(commands.length).toBeGreaterThan(0) + expect(commands.some((command) => command.script.includes('Delete($staging)'))).toBe(false) + }) + + it('prefers PowerShell 7, which reads a redirected stdin correctly', async () => { + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) + + expect(fileWrites().map((write) => write.executable)).toEqual(['pwsh.exe']) + // PowerShell 7 took 2MB through one exec when measured, so chunking it buys nothing. + expect(fileWrites()[0]!.stdin).toHaveLength(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3) + }) + + it('bounds every write when only Windows PowerShell 5.1 is available', async () => { + refusePwsh() + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) + writeFileSync(join(localDir, 'big.node'), contents) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + const writes = fileWrites().filter((write) => write.executable === 'powershell.exe') + expect(writes).toHaveLength(4) + expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( + WINDOWS_STDIN_WRITE_CHUNK_BYTES + ) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append', 'Append']) + // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. + // Count first: `every` is true of zero calls, so a wait that moved to a different helper would + // pass this silently. + expect(waitForChannelCloseSpy.mock.calls.length).toBeGreaterThan(0) + expect( + waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ).toBe(true) + }) + + it('remembers that PowerShell 7 is absent instead of re-probing per chunk', async () => { + refusePwsh() + writeFileSync(join(localDir, 'big.node'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + + await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, { + hostPlatform + }) + + expect(fileWrites().filter((write) => write.executable === 'pwsh.exe')).toHaveLength(1) }) it('leaves no truncated file under the real name when a chunk fails mid-file', async () => { writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) - // Spawns: 0 = mkdir batch, 1..3 = chunk writes. Fail the second chunk. - failAtSpawn = 2 + // Spawn 0 is the pwsh write; fail it and every retry beneath it. + failAtSpawn = 0 await expect( - uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, { + hostPlatform + }) ).rejects.toThrow() + expect(fileWrites().length).toBeGreaterThan(0) expect(fileWrites().map(writtenPath)).not.toContain(`${remoteRoot}/relay.js`) expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) }) +}) - it('enforces exclusive once at the rename, so a retry is not blocked by its own leftovers', async () => { - const localPath = join(localDir, 'import.bin') - writeFileSync(localPath, Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) +describe('last-resort Windows PowerShell failure reporting', () => { + it('names the host limitation and its remedy, not just the timeout', () => { + const timeout = new Error('write C:/x at offset 0 timed out after 60000ms with no response') - await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/import.bin`, { - hostPlatform, - exclusive: true - }) + const explained = explainWindowsPowerShellStdinFailure(timeout) as Error - // CreateNew on chunk one would fail against a leftover staging file from a failed attempt; - // `File::Move` raising on an existing destination is what carries the exclusive contract. - expect(fileWrites().map(fileMode)).toEqual(['Create', 'Append']) - const publish = commands.at(-1)! - expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') - expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + // "timed out" alone sends the user to retry a network they cannot fix; the fix is host-side. + expect(explained.message).toContain('Windows PowerShell 5.1') + expect(explained.message).toContain('Subsystem sftp sftp-server.exe') + expect(explained.cause).toBe(timeout) }) - it('keeps a single-chunk write on the destination, with the caller mode intact', async () => { - await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { - hostPlatform, - exclusive: true - }) + it('leaves a real failure alone, so a permission error is not reported as a host limitation', () => { + const denied = new Error('write C:/x at offset 0 failed (exit 1): Access to the path is denied') - expect(fileWrites()).toHaveLength(1) - expect(writtenPath(fileWrites()[0])).toBe(`${remoteRoot}/version`) - expect(fileMode(fileWrites()[0])).toBe('CreateNew') - expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) - }) - - it('appends onto the destination rather than staging, since append cannot be staged', async () => { - const remotePath = `${remoteRoot}/log.bin` - await writeBufferViaSystemSsh( - target, - remotePath, - Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1), - { hostPlatform, append: true } - ) - - expect(fileWrites().map(writtenPath)).toEqual([remotePath, remotePath]) - expect(fileWrites().map(fileMode)).toEqual(['Append', 'Append']) + expect(explainWindowsPowerShellStdinFailure(denied)).toBe(denied) }) }) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.test.ts b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts new file mode 100644 index 00000000000..ad723d592b0 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.test.ts @@ -0,0 +1,79 @@ +/** + * Whether a Windows host has an sftp subsystem is a fact about that host, so the cache is keyed by + * the endpoint that executes rather than by Orca's target id — otherwise a hardened host is + * re-probed once per file, and two targets pointing at one machine learn the same fact twice. + */ +import { afterEach, describe, expect, it } from 'vitest' +import type { SshTarget } from '../../shared/ssh-types' +import { + clearWindowsRemoteWriteCapabilitiesForTests, + getWindowsRemoteWriteCapabilities, + getWindowsRemoteWriteExecutionHostKey +} from './system-ssh-windows-write-capabilities' + +const asTarget = (fields: Partial): SshTarget => fields as SshTarget + +afterEach(() => { + clearWindowsRemoteWriteCapabilitiesForTests() +}) + +describe('getWindowsRemoteWriteExecutionHostKey', () => { + it('gives two targets on one endpoint the same key', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + // A target re-created under a new id has not changed what the host supports. + expect(getWindowsRemoteWriteExecutionHostKey(first)).toBe( + getWindowsRemoteWriteExecutionHostKey(second) + ) + }) + + it('separates hosts, ports and users', () => { + const base = { id: 'a', host: 'win.example', username: 'dev', port: 22 } + const keys = [ + asTarget(base), + asTarget({ ...base, host: 'other.example' }), + asTarget({ ...base, port: 2222 }), + asTarget({ ...base, username: 'ops' }) + ].map(getWindowsRemoteWriteExecutionHostKey) + + expect(new Set(keys).size).toBe(4) + }) + + it('keys a config alias by the alias, since ssh_config decides where it lands', () => { + const alias = asTarget({ id: 'a', host: 'stale.example', configHost: 'winbox' }) + + expect(getWindowsRemoteWriteExecutionHostKey(alias)).toBe('config:winbox') + }) +}) + +describe('getWindowsRemoteWriteCapabilities', () => { + it('shares one cache across targets that reach the same host', () => { + const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(first).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(second).shouldTry('sftp-subsystem')).toBe(false) + }) + + it('does not let one host answer for another', () => { + const hardened = asTarget({ id: 'a', host: 'hardened.example', username: 'dev', port: 22 }) + const ordinary = asTarget({ id: 'b', host: 'ordinary.example', username: 'dev', port: 22 }) + + getWindowsRemoteWriteCapabilities(hardened).rememberUnsupported('sftp-subsystem') + + expect(getWindowsRemoteWriteCapabilities(ordinary).shouldTry('sftp-subsystem')).toBe(true) + }) + + it('keeps the two capabilities independent', () => { + const target = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 }) + const capabilities = getWindowsRemoteWriteCapabilities(target) + + capabilities.rememberUnsupported('pwsh') + + // No PowerShell 7 says nothing about whether the host will serve sftp. + expect(capabilities.shouldTry('sftp-subsystem')).toBe(true) + expect(capabilities.shouldTry('pwsh')).toBe(false) + }) +}) diff --git a/src/main/ssh/system-ssh-windows-write-capabilities.ts b/src/main/ssh/system-ssh-windows-write-capabilities.ts new file mode 100644 index 00000000000..dcd03f19807 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-capabilities.ts @@ -0,0 +1,52 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { CapabilityProbeCache } from '../../shared/capability-probe-cache' + +/** + * Whether a Windows host can take a file write over the sftp subsystem, and whether it has a + * PowerShell 7 to fall back to. Both are host facts, so they are cached per execution host rather + * than per transfer — a hardened host with `Subsystem sftp` removed must not be re-probed on every + * file of a multi-file upload. + */ +export type WindowsRemoteWriteCapability = 'sftp-subsystem' | 'pwsh' + +// Why re-probe at all: an admin can enable the subsystem, or install PowerShell 7, without the +// user restarting Orca. Long enough that a hardened host costs one failed probe per half hour. +export const WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS = 30 * 60_000 + +const capabilitiesByExecutionHost = new Map< + string, + CapabilityProbeCache +>() + +/** + * Keyed by the endpoint that executes, not by target id: two Orca targets pointing at one host + * describe the same sshd, and a target re-created under a new id has not changed what that host + * supports. A config alias is its own key because ssh_config, not Orca, resolves where it lands. + */ +export function getWindowsRemoteWriteExecutionHostKey(target: SshTarget): string { + if (target.configHost) { + return `config:${target.configHost}` + } + const port = target.port ?? 22 + return target.username + ? `host:${target.username}@${target.host}:${port}` + : `host:${target.host}:${port}` +} + +export function getWindowsRemoteWriteCapabilities( + target: SshTarget +): CapabilityProbeCache { + const key = getWindowsRemoteWriteExecutionHostKey(target) + let cache = capabilitiesByExecutionHost.get(key) + if (!cache) { + cache = new CapabilityProbeCache( + WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS + ) + capabilitiesByExecutionHost.set(key, cache) + } + return cache +} + +export function clearWindowsRemoteWriteCapabilitiesForTests(): void { + capabilitiesByExecutionHost.clear() +} diff --git a/src/main/ssh/system-ssh-windows-write-strategy.ts b/src/main/ssh/system-ssh-windows-write-strategy.ts new file mode 100644 index 00000000000..f2cdca12516 --- /dev/null +++ b/src/main/ssh/system-ssh-windows-write-strategy.ts @@ -0,0 +1,329 @@ +import type { SshTarget } from '../../shared/ssh-types' +import { getSystemSshBuildArgsFromOperationOptions } from './system-ssh-args' +import { spawnSystemSshCommand } from './system-ssh-command' +import { + awaitWithSystemSshAbort, + throwIfAborted, + waitForChannelClose +} from './system-ssh-operation-lifecycle' +import { + isSftpPathUnsupportedError, + isSftpRefusalBeforeStaging, + isSftpUnavailableError, + runSftpBatch +} from './system-ssh-sftp-transfer' +import { quoteSftpBatchArgument, toSftpRemotePath } from './system-ssh-sftp-path' +import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities' +import { + makeWindowsDiscardStagedFileCommand, + makeWindowsPublishStagedFileCommand, + makeWindowsStagingPath, + makeWindowsWriteFileCommand, + windowsRemoteAncestorDirectories, + type WindowsPublishMode +} from './system-ssh-windows-file-write' + +/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ +export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 + +/** + * Bound on one stdin write for the last-resort Windows PowerShell 5.1 path. + * + * Measured on Windows 11 26200 / OpenSSH 10.0p2: a 32KB write still hangs 15 times in 120 under + * load, and no smaller value removes the risk. The defect is per blocking read, not per byte, so + * shrinking the chunk trades one risky read for more execs that each carry their own. This is a + * damage bound on a path known to be unreliable, not a safe size. + */ +export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 + +export type WindowsWriteOptions = Parameters< + typeof getSystemSshBuildArgsFromOperationOptions +>[0] & { + signal?: AbortSignal + append?: boolean + exclusive?: boolean +} + +/** Bytes to write, plus a way to present them to sftp, which can only send a local file. */ +export type WindowsWriteSource = { + totalBytes: number + readChunk: (offset: number, maxBytes: number) => Promise + withLocalFile: (send: (localPath: string) => Promise) => Promise +} + +function publishMode(options: WindowsWriteOptions): WindowsPublishMode { + return options.append ? 'append' : options.exclusive === true ? 'exclusive' : 'create' +} + +/** + * Writes one file to a Windows host, preferring transports that do not push bytes through a remote + * PowerShell's stdin. + * + * Order, and why: sftp carries the whole payload in one transfer and never has a remote process + * read a pipe. Measured on Windows 11 / OpenSSH 10.0p2: 1.9MB in a median 315ms over sftp against + * 0 of 6 completions on the chunked path, whose best case was ~62 execs at ~350ms each. PowerShell + * 7 reads a redirected stdin correctly but is not installed by default. Windows PowerShell 5.1 is + * always present and is the defective reader, so it is last and it is bounded. + * + * Every transport stages under a unique name and publishes by rename, so no partial write is ever + * visible under the real name and no retry inherits a predecessor's lock. + */ +export async function writeWindowsRemoteFile( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + throwIfAborted(options.signal) + const capabilities = getWindowsRemoteWriteCapabilities(target) + await capabilities.runWithFallback( + 'sftp-subsystem', + () => writeViaSftp(target, remotePath, source, options), + () => writeViaRemoteStdin(target, remotePath, source, options), + isSftpUnavailableError + ) +} + +/** + * Stages under a name nothing else can own, publishes it, and sweeps the staging file if either + * step fails. + * + * Shared by both transports so the cleanup contract cannot drift between them: a failed publish — + * an exclusive conflict is the ordinary case — leaves bytes on the host that no longer have a + * purpose, and the sweep is what stops them accumulating. + */ +async function stageThenPublish( + target: SshTarget, + remotePath: string, + options: WindowsWriteOptions, + stage: (stagingPath: string) => Promise, + nothingStaged: (error: unknown) => boolean = () => false +): Promise { + const stagingPath = makeWindowsStagingPath(remotePath) + try { + await stage(stagingPath) + await publishStagedWrite(target, stagingPath, remotePath, options) + } catch (error) { + // A transport that declined before it moved any bytes has nothing to sweep, and sweeping + // anyway would spend a round trip on every write to a host that has no sftp subsystem. + if (!nothingStaged(error)) { + await discardStagedWrite(target, stagingPath, options) + } + throw error + } +} + +/** + * A path sftp cannot address falls back for this write alone, without touching the host verdict. + * + * The distinction matters because the capability cache is keyed by host and holds for half an hour: + * routing one UNC destination, or one local filename containing a newline, into + * `rememberUnsupported` would send every later write to that host down the defective path too. + */ +async function writeViaSftp( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + try { + await attemptSftpWrite(target, remotePath, source, options) + } catch (error) { + if (!isSftpPathUnsupportedError(error)) { + throw error + } + await writeViaRemoteStdin(target, remotePath, source, options) + } +} + +function attemptSftpWrite( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const mkdirs = windowsRemoteAncestorDirectories(remotePath).map( + (directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}` + ) + return stageThenPublish( + target, + remotePath, + options, + (stagingPath) => + source.withLocalFile((localPath) => + // One round trip: the parent chain and the payload travel in the same batch. + runSftpBatch( + target, + [ + ...mkdirs, + `put ${quoteSftpBatchArgument(localPath)} ${quoteSftpBatchArgument(toSftpRemotePath(stagingPath))}` + ], + options + ) + ), + isSftpRefusalBeforeStaging + ) +} + +function writeViaRemoteStdin( + target: SshTarget, + remotePath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions +): Promise { + const capabilities = getWindowsRemoteWriteCapabilities(target) + return stageThenPublish(target, remotePath, options, (stagingPath) => + capabilities.runWithFallback( + 'pwsh', + () => writeStdinChunks(target, stagingPath, source, options, 'pwsh.exe'), + () => writeStdinChunks(target, stagingPath, source, options, 'powershell.exe'), + isPwshUnavailableError + ) + ) +} + +/** + * PowerShell 7 takes the whole payload in one exec — measured at 2MB — so only the 5.1 path pays + * for chunking, and only because a bounded write is the most that path can be trusted with. + */ +async function writeStdinChunks( + target: SshTarget, + stagingPath: string, + source: WindowsWriteSource, + options: WindowsWriteOptions, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + const chunkBytes = + executable === 'pwsh.exe' ? Math.max(source.totalBytes, 1) : WINDOWS_STDIN_WRITE_CHUNK_BYTES + let offset = 0 + // An empty write still has to run: it is what creates the staged file. + do { + const chunk = await source.readChunk(offset, chunkBytes) + if (chunk.length === 0 && offset < source.totalBytes) { + throw new Error(`Source ran short during upload of ${stagingPath}`) + } + await writeOneStdinChunk( + target, + stagingPath, + chunk, + { ...options, append: offset > 0, exclusive: false }, + offset, + executable + ) + offset += chunk.length + } while (offset < source.totalBytes) +} + +async function writeOneStdinChunk( + target: SshTarget, + stagingPath: string, + chunk: Buffer, + options: WindowsWriteOptions, + offset: number, + executable: 'powershell.exe' | 'pwsh.exe' +): Promise { + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand( + target, + makeWindowsWriteFileCommand(stagingPath, { + append: options.append, + exclusive: options.exclusive, + executable + }), + { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } + ) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose( + channel, + `write ${stagingPath} at offset ${offset}`, + WINDOWS_STDIN_WRITE_TIMEOUT_MS + ) + ).catch((error: unknown) => { + throw executable === 'powershell.exe' ? explainWindowsPowerShellStdinFailure(error) : error + }) + if (!options.signal?.aborted) { + channel.stdin.end(chunk) + } + await closePromise +} + +/** + * Names the cause on the one path that can hang, so the failure is not just "timed out". + * + * A user seeing this needs to know it is a host limitation with a host-side remedy, not a network + * fault they should retry into. + */ +export function explainWindowsPowerShellStdinFailure(error: unknown): unknown { + const message = error instanceof Error ? error.message : String(error) + if (!/timed out/i.test(message)) { + return error + } + return new Error( + `${message}\nWindows PowerShell 5.1 can lose a redirected stdin permanently when a read finds it momentarily empty, so this write cannot be made reliable from the client. Enable the sftp subsystem on the host (sshd_config: "Subsystem sftp sftp-server.exe"), or install PowerShell 7, and Orca will use it automatically.`, + { cause: error instanceof Error ? error : undefined } + ) +} + +function isPwshUnavailableError(error: unknown): boolean { + const message = error instanceof Error ? error.message : String(error) + // cmd.exe's "not recognized" and sshd's exit 9009 both mean "no pwsh here". A timeout does not: + // that is the stdin defect, and PowerShell 7 does not have it, so it must not be cached as absent. + return /is not recognized as an internal or external command|9009|CommandNotFoundException/i.test( + message + ) +} + +async function publishStagedWrite( + target: SshTarget, + stagingPath: string, + remotePath: string, + options: WindowsWriteOptions +): Promise { + await runWindowsCommandWithoutStdin( + target, + makeWindowsPublishStagedFileCommand(stagingPath, remotePath, publishMode(options)), + `publish ${remotePath}`, + options + ) +} + +async function discardStagedWrite( + target: SshTarget, + stagingPath: string, + options: WindowsWriteOptions +): Promise { + try { + await runWindowsCommandWithoutStdin( + target, + makeWindowsDiscardStagedFileCommand(stagingPath), + `discard ${stagingPath}`, + { ...options, signal: undefined } + ) + } catch { + // Housekeeping only. The staging name is unique, so a leftover blocks nothing, and a failure + // here says nothing about whether the abandoned writer is still alive. + } +} + +function runWindowsCommandWithoutStdin( + target: SshTarget, + command: string, + label: string, + options: WindowsWriteOptions +): Promise { + const channel = spawnSystemSshCommand(target, command, { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, label, WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end() + } + return closePromise +}