diff --git a/src/main/providers/ssh-pty-provider-terminal-repair.test.ts b/src/main/providers/ssh-pty-provider-terminal-repair.test.ts new file mode 100644 index 00000000000..530ce2804d8 --- /dev/null +++ b/src/main/providers/ssh-pty-provider-terminal-repair.test.ts @@ -0,0 +1,155 @@ +// The client half of #17830: a spawn refused for an unloadable node-pty must route into a repair +// instead of printing a paragraph. Covers the seam only — the ledger and the locked rebuild are +// tested in src/main/ssh/ssh-relay-node-pty-repair.test.ts and +// src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts. + +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { SshPtyProvider } from './ssh-pty-provider' +import { createMockMux, type MockMultiplexer } from './ssh-pty-provider-mock-multiplexer' +import { + TERMINAL_UNAVAILABLE_RPC_ERROR_CODE, + type TerminalUnavailableCause +} from '../../shared/terminal-unavailable-cause' + +const UNAVAILABLE_MESSAGE = + "Remote terminals are unavailable: this host's node-pty binary was built for Node ABI 108." + +function cause(overrides: Partial = {}): TerminalUnavailableCause { + return { + status: 'blocked', + reason: 'abi_mismatch', + detail: 'built for NODE_MODULE_VERSION 108, this Node accepts 115', + repairable: true, + host: { + platform: 'linux', + arch: 'x64', + libc: 'glibc', + glibcVersion: '2.31', + nodeAbi: '115', + nodeVersion: 'v20.11.0' + }, + ...overrides + } +} + +/** Shaped exactly as the multiplexer rebuilds a JSON-RPC error response client-side. */ +function relayRejection(data: unknown): Error { + const error = new Error(UNAVAILABLE_MESSAGE) + Object.defineProperty(error, 'code', { value: TERMINAL_UNAVAILABLE_RPC_ERROR_CODE }) + Object.defineProperty(error, 'data', { value: data }) + return error +} + +function spawnCallCount(target: MockMultiplexer): number { + return target.request.mock.calls.filter((call) => call[0] === 'pty.spawn').length +} + +function rejectSpawnOnce(mux: MockMultiplexer, error: Error): void { + mux.request.mockImplementation(async (method: string) => { + if (method === 'pty.spawn') { + throw error + } + return undefined + }) +} + +const SPAWN_OPTS = { cwd: '/repo', cols: 80, rows: 24 } + +let mux: MockMultiplexer +let provider: SshPtyProvider + +beforeEach(() => { + mux = createMockMux() + provider = new SshPtyProvider('conn-1', mux as never) +}) + +describe('terminal-unavailable spawn recovery', () => { + it('routes a repairable cause into recovery and retries once on the repaired provider', async () => { + rejectSpawnOnce(mux, relayRejection(cause())) + const repairedMux = createMockMux() + repairedMux.request.mockResolvedValue({ id: 'pty-1', incarnationId: 'incarnation-1' }) + const repairedProvider = new SshPtyProvider('conn-1', repairedMux as never) + const recover = vi.fn(async () => repairedProvider) + provider.setTerminalUnavailableRecovery(recover) + + const result = await provider.spawn(SPAWN_OPTS) + + expect(result.id).toBe('ssh:conn-1@@pty-1') + expect(recover).toHaveBeenCalledTimes(1) + expect(recover).toHaveBeenCalledWith( + expect.objectContaining({ reason: 'abi_mismatch', status: 'blocked' }) + ) + // Exactly one retry, on the post-repair channel — never a second attempt on the broken one. + expect(spawnCallCount(mux)).toBe(1) + expect(spawnCallCount(repairedMux)).toBe(1) + }) + + it('surfaces the relay message when the retry still fails, and does not recurse into a second repair', async () => { + // The lock-busy shape: the reconnect happened, the rebuild did not, so the relay says the same thing. + rejectSpawnOnce(mux, relayRejection(cause())) + const degradedMux = createMockMux() + const degradedProvider = new SshPtyProvider('conn-1', degradedMux as never) + rejectSpawnOnce(degradedMux, relayRejection(cause())) + const degradedRecover = vi.fn(async () => degradedProvider) + degradedProvider.setTerminalUnavailableRecovery(degradedRecover) + const recover = vi.fn(async () => degradedProvider) + provider.setTerminalUnavailableRecovery(recover) + + await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE) + + expect(recover).toHaveBeenCalledTimes(1) + expect(degradedRecover).not.toHaveBeenCalled() + }) + + it('rethrows without recovery when the recovery declines', async () => { + rejectSpawnOnce(mux, relayRejection(cause())) + provider.setTerminalUnavailableRecovery(async () => null) + + await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE) + }) + + it('never recovers from an unverifiable cause', async () => { + rejectSpawnOnce(mux, relayRejection(cause({ status: 'unverifiable', repairable: true }))) + const recover = vi.fn(async () => provider) + provider.setTerminalUnavailableRecovery(recover) + + await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE) + expect(recover).not.toHaveBeenCalled() + }) + + it('never recovers from a toolchain_missing cause', async () => { + rejectSpawnOnce(mux, relayRejection(cause({ reason: 'toolchain_missing', repairable: false }))) + const recover = vi.fn(async () => provider) + provider.setTerminalUnavailableRecovery(recover) + + await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE) + expect(recover).not.toHaveBeenCalled() + }) + + it('ignores a malformed cause rather than half-reading it', async () => { + rejectSpawnOnce(mux, relayRejection({ status: 'blocked', repairable: true })) + const recover = vi.fn(async () => provider) + provider.setTerminalUnavailableRecovery(recover) + + await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE) + expect(recover).not.toHaveBeenCalled() + }) + + it('leaves an old relay that publishes no cause on today behaviour', async () => { + rejectSpawnOnce(mux, new Error(UNAVAILABLE_MESSAGE)) + const recover = vi.fn(async () => provider) + provider.setTerminalUnavailableRecovery(recover) + + await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE) + expect(recover).not.toHaveBeenCalled() + }) + + it('does not swallow an ordinary spawn failure', async () => { + rejectSpawnOnce(mux, new Error('shell not found')) + const recover = vi.fn(async () => provider) + provider.setTerminalUnavailableRecovery(recover) + + await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow('shell not found') + expect(recover).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/providers/ssh-pty-provider.ts b/src/main/providers/ssh-pty-provider.ts index a8e4218ce63..f218cc43eca 100644 --- a/src/main/providers/ssh-pty-provider.ts +++ b/src/main/providers/ssh-pty-provider.ts @@ -23,6 +23,7 @@ import { SshPtySpawnExitRaceTracker } from './ssh-pty-spawn-exit-race' import { SshAgentSessionCapabilities } from './ssh-agent-session-capabilities' import type { PtyProcessInspection } from './pty-process-inspection' import { writeToSshPty, writeToSshPtyWithSettlement } from './ssh-pty-write' +import { spawnWithTerminalRuntimeRepair, type TerminalRepairHook } from './ssh-pty-spawn-repair' // Why: sequential relay teardown calls share one absolute budget; convert to the mux-relative timeout only at dispatch. function relayTimeoutOptions(deadlineMs: number | undefined): { timeoutMs: number } | undefined { @@ -38,6 +39,7 @@ export class SshPtyProvider implements IPtyProvider { private readonly agentSessionCapabilities: SshAgentSessionCapabilities private spawnExitRaces = new SshPtySpawnExitRaceTracker() private readonly outputState: SshPtyProviderOutputState + private recoverFromTerminalUnavailable: TerminalRepairHook | null = null requestHostRpc: NonNullable = (method, params, options) => this.mux.request(method, params as Record, options) @@ -76,7 +78,24 @@ export class SshPtyProvider implements IPtyProvider { private toAppPtyId = (id: string): string => toAppSshPtyId(this.connectionId, id) + /** Installed by SshRelaySession, which owns the connection, the repair lock and the reconnect. */ + setTerminalUnavailableRecovery(recover: TerminalRepairHook): void { + this.recoverFromTerminalUnavailable = recover + } + + hasLivePtys(): boolean { + return this.livePtyIds.size > 0 + } + async spawn(opts: PtySpawnOptions): Promise { + return await spawnWithTerminalRuntimeRepair({ + attempt: () => this.spawnWithoutTerminalRuntimeRepair(opts), + recover: this.recoverFromTerminalUnavailable, + retry: (provider) => provider.spawnWithoutTerminalRuntimeRepair(opts) + }) + } + + private async spawnWithoutTerminalRuntimeRepair(opts: PtySpawnOptions): Promise { if (opts.agentSessionEnsure && opts.sessionId) { throw new Error('agent_session_claim_unavailable') } diff --git a/src/main/providers/ssh-pty-spawn-repair.ts b/src/main/providers/ssh-pty-spawn-repair.ts new file mode 100644 index 00000000000..5086283f3fb --- /dev/null +++ b/src/main/providers/ssh-pty-spawn-repair.ts @@ -0,0 +1,45 @@ +/** + * The client seam for #17830: a spawn the relay refused because it cannot load node-pty. + * + * Split out of ssh-pty-provider.ts so the provider keeps only the wiring. The repair itself is + * driven by SshRelaySession, which owns the connection, the repair lock and the reconnect. + */ +import { + mayRepairFromCause, + terminalUnavailableCauseFromError, + type TerminalUnavailableCause +} from '../../shared/terminal-unavailable-cause' + +/** Resolves to the provider registered after a successful repair, or null to keep the rejection. */ +export type TerminalRepairHook = ( + cause: TerminalUnavailableCause +) => Promise + +/** + * Run a spawn, and on a proved-repairable terminal-unavailable rejection repair the host and + * retry exactly once on the provider the repair produced. + * + * Re-issuing is safe because a validated cause is the host's own statement that admission was + * refused before any PTY existed, so there is nothing to duplicate. The retry deliberately goes + * through a caller-supplied thunk that does not re-enter this wrapper, so it cannot recurse. + * Gated on `mayRepairFromCause`, never on the peer's `repairable` flag alone. + */ +export async function spawnWithTerminalRuntimeRepair(args: { + attempt: () => Promise + recover: TerminalRepairHook | null + retry: (provider: TProvider) => Promise +}): Promise { + try { + return await args.attempt() + } catch (error) { + const cause = terminalUnavailableCauseFromError(error) + if (!cause || !mayRepairFromCause(cause) || !args.recover) { + throw error + } + const repaired = await args.recover(cause) + if (!repaired) { + throw error + } + return await args.retry(repaired) + } +} diff --git a/src/main/ssh/ssh-relay-node-pty-repair.test.ts b/src/main/ssh/ssh-relay-node-pty-repair.test.ts new file mode 100644 index 00000000000..7c1b1800259 --- /dev/null +++ b/src/main/ssh/ssh-relay-node-pty-repair.test.ts @@ -0,0 +1,204 @@ +// Why: the once-only ledger is the whole safety story here — an unbounded rebuild loop against a +// remote is worse than the bug it chases, so every gate that stops one gets a test. + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + forgetRelayNodePtyRepairs, + recoverRelayNodePtyForSpawn, + relayNodePtyRepairAttempts +} from './ssh-relay-node-pty-repair' +import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause' + +const HOST: TerminalUnavailableCause['host'] = { + platform: 'linux', + arch: 'x64', + libc: 'glibc', + glibcVersion: '2.31', + nodeAbi: '115', + nodeVersion: 'v20.11.0' +} + +function cause(overrides: Partial = {}): TerminalUnavailableCause { + return { + status: 'blocked', + reason: 'abi_mismatch', + detail: 'built for NODE_MODULE_VERSION 108, this Node accepts 115', + repairable: true, + host: HOST, + ...overrides + } +} + +describe('recoverRelayNodePtyForSpawn', () => { + const TARGET = 'host-a' + let warnSpy: ReturnType + + beforeEach(() => { + forgetRelayNodePtyRepairs(TARGET) + forgetRelayNodePtyRepairs('host-b') + warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {}) + }) + + afterEach(() => { + warnSpy.mockRestore() + forgetRelayNodePtyRepairs(TARGET) + forgetRelayNodePtyRepairs('host-b') + }) + + function harness(overrides: { hasLivePtys?: boolean; reconnect?: () => Promise } = {}) { + const reconnect = vi.fn(overrides.reconnect ?? (async () => {})) + const repaired = { name: 'post-repair-provider' } + const resolveProvider = vi.fn(() => repaired) + return { + reconnect, + resolveProvider, + repaired, + run: (c: TerminalUnavailableCause | null) => + recoverRelayNodePtyForSpawn({ + targetId: TARGET, + cause: c, + hasLivePtys: () => overrides.hasLivePtys === true, + reconnect, + resolveProvider + }) + } + } + + it('repairs a repairable cause with exactly one reconnect and hands back the new provider', async () => { + const h = harness() + + const result = await h.run(cause()) + + expect(result.outcome).toBe('repaired') + expect(result.provider).toBe(h.repaired) + expect(h.reconnect).toHaveBeenCalledTimes(1) + expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual(['abi_mismatch']) + }) + + it('does not repair a second time for the same cause on the same host', async () => { + const first = harness() + await first.run(cause()) + const second = harness() + + const result = await second.run(cause()) + + expect(result.outcome).toBe('already-attempted') + expect(result.provider).toBeNull() + expect(second.reconnect).not.toHaveBeenCalled() + }) + + it('spends the attempt even when the repair reconnect fails, so it cannot loop', async () => { + const failing = harness({ + reconnect: async () => { + throw new Error('relay repair lock is wedged') + } + }) + + const first = await failing.run(cause()) + expect(first.outcome).toBe('reconnect-failed') + expect(first.provider).toBeNull() + + const second = harness() + const retry = await second.run(cause()) + + expect(retry.outcome).toBe('already-attempted') + expect(second.reconnect).not.toHaveBeenCalled() + }) + + it('keeps the ledger per host, so a second host still gets its one attempt', async () => { + const h = harness() + await h.run(cause()) + const other = vi.fn(async () => {}) + + const result = await recoverRelayNodePtyForSpawn({ + targetId: 'host-b', + cause: cause(), + hasLivePtys: () => false, + reconnect: other, + resolveProvider: () => ({}) + }) + + expect(result.outcome).toBe('repaired') + expect(other).toHaveBeenCalledTimes(1) + }) + + it('keeps the ledger per reason, so a different proved fault still gets its one attempt', async () => { + const h = harness() + await h.run(cause()) + const second = harness() + + const result = await second.run(cause({ reason: 'arch_mismatch' })) + + expect(result.outcome).toBe('repaired') + expect([...relayNodePtyRepairAttempts(TARGET)].sort()).toEqual([ + 'abi_mismatch', + 'arch_mismatch' + ]) + }) + + it('never repairs an unverifiable cause, whatever the peer claims about repairability', async () => { + const h = harness() + + // Why repairable:true here: #14830 is exactly a peer flag believed over the status. + const result = await h.run(cause({ status: 'unverifiable', repairable: true })) + + expect(result.outcome).toBe('not-repairable') + expect(h.reconnect).not.toHaveBeenCalled() + expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual([]) + }) + + it('never repairs a toolchain_missing cause — the rebuild needs the missing compiler', async () => { + const h = harness() + + const result = await h.run(cause({ reason: 'toolchain_missing', repairable: false })) + + expect(result.outcome).toBe('not-repairable') + expect(h.reconnect).not.toHaveBeenCalled() + expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual([]) + }) + + it('never repairs when the relay published no cause at all', async () => { + const h = harness() + + const result = await h.run(null) + + expect(result.outcome).toBe('not-repairable') + expect(h.reconnect).not.toHaveBeenCalled() + }) + + it('does not rebuild under live PTYs, and does not spend the attempt doing so', async () => { + const live = harness({ hasLivePtys: true }) + + const blocked = await live.run(cause()) + expect(blocked.outcome).toBe('ptys-live') + expect(live.reconnect).not.toHaveBeenCalled() + expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual([]) + + const idle = harness() + expect((await idle.run(cause())).outcome).toBe('repaired') + }) + + it('withholds the retry when the reconnect produced no provider', async () => { + const reconnect = vi.fn(async () => {}) + + const result = await recoverRelayNodePtyForSpawn({ + targetId: TARGET, + cause: cause(), + hasLivePtys: () => false, + reconnect, + resolveProvider: () => null + }) + + expect(result.outcome).toBe('no-provider') + expect(result.provider).toBeNull() + expect(reconnect).toHaveBeenCalledTimes(1) + }) + + it('gives a host a fresh attempt only after an explicit disconnect', async () => { + await harness().run(cause()) + forgetRelayNodePtyRepairs(TARGET) + const afterDisconnect = harness() + + expect((await afterDisconnect.run(cause())).outcome).toBe('repaired') + }) +}) diff --git a/src/main/ssh/ssh-relay-node-pty-repair.ts b/src/main/ssh/ssh-relay-node-pty-repair.ts new file mode 100644 index 00000000000..149b6c486cb --- /dev/null +++ b/src/main/ssh/ssh-relay-node-pty-repair.ts @@ -0,0 +1,115 @@ +/** + * Turning a spawn-time "remote terminals are unavailable" into an actual repair. + * + * The fault is proved on the relay at spawn time; the only thing that can fix it — + * `repairInstalledNativeDeps` in ssh-relay-deploy.ts — runs on the client during deploy, under + * `tryAcquireRelayRepairLock`. So the recovery here is deliberately indirect: it does not touch + * the remote `node_modules` itself, it drives one relay reconnect and lets the locked deploy path + * do the rebuild. All `node_modules` mutation stays behind that lock. + * + * The invariant that matters more than the recovery: **at most one repair per host per reason.** + * A rebuild loop against a remote is worse than the bug it is chasing, so the attempt is recorded + * before the reconnect starts, not after it succeeds. A repair that ran and failed is still an + * attempt, and this host will render the relay's message from then on. + */ +import { + mayRepairFromCause, + type TerminalUnavailableCause +} from '../../shared/terminal-unavailable-cause' + +/** targetId -> the cause reasons already spent on this host, for the life of the session. */ +const attemptedRepairsByTarget = new Map>() + +export type RelayNodePtyRepairOutcome = + /** The cause is not proved-and-rebuildable; render the relay's message. */ + | 'not-repairable' + /** This host already spent its one attempt on this reason. */ + | 'already-attempted' + /** The relay is still serving PTYs, so a rebuild under it is not accounted for. */ + | 'ptys-live' + /** No reconnect was possible, or it failed; the attempt is still spent. */ + | 'reconnect-failed' + /** Reconnected, but no provider came back to retry on. */ + | 'no-provider' + | 'repaired' + +/** Forget a host's spent attempts. Disconnect is user action, so it may earn a fresh one. */ +export function forgetRelayNodePtyRepairs(targetId: string): void { + attemptedRepairsByTarget.delete(targetId) +} + +/** Test/diagnostic view of the ledger. */ +export function relayNodePtyRepairAttempts(targetId: string): ReadonlySet { + return attemptedRepairsByTarget.get(targetId) ?? new Set() +} + +/** False when this host has already spent its attempt on this reason. Marks on success. */ +function claimRepairAttempt(targetId: string, reason: string): boolean { + const spent = attemptedRepairsByTarget.get(targetId) + if (spent) { + if (spent.has(reason)) { + return false + } + spent.add(reason) + return true + } + attemptedRepairsByTarget.set(targetId, new Set([reason])) + return true +} + +export type RelayNodePtyRepairRequest = { + targetId: string + /** Already parsed and schema-validated; null when the relay published no cause. */ + cause: TerminalUnavailableCause | null + /** Whether this client still holds live PTYs on the relay about to be rebuilt. */ + hasLivePtys: () => boolean + /** One relay reconnect, which runs the locked `repairInstalledNativeDeps`. */ + reconnect: () => Promise + /** The provider registered after the reconnect — never the one that failed. */ + resolveProvider: () => TProvider | null +} + +/** + * Drive one repair for a failed spawn, returning the provider to retry on. + * + * Gates on `mayRepairFromCause`, not on the peer's `repairable` flag: an `unverifiable` cause + * proves nothing and must never trigger a rebuild (#14830, docs/reference/ssh-execution-boundary.md). + */ +export async function recoverRelayNodePtyForSpawn( + request: RelayNodePtyRepairRequest +): Promise<{ outcome: RelayNodePtyRepairOutcome; provider: TProvider | null }> { + const { targetId, cause } = request + if (!mayRepairFromCause(cause) || !cause) { + return { outcome: 'not-repairable', provider: null } + } + // Why check before claiming: a live PTY means the rebuild's blast radius is unaccounted for, and + // that is a reason to wait, not a spent attempt. Costs nothing remote, so it cannot loop. + if (request.hasLivePtys()) { + console.warn( + `[ssh-relay-repair] Not rebuilding node-pty on ${targetId} for ${cause.reason}: the relay is still serving PTYs` + ) + return { outcome: 'ptys-live', provider: null } + } + if (!claimRepairAttempt(targetId, cause.reason)) { + console.warn( + `[ssh-relay-repair] node-pty repair for ${cause.reason} on ${targetId} already ran; not retrying` + ) + return { outcome: 'already-attempted', provider: null } + } + + console.warn( + `[ssh-relay-repair] Reconnecting ${targetId} once to rebuild node-pty (${cause.reason}): ${cause.detail}` + ) + try { + await request.reconnect() + } catch (error) { + console.warn( + `[ssh-relay-repair] Repair reconnect for ${targetId} failed: ${ + error instanceof Error ? error.message : String(error) + }` + ) + return { outcome: 'reconnect-failed', provider: null } + } + const provider = request.resolveProvider() + return provider ? { outcome: 'repaired', provider } : { outcome: 'no-provider', provider: null } +} diff --git a/src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts b/src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts new file mode 100644 index 00000000000..12a51ce6bdd --- /dev/null +++ b/src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts @@ -0,0 +1,288 @@ +// The other half of the #17830 recovery: proof that the reconnect a spawn-time cause triggers is +// the SAME locked deploy repair, not a second rebuild path. `tryAcquireRelayRepairLock` is the only +// thing standing between two clients and a concurrent `node_modules` rewrite, so a lock this path +// cannot take must degrade to the relay's message rather than proceed. + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as RelayInstallMarkerModule from './ssh-relay-install-marker' + +vi.mock('electron', () => ({ + app: { getAppPath: () => '/mock/app' } +})) + +vi.mock('fs', () => ({ + existsSync: vi.fn().mockReturnValue(true), + readFileSync: vi.fn().mockReturnValue('0.1.0+testhash') +})) + +vi.mock('./relay-protocol', () => ({ + RELAY_VERSION: '0.1.0', + RELAY_REMOTE_DIR: '.orca-remote', + parseUnameToRelayPlatform: vi.fn().mockReturnValue('linux-x64'), + RELAY_SENTINEL: 'ORCA-RELAY v0.1.0 READY\n', + RELAY_SENTINEL_TIMEOUT_MS: 10_000 +})) + +vi.mock('./ssh-relay-deploy-helpers', () => ({ + uploadDirectory: vi.fn().mockResolvedValue(undefined), + waitForSentinel: vi.fn().mockResolvedValue({ + write: vi.fn(), + onData: vi.fn(), + onClose: vi.fn() + }), + isUnconfirmedSshCommandTermination: () => false, + execCommand: vi.fn() +})) + +vi.mock('./ssh-remote-node-resolution', () => ({ + resolveRemoteNodePath: vi.fn().mockResolvedValue('/usr/bin/node') +})) + +vi.mock('./ssh-relay-install-marker', async (importOriginal) => ({ + ...(await importOriginal()), + createRelayInstallMarkerFileName: () => '.sftp-namespace-00000000000000000000000000000000' +})) + +vi.mock('./ssh-relay-versioned-install', () => ({ + readLocalFullVersion: vi.fn().mockReturnValue('0.1.0+testhash'), + computeRemoteRelayDir: (home: string, v: string) => `${home}/.orca-remote/relay-${v}`, + isRelayAlreadyInstalled: vi.fn().mockResolvedValue(true), + finalizeInstall: vi.fn().mockResolvedValue(undefined), + abandonInstall: vi.fn().mockResolvedValue(undefined), + gcOldRelayVersions: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-relay-install-lock', () => ({ + acquireInstallLock: vi.fn().mockResolvedValue(undefined), + RELAY_INSTALL_LOCK_NAME: '.install-lock' +})) + +vi.mock('./ssh-relay-repair-lock', () => ({ + tryAcquireRelayRepairLock: vi.fn().mockResolvedValue('acquired') +})) + +vi.mock('./ssh-relay-gc-claim', () => ({ + releaseRelayGcClaimWithRetry: vi.fn().mockResolvedValue('released'), + tryAcquireRelayGcClaim: vi.fn().mockResolvedValue('launch-token'), + waitForRelayGcClaimRelease: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-connection-utils', () => ({ + shellEscape: (s: string) => `'${s}'` +})) + +import { deployAndLaunchRelay } from './ssh-relay-deploy' +import { execCommand } from './ssh-relay-deploy-helpers' +import { parseUnameToRelayPlatform } from './relay-protocol' +import { isRelayAlreadyInstalled } from './ssh-relay-versioned-install' +import { tryAcquireRelayRepairLock } from './ssh-relay-repair-lock' +import { + makeMockConnection, + type ExecResponse, + type SftpWriteCapture +} from './ssh-relay-native-deps-install-fixture' +import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair' +import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause' + +const TARGET = 'repair-host' + +const ABI_MISMATCH: TerminalUnavailableCause = { + status: 'blocked', + reason: 'abi_mismatch', + detail: 'built for NODE_MODULE_VERSION 108, this Node accepts 115', + repairable: true, + host: { + platform: 'linux', + arch: 'x64', + libc: 'glibc', + glibcVersion: '2.31', + nodeAbi: '115', + nodeVersion: 'v20.11.0' + } +} + +// The relay dir is complete but node-pty will not load, which is exactly what the spawn-time cause +// describes. @parcel/watcher is healthy, so only node-pty is reset and rebuilt. +const NODE_PTY_BROKEN = 'ORCA-NATIVE-DEPS-MISSING:node-pty\nMISSING' + +function repairSucceedsResponses(): ExecResponse[] { + return [ + '__ORCA_REMOTE_PLATFORM__ Linux x86_64', + '/home/u', + NODE_PTY_BROKEN, // health probe before the lock + NODE_PTY_BROKEN, // re-probe under the repair lock + '', // SFTP-namespace install-owner marker + '', // reset node-pty + npm install + '', // chmod prebuilds + 'ORCA-NPTY-PROBE-OK\n', // node-pty loads again + '', // rm -f probe stderr + 'DEAD', + '', // publish the per-launch credential + 'READY' + ] +} + +function lockUnavailableResponses(): ExecResponse[] { + return [ + '__ORCA_REMOTE_PLATFORM__ Linux x86_64', + '/home/u', + NODE_PTY_BROKEN, // health probe before the lock + 'DEAD', + '', // publish the per-launch credential + 'READY' + ] +} + +describe('spawn-time node-pty repair through the locked deploy path', () => { + let warnSpy: ReturnType + const sftpCapture: SftpWriteCapture = { + paths: [], + contents: {}, + execCallCountAtWrite: {} + } + + beforeEach(() => { + vi.clearAllMocks() + vi.mocked(execCommand).mockReset().mockResolvedValue('') + sftpCapture.paths.length = 0 + for (const key of Object.keys(sftpCapture.contents)) { + delete sftpCapture.contents[key] + } + for (const key of Object.keys(sftpCapture.execCallCountAtWrite)) { + delete sftpCapture.execCallCountAtWrite[key] + } + vi.mocked(parseUnameToRelayPlatform).mockReturnValue('linux-x64') + vi.mocked(isRelayAlreadyInstalled).mockResolvedValue(true) + vi.mocked(tryAcquireRelayRepairLock).mockResolvedValue('acquired') + forgetRelayNodePtyRepairs(TARGET) + warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {}) + }) + + afterEach(() => { + warnSpy.mockRestore() + forgetRelayNodePtyRepairs(TARGET) + }) + + function feed(responses: ExecResponse[]): void { + const mockExec = vi.mocked(execCommand) + for (const response of responses) { + if (typeof response === 'string') { + mockExec.mockResolvedValueOnce(response) + } else { + mockExec.mockRejectedValueOnce(new Error(response.reject)) + } + } + } + + function recover(conn: ReturnType, deploys: { count: number }) { + return recoverRelayNodePtyForSpawn({ + targetId: TARGET, + cause: ABI_MISMATCH, + hasLivePtys: () => false, + reconnect: async () => { + deploys.count += 1 + await deployAndLaunchRelay(conn) + }, + resolveProvider: () => ({ generation: deploys.count }) + }) + } + + function execCalls(): string[] { + return vi.mocked(execCommand).mock.calls.map(([, command]) => String(command)) + } + + it('rebuilds node-pty under the repair lock and returns the post-reconnect provider', async () => { + const conn = makeMockConnection(sftpCapture) + feed(repairSucceedsResponses()) + const deploys = { count: 0 } + + const result = await recover(conn, deploys) + + expect(result.outcome).toBe('repaired') + expect(result.provider).toEqual({ generation: 1 }) + expect(deploys.count).toBe(1) + expect(vi.mocked(tryAcquireRelayRepairLock)).toHaveBeenCalledTimes(1) + const install = execCalls().find((command) => command.includes('npm install')) ?? '' + expect(install).toContain('npm install') + // The reset is what makes an ABI-mismatched binding recompile instead of being reported up to date. + expect(install).toContain("rm -rf 'node_modules/node-pty'") + }) + + it('does not repair or reconnect a second time for the same cause on the same host', async () => { + const conn = makeMockConnection(sftpCapture) + feed(repairSucceedsResponses()) + const deploys = { count: 0 } + await recover(conn, deploys) + vi.mocked(execCommand).mockReset().mockResolvedValue('') + + const second = await recover(conn, deploys) + + expect(second.outcome).toBe('already-attempted') + expect(second.provider).toBeNull() + expect(deploys.count).toBe(1) + expect(execCalls()).toEqual([]) + expect(vi.mocked(tryAcquireRelayRepairLock)).toHaveBeenCalledTimes(1) + }) + + it.each(['busy', 'error'] as const)( + 'leaves the host untouched and degrades to the relay message when the repair lock is %s', + async (lockResult) => { + const conn = makeMockConnection(sftpCapture) + vi.mocked(tryAcquireRelayRepairLock).mockResolvedValue(lockResult) + feed(lockUnavailableResponses()) + const deploys = { count: 0 } + + const result = await recover(conn, deploys) + + // The reconnect happened; the rebuild did not, so the retried spawn hits the same relay + // rejection and the user reads today's message. Nothing wrote to node_modules unlocked. + expect(deploys.count).toBe(1) + expect(execCalls().some((command) => command.includes('npm install'))).toBe(false) + expect(execCalls().some((command) => command.includes('node_modules/node-pty'))).toBe(false) + expect(result.outcome).toBe('repaired') + const warnings = warnSpy.mock.calls.map((args) => String(args[0] ?? '')) + expect(warnings.some((line) => line.includes(`repair lock is ${lockResult}`))).toBe(true) + } + ) + + it('never reaches the deploy path for an unverifiable cause', async () => { + const conn = makeMockConnection(sftpCapture) + const deploys = { count: 0 } + + const result = await recoverRelayNodePtyForSpawn({ + targetId: TARGET, + cause: { ...ABI_MISMATCH, status: 'unverifiable' }, + hasLivePtys: () => false, + reconnect: async () => { + deploys.count += 1 + await deployAndLaunchRelay(conn) + }, + resolveProvider: () => ({ generation: deploys.count }) + }) + + expect(result.outcome).toBe('not-repairable') + expect(deploys.count).toBe(0) + expect(vi.mocked(tryAcquireRelayRepairLock)).not.toHaveBeenCalled() + expect(execCalls()).toEqual([]) + }) + + it('never reaches the deploy path for a toolchain_missing cause', async () => { + const conn = makeMockConnection(sftpCapture) + const deploys = { count: 0 } + + const result = await recoverRelayNodePtyForSpawn({ + targetId: TARGET, + cause: { ...ABI_MISMATCH, reason: 'toolchain_missing', repairable: false }, + hasLivePtys: () => false, + reconnect: async () => { + deploys.count += 1 + await deployAndLaunchRelay(conn) + }, + resolveProvider: () => ({ generation: deploys.count }) + }) + + expect(result.outcome).toBe('not-repairable') + expect(deploys.count).toBe(0) + expect(execCalls()).toEqual([]) + }) +}) diff --git a/src/main/ssh/ssh-relay-session.ts b/src/main/ssh/ssh-relay-session.ts index ed5ae3f9cfd..68254fbb108 100644 --- a/src/main/ssh/ssh-relay-session.ts +++ b/src/main/ssh/ssh-relay-session.ts @@ -7,6 +7,8 @@ import { deployAndLaunchRelay } from './ssh-relay-deploy' import { execCommand } from './ssh-relay-deploy-helpers' import { isRelayVersionMismatchError } from './ssh-relay-version-mismatch-error' import { isRelayEndpointHeldError } from './ssh-relay-endpoint-incumbent' +import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair' +import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause' import { replayPendingSshPtyKills } from './ssh-pending-pty-kill-replay' import { SshChannelMultiplexer } from './ssh-channel-multiplexer' import { SshPtyProvider } from '../providers/ssh-pty-provider' @@ -319,6 +321,8 @@ export class SshRelaySession { private _onReady: ((targetId: string) => void) | null = null private portScanner: PortScanner | null = null private currentConnection: SshConnection | null = null + // Why: a self-driven repair reconnect must not silently re-negotiate the target's grace window. + private lastGraceTimeSeconds: number | undefined = undefined private hostPlatform: RemoteHostPlatform | null = null private remoteCliBridgeEnv: RemoteCliBridgeEnv | null = null private aiVaultListMethodSupported: boolean | null = null @@ -517,6 +521,7 @@ export class SshRelaySession { this.aiVaultListMethodSupported = null this.aiVaultTitleMethodSupported = null this.currentConnection = conn + this.lastGraceTimeSeconds = graceTimeSeconds try { const { @@ -663,6 +668,7 @@ export class SshRelaySession { this.aiVaultListMethodSupported = null this.aiVaultTitleMethodSupported = null this.currentConnection = conn + this.lastGraceTimeSeconds = graceTimeSeconds // Why: stop scanning before teardownProviders so the poll timer can't fire against a disposed multiplexer. this.stopPortScanning() @@ -862,6 +868,9 @@ export class SshRelaySession { this.teardownProviders('shutdown') this.currentConnection = null this._state = 'disposed' + // Why here and not on reconnect: an explicit disconnect is user action, so the host earns a + // fresh node-pty repair attempt. A reconnect must not, or the repair becomes a loop. + forgetRelayNodePtyRepairs(this.targetId) const recoveryRemoval = forgetSshPtyConsumerRecovery( this.targetId, this.ptyConsumerClientInstanceId, @@ -995,6 +1004,44 @@ export class SshRelaySession { }) } + /** + * A spawn was refused because the relay cannot load node-pty. Reconnect once so the deploy + * path's `repairInstalledNativeDeps` rebuilds it under `tryAcquireRelayRepairLock`, then hand + * back the provider registered by that reconnect for a single retry. + * + * Nothing here mutates the remote directly — a lock-less rebuild could collide with a + * concurrent reconnect's repair, so the locked deploy path stays the only writer. If the lock + * is busy it launches degraded, the retry hits the same rejection, and the user sees the + * relay's message. The attempt is spent either way. + */ + private async recoverRemoteTerminalRuntime( + requestingProvider: SshPtyProvider, + cause: TerminalUnavailableCause + ): Promise { + const { provider } = await recoverRelayNodePtyForSpawn({ + targetId: this.targetId, + cause, + hasLivePtys: () => requestingProvider.hasLivePtys(), + reconnect: async () => { + const conn = this.currentConnection + if (!conn || this.isDisposed()) { + throw new Error('no_live_ssh_connection') + } + await this.reconnect(conn, this.lastGraceTimeSeconds) + }, + resolveProvider: () => { + if (this._state !== 'ready' || this.isDisposed()) { + return null + } + const current = getSshPtyProvider(this.targetId) as SshPtyProvider | undefined + // Why identity-checked: a reconnect that fell back to the same provider would retry + // against the same unrepaired relay. + return current && current !== requestingProvider ? current : null + } + }) + return provider + } + // Why: shared by establish() and reconnect() so both use the exact same registration sequence. private async registerProviders( mux: SshChannelMultiplexer, @@ -1034,6 +1081,10 @@ export class SshRelaySession { this.remoteCliBridgeEnv ?? undefined, providerGeneration ) + // Why optional-call: session tests register partial provider stubs, same as the pause adapter below. + ptyProvider.setTerminalUnavailableRecovery?.((cause) => + this.recoverRemoteTerminalRuntime(ptyProvider, cause) + ) const consumerOwnerState = this.activePtyConsumerOwner() if (consumerOwnerState) { ptyProvider.setPtyDeliveryPauseAdapter?.(({ id, providerGeneration: generation, paused }) => { diff --git a/src/shared/terminal-unavailable-cause.ts b/src/shared/terminal-unavailable-cause.ts index 2671941da30..bb2478e21b0 100644 --- a/src/shared/terminal-unavailable-cause.ts +++ b/src/shared/terminal-unavailable-cause.ts @@ -62,6 +62,20 @@ export function parseTerminalUnavailableCause(value: unknown): TerminalUnavailab return parsed.success ? parsed.data : null } +/** + * The cause carried by a rejected JSON-RPC call, or null when there is none to act on. + * + * Reads `data`, never `code`: the dispatcher coerces a non-numeric error code to -32000 on the + * way out, so the string code does not survive the wire. The strict schema is the whole gate — + * no other published `data` shape validates against it. + */ +export function terminalUnavailableCauseFromError(error: unknown): TerminalUnavailableCause | null { + if (typeof error !== 'object' || error === null || !('data' in error)) { + return null + } + return parseTerminalUnavailableCause((error as { data: unknown }).data) +} + /** * Whether the client may rewrite the host's `node_modules` on the strength of this cause. *