Merge remote-tracking branch 'origin/fix/node-pty-spawn-repair' into adhoc/ssh-sweep-combined

# Conflicts:
#	src/main/ssh/ssh-relay-session.ts
This commit is contained in:
Neil
2026-09-01 15:40:28 -07:00
8 changed files with 891 additions and 0 deletions
@@ -0,0 +1,155 @@
// The client half of #17830: a spawn refused for an unloadable node-pty must route into a repair
// instead of printing a paragraph. Covers the seam only — the ledger and the locked rebuild are
// tested in src/main/ssh/ssh-relay-node-pty-repair.test.ts and
// src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts.
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { SshPtyProvider } from './ssh-pty-provider'
import { createMockMux, type MockMultiplexer } from './ssh-pty-provider-mock-multiplexer'
import {
TERMINAL_UNAVAILABLE_RPC_ERROR_CODE,
type TerminalUnavailableCause
} from '../../shared/terminal-unavailable-cause'
const UNAVAILABLE_MESSAGE =
"Remote terminals are unavailable: this host's node-pty binary was built for Node ABI 108."
function cause(overrides: Partial<TerminalUnavailableCause> = {}): TerminalUnavailableCause {
return {
status: 'blocked',
reason: 'abi_mismatch',
detail: 'built for NODE_MODULE_VERSION 108, this Node accepts 115',
repairable: true,
host: {
platform: 'linux',
arch: 'x64',
libc: 'glibc',
glibcVersion: '2.31',
nodeAbi: '115',
nodeVersion: 'v20.11.0'
},
...overrides
}
}
/** Shaped exactly as the multiplexer rebuilds a JSON-RPC error response client-side. */
function relayRejection(data: unknown): Error {
const error = new Error(UNAVAILABLE_MESSAGE)
Object.defineProperty(error, 'code', { value: TERMINAL_UNAVAILABLE_RPC_ERROR_CODE })
Object.defineProperty(error, 'data', { value: data })
return error
}
function spawnCallCount(target: MockMultiplexer): number {
return target.request.mock.calls.filter((call) => call[0] === 'pty.spawn').length
}
function rejectSpawnOnce(mux: MockMultiplexer, error: Error): void {
mux.request.mockImplementation(async (method: string) => {
if (method === 'pty.spawn') {
throw error
}
return undefined
})
}
const SPAWN_OPTS = { cwd: '/repo', cols: 80, rows: 24 }
let mux: MockMultiplexer
let provider: SshPtyProvider
beforeEach(() => {
mux = createMockMux()
provider = new SshPtyProvider('conn-1', mux as never)
})
describe('terminal-unavailable spawn recovery', () => {
it('routes a repairable cause into recovery and retries once on the repaired provider', async () => {
rejectSpawnOnce(mux, relayRejection(cause()))
const repairedMux = createMockMux()
repairedMux.request.mockResolvedValue({ id: 'pty-1', incarnationId: 'incarnation-1' })
const repairedProvider = new SshPtyProvider('conn-1', repairedMux as never)
const recover = vi.fn(async () => repairedProvider)
provider.setTerminalUnavailableRecovery(recover)
const result = await provider.spawn(SPAWN_OPTS)
expect(result.id).toBe('ssh:conn-1@@pty-1')
expect(recover).toHaveBeenCalledTimes(1)
expect(recover).toHaveBeenCalledWith(
expect.objectContaining({ reason: 'abi_mismatch', status: 'blocked' })
)
// Exactly one retry, on the post-repair channel — never a second attempt on the broken one.
expect(spawnCallCount(mux)).toBe(1)
expect(spawnCallCount(repairedMux)).toBe(1)
})
it('surfaces the relay message when the retry still fails, and does not recurse into a second repair', async () => {
// The lock-busy shape: the reconnect happened, the rebuild did not, so the relay says the same thing.
rejectSpawnOnce(mux, relayRejection(cause()))
const degradedMux = createMockMux()
const degradedProvider = new SshPtyProvider('conn-1', degradedMux as never)
rejectSpawnOnce(degradedMux, relayRejection(cause()))
const degradedRecover = vi.fn(async () => degradedProvider)
degradedProvider.setTerminalUnavailableRecovery(degradedRecover)
const recover = vi.fn(async () => degradedProvider)
provider.setTerminalUnavailableRecovery(recover)
await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE)
expect(recover).toHaveBeenCalledTimes(1)
expect(degradedRecover).not.toHaveBeenCalled()
})
it('rethrows without recovery when the recovery declines', async () => {
rejectSpawnOnce(mux, relayRejection(cause()))
provider.setTerminalUnavailableRecovery(async () => null)
await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE)
})
it('never recovers from an unverifiable cause', async () => {
rejectSpawnOnce(mux, relayRejection(cause({ status: 'unverifiable', repairable: true })))
const recover = vi.fn(async () => provider)
provider.setTerminalUnavailableRecovery(recover)
await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE)
expect(recover).not.toHaveBeenCalled()
})
it('never recovers from a toolchain_missing cause', async () => {
rejectSpawnOnce(mux, relayRejection(cause({ reason: 'toolchain_missing', repairable: false })))
const recover = vi.fn(async () => provider)
provider.setTerminalUnavailableRecovery(recover)
await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE)
expect(recover).not.toHaveBeenCalled()
})
it('ignores a malformed cause rather than half-reading it', async () => {
rejectSpawnOnce(mux, relayRejection({ status: 'blocked', repairable: true }))
const recover = vi.fn(async () => provider)
provider.setTerminalUnavailableRecovery(recover)
await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE)
expect(recover).not.toHaveBeenCalled()
})
it('leaves an old relay that publishes no cause on today behaviour', async () => {
rejectSpawnOnce(mux, new Error(UNAVAILABLE_MESSAGE))
const recover = vi.fn(async () => provider)
provider.setTerminalUnavailableRecovery(recover)
await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow(UNAVAILABLE_MESSAGE)
expect(recover).not.toHaveBeenCalled()
})
it('does not swallow an ordinary spawn failure', async () => {
rejectSpawnOnce(mux, new Error('shell not found'))
const recover = vi.fn(async () => provider)
provider.setTerminalUnavailableRecovery(recover)
await expect(provider.spawn(SPAWN_OPTS)).rejects.toThrow('shell not found')
expect(recover).not.toHaveBeenCalled()
})
})
+19
View File
@@ -23,6 +23,7 @@ import { SshPtySpawnExitRaceTracker } from './ssh-pty-spawn-exit-race'
import { SshAgentSessionCapabilities } from './ssh-agent-session-capabilities'
import type { PtyProcessInspection } from './pty-process-inspection'
import { writeToSshPty, writeToSshPtyWithSettlement } from './ssh-pty-write'
import { spawnWithTerminalRuntimeRepair, type TerminalRepairHook } from './ssh-pty-spawn-repair'
// Why: sequential relay teardown calls share one absolute budget; convert to the mux-relative timeout only at dispatch.
function relayTimeoutOptions(deadlineMs: number | undefined): { timeoutMs: number } | undefined {
@@ -38,6 +39,7 @@ export class SshPtyProvider implements IPtyProvider {
private readonly agentSessionCapabilities: SshAgentSessionCapabilities
private spawnExitRaces = new SshPtySpawnExitRaceTracker()
private readonly outputState: SshPtyProviderOutputState
private recoverFromTerminalUnavailable: TerminalRepairHook<SshPtyProvider> | null = null
requestHostRpc: NonNullable<IPtyProvider['requestHostRpc']> = (method, params, options) =>
this.mux.request(method, params as Record<string, unknown>, options)
@@ -76,7 +78,24 @@ export class SshPtyProvider implements IPtyProvider {
private toAppPtyId = (id: string): string => toAppSshPtyId(this.connectionId, id)
/** Installed by SshRelaySession, which owns the connection, the repair lock and the reconnect. */
setTerminalUnavailableRecovery(recover: TerminalRepairHook<SshPtyProvider>): void {
this.recoverFromTerminalUnavailable = recover
}
hasLivePtys(): boolean {
return this.livePtyIds.size > 0
}
async spawn(opts: PtySpawnOptions): Promise<PtySpawnResult> {
return await spawnWithTerminalRuntimeRepair<SshPtyProvider, PtySpawnResult>({
attempt: () => this.spawnWithoutTerminalRuntimeRepair(opts),
recover: this.recoverFromTerminalUnavailable,
retry: (provider) => provider.spawnWithoutTerminalRuntimeRepair(opts)
})
}
private async spawnWithoutTerminalRuntimeRepair(opts: PtySpawnOptions): Promise<PtySpawnResult> {
if (opts.agentSessionEnsure && opts.sessionId) {
throw new Error('agent_session_claim_unavailable')
}
@@ -0,0 +1,45 @@
/**
* The client seam for #17830: a spawn the relay refused because it cannot load node-pty.
*
* Split out of ssh-pty-provider.ts so the provider keeps only the wiring. The repair itself is
* driven by SshRelaySession, which owns the connection, the repair lock and the reconnect.
*/
import {
mayRepairFromCause,
terminalUnavailableCauseFromError,
type TerminalUnavailableCause
} from '../../shared/terminal-unavailable-cause'
/** Resolves to the provider registered after a successful repair, or null to keep the rejection. */
export type TerminalRepairHook<TProvider> = (
cause: TerminalUnavailableCause
) => Promise<TProvider | null>
/**
* Run a spawn, and on a proved-repairable terminal-unavailable rejection repair the host and
* retry exactly once on the provider the repair produced.
*
* Re-issuing is safe because a validated cause is the host's own statement that admission was
* refused before any PTY existed, so there is nothing to duplicate. The retry deliberately goes
* through a caller-supplied thunk that does not re-enter this wrapper, so it cannot recurse.
* Gated on `mayRepairFromCause`, never on the peer's `repairable` flag alone.
*/
export async function spawnWithTerminalRuntimeRepair<TProvider, TResult>(args: {
attempt: () => Promise<TResult>
recover: TerminalRepairHook<TProvider> | null
retry: (provider: TProvider) => Promise<TResult>
}): Promise<TResult> {
try {
return await args.attempt()
} catch (error) {
const cause = terminalUnavailableCauseFromError(error)
if (!cause || !mayRepairFromCause(cause) || !args.recover) {
throw error
}
const repaired = await args.recover(cause)
if (!repaired) {
throw error
}
return await args.retry(repaired)
}
}
@@ -0,0 +1,204 @@
// Why: the once-only ledger is the whole safety story here — an unbounded rebuild loop against a
// remote is worse than the bug it chases, so every gate that stops one gets a test.
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import {
forgetRelayNodePtyRepairs,
recoverRelayNodePtyForSpawn,
relayNodePtyRepairAttempts
} from './ssh-relay-node-pty-repair'
import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause'
const HOST: TerminalUnavailableCause['host'] = {
platform: 'linux',
arch: 'x64',
libc: 'glibc',
glibcVersion: '2.31',
nodeAbi: '115',
nodeVersion: 'v20.11.0'
}
function cause(overrides: Partial<TerminalUnavailableCause> = {}): TerminalUnavailableCause {
return {
status: 'blocked',
reason: 'abi_mismatch',
detail: 'built for NODE_MODULE_VERSION 108, this Node accepts 115',
repairable: true,
host: HOST,
...overrides
}
}
describe('recoverRelayNodePtyForSpawn', () => {
const TARGET = 'host-a'
let warnSpy: ReturnType<typeof vi.spyOn>
beforeEach(() => {
forgetRelayNodePtyRepairs(TARGET)
forgetRelayNodePtyRepairs('host-b')
warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {})
})
afterEach(() => {
warnSpy.mockRestore()
forgetRelayNodePtyRepairs(TARGET)
forgetRelayNodePtyRepairs('host-b')
})
function harness(overrides: { hasLivePtys?: boolean; reconnect?: () => Promise<void> } = {}) {
const reconnect = vi.fn(overrides.reconnect ?? (async () => {}))
const repaired = { name: 'post-repair-provider' }
const resolveProvider = vi.fn(() => repaired)
return {
reconnect,
resolveProvider,
repaired,
run: (c: TerminalUnavailableCause | null) =>
recoverRelayNodePtyForSpawn({
targetId: TARGET,
cause: c,
hasLivePtys: () => overrides.hasLivePtys === true,
reconnect,
resolveProvider
})
}
}
it('repairs a repairable cause with exactly one reconnect and hands back the new provider', async () => {
const h = harness()
const result = await h.run(cause())
expect(result.outcome).toBe('repaired')
expect(result.provider).toBe(h.repaired)
expect(h.reconnect).toHaveBeenCalledTimes(1)
expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual(['abi_mismatch'])
})
it('does not repair a second time for the same cause on the same host', async () => {
const first = harness()
await first.run(cause())
const second = harness()
const result = await second.run(cause())
expect(result.outcome).toBe('already-attempted')
expect(result.provider).toBeNull()
expect(second.reconnect).not.toHaveBeenCalled()
})
it('spends the attempt even when the repair reconnect fails, so it cannot loop', async () => {
const failing = harness({
reconnect: async () => {
throw new Error('relay repair lock is wedged')
}
})
const first = await failing.run(cause())
expect(first.outcome).toBe('reconnect-failed')
expect(first.provider).toBeNull()
const second = harness()
const retry = await second.run(cause())
expect(retry.outcome).toBe('already-attempted')
expect(second.reconnect).not.toHaveBeenCalled()
})
it('keeps the ledger per host, so a second host still gets its one attempt', async () => {
const h = harness()
await h.run(cause())
const other = vi.fn(async () => {})
const result = await recoverRelayNodePtyForSpawn({
targetId: 'host-b',
cause: cause(),
hasLivePtys: () => false,
reconnect: other,
resolveProvider: () => ({})
})
expect(result.outcome).toBe('repaired')
expect(other).toHaveBeenCalledTimes(1)
})
it('keeps the ledger per reason, so a different proved fault still gets its one attempt', async () => {
const h = harness()
await h.run(cause())
const second = harness()
const result = await second.run(cause({ reason: 'arch_mismatch' }))
expect(result.outcome).toBe('repaired')
expect([...relayNodePtyRepairAttempts(TARGET)].sort()).toEqual([
'abi_mismatch',
'arch_mismatch'
])
})
it('never repairs an unverifiable cause, whatever the peer claims about repairability', async () => {
const h = harness()
// Why repairable:true here: #14830 is exactly a peer flag believed over the status.
const result = await h.run(cause({ status: 'unverifiable', repairable: true }))
expect(result.outcome).toBe('not-repairable')
expect(h.reconnect).not.toHaveBeenCalled()
expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual([])
})
it('never repairs a toolchain_missing cause — the rebuild needs the missing compiler', async () => {
const h = harness()
const result = await h.run(cause({ reason: 'toolchain_missing', repairable: false }))
expect(result.outcome).toBe('not-repairable')
expect(h.reconnect).not.toHaveBeenCalled()
expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual([])
})
it('never repairs when the relay published no cause at all', async () => {
const h = harness()
const result = await h.run(null)
expect(result.outcome).toBe('not-repairable')
expect(h.reconnect).not.toHaveBeenCalled()
})
it('does not rebuild under live PTYs, and does not spend the attempt doing so', async () => {
const live = harness({ hasLivePtys: true })
const blocked = await live.run(cause())
expect(blocked.outcome).toBe('ptys-live')
expect(live.reconnect).not.toHaveBeenCalled()
expect([...relayNodePtyRepairAttempts(TARGET)]).toEqual([])
const idle = harness()
expect((await idle.run(cause())).outcome).toBe('repaired')
})
it('withholds the retry when the reconnect produced no provider', async () => {
const reconnect = vi.fn(async () => {})
const result = await recoverRelayNodePtyForSpawn({
targetId: TARGET,
cause: cause(),
hasLivePtys: () => false,
reconnect,
resolveProvider: () => null
})
expect(result.outcome).toBe('no-provider')
expect(result.provider).toBeNull()
expect(reconnect).toHaveBeenCalledTimes(1)
})
it('gives a host a fresh attempt only after an explicit disconnect', async () => {
await harness().run(cause())
forgetRelayNodePtyRepairs(TARGET)
const afterDisconnect = harness()
expect((await afterDisconnect.run(cause())).outcome).toBe('repaired')
})
})
+115
View File
@@ -0,0 +1,115 @@
/**
* Turning a spawn-time "remote terminals are unavailable" into an actual repair.
*
* The fault is proved on the relay at spawn time; the only thing that can fix it —
* `repairInstalledNativeDeps` in ssh-relay-deploy.ts — runs on the client during deploy, under
* `tryAcquireRelayRepairLock`. So the recovery here is deliberately indirect: it does not touch
* the remote `node_modules` itself, it drives one relay reconnect and lets the locked deploy path
* do the rebuild. All `node_modules` mutation stays behind that lock.
*
* The invariant that matters more than the recovery: **at most one repair per host per reason.**
* A rebuild loop against a remote is worse than the bug it is chasing, so the attempt is recorded
* before the reconnect starts, not after it succeeds. A repair that ran and failed is still an
* attempt, and this host will render the relay's message from then on.
*/
import {
mayRepairFromCause,
type TerminalUnavailableCause
} from '../../shared/terminal-unavailable-cause'
/** targetId -> the cause reasons already spent on this host, for the life of the session. */
const attemptedRepairsByTarget = new Map<string, Set<string>>()
export type RelayNodePtyRepairOutcome =
/** The cause is not proved-and-rebuildable; render the relay's message. */
| 'not-repairable'
/** This host already spent its one attempt on this reason. */
| 'already-attempted'
/** The relay is still serving PTYs, so a rebuild under it is not accounted for. */
| 'ptys-live'
/** No reconnect was possible, or it failed; the attempt is still spent. */
| 'reconnect-failed'
/** Reconnected, but no provider came back to retry on. */
| 'no-provider'
| 'repaired'
/** Forget a host's spent attempts. Disconnect is user action, so it may earn a fresh one. */
export function forgetRelayNodePtyRepairs(targetId: string): void {
attemptedRepairsByTarget.delete(targetId)
}
/** Test/diagnostic view of the ledger. */
export function relayNodePtyRepairAttempts(targetId: string): ReadonlySet<string> {
return attemptedRepairsByTarget.get(targetId) ?? new Set<string>()
}
/** False when this host has already spent its attempt on this reason. Marks on success. */
function claimRepairAttempt(targetId: string, reason: string): boolean {
const spent = attemptedRepairsByTarget.get(targetId)
if (spent) {
if (spent.has(reason)) {
return false
}
spent.add(reason)
return true
}
attemptedRepairsByTarget.set(targetId, new Set([reason]))
return true
}
export type RelayNodePtyRepairRequest<TProvider> = {
targetId: string
/** Already parsed and schema-validated; null when the relay published no cause. */
cause: TerminalUnavailableCause | null
/** Whether this client still holds live PTYs on the relay about to be rebuilt. */
hasLivePtys: () => boolean
/** One relay reconnect, which runs the locked `repairInstalledNativeDeps`. */
reconnect: () => Promise<void>
/** The provider registered after the reconnect — never the one that failed. */
resolveProvider: () => TProvider | null
}
/**
* Drive one repair for a failed spawn, returning the provider to retry on.
*
* Gates on `mayRepairFromCause`, not on the peer's `repairable` flag: an `unverifiable` cause
* proves nothing and must never trigger a rebuild (#14830, docs/reference/ssh-execution-boundary.md).
*/
export async function recoverRelayNodePtyForSpawn<TProvider>(
request: RelayNodePtyRepairRequest<TProvider>
): Promise<{ outcome: RelayNodePtyRepairOutcome; provider: TProvider | null }> {
const { targetId, cause } = request
if (!mayRepairFromCause(cause) || !cause) {
return { outcome: 'not-repairable', provider: null }
}
// Why check before claiming: a live PTY means the rebuild's blast radius is unaccounted for, and
// that is a reason to wait, not a spent attempt. Costs nothing remote, so it cannot loop.
if (request.hasLivePtys()) {
console.warn(
`[ssh-relay-repair] Not rebuilding node-pty on ${targetId} for ${cause.reason}: the relay is still serving PTYs`
)
return { outcome: 'ptys-live', provider: null }
}
if (!claimRepairAttempt(targetId, cause.reason)) {
console.warn(
`[ssh-relay-repair] node-pty repair for ${cause.reason} on ${targetId} already ran; not retrying`
)
return { outcome: 'already-attempted', provider: null }
}
console.warn(
`[ssh-relay-repair] Reconnecting ${targetId} once to rebuild node-pty (${cause.reason}): ${cause.detail}`
)
try {
await request.reconnect()
} catch (error) {
console.warn(
`[ssh-relay-repair] Repair reconnect for ${targetId} failed: ${
error instanceof Error ? error.message : String(error)
}`
)
return { outcome: 'reconnect-failed', provider: null }
}
const provider = request.resolveProvider()
return provider ? { outcome: 'repaired', provider } : { outcome: 'no-provider', provider: null }
}
@@ -0,0 +1,288 @@
// The other half of the #17830 recovery: proof that the reconnect a spawn-time cause triggers is
// the SAME locked deploy repair, not a second rebuild path. `tryAcquireRelayRepairLock` is the only
// thing standing between two clients and a concurrent `node_modules` rewrite, so a lock this path
// cannot take must degrade to the relay's message rather than proceed.
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type * as RelayInstallMarkerModule from './ssh-relay-install-marker'
vi.mock('electron', () => ({
app: { getAppPath: () => '/mock/app' }
}))
vi.mock('fs', () => ({
existsSync: vi.fn().mockReturnValue(true),
readFileSync: vi.fn().mockReturnValue('0.1.0+testhash')
}))
vi.mock('./relay-protocol', () => ({
RELAY_VERSION: '0.1.0',
RELAY_REMOTE_DIR: '.orca-remote',
parseUnameToRelayPlatform: vi.fn().mockReturnValue('linux-x64'),
RELAY_SENTINEL: 'ORCA-RELAY v0.1.0 READY\n',
RELAY_SENTINEL_TIMEOUT_MS: 10_000
}))
vi.mock('./ssh-relay-deploy-helpers', () => ({
uploadDirectory: vi.fn().mockResolvedValue(undefined),
waitForSentinel: vi.fn().mockResolvedValue({
write: vi.fn(),
onData: vi.fn(),
onClose: vi.fn()
}),
isUnconfirmedSshCommandTermination: () => false,
execCommand: vi.fn()
}))
vi.mock('./ssh-remote-node-resolution', () => ({
resolveRemoteNodePath: vi.fn().mockResolvedValue('/usr/bin/node')
}))
vi.mock('./ssh-relay-install-marker', async (importOriginal) => ({
...(await importOriginal<typeof RelayInstallMarkerModule>()),
createRelayInstallMarkerFileName: () => '.sftp-namespace-00000000000000000000000000000000'
}))
vi.mock('./ssh-relay-versioned-install', () => ({
readLocalFullVersion: vi.fn().mockReturnValue('0.1.0+testhash'),
computeRemoteRelayDir: (home: string, v: string) => `${home}/.orca-remote/relay-${v}`,
isRelayAlreadyInstalled: vi.fn().mockResolvedValue(true),
finalizeInstall: vi.fn().mockResolvedValue(undefined),
abandonInstall: vi.fn().mockResolvedValue(undefined),
gcOldRelayVersions: vi.fn().mockResolvedValue(undefined)
}))
vi.mock('./ssh-relay-install-lock', () => ({
acquireInstallLock: vi.fn().mockResolvedValue(undefined),
RELAY_INSTALL_LOCK_NAME: '.install-lock'
}))
vi.mock('./ssh-relay-repair-lock', () => ({
tryAcquireRelayRepairLock: vi.fn().mockResolvedValue('acquired')
}))
vi.mock('./ssh-relay-gc-claim', () => ({
releaseRelayGcClaimWithRetry: vi.fn().mockResolvedValue('released'),
tryAcquireRelayGcClaim: vi.fn().mockResolvedValue('launch-token'),
waitForRelayGcClaimRelease: vi.fn().mockResolvedValue(undefined)
}))
vi.mock('./ssh-connection-utils', () => ({
shellEscape: (s: string) => `'${s}'`
}))
import { deployAndLaunchRelay } from './ssh-relay-deploy'
import { execCommand } from './ssh-relay-deploy-helpers'
import { parseUnameToRelayPlatform } from './relay-protocol'
import { isRelayAlreadyInstalled } from './ssh-relay-versioned-install'
import { tryAcquireRelayRepairLock } from './ssh-relay-repair-lock'
import {
makeMockConnection,
type ExecResponse,
type SftpWriteCapture
} from './ssh-relay-native-deps-install-fixture'
import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair'
import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause'
const TARGET = 'repair-host'
const ABI_MISMATCH: TerminalUnavailableCause = {
status: 'blocked',
reason: 'abi_mismatch',
detail: 'built for NODE_MODULE_VERSION 108, this Node accepts 115',
repairable: true,
host: {
platform: 'linux',
arch: 'x64',
libc: 'glibc',
glibcVersion: '2.31',
nodeAbi: '115',
nodeVersion: 'v20.11.0'
}
}
// The relay dir is complete but node-pty will not load, which is exactly what the spawn-time cause
// describes. @parcel/watcher is healthy, so only node-pty is reset and rebuilt.
const NODE_PTY_BROKEN = 'ORCA-NATIVE-DEPS-MISSING:node-pty\nMISSING'
function repairSucceedsResponses(): ExecResponse[] {
return [
'__ORCA_REMOTE_PLATFORM__ Linux x86_64',
'/home/u',
NODE_PTY_BROKEN, // health probe before the lock
NODE_PTY_BROKEN, // re-probe under the repair lock
'', // SFTP-namespace install-owner marker
'', // reset node-pty + npm install
'', // chmod prebuilds
'ORCA-NPTY-PROBE-OK\n', // node-pty loads again
'', // rm -f probe stderr
'DEAD',
'', // publish the per-launch credential
'READY'
]
}
function lockUnavailableResponses(): ExecResponse[] {
return [
'__ORCA_REMOTE_PLATFORM__ Linux x86_64',
'/home/u',
NODE_PTY_BROKEN, // health probe before the lock
'DEAD',
'', // publish the per-launch credential
'READY'
]
}
describe('spawn-time node-pty repair through the locked deploy path', () => {
let warnSpy: ReturnType<typeof vi.spyOn>
const sftpCapture: SftpWriteCapture = {
paths: [],
contents: {},
execCallCountAtWrite: {}
}
beforeEach(() => {
vi.clearAllMocks()
vi.mocked(execCommand).mockReset().mockResolvedValue('')
sftpCapture.paths.length = 0
for (const key of Object.keys(sftpCapture.contents)) {
delete sftpCapture.contents[key]
}
for (const key of Object.keys(sftpCapture.execCallCountAtWrite)) {
delete sftpCapture.execCallCountAtWrite[key]
}
vi.mocked(parseUnameToRelayPlatform).mockReturnValue('linux-x64')
vi.mocked(isRelayAlreadyInstalled).mockResolvedValue(true)
vi.mocked(tryAcquireRelayRepairLock).mockResolvedValue('acquired')
forgetRelayNodePtyRepairs(TARGET)
warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {})
})
afterEach(() => {
warnSpy.mockRestore()
forgetRelayNodePtyRepairs(TARGET)
})
function feed(responses: ExecResponse[]): void {
const mockExec = vi.mocked(execCommand)
for (const response of responses) {
if (typeof response === 'string') {
mockExec.mockResolvedValueOnce(response)
} else {
mockExec.mockRejectedValueOnce(new Error(response.reject))
}
}
}
function recover(conn: ReturnType<typeof makeMockConnection>, deploys: { count: number }) {
return recoverRelayNodePtyForSpawn({
targetId: TARGET,
cause: ABI_MISMATCH,
hasLivePtys: () => false,
reconnect: async () => {
deploys.count += 1
await deployAndLaunchRelay(conn)
},
resolveProvider: () => ({ generation: deploys.count })
})
}
function execCalls(): string[] {
return vi.mocked(execCommand).mock.calls.map(([, command]) => String(command))
}
it('rebuilds node-pty under the repair lock and returns the post-reconnect provider', async () => {
const conn = makeMockConnection(sftpCapture)
feed(repairSucceedsResponses())
const deploys = { count: 0 }
const result = await recover(conn, deploys)
expect(result.outcome).toBe('repaired')
expect(result.provider).toEqual({ generation: 1 })
expect(deploys.count).toBe(1)
expect(vi.mocked(tryAcquireRelayRepairLock)).toHaveBeenCalledTimes(1)
const install = execCalls().find((command) => command.includes('npm install')) ?? ''
expect(install).toContain('npm install')
// The reset is what makes an ABI-mismatched binding recompile instead of being reported up to date.
expect(install).toContain("rm -rf 'node_modules/node-pty'")
})
it('does not repair or reconnect a second time for the same cause on the same host', async () => {
const conn = makeMockConnection(sftpCapture)
feed(repairSucceedsResponses())
const deploys = { count: 0 }
await recover(conn, deploys)
vi.mocked(execCommand).mockReset().mockResolvedValue('')
const second = await recover(conn, deploys)
expect(second.outcome).toBe('already-attempted')
expect(second.provider).toBeNull()
expect(deploys.count).toBe(1)
expect(execCalls()).toEqual([])
expect(vi.mocked(tryAcquireRelayRepairLock)).toHaveBeenCalledTimes(1)
})
it.each(['busy', 'error'] as const)(
'leaves the host untouched and degrades to the relay message when the repair lock is %s',
async (lockResult) => {
const conn = makeMockConnection(sftpCapture)
vi.mocked(tryAcquireRelayRepairLock).mockResolvedValue(lockResult)
feed(lockUnavailableResponses())
const deploys = { count: 0 }
const result = await recover(conn, deploys)
// The reconnect happened; the rebuild did not, so the retried spawn hits the same relay
// rejection and the user reads today's message. Nothing wrote to node_modules unlocked.
expect(deploys.count).toBe(1)
expect(execCalls().some((command) => command.includes('npm install'))).toBe(false)
expect(execCalls().some((command) => command.includes('node_modules/node-pty'))).toBe(false)
expect(result.outcome).toBe('repaired')
const warnings = warnSpy.mock.calls.map((args) => String(args[0] ?? ''))
expect(warnings.some((line) => line.includes(`repair lock is ${lockResult}`))).toBe(true)
}
)
it('never reaches the deploy path for an unverifiable cause', async () => {
const conn = makeMockConnection(sftpCapture)
const deploys = { count: 0 }
const result = await recoverRelayNodePtyForSpawn({
targetId: TARGET,
cause: { ...ABI_MISMATCH, status: 'unverifiable' },
hasLivePtys: () => false,
reconnect: async () => {
deploys.count += 1
await deployAndLaunchRelay(conn)
},
resolveProvider: () => ({ generation: deploys.count })
})
expect(result.outcome).toBe('not-repairable')
expect(deploys.count).toBe(0)
expect(vi.mocked(tryAcquireRelayRepairLock)).not.toHaveBeenCalled()
expect(execCalls()).toEqual([])
})
it('never reaches the deploy path for a toolchain_missing cause', async () => {
const conn = makeMockConnection(sftpCapture)
const deploys = { count: 0 }
const result = await recoverRelayNodePtyForSpawn({
targetId: TARGET,
cause: { ...ABI_MISMATCH, reason: 'toolchain_missing', repairable: false },
hasLivePtys: () => false,
reconnect: async () => {
deploys.count += 1
await deployAndLaunchRelay(conn)
},
resolveProvider: () => ({ generation: deploys.count })
})
expect(result.outcome).toBe('not-repairable')
expect(deploys.count).toBe(0)
expect(execCalls()).toEqual([])
})
})
+51
View File
@@ -7,6 +7,8 @@ import { deployAndLaunchRelay } from './ssh-relay-deploy'
import { execCommand } from './ssh-relay-deploy-helpers'
import { isRelayVersionMismatchError } from './ssh-relay-version-mismatch-error'
import { isRelayEndpointHeldError } from './ssh-relay-endpoint-incumbent'
import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair'
import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause'
import { replayPendingSshPtyKills } from './ssh-pending-pty-kill-replay'
import { SshChannelMultiplexer } from './ssh-channel-multiplexer'
import { SshPtyProvider } from '../providers/ssh-pty-provider'
@@ -319,6 +321,8 @@ export class SshRelaySession {
private _onReady: ((targetId: string) => void) | null = null
private portScanner: PortScanner | null = null
private currentConnection: SshConnection | null = null
// Why: a self-driven repair reconnect must not silently re-negotiate the target's grace window.
private lastGraceTimeSeconds: number | undefined = undefined
private hostPlatform: RemoteHostPlatform | null = null
private remoteCliBridgeEnv: RemoteCliBridgeEnv | null = null
private aiVaultListMethodSupported: boolean | null = null
@@ -517,6 +521,7 @@ export class SshRelaySession {
this.aiVaultListMethodSupported = null
this.aiVaultTitleMethodSupported = null
this.currentConnection = conn
this.lastGraceTimeSeconds = graceTimeSeconds
try {
const {
@@ -663,6 +668,7 @@ export class SshRelaySession {
this.aiVaultListMethodSupported = null
this.aiVaultTitleMethodSupported = null
this.currentConnection = conn
this.lastGraceTimeSeconds = graceTimeSeconds
// Why: stop scanning before teardownProviders so the poll timer can't fire against a disposed multiplexer.
this.stopPortScanning()
@@ -862,6 +868,9 @@ export class SshRelaySession {
this.teardownProviders('shutdown')
this.currentConnection = null
this._state = 'disposed'
// Why here and not on reconnect: an explicit disconnect is user action, so the host earns a
// fresh node-pty repair attempt. A reconnect must not, or the repair becomes a loop.
forgetRelayNodePtyRepairs(this.targetId)
const recoveryRemoval = forgetSshPtyConsumerRecovery(
this.targetId,
this.ptyConsumerClientInstanceId,
@@ -995,6 +1004,44 @@ export class SshRelaySession {
})
}
/**
* A spawn was refused because the relay cannot load node-pty. Reconnect once so the deploy
* path's `repairInstalledNativeDeps` rebuilds it under `tryAcquireRelayRepairLock`, then hand
* back the provider registered by that reconnect for a single retry.
*
* Nothing here mutates the remote directly — a lock-less rebuild could collide with a
* concurrent reconnect's repair, so the locked deploy path stays the only writer. If the lock
* is busy it launches degraded, the retry hits the same rejection, and the user sees the
* relay's message. The attempt is spent either way.
*/
private async recoverRemoteTerminalRuntime(
requestingProvider: SshPtyProvider,
cause: TerminalUnavailableCause
): Promise<SshPtyProvider | null> {
const { provider } = await recoverRelayNodePtyForSpawn<SshPtyProvider>({
targetId: this.targetId,
cause,
hasLivePtys: () => requestingProvider.hasLivePtys(),
reconnect: async () => {
const conn = this.currentConnection
if (!conn || this.isDisposed()) {
throw new Error('no_live_ssh_connection')
}
await this.reconnect(conn, this.lastGraceTimeSeconds)
},
resolveProvider: () => {
if (this._state !== 'ready' || this.isDisposed()) {
return null
}
const current = getSshPtyProvider(this.targetId) as SshPtyProvider | undefined
// Why identity-checked: a reconnect that fell back to the same provider would retry
// against the same unrepaired relay.
return current && current !== requestingProvider ? current : null
}
})
return provider
}
// Why: shared by establish() and reconnect() so both use the exact same registration sequence.
private async registerProviders(
mux: SshChannelMultiplexer,
@@ -1034,6 +1081,10 @@ export class SshRelaySession {
this.remoteCliBridgeEnv ?? undefined,
providerGeneration
)
// Why optional-call: session tests register partial provider stubs, same as the pause adapter below.
ptyProvider.setTerminalUnavailableRecovery?.((cause) =>
this.recoverRemoteTerminalRuntime(ptyProvider, cause)
)
const consumerOwnerState = this.activePtyConsumerOwner()
if (consumerOwnerState) {
ptyProvider.setPtyDeliveryPauseAdapter?.(({ id, providerGeneration: generation, paused }) => {
+14
View File
@@ -62,6 +62,20 @@ export function parseTerminalUnavailableCause(value: unknown): TerminalUnavailab
return parsed.success ? parsed.data : null
}
/**
* The cause carried by a rejected JSON-RPC call, or null when there is none to act on.
*
* Reads `data`, never `code`: the dispatcher coerces a non-numeric error code to -32000 on the
* way out, so the string code does not survive the wire. The strict schema is the whole gate —
* no other published `data` shape validates against it.
*/
export function terminalUnavailableCauseFromError(error: unknown): TerminalUnavailableCause | null {
if (typeof error !== 'object' || error === null || !('data' in error)) {
return null
}
return parseTerminalUnavailableCause((error as { data: unknown }).data)
}
/**
* Whether the client may rewrite the host's `node_modules` on the strength of this cause.
*