From fb48a9771b4f9eb87c4a6bfe1aadaf021034257c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:12:21 -0700 Subject: [PATCH 01/77] fix(gh): reap the whole gh/glab process tree at the deadline on POSIX (#18258) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `gh` and `glab` on PATH are routinely shims — mise, asdf, volta, or a hand-written wrapper — so a timed-out invocation has a chain to stop, not one process. `execFileCapture`'s POSIX kill path signals only the direct child; the descendants are orphaned to init and keep running. #18234 is exactly that shape: `bash ~/.local/bin/gh` -> `mise x gh` -> `gh`, where the reporter found the tail reparented to `systemd --user` and still at 100% CPU nearly two hours later. The 15s deadline #18239 added bounds Orca's semaphore slot and its promise; it does not bound the CPU burn. Route both CLIs through `execFileCaptureToTermination`, the primitive git's barrier path already uses: POSIX children spawn `detached`, the deadline signals `-pgid` and escalates to SIGKILL, and the promise waits for verified termination. Windows behaviour is unchanged (`taskkill /t` either way). Switching primitives also swapped execFile's hard maxBuffer failure for `runProcess`'s silent clipping, which would have turned an oversized gh response into a shorter valid-looking one. `ProcessResult` now reports truncation and the capture rejects on it, restoring the old contract and closing the same latent gap on git's barrier path. --- .../git/command-runner/exec-file-capture.ts | 22 +- .../gh-exec-file-deadline.test.ts | 102 ++-- src/main/git/command-runner/gh-exec-file.ts | 30 +- src/main/git/command-runner/glab-exec-file.ts | 25 +- src/main/git/runner-command-exec.test.ts | 160 ++++-- .../git/runner-gh-rate-limit-breaker.test.ts | 26 +- src/main/git/runner-wsl-gh-fallback.test.ts | 461 ++++++------------ ...tlab-known-host-probe-wsl-fallback.test.ts | 33 +- .../__fixtures__/fake-spawned-child.ts | 75 +++ .../child-process/bounded-output-sink.ts | 7 +- src/shared/child-process/process-spec.ts | 2 + src/shared/child-process/run-process.test.ts | 21 + src/shared/child-process/run-process.ts | 12 +- 13 files changed, 508 insertions(+), 468 deletions(-) create mode 100644 src/shared/child-process/__fixtures__/fake-spawned-child.ts diff --git a/src/main/git/command-runner/exec-file-capture.ts b/src/main/git/command-runner/exec-file-capture.ts index b1b9c664176..e9ae815ff34 100644 --- a/src/main/git/command-runner/exec-file-capture.ts +++ b/src/main/git/command-runner/exec-file-capture.ts @@ -25,7 +25,11 @@ export async function execFileCaptureToTermination( options: ExecFileCaptureOptions, termination?: WslProcessGroupTermination ): Promise<{ stdout: string | Buffer; stderr: string | Buffer }> { - const result = await runProcess({ + // Why measured here: runProcess spawns inside its promise executor, which runs + // synchronously, so this brackets exactly the main-thread block execFileCapture + // reports for its own spawns. + const spawnStartedAt = performance.now() + const pending = runProcess({ program: command, args, cwd: typeof options.cwd === 'string' ? options.cwd : undefined, @@ -37,10 +41,17 @@ export async function execFileCaptureToTermination( onChildTerminated: options.onChildTerminated, ...(options.stdin === undefined ? {} : { input: options.stdin }) }) + recordSubprocessSpawn(command, args, performance.now() - spawnStartedAt) + const result = await pending const stdout = options.encoding === 'buffer' ? Buffer.from(result.stdout) : result.stdout const cleanStderr = termination?.stripControlOutput(result.stderr) ?? result.stderr const stderr = options.encoding === 'buffer' ? Buffer.from(cleanStderr) : cleanStderr - if (result.code === 0 && !result.timedOut && !options.signal?.aborted) { + if ( + result.code === 0 && + !result.timedOut && + !result.outputTruncated && + !options.signal?.aborted + ) { return { stdout, stderr } } const error = result.timedOut @@ -48,7 +59,12 @@ export async function execFileCaptureToTermination( : new Error( options.signal?.aborted ? 'The operation was aborted.' - : cleanStderr.trim() || `${command} exited with ${result.code}.` + : result.outputTruncated + ? // Why fail instead of returning the clipped text: callers parse this + // as JSON or JSONL, where a clipped answer reads as a shorter valid + // one. execFile's own maxBuffer overrun errored for the same reason. + `${command} produced more than ${options.maxBuffer ?? DEFAULT_GIT_MAX_BUFFER} bytes of output.` + : cleanStderr.trim() || `${command} exited with ${result.code}.` ) if (options.signal?.aborted) { error.name = 'AbortError' diff --git a/src/main/git/command-runner/gh-exec-file-deadline.test.ts b/src/main/git/command-runner/gh-exec-file-deadline.test.ts index 3775b67a7ed..fa07dec32be 100644 --- a/src/main/git/command-runner/gh-exec-file-deadline.test.ts +++ b/src/main/git/command-runner/gh-exec-file-deadline.test.ts @@ -2,20 +2,15 @@ import { EventEmitter } from 'node:events' import type { ChildProcess } from 'node:child_process' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -const { execFileMock, spawnMock, killSpawnedCommandTreeMock } = vi.hoisted(() => ({ - execFileMock: vi.fn(), +const { spawnMock, processKillMock } = vi.hoisted(() => ({ spawnMock: vi.fn(), - killSpawnedCommandTreeMock: vi.fn().mockResolvedValue(undefined) + processKillMock: vi.fn() })) vi.mock('node:child_process', async (importOriginal) => ({ ...(await importOriginal()), - execFile: execFileMock, spawn: spawnMock })) -vi.mock('./spawned-command-tree-kill', () => ({ - killSpawnedCommandTree: killSpawnedCommandTreeMock -})) import { ghExecFileAsync } from './gh-exec-file' @@ -29,66 +24,87 @@ function mockChild(pid = 4321): ChildProcess { return child as unknown as ChildProcess } +function settleChild(child: ChildProcess, stdout: string): void { + child.stdout?.emit('data', Buffer.from(stdout)) + child.emit('exit', 0, null) + child.emit('close', 0, null) +} + /** * The contract the star check depends on after #18234: a `gh` that never exits - * is killed at the deadline, tree and all, rather than running forever. + * is killed at the deadline, and the kill reaches the whole chain. On the + * reporter's box `gh` was a shell wrapper calling `mise x gh`, so signalling + * only the direct child left the rest of the chain running under init. */ describe('gh exec deadline', () => { beforeEach(() => { vi.useFakeTimers() - execFileMock.mockReset() spawnMock.mockReset() - killSpawnedCommandTreeMock.mockClear() + processKillMock.mockReset() + vi.spyOn(process, 'kill').mockImplementation(processKillMock as unknown as typeof process.kill) }) afterEach(() => { vi.useRealTimers() + vi.restoreAllMocks() }) - it('kills the process tree and rejects when gh never exits', async () => { - const child = mockChild() - // Why never invoking the callback: this is exactly the stuck child from - // #18234 — spawned, spinning, and never reporting an exit. - execFileMock.mockReturnValue(child) + it.runIf(process.platform !== 'win32')( + 'signals the whole process group, not just the child, when gh never exits', + async () => { + const child = mockChild() + // Why never emitting exit: this is exactly the stuck child from #18234 — + // spawned, spinning, and never reporting an exit. + spawnMock.mockReturnValue(child) - const pending = ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { - timeout: 15_000 - }) - const rejection = expect(pending).rejects.toThrow('timed out') - await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledOnce()) + const pending = ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { + timeout: 15_000 + }) + const rejection = expect(pending).rejects.toThrow('timed out') + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledOnce()) - // Not yet: the deadline has not elapsed. - expect(killSpawnedCommandTreeMock).not.toHaveBeenCalled() + // The child must be its own group leader, or the signal below would go to + // whatever group it inherited — Orca's own. + expect(spawnMock.mock.calls[0][2].detached).toBe(true) + expect(processKillMock).not.toHaveBeenCalled() - await vi.advanceTimersByTimeAsync(15_000) - await rejection + await vi.advanceTimersByTimeAsync(15_000) + await vi.advanceTimersByTimeAsync(15_000) + await rejection - expect(killSpawnedCommandTreeMock).toHaveBeenCalledWith(child) - }) + expect(processKillMock).toHaveBeenCalledWith(-4321, undefined) + } + ) it('spawns with hidden console and captured stdio, never an inherited or shell stdio', async () => { const child = mockChild() - execFileMock.mockImplementation( - ( - _command: string, - _args: string[], - _options: unknown, - callback: (error: Error | null, stdout: string, stderr: string) => void - ) => { - callback(null, 'HTTP/2.0 204 No Content\r\n', '') - return child - } - ) + spawnMock.mockImplementation(() => { + queueMicrotask(() => settleChild(child, 'HTTP/2.0 204 No Content\r\n')) + return child + }) - await ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { timeout: 15_000 }) + const result = await ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { + timeout: 15_000 + }) - const [command, args, options] = execFileMock.mock.calls[0] + expect(result.stdout).toContain('204 No Content') + const [command, args, options] = spawnMock.mock.calls[0] expect(command).toBe('gh') expect(args).toEqual(['api', '--include', 'user/starred/stablyai/orca']) - // `execFile` captures stdout/stderr over pipes and never inherits Orca's; - // `shell` is never set, and the console stays hidden on Windows. expect(options.windowsHide).toBe(true) - expect(options.stdio).toBeUndefined() - expect(options.shell).toBeUndefined() + expect(options.stdio).toEqual(['pipe', 'pipe', 'pipe']) + expect(options.shell).toBe(false) + }) + + it('fails rather than returning a clipped answer when gh overruns maxBuffer', async () => { + const child = mockChild() + spawnMock.mockImplementation(() => { + queueMicrotask(() => settleChild(child, '['.padEnd(64, 'x'))) + return child + }) + + await expect( + ghExecFileAsync(['api', 'repos/stablyai/orca/issues'], { timeout: 15_000, maxBuffer: 8 }) + ).rejects.toThrow('more than 8 bytes') }) }) diff --git a/src/main/git/command-runner/gh-exec-file.ts b/src/main/git/command-runner/gh-exec-file.ts index b8f13be5d5e..e8308a8e4b4 100644 --- a/src/main/git/command-runner/gh-exec-file.ts +++ b/src/main/git/command-runner/gh-exec-file.ts @@ -19,7 +19,7 @@ import { isHostCommandMissing, resolveHostGitHubCli } from './github-cli-host-fallback' -import { execFileCapture } from './exec-file-capture' +import { execFileCaptureToTermination } from './exec-file-capture' import type { GitExecOptions } from './git-exec-options' import { argsLookIdempotent } from './gh-idempotency' import { applyGhHostToArgs, explicitGhHostname, explicitGhRepoHostname } from './gh-host-args' @@ -115,15 +115,25 @@ export async function ghExecFileAsync( let attemptedDefaultWslFallback = false for (let attempt = 0; attempt <= GH_RETRY_DELAYS_MS.length; attempt++) { try { - const { stdout, stderr } = await execFileCapture(resolved.binary, resolved.args, { - cwd: resolved.cwd, - encoding: (options.encoding ?? 'utf-8') as BufferEncoding, - maxBuffer: options.maxBuffer, - // Why: bound gh so one stuck child fails visibly instead of wedging the IPC lane. - timeout: options.timeout ?? defaultGhExecTimeoutMs(options.env), - env: nonInteractiveGhEnv(options.env), - signal: options.signal - }) + // Why to-termination and not execFileCapture: `gh` on PATH is routinely a + // shim (mise, asdf, volta, a hand-written wrapper), so the deadline below + // has a chain to reap, not one process. execFileCapture's POSIX kill only + // signals the direct child, which orphans the rest to init — a wedged + // helper then outlives the timeout that was supposed to bound it (#18234). + const { stdout, stderr } = await execFileCaptureToTermination( + resolved.binary, + resolved.args, + { + cwd: resolved.cwd, + encoding: (options.encoding ?? 'utf-8') as BufferEncoding, + maxBuffer: options.maxBuffer, + // Why: bound gh so one stuck child fails visibly instead of wedging the IPC lane. + timeout: options.timeout ?? defaultGhExecTimeoutMs(options.env), + env: nonInteractiveGhEnv(options.env), + signal: options.signal + }, + resolved.termination + ) return { stdout: stdout as string, stderr: stderr as string } } catch (err) { lastError = err diff --git a/src/main/git/command-runner/glab-exec-file.ts b/src/main/git/command-runner/glab-exec-file.ts index 3a4b4467ba9..3257dd9e818 100644 --- a/src/main/git/command-runner/glab-exec-file.ts +++ b/src/main/git/command-runner/glab-exec-file.ts @@ -2,7 +2,7 @@ import { addWslEnvKeys } from '../../wsl-env' import { extractExecError, parseRetryAfterMs } from '../exec-error' import { resolveCommand, resolveDefaultWslCli } from './wsl-command-resolution' import { isHostCommandMissing } from './github-cli-host-fallback' -import { execFileCapture } from './exec-file-capture' +import { execFileCaptureToTermination } from './exec-file-capture' import type { GitExecOptions } from './git-exec-options' import { argsLookIdempotent } from './gh-idempotency' import { @@ -63,14 +63,21 @@ export async function glabExecFileAsync( let attemptedDefaultWslFallback = false for (let attempt = 0; attempt <= GH_RETRY_DELAYS_MS.length; attempt++) { try { - const { stdout, stderr } = await execFileCapture(resolved.binary, resolved.args, { - cwd: resolved.cwd, - encoding: (options.encoding ?? 'utf-8') as BufferEncoding, - maxBuffer: options.maxBuffer, - timeout: options.timeout ?? DEFAULT_GLAB_EXEC_TIMEOUT_MS, - env: options.env, - signal: options.signal - }) + // Why to-termination: same shim chain as gh — the deadline has to reap the + // whole tree, not just the wrapper that spawned it (#18234). + const { stdout, stderr } = await execFileCaptureToTermination( + resolved.binary, + resolved.args, + { + cwd: resolved.cwd, + encoding: (options.encoding ?? 'utf-8') as BufferEncoding, + maxBuffer: options.maxBuffer, + timeout: options.timeout ?? DEFAULT_GLAB_EXEC_TIMEOUT_MS, + env: options.env, + signal: options.signal + }, + resolved.termination + ) return { stdout: stdout as string, stderr: stderr as string } } catch (err) { lastError = err diff --git a/src/main/git/runner-command-exec.test.ts b/src/main/git/runner-command-exec.test.ts index 27d7a72c747..89bcc4dca15 100644 --- a/src/main/git/runner-command-exec.test.ts +++ b/src/main/git/runner-command-exec.test.ts @@ -46,6 +46,31 @@ function createMockChildProcess(pid: number): MockChildProcess { return child } +/** + * Spawn stand-in for the gh/glab deadline tests: the CLI hangs, while the `ps` + * quiescence probe the tree termination runs answers immediately. + */ +function mockWedgedCliSpawn(child: MockChildProcess): void { + spawnMock.mockImplementation((program: string) => { + if (program !== 'ps') { + return child + } + const probe = createMockChildProcess(9100) + queueMicrotask(() => probe.emit('close', 0, null)) + return probe + }) +} + +/** Signals succeed; the existence probe reports the group already gone. */ +function mockProcessGroupSignals(): ReturnType { + return vi.spyOn(process, 'kill').mockImplementation(((_pid: number, signal?: unknown) => { + if (signal === 0) { + throw Object.assign(new Error('ESRCH'), { code: 'ESRCH' }) + } + return true + }) as typeof process.kill) +} + function createMockTaskkillProcess(): MockChildProcess { const child = createMockChildProcess(9000) child.unref = vi.fn() @@ -271,32 +296,46 @@ describe('runner execFile timeout handling', () => { } ) - it('rejects gh executions that never call back using the default timeout', async () => { + // Why the group and not the child (#18234): `gh` and `glab` on PATH are often + // shims, so the deadline has a chain to reap. Signalling only the direct child + // leaves the rest of it running under init long after the deadline passed. + it('signals the whole gh process group when gh never calls back', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { + cwd: '/repo' + }) + const rejection = expect(promise).rejects.toThrow('gh timed out.') + await vi.advanceTimersByTimeAsync(30_000) + expect(spawnMock.mock.calls[0][2].detached).toBe(true) + await vi.advanceTimersByTimeAsync(2_000) - const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { - cwd: '/repo' - }) - const rejection = expect(promise).rejects.toThrow('gh timed out.') - await vi.advanceTimersByTimeAsync(30_000) - - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) - it('rejects glab executions that never call back using the default timeout', async () => { + it('signals the whole glab process group when glab never calls back', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const promise = glabExecFileAsync(['api', 'projects/stablyai%2Forca/issues'], { + cwd: '/repo' + }) + const rejection = expect(promise).rejects.toThrow('glab timed out.') + await vi.advanceTimersByTimeAsync(30_000) + await vi.advanceTimersByTimeAsync(2_000) - const promise = glabExecFileAsync(['api', 'projects/stablyai%2Forca/issues'], { - cwd: '/repo' - }) - const rejection = expect(promise).rejects.toThrow('glab timed out.') - await vi.advanceTimersByTimeAsync(30_000) - - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) it('aborts glab retry backoff instead of starting another attempt', async () => { @@ -304,9 +343,14 @@ describe('runner execFile timeout handling', () => { const transient = Object.assign(new Error('glab failed'), { stderr: 'HTTP 503 Service Unavailable' }) - execFileMock.mockImplementationOnce((_command, _args, _options, callback) => { - callback(transient) - return createMockChildProcess(1234) + spawnMock.mockImplementationOnce(() => { + const child = createMockChildProcess(1234) + queueMicrotask(() => { + child.stderr.emit('data', Buffer.from(transient.stderr)) + child.emit('exit', 1, null) + child.emit('close', 1, null) + }) + return child }) const promise = glabExecFileAsync(['api', 'projects'], { @@ -314,52 +358,68 @@ describe('runner execFile timeout handling', () => { signal: controller.signal }) const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) - await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledTimes(1)) + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledTimes(1)) controller.abort() await rejection - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('kills an active gh execution when its caller aborts', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) - const controller = new AbortController() - const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { - cwd: '/repo', - signal: controller.signal - }) - const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const controller = new AbortController() + const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { + cwd: '/repo', + signal: controller.signal + }) + const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) - controller.abort() + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalled()) + controller.abort() + await vi.advanceTimersByTimeAsync(2_000) - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) it('honors explicit gh timeouts', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { + cwd: '/repo', + timeout: 1234 + }) + const rejection = expect(promise).rejects.toThrow('gh timed out.') + await vi.advanceTimersByTimeAsync(1233) + expect(processKill).not.toHaveBeenCalled() + await vi.advanceTimersByTimeAsync(1) + await vi.advanceTimersByTimeAsync(2_000) - const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { - cwd: '/repo', - timeout: 1234 - }) - const rejection = expect(promise).rejects.toThrow('gh timed out.') - await vi.advanceTimersByTimeAsync(1233) - expect(child.kill).not.toHaveBeenCalled() - await vi.advanceTimersByTimeAsync(1) - - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) it('runs gh non-interactively while preserving explicit env', async () => { - const child = createMockChildProcess(1234) let capturedEnv: NodeJS.ProcessEnv | undefined - execFileMock.mockImplementation((_cmd, _args, opts, cb) => { + spawnMock.mockImplementation((_cmd, _args, opts) => { capturedEnv = opts.env - cb(null, 'ok', '') + const child = createMockChildProcess(1234) + queueMicrotask(() => { + child.stdout.emit('data', Buffer.from('ok')) + child.emit('exit', 0, null) + child.emit('close', 0, null) + }) return child }) diff --git a/src/main/git/runner-gh-rate-limit-breaker.test.ts b/src/main/git/runner-gh-rate-limit-breaker.test.ts index 17e14b8e7d6..e450afd7939 100644 --- a/src/main/git/runner-gh-rate-limit-breaker.test.ts +++ b/src/main/git/runner-gh-rate-limit-breaker.test.ts @@ -12,6 +12,7 @@ vi.mock('child_process', () => ({ spawn: spawnMock })) +import { fakeSpawnReturning } from '../../shared/child-process/__fixtures__/fake-spawned-child' import { ghExecFileAsync } from './runner' import { _resetGhRateLimitBreaker, @@ -23,25 +24,16 @@ const PRIMARY_RATE_LIMIT_STDERR = 'gh: API rate limit exceeded for user ID 1775218. Please wait. (HTTP 403)' function mockGhFailure(stderr: string): void { - execFileMock.mockImplementation((_binary, _args, options, callback) => { - const done = typeof options === 'function' ? options : callback - queueMicrotask(() => - done(Object.assign(new Error(`Command failed: gh\n${stderr}`), { stderr }), '', stderr) - ) - return { once: vi.fn() } - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr, code: 1 })) } function mockGhSuccess(stdout: string): void { - execFileMock.mockImplementation((_binary, _args, options, callback) => { - const done = typeof options === 'function' ? options : callback - queueMicrotask(() => done(null, stdout, '')) - return { once: vi.fn() } - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stdout })) } beforeEach(() => { execFileMock.mockReset() + spawnMock.mockReset() }) afterEach(() => { @@ -54,14 +46,14 @@ describe('ghExecFileAsync rate-limit breaker', () => { await expect( ghExecFileAsync(['api', '--cache', '120s', 'search/issues?q=repo:a/b&per_page=1']) ).rejects.toThrow('rate limit') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) // The 90-repo storm case: every further search-bucket call must fail fast // without a subprocess. await expect( ghExecFileAsync(['api', '--cache', '120s', 'search/issues?q=repo:c/d&per_page=1']) ).rejects.toMatchObject({ ghRateLimitBlocked: true }) - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('keeps other buckets working while one bucket is blocked', async () => { @@ -72,7 +64,7 @@ describe('ghExecFileAsync rate-limit breaker', () => { stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('keeps other GitHub hosts and WSL runtimes working when github.com is blocked', async () => { @@ -105,7 +97,7 @@ describe('ghExecFileAsync rate-limit breaker', () => { value: originalPlatform }) } - expect(execFileMock).toHaveBeenCalledTimes(5) + expect(spawnMock).toHaveBeenCalledTimes(5) }) it.each([ @@ -195,7 +187,7 @@ describe('ghExecFileAsync rate-limit breaker', () => { ).resolves.toMatchObject({ stdout: '{"resources":{}}' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('does not trip the breaker on secondary rate limits', async () => { diff --git a/src/main/git/runner-wsl-gh-fallback.test.ts b/src/main/git/runner-wsl-gh-fallback.test.ts index 5f933c60620..9f4566f56ae 100644 --- a/src/main/git/runner-wsl-gh-fallback.test.ts +++ b/src/main/git/runner-wsl-gh-fallback.test.ts @@ -1,16 +1,19 @@ -import { EventEmitter } from 'node:events' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + createFakeSpawnedChild, + fakeSpawnDispatch, + fakeSpawnReturning +} from '../../shared/child-process/__fixtures__/fake-spawned-child' import type * as WslModule from '../wsl' -const { execFileMock, execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ - execFileMock: vi.fn(), +const { execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ execFileSyncMock: vi.fn(), spawnMock: vi.fn(), getDefaultWslDistroMock: vi.fn() })) vi.mock('child_process', () => ({ - execFile: execFileMock, + execFile: vi.fn(), execFileSync: execFileSyncMock, spawn: spawnMock })) @@ -26,25 +29,18 @@ import { _resetGhRateLimitBreaker } from './gh-rate-limit-breaker' const PRIMARY_RATE_LIMIT_STDERR = 'gh: API rate limit exceeded for user ID 1775218. Please wait. (HTTP 403)' -type MockChildProcess = EventEmitter & { - pid: number - kill: ReturnType - unref: ReturnType -} +// What the distro prints when the CLI is absent inside WSL but present on the host. +const WSL_GH_MISSING = 'bash: line 1: gh: command not found\n' +const TRANSIENT_502 = 'HTTP 502 Bad Gateway' -function createMockChildProcess(pid: number): MockChildProcess { - const child = new EventEmitter() as MockChildProcess - child.pid = pid - child.kill = vi.fn() - child.unref = vi.fn() - return child +function spawnEnoent(command: string): { spawnError: Error } { + return { spawnError: Object.assign(new Error(`spawn ${command} ENOENT`), { code: 'ENOENT' }) } } describe('ghExecFileAsync WSL fallback', () => { const originalPlatform = process.platform beforeEach(() => { - execFileMock.mockReset() spawnMock.mockReset() getDefaultWslDistroMock.mockReset() getDefaultWslDistroMock.mockReturnValue(null) @@ -66,21 +62,11 @@ describe('ghExecFileAsync WSL fallback', () => { }) it('falls back to host gh for explicit-repo WSL calls when gh is missing in the distro', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '[]' } + ) + ) await expect( ghExecFileAsync(['issue', 'list', '--repo', 'stablyhq/noqa', '--json', 'number,title'], { @@ -88,7 +74,7 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 1, 'wsl.exe', [ @@ -102,53 +88,34 @@ describe('ghExecFileAsync WSL fallback', () => { // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. The Linux directory still rides inside the command. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['issue', 'list', '--repo', 'stablyhq/noqa', '--json', 'number,title'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('does not fall back for repo-context gh calls without explicit repo context', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: WSL_GH_MISSING, code: 1 })) await expect( ghExecFileAsync(['issue', 'list'], { cwd: String.raw`\\wsl.localhost\Ubuntu\home\jinwoo\stably\noqa` }) - ).rejects.toThrow('Command failed: wsl.exe') + ).rejects.toThrow('gh: command not found') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('falls back for short-form explicit repo flags used by gh', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '[]' } + ) + ) await expect( ghExecFileAsync(['issue', 'list', '-R', 'stablyhq/noqa'], { @@ -156,31 +123,20 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['issue', 'list', '-R', 'stablyhq/noqa'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('falls back for compact short-form repo flags used by gh', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '[]' } + ) + ) await expect( ghExecFileAsync(['issue', 'list', '-Rstablyhq/noqa'], { @@ -188,31 +144,20 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['issue', 'list', '-Rstablyhq/noqa'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('falls back for repo view with an explicit positional repository', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '{"isFork":false}', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '{"isFork":false}' } + ) + ) await expect( ghExecFileAsync( @@ -224,68 +169,42 @@ describe('ghExecFileAsync WSL fallback', () => { ) ).resolves.toEqual({ stdout: '{"isFork":false}', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['repo', 'view', 'github.acme-corp.com/stablyhq/noqa', '--json', 'isFork,parent'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('does not fall back for gh api calls that depend on repo-context placeholders', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: WSL_GH_MISSING, code: 1 })) await expect( ghExecFileAsync(['api', 'repos/stablyhq/noqa/branches/{branch}'], { cwd: String.raw`\\wsl.localhost\Ubuntu\home\jinwoo\stably\noqa` }) - ).rejects.toThrow('Command failed: wsl.exe') + ).rejects.toThrow('gh: command not found') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('retries idempotent gh GraphQL query transient failures', async () => { - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '{"data":{}}', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '{"data":{}}' })) await expect( ghExecFileAsync(['api', 'graphql', '-f', 'query=query { viewer { login } }']) ).resolves.toEqual({ stdout: '{"data":{}}', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('retries a host-pinned idempotent gh GraphQL query after host injection', async () => { - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '{"data":{}}', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '{"data":{}}' })) await expect( ghExecFileAsync(['api', 'graphql', '-f', 'query=query { viewer { login } }'], { @@ -293,8 +212,8 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '{"data":{}}', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenNthCalledWith( 1, 'gh', [ @@ -305,37 +224,22 @@ describe('ghExecFileAsync WSL fallback', () => { '-f', 'query=query { viewer { login } }' ], - expect.any(Object), - expect.any(Function) + expect.any(Object) ) }) it('does not retry non-idempotent gh API transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( ghExecFileAsync(['api', '-X', 'POST', 'repos/stablyai/orca/issues']) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('does not retry gh GraphQL mutation transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( ghExecFileAsync([ @@ -346,42 +250,31 @@ describe('ghExecFileAsync WSL fallback', () => { ]) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('does not retry high-level gh edit transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( ghExecFileAsync(['issue', 'edit', '5', '--repo', 'stablyai/orca']) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('retries cwd-less gh calls through the default WSL distro when host gh is missing', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '{"resources":{}}', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('gh'))) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '{"resources":{}}' })) await expect(ghExecFileAsync(['api', 'rate_limit'])).resolves.toEqual({ stdout: '{"resources":{}}', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'gh' 'api' 'rate_limit'"], @@ -389,53 +282,35 @@ describe('ghExecFileAsync WSL fallback', () => { // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. This global call has no repo directory at all, so nothing about // where it runs changes. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) }) it('checks a blocked WSL scope before repeating a native-to-WSL fallback', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock.mockImplementation((binary, _args, _options, callback) => { - if (binary === 'gh') { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT', stderr: '' })) - return - } - callback( - Object.assign(new Error(PRIMARY_RATE_LIMIT_STDERR), { - stdout: '', - stderr: PRIMARY_RATE_LIMIT_STDERR - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'gh' ? spawnEnoent('gh') : { stderr: PRIMARY_RATE_LIMIT_STDERR, code: 1 } ) - }) + ) await expect(ghExecFileAsync(['api', 'repos/acme/widgets/pulls'])).rejects.toThrow('rate limit') await expect(ghExecFileAsync(['api', 'repos/acme/widgets/pulls'])).rejects.toMatchObject({ ghRateLimitBlocked: true }) - expect(execFileMock).toHaveBeenCalledTimes(3) - expect(execFileMock.mock.calls.map(([binary]) => binary)).toEqual(['gh', 'wsl.exe', 'gh']) + expect(spawnMock).toHaveBeenCalledTimes(3) + expect(spawnMock.mock.calls.map(([binary]) => binary)).toEqual(['gh', 'wsl.exe', 'gh']) }) it('checks a blocked native scope before repeating a WSL-to-native fallback', async () => { - execFileMock.mockImplementation((binary, _args, _options, callback) => { - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback( - Object.assign(new Error(PRIMARY_RATE_LIMIT_STDERR), { - stdout: '', - stderr: PRIMARY_RATE_LIMIT_STDERR - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' + ? { stderr: WSL_GH_MISSING, code: 1 } + : { stderr: PRIMARY_RATE_LIMIT_STDERR, code: 1 } ) - }) + ) const options = { cwd: String.raw`\\wsl.localhost\Ubuntu\home\jinwoo\stably\noqa` @@ -447,19 +322,12 @@ describe('ghExecFileAsync WSL fallback', () => { ghExecFileAsync(['api', 'repos/acme/widgets/pulls'], options) ).rejects.toMatchObject({ ghRateLimitBlocked: true }) - expect(execFileMock).toHaveBeenCalledTimes(3) - expect(execFileMock.mock.calls.map(([binary]) => binary)).toEqual(['wsl.exe', 'gh', 'wsl.exe']) + expect(spawnMock).toHaveBeenCalledTimes(3) + expect(spawnMock.mock.calls.map(([binary]) => binary)).toEqual(['wsl.exe', 'gh', 'wsl.exe']) }) it('does not retry non-idempotent glab transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( glabExecFileAsync(['api', '-X', 'POST', 'projects/stablyai%2Forca/issues/5/notes'], { @@ -467,18 +335,11 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('does not retry high-level glab update transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( glabExecFileAsync(['issue', 'update', '5', '-R', 'stablyai/orca'], { @@ -486,25 +347,21 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('retries cwd-less glab calls through the default WSL distro when host glab is missing', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '[]' })) await expect(glabExecFileAsync(['api', 'projects'])).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'api' 'projects'"], @@ -512,24 +369,21 @@ describe('ghExecFileAsync WSL fallback', () => { // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. This global call has no repo directory at all, so nothing about // where it runs changes. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) }) it('times out the default-WSL glab fallback and waits for full tree cleanup', async () => { vi.useFakeTimers() getDefaultWslDistroMock.mockReturnValue('Ubuntu') - const nativeChild = createMockChildProcess(1200) - const wslChild = createMockChildProcess(2400) - const taskkill = createMockChildProcess(3600) - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - return nativeChild - }) - .mockReturnValueOnce(wslChild) - spawnMock.mockReturnValue(taskkill) + const wslChild = createFakeSpawnedChild(2400) + const taskkill = createFakeSpawnedChild(3600) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + // Why a child that never exits: this is the wedged WSL helper the deadline + // has to reap, so nothing must settle the promise before taskkill reports. + .mockImplementationOnce(() => wslChild) + .mockImplementation(() => taskkill) const promise = glabExecFileAsync(['auth', 'status'], { timeout: 1000 }) const rejection = expect(promise).rejects.toThrow('wsl.exe timed out.') @@ -539,7 +393,7 @@ describe('ghExecFileAsync WSL fallback', () => { }) await vi.advanceTimersByTimeAsync(999) - expect(spawnMock).not.toHaveBeenCalled() + expect(spawnMock).toHaveBeenCalledTimes(2) await vi.advanceTimersByTimeAsync(1) expect(spawnMock).toHaveBeenCalledWith( 'taskkill', @@ -556,29 +410,24 @@ describe('ghExecFileAsync WSL fallback', () => { it('aborts the default-WSL glab fallback with full process-tree cleanup', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - const nativeChild = createMockChildProcess(1200) - const wslChild = createMockChildProcess(2400) - const taskkill = createMockChildProcess(3600) - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - return nativeChild - }) - .mockReturnValueOnce(wslChild) - spawnMock.mockReturnValue(taskkill) + const wslChild = createFakeSpawnedChild(2400) + const taskkill = createFakeSpawnedChild(3600) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + .mockImplementationOnce(() => wslChild) + .mockImplementation(() => taskkill) const controller = new AbortController() const promise = glabExecFileAsync(['auth', 'status'], { signal: controller.signal }) const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) - await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledTimes(2)) + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledTimes(2)) controller.abort() - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'auth' 'status'"], - expect.not.objectContaining({ signal: controller.signal }), - expect.any(Function) + expect.not.objectContaining({ signal: controller.signal }) ) expect(spawnMock).toHaveBeenCalledWith( 'taskkill', @@ -593,40 +442,26 @@ describe('ghExecFileAsync WSL fallback', () => { it('does not wake the default WSL distro for host-only GitLab diagnostics', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: 'Logged in to gitlab.com', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + .mockImplementationOnce(fakeSpawnReturning({ stdout: 'Logged in to gitlab.com' })) await expect( glabExecFileAsync(['auth', 'status'], { allowDefaultWslFallback: false }) ).rejects.toThrow('spawn glab ENOENT') - expect(execFileMock).toHaveBeenCalledTimes(1) - expect(execFileMock).toHaveBeenCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledWith( 'glab', ['auth', 'status'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('still retries idempotent glab transient failures', async () => { - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '[]' })) await expect( glabExecFileAsync(['api', 'projects/stablyai%2Forca/issues'], { @@ -634,7 +469,7 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('resolves fallback to the overridden distro if configured, and falls back to default WSL distro otherwise', async () => { @@ -642,60 +477,54 @@ describe('ghExecFileAsync WSL fallback', () => { setDefaultWslDistroOverride('Debian') getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((binary, args, _options, callback) => { - if (binary === 'wsl.exe' && args.includes('Debian')) { - callback(null, { stdout: 'Logged in to github.com as override', stderr: '' }) - return - } - callback(new Error('Wrong distro fallback')) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('gh'))) + .mockImplementationOnce( + fakeSpawnDispatch((program, args) => + program === 'wsl.exe' && args.includes('Debian') + ? { stdout: 'Logged in to github.com as override' } + : { stderr: 'Wrong distro fallback', code: 1 } + ) + ) await expect(ghExecFileAsync(['auth', 'status'])).resolves.toEqual({ stdout: 'Logged in to github.com as override', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Debian', '--exec', 'bash', '-c', "'gh' 'auth' 'status'"], - expect.any(Object), - expect.any(Function) + expect.any(Object) ) // 2) Test without override (should use default 'Ubuntu') - execFileMock.mockClear() + spawnMock.mockClear() setDefaultWslDistroOverride(null) - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((binary, args, _options, callback) => { - if (binary === 'wsl.exe' && args.includes('Ubuntu')) { - callback(null, { stdout: 'Logged in to github.com as default', stderr: '' }) - return - } - callback(new Error('Wrong distro fallback')) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('gh'))) + .mockImplementationOnce( + fakeSpawnDispatch((program, args) => + program === 'wsl.exe' && args.includes('Ubuntu') + ? { stdout: 'Logged in to github.com as default' } + : { stderr: 'Wrong distro fallback', code: 1 } + ) + ) await expect(ghExecFileAsync(['auth', 'status'])).resolves.toEqual({ stdout: 'Logged in to github.com as default', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'gh' 'auth' 'status'"], - expect.any(Object), - expect.any(Function) + expect.any(Object) ) }) }) diff --git a/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts b/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts index 40da88f0c91..f9f3c573629 100644 --- a/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts +++ b/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts @@ -1,15 +1,15 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { fakeSpawnDispatch } from '../../shared/child-process/__fixtures__/fake-spawned-child' import type * as WslModule from '../wsl' -const { execFileMock, execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ - execFileMock: vi.fn(), +const { execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ execFileSyncMock: vi.fn(), spawnMock: vi.fn(), getDefaultWslDistroMock: vi.fn() })) vi.mock('child_process', () => ({ - execFile: execFileMock, + execFile: vi.fn(), execFileSync: execFileSyncMock, spawn: spawnMock })) @@ -26,17 +26,16 @@ describe('glab known-hosts probe on Windows', () => { const originalPlatform = process.platform const hostGlabMissingWslLoggedIn = (): void => { - execFileMock.mockImplementation((binary, _args, _options, callback) => { - if (binary === 'wsl.exe') { - callback(null, { stdout: 'Logged in to gitlab.wsl.test as user', stderr: '' }) - return - } - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' + ? { stdout: 'Logged in to gitlab.wsl.test as user' } + : { spawnError: Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' }) } + ) + ) } beforeEach(() => { - execFileMock.mockReset() spawnMock.mockReset() getDefaultWslDistroMock.mockReset() getDefaultWslDistroMock.mockReturnValue('Ubuntu') @@ -56,12 +55,11 @@ describe('glab known-hosts probe on Windows', () => { await expect(getGlabKnownHosts()).resolves.toEqual(['gitlab.com']) - expect(execFileMock).toHaveBeenCalledTimes(1) - expect(execFileMock).toHaveBeenCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledWith( 'glab', ['auth', 'status'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) @@ -73,15 +71,14 @@ describe('glab known-hosts probe on Windows', () => { await expect(getGlabKnownHosts('conn-1')).resolves.toEqual(['gitlab.com', 'gitlab.wsl.test']) - expect(execFileMock).toHaveBeenCalledWith( + expect(spawnMock).toHaveBeenCalledWith( 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'auth' 'status'"], // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. This probe has no repo directory at all, so nothing about where // it runs changes. The native `glab` assertion above keeps `undefined`. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) }) }) diff --git a/src/shared/child-process/__fixtures__/fake-spawned-child.ts b/src/shared/child-process/__fixtures__/fake-spawned-child.ts new file mode 100644 index 00000000000..7183bc12e07 --- /dev/null +++ b/src/shared/child-process/__fixtures__/fake-spawned-child.ts @@ -0,0 +1,75 @@ +import { EventEmitter } from 'node:events' +import type { ChildProcess } from 'node:child_process' +import { vi } from 'vitest' + +/** + * A `child_process.spawn` stand-in for suites that drive gh/glab/git runners. + * + * Those runners capture output through `runProcess`, which reads the streams + * and waits for `close`, so a bare EventEmitter is not enough — a test child + * has to carry stdio and report an exit or the promise never settles. + */ +export function createFakeSpawnedChild(pid = 4321): ChildProcess { + const child = new EventEmitter() as EventEmitter & Record + child.pid = pid + child.kill = vi.fn(() => true) + child.stdin = Object.assign(new EventEmitter(), { end: vi.fn() }) + child.stdout = new EventEmitter() + child.stderr = new EventEmitter() + return child as unknown as ChildProcess +} + +/** Emit output and a clean exit, the way a CLI that answered would. */ +export function completeFakeSpawn( + child: ChildProcess, + result: { stdout?: string; stderr?: string; code?: number } = {} +): void { + if (result.stdout) { + child.stdout?.emit('data', Buffer.from(result.stdout)) + } + if (result.stderr) { + child.stderr?.emit('data', Buffer.from(result.stderr)) + } + const code = result.code ?? 0 + child.emit('exit', code, null) + child.emit('close', code, null) +} + +/** What a faked spawn should do: answer, or fail to start at all. */ +export type FakeSpawnOutcome = + | { stdout?: string; stderr?: string; code?: number } + | { spawnError: Error } + +function settleFakeSpawn(child: ChildProcess, outcome: FakeSpawnOutcome): void { + if ('spawnError' in outcome) { + // Why an event and not a throw: an unresolvable program fails asynchronously + // in libuv, which is what makes ENOENT reach callers as a rejection. + child.emit('error', outcome.spawnError) + return + } + completeFakeSpawn(child, outcome) +} + +/** + * Build a `spawn` implementation that answers every call the same way. + * + * Why a fresh child per call: the runners retry and fall back, and a shared + * emitter would replay the first call's exit into the second's listeners. + */ +export function fakeSpawnReturning( + outcome: FakeSpawnOutcome = {} +): (program: string, args: readonly string[]) => ChildProcess { + return fakeSpawnDispatch(() => outcome) +} + +/** Build a `spawn` implementation that answers per invoked program and argv. */ +export function fakeSpawnDispatch( + resolve: (program: string, args: readonly string[]) => FakeSpawnOutcome +): (program: string, args: readonly string[]) => ChildProcess { + return (program, args) => { + const child = createFakeSpawnedChild() + const outcome = resolve(program, args) + queueMicrotask(() => settleFakeSpawn(child, outcome)) + return child + } +} diff --git a/src/shared/child-process/bounded-output-sink.ts b/src/shared/child-process/bounded-output-sink.ts index c231234ef1a..195e466fdf6 100644 --- a/src/shared/child-process/bounded-output-sink.ts +++ b/src/shared/child-process/bounded-output-sink.ts @@ -10,6 +10,7 @@ import { Buffer } from 'node:buffer' export function createOutputSink(maxBytes: number): { write: (chunk: Buffer | string) => void text: () => string + truncated: () => boolean } { const chunks: Buffer[] = [] let bytes = 0 @@ -18,11 +19,15 @@ export function createOutputSink(maxBytes: number): { const chunk = Buffer.isBuffer(raw) ? raw : Buffer.from(raw) const remaining = maxBytes - bytes if (remaining <= 0) { + bytes += chunk.length return } chunks.push(chunk.length > remaining ? chunk.subarray(0, remaining) : chunk) bytes += chunk.length }, - text: () => Buffer.concat(chunks).toString('utf8') + text: () => Buffer.concat(chunks).toString('utf8'), + // Why: callers that parse the output need to tell a short answer from a + // clipped one -- truncated JSON or JSONL parses as a smaller valid result. + truncated: () => bytes > maxBytes } } diff --git a/src/shared/child-process/process-spec.ts b/src/shared/child-process/process-spec.ts index ac705974efe..2acfe82d61d 100644 --- a/src/shared/child-process/process-spec.ts +++ b/src/shared/child-process/process-spec.ts @@ -65,6 +65,8 @@ export type ProcessResult = { stderr: string /** True when the process was killed by `timeoutMs` rather than exiting. */ timedOut: boolean + /** True when stdout or stderr exceeded `maxOutputBytes` and was clipped. */ + outputTruncated?: boolean } export const DEFAULT_PROCESS_TIMEOUT_MS = 30_000 diff --git a/src/shared/child-process/run-process.test.ts b/src/shared/child-process/run-process.test.ts index 5fad9b5dc70..d36de3c688a 100644 --- a/src/shared/child-process/run-process.test.ts +++ b/src/shared/child-process/run-process.test.ts @@ -99,6 +99,27 @@ describe('runProcessSync', () => { }) }) +describe('bounded output', () => { + it('reports a clipped answer instead of passing it off as the whole one', async () => { + const result = await runProcess({ + program: process.execPath, + args: ['-e', 'process.stdout.write("x".repeat(64))'], + maxOutputBytes: 8 + }) + expect(result.stdout).toBe('xxxxxxxx') + expect(result.outputTruncated).toBe(true) + }) + + it('does not call output that exactly fills the cap truncated', async () => { + const result = await runProcess({ + program: process.execPath, + args: ['-e', 'process.stdout.write("x".repeat(8))'], + maxOutputBytes: 8 + }) + expect(result.outputTruncated).toBe(false) + }) +}) + describe('unkillable children', () => { it('settles after the grace period rather than outliving its own deadline', async () => { // `close` only fires once the child is gone, so a child that ignores the diff --git a/src/shared/child-process/run-process.ts b/src/shared/child-process/run-process.ts index ec83c5beed0..67bd48f5b81 100644 --- a/src/shared/child-process/run-process.ts +++ b/src/shared/child-process/run-process.ts @@ -186,7 +186,14 @@ export function runProcess(spec: ProcessSpec): Promise { const resolveFromClose = (code: number | null, signal: NodeJS.Signals | null): void => settle(() => - resolve({ code, signal, stdout: stdout.text(), stderr: stderr.text(), timedOut }) + resolve({ + code, + signal, + stdout: stdout.text(), + stderr: stderr.text(), + timedOut, + outputTruncated: stdout.truncated() || stderr.truncated() + }) ) const settleBarrierOutcome = (): void => { @@ -381,6 +388,9 @@ export function runProcessSync(spec: ProcessSpec): ProcessResult { signal: result.signal, stdout: result.stdout?.toString('utf8') ?? '', stderr: result.stderr?.toString('utf8') ?? '', + // Why always false: spawnSync reports an overrun as an ENOBUFS error, and + // the guard above rethrows it, so no truncated result reaches this point. + outputTruncated: false, // Why ETIMEDOUT and not the signal: a timeout kills with SIGTERM, but so // does anything else that terminates the child, and only a timeout also // sets this error. Reading the signal alone reports a deliberately From 31007c0d86cb4d5c2ab6f8621621bf8bf1800eb4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:14 -0700 Subject: [PATCH 02/77] fix(ssh): reclaim relay PTYs the client has provably lost, on host attestation only (#17831) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): reclaim relay PTYs the host attests this client orphaned (#9819) Orca could lose track of terminals running on an SSH relay until the 50-slot cap refused to open any more. This reclaims them, and the whole design is built around the fact that getting it wrong destroys a user's running process on their remote machine: the failure mode is leak, never kill. A stop requires all nine of: 1. the relay published an `ownerClientInstanceId` read from the live authenticated consumer grant of the connection that requested the spawn — never from a spawn parameter, since an echoed claim is no evidence; absent means skip 2. that id equals this client's persisted consumer identity 3. this connection holds the negotiated `session-owner` grant 4. `paneBound === true`, host-published 5. no `agentSessionOwners` — the host still advertises it as adoptable 6. `hostAgeMs >= 30s`, measured on the host's clock 7. this client has no route: not reattached, no lease outside terminated/expired, no pending kill, and no `expired` lease either — an expired lease is the record of a process deliberately left running, never a licence to kill it 8. every stop is fenced on the incarnation the same listing published, and on the owner identity, both re-checked by the host 9. a pass wanting to stop more than 8 refuses entirely Absence from a client-side set is `unverifiable` by construction (docs/reference/ssh-execution-boundary.md): a second machine attaches to the same relay and displaces the session owner, and its live agents are missing from this client's store for exactly the reason a genuine orphan is. So the host has to attest ownership, and the host has to attest that nothing is running. That second attestation is measured over the pane's whole tty, not its foreground process group. `tpgid == pgid` is foreground-only: on a real `bash -i` on a real pty, a shell holding `sleep 300 &` and a shell holding a Ctrl-Z'd job both read `pgid == tpgid`, `Ss+` — byte-identical to an idle prompt, with only the job's own row differing. A foreground-only gate therefore attests `pnpm build &` and a suspended editor as idle, and the stop that follows SIGKILLs every process group on the tty. `shellOwnsEveryTtyProcessGroup` is measured over that same set of groups, so the evidence and the kill describe the same thing. No new probe: `tpgid` already identifies the terminal, because a process group belongs to one session and a session to at most one controlling terminal. The freshness field is real rather than decorative. `capturedAgeMs` is stamped from when the capture was taken, deliberately as an upper bound since the process table is TTL-shared, and the sweep refuses an observation older than its own pass budget, counting its own elapsed time since the listing arrived. Stale evidence degrades to "do not sweep", never to "sweep". The display consumer of the same measurement keeps no age budget, as a stated decision: a stale pane title costs a redraw and self-corrects. `pty.shutdown` is authorized on the host that owns the process. `pty.spawn` and `pty.attach` both take a request context and check it; the one irreversible call took none, so the rule above lived entirely on the client that decided to make the call. It gains an optional `expectedOwnerClientInstanceId` and refuses unless the connection still authenticates as that identity AND this host recorded it at spawn. Finally, a reattach refusal now says whether it observed the process. Three refusals carry the same `SSH_SESSION_EXPIRED` text and only one is absence; `restoreRequired` means the PTY is live and only its source stream is not. Testing that text with `.includes()` expired the lease and deleted ownership for a running process, erasing this client's only record of it — and a PTY with no record is one the sweep may stop. Wire compatibility: four new optional fields and one new optional param on existing methods, no new method and no new stream opcode (Rule 1, and Rule 2 does not apply). Rule 1's caveat is discharged explicitly — no reader requires any of them, each absence is a named skip reason, and an ordinary pane teardown must omit the owner fence because a revived PTY carries no attested owner at all. New client plus old relay stops zero PTYs; old client plus new relay never reads the fields. Windows relay hosts publish no evidence and therefore never sweep. Verified by joining the real publisher to the real client reader over `ps` captured verbatim from a Linux container, and by driving a real group-for-group SIGKILL against a real pty: backgrounded and suspended jobs survive by pid, and an idle shell is still reclaimed, so the narrowed predicate is not a silent no-op. Squashed deliberately. The sweep is unsafe at every intermediate commit of its own history — before the foreground gate it reaps a hand-launched `claude`, and with a foreground-only gate it reaps a backgrounded build — so this ships as one commit with no bisectable state that kills live work. Refs #9819. Folds in #17939. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- .../pty-daemon-spawn-session-identity.test.ts | 67 +++- ...ty-runtime-ssh-binding-persistence.test.ts | 107 +++++- src/main/ipc/pty/ipc/spawn-execute.ts | 16 +- src/main/ipc/pty/runtime/spawn-execute.ts | 16 +- .../agent-foreground-process-batch.test.ts | 23 +- .../agent-foreground-process-batch.ts | 76 ++++- src/main/providers/pty-process-info.ts | 9 + src/main/providers/pty-provider-contract.ts | 6 + src/main/providers/ssh-pty-provider.ts | 8 +- ...ty-reattach-absence-discrimination.test.ts | 75 +++++ .../ssh/ssh-orphan-relay-pty-sweep.test.ts | 310 ++++++++++++++++++ src/main/ssh/ssh-orphan-relay-pty-sweep.ts | 164 +++++++++ ...h-orphan-sweep-pane-state-verdicts.test.ts | 238 ++++++++++++++ .../ssh/ssh-pty-consumer-recovery.test.ts | 15 + src/main/ssh/ssh-pty-consumer-recovery.ts | 9 + .../ssh-relay-session-orphan-sweep.test.ts | 293 +++++++++++++++++ src/main/ssh/ssh-relay-session.ts | 12 + ...handler-inventory-process-evidence.test.ts | 44 ++- .../pty-handler-ownership-attestation.test.ts | 283 ++++++++++++++++ src/relay/pty-handler.ts | 87 ++++- src/relay/relay-runtime-services.ts | 5 + src/relay/ssh-pty-consumer-session-adapter.ts | 6 + src/shared/foreground-process-evidence.ts | 33 +- src/shared/process-table-snapshot.ts | 5 + src/shared/pty-consumer-session.ts | 9 + .../ssh-relay-pty-ownership-proof.test.ts | 265 +++++++++++++++ src/shared/ssh-relay-pty-ownership-proof.ts | 226 +++++++++++++ 27 files changed, 2371 insertions(+), 36 deletions(-) create mode 100644 src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts create mode 100644 src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts create mode 100644 src/main/ssh/ssh-orphan-relay-pty-sweep.ts create mode 100644 src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts create mode 100644 src/main/ssh/ssh-relay-session-orphan-sweep.test.ts create mode 100644 src/relay/pty-handler-ownership-attestation.test.ts create mode 100644 src/shared/ssh-relay-pty-ownership-proof.test.ts create mode 100644 src/shared/ssh-relay-pty-ownership-proof.ts diff --git a/src/main/ipc/pty-daemon-spawn-session-identity.test.ts b/src/main/ipc/pty-daemon-spawn-session-identity.test.ts index b32ad6426f1..90ea56049fa 100644 --- a/src/main/ipc/pty-daemon-spawn-session-identity.test.ts +++ b/src/main/ipc/pty-daemon-spawn-session-identity.test.ts @@ -8,6 +8,7 @@ import { import { setupPtyIpcSuite } from './pty-ipc-test-harness' import { createDaemonActiveProviderFixtures } from './pty-ipc-daemon-provider-fixtures' import { makePaneKey } from '../../shared/stable-pane-id' +import { SshPtyAbsentFromRelayError } from '../providers/ssh-pty-errors' import { registerPtyHandlers, registerSshPtyProvider, @@ -459,7 +460,9 @@ describe('registerPtyHandlers', () => { }) it('marks a caller-supplied SSH session expired when remote reattach is gone', async () => { const sshSpawn = vi.fn(async () => { - throw new Error('SSH_SESSION_EXPIRED: remote-pty') + // The class, not the message: `SSH_SESSION_EXPIRED` is also what a live PTY whose + // source stream needs restoring refuses with, and only this one is host-reported absence. + throw new SshPtyAbsentFromRelayError('SSH_SESSION_EXPIRED: remote-pty') }) const store = { markSshRemotePtyLease: vi.fn(), @@ -509,10 +512,70 @@ describe('registerPtyHandlers', () => { expect(store.markSshRemotePtyLease).toHaveBeenCalledWith('ssh-1', 'remote-pty', 'expired') }) + it('leaves the lease alone when the refusal did not observe the process', async () => { + // A `restoreRequired` reattach is refused with the SAME `SSH_SESSION_EXPIRED` text, and it + // means the opposite: the PTY is live, only its source stream could not be resumed. The + // lease and the in-memory ownership are between them this client's only record that the + // remote process exists, and #9819's sweep reads a PTY it has no record of as one it may + // SIGKILL on the next connect. Erasing them here is how a live shell gets reaped. + const sshSpawn = vi.fn(async () => { + throw new Error('SSH_SESSION_EXPIRED: remote-pty') + }) + const store = { + markSshRemotePtyLease: vi.fn(), + clearSshRemotePtyKillIntent: vi.fn() + } + registerSshPtyProvider('ssh-1', { + spawn: sshSpawn, + write: vi.fn(), + resize: vi.fn(), + shutdown: vi.fn(), + sendSignal: vi.fn(), + getCwd: vi.fn(), + getInitialCwd: vi.fn(), + clearBuffer: vi.fn(), + acknowledgeDataEvent: vi.fn(), + hasChildProcesses: vi.fn(), + getForegroundProcess: vi.fn(), + serialize: vi.fn(), + revive: vi.fn(), + onData: vi.fn(() => () => {}), + onReplay: vi.fn(() => () => {}), + onExit: vi.fn(() => () => {}), + listProcesses: vi.fn(async () => []), + attach: vi.fn(), + getDefaultShell: vi.fn(), + getProfiles: vi.fn() + } as never) + handlers.clear() + registerPtyHandlers( + mainWindow as never, + undefined, + undefined, + undefined, + undefined, + store as never + ) + + await expect( + handlers.get('pty:spawn')!(null, { + cols: 80, + rows: 24, + env: {}, + connectionId: 'ssh-1', + sessionId: 'remote-pty' + }) + ).rejects.toThrow('SSH_SESSION_EXPIRED: remote-pty') + + // The spawn still fails; what must not happen is the destructive bookkeeping. + expect(store.markSshRemotePtyLease).not.toHaveBeenCalled() + }) it('marks a scoped SSH session expired using the raw relay lease id', async () => { const scopedPtyId = 'ssh:ssh-1@@remote-pty' const sshSpawn = vi.fn(async () => { - throw new Error('SSH_SESSION_EXPIRED: remote-pty') + // The class, not the message: `SSH_SESSION_EXPIRED` is also what a live PTY whose + // source stream needs restoring refuses with, and only this one is host-reported absence. + throw new SshPtyAbsentFromRelayError('SSH_SESSION_EXPIRED: remote-pty') }) const store = { markSshRemotePtyLease: vi.fn(), diff --git a/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts b/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts index 708d41c3e08..3fd1f89b11d 100644 --- a/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts +++ b/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts @@ -2,7 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { spawnMock, openCodeClearPtyMock, piClearPtyMock } from './pty-ipc-mock-registry' import { setupPtyIpcSuite } from './pty-ipc-test-harness' import { makePaneKey } from '../../shared/stable-pane-id' -import { SSH_SESSION_EXPIRED_ERROR } from '../providers/ssh-pty-errors' +import { SSH_SESSION_EXPIRED_ERROR, SshPtyAbsentFromRelayError } from '../providers/ssh-pty-errors' import { registerPtyHandlers, registerSshPtyProvider, @@ -446,7 +446,9 @@ describe('registerPtyHandlers', () => { const remoteWrite = vi.fn() registerSshPtyProvider('ssh-expired-runtime', { spawn: vi.fn(async () => { - throw new Error(`${SSH_SESSION_EXPIRED_ERROR}: relay-pty`) + // The class, not the message: `SSH_SESSION_EXPIRED` is also what a live PTY whose source + // stream needs restoring refuses with, and only this one is host-reported absence. + throw new SshPtyAbsentFromRelayError(`${SSH_SESSION_EXPIRED_ERROR}: relay-pty`) }), write: remoteWrite, resize: vi.fn(), @@ -531,4 +533,105 @@ describe('registerPtyHandlers', () => { unregisterSshPtyProvider('ssh-expired-runtime') } }) + it('leaves a runtime-owned lease alone when the refusal did not observe the process', async () => { + // The `restoreRequired` twin of the case above: same `SSH_SESSION_EXPIRED` text, opposite + // meaning — the PTY is live and only its source stream needs rebuilding. Expiring the lease + // and dropping ownership erases this client's only record of a running remote process, and + // #9819's sweep reads a PTY it has no record of as one it may SIGKILL on the next connect. + type RuntimeSpawnController = { + spawn(args: { + cols: number + rows: number + worktreeId?: string + connectionId?: string + tabId?: string + leafId?: string + sessionId?: string + persistHostSessionBinding?: boolean + }): Promise<{ id: string }> + } + const appPtyId = 'ssh:ssh-live-runtime@@relay-pty' + const remoteWrite = vi.fn() + registerSshPtyProvider('ssh-live-runtime', { + spawn: vi.fn(async () => { + throw new Error(`${SSH_SESSION_EXPIRED_ERROR}: relay-pty`) + }), + write: remoteWrite, + resize: vi.fn(), + shutdown: vi.fn(), + sendSignal: vi.fn(), + getCwd: vi.fn(), + getInitialCwd: vi.fn(), + clearBuffer: vi.fn(), + acknowledgeDataEvent: vi.fn(), + onData: vi.fn(() => () => {}), + onReplay: vi.fn(() => () => {}), + onExit: vi.fn(() => () => {}), + listProcesses: vi.fn(), + hasChildProcesses: vi.fn(), + getForegroundProcess: vi.fn(), + serialize: vi.fn(), + revive: vi.fn(), + getDefaultShell: vi.fn(), + getProfiles: vi.fn() + } as never) + const store = { + upsertSshRemotePtyLease: vi.fn(), + persistPtyBinding: vi.fn(), + removeSshRemotePtyLease: vi.fn(), + markSshRemotePtyLease: vi.fn(), + clearSshRemotePtyKillIntent: vi.fn() + } + let controller: RuntimeSpawnController | null = null + const runtime = { + setPtyController: vi.fn((value) => { + controller = value + }), + createPreAllocatedTerminalHandle: vi.fn(() => 'term_remote'), + registerPreAllocatedHandleForPty: vi.fn(), + registerPty: vi.fn(), + noteTerminalSpawnCommand: vi.fn(), + getDriver: vi.fn(() => ({ kind: 'host' })), + onPtySpawned: vi.fn(), + onPtyExit: vi.fn(), + onPtyData: vi.fn() + } + + try { + setPtyOwnership(appPtyId, 'ssh-live-runtime') + registerPtyHandlers( + mainWindow as never, + runtime as never, + undefined, + undefined, + undefined, + store as never + ) + const spawnController = controller as unknown as RuntimeSpawnController + const leafId = '11111111-1111-4111-8111-111111111111' + + await expect( + spawnController.spawn({ + cols: 80, + rows: 24, + connectionId: 'ssh-live-runtime', + worktreeId: 'wt-remote', + tabId: 'tab-remote', + leafId, + sessionId: appPtyId, + persistHostSessionBinding: true + }) + ).rejects.toThrow(SSH_SESSION_EXPIRED_ERROR) + + expect(store.markSshRemotePtyLease).not.toHaveBeenCalled() + expect(store.upsertSshRemotePtyLease).not.toHaveBeenCalled() + expect(store.persistPtyBinding).not.toHaveBeenCalled() + // Still routable: the client kept its handle on a process that is still running. + getPtyWriteListener()(mainWindowIpcEvent, { id: appPtyId, data: 'echo still-here' }) + expect(remoteWrite).toHaveBeenCalledWith(appPtyId, 'echo still-here') + } finally { + deletePtyOwnership(appPtyId) + unregisterSshPtyProvider('ssh-live-runtime') + } + }) }) diff --git a/src/main/ipc/pty/ipc/spawn-execute.ts b/src/main/ipc/pty/ipc/spawn-execute.ts index 58525e162aa..2d9de676c39 100644 --- a/src/main/ipc/pty/ipc/spawn-execute.ts +++ b/src/main/ipc/pty/ipc/spawn-execute.ts @@ -1,6 +1,7 @@ import { ensureWslHookRelayForReattach } from '../../../agent-hooks/wsl-hook-relay-reattach' import { SSH_SESSION_EXPIRED_ERROR, + isSshPtyAbsentFromRelayError, isSshPtyIdentityMismatchError } from '../../../providers/ssh-pty-errors' import { classifyError } from '../../../telemetry/classify-error' @@ -144,6 +145,15 @@ export async function executePtyIpcSpawn(ctx: PtyIpcSpawnState): Promise { Boolean(args.connectionId) && (spawnError.message.includes(SSH_SESSION_EXPIRED_ERROR) || rawMessage.includes(SSH_SESSION_EXPIRED_ERROR)) + // The message alone cannot carry this decision. All three reattach refusals are minted with the + // same `SSH_SESSION_EXPIRED` text, and only one of them observed the process: `restoreRequired` + // means the PTY is LIVE and only its source stream needs rebuilding, which + // `ssh-pty-errors.ts` states outright. Expiring its lease and deleting its ownership erases + // this client's last record of a running remote process, and #9819's sweep reads a PTY it has + // no record of as one it may SIGKILL on the next connect. Only positive host-reported absence + // may reach that bookkeeping; being too strict here merely leaves a dead lease for the next + // reattach to retire on real host evidence. + const relayReportedSessionAbsent = isExpiredSshSession && isSshPtyAbsentFromRelayError(err) const exitedBeforeSpawnReply = ctx.rejectedRegistrationCandidate?.exitedBeforeSpawnReply === true if (ctx.effectiveSessionAppId !== undefined) { @@ -157,7 +167,11 @@ export async function executePtyIpcSpawn(ctx: PtyIpcSpawnState): Promise { ptySizes.delete(ctx.effectiveSessionAppId) } } - if (args.connectionId && ctx.effectiveSessionRelayId !== undefined && isExpiredSshSession) { + if ( + args.connectionId && + ctx.effectiveSessionRelayId !== undefined && + relayReportedSessionAbsent + ) { // Why: expired remote reattach = relay already dropped the PTY; clear the lease so writes can't restore the stale binding. if (ctx.effectiveSessionAppId !== undefined && !isIdentityMismatch) { clearProviderPtyState(ctx.effectiveSessionAppId) diff --git a/src/main/ipc/pty/runtime/spawn-execute.ts b/src/main/ipc/pty/runtime/spawn-execute.ts index 99829f119bb..05a9ca82d15 100644 --- a/src/main/ipc/pty/runtime/spawn-execute.ts +++ b/src/main/ipc/pty/runtime/spawn-execute.ts @@ -13,6 +13,7 @@ import { clearProviderPtyState } from '../provider/state-cleanup' import { isProviderAgentSessionOwnerLive, normalizeNodePtySpawnError } from '../provider/liveness' import { SSH_SESSION_EXPIRED_ERROR, + isSshPtyAbsentFromRelayError, isSshPtyIdentityMismatchError } from '../../../providers/ssh-pty-errors' import type { RuntimePtySpawnState } from './spawn-state' @@ -224,6 +225,15 @@ export async function executeRuntimePtySpawn(ctx: RuntimePtySpawnState): Promise Boolean(args.connectionId) && (spawnError.message.includes(SSH_SESSION_EXPIRED_ERROR) || rawMessage.includes(SSH_SESSION_EXPIRED_ERROR)) + // The message alone cannot carry this decision. All three reattach refusals are minted with the + // same `SSH_SESSION_EXPIRED` text, and only one of them observed the process: `restoreRequired` + // means the PTY is LIVE and only its source stream needs rebuilding, which + // `ssh-pty-errors.ts` states outright. Expiring its lease and deleting its ownership erases + // this client's last record of a running remote process, and #9819's sweep reads a PTY it has + // no record of as one it may SIGKILL on the next connect. Only positive host-reported absence + // may reach that bookkeeping; being too strict here merely leaves a dead lease for the next + // reattach to retire on real host evidence. + const relayReportedSessionAbsent = isExpiredSshSession && isSshPtyAbsentFromRelayError(err) const exitedBeforeSpawnReply = ctx.rejectedRegistrationCandidate?.exitedBeforeSpawnReply === true if (ctx.effectiveSessionAppId !== undefined) { @@ -237,7 +247,11 @@ export async function executeRuntimePtySpawn(ctx: RuntimePtySpawnState): Promise ptySizes.delete(ctx.effectiveSessionAppId) } } - if (args.connectionId && ctx.effectiveSessionRelayId !== undefined && isExpiredSshSession) { + if ( + args.connectionId && + ctx.effectiveSessionRelayId !== undefined && + relayReportedSessionAbsent + ) { if (ctx.effectiveSessionAppId !== undefined && !isIdentityMismatch) { clearProviderPtyState(ctx.effectiveSessionAppId) deletePtyOwnership(ctx.effectiveSessionAppId) diff --git a/src/main/providers/agent-foreground-process-batch.test.ts b/src/main/providers/agent-foreground-process-batch.test.ts index 8baa37a7499..4107a96c2f1 100644 --- a/src/main/providers/agent-foreground-process-batch.test.ts +++ b/src/main/providers/agent-foreground-process-batch.test.ts @@ -22,7 +22,28 @@ describe('batched foreground process correlation', () => { resolveAgentForegroundProcessesFromIndex(buildProcessTableIndex(rows), [ { rootPid: 100, fallbackProcess: 'zsh' } ]) - ).toEqual([{ available: true, processName: 'codex' }]) + ).toEqual([{ available: true, processName: 'codex', shellOwnsEveryTtyProcessGroup: false }]) + }) + + it('reports whether the shell itself owns the terminal, named process or not', () => { + // The only host-observable "nothing is running here". pid 200's own pgid owns the terminal; + // pid 300 has an unrecognized command in the foreground, which nothing else here can see. + const rows = parseStrictProcessTableRows( + [ + '200 1 200 200 Ss /bin/zsh', + '300 1 300 301 Ss /bin/zsh', + '301 300 301 301 S+ vim notes.md' + ].join('\n') + ) + expect( + resolveAgentForegroundProcessesFromIndex(buildProcessTableIndex(rows), [ + { rootPid: 200, fallbackProcess: 'zsh' }, + { rootPid: 300, fallbackProcess: 'zsh' } + ]) + ).toEqual([ + { available: true, processName: null, shellOwnsEveryTtyProcessGroup: true }, + { available: true, processName: null, shellOwnsEveryTtyProcessGroup: false } + ]) }) it('returns unverifiable for a missing root or no controlling tty', () => { diff --git a/src/main/providers/agent-foreground-process-batch.ts b/src/main/providers/agent-foreground-process-batch.ts index 2e0036bb603..e2d9e5a1372 100644 --- a/src/main/providers/agent-foreground-process-batch.ts +++ b/src/main/providers/agent-foreground-process-batch.ts @@ -26,6 +26,9 @@ export type BatchedForegroundProcessResult = { available: boolean processName: string | null reason?: string + /** Set only when the table was readable: every process group attached to this PTY's terminal is + * the shell's own, and none of them is stopped. Left absent when we could not observe it. */ + shellOwnsEveryTtyProcessGroup?: boolean } export type BatchedForegroundProcessOptions = { @@ -34,6 +37,48 @@ export type BatchedForegroundProcessOptions = { stats?: ProcessTableIndexStats } +/** Which process groups occupy each controlling terminal, and which terminals hold a stopped + * process. */ +type TtyOccupancy = { + processGroupsByTty: ReadonlyMap> + stoppedTtys: ReadonlySet +} + +const ttyOccupancyByCapture = new WeakMap() + +/** Index the capture by controlling terminal. + * + * Keyed on `tpgid` because the snapshot carries no tty column and does not need one: a process + * group belongs to exactly one session, a session to at most one controlling terminal, so two + * rows reporting the same live `tpgid` are on the same tty. Memoized per capture, since the + * per-pane cadence poll and `pty.listProcesses` share one TTL-cached table. */ +function getTtyOccupancy(rows: readonly ProcessTableRow[]): TtyOccupancy { + const cached = ttyOccupancyByCapture.get(rows) + if (cached) { + return cached + } + const processGroupsByTty = new Map>() + const stoppedTtys = new Set() + for (const row of rows) { + if (row.pgid === undefined || row.tpgid === undefined || row.tpgid <= 0) { + continue + } + let groups = processGroupsByTty.get(row.tpgid) + if (!groups) { + groups = new Set() + processGroupsByTty.set(row.tpgid, groups) + } + groups.add(row.pgid) + // `T` is a job-control stop (Ctrl-Z), `t` a tracing stop. Both are work the pane still holds. + if (row.stat.startsWith('T') || row.stat.startsWith('t')) { + stoppedTtys.add(row.tpgid) + } + } + const occupancy: TtyOccupancy = { processGroupsByTty, stoppedTtys } + ttyOccupancyByCapture.set(rows, occupancy) + return occupancy +} + export async function resolveAgentForegroundProcessesBatch( requests: readonly BatchedForegroundProcessRequest[], options: BatchedForegroundProcessOptions = {} @@ -91,6 +136,7 @@ export function resolveAgentForegroundProcessesFromIndex( } } + const occupancy = getTtyOccupancy(index.rows) return requests.map((request) => { const root = lookupProcessTableIndex(index, (value) => value.byPid.get(request.rootPid)) if (!root) { @@ -114,6 +160,20 @@ export function resolveAgentForegroundProcessesFromIndex( reason: 'no_controlling_tty' } } + // The only host-observable "nothing is running here" signal, and it has to be read off the + // whole tty rather than off `tpgid === pgid`. A backgrounded `pnpm build &` and a Ctrl-Z'd + // editor both leave the shell owning the foreground group, byte-identical to an idle prompt; + // what separates them is a second process group attached to the pane's terminal. That is also + // exactly the blast radius of the stop this attests to — `forceKillPosixPtyProcessGroups` + // SIGKILLs every process group on the tty — so the evidence and the kill now measure the same + // thing. A reader may treat `false` as "busy" and must never treat absence as "idle". + const ttyProcessGroups = occupancy.processGroupsByTty.get(root.tpgid) + const shellOwnsEveryTtyProcessGroup = + root.tpgid === root.pgid && + ttyProcessGroups !== undefined && + ttyProcessGroups.size === 1 && + ttyProcessGroups.has(root.pgid) && + !occupancy.stoppedTtys.has(root.tpgid) const allCandidates = rowsByOwner.get(root.pid) ?? [] const foregroundCandidates = allCandidates.filter((row) => row.pgid === root.tpgid) const fallbackProcess = request.fallbackProcess @@ -125,7 +185,7 @@ export function resolveAgentForegroundProcessesFromIndex( ) : foregroundCandidates if (wrapperFallback && candidates.length !== 1) { - return { available: true, processName: null } + return { available: true, processName: null, shellOwnsEveryTtyProcessGroup } } const selected = selectForegroundProcessCandidate(candidates, allCandidates) if (selected) { @@ -135,10 +195,11 @@ export function resolveAgentForegroundProcessesFromIndex( selected.recognized, selected.candidate, allCandidates - ) + ), + shellOwnsEveryTtyProcessGroup } } - return { available: true, processName: null } + return { available: true, processName: null, shellOwnsEveryTtyProcessGroup } }) } @@ -147,7 +208,14 @@ export function toForegroundProcessEvidence( metadata: { authorityGeneration: string; observationEpoch: number; capturedAgeMs: number } ): ForegroundProcessEvidence { return result.available - ? { ...metadata, verdict: 'live', processName: result.processName } + ? { + ...metadata, + verdict: 'live', + processName: result.processName, + ...(result.shellOwnsEveryTtyProcessGroup !== undefined + ? { shellOwnsEveryTtyProcessGroup: result.shellOwnsEveryTtyProcessGroup } + : {}) + } : { ...metadata, verdict: 'unverifiable', diff --git a/src/main/providers/pty-process-info.ts b/src/main/providers/pty-process-info.ts index 848ff07c78a..942cab84f73 100644 --- a/src/main/providers/pty-process-info.ts +++ b/src/main/providers/pty-process-info.ts @@ -18,4 +18,13 @@ export type PtyProcessInfo = { /** Optional host-side process evidence attached to an inventory seed. */ foregroundProcessEvidence?: ForegroundProcessEvidence agentSessionOwners?: AgentSessionOwnerBinding[] + /** Age measured on the OWNING host's clock. Absent means the host did not measure it, which is + * not the same as "new" or "old" — a reader that needs an age must defer instead of assuming. */ + hostAgeMs?: number + /** True when the host spawned this PTY for an Orca pane, false for a bare host shell. Absent from + * a host that never published it; absence is neither value. */ + paneBound?: boolean + /** The client identity the OWNING host recorded as having asked it to create this PTY. Absent + * whenever the host could not attest one, and absence must never be read as "unowned". */ + ownerClientInstanceId?: string } diff --git a/src/main/providers/pty-provider-contract.ts b/src/main/providers/pty-provider-contract.ts index 35fca5b0b34..cec4dbcb30a 100644 --- a/src/main/providers/pty-provider-contract.ts +++ b/src/main/providers/pty-provider-contract.ts @@ -204,6 +204,12 @@ export type IPtyProvider = { keepHistory?: boolean deadlineMs?: number expectedIncarnationId?: PtyIncarnationId + /** Ask the execution host to refuse this stop unless it recorded this exact client identity + * as the PTY's creator AND this connection still authenticates as it. Optional because a + * host that predates it ignores the field, and because most stops are ordinary teardown of a + * pane whose owner the host may never have attested (a revived PTY carries none). Set it + * wherever the caller's authority to destroy comes from that attestation. */ + expectedOwnerClientInstanceId?: string } ): Promise sendSignal(id: string, signal: string): Promise diff --git a/src/main/providers/ssh-pty-provider.ts b/src/main/providers/ssh-pty-provider.ts index f218cc43eca..d48e09bba06 100644 --- a/src/main/providers/ssh-pty-provider.ts +++ b/src/main/providers/ssh-pty-provider.ts @@ -225,15 +225,17 @@ export class SshPtyProvider implements IPtyProvider { } async shutdown(id: string, opts: Parameters[1]): Promise { + // Both fences are omitted rather than sent undefined: a host that predates either must see no + // key at all, and the owner fence in particular must never reach it as a falsy claim. + const { expectedIncarnationId, expectedOwnerClientInstanceId } = opts await this.mux.request( 'pty.shutdown', { id: this.toRelayPtyId(id), immediate: opts.immediate ?? false, keepHistory: opts.keepHistory ?? false, - ...(opts.expectedIncarnationId === undefined - ? {} - : { expectedIncarnationId: opts.expectedIncarnationId }) + ...(expectedIncarnationId === undefined ? {} : { expectedIncarnationId }), + ...(expectedOwnerClientInstanceId === undefined ? {} : { expectedOwnerClientInstanceId }) }, relayTimeoutOptions(opts.deadlineMs) ) diff --git a/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts b/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts new file mode 100644 index 00000000000..13289416491 --- /dev/null +++ b/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts @@ -0,0 +1,75 @@ +// Three different refusals leave `reattachSshPtySessionForSpawn` carrying the same +// `SSH_SESSION_EXPIRED` text, and only one of them observed the process. That text is therefore not +// a verdict, and a caller that tests it with `.includes()` cannot tell "the host says this PTY is +// gone" from "the PTY is fine, its source stream needs rebuilding". +// +// It matters because the callers that DO test it act destructively: `spawn-execute.ts` expires the +// lease and deletes the in-memory ownership, which between them are the client's only record that a +// remote process exists. Erase both for a live PTY and #9819's sweep finds a host-attested, +// route-less, pane-bound shell on the next connect and SIGKILLs it. +// +// So this pins the discriminator the destructive branch keys on: the type, not the message. +import { describe, expect, it, vi } from 'vitest' +import { isSshPtyAbsentFromRelayError, SSH_SESSION_EXPIRED_ERROR } from './ssh-pty-errors' +import { reattachSshPtySessionForSpawn } from './ssh-pty-session-reattach' +import type { SshChannelMultiplexer } from '../ssh/ssh-channel-multiplexer' + +const CONNECTION = 'conn-1' +const SESSION = 'pty-1' + +function reattachAgainst(attach: () => Promise): Promise { + return reattachSshPtySessionForSpawn({ + mux: { request: vi.fn(attach) } as unknown as SshChannelMultiplexer, + connectionId: CONNECTION, + sessionId: SESSION, + options: { cols: 80, rows: 24 }, + exitRaceTracker: { + begin: () => 1, + didMatchingExitArrive: () => false, + finish: () => {} + } as never, + acceptLivePty: () => {} + }) +} + +async function refusalFrom(attach: () => Promise): Promise { + try { + await reattachAgainst(attach) + } catch (error) { + return error as Error + } + throw new Error('expected the reattach to be refused') +} + +describe('an SSH reattach refusal says whether the host observed the PTY', () => { + it('marks a relay that answered "not found" as positive evidence of absence', async () => { + const error = await refusalFrom(async () => { + throw new Error(`PTY "${SESSION}" not found`) + }) + + expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) + expect(isSshPtyAbsentFromRelayError(error)).toBe(true) + }) + + it('does not mark a restoreRequired refusal as absence, though it reads identically', async () => { + // The PTY attached. The relay answered about it. It is running. Only the source stream could + // not be resumed — see the `restoreRequired` carve-out in ssh-pty-errors.ts. + const error = await refusalFrom(async () => ({ + incarnationId: '11111111-1111-4111-8111-111111111111', + sourceRecovery: { status: 'restoreRequired', reason: 'checkpoint_unavailable' } + })) + + expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) + expect(isSshPtyAbsentFromRelayError(error)).toBe(false) + }) + + it('does not mark an identity mismatch as absence either', async () => { + // The id names a LIVE PTY that belongs to a different pane. + const error = await refusalFrom(async () => { + throw new Error(`PTY "${SESSION}" not found (identity mismatch)`) + }) + + expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) + expect(isSshPtyAbsentFromRelayError(error)).toBe(false) + }) +}) diff --git a/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts b/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts new file mode 100644 index 00000000000..e2d8ef6c9dc --- /dev/null +++ b/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts @@ -0,0 +1,310 @@ +// #9819, the client half: what the sweep actually asks the store and the host, and what it does +// with the answers. The rule itself is covered in ssh-relay-pty-ownership-proof.test.ts. +import { describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import type { IPtyProvider } from '../providers/types' +import type { PtyProcessInfo } from '../providers/pty-process-info' +import type { SshRemotePtyLease } from '../../shared/ssh-types' +import type { ForegroundProcessEvidence } from '../../shared/foreground-process-evidence' +import type { PersistedState } from '../../shared/persisted-state-types' +import { + upsertSshRemotePtyLease, + type SshPtyLeaseOperations +} from '../persistence/leasing-ssh-ptys/ssh-pty-lease-operations' +import { + RELAY_PTY_SWEEP_PASS_BUDGET_MS, + sweepOrphanedRelayPtys +} from './ssh-orphan-relay-pty-sweep' +import { RELAY_PTY_SWEEP_MIN_AGE_MS } from '../../shared/ssh-relay-pty-ownership-proof' + +const TARGET = 'target-1' +const OURS = 'client-instance-ours' +// A stable pane id is a UUID; anything else is stripped before supersede can match on it. +const LEAF = '11111111-2222-4333-8444-555555555555' + +const OBSERVATION = { authorityGeneration: 'gen-1', observationEpoch: 1, capturedAgeMs: 0 } + +/** The host looked and saw its own shell owning the terminal: nothing is running in the pane. */ +function idleShell(): ForegroundProcessEvidence { + return { ...OBSERVATION, verdict: 'live', processName: null, shellOwnsEveryTtyProcessGroup: true } +} + +function hostEntry(overrides: Partial = {}): PtyProcessInfo { + return { + id: `ssh:${TARGET}@@pty-1`, + incarnationId: 'inc-1', + cwd: '/home/user', + title: 'zsh', + ownerClientInstanceId: OURS, + hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS * 2, + paneBound: true, + foregroundProcessEvidence: idleShell(), + ...overrides + } +} + +function createHarness( + processes: PtyProcessInfo[], + leases: SshRemotePtyLease[] = [] +): { provider: IPtyProvider; store: Store; shutdown: ReturnType } { + const shutdown = vi.fn().mockResolvedValue(undefined) + const provider = { + listProcesses: vi.fn().mockResolvedValue(processes), + shutdown + } as unknown as IPtyProvider + const store = { + getSshRemotePtyLeases: vi.fn().mockReturnValue(leases) + } as unknown as Store + return { provider, store, shutdown } +} + +function run( + harness: ReturnType, + overrides: Partial[0]> = {} +): Promise { + return sweepOrphanedRelayPtys({ + targetId: TARGET, + store: harness.store, + provider: harness.provider, + clientInstanceId: OURS, + isSessionOwner: true, + routedPtyIds: [], + shouldContinue: () => true, + ...overrides + }) +} + +function lease(ptyId: string, state: SshRemotePtyLease['state']): SshRemotePtyLease { + return { ptyId, state } as SshRemotePtyLease +} + +describe('sweepOrphanedRelayPtys', () => { + it('stops an attested orphan, fenced on the incarnation the same listing published', async () => { + const harness = createHarness([hostEntry()]) + + await run(harness) + + expect(harness.shutdown).toHaveBeenCalledWith( + `ssh:${TARGET}@@pty-1`, + expect.objectContaining({ immediate: true, expectedIncarnationId: 'inc-1' }) + ) + }) + + it('asks the host to re-check ownership on the one call that cannot be undone', async () => { + // The stop is the only irreversible step in this flow, and until now the whole nine-condition + // rule was enforced only here, on the client that decided to make it. Naming the owner makes + // the host re-decide where the processes actually live. + const harness = createHarness([hostEntry()]) + + await run(harness) + + expect(harness.shutdown).toHaveBeenCalledWith( + `ssh:${TARGET}@@pty-1`, + expect.objectContaining({ expectedOwnerClientInstanceId: OURS }) + ) + }) + + it('does not act on an observation that aged out between the listing and the plan', async () => { + // The listing answered inside the budget, but this pass then spent longer than the evidence is + // good for. Staleness has to degrade to "leave it running". + let clock = 1_000_000 + const harness = createHarness([hostEntry()]) + harness.provider.listProcesses = vi.fn().mockImplementation(async () => { + clock += 1 + return [hostEntry()] + }) + + const pass = (maximumEvidenceAgeMs: number): Promise => + run(harness, { + now: () => clock, + maximumEvidenceAgeMs, + passBudgetMs: 60_000, + shouldContinue: () => { + clock += 20 + return true + } + }) + + await pass(10) + expect(harness.shutdown).not.toHaveBeenCalled() + + // Positive control: the same entry, the same elapsed time, a budget that covers it. + await pass(10_000) + expect(harness.shutdown).toHaveBeenCalledTimes(1) + }) + + it('leaves a PTY the caller just reattached alone', async () => { + const harness = createHarness([hostEntry()]) + + await run(harness, { routedPtyIds: ['pty-1'] }) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it.each([['attached'], ['detached']] as const)( + 'leaves a PTY holding a live %s lease alone', + async (state) => { + const harness = createHarness([hostEntry()], [lease('pty-1', state)]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + } + ) + + it('leaves a PTY with an undelivered stop to the replay pass', async () => { + // The kill-intent journal owns those: it re-fences and retries them, and a second stop issued + // from here would race that decision with weaker evidence. + const tombstoned = { + ...lease('pty-1', 'terminated'), + pendingKill: { requestedAt: 1, incarnationId: 'inc-1', attempts: 0 } + } as SshRemotePtyLease + const harness = createHarness([hostEntry()], [tombstoned]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('leaves a PTY whose lease this client expired alone', async () => { + // The reversal this guards: supersedeSiblingLeasesForPane, dropStalePty and the missing-surface + // refusal all write `expired` precisely BECAUSE they will not stop the remote process. + const harness = createHarness([hostEntry()], [lease('pty-1', 'expired')]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('leaves alone a lease the real supersede path expired when a pane re-leased', async () => { + // Drives the actual persistence operation rather than asserting the state by hand, so this + // stays true only while supersede really does leave the predecessor's process running. + const state: PersistedState = { + sshRemotePtyLeases: [ + { + targetId: TARGET, + ptyId: 'pty-1', + state: 'attached', + worktreeId: 'wt-1', + leafId: LEAF, + createdAt: 1, + updatedAt: 1 + } + ] + } as unknown as PersistedState + const operations: SshPtyLeaseOperations = { + state, + toStoredPtyId: (_targetId, ptyId) => ptyId, + toComparablePtyId: (_targetId, ptyId) => ptyId, + clearBindingsForTarget: () => {}, + clearBindingsForLeases: () => false, + flush: () => {}, + flushDurableStateOrThrowAsync: async () => {} + } + // The same pane re-leases under a new relay id; pty-1 is expired, never terminated. + upsertSshRemotePtyLease(operations, { + targetId: TARGET, + ptyId: 'pty-2', + state: 'attached', + worktreeId: 'wt-1', + leafId: LEAF + }) + expect(state.sshRemotePtyLeases?.find((entry) => entry.ptyId === 'pty-1')?.state).toBe( + 'expired' + ) + const harness = createHarness([hostEntry()], state.sshRemotePtyLeases ?? []) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('forwards the host foreground observation, so a busy pane is never swept', async () => { + // A `claude` the user launched by hand: Orca registered no agent session, so the entry carries + // no agentSessionOwners and only the host's own observation can save it. + const harness = createHarness([ + hostEntry({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: 'claude', + shellOwnsEveryTtyProcessGroup: false + } + }) + ]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('bounds the listing and every stop with one connect budget', async () => { + const harness = createHarness([hostEntry()]) + const start = 1_000_000 + + await run(harness, { now: () => start }) + + const deadline = vi.mocked(harness.provider.listProcesses).mock.calls[0]?.[0]?.deadlineMs + expect(deadline).toBe(start + RELAY_PTY_SWEEP_PASS_BUDGET_MS) + expect(harness.shutdown).toHaveBeenCalledWith( + `ssh:${TARGET}@@pty-1`, + expect.objectContaining({ deadlineMs: deadline }) + ) + }) + + it('does sweep a PTY whose lease this client already tombstoned without an order', async () => { + const harness = createHarness([hostEntry()], [lease('pty-1', 'terminated')]) + + await run(harness) + + expect(harness.shutdown).toHaveBeenCalledTimes(1) + }) + + it('asks the host nothing when this connection is not the session owner', async () => { + const harness = createHarness([hostEntry()]) + + await run(harness, { isSessionOwner: false }) + + expect(harness.provider.listProcesses).not.toHaveBeenCalled() + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('stops nothing against a host that publishes no attestation', async () => { + const legacy = hostEntry() + delete legacy.ownerClientInstanceId + delete legacy.hostAgeMs + delete legacy.paneBound + const harness = createHarness([legacy]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('swallows a failed listing rather than failing the connect it runs on', async () => { + const harness = createHarness([]) + vi.mocked(harness.provider.listProcesses).mockRejectedValue(new Error('relay went away')) + + await expect(run(harness)).resolves.toBeUndefined() + }) + + it('swallows a failed stop and leaves the order to the next connect', async () => { + const harness = createHarness([hostEntry()]) + harness.shutdown.mockRejectedValue(new Error('connection lost')) + + await expect(run(harness)).resolves.toBeUndefined() + }) + + it('abandons the pass when the attempt is superseded mid-flight', async () => { + const harness = createHarness([hostEntry()]) + let alive = true + vi.mocked(harness.provider.listProcesses).mockImplementation(async () => { + alive = false + return [hostEntry()] + }) + + await run(harness, { shouldContinue: () => alive }) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/ssh/ssh-orphan-relay-pty-sweep.ts b/src/main/ssh/ssh-orphan-relay-pty-sweep.ts new file mode 100644 index 00000000000..fbc72bf5ec8 --- /dev/null +++ b/src/main/ssh/ssh-orphan-relay-pty-sweep.ts @@ -0,0 +1,164 @@ +import type { Store } from '../persistence' +import type { IPtyProvider } from '../providers/types' +import { toAppSshPtyId, toRelaySshPtyId } from '../providers/ssh-pty-id' +import { + planRelayPtySweep, + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + RELAY_PTY_SWEEP_MIN_AGE_MS, + type RelayPtyOwnershipEvidence +} from '../../shared/ssh-relay-pty-ownership-proof' + +export type SshOrphanRelayPtySweepArgs = { + targetId: string + store: Store + provider: IPtyProvider + /** This client's persisted consumer identity for the target. */ + clientInstanceId: string + /** True only when the relay granted this connection the negotiated `session-owner` role. */ + isSessionOwner: boolean + /** Relay PTY ids this connect just reattached, plus any the caller otherwise knows are live. */ + routedPtyIds: Iterable + shouldContinue: () => boolean + now?: () => number + minimumHostAgeMs?: number + /** Absolute budget for the whole pass, in ms from its start. */ + passBudgetMs?: number + maximumEvidenceAgeMs?: number +} + +/** The two client-side claims the plan needs, read in one pass over the leases. + * + * `routed` is every relay PTY id this client still has a route to: a live lease, an id this + * connect reattached, or a stop it recorded and has not delivered. + * + * `expired` is separate on purpose. It is written by four paths, and every one of them + * deliberately leaves the remote process running: a pane re-leasing under a new relay id, a pane + * surface missing from the layout, a retired reattach, and a reattach the HOST answered "not + * found" for. Only the second of those is reached with the process provably alive — the layout + * refusal runs after `pty.attach` already succeeded (`restoreReattachedPtyRuntime`), which is + * precisely why its own comment reads "topology absence alone is not authority to kill a + * process". A reattach that failed on the transport writes nothing at all: it early-returns as + * `reattachAttemptsExhausted` and the lease stays `attached`, hence routed. + * + * Folding it into `routed` would work, but it would also lose the reason in the skip log, and this + * is the distinction the sweep most needs to be able to explain. */ +function clientClaims(args: SshOrphanRelayPtySweepArgs): { + routed: Set + expired: Set +} { + const routed = new Set(args.routedPtyIds) + const expired = new Set() + for (const lease of args.store.getSshRemotePtyLeases(args.targetId)) { + if (lease.state === 'expired') { + expired.add(lease.ptyId) + } else if (lease.state !== 'terminated') { + routed.add(lease.ptyId) + } + if (lease.pendingKill) { + routed.add(lease.ptyId) + } + } + return { routed, expired } +} + +function toEvidence( + targetId: string, + process: Awaited>[number] +): RelayPtyOwnershipEvidence { + return { + ptyId: toRelaySshPtyId(targetId, process.id), + ...(process.incarnationId ? { incarnationId: process.incarnationId } : {}), + ...(process.ownerClientInstanceId + ? { ownerClientInstanceId: process.ownerClientInstanceId } + : {}), + ...(typeof process.hostAgeMs === 'number' ? { hostAgeMs: process.hostAgeMs } : {}), + ...(typeof process.paneBound === 'boolean' ? { paneBound: process.paneBound } : {}), + ...(process.agentSessionOwners ? { agentSessionOwners: process.agentSessionOwners } : {}), + ...(process.foregroundProcessEvidence + ? { foregroundProcessEvidence: process.foregroundProcessEvidence } + : {}) + } +} + +/** One budget for the whole pass, because this is opportunistic cleanup bolted onto the most + * latency-sensitive and most failure-prone path in the app (#14830, #17830). Without it the + * listing and up to eight stops inherit the mux default and connect waits on all of them. + * Overrunning it yields an empty pass — the same outcome as finding nothing, never a failed + * connect. */ +export const RELAY_PTY_SWEEP_PASS_BUDGET_MS = 5_000 + +/** Stops the relay PTYs this client can prove it created and has since lost every route to. + * + * Runs after reattach, so a PTY this connect reclaimed is already routed and can never be a + * candidate. Best-effort and never throws: it is opportunistic cleanup on the connect path, and a + * failed connection is a much worse outcome than a slot left leaked for another session. + * + * Costs one `pty.listProcesses` per connect. That is the price of reconciling at all — there is no + * cheaper question than asking the authoritative host what it is holding. */ +export async function sweepOrphanedRelayPtys(args: SshOrphanRelayPtySweepArgs): Promise { + if (!args.isSessionOwner || !args.clientInstanceId || !args.shouldContinue()) { + return + } + const now = args.now ?? Date.now + const deadlineMs = now() + (args.passBudgetMs ?? RELAY_PTY_SWEEP_PASS_BUDGET_MS) + try { + const processes = await args.provider.listProcesses({ deadlineMs }) + // The instant the host's observations reached this client. Every later step — reading the + // leases, planning, issuing the stops — ages them, and the plan has to see that age. + const listedAtMs = now() + if (!args.shouldContinue() || now() >= deadlineMs) { + return + } + const claims = clientClaims(args) + const plan = planRelayPtySweep( + processes.map((process) => toEvidence(args.targetId, process)), + { + clientInstanceId: args.clientInstanceId, + isSessionOwner: args.isSessionOwner, + routedPtyIds: claims.routed, + expiredLeasePtyIds: claims.expired, + minimumHostAgeMs: args.minimumHostAgeMs ?? RELAY_PTY_SWEEP_MIN_AGE_MS, + evidenceAgeSinceListingMs: Math.max(0, now() - listedAtMs), + maximumEvidenceAgeMs: args.maximumEvidenceAgeMs ?? RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS + } + ) + if (plan.sweep.length === 0) { + return + } + await Promise.all( + plan.sweep.map(async (target) => { + if (!args.shouldContinue()) { + return + } + try { + // Two fences, both enforced by the host that owns the process. The incarnation stops a + // relay that renumbered its ids between the read and this call from hitting a stranger; + // the owner id makes the host re-check the ownership rule itself, so the one irreversible + // call in this flow is not authorized by the client alone. + await args.provider.shutdown(toAppSshPtyId(args.targetId, target.ptyId), { + immediate: true, + deadlineMs, + expectedIncarnationId: target.incarnationId, + expectedOwnerClientInstanceId: args.clientInstanceId + }) + console.log( + `[ssh-orphan-sweep] stopped orphaned relay PTY ${args.targetId}/${target.ptyId}` + ) + } catch (err) { + // Unverifiable, not failed: the next connect re-reads the inventory and decides again. + console.warn( + `[ssh-orphan-sweep] stop for ${args.targetId}/${target.ptyId} is unverifiable: ${ + err instanceof Error ? err.message : String(err) + }` + ) + } + }) + ) + } catch (err) { + console.warn( + `[ssh-orphan-sweep] pass on ${args.targetId} stopped early: ${ + err instanceof Error ? err.message : String(err) + }` + ) + } +} diff --git a/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts b/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts new file mode 100644 index 00000000000..ab069240681 --- /dev/null +++ b/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts @@ -0,0 +1,238 @@ +// The one test that spans both halves of the sweep. Every other test in this feature asserts on a +// hand-written `ForegroundProcessEvidence` literal, which is exactly how a foreground-only idle +// predicate survived review: the literals said `shellIsForeground: true` for an idle shell because +// that is what the author believed, and nothing ever produced one from a real process table. +// +// So this runs the REAL publisher (`resolveAgentForegroundProcessesBatch` -> +// `toForegroundProcessEvidence`, what `pty.listProcesses` calls) against the REAL client reader +// (`planRelayPtySweep`), over `ps` output captured verbatim from a Linux container driving a real +// `bash -i` on a real pty. The fixtures below are transcripts, not constructions. +// +// Read the shell's own row in each fixture. In `background` and `ctrlz` it is +// `pgid == tpgid`, `Ss+` — byte-identical to `idle`. That is the defect: a foreground-only +// predicate cannot see a job the user backgrounded or suspended, and the stop it authorizes +// SIGKILLs every process group on the tty. +import { describe, expect, it } from 'vitest' +import { + resolveAgentForegroundProcessesBatch, + toForegroundProcessEvidence +} from '../providers/agent-foreground-process-batch' +import { parseStrictProcessTableRows } from '../../shared/process-table-snapshot' +import { + planRelayPtySweep, + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + RELAY_PTY_SWEEP_MIN_AGE_MS, + type RelayPtySweepContext +} from '../../shared/ssh-relay-pty-ownership-proof' + +const OURS = 'client-instance-ours' + +/** `ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=` on debian:bookworm-slim, one capture per pane + * state, each with a `bash -i` on a pty forked by the harness. */ +const CAPTURES = { + /** Nothing running. The only sweepable state. */ + idle: { + rootPid: 3150, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3150 1 3150 3150 Ss+ bash -i', + ' 3151 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** `sleep 300` in the foreground. The shell's tpgid moved off its own pgid. */ + foreground: { + rootPid: 3155, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3155 1 3155 3156 Ss bash -i', + ' 3156 3155 3156 3156 S+ sleep 300', + ' 3157 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** `sleep 300 &`. The shell reads IDENTICALLY to `idle`; only the job's own row differs. */ + background: { + rootPid: 3152, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3152 1 3152 3152 Ss+ bash -i', + ' 3153 3152 3153 3152 S sleep 300', + ' 3154 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** `sleep 300` then Ctrl-Z. The shell again reads IDENTICALLY to `idle`. */ + ctrlz: { + rootPid: 3158, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3158 1 3158 3158 Ss+ bash -i', + ' 3159 3158 3159 3158 T sleep 300', + ' 3160 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + } +} as const + +function context(overrides: Partial = {}): RelayPtySweepContext { + return { + clientInstanceId: OURS, + isSessionOwner: true, + routedPtyIds: new Set(), + expiredLeasePtyIds: new Set(), + minimumHostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS, + evidenceAgeSinceListingMs: 0, + maximumEvidenceAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + ...overrides + } +} + +/** Everything the host does between reading `ps` and putting a record on the wire. */ +async function publish( + capture: { rootPid: number; table: readonly string[] }, + capturedAgeMs = 0 +): Promise> { + const rows = parseStrictProcessTableRows(capture.table.join('\n')) + const [result] = await resolveAgentForegroundProcessesBatch( + [{ rootPid: capture.rootPid, fallbackProcess: 'bash' }], + { rows } + ) + return toForegroundProcessEvidence(result, { + authorityGeneration: 'relay-generation-1', + observationEpoch: 1, + capturedAgeMs + }) +} + +/** Everything the client does with that record. Returns the plan for one orphan entry. */ +async function planFor( + capture: { rootPid: number; table: readonly string[] }, + overrides: { capturedAgeMs?: number; context?: Partial } = {} +): Promise> { + return planRelayPtySweep( + [ + { + ptyId: 'pty-1', + incarnationId: 'inc-1', + ownerClientInstanceId: OURS, + hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS * 2, + paneBound: true, + foregroundProcessEvidence: await publish(capture, overrides.capturedAgeMs) + } + ], + context(overrides.context) + ) +} + +function skipReason(plan: ReturnType): string | undefined { + return plan.skipped.find((entry) => entry.ptyId === 'pty-1')?.reason +} + +describe('what the host publishes about a pane, read by the sweep', () => { + it('records that a backgrounded and a suspended shell are indistinguishable at tpgid/pgid', () => { + // The premise of the whole file. If this ever fails, the fixtures drifted and every verdict + // below is testing something other than the defect. Pids differ between captures, so the + // comparison is of the shell row's shape: who its parent is, whether it leads its own process + // group, whether that group owns the terminal, and its state flags. + const shellShape = (capture: { rootPid: number; table: readonly string[] }): string => { + const row = parseStrictProcessTableRows(capture.table.join('\n')).find( + (candidate) => candidate.pid === capture.rootPid + )! + return [ + `ppid=${row.ppid}`, + `leadsOwnGroup=${row.pgid === row.pid}`, + `ownsTerminal=${row.tpgid === row.pgid}`, + `stat=${row.stat}` + ].join(' ') + } + + expect(shellShape(CAPTURES.idle)).toBe('ppid=1 leadsOwnGroup=true ownsTerminal=true stat=Ss+') + expect(shellShape(CAPTURES.background)).toBe(shellShape(CAPTURES.idle)) + expect(shellShape(CAPTURES.ctrlz)).toBe(shellShape(CAPTURES.idle)) + expect(shellShape(CAPTURES.foreground)).not.toBe(shellShape(CAPTURES.idle)) + }) + + it('sweeps an idle shell', async () => { + const evidence = await publish(CAPTURES.idle) + expect(evidence).toMatchObject({ + verdict: 'live', + processName: null, + shellOwnsEveryTtyProcessGroup: true + }) + + const plan = await planFor(CAPTURES.idle) + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) + + it('never sweeps a pane running a foreground job', async () => { + const evidence = await publish(CAPTURES.foreground) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.foreground) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('never sweeps a pane holding a backgrounded job', async () => { + // `sleep 300 &`, i.e. `pnpm build &` or `npm run dev &`. The shell handed the terminal back, + // so the pane looks idle; the job is alive in its own process group on the same tty and a + // stop would SIGKILL it. + const evidence = await publish(CAPTURES.background) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.background) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('never sweeps a pane holding a Ctrl-Z suspended job', async () => { + const evidence = await publish(CAPTURES.ctrlz) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.ctrlz) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('refuses an observation older than the pass it would authorize', async () => { + // Same idle capture that sweeps above; only its age differs. Staleness degrades to "leave it + // running", never to "stop it". + const stale = await planFor(CAPTURES.idle, { + capturedAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS + 1 + }) + expect(stale.sweep).toEqual([]) + expect(skipReason(stale)).toBe('host foreground observation is too old to authorize a stop') + + // And the client's own share of the age counts: a host stamp inside the budget still ages out + // while this pass reads leases and plans. + const agedOnTheClient = await planFor(CAPTURES.idle, { + capturedAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + context: { evidenceAgeSinceListingMs: 1 } + }) + expect(agedOnTheClient.sweep).toEqual([]) + expect(skipReason(agedOnTheClient)).toBe( + 'host foreground observation is too old to authorize a stop' + ) + }) + + it('never sweeps when the host itself is too degraded to answer', async () => { + // `main` has since added `recoverRemoteTerminalRuntime`, a self-driven reconnect on relay + // node-pty failure — a sweep trigger that fires exactly when the host is unwell. The publisher + // has to fail closed there: a capture that cannot locate the shell is `unverifiable`, which is + // its own verdict and never collapses into "idle" (docs/reference/ssh-execution-boundary.md). + const evidence = await publish({ rootPid: 999_999, table: CAPTURES.idle.table }) + expect(evidence).toMatchObject({ verdict: 'unverifiable', reason: 'root_missing' }) + + const plan = await planFor({ rootPid: 999_999, table: CAPTURES.idle.table }) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host could not observe the pane foreground process') + }) + + it('still reclaims a shell whose work really did finish', async () => { + // The feature must not degrade into a no-op. The background fixture's job is gone; what is + // left is the same orphaned shell, and it is swept. + const finished = { + rootPid: CAPTURES.background.rootPid, + table: CAPTURES.background.table.filter((line) => !line.includes('sleep 300')) + } + const plan = await planFor(finished) + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) +}) diff --git a/src/main/ssh/ssh-pty-consumer-recovery.test.ts b/src/main/ssh/ssh-pty-consumer-recovery.test.ts index 09405f9ee0b..d77e8c3c8dc 100644 --- a/src/main/ssh/ssh-pty-consumer-recovery.test.ts +++ b/src/main/ssh/ssh-pty-consumer-recovery.test.ts @@ -8,6 +8,21 @@ import { } from './ssh-pty-consumer-recovery' describe('SSH PTY consumer recovery', () => { + it('says out loud when it mints a new identity, because the sweep goes quiet after', () => { + // The id the host attests on every PTY it holds for us. Minting a new one is fail-safe — it + // can only under-sweep — but it makes the reaper silently stop working, so it must be visible. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const store = { + getSshPtyConsumerRecovery: vi.fn().mockReturnValue(null), + upsertSshPtyConsumerRecovery: vi.fn() + } as unknown as Store + + claimSshPtyConsumerRecovery('mint-logs-the-loss', store) + + expect(warn).toHaveBeenCalledWith(expect.stringContaining('minting a new consumer identity')) + warn.mockRestore() + }) + it('keeps a detached identity when a concurrent open finishes late', async () => { const targetId = 'remember-detach-race' const store = { diff --git a/src/main/ssh/ssh-pty-consumer-recovery.ts b/src/main/ssh/ssh-pty-consumer-recovery.ts index 1c357d32a12..db49a45c399 100644 --- a/src/main/ssh/ssh-pty-consumer-recovery.ts +++ b/src/main/ssh/ssh-pty-consumer-recovery.ts @@ -37,6 +37,15 @@ export function claimSshPtyConsumerRecovery( return current } const persisted = current ? null : store.getSshPtyConsumerRecovery(targetId) + if (!persisted) { + // Deliberately not stabilized: this id is also the consumer session's ownership identity, and + // reusing it without the generations the same dropped record carried would replay a stale + // owner generation at the relay. The cost is visible instead of silent — every relay PTY the + // host still attributes to the previous identity becomes permanently unsweepable (#9819). + console.warn( + `[ssh-pty-consumer] no recovery record for ${targetId}; minting a new consumer identity. Relay PTYs the host attributes to this client's previous identity can no longer be swept.` + ) + } const created: SshPtyConsumerRecoveryState = { clientInstanceId: persisted?.clientInstanceId ?? randomUUID(), detached: false, diff --git a/src/main/ssh/ssh-relay-session-orphan-sweep.test.ts b/src/main/ssh/ssh-relay-session-orphan-sweep.test.ts new file mode 100644 index 00000000000..fd4cbea8c34 --- /dev/null +++ b/src/main/ssh/ssh-relay-session-orphan-sweep.test.ts @@ -0,0 +1,293 @@ +// #9819 end to end on the client: the sweep runs only after reattach, only under a negotiated +// session-owner grant, and only against PTYs this relay itself attributes to this client. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { SshRelaySession } from './ssh-relay-session' +import { createMockDeps, mockDeploySuccess } from './ssh-relay-session-test-fixtures' + +const { muxRequestMock, openConsumerSessionMock } = vi.hoisted(() => ({ + muxRequestMock: vi.fn(), + openConsumerSessionMock: vi.fn(async (_mux: unknown, options: { clientInstanceId: string }) => ({ + state: { + mode: 'negotiated' as const, + clientInstanceId: options.clientInstanceId, + clientGeneration: 1, + ownerGeneration: 1, + ownerLease: 'test-owner-lease' + }, + resumed: false + })) +})) + +vi.mock('./ssh-relay-deploy', () => ({ deployAndLaunchRelay: vi.fn() })) +vi.mock('./ssh-pty-consumer-session', () => ({ + openSshPtyConsumerSession: openConsumerSessionMock +})) +vi.mock('../ipc/ssh-pty-output-intake-registry', () => ({ + acceptSshPtyOutputData: vi.fn().mockResolvedValue(undefined), + acceptSshPtyOutputExit: vi.fn().mockResolvedValue(undefined), + allocateSshPtyProviderGeneration: vi.fn(() => 17), + beginSshPtyOutputGenerationMigration: vi.fn(() => ({ + byPty: new Map(), + completion: Promise.resolve() + })), + closeSshPtyOutputGeneration: vi.fn(), + getSshPtyAcceptedSourceCheckpoints: vi.fn(() => []), + applySshPtySourceCancellationProof: vi.fn(() => true), + applySshPtySourceRecoveryCancellationProof: vi.fn(() => true), + installSshPtySourceAckPublisher: vi.fn(() => () => {}), + installSshPtySourceCancellationPublisher: vi.fn(() => () => {}) +})) +vi.mock('./ssh-relay-deploy-helpers', () => ({ execCommand: vi.fn().mockResolvedValue('') })) +vi.mock('./ssh-remote-orca-cli', () => ({ + runRemoteOrcaCli: vi.fn().mockResolvedValue({ exitCode: 0, stdout: '', stderr: '' }) +})) +vi.mock('./ssh-channel-multiplexer', () => ({ + SshChannelMultiplexer: class MockSshChannelMultiplexer { + notify = vi.fn() + notifyWithSettlement = vi.fn() + request = muxRequestMock + onNotification = vi.fn().mockReturnValue(() => {}) + onNotificationByMethod = vi.fn().mockReturnValue(() => {}) + onRequest = vi.fn().mockReturnValue(() => {}) + onDispose = vi.fn().mockReturnValue(() => {}) + dispose = vi.fn() + isDisposed = vi.fn().mockReturnValue(false) + } +})) +vi.mock('../agent-hooks/remote-managed-hook-installers', () => ({ + installRemoteManagedAgentHooks: vi.fn() +})) +vi.mock('../providers/ssh-pty-provider', () => ({ + SshPtyProvider: class MockSshPtyProvider { + onData = vi.fn().mockReturnValue(() => {}) + onReplay = vi.fn().mockReturnValue(() => {}) + onExit = vi.fn().mockReturnValue(() => {}) + attach = vi.fn().mockResolvedValue(undefined) + attachForReconnect = vi.fn().mockResolvedValue({}) + dispose = vi.fn() + } +})) +vi.mock('../providers/ssh-filesystem-provider', () => ({ + SshFilesystemProvider: class MockSshFilesystemProvider { + dispose = vi.fn() + } +})) +vi.mock('../providers/ssh-git-provider', () => ({ + SshGitProvider: class MockSshGitProvider {} +})) +vi.mock('../ipc/pty', () => ({ + registerSshPtyProvider: vi.fn(), + unregisterSshPtyProvider: vi.fn(), + getSshPtyProvider: vi.fn(), + getPtyIdsForConnection: vi.fn().mockReturnValue([]), + clearPtyOwnershipForConnection: vi.fn(), + clearProviderPtyState: vi.fn(), + deletePtyOwnership: vi.fn(), + setPtyOwnership: vi.fn(), + restorePtyIncarnation: vi.fn(), + isCurrentPtyExit: vi.fn(() => true), + answerStartupTerminalColorQueriesForPty: vi.fn((_id: string, data: string) => data) +})) +vi.mock('../providers/ssh-filesystem-dispatch', () => ({ + registerSshFilesystemProvider: vi.fn(), + unregisterSshFilesystemProvider: vi.fn(), + getSshFilesystemProvider: vi.fn().mockReturnValue({ dispose: vi.fn() }) +})) +vi.mock('../providers/ssh-git-dispatch', () => ({ + registerSshGitProvider: vi.fn(), + unregisterSshGitProvider: vi.fn() +})) + +const { getSshPtyProvider, getPtyIdsForConnection } = await import('../ipc/pty') + +const OUR_CLIENT = 'client-instance-1' + +// One target per test. `claimSshPtyConsumerRecovery` keeps a module-level map keyed on target and +// mints a FRESH clientInstanceId whenever it is asked for a target it already holds a live entry +// for — so a second `establish` on a shared target silently stops matching the host attestation and +// every assertion after the first passes for the wrong reason. +let targetSeq = 0 +function nextTarget(): string { + targetSeq += 1 + return `target-${targetSeq}` +} + +const OBSERVATION = { authorityGeneration: 'gen-1', observationEpoch: 1, capturedAgeMs: 0 } + +function hostEntry( + target: string, + overrides: Record = {} +): Record { + return { + id: `ssh:${target}@@pty-orphan`, + incarnationId: 'inc-orphan', + cwd: '/home/user', + title: 'zsh', + // The relay stamps this from the live consumer grant, so it names THIS client. + ownerClientInstanceId: OUR_CLIENT, + hostAgeMs: 120_000, + paneBound: true, + // The same listing's host observation: the shell owns the terminal, nothing is running. + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: null, + shellOwnsEveryTtyProcessGroup: true + }, + ...overrides + } +} + +describe('SshRelaySession orphaned relay PTY sweep', () => { + let warn: ReturnType + + beforeEach(() => { + vi.clearAllMocks() + warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + muxRequestMock.mockReset() + muxRequestMock.mockResolvedValue([]) + mockDeploySuccess() + vi.mocked(getPtyIdsForConnection).mockReturnValue([]) + }) + + afterEach(() => { + warn.mockRestore() + }) + + async function establish( + target: string, + processes: Record[], + leases: { ptyId: string; state: string }[] = [] + ): Promise<{ shutdown: ReturnType; listProcesses: ReturnType }> { + const deps = createMockDeps() + // Why the recovery row: it pins this session's clientInstanceId, and the comparison is + // meaningless unless the id it uses is the persisted one. + vi.mocked(deps.mockStore.getSshPtyConsumerRecovery).mockReturnValue({ + targetId: target, + clientInstanceId: OUR_CLIENT, + serverBuildId: 'build-1', + clientGeneration: 1, + ownerGeneration: 1, + ownerLease: 'test-owner-lease' + } as ReturnType) + vi.mocked(deps.mockStore.getSshRemotePtyLeases).mockReturnValue( + leases.map((lease) => ({ targetId: target, ...lease })) as ReturnType< + typeof deps.mockStore.getSshRemotePtyLeases + > + ) + const shutdown = vi.fn().mockResolvedValue(undefined) + const listProcesses = vi.fn().mockResolvedValue(processes) + vi.mocked(getSshPtyProvider).mockReturnValue({ + attachForReconnect: vi.fn().mockResolvedValue({}), + listProcesses, + shutdown, + dispose: vi.fn() + } as unknown as ReturnType) + + const session = new SshRelaySession( + target, + deps.getMainWindow, + deps.mockStore, + deps.mockPortForward + ) + await session.establish(deps.mockConn) + // Both guards exist because every "never stops" case below is trivially satisfiable. The pass + // has to have run, and it has to have run under the identity the host attests — a session that + // minted a fresh one compares against nothing and skips everything for the wrong reason. + expect(listProcesses).toHaveBeenCalledTimes(1) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('minting a new consumer identity') + ) + return { shutdown, listProcesses } + } + + it('stops an attested orphan the client has no lease for', async () => { + const target = nextTarget() + const { shutdown } = await establish(target, [hostEntry(target)]) + + expect(shutdown).toHaveBeenCalledWith( + `ssh:${target}@@pty-orphan`, + expect.objectContaining({ immediate: true, expectedIncarnationId: 'inc-orphan' }) + ) + }) + + it('never stops a PTY that still holds a live lease', async () => { + const target = nextTarget() + const { shutdown } = await establish( + target, + [hostEntry(target, { id: `ssh:${target}@@pty-live`, incarnationId: 'inc-live' })], + [{ ptyId: 'pty-live', state: 'detached' }] + ) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a PTY whose lease this client expired rather than ordered stopped', async () => { + // What reaches this state in the field: a pane re-leased under a new relay id, a reattach that + // failed on the transport (dropStalePty), or a pane surface missing from the layout. All three + // leave the remote process running on purpose. + const target = nextTarget() + const { shutdown } = await establish( + target, + [hostEntry(target, { id: `ssh:${target}@@pty-gone`, incarnationId: 'inc-gone' })], + [{ ptyId: 'pty-gone', state: 'expired' }] + ) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a pane this relay observes running a foreground process', async () => { + // agentSessionOwners is empty here — the user typed `claude` themselves — so the only thing + // between a live agent and a stop is the host's own foreground observation. + const target = nextTarget() + const { shutdown } = await establish(target, [ + hostEntry(target, { + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: 'claude', + shellOwnsEveryTtyProcessGroup: false + } + }) + ]) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a pane whose foreground observation the relay could not make', async () => { + const target = nextTarget() + const { shutdown } = await establish(target, [ + hostEntry(target, { + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'unverifiable', + reason: 'table_unreadable' + } + }) + ]) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a PTY this relay attributes to a different client', async () => { + const target = nextTarget() + const { shutdown } = await establish(target, [ + hostEntry(target, { ownerClientInstanceId: 'someone-elses-laptop' }) + ]) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops anything a relay predating the attestation lists', async () => { + const target = nextTarget() + const legacy = hostEntry(target) + delete legacy.ownerClientInstanceId + delete legacy.hostAgeMs + delete legacy.paneBound + delete legacy.foregroundProcessEvidence + + const { shutdown } = await establish(target, [legacy]) + + expect(shutdown).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/ssh/ssh-relay-session.ts b/src/main/ssh/ssh-relay-session.ts index 68254fbb108..04fe6020749 100644 --- a/src/main/ssh/ssh-relay-session.ts +++ b/src/main/ssh/ssh-relay-session.ts @@ -10,6 +10,7 @@ import { isRelayEndpointHeldError } from './ssh-relay-endpoint-incumbent' import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair' import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause' import { replayPendingSshPtyKills } from './ssh-pending-pty-kill-replay' +import { sweepOrphanedRelayPtys } from './ssh-orphan-relay-pty-sweep' import { SshChannelMultiplexer } from './ssh-channel-multiplexer' import { SshPtyProvider } from '../providers/ssh-pty-provider' import type { SshPtyAttachResult } from '../providers/ssh-pty-session-reattach' @@ -2398,6 +2399,17 @@ export class SshRelaySession { Array.from(attachedLeaseIds) ) } + // Why last: reclaiming comes first, so every PTY this connect could route to is routed before + // anything asks which ones are unreachable (#9819). + await sweepOrphanedRelayPtys({ + targetId: this.targetId, + store: this.store, + provider: ptyProvider, + clientInstanceId: this.ptyConsumerClientInstanceId, + isSessionOwner: this.activePtyConsumerOwner() !== null, + routedPtyIds: ptyIds, + shouldContinue + }) } private async reattachKnownPty(args: { diff --git a/src/relay/pty-handler-inventory-process-evidence.test.ts b/src/relay/pty-handler-inventory-process-evidence.test.ts index 17896f67c72..1c8b59bc6de 100644 --- a/src/relay/pty-handler-inventory-process-evidence.test.ts +++ b/src/relay/pty-handler-inventory-process-evidence.test.ts @@ -157,22 +157,32 @@ describe('PtyHandler inventory foreground evidence', () => { expect((await listProcesses())[0].title).toBe('node') }) - it.each([1, 8])('visits the host table exactly once for %s panes', async (paneCount) => { - const table = Array.from({ length: paneCount }, (_, index) => - paneRows(10_000 + index * 10, ['node /opt/codex']) - ).flat() - const { rows, reads } = countingRows(table) - mockGetStrictProcessTableSnapshot.mockResolvedValue(rows) - for (let index = 0; index < paneCount; index += 1) { - await spawnPane(10_000 + index * 10, 'zsh') + // The cost that matters is per-CAPTURE, not per-pane: the defect this guards against is a + // full-table walk for every pane, which is what an O(PTY x rows) inventory looked like. Two + // linear passes build the two indexes the resolver reads — parent/child correlation, and which + // process groups occupy each controlling terminal — and neither grows with the pane count. + const CAPTURE_PASSES = 2 + + it.each([1, 8])( + 'walks the host table a fixed number of times for %s panes', + async (paneCount) => { + const table = Array.from({ length: paneCount }, (_, index) => + paneRows(10_000 + index * 10, ['node /opt/codex']) + ).flat() + const { rows, reads } = countingRows(table) + mockGetStrictProcessTableSnapshot.mockResolvedValue(rows) + for (let index = 0; index < paneCount; index += 1) { + await spawnPane(10_000 + index * 10, 'zsh') + } + + const listed = await listProcesses() + + expect(listed).toHaveLength(paneCount) + expect(listed.every((entry) => entry.title === 'codex')).toBe(true) + expect(mockGetStrictProcessTableSnapshot).toHaveBeenCalledTimes(1) + // Linear in the capture — NOT one full-table walk per pane, which would be + // `table.length * paneCount` here. + expect(reads()).toBe(table.length * CAPTURE_PASSES) } - - const listed = await listProcesses() - - expect(listed).toHaveLength(paneCount) - expect(listed.every((entry) => entry.title === 'codex')).toBe(true) - expect(mockGetStrictProcessTableSnapshot).toHaveBeenCalledTimes(1) - // One linear index pass — NOT one full-table walk per pane. - expect(reads()).toBe(table.length) - }) + ) }) diff --git a/src/relay/pty-handler-ownership-attestation.test.ts b/src/relay/pty-handler-ownership-attestation.test.ts new file mode 100644 index 00000000000..ec5f1beb5e3 --- /dev/null +++ b/src/relay/pty-handler-ownership-attestation.test.ts @@ -0,0 +1,283 @@ +// The host half of #9819: a client may only reap a relay PTY it can prove it created, so the relay +// has to say who created each one. The attestation is read from the live consumer grant, never from +// a spawn parameter — otherwise it would just echo the caller's claim back at it. +import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' + +const { mockPtySpawn, mockPtyInstance, mockCreateShellPromptReadinessProbe } = vi.hoisted(() => ({ + mockPtySpawn: vi.fn(), + mockCreateShellPromptReadinessProbe: vi.fn(), + mockPtyInstance: { + pid: process.pid, + onData: vi.fn(), + onExit: vi.fn(), + write: vi.fn(), + resize: vi.fn(), + kill: vi.fn(), + clear: vi.fn(), + pause: vi.fn(), + resume: vi.fn() + } +})) + +vi.mock('node-pty', () => ({ spawn: mockPtySpawn })) +vi.mock('../main/pty/posix-pty-process-groups', () => ({ + forceKillPosixPtyProcessGroups: vi.fn((_pid: number, fallback: () => void) => fallback()) +})) +vi.mock('../main/shell-prompt-readiness-probe', () => ({ + createShellPromptReadinessProbe: mockCreateShellPromptReadinessProbe +})) + +import type { PtyHandler } from './pty-handler' +import { + beginPtyHandlerTest, + endPtyHandlerTest, + type MockDispatcher +} from './pty-handler-test-harness' +import { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } from '../shared/process-table-snapshot' + +const PANE_KEY = 'tab-agent:22222222-2222-4222-8222-222222222222' + +type Summary = { + id: string + paneBound?: boolean + hostAgeMs?: number + ownerClientInstanceId?: string + foregroundProcessEvidence?: { capturedAgeMs: number } +} + +describe('PtyHandler publishes host-attested PTY ownership', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + + async function spawnFrom( + clientId: number, + params: Record = {} + ): Promise<{ id: string }> { + mockPtySpawn.mockReturnValue({ ...mockPtyInstance, onData: vi.fn(), onExit: vi.fn() }) + return (await dispatcher.callRequest('pty.spawn', params, { + clientId, + isStale: () => false + } as never)) as { id: string } + } + + async function listProcesses(): Promise { + return (await dispatcher.callRequest('pty.listProcesses', {})) as Summary[] + } + + beforeEach(() => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + handler.setConsumerIdentityResolver((clientId) => (clientId === 7 ? 'client-A' : null)) + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('attributes a pane spawn to the identity the consumer grant names', async () => { + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + vi.advanceTimersByTime(45_000) + + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(entry?.ownerClientInstanceId).toBe('client-A') + expect(entry?.paneBound).toBe(true) + expect(entry?.hostAgeMs).toBeGreaterThanOrEqual(45_000) + }) + + it('omits the attestation entirely when the connection holds no active grant', async () => { + const { id } = await spawnFrom(9, { env: { ORCA_PANE_KEY: PANE_KEY } }) + + const entry = (await listProcesses()).find((process) => process.id === id) + + // Absent, not empty-string or null: a reader must be able to tell "unattested" from any value. + expect(entry).not.toHaveProperty('ownerClientInstanceId') + }) + + it('never attests a revived PTY, so a restored session is not sweepable', async () => { + // The load-bearing invariant of #9819's host half, and the one most likely to be "helpfully" + // broken later: revive replays state a client serialized, which is not this host observing who + // asked for the shell. A revived PTY *does* get paneBound: true (paneKey is restored) and a + // fresh createdAt, so the omitted attestation is the only thing standing between a relay + // restart and a sweep of the entire restored session. + const revivedId = 'pty-revived-1' + await dispatcher.callRequest( + 'pty.revive', + { + state: JSON.stringify([ + { + id: revivedId, + pid: process.pid, + cwd: process.cwd(), + paneKey: PANE_KEY, + cols: 80, + rows: 24 + } + ]) + }, + { clientId: 7, isStale: () => false } as never + ) + + const entry = (await listProcesses()).find((process) => process.id === revivedId) + expect(entry, 'revive should have produced a live PTY entry').toBeDefined() + // paneBound is true, which is exactly why the missing attestation has to be asserted: + // every other sweep precondition is satisfied by a revived pane. + expect(entry?.paneBound).toBe(true) + expect(entry?.ownerClientInstanceId).toBeUndefined() + }) + + it('reports a bare shell as not pane-bound', async () => { + const { id } = await spawnFrom(7, {}) + + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(entry?.paneBound).toBe(false) + expect(entry?.ownerClientInstanceId).toBe('client-A') + }) + + it('dates the foreground observation instead of stamping it fresh', async () => { + // `capturedAgeMs` used to be a hardcoded 0 with no reader anywhere, so the one field that + // exists to bound staleness asserted the evidence was never stale. It now carries the + // worst-case age of the TTL-shared capture the record was derived from. + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBe( + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + ) + }) +}) + +// CodeRabbit's unaddressed note, and the asymmetry behind it: `pty.spawn` and `pty.attach` both +// take a request context and both check it, while `pty.shutdown` — the one call that irreversibly +// destroys a user's running process — took none, so the entire ownership rule was enforced only on +// the client that decided to make the call. +describe('PtyHandler authorizes a fenced stop against its own attestation', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + + async function spawnFrom(clientId: number): Promise<{ id: string }> { + mockPtySpawn.mockReturnValue({ ...mockPtyInstance, onData: vi.fn(), onExit: vi.fn() }) + return (await dispatcher.callRequest('pty.spawn', { env: { ORCA_PANE_KEY: PANE_KEY } }, { + clientId, + isStale: () => false + } as never)) as { id: string } + } + + async function isStillHeld(id: string): Promise { + const entries = (await dispatcher.callRequest('pty.listProcesses', {})) as Summary[] + return entries.some((entry) => entry.id === id) + } + + /** What actually reaches the process. A refusal has to leave this untouched — the point of the + * check is the process, not the error. */ + function killSignals(): unknown[][] { + return mockPtyInstance.kill.mock.calls + } + + function stop(id: string, params: Record, clientId: number): Promise { + return dispatcher.callRequest('pty.shutdown', { id, immediate: false, ...params }, { + clientId, + isStale: () => false + } as never) + } + + beforeEach(() => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + handler.setConsumerIdentityResolver((clientId) => + clientId === 7 ? 'client-A' : clientId === 8 ? 'client-B' : null + ) + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('stops a PTY when the connection and the host agree on the owner', async () => { + const { id } = await spawnFrom(7) + + await expect( + stop(id, { expectedOwnerClientInstanceId: 'client-A' }, 7) + ).resolves.toBeUndefined() + expect(killSignals()).toEqual([['SIGTERM']]) + }) + + it('refuses when another client asserts our identity, and leaves the process running', async () => { + // The claim is a parameter, so a confused or displaced client can send any value it likes. + // What it cannot do is authenticate as that identity on this connection. + const { id } = await spawnFrom(7) + + await expect(stop(id, { expectedOwnerClientInstanceId: 'client-A' }, 8)).rejects.toThrow( + /requester is not the attested owner/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(id)).toBe(true) + }) + + it('refuses when the connection holds no grant at all', async () => { + const { id } = await spawnFrom(7) + + await expect(stop(id, { expectedOwnerClientInstanceId: 'client-A' }, 99)).rejects.toThrow( + /requester is not the attested owner/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(id)).toBe(true) + }) + + it('refuses a PTY this host never attested, even to the client that asked', async () => { + // The revived-PTY case. Both sweep preconditions a client can see are satisfied — pane-bound, + // old enough — and only the host knows it never recorded a creator for it. + const revivedId = 'pty-revived-fence' + await dispatcher.callRequest( + 'pty.revive', + { + state: JSON.stringify([ + { + id: revivedId, + pid: process.pid, + cwd: process.cwd(), + paneKey: PANE_KEY, + cols: 80, + rows: 24 + } + ]) + }, + { clientId: 7, isStale: () => false } as never + ) + + await expect(stop(revivedId, { expectedOwnerClientInstanceId: 'client-A' }, 7)).rejects.toThrow( + /this host attested no such owner/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(revivedId)).toBe(true) + }) + + it('leaves an ordinary teardown that names no owner exactly as it was', async () => { + // Rule 1's obligation: an old client, and every non-sweep caller on a current one, omits the + // field. The host must not start refusing a stop it is obliged to honour. + const { id } = await spawnFrom(7) + + await expect(stop(id, {}, 7)).resolves.toBeUndefined() + expect(killSignals()).toEqual([['SIGTERM']]) + }) + + it('rejects a malformed owner claim rather than ignoring it', async () => { + const { id } = await spawnFrom(7) + + await expect(stop(id, { expectedOwnerClientInstanceId: '' }, 7)).rejects.toThrow( + /Invalid expectedOwnerClientInstanceId/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(id)).toBe(true) + }) +}) diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index f8bd2351cf2..25945203878 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -70,6 +70,7 @@ import { } from '../main/providers/agent-foreground-process' import { getStrictProcessTableSnapshot, + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS, type ProcessTableRow } from '../shared/process-table-snapshot' import type { ForegroundProcessEvidence } from '../shared/foreground-process-evidence' @@ -190,6 +191,13 @@ type ManagedPty = { startupIngressIntent?: ReturnType ownerBackend: PtyOwnerBackend agentSessionOwners?: AgentSessionOwnerBinding[] + /** Host clock, host-relative only: published as an age so no client has to trust our wall clock. */ + createdAt: number + /** The authenticated consumer identity that asked this host to create this PTY, read from the + * live grant rather than from a spawn parameter. Absent whenever the host could not attest one + * (no consumer session, or a revive replaying state some other client serialized), and absence + * must never be read as "nobody owns it". */ + ownerClientInstanceId?: string } type RelayAgentSessionCreateResult = { @@ -369,6 +377,14 @@ type PtyProcessSummary = { terminalHandle?: string foregroundProcessEvidence?: ForegroundProcessEvidence agentSessionOwners?: AgentSessionOwnerBinding[] + /** Age on the HOST's clock. Published instead of a creation timestamp so a client with a skewed + * clock cannot compute a negative or enormous age and act on it. */ + hostAgeMs?: number + /** True when this PTY was spawned for an Orca pane (`ORCA_PANE_KEY`). False means a bare relay + * shell. Absent from a host that predates the field — which is neither. */ + paneBound?: boolean + /** See {@link ManagedPty.ownerClientInstanceId}. Omitted when this host cannot attest one. */ + ownerClientInstanceId?: string } type SerializedPtyEntry = { @@ -456,6 +472,7 @@ export class PtyHandler { private consumerPausedOutputPtys = new Set() private removeLegacyCapacityListener: (() => void) | null = null private sourcePublication: RelayPtySourcePublication | null = null + private consumerIdentityResolver: ((clientId: number) => string | null) | null = null private lastInputAtByPty = new Map() private interactiveOutputCharsByPty = new Map() private pendingSpawnCount = 0 @@ -510,6 +527,12 @@ export class PtyHandler { this.sourcePublication = publication } + /** Supplies the authenticated client identity behind a transport connection, so a spawn can be + * attributed to the consumer session that requested it. */ + setConsumerIdentityResolver(resolve: ((clientId: number) => string | null) | null): void { + this.consumerIdentityResolver = resolve + } + handleSourceCreditAvailable(id: string): void { this.sourcePublication?.onCreditAvailable(id) } @@ -991,7 +1014,7 @@ export class PtyHandler { private registerHandlers(): void { this.dispatcher.onRequest('pty.spawn', (p, context) => this.spawn(p, context)) this.dispatcher.onRequest('pty.attach', (p, context) => this.attach(p, context)) - this.dispatcher.onRequest('pty.shutdown', (p) => this.shutdown(p)) + this.dispatcher.onRequest('pty.shutdown', (p, context) => this.shutdown(p, context)) this.dispatcher.onRequest('pty.sendSignal', (p) => this.sendSignal(p)) this.dispatcher.onRequest('pty.getCwd', (p) => this.getCwd(p)) this.dispatcher.onRequest('pty.getInitialCwd', (p) => this.getInitialCwd(p)) @@ -1871,11 +1894,15 @@ export class PtyHandler { params.startupIngressVersion === PTY_STARTUP_INGRESS_VERSION ? parsePtyStartupIngressIntent(params.startupIngress) : undefined + const ownerClientInstanceId = + context === undefined ? null : (this.consumerIdentityResolver?.(context.clientId) ?? null) const managed: ManagedPty = { id, incarnationId: randomUUID(), pty: term, initialCwd: cwd, + createdAt: Date.now(), + ...(ownerClientInstanceId ? { ownerClientInstanceId } : {}), buffered: new RecentPtyOutputBuffer({ preserveChunkBoundaries: false, limit: REPLAY_BUFFER_MAX @@ -2097,7 +2124,7 @@ export class PtyHandler { return { cols: managed.pty.cols, rows: managed.pty.rows } } - private async shutdown(params: Record): Promise { + private async shutdown(params: Record, context?: RequestContext): Promise { const id = params.id as string const immediate = params.immediate as boolean const expectedIncarnationId = params.expectedIncarnationId @@ -2107,6 +2134,14 @@ export class PtyHandler { ) { throw new Error('Invalid expectedIncarnationId') } + const expectedOwnerClientInstanceId = params.expectedOwnerClientInstanceId + if ( + expectedOwnerClientInstanceId !== undefined && + (typeof expectedOwnerClientInstanceId !== 'string' || + expectedOwnerClientInstanceId.length === 0) + ) { + throw new Error('Invalid expectedOwnerClientInstanceId') + } const managed = this.ptys.get(id) if (!managed) { return @@ -2114,6 +2149,9 @@ export class PtyHandler { if (expectedIncarnationId !== undefined && expectedIncarnationId !== managed.incarnationId) { throw new Error(`PTY incarnation mismatch for ${id}`) } + if (expectedOwnerClientInstanceId !== undefined) { + this.assertShutdownOwnership(id, managed, expectedOwnerClientInstanceId, context) + } // Why: `pty.shutdown` is the only authoritative statement this host ever gets that a tab is // gone. Record it before the kill request, because the kill is the part that can fail: an agent // that survives teardown otherwise keeps posting hooks the relay forwards as a live agent pane @@ -2138,6 +2176,34 @@ export class PtyHandler { } } + /** Re-decide, on the host, whether the caller may destroy this PTY. + * + * `pty.shutdown` is irreversible and its siblings `pty.spawn`/`pty.attach` already take a + * request context; without this the whole ownership rule lived on the client, on the one call + * that cannot be taken back. Both halves are checked here because either alone is an echo: the + * connection must still authenticate as that consumer identity (so a claim cannot be asserted), + * and this host must have recorded that same identity as the PTY's creator at spawn (so the + * caller cannot reach a PTY it never made). + * + * Only callers that opt in are checked. An ordinary pane teardown does not pass the field, and + * must not: a revived PTY carries no attested owner at all, and a host predating the attestation + * would refuse stops it is obliged to honour. */ + private assertShutdownOwnership( + id: string, + managed: ManagedPty, + expectedOwnerClientInstanceId: string, + context: RequestContext | undefined + ): void { + const requester = + context === undefined ? null : (this.consumerIdentityResolver?.(context.clientId) ?? null) + if (requester !== expectedOwnerClientInstanceId) { + throw new Error(`PTY "${id}" stop refused: requester is not the attested owner`) + } + if (managed.ownerClientInstanceId !== expectedOwnerClientInstanceId) { + throw new Error(`PTY "${id}" stop refused: this host attested no such owner`) + } + } + /** Record that this pane's client surface is gone, and tell the hook server so the pane's cached * agent status stops being replayed to reconnecting clients. Returns false when there is no pane * surface to retire. */ @@ -2406,9 +2472,15 @@ export class PtyHandler { let evidenceRows: readonly ProcessTableRow[] | null = null let evidenceResults: BatchedForegroundProcessResult[] = [] const evidenceEpoch = ++this.foregroundEvidenceEpoch + // Worst-case capture time for the snapshot below, not the instant its await settled: the + // reader may serve a TTL-cached table, so the observation can already be one window old. The + // loop that follows can await per entry, so each record is stamped against this rather than + // carrying a shared constant. + let evidenceCapturedAtMs = Date.now() if (process.platform !== 'win32' && managedEntries.length > 0) { try { evidenceRows = await getStrictProcessTableSnapshot() + evidenceCapturedAtMs = Date.now() - PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS evidenceResults = await resolveAgentForegroundProcessesBatch( managedEntries.map(([, managed]) => ({ rootPid: managed.pty.pid, @@ -2442,7 +2514,7 @@ export class PtyHandler { { authorityGeneration: this.ptyIdMintEpoch, observationEpoch: evidenceEpoch, - capturedAgeMs: 0 + capturedAgeMs: Math.max(0, Date.now() - evidenceCapturedAtMs) } ) : undefined @@ -2451,6 +2523,11 @@ export class PtyHandler { incarnationId: managed.incarnationId, cwd: managed.initialCwd, title, + hostAgeMs: Math.max(0, Date.now() - managed.createdAt), + paneBound: Boolean(managed.paneKey ?? managed.attachIdentity?.paneKey), + ...(managed.ownerClientInstanceId + ? { ownerClientInstanceId: managed.ownerClientInstanceId } + : {}), ...(managed.worktreeId ? { worktreeId: managed.worktreeId } : {}), ...(managed.terminalHandle ? { terminalHandle: managed.terminalHandle } : {}), ...(foregroundProcessEvidence ? { foregroundProcessEvidence } : {}), @@ -2624,6 +2701,10 @@ export class PtyHandler { incarnationId: randomUUID(), pty: term, initialCwd: entry.cwd, + createdAt: Date.now(), + // Deliberately no ownerClientInstanceId: revive replays state a client serialized, which is + // not this host observing who asked for the shell. Unattested means never swept. + buffered: new RecentPtyOutputBuffer({ preserveChunkBoundaries: false, limit: REPLAY_BUFFER_MAX diff --git a/src/relay/relay-runtime-services.ts b/src/relay/relay-runtime-services.ts index 485c9e787d1..73e03242af6 100644 --- a/src/relay/relay-runtime-services.ts +++ b/src/relay/relay-runtime-services.ts @@ -45,6 +45,11 @@ export class RelayRuntimeServices { (id, paused) => this.ptyHandler.setConsumerDeliveryPaused(id, paused), (id) => this.ptyHandler.handleSourceCreditAvailable(id) ) + // Why wired after construction: the handler is built first, but PTY ownership has to be + // attested from the consumer grant the adapter holds. + this.ptyHandler.setConsumerIdentityResolver((clientId) => + this.ptyConsumerSessionAdapter.clientInstanceIdFor(clientId) + ) this.ptySourcePublication = new RelayPtySourcePublication( dispatcher, this.ptyConsumerSessionAdapter, diff --git a/src/relay/ssh-pty-consumer-session-adapter.ts b/src/relay/ssh-pty-consumer-session-adapter.ts index 83e61f17fa3..906edd47ff8 100644 --- a/src/relay/ssh-pty-consumer-session-adapter.ts +++ b/src/relay/ssh-pty-consumer-session-adapter.ts @@ -112,6 +112,12 @@ export class SshPtyConsumerSessionAdapter { }) } + /** The authenticated client identity behind a transport connection, or null when it holds no + * active grant. Used to stamp host-attested ownership on a PTY at spawn. */ + clientInstanceIdFor(clientId: number): string | null { + return this.session.activeClientInstanceId(String(clientId)) + } + openDelivery( clientId: number, id: string, diff --git a/src/shared/foreground-process-evidence.ts b/src/shared/foreground-process-evidence.ts index 08f9f208dbb..bbcfd091d60 100644 --- a/src/shared/foreground-process-evidence.ts +++ b/src/shared/foreground-process-evidence.ts @@ -2,12 +2,35 @@ export type ForegroundEvidenceObservation = { authorityGeneration: string observationEpoch: number - /** Age at serialization; receivers rebase this onto their monotonic clock. */ + /** How old the underlying process-table capture was when this record was serialized, measured on + * the OBSERVING host's clock so no clock skew enters it. Receivers rebase it onto their own + * monotonic clock by adding the time since the carrying response arrived. + * + * It is an upper bound, not an estimate: the capture is TTL-shared, so a reader may be served + * one up to `PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS` older than its own await, and the + * producer stamps for that worst case. Erring old is the safe direction for every consumer — + * the only one that acts destructively refuses stale evidence. */ capturedAgeMs: number } export type ForegroundProcessEvidence = - | ({ verdict: 'live'; processName: string | null } & ForegroundEvidenceObservation) + | ({ + verdict: 'live' + processName: string | null + /** True only when the host observed every process group attached to this PTY's terminal to be + * the shell's own, with none of them stopped — i.e. nothing is running in the pane, in the + * foreground OR the background, and nothing sits suspended. + * + * Deliberately not `tpgid === pgid`: a job the user backgrounded with `&` and a job the user + * suspended with Ctrl-Z both hand the terminal back to the shell, so a foreground-only + * predicate reads them as idle. This one is measured against the same set of process groups + * a forced stop would SIGKILL. + * + * False means something IS running, named or not. Absent from a host that predates the + * field, which is neither: a reader deciding whether the pane is idle must require `true` + * and defer on anything else. */ + shellOwnsEveryTtyProcessGroup?: boolean + } & ForegroundEvidenceObservation) | ({ verdict: 'unverifiable'; reason: string } & ForegroundEvidenceObservation) export function isForegroundProcessEvidence(value: unknown): value is ForegroundProcessEvidence { @@ -30,6 +53,12 @@ export function isForegroundProcessEvidence(value: unknown): value is Foreground return false } if (input.verdict === 'live') { + if ( + input.shellOwnsEveryTtyProcessGroup !== undefined && + typeof input.shellOwnsEveryTtyProcessGroup !== 'boolean' + ) { + return false + } return input.processName === null || typeof input.processName === 'string' } return ( diff --git a/src/shared/process-table-snapshot.ts b/src/shared/process-table-snapshot.ts index 17243d17b96..599489f0c99 100644 --- a/src/shared/process-table-snapshot.ts +++ b/src/shared/process-table-snapshot.ts @@ -323,6 +323,11 @@ const processTableReader = createProcessTableSnapshotReader now: () => Date.now() }) +/** How much older than its own await a snapshot from the shared reader may be. The reader serves a + * capture from its TTL cache, so a caller that needs to state the observation's age must assume + * this whole window rather than the instant its await settled. */ +export const PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS = DEFAULT_SNAPSHOT_TTL_MS + /** * Run (or reuse a recent) `ps -axo` process-table scan and return * its parsed rows. Per-process singleton: the relay and local main processes diff --git a/src/shared/pty-consumer-session.ts b/src/shared/pty-consumer-session.ts index 6b040be47cf..7c5b29aacfd 100644 --- a/src/shared/pty-consumer-session.ts +++ b/src/shared/pty-consumer-session.ts @@ -153,6 +153,15 @@ export class PtyConsumerSession { return client?.state === 'active' ? client.grant : null } + /** The authenticated client identity behind an active connection, or null. + * + * Why the host reads it here instead of taking a spawn parameter: this is what makes a later + * "this PTY belongs to you" attestation evidence rather than an echo of what a caller claimed. */ + activeClientInstanceId(connectionId: string): string | null { + const client = this.clients.get(connectionId) + return client?.state === 'active' ? client.clientInstanceId : null + } + private admissionFor( client: ClientRecord, displacedOwner?: Readonly diff --git a/src/shared/ssh-relay-pty-ownership-proof.test.ts b/src/shared/ssh-relay-pty-ownership-proof.test.ts new file mode 100644 index 00000000000..1abfc0ef7f5 --- /dev/null +++ b/src/shared/ssh-relay-pty-ownership-proof.test.ts @@ -0,0 +1,265 @@ +// #9819. Every case here is the same question asked from a different angle: can this client PROVE +// the host is holding a process nobody can reach? A "no" has to mean "leave it running". +import { describe, expect, it } from 'vitest' +import type { ForegroundProcessEvidence } from './foreground-process-evidence' +import { + planRelayPtySweep, + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + RELAY_PTY_SWEEP_MAX_PER_PASS, + RELAY_PTY_SWEEP_MIN_AGE_MS, + type RelayPtyOwnershipEvidence, + type RelayPtySweepContext +} from './ssh-relay-pty-ownership-proof' + +const OURS = 'client-instance-ours' + +const OBSERVATION = { authorityGeneration: 'gen-1', observationEpoch: 1, capturedAgeMs: 0 } + +/** The host looked at the pane and saw its own shell owning the terminal: nothing is running. */ +function idleShell(): ForegroundProcessEvidence { + return { ...OBSERVATION, verdict: 'live', processName: null, shellOwnsEveryTtyProcessGroup: true } +} + +function orphan(overrides: Partial = {}): RelayPtyOwnershipEvidence { + return { + ptyId: 'pty-1', + incarnationId: 'inc-1', + ownerClientInstanceId: OURS, + hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS * 2, + paneBound: true, + foregroundProcessEvidence: idleShell(), + ...overrides + } +} + +function context(overrides: Partial = {}): RelayPtySweepContext { + return { + clientInstanceId: OURS, + isSessionOwner: true, + routedPtyIds: new Set(), + expiredLeasePtyIds: new Set(), + minimumHostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS, + evidenceAgeSinceListingMs: 0, + maximumEvidenceAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + ...overrides + } +} + +function reasonFor(plan: ReturnType, ptyId: string): string | undefined { + return plan.skipped.find((entry) => entry.ptyId === ptyId)?.reason +} + +describe('planRelayPtySweep', () => { + it('sweeps a pane PTY this host attests we created and we have lost every route to', () => { + const plan = planRelayPtySweep([orphan()], context()) + + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) + + it('never sweeps a PTY this client still routes to', () => { + const plan = planRelayPtySweep([orphan()], context({ routedPtyIds: new Set(['pty-1']) })) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('this client still has a route to it') + }) + + it('never sweeps a PTY whose lease this client expired without ordering a stop', () => { + // `expired` is what supersedeSiblingLeasesForPane, a reattach that failed on the transport, and + // a pane whose surface left the layout all write, and every one of them deliberately leaves the + // remote process running. Losing our handle is `unverifiable`; it is not abandonment. + const plan = planRelayPtySweep([orphan()], context({ expiredLeasePtyIds: new Set(['pty-1']) })) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('this client expired its lease without ordering a stop') + }) + + it('never sweeps a pane the host observes running a named foreground process', () => { + // The hand-launched agent: the user typed `claude` in a pane, so Orca registered no agent + // session and agentSessionOwners is empty. Only the host's own observation can see it. + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: 'claude', + shellOwnsEveryTtyProcessGroup: false + } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host observes a named foreground process') + }) + + it('never sweeps a pane whose foreground group is not the shell, even unnamed', () => { + // A build, a test run, an editor: nothing recognizes it, but the host can still see that the + // terminal's foreground process group is not the shell's own. + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: null, + shellOwnsEveryTtyProcessGroup: false + } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host does not attest an idle shell') + }) + + it('never sweeps when the host could not observe the pane at all', () => { + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'unverifiable', + reason: 'table_unreadable' + } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host could not observe the pane foreground process') + }) + + it('never sweeps a PTY the host attributes to another client instance', () => { + // The case that makes local absence useless as evidence: a second machine on the same build + // connects to the same relay, and its live agents are missing from our store exactly like an + // orphan is. + const plan = planRelayPtySweep( + [orphan({ ownerClientInstanceId: 'client-instance-theirs' })], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host attests another client created it') + }) + + it('never sweeps a PTY younger than the floor', () => { + const plan = planRelayPtySweep( + [orphan({ hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS - 1 })], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('younger than the sweep floor') + }) + + it('never sweeps a bare host shell', () => { + const plan = planRelayPtySweep([orphan({ paneBound: false })], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('not a pane-bound PTY') + }) + + it('never sweeps a PTY whose agent session the host still advertises as adoptable', () => { + const plan = planRelayPtySweep( + [orphan({ agentSessionOwners: [{ ptyId: 'pty-1' }] })], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host still advertises an adoptable agent session') + }) + + it('never sweeps without the negotiated session-owner grant', () => { + const plan = planRelayPtySweep([orphan()], context({ isSessionOwner: false })) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('this client does not hold the relay session-owner grant') + }) + + it('refuses a pass larger than the per-pass ceiling instead of truncating it', () => { + const entries = Array.from({ length: RELAY_PTY_SWEEP_MAX_PER_PASS + 1 }, (_, index) => + orphan({ ptyId: `pty-${index}`, incarnationId: `inc-${index}` }) + ) + + const plan = planRelayPtySweep(entries, context()) + + expect(plan.sweep).toEqual([]) + expect(plan.skipped).toHaveLength(entries.length) + }) + + describe('against a host that predates the attestation', () => { + // Mixed versions: every new field is optional, and an older host publishes none of them. The + // sweep has to read each absence as "unknown", never as a permissive default. + it('skips an entry with no owner attestation', () => { + const { ownerClientInstanceId: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host attested no owning client') + }) + + it('skips an entry with no published age', () => { + const { hostAgeMs: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host published no age') + }) + + it('skips an entry with no paneBound field', () => { + const { paneBound: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('not a pane-bound PTY') + }) + + it('skips an entry with no foreground observation', () => { + const { foregroundProcessEvidence: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host published no foreground-process observation') + }) + + it('skips an entry from a host that observes the pane but cannot say the shell is idle', () => { + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { ...OBSERVATION, verdict: 'live', processName: null } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host does not attest an idle shell') + }) + + it('skips an entry with no incarnation, so no stop is ever unfenced', () => { + const { incarnationId: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host published no PTY incarnation') + }) + + it('sweeps nothing at all when the whole listing predates the fields', () => { + const legacy = [ + { ptyId: 'pty-1', incarnationId: 'inc-1' }, + { ptyId: 'pty-2', incarnationId: 'inc-2' } + ] + + expect(planRelayPtySweep(legacy, context()).sweep).toEqual([]) + }) + }) +}) diff --git a/src/shared/ssh-relay-pty-ownership-proof.ts b/src/shared/ssh-relay-pty-ownership-proof.ts new file mode 100644 index 00000000000..ed87be13610 --- /dev/null +++ b/src/shared/ssh-relay-pty-ownership-proof.ts @@ -0,0 +1,226 @@ +import type { ForegroundProcessEvidence } from './foreground-process-evidence' + +/** Which relay PTYs a client may prove it orphaned, and therefore may stop (#9819). + * + * A relay PTY is a child of the detached relay daemon. Stopping one destroys a running process — + * often a running agent — on the user's remote machine, and the relay's 50-slot cap is a far + * cheaper failure than that. So every rule below is written to answer "can this client PROVE + * nobody owns this?" and to answer "no" whenever it cannot. + * + * The rule #9819 proposed — "pane-bound and the app no longer owns or leases it" — is not that + * proof. Absence from a client-side set is `unverifiable` by construction + * (`docs/reference/ssh-execution-boundary.md`): a second machine running the same Orca build + * connects to the SAME relay and displaces the session owner, and its PTYs are missing from THIS + * client's store for exactly the same reason a genuine orphan is. Sweeping on local absence alone + * would let one laptop reap another laptop's live agents. + * + * What replaces it: the host itself records which authenticated consumer identity asked it to + * create each PTY, and publishes that back. A PTY is sweepable only when the OWNING HOST names + * this client as its creator and this client's own durable state has no route to it. Both halves + * are required; either alone is a guess. + */ + +/** One `pty.listProcesses` entry, as far as this decision is concerned. Every field a host may + * omit is optional here, because a host predating it publishes nothing rather than a default. */ +export type RelayPtyOwnershipEvidence = { + /** Relay-scoped PTY id. */ + ptyId: string + incarnationId?: string + ownerClientInstanceId?: string + hostAgeMs?: number + paneBound?: boolean + /** Non-empty when the host still advertises an adoptable agent session on this PTY. */ + agentSessionOwners?: readonly unknown[] + /** What the OWNING host saw in the pane on the same listing. This answers a different question + * from `agentSessionOwners`: that one asks whether Orca REGISTERED an agent session here, this + * one asks whether anything at all is running. A `claude` the user typed by hand registers + * nothing, so only this can see it. */ + foregroundProcessEvidence?: ForegroundProcessEvidence +} + +export type RelayPtySweepContext = { + /** This client's persisted consumer identity for the target. */ + clientInstanceId: string + /** Whether the relay granted THIS connection the `session-owner` role. A subscriber, or a client + * that fell back to the unnegotiated legacy path, never sweeps. */ + isSessionOwner: boolean + /** Every relay PTY id this client still has any route to: a live provider PTY, a lease it has + * not tombstoned, an id it just reattached, or a stop it has recorded and not yet delivered. */ + routedPtyIds: ReadonlySet + /** Relay PTY ids this client holds an `expired` lease for. + * + * Separate from {@link routedPtyIds} because it is a different fact with the same verdict: an + * expired lease records that THIS CLIENT lost its handle — a pane re-leased under a new relay + * id, a pane surface that is no longer in the layout, a retired reattach. The layout case is the + * one that matters most, because it is reached only AFTER `pty.attach` succeeded: the process is + * not merely unproven, it is known to be alive. Every one of those writers deliberately declines + * to stop it, and client-side absence is `unverifiable` by construction + * (`docs/reference/ssh-execution-boundary.md`). So an expired lease is the record of a process + * left running on purpose, never a licence to kill it. */ + expiredLeasePtyIds: ReadonlySet + /** Host-measured age a PTY must exceed. Guards a spawn that is in flight from another window of + * this same client and has not written its lease yet. */ + minimumHostAgeMs: number + /** How long ago, on THIS client's clock, the listing that carried the evidence arrived. Added to + * each entry's host-stamped `capturedAgeMs` so {@link maximumEvidenceAgeMs} bounds staleness at + * the moment of the decision rather than at the moment of serialization. The transit itself is + * unmeasured — the two clocks are not synchronized — but it is bounded by the listing's own RPC + * deadline, and both halves that ARE measurable are counted. */ + evidenceAgeSinceListingMs: number + /** Oldest foreground observation that may authorize a stop. Stale evidence degrades to "do not + * sweep", never to "sweep". */ + maximumEvidenceAgeMs: number +} + +export type RelayPtySweepTarget = { ptyId: string; incarnationId: string } + +export type RelayPtySweepSkip = { ptyId: string; reason: string } + +export type RelayPtySweepPlan = { + sweep: RelayPtySweepTarget[] + skipped: RelayPtySweepSkip[] +} + +/** Deliberately longer than any single connect round trip. A PTY younger than this is never worth + * the risk: the leak it represents costs one slot for 30 more seconds, and reaping a shell that a + * concurrent spawn is still recording costs the user a terminal. */ +export const RELAY_PTY_SWEEP_MIN_AGE_MS = 30_000 + +/** Bounds one pass. A relay is capped at 50 PTYs, so a pass that wants to stop more than this is + * not reclaiming a leak — it is a disagreement about ownership, and stopping is the wrong move. */ +export const RELAY_PTY_SWEEP_MAX_PER_PASS = 8 + +/** The oldest foreground observation this sweep will treat as authorization to SIGKILL. + * + * Sized to the pass budget rather than to the 30s spawn floor: those answer different questions. + * The floor guards a concurrent spawn this client has not recorded yet; this one guards the pane + * the user started working in AFTER the host looked. An observation older than the whole pass it + * is meant to authorize cannot have been taken for this pass, so it is not evidence about now. + * + * It does not remove the race — nothing can, the host cannot re-check between the answer and the + * signal — it bounds it. The display consumer of the same measurement deliberately keeps NO age + * budget: a stale pane title costs a redraw and self-corrects on the next poll, so one truthful + * number carries two explicit budgets rather than one implicit one. */ +export const RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS = 5_000 + +/** The host's own answer to "is anything running in this pane?". Only a positive "no" clears the + * sweep; every other shape — an older host, an unreadable process table, an observation too old to + * describe now, a named foreground process, any other process group on the pane's terminal — is a + * reason to leave the process alone. */ +function foregroundSkipReason( + evidence: ForegroundProcessEvidence | undefined, + context: RelayPtySweepContext +): string | null { + if (evidence === undefined) { + // A host that never published it, or a Windows host where it is not collected. Absence of the + // observation is not the observation of absence. + return 'host published no foreground-process observation' + } + // Before anything is read out of it: an observation is only a claim about the instant it was + // taken. Age is checked on both verdicts because a stale `unverifiable` is no better. + if (evidence.capturedAgeMs + context.evidenceAgeSinceListingMs > context.maximumEvidenceAgeMs) { + return 'host foreground observation is too old to authorize a stop' + } + if (evidence.verdict !== 'live') { + return 'host could not observe the pane foreground process' + } + if (evidence.processName !== null) { + // The host named something running in the pane. It registered no agent session, which is + // exactly the hand-launched `claude`/`codex` case agentSessionOwners cannot see. + return 'host observes a named foreground process' + } + if (evidence.shellOwnsEveryTtyProcessGroup !== true) { + // Something other than the shell's own process group is attached to the pane's terminal — a + // foreground command, a job backgrounded with `&`, a Ctrl-Z'd editor — or this host predates + // the field. The stop would SIGKILL that group, so none of those is a pane to reclaim. + return 'host does not attest an idle shell' + } + return null +} + +function skipReason( + entry: RelayPtyOwnershipEvidence, + context: RelayPtySweepContext +): string | null { + if (typeof entry.incarnationId !== 'string' || entry.incarnationId.length === 0) { + // Without the host's own incarnation there is no fence, and an unfenced stop aimed at a relay + // id can hit whatever holds that id by the time it lands. + return 'host published no PTY incarnation' + } + if (typeof entry.ownerClientInstanceId !== 'string' || entry.ownerClientInstanceId.length === 0) { + return 'host attested no owning client' + } + if (entry.ownerClientInstanceId !== context.clientInstanceId) { + return 'host attests another client created it' + } + if (entry.paneBound !== true) { + // Covers both a bare host shell (a remote CLI terminal nobody's pane owns) and a host that + // never published the field. Neither is a pane this client lost. + return 'not a pane-bound PTY' + } + if (entry.agentSessionOwners !== undefined && entry.agentSessionOwners.length > 0) { + // The host still advertises this session as adoptable, so a later spawn can reclaim the running + // agent. Reaping it converts a recoverable session into a destroyed one. + return 'host still advertises an adoptable agent session' + } + const foregroundSkip = foregroundSkipReason(entry.foregroundProcessEvidence, context) + if (foregroundSkip !== null) { + return foregroundSkip + } + if (typeof entry.hostAgeMs !== 'number' || !Number.isFinite(entry.hostAgeMs)) { + return 'host published no age' + } + if (entry.hostAgeMs < context.minimumHostAgeMs) { + return 'younger than the sweep floor' + } + if (context.routedPtyIds.has(entry.ptyId)) { + return 'this client still has a route to it' + } + if (context.expiredLeasePtyIds.has(entry.ptyId)) { + return 'this client expired its lease without ordering a stop' + } + return null +} + +/** Plans one sweep pass. Pure: every input is evidence the caller already gathered, so the rule can + * be tested without a relay, and the irreversible call sits with the caller. */ +export function planRelayPtySweep( + entries: readonly RelayPtyOwnershipEvidence[], + context: RelayPtySweepContext +): RelayPtySweepPlan { + if (!context.isSessionOwner || !context.clientInstanceId) { + return { + sweep: [], + skipped: entries.map((entry) => ({ + ptyId: entry.ptyId, + reason: 'this client does not hold the relay session-owner grant' + })) + } + } + const sweep: RelayPtySweepTarget[] = [] + const skipped: RelayPtySweepSkip[] = [] + for (const entry of entries) { + const reason = skipReason(entry, context) + if (reason !== null) { + skipped.push({ ptyId: entry.ptyId, reason }) + } else { + sweep.push({ ptyId: entry.ptyId, incarnationId: entry.incarnationId as string }) + } + } + if (sweep.length > RELAY_PTY_SWEEP_MAX_PER_PASS) { + // Why refuse rather than truncate: at this size the disagreement is about ownership, not about + // a handful of leaked slots, and a truncated pass would work through the same list one connect + // at a time and destroy it all anyway. + return { + sweep: [], + skipped: [ + ...skipped, + ...sweep.map((target) => ({ + ptyId: target.ptyId, + reason: `refusing a ${sweep.length}-PTY sweep; over the ${RELAY_PTY_SWEEP_MAX_PER_PASS} per-pass ceiling` + })) + ] + } + } + return { sweep, skipped } +} From 7c6c8ef85e143ae4062475f13b418cc6ba45e42e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:17 -0700 Subject: [PATCH 03/77] fix(ssh): stop a late SFTP stream error crashing main, and keep the relay socket inside sun_path (#17862) * fix(ssh): stop SFTP stream errors crashing main and bound the relay socket path inside the protocol parser. Every transfer removed its listener on settle, so a STATUS reply that arrived late - the normal case behind a jump host that chroots its SFTP subsystem - threw synchronously out of Socket.emit('data') and killed the main process. Keep one durable listener per stream, and report a sandboxed SFTP namespace with an actionable message instead of a bare 'file does not exist'. 104 macOS) and bind failed with a bare 'listen EINVAL'. Fall back to a per-uid base whose length does not depend on $HOME, keeping the hashed socket name intact. * fix(ssh): validate the short socket dir before mutating it * fix(ssh): keep the SFTP session guarded, scope the relocated socket, narrow the chroot verdict Three review findings. The CLI-launcher install ran writeStringViaSftp in a loop over a bare conn.sftp(). That helper removes its own session 'error' listener at each settle, so between files and after the last one the emitter carried none -- and ssh2 raises a late STATUS reply synchronously out of Protocol.parse, which is the uncaught exception that kills main (#15479). The inline loop it replaced leaked one listener per file and covered this by accident. Extract writeStringsViaSftp, which owns the session latch, and share that latch with runSftpFallbackTransfer. SSH_FX_PERMISSION_DENIED is a mode/ownership refusal on a path the subsystem can see, not evidence of a chroot; sftp-namespace-resolution already treats only NO_SUCH_FILE as conclusive. Narrow the predicate to code 2 so a read-only home stops being reported as a bastion misconfiguration. The relocated socket had no version dimension. relaySocketNameForInstanceId hashes the target, not the build, and under $HOME the enclosing relay- dir supplied the rest -- so the short form made the path stable across updates. The next build would bind the path the previous relay still holds, the handshake would mismatch, and a relay holding live work would raise RelayEndpointHeldError with no way through. Add a hashed version segment under the short base, mirroring the relay-*/ shape so one pattern serves both, and teach the superseded sweep and force-stop about that base. The relocated tree now also gets reclaimed: nothing else walks it. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- ...ocket-path-limit-shell.integration.test.ts | 103 ++++++++ src/main/ssh/relay-socket-path-limit.test.ts | 250 ++++++++++++++++++ src/main/ssh/relay-socket-path-limit.ts | 128 +++++++++ src/main/ssh/sftp-stream-late-error.test.ts | 172 ++++++++++++ src/main/ssh/sftp-stream-late-error.ts | 108 ++++++++ src/main/ssh/sftp-upload.test.ts | 8 +- src/main/ssh/sftp-upload.ts | 37 +++ src/main/ssh/ssh-relay-deploy.ts | 66 ++++- src/main/ssh/ssh-relay-install-transfers.ts | 54 +++- src/main/ssh/ssh-relay-reset.ts | 7 +- src/main/ssh/ssh-relay-session.ts | 16 +- .../ssh/ssh-relay-superseded-endpoints.ts | 22 +- 12 files changed, 944 insertions(+), 27 deletions(-) create mode 100644 src/main/ssh/relay-socket-path-limit-shell.integration.test.ts create mode 100644 src/main/ssh/relay-socket-path-limit.test.ts create mode 100644 src/main/ssh/relay-socket-path-limit.ts create mode 100644 src/main/ssh/sftp-stream-late-error.test.ts create mode 100644 src/main/ssh/sftp-stream-late-error.ts diff --git a/src/main/ssh/relay-socket-path-limit-shell.integration.test.ts b/src/main/ssh/relay-socket-path-limit-shell.integration.test.ts new file mode 100644 index 00000000000..ce08556c3d7 --- /dev/null +++ b/src/main/ssh/relay-socket-path-limit-shell.integration.test.ts @@ -0,0 +1,103 @@ +import { execFile } from 'node:child_process' +import { chmod, mkdtemp, mkdir, rm, symlink, stat } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { promisify } from 'node:util' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + resolveShortRelaySocketDirCommand, + shortRelayVersionSegment +} from './relay-socket-path-limit' + +const run = promisify(execFile) + +// The generated script runs on the remote host's /bin/sh, so assert against a real shell rather +// than a string match: the hazard here is an ordering bug that only a filesystem can observe. +describe('short relay socket dir guard, against a real shell', () => { + let root: string + + const VERSION_SEGMENT = shortRelayVersionSegment('relay-0.1.0+test') + + // Retarget the generated script at a sandbox instead of the real /tmp path. + function scriptFor(dir: string): string { + return resolveShortRelaySocketDirCommand(VERSION_SEGMENT).replace( + /^dir=.*$/m, + `dir=${JSON.stringify(dir)}` + ) + } + + async function attempt(dir: string): Promise<{ ok: boolean; stdout: string }> { + try { + const { stdout } = await run('/bin/sh', ['-c', scriptFor(dir)]) + return { ok: true, stdout } + } catch { + return { ok: false, stdout: '' } + } + } + + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-relay-dir-guard-')) + }) + + afterEach(async () => { + await rm(root, { recursive: true, force: true }) + }) + + it('creates the directory and its version segment when neither exists', async () => { + const dir = join(root, 'fresh') + const attempted = await attempt(dir) + expect(attempted.ok).toBe(true) + expect((await stat(dir)).mode & 0o777).toBe(0o700) + // The segment is what keeps a later build off the path this one binds. + expect((await stat(join(dir, VERSION_SEGMENT))).mode & 0o777).toBe(0o700) + expect(attempted.stdout.trim().endsWith(`${dir}/${VERSION_SEGMENT}`)).toBe(true) + }) + + it('refuses a planted symlink in the version segment without following it', async () => { + const dir = join(root, 'mine') + const victim = join(root, 'segment-victim') + await mkdir(dir) + await chmod(dir, 0o700) + await mkdir(victim) + await chmod(victim, 0o755) + await symlink(victim, join(dir, VERSION_SEGMENT)) + + const before = (await stat(victim)).mode & 0o777 + expect((await attempt(dir)).ok).toBe(false) + expect((await stat(victim)).mode & 0o777).toBe(before) + }) + + it('adopts a directory we already own at 0700, so reconnects keep working', async () => { + // Regression guard: `ls` decorates the mode with @ (xattrs), + (ACL) or . (SELinux), and an + // exact match refused a directory we own — which would have broken every reconnect. + const dir = join(root, 'mine') + await mkdir(dir) + await chmod(dir, 0o700) + expect((await attempt(dir)).ok).toBe(true) + expect((await attempt(dir)).ok).toBe(true) + }) + + it('refuses a planted symlink without changing what it points at', async () => { + const victim = join(root, 'victim') + const link = join(root, 'link') + await mkdir(victim) + await chmod(victim, 0o755) + await symlink(victim, link) + + const before = (await stat(victim)).mode & 0o777 + expect((await attempt(link)).ok).toBe(false) + // The point of the ordering: an unconditional chmod would have followed the link and + // rewritten the victim's mode before the owner check ever ran. + expect((await stat(victim)).mode & 0o777).toBe(before) + }) + + it('refuses an existing directory that is not 0700', async () => { + const dir = join(root, 'loose') + await mkdir(dir) + // Explicit chmod: mkdir's mode is masked by the process umask, so the fixture would not + // actually be world-writable and the test would not be testing what it claims. + await chmod(dir, 0o777) + expect((await attempt(dir)).ok).toBe(false) + expect((await stat(dir)).mode & 0o777).toBe(0o777) + }) +}) diff --git a/src/main/ssh/relay-socket-path-limit.test.ts b/src/main/ssh/relay-socket-path-limit.test.ts new file mode 100644 index 00000000000..7334ff9d914 --- /dev/null +++ b/src/main/ssh/relay-socket-path-limit.test.ts @@ -0,0 +1,250 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + app: { getAppPath: () => '/mock/app' } +})) + +vi.mock('fs', () => ({ + existsSync: vi.fn().mockReturnValue(true), + readFileSync: vi.fn().mockReturnValue('0.1.0+abcdef012345') +})) + +vi.mock('./relay-protocol', () => ({ + RELAY_VERSION: '0.1.0', + RELAY_REMOTE_DIR: '.orca-remote', + parseUnameToRelayPlatform: vi.fn(() => 'linux-x64'), + RELAY_SENTINEL: 'ORCA-RELAY v0.1.0 READY\n', + RELAY_SENTINEL_TIMEOUT_MS: 10_000 +})) + +vi.mock('./ssh-relay-deploy-helpers', () => ({ + uploadDirectory: vi.fn().mockResolvedValue(undefined), + waitForSentinel: vi.fn().mockResolvedValue({ + write: vi.fn(), + onData: vi.fn(), + onClose: vi.fn() + }), + isUnconfirmedSshCommandTermination: () => false, + execCommand: vi.fn().mockResolvedValue('') +})) + +vi.mock('./ssh-remote-node-resolution', () => ({ + resolveRemoteNodePath: vi.fn().mockResolvedValue('/usr/bin/node') +})) + +vi.mock('./ssh-relay-endpoint-credential', () => ({ + writeRelayEndpointCredential: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-relay-versioned-install', () => ({ + readLocalFullVersion: vi.fn().mockReturnValue('0.1.0+8d4e15ad63eb'), + computeRemoteRelayDir: (home: string, v: string) => `${home}/.orca-remote/relay-${v}`, + isRelayAlreadyInstalled: vi.fn().mockResolvedValue(true), + finalizeInstall: vi.fn().mockResolvedValue(undefined), + abandonInstall: vi.fn().mockResolvedValue(undefined), + gcOldRelayVersions: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-relay-install-lock', () => ({ + acquireInstallLock: vi.fn().mockResolvedValue(undefined), + RELAY_INSTALL_LOCK_NAME: '.install-lock' +})) + +vi.mock('./ssh-relay-repair-lock', () => ({ + tryAcquireRelayRepairLock: vi.fn().mockResolvedValue('acquired') +})) + +vi.mock('./ssh-connection-utils', () => ({ + shellEscape: (s: string) => `'${s}'`, + createSshOperationAbortError: () => + Object.assign(new Error('SSH operation was cancelled'), { name: 'AbortError' }) +})) + +import { deployAndLaunchRelay } from './ssh-relay-deploy' +import { execCommand } from './ssh-relay-deploy-helpers' +import { forceStopRelayForTarget } from './ssh-relay-reset' +import { relaySocketNameForInstanceId } from './ssh-relay-instance-id' +import { + parseShortRelaySocketDir, + remoteSocketPathFitsLimit, + remoteUnixSocketPathByteLimit, + shortRelayVersionSegment, + SHORT_RELAY_SOCKET_DIR_PREFIX +} from './relay-socket-path-limit' +import { supersededRelayEndpointListCommand } from './ssh-relay-superseded-endpoints' +import { getRemoteHostPlatform } from './ssh-remote-platform' +import type { SshConnection } from './ssh-connection' + +const LINUX = getRemoteHostPlatform('linux-x64') +const DARWIN = getRemoteHostPlatform('darwin-arm64') +const WINDOWS = getRemoteHostPlatform('win32-x64') + +// The reporter's host: a managed-hosting container whose $HOME is 45 bytes (#10726). +const LONG_HOME = '/var/www/611f7cf9-f715-49e6-91d9-0ffac1d7c4c0' + +/** Matches the version this suite's mocked build reports. */ +const RELAY_VERSION_DIR_NAME = 'relay-0.1.0+8d4e15ad63eb' + +function makeMockConnection(): SshConnection { + return { + canRunConcurrentExecCommands: vi.fn().mockReturnValue(true), + exec: vi.fn().mockResolvedValue({ + on: vi.fn(), + stderr: { on: vi.fn() }, + stdin: {}, + stdout: { on: vi.fn() }, + close: vi.fn() + }), + writeFile: vi.fn().mockResolvedValue(undefined), + sftp: vi.fn().mockResolvedValue({ + mkdir: vi.fn((_p: string, cb: (err: Error | null) => void) => cb(null)), + createWriteStream: vi.fn().mockReturnValue({ + on: vi.fn((event: string, cb: () => void) => { + if (event === 'close') { + setTimeout(cb, 0) + } + }), + end: vi.fn() + }), + end: vi.fn() + }) + } as unknown as SshConnection +} + +function launchedSockPath(conn: SshConnection): string { + const launch = vi + .mocked(conn.exec) + .mock.calls.map(([command]) => command as string) + .find((command) => command.includes('--detached')) + return /--sock-path\s+'([^']+)'/.exec(launch ?? '')?.[1] ?? '' +} + +describe('remote unix socket path limit', () => { + it('uses the per-OS sun_path budget and ignores Windows named pipes', () => { + expect(remoteUnixSocketPathByteLimit(LINUX)).toBe(107) + expect(remoteUnixSocketPathByteLimit(DARWIN)).toBe(103) + expect(remoteUnixSocketPathByteLimit(WINDOWS)).toBeNull() + expect(remoteSocketPathFitsLimit(WINDOWS, `\\\\.\\pipe\\orca-relay-${'a'.repeat(400)}`)).toBe( + true + ) + }) + + it('measures bytes, not characters', () => { + // 1 + 52 two-byte characters = 105 bytes: fits Linux (107), not macOS (103). + const path = `/${'é'.repeat(52)}` + expect(path.length).toBe(53) + expect(remoteSocketPathFitsLimit(LINUX, path)).toBe(true) + expect(remoteSocketPathFitsLimit(DARWIN, path)).toBe(false) + }) + + it('accepts only the marker line as the short directory', () => { + const segment = shortRelayVersionSegment(RELAY_VERSION_DIR_NAME) + expect( + parseShortRelaySocketDir( + `Welcome to Ubuntu\nORCA-RELAY-SHORT-SOCKET-DIR /tmp/.orca-relay-1000/${segment}\n`, + segment + ) + ).toBe(`/tmp/.orca-relay-1000/${segment}`) + expect(parseShortRelaySocketDir('mkdir: permission denied\n', segment)).toBeNull() + expect( + parseShortRelaySocketDir(`ORCA-RELAY-SHORT-SOCKET-DIR /etc/${segment}\n`, segment) + ).toBeNull() + // A directory belonging to another build must not be adopted as this build's. + expect( + parseShortRelaySocketDir( + `ORCA-RELAY-SHORT-SOCKET-DIR /tmp/.orca-relay-1000/${shortRelayVersionSegment('relay-9.9.9+other')}\n`, + segment + ) + ).toBeNull() + }) +}) + +describe('relay launch with a long remote $HOME', () => { + beforeEach(() => { + vi.clearAllMocks() + }) + + it('keeps the launched socket path inside the remote sun_path limit', async () => { + const conn = makeMockConnection() + vi.mocked(execCommand) + .mockReset() + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ Linux x86_64') + .mockResolvedValueOnce(LONG_HOME) + .mockResolvedValueOnce('ORCA-NATIVE-DEPS-OK') + .mockResolvedValueOnce('') // launch namespace marker + .mockResolvedValueOnce( + `ORCA-RELAY-SHORT-SOCKET-DIR ${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/${shortRelayVersionSegment(RELAY_VERSION_DIR_NAME)}` + ) + .mockResolvedValueOnce('DEAD') + .mockResolvedValueOnce('READY') + .mockResolvedValue('') + + // A per-target relay instance id is what pushes the default path past the limit: + // 45-byte $HOME + `/.orca-remote/relay-0.1.0+8d4e15ad63eb` + `/relay-.sock` = 110 bytes. + const result = await deployAndLaunchRelay(conn, undefined, undefined, 'ssh-target-1') + + const sockPath = launchedSockPath(conn) + expect(sockPath).not.toBe('') + expect(Buffer.byteLength(sockPath, 'utf8')).toBeLessThanOrEqual( + remoteUnixSocketPathByteLimit(LINUX) as number + ) + expect(sockPath.startsWith(`${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/`)).toBe(true) + expect(result.sockPath).toBe(sockPath) + // The hashed socket name survives intact, so two targets cannot collide -- and the + // build's version segment sits above it, so the next Orca release binds a path of + // its own instead of the one this relay is still holding. + expect(sockPath).toBe( + `${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/${shortRelayVersionSegment(RELAY_VERSION_DIR_NAME)}/${relaySocketNameForInstanceId('ssh-target-1')}` + ) + expect(shortRelayVersionSegment('relay-0.1.0+next')).not.toBe( + shortRelayVersionSegment(RELAY_VERSION_DIR_NAME) + ) + }) + + it('sweeps superseded relays under the short base too, but never the live one', () => { + const currentShortSocketDir = `${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/${shortRelayVersionSegment(RELAY_VERSION_DIR_NAME)}` + const script = supersededRelayEndpointListCommand({ + remoteHome: LONG_HOME, + currentRelayDir: `${LONG_HOME}/.orca-remote/${RELAY_VERSION_DIR_NAME}`, + sockName: relaySocketNameForInstanceId('ssh-target-1'), + currentShortSocketDir + }) + + // A relocated orphan lives outside $HOME, so the sweep that exists to make orphans + // visible has to look at the short base as well. + expect(script).toContain(`short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`) + expect(script).toContain('"$short_base"/relay-*/"$sock_name"') + expect(script).toContain(`short_current='${currentShortSocketDir}'`) + expect(script).toContain('[ -n "$short_current" ] && [ "$dir" = "$short_current" ] && continue') + }) + + it('leaves the socket in the versioned relay dir when it already fits', async () => { + const conn = makeMockConnection() + vi.mocked(execCommand) + .mockReset() + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ Linux x86_64') + .mockResolvedValueOnce('/home/user') + .mockResolvedValueOnce('ORCA-NATIVE-DEPS-OK') + .mockResolvedValueOnce('') + .mockResolvedValueOnce('DEAD') + .mockResolvedValueOnce('READY') + .mockResolvedValue('') + + await deployAndLaunchRelay(conn) + + expect(launchedSockPath(conn)).toBe( + '/home/user/.orca-remote/relay-0.1.0+8d4e15ad63eb/relay.sock' + ) + }) + + it('force-stop also looks for the socket under the short base', async () => { + const conn = makeMockConnection() + vi.mocked(execCommand).mockReset().mockResolvedValue('') + + await forceStopRelayForTarget(conn, 'ssh-1') + + const script = vi.mocked(execCommand).mock.calls[0]?.[1] as string + expect(script).toContain(`short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`) + expect(script).toContain('"$short_base"/relay-*/"$sock_name"') + }) +}) diff --git a/src/main/ssh/relay-socket-path-limit.ts b/src/main/ssh/relay-socket-path-limit.ts new file mode 100644 index 00000000000..417366e2bf6 --- /dev/null +++ b/src/main/ssh/relay-socket-path-limit.ts @@ -0,0 +1,128 @@ +/** + * Keeps the remote relay's Unix socket path inside `sockaddr_un.sun_path`. + * + * The default endpoint is `$HOME/.orca-remote/relay-/relay-.sock`, + * whose fixed suffix already costs ~66 bytes. A managed-hosting `$HOME` such as + * `/var/www/` pushes the whole path past the kernel cap and libuv reports only + * `listen EINVAL`, so the relay never starts (#10726). When that happens the socket + * moves to a fixed-length base whose length no longer depends on `$HOME`. + * + * Windows relays bind named pipes (`\\.\pipe\...`), which have no `sun_path` limit. + */ +import { createHash } from 'node:crypto' +import { isWindowsRemoteHost, type RemoteHostPlatform } from './ssh-remote-platform' + +/** + * `sizeof(sun_path)` per remote OS, including the terminating NUL: 108 on Linux, + * 104 on macOS/BSD. Compared against byte length, not character count — a non-ASCII + * `$HOME` costs more bytes than characters. + */ +const SUN_PATH_SIZE: Record<'linux' | 'darwin', number> = { linux: 108, darwin: 104 } + +export function remoteUnixSocketPathByteLimit(host: RemoteHostPlatform): number | null { + if (isWindowsRemoteHost(host)) { + return null + } + return SUN_PATH_SIZE[host.os === 'darwin' ? 'darwin' : 'linux'] - 1 +} + +export function remoteSocketPathFitsLimit(host: RemoteHostPlatform, sockPath: string): boolean { + const limit = remoteUnixSocketPathByteLimit(host) + return limit === null || Buffer.byteLength(sockPath, 'utf8') <= limit +} + +/** Fixed-length, per-uid base. `/tmp` is the only POSIX directory whose length is not user-dependent. */ +export const SHORT_RELAY_SOCKET_DIR_PREFIX = '/tmp/.orca-relay-' + +export function shortRelaySocketDirForUid(uid: string): string { + return `${SHORT_RELAY_SOCKET_DIR_PREFIX}${uid}` +} + +/** + * The version segment the relocated socket lives under, named to match the version + * directories in `$HOME/.orca-remote` so one sweep pattern covers both bases. + * + * Why it has to exist: `relaySocketNameForInstanceId` hashes the *target*, not the + * build, so the filename alone is version-independent. Under `$HOME` the enclosing + * `relay-` directory supplies that dimension; without it here, the next + * Orca build would bind the exact path the previous build's relay still holds. The + * daemon handshake compares build hashes exactly, so that meeting is a version + * mismatch — and if the incumbent holds live work, `resolveRelayEndpointBeforeRelaunch` + * raises `RelayEndpointHeldError` and the user cannot connect at all until the old + * relay is stopped. The version is hashed rather than spelled out because the whole + * point of this base is a bounded length. + */ +export function shortRelayVersionSegment(relayVersionDirName: string): string { + return `relay-${createHash('sha256').update(relayVersionDirName).digest('hex').slice(0, 12)}` +} + +/** + * The whole hashed socket name is kept — shortening happens by replacing the + * variable-length directory, never by truncating the hash, so two targets on one + * host can never land on the same socket. + */ +export function shortRelaySocketPath(shortVersionDir: string, sockName: string): string { + return `${shortVersionDir}/${sockName}` +} + +const SHORT_DIR_MARKER = 'ORCA-RELAY-SHORT-SOCKET-DIR' + +/** + * Create (or adopt) the per-uid short socket directory and its version segment, and + * print the segment's path. + * + * Validate before mutating, never the other way round: an unconditional `chmod` follows a + * symlink, so a path planted by another user would have its *target's* mode rewritten before + * the owner check could reject it. A fresh `mkdir` under `umask 077` already yields 0700 and + * proves we own it, so the only path that adopts an existing entry is the one that first + * proves — via `ls -ldn`, which reports the entry itself rather than what it points at — that + * it is a real directory, owned by this uid, already 0700. Nothing else is touched. + */ +export function resolveShortRelaySocketDirCommand(versionSegment: string): string { + return [ + 'uid=$(id -u) || exit 1', + `dir="${SHORT_RELAY_SOCKET_DIR_PREFIX}$uid"`, + 'umask 077', + ...adoptOwnedDirectoryCommand('$dir'), + // The version segment is validated the same way rather than trusted: `$dir` being + // 0700 and ours does not prove what an earlier run left inside it still is. + `ver="$dir/${versionSegment}"`, + ...adoptOwnedDirectoryCommand('$ver'), + `printf '%s %s\n' '${SHORT_DIR_MARKER}' "$ver"` + ].join('\n') +} + +function adoptOwnedDirectoryCommand(target: string): string[] { + return [ + `if mkdir "${target}" 2>/dev/null; then`, + ' :', + 'else', + // Why the sub(): ls decorates the mode with a trailing marker for extended attributes (@), + // ACLs (+) or an SELinux context (.), so an exact match would refuse a directory we own. + ` entry=$(ls -ldn "${target}" 2>/dev/null | awk 'NR==1{sub(/[.@+]$/, "", $1); print $1" "$3}')`, + ' case "$entry" in', + ' "drwx------ $uid") ;;', + ' *) exit 1 ;;', + ' esac', + 'fi' + ] +} + +/** Tolerates login-shell banner noise ahead of the marker line. */ +export function parseShortRelaySocketDir(output: string, versionSegment: string): string | null { + for (const line of output.split('\n')) { + const trimmed = line.trim() + if (!trimmed.startsWith(`${SHORT_DIR_MARKER} `)) { + continue + } + const dir = trimmed.slice(SHORT_DIR_MARKER.length + 1).trim() + if ( + dir.startsWith(`${SHORT_RELAY_SOCKET_DIR_PREFIX}`) && + dir.endsWith(`/${versionSegment}`) && + !/[\r\n]/.test(dir) + ) { + return dir + } + } + return null +} diff --git a/src/main/ssh/sftp-stream-late-error.test.ts b/src/main/ssh/sftp-stream-late-error.test.ts new file mode 100644 index 00000000000..4ed3bcf8b32 --- /dev/null +++ b/src/main/ssh/sftp-stream-late-error.test.ts @@ -0,0 +1,172 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import type { SFTPWrapper } from 'ssh2' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { uploadBuffer, uploadFile, writeStringViaSftp, writeStringsViaSftp } from './sftp-upload' +import { writeRelayFile } from './ssh-relay-install-transfers' +import type { SshConnection } from './ssh-connection' +import { getRemoteHostPlatform } from './ssh-remote-platform' + +/** The exact error ssh2 builds from a STATUS reply of SSH_FX_NO_SUCH_FILE. */ +function sftpNoSuchFileError(): Error { + return Object.assign(new Error('file does not exist'), { code: 2 }) +} + +let tempDir = '' +let localFile = '' + +beforeEach(async () => { + tempDir = await mkdtemp(join(tmpdir(), 'orca-sftp-late-')) + localFile = join(tempDir, 'relay.js') + await writeFile(localFile, 'console.log(1)\n') +}) + +afterEach(async () => { + await rm(tempDir, { recursive: true, force: true }) +}) + +function sftpDoubleReturning(stream: PassThrough): SFTPWrapper { + return Object.assign(new EventEmitter(), { + createWriteStream: () => stream + }) as unknown as SFTPWrapper +} + +describe('late SFTP stream errors', () => { + // ssh2 emits the OPEN failure from inside the protocol parser. If no listener is left, + // Node throws it synchronously up through Socket.emit('data') and the main process dies + // (#15479) — uncaught exceptions are re-thrown by installUncaughtPipeErrorGuard, unlike + // rejections, which are only logged. + it('does not throw when a write stream fails after uploadFile settles', async () => { + const stream = new PassThrough() + stream.resume() + + await uploadFile(sftpDoubleReturning(stream), localFile, '/home/user/.orca-remote/relay.js') + + expect(() => stream.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + it('does not throw when a write stream fails after writeStringViaSftp settles', async () => { + const stream = new PassThrough() + stream.resume() + + await writeStringViaSftp(sftpDoubleReturning(stream), '/home/user/.orca-remote/.version', 'v1') + + expect(() => stream.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + it('does not throw when a write stream fails after uploadBuffer settles', async () => { + const stream = new PassThrough() + stream.resume() + + await uploadBuffer(sftpDoubleReturning(stream), Buffer.from('x'), '/home/user/x') + + expect(() => stream.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + // The failure mode a per-file loop reintroduces: writeStringViaSftp removes its own + // session listener at each settle, so a session that ran N transfers ends up with zero + // listeners while it is still open and still able to deliver a STATUS reply. + it('does not throw when a session error arrives after a multi-file write settles', async () => { + const sftp = Object.assign(new EventEmitter(), { + createWriteStream: () => { + const stream = new PassThrough() + stream.resume() + return stream + }, + end: () => {} + }) as unknown as SFTPWrapper + + await writeStringsViaSftp({ sftp: () => Promise.resolve(sftp) }, [ + { path: '/home/user/.local/bin/orca', contents: '#!/bin/sh\n' }, + { path: '/home/user/.local/bin/orca.mjs', contents: 'export {}\n' } + ]) + + expect(() => sftp.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + it('still rejects a multi-file write with a session error raised during it', async () => { + const sftp = Object.assign(new EventEmitter(), { + createWriteStream: () => { + const stream = new PassThrough() + queueMicrotask(() => sftp.emit('error', sftpNoSuchFileError())) + return stream + }, + end: () => {} + }) as unknown as SFTPWrapper + + // The latch must sit behind the transfer's own prepended listener, or a real + // mid-transfer failure would be swallowed into a hang. + await expect( + writeStringsViaSftp({ sftp: () => Promise.resolve(sftp) }, [ + { path: '/home/user/.local/bin/orca', contents: '#!/bin/sh\n' } + ]) + ).rejects.toThrow('file does not exist') + }) + + it('still rejects with the SFTP error when it arrives during the transfer', async () => { + const stream = new PassThrough() + stream.resume() + const failing = Object.assign(new EventEmitter(), { + createWriteStream: () => { + queueMicrotask(() => stream.emit('error', sftpNoSuchFileError())) + return stream + } + }) as unknown as SFTPWrapper + + await expect(writeStringViaSftp(failing, '/home/user/x', 'v1')).rejects.toThrow( + 'file does not exist' + ) + }) +}) + +describe('sandboxed SFTP subsystem diagnosis', () => { + it('leaves a permission refusal as itself rather than blaming a chroot', async () => { + // SSH_FX_PERMISSION_DENIED is a mode/ownership refusal on a path the subsystem can + // see -- a read-only home, a root-owned parent, a quota. Rewriting it into "your + // bastion chroots SFTP" sends the user to fix ProxyJump for a chmod. + const conn = { + writeFile: () => Promise.reject(Object.assign(new Error('permission denied'), { code: 3 })) + } as unknown as SshConnection + + const failure: unknown = await writeRelayFile( + conn, + getRemoteHostPlatform('linux-x64'), + '/home/user/.orca-remote/relay-1/.version', + 'v1' + ).then( + () => null, + (err: unknown) => err + ) + + expect((failure as Error).message).toBe('permission denied') + expect(failure).not.toHaveProperty('sandboxedSftpNamespace') + }) + + it('replaces the bare SFTP status with an actionable relay-install message', async () => { + const conn = { + writeFile: () => Promise.reject(sftpNoSuchFileError()) + } as unknown as SshConnection + + await expect( + writeRelayFile( + conn, + getRemoteHostPlatform('linux-x64'), + '/home/user/.orca-remote/relay-1/.version', + 'v1' + ) + ).rejects.toThrow(/SFTP subsystem sees a different filesystem/) + }) + + it('leaves unrelated transfer failures untouched', async () => { + const conn = { + writeFile: () => Promise.reject(new Error('Connection lost')) + } as unknown as SshConnection + + await expect( + writeRelayFile(conn, getRemoteHostPlatform('linux-x64'), '/home/user/x', 'v1') + ).rejects.toThrow('Connection lost') + }) +}) diff --git a/src/main/ssh/sftp-stream-late-error.ts b/src/main/ssh/sftp-stream-late-error.ts new file mode 100644 index 00000000000..1630668f874 --- /dev/null +++ b/src/main/ssh/sftp-stream-late-error.ts @@ -0,0 +1,108 @@ +/** + * Why an SFTP stream needs an `'error'` listener that outlives its transfer. + * + * ssh2 answers an SFTP request by invoking the pending request's callback from inside + * the protocol parser, on the socket's `data` handler stack. For a write stream that + * callback is `WriteStream.open`'s, and it does a bare `this.emit('error', err)`. Node + * throws when `'error'` is emitted on an emitter with no listener, so once a transfer + * has settled and removed its listener, a late STATUS reply becomes a *synchronous + * throw* out of `Protocol.parse` -> `Socket.emit('data')`. + * + * That is an uncaught exception, not a rejection: `installUnhandledRejectionLogging` + * absorbs rejections, but `installUncaughtPipeErrorGuard` re-throws uncaught exceptions + * and the app dies (#15479). A jump host that sandboxes the SFTP subsystem into its own + * chroot makes a late `SSH_FX_NO_SUCH_FILE` the normal answer, so the listener has to + * outlive the transfer rather than the other way round. + * + * `runSftpFallbackTransfer` already does this for the session emitter; this is the same + * guarantee one level down, on the streams. + */ + +/** SSH_FX_* status code ssh2 copies onto the `Error` it builds from a STATUS reply. */ +const SSH_FX_NO_SUCH_FILE = 2 + +type ErrorEmitter = { + on(event: 'error', listener: (err: Error) => void): unknown +} + +type SftpSessionEmitter = ErrorEmitter & { + once(event: 'close', listener: () => void): unknown + removeListener(event: 'error', listener: (err: Error) => void): unknown +} + +export type SftpStreamErrorLatch = { + /** Call once the transfer has settled; any error after this point is the late one. */ + markTransferSettled(): void +} + +export function latchLateSftpStreamErrors( + stream: ErrorEmitter, + remotePath: string +): SftpStreamErrorLatch { + let settled = false + stream.on('error', (err: Error) => { + if (!settled) { + // The transfer's own listener owns this error and will reject with it. + return + } + console.warn( + `[sftp] Ignored late stream error for ${remotePath}: ${err instanceof Error ? err.message : String(err)}` + ) + }) + return { + markTransferSettled: () => { + settled = true + } + } +} + +/** + * Hold one `'error'` listener on the SFTP *session* for as long as the session lives. + * + * A transfer that attaches and removes its own session listener — `writeStringViaSftp` + * does, so a session error can reject the write in flight — leaves the emitter with zero + * listeners between transfers and after the last one. A late STATUS reply arriving in + * that window is the synchronous throw described above. Errors during a transfer still + * reach that transfer first: it prepends its listener ahead of this one. + * + * Attach this once, right after `conn.sftp()`, on every path that runs transfers over a + * session it owns. + */ +export function latchLateSftpSessionErrors(sftp: SftpSessionEmitter): void { + const swallowLateSftpError = (): void => {} + sftp.on('error', swallowLateSftpError) + sftp.once('close', () => sftp.removeListener('error', swallowLateSftpError)) +} + +/** + * A chrooted SFTP subsystem answers a path outside its namespace with + * `SSH_FX_NO_SUCH_FILE`, because the path genuinely does not exist in the view it + * serves. `SSH_FX_PERMISSION_DENIED` is not that: it is an ordinary mode/ownership + * refusal on a path the subsystem *can* see — a read-only home, a root-owned parent, + * a quota — and rewriting it into "your bastion chroots SFTP" would send the user to + * fix ProxyJump for a `chmod`. + */ +export function isSandboxedSftpNamespaceError(error: unknown): boolean { + return (error as { code?: unknown } | null)?.code === SSH_FX_NO_SUCH_FILE +} + +/** + * SFTP is not optional for a bundled-ssh2 relay install — `SshConnection.sftp()` is the + * only transfer route on that transport, and the exec-based `tar`/`cat` transfers are + * bound to the system-SSH transport, not selectable per operation. So a sandboxed SFTP + * subsystem is a clean failure with an actionable message, not a degraded mode. + */ +export function describeSandboxedSftpFailure(error: unknown, remotePath: string): Error { + const detail = error instanceof Error ? error.message : String(error) + return Object.assign( + new Error( + `Relay install could not reach ${remotePath} over SFTP (${detail}). ` + + 'The host answered the shell channel but its SFTP subsystem sees a different filesystem — ' + + 'typically a bastion or jump host that chroots SFTP to a transfer directory. ' + + 'Orca cannot install the relay through a sandboxed SFTP subsystem; connect to the target ' + + 'host directly (for example with ProxyJump) or allow SFTP access to the account home.', + { cause: error } + ), + { sandboxedSftpNamespace: true } + ) +} diff --git a/src/main/ssh/sftp-upload.test.ts b/src/main/ssh/sftp-upload.test.ts index db86fa17f51..d5cf25abe70 100644 --- a/src/main/ssh/sftp-upload.test.ts +++ b/src/main/ssh/sftp-upload.test.ts @@ -40,7 +40,9 @@ describe('sftp-upload', () => { }) const writeStream = vi.mocked(sftp.createWriteStream).mock.results[0]?.value as Writable expect(writeStream.listenerCount('close')).toBe(0) - expect(writeStream.listenerCount('error')).toBe(0) + // One durable 'error' listener stays for the stream's whole life: a STATUS reply that + // lands after the transfer settles must not throw into ssh2's parser (#15479). + expect(writeStream.listenerCount('error')).toBe(1) }) it('uses no-clobber writes for nested files during exclusive directory upload', async () => { @@ -59,7 +61,9 @@ describe('sftp-upload', () => { }) const writeStream = vi.mocked(sftp.createWriteStream).mock.results[0]?.value as Writable expect(writeStream.listenerCount('close')).toBe(0) - expect(writeStream.listenerCount('error')).toBe(0) + // One durable 'error' listener stays for the stream's whole life: a STATUS reply that + // lands after the transfer settles must not throw into ssh2's parser (#15479). + expect(writeStream.listenerCount('error')).toBe(1) }) it('uploads files from valid dot-dot-prefixed local directories', async () => { diff --git a/src/main/ssh/sftp-upload.ts b/src/main/ssh/sftp-upload.ts index 6344a7c1457..514df81ed1d 100644 --- a/src/main/ssh/sftp-upload.ts +++ b/src/main/ssh/sftp-upload.ts @@ -4,6 +4,11 @@ import { lstat, open, readdir, realpath } from 'node:fs/promises' import { isAbsolute, join as pathJoin, relative, sep } from 'node:path' import { finished } from 'node:stream/promises' import type { SFTPWrapper } from 'ssh2' +import { + latchLateSftpSessionErrors, + latchLateSftpStreamErrors, + type SftpStreamErrorLatch +} from './sftp-stream-late-error' export function mkdirSftp( sftp: SFTPWrapper, @@ -44,6 +49,7 @@ async function uploadFileAndJoinTeardown( let handleClose: Promise | undefined let readStream: ReadStream | undefined let writeStream: ReturnType | undefined + let writeStreamErrors: SftpStreamErrorLatch | undefined const closeHandle = (): Promise => { handleClose ??= handle.close() return handleClose @@ -67,6 +73,9 @@ async function uploadFileAndJoinTeardown( writeStream = sftp.createWriteStream(remotePath, { flags: options?.exclusive ? 'wx' : 'w' }) + // Why: the OPEN reply can land after this transfer settles; without a listener that + // outlives it, ssh2 throws it synchronously into the socket handler (#15479). + writeStreamErrors = latchLateSftpStreamErrors(writeStream, remotePath) readStream = handle.createReadStream({ autoClose: false }) const abortTransfer = (): void => { const reason = @@ -100,6 +109,7 @@ async function uploadFileAndJoinTeardown( options?.signal?.removeEventListener('abort', abortTransfer) } } finally { + writeStreamErrors?.markTransferSettled() readStream?.destroy() writeStream?.destroy() await closeHandle() @@ -117,8 +127,10 @@ export function uploadBuffer( const writeStream = sftp.createWriteStream(remotePath, { flags: options?.append ? 'a' : options?.exclusive ? 'wx' : 'w' }) + const lateErrors = latchLateSftpStreamErrors(writeStream, remotePath) const cleanupListeners = (): void => { + lateErrors.markTransferSettled() writeStream.off('close', onClose) writeStream.off('error', onError) } @@ -147,8 +159,10 @@ export function writeStringViaSftp( ): Promise { return new Promise((resolve, reject) => { const ws = sftp.createWriteStream(remotePath) + const lateErrors = latchLateSftpStreamErrors(ws, remotePath) let settled = false const cleanup = (): void => { + lateErrors.markTransferSettled() sftp.removeListener('error', onError) ws.removeListener('close', onClose) ws.removeListener('error', onError) @@ -177,6 +191,29 @@ export function writeStringViaSftp( }) } +/** + * Write several files over one SFTP session, ending it when they are all done. + * + * Owns the session's late-error latch, which is why a caller must not hand-roll this + * loop: `writeStringViaSftp` drops its own session listener at each settle, so between + * files and after the last one the emitter would carry none, and a late STATUS reply + * throws synchronously out of ssh2's parser into main (#15479). + */ +export async function writeStringsViaSftp( + conn: { sftp(): Promise }, + files: readonly { path: string; contents: string }[] +): Promise { + const sftp = await conn.sftp() + latchLateSftpSessionErrors(sftp) + try { + for (const file of files) { + await writeStringViaSftp(sftp, file.path, file.contents) + } + } finally { + sftp.end() + } +} + export async function uploadDirectory( sftp: SFTPWrapper, localDir: string, diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index fc65af35c0a..af7346f5d35 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -85,6 +85,14 @@ import { powerShellCommand, powerShellLiteral, powerShellNativeArg } from './ssh import { relaySocketNameForInstanceId } from './ssh-relay-instance-id' import { resolveRelayEndpointBeforeRelaunch } from './ssh-relay-endpoint-takeover' import { sweepSupersededRelayEndpoints } from './ssh-relay-superseded-endpoints' +import { + parseShortRelaySocketDir, + remoteSocketPathFitsLimit, + resolveShortRelaySocketDirCommand, + shortRelaySocketPath, + shortRelayVersionSegment, + SHORT_RELAY_SOCKET_DIR_PREFIX +} from './relay-socket-path-limit' import { isSshSessionLimitError } from './ssh-session-limit-error' import { isWindowsRelayPipePath, @@ -598,6 +606,13 @@ async function deployAndLaunchRelayAttempt( remoteHome, currentRelayDir: remoteRelayDir, sockName: relaySocketNameForInstanceId(relayInstanceId), + // Set only when this launch relocated past sun_path; the sweep must not reap + // the socket the transport it just handed back is talking to. + ...(launched.sockPath.startsWith(SHORT_RELAY_SOCKET_DIR_PREFIX) + ? { + currentShortSocketDir: launched.sockPath.slice(0, launched.sockPath.lastIndexOf('/')) + } + : {}), nodePath: launched.nodePath }) ) @@ -1636,9 +1651,13 @@ async function launchRelay( const escapedNode = shellEscape(nodePath) // Why: remoteRelayDir is shared across Orca targets for one account; hashing the target ID into the socket name stops cross-target attach. const sockName = relaySocketNameForInstanceId(relayInstanceId) - const sockFile = relayEndpointForHost(hostPlatform, remoteDir, sockName) - const endpointDir = relayHookEndpointDirForHost(hostPlatform, remoteDir, sockFile) + const defaultSockFile = relayEndpointForHost(hostPlatform, remoteDir, sockName) + const endpointDir = relayHookEndpointDirForHost(hostPlatform, remoteDir, defaultSockFile) const credentialFile = joinRemotePath(hostPlatform, remoteDir, `${sockName}.credential`) + // Why: a long remote $HOME pushes the default endpoint past sun_path and bind fails with a bare `listen EINVAL` (#10726). + const sockFile = remoteSocketPathFitsLimit(hostPlatform, defaultSockFile) + ? defaultSockFile + : await resolveShortPosixRelaySocketPath(conn, remoteDir, sockName, defaultSockFile, signal) if (isWindowsRemoteHost(hostPlatform)) { const activePipeMarkerPath = windowsActivePipeMarkerPath(hostPlatform, remoteDir, sockName) @@ -1720,7 +1739,10 @@ async function launchRelay( signal }) // Why: --log-file lets the relay rotate relay.log in-process; the shell redirect stays to capture pre-JS boot/crash output. - const launchCmd = `cd ${escapedDir} && chmod 600 ${shellEscape(credentialFile)} && nohup ${escapedNode} relay.js --detached --grace-time ${graceTime} --sock-path ${shellEscape(sockFile)} --credential-file ${shellEscape(credentialFile)} --log-file ${shellEscape(logFile)} > ${shellEscape(logFile)} 2>&1 ${shellEscape(logFile)} 2>&1 {}) launchChannel.on('error', () => {}) @@ -1781,6 +1803,44 @@ async function launchRelay( } } +/** + * Move the endpoint under a `$HOME`-independent base so its length is bounded. + * + * The hashed socket name is preserved in full: only the directory shrinks, so the + * short form stays deterministic per target and cannot collide with another target. + * The version directory's identity comes along as a hashed segment, so a later build + * still binds a path of its own rather than the one its predecessor is holding. + */ +async function resolveShortPosixRelaySocketPath( + conn: SshConnection, + remoteDir: string, + sockName: string, + defaultSockFile: string, + signal?: AbortSignal +): Promise { + const versionSegment = shortRelayVersionSegment(remoteDir.slice(remoteDir.lastIndexOf('/') + 1)) + const output = await execCommand(conn, resolveShortRelaySocketDirCommand(versionSegment), { + signal + }).catch((err: unknown) => { + if (isUnconfirmedSshCommandTermination(err)) { + throw err + } + signal?.throwIfAborted() + return '' + }) + const shortDir = parseShortRelaySocketDir(output, versionSegment) + if (!shortDir) { + throw new Error( + `Relay socket path ${defaultSockFile} exceeds the remote Unix socket limit and no short socket directory could be created on the host.` + ) + } + const shortSockFile = shortRelaySocketPath(shortDir, sockName) + console.warn( + `[ssh-relay] Socket path too long for sun_path; using ${shortSockFile} instead of ${defaultSockFile}` + ) + return shortSockFile +} + function waitForRelayPoll(delayMs: number, signal?: AbortSignal): Promise { return new Promise((resolve, reject) => { const onAbort = (): void => { diff --git a/src/main/ssh/ssh-relay-install-transfers.ts b/src/main/ssh/ssh-relay-install-transfers.ts index c5c99f99b70..6f8ede57984 100644 --- a/src/main/ssh/ssh-relay-install-transfers.ts +++ b/src/main/ssh/ssh-relay-install-transfers.ts @@ -13,6 +13,11 @@ import { type SftpNamespacePathMapping } from './sftp-namespace-resolution' import type { RemoteHostPlatform } from './ssh-remote-platform' +import { + describeSandboxedSftpFailure, + isSandboxedSftpNamespaceError, + latchLateSftpSessionErrors +} from './sftp-stream-late-error' export type RelayTransferOptions = { signal?: AbortSignal @@ -25,6 +30,18 @@ export async function uploadRelayDirectory( shellRemoteDir: string, hostPlatform: RemoteHostPlatform, options?: RelayTransferOptions +): Promise { + await withSandboxedSftpDiagnosis(shellRemoteDir, () => + uploadRelayDirectoryTransfer(conn, localRelayDir, shellRemoteDir, hostPlatform, options) + ) +} + +async function uploadRelayDirectoryTransfer( + conn: SshConnection, + localRelayDir: string, + shellRemoteDir: string, + hostPlatform: RemoteHostPlatform, + options?: RelayTransferOptions ): Promise { if (typeof conn.uploadDirectory === 'function') { await conn.uploadDirectory(localRelayDir, shellRemoteDir, { @@ -52,6 +69,18 @@ export async function writeRelayFile( shellRemotePath: string, contents: string, options?: RelayTransferOptions +): Promise { + await withSandboxedSftpDiagnosis(shellRemotePath, () => + writeRelayFileTransfer(conn, hostPlatform, shellRemotePath, contents, options) + ) +} + +async function writeRelayFileTransfer( + conn: SshConnection, + hostPlatform: RemoteHostPlatform, + shellRemotePath: string, + contents: string, + options?: RelayTransferOptions ): Promise { if (typeof conn.writeFile === 'function') { await conn.writeFile(shellRemotePath, contents, { @@ -77,7 +106,6 @@ async function runSftpFallbackTransfer( transfer: (sftp: SFTPWrapper) => Promise ): Promise { const sftp = await conn.sftp(options?.signal) - const swallowLateSftpError = (): void => {} let sftpEndRequested = false const endSftp = (): void => { if (!sftpEndRequested) { @@ -85,9 +113,7 @@ async function runSftpFallbackTransfer( sftp.end() } } - // A late session 'error' after settle would otherwise be unhandled and crash main. - sftp.on('error', swallowLateSftpError) - sftp.once('close', () => sftp.removeListener('error', swallowLateSftpError)) + latchLateSftpSessionErrors(sftp) try { await raceSftpFileTransferWithAbort( transfer(sftp), @@ -102,3 +128,23 @@ async function runSftpFallbackTransfer( endSftp() } } + +/** + * A jump host whose SFTP subsystem is chrooted answers a home path with + * SSH_FX_NO_SUCH_FILE even though the shell channel resolves it (#15479). SFTP is the + * only install route on the bundled-ssh2 transport, so say what the host did rather + * than surfacing a bare "file does not exist". + */ +async function withSandboxedSftpDiagnosis( + remotePath: string, + transfer: () => Promise +): Promise { + try { + return await transfer() + } catch (error) { + if (isSandboxedSftpNamespaceError(error)) { + throw describeSandboxedSftpFailure(error, remotePath) + } + throw error + } +} diff --git a/src/main/ssh/ssh-relay-reset.ts b/src/main/ssh/ssh-relay-reset.ts index 50101084478..3d585b525f2 100644 --- a/src/main/ssh/ssh-relay-reset.ts +++ b/src/main/ssh/ssh-relay-reset.ts @@ -2,6 +2,7 @@ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' import { execCommand } from './ssh-relay-deploy-helpers' import { relaySocketNameForInstanceId } from './ssh-relay-instance-id' +import { SHORT_RELAY_SOCKET_DIR_PREFIX } from './relay-socket-path-limit' export async function forceStopRelayForTarget( conn: SshConnection, @@ -12,8 +13,10 @@ export async function forceStopRelayForTarget( const script = [ `sock_name=${escapedSockName}`, 'base="${HOME}/.orca-remote"', - 'if [ -d "$base" ]; then', - ' for sock in "$base"/relay-*/"$sock_name" "$base"/"$sock_name"; do', + // Why: a long $HOME moves the socket to the sun_path-safe short base (#10726); reset must reach it there too. + `short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`, + 'if [ -d "$base" ] || [ -d "$short_base" ]; then', + ' for sock in "$base"/relay-*/"$sock_name" "$base"/"$sock_name" "$short_base"/relay-*/"$sock_name"; do', ' [ -S "$sock" ] || continue', ' pid=""', // Why: lsof ORs selectors by default; -a prevents reset from targeting diff --git a/src/main/ssh/ssh-relay-session.ts b/src/main/ssh/ssh-relay-session.ts index 04fe6020749..c303d33b428 100644 --- a/src/main/ssh/ssh-relay-session.ts +++ b/src/main/ssh/ssh-relay-session.ts @@ -5,6 +5,7 @@ import { randomUUID } from 'node:crypto' import type { BrowserWindow } from 'electron' import { deployAndLaunchRelay } from './ssh-relay-deploy' import { execCommand } from './ssh-relay-deploy-helpers' +import { writeStringsViaSftp } from './sftp-upload' import { isRelayVersionMismatchError } from './ssh-relay-version-mismatch-error' import { isRelayEndpointHeldError } from './ssh-relay-endpoint-incumbent' import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair' @@ -1414,20 +1415,7 @@ export class SshRelaySession { await conn.writeFile(file.path, file.contents, { hostPlatform }) } } else { - const sftp = await conn.sftp() - try { - for (const file of plan.files) { - await new Promise((resolve, reject) => { - const ws = sftp.createWriteStream(file.path) - sftp.once('error', reject) - ws.once('close', resolve) - ws.once('error', reject) - ws.end(file.contents) - }) - } - } finally { - sftp.end() - } + await writeStringsViaSftp(conn, plan.files) } for (const command of plan.postWriteCommands) { await execCommand(conn, command, { wrapCommand: !isWindowsRemoteHost(hostPlatform) }) diff --git a/src/main/ssh/ssh-relay-superseded-endpoints.ts b/src/main/ssh/ssh-relay-superseded-endpoints.ts index 1a9554f7b82..4b1ad5637ef 100644 --- a/src/main/ssh/ssh-relay-superseded-endpoints.ts +++ b/src/main/ssh/ssh-relay-superseded-endpoints.ts @@ -19,6 +19,7 @@ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' import { RELAY_REMOTE_DIR } from './relay-protocol' +import { SHORT_RELAY_SOCKET_DIR_PREFIX } from './relay-socket-path-limit' import { execCommand } from './ssh-relay-deploy-helpers' import { describeRelayEndpointIncumbent, @@ -53,6 +54,8 @@ export type SupersededRelaySweepOptions = { currentRelayDir: string /** Stable per-target socket filename, from `relaySocketNameForInstanceId`. */ sockName: string + /** Set only when this launch relocated its socket; that directory is never swept. */ + currentShortSocketDir?: string nodePath: string signal?: AbortSignal } @@ -63,15 +66,23 @@ export function supersededRelayEndpointListCommand(options: { remoteHome: string currentRelayDir: string sockName: string + currentShortSocketDir?: string }): string { return [ `base=${shellEscape(`${options.remoteHome}/${RELAY_REMOTE_DIR}`)}`, `sock_name=${shellEscape(options.sockName)}`, `current=${shellEscape(options.currentRelayDir)}`, - 'for sock in "$base"/relay-*/"$sock_name"; do', + // Why the second base: a host whose `$HOME` pushes the endpoint past `sun_path` binds + // under `/tmp/.orca-relay-/relay-/` instead (relay-socket-path-limit.ts). + // Those orphans are the same population this sweep exists to make visible, and the + // `$HOME` glob cannot see them. The uid is resolved on the host; the client never knows it. + `short_current=${shellEscape(options.currentShortSocketDir ?? '')}`, + `short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`, + 'for sock in "$base"/relay-*/"$sock_name" "$short_base"/relay-*/"$sock_name"; do', ' [ -S "$sock" ] || continue', ' dir=${sock%/*}', ' [ "$dir" = "$current" ] && continue', + ' [ -n "$short_current" ] && [ "$dir" = "$short_current" ] && continue', ' printf \'%s\\n\' "$sock"', 'done' ].join('\n') @@ -79,7 +90,14 @@ export function supersededRelayEndpointListCommand(options: { /** Remove a socket inode proven to have no holder, so version-dir GC can reclaim the tree. */ export function removeStaleRelayEndpointCommand(sockPath: string): string { - return `rm -f ${shellEscape(sockPath)}` + const remove = `rm -f ${shellEscape(sockPath)}` + if (!sockPath.startsWith(SHORT_RELAY_SOCKET_DIR_PREFIX)) { + return remove + } + // `gcOldRelayVersions` only walks `$HOME/.orca-remote`, so nothing else would ever + // reclaim a relocated version segment. `rmdir` fails while another target of the same + // build still has a socket there, which is exactly the condition for keeping it. + return `${remove}; rmdir ${shellEscape(sockPath.slice(0, sockPath.lastIndexOf('/')))} 2>/dev/null || true` } export function classifySupersededRelay( From 710405698472c3d9618c3114c7f15088be4d72cc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:21 -0700 Subject: [PATCH 04/77] fix(watcher): route relay watch-root capacity refusals off the fast ladder (#17950) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): stop two unrecoverable relay refusal loops A relay refusal that is a pure function of state the client cannot change was being retried forever, on two different paths. - pty.openClient: a superseded owner proof is refuted evidence, not a transient fault. The client kept re-presenting the identical proof, so every reconnect reproduced the same refusal until the relay was redeployed (#12895, #12931). It is now dropped exactly as a stale lease already is, and the claim re-asked without it. - fs.watch: the relay's watch-root capacity refusal was classified 'unavailable' and retried at 1 Hz per root for 60s, re-armed indefinitely. A folder workspace with more repos than the cap turns that into a permanent install storm scaled by the excess root count (#11196). It is now its own 'capacity' result that goes straight to the existing dormant backoff, mirroring what the local watcher path already does. * fix(watcher): route relay watch-root capacity refusals off the fast ladder A full watch-root cap is a decision, not a fault, so a 1 Hz reinstall per refused root only bills the relay the load that keeps the cap busy (#11196). Capacity refusals now go straight to the dormant backoff. The relay side no longer refuses on a slot it is about to hand back: an over-cap caused by roots still unsubscribing waits once on the teardowns settling — the release event, mirroring WatcherSupervisorCapacityWait — before it answers. A parked waiter is excluded from the accounting so it cannot take a slot from the root already reclaiming one. Drops the SSH owner-recovery half of this branch. Its premise — that a -32043 SUPERSEDED refusal is permanent — is false: the refusal fires only while the incumbent is 'active', and assertPtyConsumerOwnerRecovery explicitly admits the identical lower-generation proof once the incumbent flips to 'disconnected' (relay-pty-consumer-owner-displacement.test.ts proves it). The remedy could not work either: the proofless re-ask routes into refuseHeldPtyConsumerOwner, which is declared `: never` and, with sameClient true by construction, always throws. It would have traded one refusal loop for another, minus the checkpoints and minus the proof that resumes the claim once the relay reaps the incumbent. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ipc/filesystem-watcher-handlers.ts | 7 + .../ipc/filesystem-watcher-lifecycle-state.ts | 5 +- ...filesystem-watcher-remote-capacity.test.ts | 91 +++++++++++++ .../filesystem-watcher-remote-controller.ts | 3 +- .../ipc/filesystem-watcher-remote-dormant.ts | 4 +- .../ipc/filesystem-watcher-remote-install.ts | 5 + ...ilesystem-watcher-remote-provider-rearm.ts | 5 + .../ipc/filesystem-watcher-remote-removal.ts | 5 +- .../ipc/filesystem-watcher-remote-retry.ts | 6 + src/relay/fs-handler.test.ts | 114 +---------------- src/relay/relay-filesystem-watch-registry.ts | 18 ++- src/relay/relay-fs-test-dispatcher.ts | 120 ++++++++++++++++++ src/relay/relay-watch-root-capacity-gate.ts | 86 +++++++++++++ src/relay/relay-watch-root-capacity.test.ts | 107 ++++++++++++++++ src/relay/relay-watcher-root-capacity.ts | 20 ++- src/relay/relay-watcher-teardown-tracker.ts | 11 ++ src/shared/watch-root-capacity-refusal.ts | 9 ++ 17 files changed, 487 insertions(+), 129 deletions(-) create mode 100644 src/main/ipc/filesystem-watcher-remote-capacity.test.ts create mode 100644 src/relay/relay-fs-test-dispatcher.ts create mode 100644 src/relay/relay-watch-root-capacity-gate.ts create mode 100644 src/relay/relay-watch-root-capacity.test.ts create mode 100644 src/shared/watch-root-capacity-refusal.ts diff --git a/src/main/ipc/filesystem-watcher-handlers.ts b/src/main/ipc/filesystem-watcher-handlers.ts index 7a2b6230abd..bac74f63589 100644 --- a/src/main/ipc/filesystem-watcher-handlers.ts +++ b/src/main/ipc/filesystem-watcher-handlers.ts @@ -10,6 +10,7 @@ import { import { installRemoteWatcher, reinstallRemoteWatchersForConnection, + scheduleDormantRemoteWatcherRearm, scheduleRemoteWatcherRetry } from './filesystem-watcher-remote-controller' import { rememberDesiredRemoteWatcher } from './filesystem-watcher-remote-desired' @@ -41,6 +42,12 @@ export function registerFilesystemWatcherHandlers(): void { args.connectionId, args.worktreePath ) + if (result === 'capacity') { + // Why straight to the dormant backoff: the cap is full until some other root is released, + // which a 1 Hz reinstall cannot bring about — it only adds relay load per refused root. + scheduleDormantRemoteWatcherRearm(args.connectionId, args.worktreePath) + return + } if (result === 'unavailable') { if (!watcherLifecycleState.loggedUnavailableRemoteWatchers.has(key)) { watcherLifecycleState.loggedUnavailableRemoteWatchers.add(key) diff --git a/src/main/ipc/filesystem-watcher-lifecycle-state.ts b/src/main/ipc/filesystem-watcher-lifecycle-state.ts index 548d958f90c..7d6a120c0a7 100644 --- a/src/main/ipc/filesystem-watcher-lifecycle-state.ts +++ b/src/main/ipc/filesystem-watcher-lifecycle-state.ts @@ -36,7 +36,10 @@ export type RemoteWatcherState = { batch: RemoteWatcherEventBatch } -export type RemoteWatcherInstallResult = 'installed' | 'unavailable' | 'cancelled' +// Why 'capacity' is not 'unavailable': the relay refused because its watch-root cap is full, which is +// a decision, not a fault. The 1 Hz unavailable retry cannot change that answer, and a folder +// workspace whose repo count exceeds the cap turns it into a permanent per-root storm (#11196). +export type RemoteWatcherInstallResult = 'installed' | 'unavailable' | 'capacity' | 'cancelled' export type RemoteWatcherResyncState = { lastSentAt: number diff --git a/src/main/ipc/filesystem-watcher-remote-capacity.test.ts b/src/main/ipc/filesystem-watcher-remote-capacity.test.ts new file mode 100644 index 00000000000..97a8f82bf33 --- /dev/null +++ b/src/main/ipc/filesystem-watcher-remote-capacity.test.ts @@ -0,0 +1,91 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { handleMock, getSshFilesystemProviderMock } = vi.hoisted(() => ({ + handleMock: vi.fn(), + getSshFilesystemProviderMock: vi.fn() +})) + +vi.mock('electron', () => ({ + ipcMain: { handle: handleMock } +})) + +vi.mock('fs/promises', () => ({ stat: vi.fn() })) +vi.mock('@parcel/watcher', () => ({ subscribe: vi.fn() })) +vi.mock('./filesystem-watcher-wsl', () => ({ createWslWatcher: vi.fn() })) +vi.mock('../providers/ssh-filesystem-dispatch', () => ({ + getSshFilesystemProvider: getSshFilesystemProviderMock, + onSshFilesystemProviderRegistered: () => () => {} +})) + +import { WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE } from '../../shared/watch-root-capacity-refusal' +import { closeAllWatchers, registerFilesystemWatcherHandlers } from './filesystem-watcher' +import { watcherLifecycleState } from './filesystem-watcher-lifecycle-state' +import { getRemoteWatcherKey } from './filesystem-watcher-paths' + +type HandlerMap = Record unknown> + +describe('remote filesystem watcher capacity refusals', () => { + const handlers: HandlerMap = {} + + beforeEach(async () => { + handleMock.mockReset() + getSshFilesystemProviderMock.mockReset() + for (const key of Object.keys(handlers)) { + delete handlers[key] + } + handleMock.mockImplementation((channel, handler) => { + handlers[channel] = handler + }) + registerFilesystemWatcherHandlers() + await closeAllWatchers() + }) + + afterEach(async () => { + for (const dormant of watcherLifecycleState.dormantRemoteWatchers.values()) { + clearTimeout(dormant.timer) + } + watcherLifecycleState.dormantRemoteWatchers.clear() + await closeAllWatchers() + vi.useRealTimers() + }) + + // A folder workspace with more repos than the relay's watch-root cap leaves every excess root + // permanently refused; the 1 Hz unavailable ladder then bills the relay one install per root per + // second, which is the load that pinned it (#11196). + it('does not retry a relay watch-root capacity refusal on the fast ladder', async () => { + vi.useFakeTimers() + const watchMock = vi.fn(async () => { + throw new Error(WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE) + }) + getSshFilesystemProviderMock.mockReturnValue({ watch: watchMock }) + const sender = { isDestroyed: () => false, send: vi.fn(), once: vi.fn(), id: 1 } + const args = { worktreePath: '/home/me/repos/one', connectionId: 'conn-capacity' } + + await handlers['fs:watchWorktree']({ sender }, args) + const key = getRemoteWatcherKey(args.connectionId, args.worktreePath) + + expect(watchMock).toHaveBeenCalledTimes(1) + expect(watcherLifecycleState.pendingRemoteWatcherRetries.has(key)).toBe(false) + expect(watcherLifecycleState.dormantRemoteWatchers.has(key)).toBe(true) + + await vi.advanceTimersByTimeAsync(10_000) + expect(watchMock).toHaveBeenCalledTimes(1) + }) + + it('still retries an ordinary unavailable install on the fast ladder', async () => { + vi.useFakeTimers() + const watchMock = vi.fn(async () => { + throw new Error('Relay channel lost') + }) + getSshFilesystemProviderMock.mockReturnValue({ watch: watchMock }) + const sender = { isDestroyed: () => false, send: vi.fn(), once: vi.fn(), id: 2 } + const args = { worktreePath: '/home/me/repos/two', connectionId: 'conn-unavailable' } + + await handlers['fs:watchWorktree']({ sender }, args) + const key = getRemoteWatcherKey(args.connectionId, args.worktreePath) + + expect(watcherLifecycleState.pendingRemoteWatcherRetries.has(key)).toBe(true) + await vi.advanceTimersByTimeAsync(2_500) + expect(watchMock.mock.calls.length).toBeGreaterThan(1) + }) +}) diff --git a/src/main/ipc/filesystem-watcher-remote-controller.ts b/src/main/ipc/filesystem-watcher-remote-controller.ts index 75bc4aa75d5..1b6953cd292 100644 --- a/src/main/ipc/filesystem-watcher-remote-controller.ts +++ b/src/main/ipc/filesystem-watcher-remote-controller.ts @@ -63,7 +63,8 @@ export function reinstallRemoteWatchersForConnection(connectionId: string): void reinstallRemoteWatchersForConnectionCore(connectionId, { install: installRemoteWatcher, requestResync: requestRemoteWatcherResync, - scheduleRetry: scheduleRemoteWatcherRetry + scheduleRetry: scheduleRemoteWatcherRetry, + scheduleDormant: scheduleDormantRemoteWatcherRearm }) } diff --git a/src/main/ipc/filesystem-watcher-remote-dormant.ts b/src/main/ipc/filesystem-watcher-remote-dormant.ts index 741753b42e5..7c168114f51 100644 --- a/src/main/ipc/filesystem-watcher-remote-dormant.ts +++ b/src/main/ipc/filesystem-watcher-remote-dormant.ts @@ -97,8 +97,8 @@ async function rearmDormantRemoteWatcher( worktreePath, listeners.filter((_, index) => results[index] === 'installed') ) - // Why: 'cancelled' means shutdown or the last listener left, so only 'unavailable' stays dormant. - if (results.some((result) => result === 'unavailable')) { + // Why: 'cancelled' means shutdown or the last listener left, so only a refusal stays dormant. + if (results.some((result) => result === 'unavailable' || result === 'capacity')) { scheduleDormantRemoteWatcherRearmCore( connectionId, worktreePath, diff --git a/src/main/ipc/filesystem-watcher-remote-install.ts b/src/main/ipc/filesystem-watcher-remote-install.ts index c069c8dc16b..3ab5babbd91 100644 --- a/src/main/ipc/filesystem-watcher-remote-install.ts +++ b/src/main/ipc/filesystem-watcher-remote-install.ts @@ -1,5 +1,6 @@ import type { WebContents } from 'electron' import type { FsChangedPayload } from '../../shared/filesystem-entry-types' +import { isWatchRootCapacityRefusal } from '../../shared/watch-root-capacity-refusal' import { WATCH_BATCH_MAX_WAIT_MS, WATCH_BATCH_TRAILING_MS @@ -202,6 +203,10 @@ async function doInstallRemoteWatcher( if (cancelToken.cancelled || cancelToken.abortController.signal.aborted) { return 'cancelled' } + if (isWatchRootCapacityRefusal(err)) { + console.warn(`[filesystem-watcher] relay watch-root capacity reached for ${key}`) + return 'capacity' + } console.warn(`[filesystem-watcher] SSH watcher unavailable for ${key}:`, err) return 'unavailable' } finally { diff --git a/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts b/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts index 6b0ea28ef9b..3cc657c4086 100644 --- a/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts +++ b/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts @@ -29,6 +29,7 @@ export function reinstallRemoteWatchersForConnectionCore( install: InstallRemoteWatcher requestResync: RequestRemoteWatcherResync scheduleRetry: ScheduleRemoteWatcherRetry + scheduleDormant: (connectionId: string, worktreePath: string) => void } ): void { if (watcherLifecycleState.remoteWatchersClosed) { @@ -90,6 +91,10 @@ export function reinstallRemoteWatchersForConnectionCore( desired.worktreePath, listeners.filter((_, index) => results[index] === 'installed') ) + if (results.some((result) => result === 'capacity')) { + dependencies.scheduleDormant(desired.connectionId, desired.worktreePath) + return + } if (results.some((result) => result === 'unavailable')) { for (const listener of listeners) { dependencies.scheduleRetry( diff --git a/src/main/ipc/filesystem-watcher-remote-removal.ts b/src/main/ipc/filesystem-watcher-remote-removal.ts index 61424dc1db9..0e9c741851a 100644 --- a/src/main/ipc/filesystem-watcher-remote-removal.ts +++ b/src/main/ipc/filesystem-watcher-remote-removal.ts @@ -9,6 +9,7 @@ import { } from './filesystem-watcher-listener-lifecycle' import { installRemoteWatcher, + scheduleDormantRemoteWatcherRearm, scheduleRemoteWatcherRetry } from './filesystem-watcher-remote-controller' @@ -75,7 +76,9 @@ export async function restoreRemoteWatcherAfterFailedRemoval( continue } const result = await installRemoteWatcher(sender, connectionId, worktreePath) - if (result === 'unavailable') { + if (result === 'capacity') { + scheduleDormantRemoteWatcherRearm(connectionId, worktreePath) + } else if (result === 'unavailable') { scheduleRemoteWatcherRetry(sender, connectionId, worktreePath) } sender.send('fs:changed', { diff --git a/src/main/ipc/filesystem-watcher-remote-retry.ts b/src/main/ipc/filesystem-watcher-remote-retry.ts index 0151ecf4419..d73f9ed4db4 100644 --- a/src/main/ipc/filesystem-watcher-remote-retry.ts +++ b/src/main/ipc/filesystem-watcher-remote-retry.ts @@ -100,6 +100,12 @@ export function scheduleRemoteWatcherRetryCore( listeners.filter((_, index) => results[index] === 'installed') ) } + // Why capacity leaves the fast window: the relay is refusing on a full watch-root cap, and a + // 1 Hz reinstall per refused root is exactly the load that keeps the cap busy (#11196). + if (results.some((result) => result === 'capacity')) { + dependencies.scheduleDormant(connectionId, worktreePath) + return + } // Why: don't re-arm on 'cancelled' (renderer stopped watching) — it would fire a stale overflow when the 60s window expires. if (results.some((result) => result === 'unavailable')) { for (const listener of listeners) { diff --git a/src/relay/fs-handler.test.ts b/src/relay/fs-handler.test.ts index 9dc4812ef98..fcea45c898c 100644 --- a/src/relay/fs-handler.test.ts +++ b/src/relay/fs-handler.test.ts @@ -8,6 +8,7 @@ import * as path from 'node:path' import { mkdtempSync, writeFileSync, mkdirSync, symlinkSync } from 'node:fs' import { tmpdir } from 'node:os' import { subscribeWithInProcessWatcher } from '../main/ipc/parcel-watcher-in-process-fallback' +import { createMockDispatcher } from './relay-fs-test-dispatcher' const { mockSubscribe } = vi.hoisted(() => ({ mockSubscribe: vi.fn() @@ -17,91 +18,6 @@ vi.mock('@parcel/watcher', () => ({ subscribe: mockSubscribe })) -function createMockDispatcher() { - const requestHandlers = new Map< - string, - ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => Promise - >() - const notificationHandlers = new Map< - string, - ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => void - >() - const detachListeners = new Set<(clientId: number) => void>() - const notifications: { method: string; params?: Record }[] = [] - - return { - onRequest: vi.fn( - ( - method: string, - handler: ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => Promise - ) => { - requestHandlers.set(method, handler) - } - ), - onNotification: vi.fn( - ( - method: string, - handler: ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => void - ) => { - notificationHandlers.set(method, handler) - } - ), - notify: vi.fn((method: string, params?: Record) => { - notifications.push({ method, params }) - }), - notifyClient: vi.fn(), - onClientDetached: vi.fn((listener: (clientId: number) => void) => { - detachListeners.add(listener) - return () => detachListeners.delete(listener) - }), - _requestHandlers: requestHandlers, - _notificationHandlers: notificationHandlers, - _notifications: notifications, - async callRequest( - method: string, - params: Record = {}, - context?: { clientId?: number; isStale: () => boolean } - ) { - const handler = requestHandlers.get(method) - if (!handler) { - throw new Error(`No handler for ${method}`) - } - return handler(params, { - clientId: context?.clientId ?? 1, - isStale: context?.isStale ?? (() => false) - }) - }, - callNotification( - method: string, - params: Record = {}, - context?: { clientId: number; isStale: () => boolean } - ) { - const handler = notificationHandlers.get(method) - if (!handler) { - throw new Error(`No handler for ${method}`) - } - handler(params, context ?? { clientId: 1, isStale: () => false }) - }, - detachClient(clientId: number) { - for (const listener of detachListeners) { - listener(clientId) - } - } - } -} - function statIdentity(stats: { dev?: number ino?: number @@ -837,34 +753,6 @@ describe('FsHandler', () => { await joined }) - it('blocks replacement watches behind physical unsubscribe and counts the pending slot', async () => { - let resolveUnsubscribe: () => void = () => {} - const unsubscribe = vi.fn( - () => - new Promise((resolve) => { - resolveUnsubscribe = resolve - }) - ) - mockSubscribe.mockResolvedValue({ unsubscribe }) - await dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) - dispatcher.callNotification('fs.unwatch', { rootPath: tmpDir }) - - const replacement = dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) - for (let index = 0; index < 19; index += 1) { - await dispatcher.callRequest('fs.watch', { - rootPath: path.join(tmpDir, `pending-cap-${index}`) - }) - } - await expect( - dispatcher.callRequest('fs.watch', { rootPath: path.join(tmpDir, 'over-pending-cap') }) - ).rejects.toThrow('Maximum number of file watchers reached') - expect(mockSubscribe).toHaveBeenCalledTimes(20) - - resolveUnsubscribe() - await replacement - expect(mockSubscribe).toHaveBeenCalledTimes(21) - }) - it('retains a failed native unsubscribe slot until acknowledged retry succeeds', async () => { const unsubscribe = vi .fn() diff --git a/src/relay/relay-filesystem-watch-registry.ts b/src/relay/relay-filesystem-watch-registry.ts index de583b2f50a..14f5d5b2ee8 100644 --- a/src/relay/relay-filesystem-watch-registry.ts +++ b/src/relay/relay-filesystem-watch-registry.ts @@ -16,7 +16,7 @@ import { type RelayWatcherTeardownState } from './relay-watcher-teardown-tracker' import { emitRelayWatcherTerminalFailure } from './relay-watcher-terminal-notifier' -import { assertRelayWatcherRootCapacity } from './relay-watcher-root-capacity' +import { RelayWatchRootCapacityGate } from './relay-watch-root-capacity-gate' import { normalizeRuntimePathForComparison } from '../shared/cross-platform-path' import { trackRelayWatcherSetup, @@ -33,6 +33,11 @@ const RELAY_WATCH_OPTIONS = buildParcelWatcherIgnoreOptions(WATCHER_IGNORE_DIRS) export class RelayFilesystemWatchRegistry { private readonly watches = new Map() private readonly pendingSetups = new Map() + private readonly capacityGate = new RelayWatchRootCapacityGate( + this.watches, + this.pendingSetups, + () => this.teardownTracker + ) private readonly teardownTracker: RelayWatcherTeardownTracker private readonly removalFence: RelayWatcherRemovalFence @@ -95,6 +100,10 @@ export class RelayFilesystemWatchRegistry { if (rootTeardown) { await rootTeardown } + const capacityRelease = this.capacityGate.release(rootKey, context?.signal) + if (capacityRelease) { + await capacityRelease + } const clientId = context?.clientId ?? 0 const isStale = context?.isStale ?? (() => false) const existing = this.watches.get(rootKey) @@ -107,12 +116,7 @@ export class RelayFilesystemWatchRegistry { return } - assertRelayWatcherRootCapacity( - this.watches.keys(), - this.pendingSetups.keys(), - this.teardownTracker.rootPaths(), - rootKey - ) + this.capacityGate.assert(rootKey) const state = createRelayWatcherState(rootKey, rootPath, clientId, isStale, watchId) this.watches.set(rootKey, state) diff --git a/src/relay/relay-fs-test-dispatcher.ts b/src/relay/relay-fs-test-dispatcher.ts new file mode 100644 index 00000000000..17d0e20bd19 --- /dev/null +++ b/src/relay/relay-fs-test-dispatcher.ts @@ -0,0 +1,120 @@ +import { vi, type Mock } from 'vitest' + +type MockRequestHandler = ( + params: Record, + context?: { clientId: number; isStale: () => boolean } +) => Promise +type MockNotificationHandler = ( + params: Record, + context?: { clientId: number; isStale: () => boolean } +) => void +type MockCallContext = { clientId?: number; isStale: () => boolean } + +// Explicit rather than inferred: vi.fn()'s inferred type is not nameable across project boundaries. +export type MockRelayFsDispatcher = { + onRequest: Mock + onNotification: Mock + notify: Mock + notifyClient: Mock + onClientDetached: Mock + _requestHandlers: Map + _notificationHandlers: Map + _notifications: { method: string; params?: Record }[] + callRequest: ( + method: string, + params?: Record, + context?: MockCallContext + ) => Promise + callNotification: ( + method: string, + params?: Record, + context?: { clientId: number; isStale: () => boolean } + ) => void + detachClient: (clientId: number) => void +} + +/** Records handlers and notifications so a test can drive FsHandler without a real transport. */ +export function createMockDispatcher(): MockRelayFsDispatcher { + const requestHandlers = new Map< + string, + ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => Promise + >() + const notificationHandlers = new Map< + string, + ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => void + >() + const detachListeners = new Set<(clientId: number) => void>() + const notifications: { method: string; params?: Record }[] = [] + + return { + onRequest: vi.fn( + ( + method: string, + handler: ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => Promise + ) => { + requestHandlers.set(method, handler) + } + ), + onNotification: vi.fn( + ( + method: string, + handler: ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => void + ) => { + notificationHandlers.set(method, handler) + } + ), + notify: vi.fn((method: string, params?: Record) => { + notifications.push({ method, params }) + }), + notifyClient: vi.fn(), + onClientDetached: vi.fn((listener: (clientId: number) => void) => { + detachListeners.add(listener) + return () => detachListeners.delete(listener) + }), + _requestHandlers: requestHandlers, + _notificationHandlers: notificationHandlers, + _notifications: notifications, + async callRequest( + method: string, + params: Record = {}, + context?: { clientId?: number; isStale: () => boolean } + ) { + const handler = requestHandlers.get(method) + if (!handler) { + throw new Error(`No handler for ${method}`) + } + return handler(params, { + clientId: context?.clientId ?? 1, + isStale: context?.isStale ?? (() => false) + }) + }, + callNotification( + method: string, + params: Record = {}, + context?: { clientId: number; isStale: () => boolean } + ) { + const handler = notificationHandlers.get(method) + if (!handler) { + throw new Error(`No handler for ${method}`) + } + handler(params, context ?? { clientId: 1, isStale: () => false }) + }, + detachClient(clientId: number) { + for (const listener of detachListeners) { + listener(clientId) + } + } + } +} diff --git a/src/relay/relay-watch-root-capacity-gate.ts b/src/relay/relay-watch-root-capacity-gate.ts new file mode 100644 index 00000000000..4e430332be2 --- /dev/null +++ b/src/relay/relay-watch-root-capacity-gate.ts @@ -0,0 +1,86 @@ +import { + assertRelayWatcherRootCapacity, + exceedsRelayWatcherRootCapacity +} from './relay-watcher-root-capacity' + +type RelayWatchRootTeardowns = { + rootPaths: () => string[] + /** Resolves when every teardown in flight has settled, or undefined when none is. */ + settlePending: () => Promise | undefined +} + +/** + * Decides whether a prospective watch root fits, and waits out an over-cap that only unsubscribing + * roots are causing. + * + * Why waiting beats refusing: a reconnect tears the old roots down as it installs the new ones, so + * the cap is briefly full of slots already promised back. The client answers a capacity refusal + * with a 60s-to-30min dormancy that no release event can shorten, so refusing on a transient + * overlap costs half an hour of blindness. Mirrors WatcherSupervisorCapacityWait. + */ +export class RelayWatchRootCapacityGate { + // Why tracked: a root parked on the wait has been granted nothing, so counting its setup entry + // would let it hold a slot away from the root already reclaiming one. + private readonly waiting = new Set() + + constructor( + private readonly activeRoots: ReadonlyMap, + private readonly setupRoots: ReadonlyMap, + // Thunk: the registry builds its teardown tracker after this field initializes. + private readonly teardowns: () => RelayWatchRootTeardowns + ) {} + + assert(rootKey: string): void { + assertRelayWatcherRootCapacity( + this.activeRoots.keys(), + this.claimedSetupRoots(rootKey), + this.teardowns().rootPaths(), + rootKey + ) + } + + /** + * The wait to hold before {@link assert}, or undefined when there is nothing to wait for. + * + * Undefined rather than a resolved promise so an install that already fits stays synchronous — + * a suspension here would let a concurrent watch of the same root join the setup, not the watch. + */ + release(rootKey: string, signal?: AbortSignal): Promise | undefined { + if ( + !exceedsRelayWatcherRootCapacity( + this.activeRoots.keys(), + this.claimedSetupRoots(rootKey), + this.teardowns().rootPaths(), + rootKey + ) + ) { + return undefined + } + const released = this.teardowns().settlePending() + if (!released) { + return undefined + } + this.waiting.add(rootKey) + // Once, and never past the caller: a genuinely full cap must still reach the refusal that sends + // the client dormant, and an unsubscribe that never settles must not park the request with it. + return (signal ? Promise.race([released, abortSignalSettled(signal)]) : released).finally( + () => { + this.waiting.delete(rootKey) + } + ) + } + + /** Setup roots that currently hold a slot — a parked capacity waiter holds none. */ + private claimedSetupRoots(rootKey: string): string[] { + return [...this.setupRoots.keys()].filter((key) => key === rootKey || !this.waiting.has(key)) + } +} + +/** Resolves (never rejects) when the request is abandoned, so a race can drop out of a wait. */ +function abortSignalSettled(signal: AbortSignal): Promise { + return signal.aborted + ? Promise.resolve() + : new Promise((resolve) => + signal.addEventListener('abort', () => resolve(), { once: true }) + ) +} diff --git a/src/relay/relay-watch-root-capacity.test.ts b/src/relay/relay-watch-root-capacity.test.ts new file mode 100644 index 00000000000..de5a5fc2b4b --- /dev/null +++ b/src/relay/relay-watch-root-capacity.test.ts @@ -0,0 +1,107 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import * as fs from 'node:fs/promises' +import * as path from 'node:path' +import { mkdtempSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { RelayContext } from './context' +import type { RelayDispatcher } from './dispatcher' +import { FsHandler } from './fs-handler' +import { subscribeWithInProcessWatcher } from '../main/ipc/parcel-watcher-in-process-fallback' +import { createMockDispatcher } from './relay-fs-test-dispatcher' + +const { mockSubscribe } = vi.hoisted(() => ({ + mockSubscribe: vi.fn() +})) + +vi.mock('@parcel/watcher', () => ({ + subscribe: mockSubscribe +})) + +describe('relay watch-root capacity', () => { + let dispatcher: ReturnType + let handler: FsHandler + let tmpDir: string + + beforeEach(() => { + mockSubscribe.mockReset() + mockSubscribe.mockResolvedValue({ unsubscribe: vi.fn() }) + tmpDir = mkdtempSync(path.join(tmpdir(), 'relay-fs-cap-')) + dispatcher = createMockDispatcher() + handler = new FsHandler(dispatcher as unknown as RelayDispatcher, new RelayContext(), { + dispose: vi.fn(), + forgetRoot: vi.fn(), + subscribe: subscribeWithInProcessWatcher + }) + }) + + afterEach(async () => { + handler.dispose() + await fs.rm(tmpDir, { recursive: true, force: true }) + }) + + it('blocks replacement watches behind physical unsubscribe and counts the pending slot', async () => { + let resolveUnsubscribe: () => void = () => {} + const unsubscribe = vi.fn( + () => + new Promise((resolve) => { + resolveUnsubscribe = resolve + }) + ) + mockSubscribe.mockResolvedValue({ unsubscribe }) + await dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) + dispatcher.callNotification('fs.unwatch', { rootPath: tmpDir }) + + const replacement = dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) + for (let index = 0; index < 19; index += 1) { + await dispatcher.callRequest('fs.watch', { + rootPath: path.join(tmpDir, `pending-cap-${index}`) + }) + } + // The replacement claims the slot the teardown releases, so this cap is genuinely full: the + // request waits for the release event and is still refused once it has happened. + const overCap = dispatcher + .callRequest('fs.watch', { rootPath: path.join(tmpDir, 'over-pending-cap') }) + .then( + () => null, + (error: Error) => error + ) + expect(mockSubscribe).toHaveBeenCalledTimes(20) + + resolveUnsubscribe() + await replacement + expect(await overCap).toMatchObject({ message: 'Maximum number of file watchers reached' }) + expect(mockSubscribe).toHaveBeenCalledTimes(21) + }) + + it('waits out a teardown that frees a slot instead of refusing on it', async () => { + let resolveUnsubscribe: () => void = () => {} + mockSubscribe.mockResolvedValue({ + unsubscribe: vi.fn( + () => + new Promise((resolve) => { + resolveUnsubscribe = resolve + }) + ) + }) + for (let index = 0; index < 20; index += 1) { + await dispatcher.callRequest('fs.watch', { rootPath: path.join(tmpDir, `full-${index}`) }) + } + dispatcher.callNotification('fs.unwatch', { rootPath: path.join(tmpDir, 'full-0') }) + + // Why not a refusal: the slot is already promised back, and the client answers a capacity + // refusal with a 60s-to-30min dormancy that no release event can shorten. + let settled = false + const fresh = dispatcher + .callRequest('fs.watch', { rootPath: path.join(tmpDir, 'fresh') }) + .then(() => { + settled = true + }) + await Promise.resolve() + expect(settled).toBe(false) + expect(mockSubscribe).toHaveBeenCalledTimes(20) + + resolveUnsubscribe() + await fresh + expect(mockSubscribe).toHaveBeenCalledTimes(21) + }) +}) diff --git a/src/relay/relay-watcher-root-capacity.ts b/src/relay/relay-watcher-root-capacity.ts index 025462e7bd3..35178166f7c 100644 --- a/src/relay/relay-watcher-root-capacity.ts +++ b/src/relay/relay-watcher-root-capacity.ts @@ -1,14 +1,26 @@ +import { WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE } from '../shared/watch-root-capacity-refusal' + const MAX_RELAY_WATCH_ROOTS = 20 +// Why teardown roots count: a root still unsubscribing owns its native handles until it settles. +export function exceedsRelayWatcherRootCapacity( + activeRoots: Iterable, + pendingRoots: Iterable, + teardownRoots: Iterable, + prospectiveRoot: string +): boolean { + const physicalRoots = new Set([...activeRoots, ...pendingRoots, ...teardownRoots]) + physicalRoots.add(prospectiveRoot) + return physicalRoots.size > MAX_RELAY_WATCH_ROOTS +} + export function assertRelayWatcherRootCapacity( activeRoots: Iterable, pendingRoots: Iterable, teardownRoots: Iterable, prospectiveRoot: string ): void { - const physicalRoots = new Set([...activeRoots, ...pendingRoots, ...teardownRoots]) - physicalRoots.add(prospectiveRoot) - if (physicalRoots.size > MAX_RELAY_WATCH_ROOTS) { - throw new Error('Maximum number of file watchers reached') + if (exceedsRelayWatcherRootCapacity(activeRoots, pendingRoots, teardownRoots, prospectiveRoot)) { + throw new Error(WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE) } } diff --git a/src/relay/relay-watcher-teardown-tracker.ts b/src/relay/relay-watcher-teardown-tracker.ts index 91b1460054d..785e4dd16ea 100644 --- a/src/relay/relay-watcher-teardown-tracker.ts +++ b/src/relay/relay-watcher-teardown-tracker.ts @@ -98,6 +98,17 @@ export class RelayWatcherTeardownTracker { rootPaths(): string[] { return [...this.pending.keys(), ...this.failed.keys()] } + + /** + * The capacity-release event: resolves once every teardown in flight right now has settled. + * + * `undefined` when nothing is unsubscribing, which is the only honest answer to "could a slot + * still come back?" — a failed teardown keeps its handles and releases nothing. + */ + settlePending(): Promise | undefined { + const inFlight = [...this.pending.values()] + return inFlight.length === 0 ? undefined : Promise.allSettled(inFlight).then(() => undefined) + } } function callUnsubscribe(subscription: WatcherProcessSubscription): Promise { diff --git a/src/shared/watch-root-capacity-refusal.ts b/src/shared/watch-root-capacity-refusal.ts new file mode 100644 index 00000000000..234fecb4ef7 --- /dev/null +++ b/src/shared/watch-root-capacity-refusal.ts @@ -0,0 +1,9 @@ +// Why a shared string rather than an error code: the refusal crosses the relay wire as a JSON-RPC +// error message, and relays deploy independently of clients. Both sides must spell it the same way, +// and a client that does not recognise it simply falls back to the ordinary unavailable handling. +export const WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE = 'Maximum number of file watchers reached' + +export function isWatchRootCapacityRefusal(error: unknown): boolean { + const message = (error as { message?: unknown } | null | undefined)?.message + return typeof message === 'string' && message.includes(WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE) +} From 07e1f953a522081f9ff3be95933f2834f5eb3d0e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:25 -0700 Subject: [PATCH 05/77] fix(ssh): log an unanswered native-deps probe instead of launching silently (#18000) * fix(ssh): log an unanswered native-deps probe instead of launching silently The wrongful rebuild used to be the only visible symptom of a dropped exec channel; #17979 removed it, so a real transport failure now leaves no trace. Matches the install-path sibling, whose callers log the same class of failure. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ssh/ssh-relay-deploy.ts | 9 ++++++++- src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts | 5 +++++ 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index af7346f5d35..dda85057205 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -869,10 +869,17 @@ async function probeRequiredNativeDeps( return { status: 'unverifiable', missing: [] } } return { status: 'blocked', missing } - } catch { + } catch (error) { signal?.throwIfAborted() // Why: an unanswered probe says nothing about the deps; reporting MISSING here reset and // recompiled healthy relays, turning one dropped exec channel into a multi-minute reconnect. + // Why: the wrongful rebuild was the only visible symptom, so without this line a dropped exec + // channel leaves no trace at all. + console.warn( + `[ssh-relay] Native deps probe unanswered at ${remoteDir}; treating as unverifiable: ${ + error instanceof Error ? error.message : String(error) + }` + ) return { status: 'unverifiable', missing: [] } } } diff --git a/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts b/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts index 3939c954f65..cb8f8c39c7b 100644 --- a/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts @@ -153,6 +153,11 @@ describe('native-deps repair probe verdicts', () => { expect(warnings().some((message) => message.includes('Repairing missing native deps'))).toBe( false ) + // Why: the wrongful rebuild used to be the only visible symptom of a dropped exec channel. + expect( + warnings().some((message) => message.includes('Native deps probe unanswered')), + 'an unanswered probe must still leave a trace' + ).toBe(true) expect(commands.some((command) => command.includes(NODE_PTY_RESET))).toBe(false) expect(commands.some((command) => command.includes(WATCHER_RESET))).toBe(false) expect(commands.some((command) => command.includes('npm install'))).toBe(false) From 7458c3918167479568b5b29460207e6f0432c746 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:33 -0700 Subject: [PATCH 06/77] fix(ssh): declare a wedged relay link lost, and stop reading silence as a verdict (#17817) * fix(ssh): declare a wedged relay link lost instead of suppressing the dead-link check * fix(ssh): make the Windows deps probe exit 0 on a real load failure, like its POSIX twin * fix(relay): reap a client that has stopped answering instead of holding its leases forever * test(relay): feed the primary before asserting the reaper exemption holds * fix(ssh): keep a lost link's verdict unverifiable instead of reporting absence * refactor(ssh): read the exec timeout from its typed code, not the message text * fix(relay): bound a client that clears the handshake and then never frames anything * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ai-vault/ssh-session-list.test.ts | 23 ++- src/main/ai-vault/ssh-session-list.ts | 4 +- ...annel-multiplexer-saturation-wedge.test.ts | 82 +++++++++ src/main/ssh/ssh-channel-multiplexer.test.ts | 20 ++- src/main/ssh/ssh-channel-multiplexer.ts | 27 ++- src/main/ssh/ssh-relay-deploy.ts | 2 +- src/main/ssh/ssh-relay-exec-command.ts | 19 +- .../ssh/ssh-relay-native-deps-install.test.ts | 25 +++ .../ssh/ssh-remote-platform-detection.test.ts | 27 +++ src/main/ssh/ssh-remote-platform-detection.ts | 5 +- .../ssh/ssh-request-outcome-verdict.test.ts | 34 ++++ .../commit-message-model-discovery.ts | 4 +- ...ge-text-generation-model-discovery.test.ts | 25 ++- ...e-text-generation-remote-execution.test.ts | 39 ++++- .../source-control-remote-generation.ts | 4 +- src/relay/dispatcher-client-lifecycle.ts | 72 +++++++- src/relay/dispatcher-contract.ts | 12 ++ src/relay/dispatcher-frame-codec.ts | 3 + .../dispatcher-silent-client-reaper.test.ts | 163 ++++++++++++++++++ 19 files changed, 560 insertions(+), 30 deletions(-) create mode 100644 src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts create mode 100644 src/main/ssh/ssh-request-outcome-verdict.test.ts create mode 100644 src/relay/dispatcher-silent-client-reaper.test.ts diff --git a/src/main/ai-vault/ssh-session-list.test.ts b/src/main/ai-vault/ssh-session-list.test.ts index 4d1abd483c3..52454dda966 100644 --- a/src/main/ai-vault/ssh-session-list.test.ts +++ b/src/main/ai-vault/ssh-session-list.test.ts @@ -1,6 +1,9 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' import type { AiVaultListResult, AiVaultSession } from '../../shared/ai-vault-types' -import { SSH_MUX_REQUEST_TIMEOUT_CODE } from '../ssh/ssh-channel-multiplexer' +import { + createSshDisposalError, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from '../ssh/ssh-channel-multiplexer' const requestActiveSshAiVaultSessionList = vi.fn() const getActiveSshAiVaultHostInfo = vi.fn() @@ -144,6 +147,24 @@ describe('scanSshAiVaultSessions', () => { ]) }) + it('reports a host issue when the relay link was declared lost on a real scan budget', async () => { + // Declaring a wedged link lost trades SSH_MUX_REQUEST_TIMEOUT for CONNECTION_LOST on this leg. + // Both are unverifiable, so both must surface as a host issue rather than falling through to a + // crawl that would publish an authoritative-looking empty list + // (docs/reference/ssh-execution-boundary.md). + requestActiveSshAiVaultSessionList.mockRejectedValue(createSshDisposalError('connection_lost')) + + const result = await scanSshAiVaultSessions('dev-box', undefined, { + timeoutMs: 20_000, + relayTimeoutMs: 15_000 + }) + + expect(scanRemoteAiVaultSessions).not.toHaveBeenCalled() + expect(result.issues).toEqual([ + expect.objectContaining({ executionHostId: 'ssh:dev-box', kind: 'host' }) + ]) + }) + it('still falls back when the relay budget was too short for a fair attempt', async () => { requestActiveSshAiVaultSessionList.mockRejectedValue(relayTimeoutError()) scanRemoteAiVaultSessions.mockResolvedValue({ diff --git a/src/main/ai-vault/ssh-session-list.ts b/src/main/ai-vault/ssh-session-list.ts index 7a02883f812..79359873d69 100644 --- a/src/main/ai-vault/ssh-session-list.ts +++ b/src/main/ai-vault/ssh-session-list.ts @@ -10,7 +10,7 @@ import { SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE } from '../providers/ssh-filesystem-dispatch' import { getActiveSshAiVaultHostInfo, requestActiveSshAiVaultSessionList } from '../ipc/ssh' -import { isSshMuxRequestTimeoutError } from '../ssh/ssh-channel-multiplexer' +import { isSshRequestOutcomeUnverifiable } from '../ssh/ssh-channel-multiplexer' import { createAiVaultScanCancelledError } from './ai-vault-scan-cancellation' import { scanRemoteAiVaultSessions } from './remote-session-scanner' import { parseAiVaultListResult } from './session-list-result-validation' @@ -84,7 +84,7 @@ async function scanOneSshHost( throw error } if ( - isSshMuxRequestTimeoutError(error) && + isSshRequestOutcomeUnverifiable(error) && (relayTimeoutMs === undefined || relayTimeoutMs >= MEANINGFUL_RELAY_SCAN_ATTEMPT_MS) ) { return sshScanIssueResult(executionHostId, targetId, errorMessage(error)) diff --git a/src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts b/src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts new file mode 100644 index 00000000000..fd9034f27f9 --- /dev/null +++ b/src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts @@ -0,0 +1,82 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { encodeKeepAliveFrame, KEEPALIVE_SEND_MS, TIMEOUT_MS } from './relay-protocol' +import { SshChannelMultiplexer, type MultiplexerTransport } from './ssh-channel-multiplexer' + +type WedgedTransport = MultiplexerTransport & { writes: Buffer[]; feed: (chunk: Buffer) => void } + +/** + * A transport that accepts the first write, then reports backpressure forever: no drain, and no + * write settlement. This is a half-open TCP link — the socket buffer filled and the peer's FIN + * never arrived — which is what sleep/resume and a dropped NAT mapping produce in the field. + */ +function createWedgedTransport(): WedgedTransport { + const writes: Buffer[] = [] + let onData: (chunk: Buffer) => void = () => {} + return { + write: (data) => { + writes.push(data) + return false + }, + onData: (callback) => { + onData = callback + }, + onClose: () => {}, + onDrain: () => () => {}, + supportsWriteSettlement: true, + writes, + feed: (chunk) => onData(chunk) + } +} + +describe('SshChannelMultiplexer on a transport that saturates and never drains', () => { + let transport: WedgedTransport + let mux: SshChannelMultiplexer + + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(0) + transport = createWedgedTransport() + mux = new SshChannelMultiplexer(transport) + }) + + afterEach(() => { + mux.dispose() + vi.restoreAllMocks() + vi.useRealTimers() + }) + + it('declares the link lost instead of suppressing the dead-link check forever', async () => { + // Drive well past every health window: keepalive interval, dead-link timeout, and the + // wake-gap grace that resets staleness after a suspend. + await vi.advanceTimersByTimeAsync(TIMEOUT_MS * 10 + KEEPALIVE_SEND_MS) + + expect(mux.isDisposed()).toBe(true) + }) + + it('fails a request parked behind saturation rather than leaving it pending forever', async () => { + const settled = vi.fn() + mux.request('pty.spawn', {}).then( + () => settled('resolved'), + () => settled('rejected') + ) + + await vi.advanceTimersByTimeAsync(TIMEOUT_MS * 10 + KEEPALIVE_SEND_MS) + + expect(settled).toHaveBeenCalledWith('rejected') + }) + + it('keeps a slow-but-alive peer connected while its own keepalives arrive', async () => { + // The regression guard for the fix above: backpressure on our uplink is not evidence of + // death, and the relay's own keepalive is what proves it. + let seq = 1 + const inbound = setInterval(() => { + transport.feed(encodeKeepAliveFrame(seq++, 0)) + }, KEEPALIVE_SEND_MS) + try { + await vi.advanceTimersByTimeAsync(TIMEOUT_MS * 10 + KEEPALIVE_SEND_MS) + expect(mux.isDisposed()).toBe(false) + } finally { + clearInterval(inbound) + } + }) +}) diff --git a/src/main/ssh/ssh-channel-multiplexer.test.ts b/src/main/ssh/ssh-channel-multiplexer.test.ts index eeca0ce5108..c1c2e960439 100644 --- a/src/main/ssh/ssh-channel-multiplexer.test.ts +++ b/src/main/ssh/ssh-channel-multiplexer.test.ts @@ -71,7 +71,6 @@ type MuxInternals = { disposeHandlers: unknown[] lastReceivedAt: number unackedTimestamps: Map - writerSaturated: boolean } function getMuxInternals(instance: SshChannelMultiplexer): MuxInternals { @@ -354,9 +353,10 @@ describe('SshChannelMultiplexer', () => { expect(mux.isDisposed()).toBe(true) }) - it('suppresses false death while locally saturated and rebases both clocks on drain', () => { + it('survives local saturation while the peer keeps talking, and rebases both clocks on drain', () => { mux.dispose() let drain = (): void => {} + let feed: (chunk: Buffer) => void = () => {} const written: Buffer[] = [] const saturatedTransport: MultiplexerTransport = { write: (data) => { @@ -367,21 +367,29 @@ describe('SshChannelMultiplexer', () => { onDrain: (callback) => { drain = callback }, - onData: vi.fn(), + onData: (callback) => { + feed = callback + }, onClose: vi.fn() } mux = new SshChannelMultiplexer(saturatedTransport) vi.advanceTimersByTime(5_000) - expect(getMuxInternals(mux).writerSaturated).toBe(true) - vi.advanceTimersByTime(25_000) + // The writer parked after its first frame: that is the saturation this test is about. + expect(written).toHaveLength(1) + // Why: backpressure on our uplink is not evidence of death. The relay's own keepalive is, + // and only that inbound traffic may keep the link alive — suppressing the check on + // saturation alone wedged a half-open link forever (see the saturation-wedge suite). + for (let tick = 0; tick < 5; tick++) { + feed(encodeKeepAliveFrame(0, 0)) + vi.advanceTimersByTime(5_000) + } expect(mux.isDisposed()).toBe(false) expect(written).toHaveLength(1) drain() const resumedAt = Date.now() const internals = getMuxInternals(mux) - expect(internals.writerSaturated).toBe(false) expect(internals.lastReceivedAt).toBe(resumedAt) expect(new Set(internals.unackedTimestamps.values())).toEqual(new Set([resumedAt])) diff --git a/src/main/ssh/ssh-channel-multiplexer.ts b/src/main/ssh/ssh-channel-multiplexer.ts index d87a7ad8f5b..a8443f86f88 100644 --- a/src/main/ssh/ssh-channel-multiplexer.ts +++ b/src/main/ssh/ssh-channel-multiplexer.ts @@ -74,11 +74,19 @@ function sshMuxRequestTimeoutError(method: string, timeoutMs: number): Error { }) } -export function isSshMuxRequestTimeoutError(error: unknown): boolean { - return ( - error instanceof Error && - (error as Error & { code?: unknown }).code === SSH_MUX_REQUEST_TIMEOUT_CODE - ) +/** + * True when a request may have run on the host despite failing here. + * + * A response deadline and a link declared lost are the same verdict: the frame reached the wire and + * the peer's answer did not come back, so the work is `unverifiable`, never absent. Declaring a + * wedged link lost at TIMEOUT_MS turned what used to surface as SSH_MUX_REQUEST_TIMEOUT into + * CONNECTION_LOST, so callers that phrase the verdict to a user must branch on this rather than on + * the timeout alone or they silently start reporting absence + * (docs/reference/ssh-execution-boundary.md). + */ +export function isSshRequestOutcomeUnverifiable(error: unknown): boolean { + const code = error instanceof Error ? (error as Error & { code?: unknown }).code : undefined + return code === SSH_MUX_REQUEST_TIMEOUT_CODE || code === 'CONNECTION_LOST' } export class SshChannelMultiplexer { @@ -102,7 +110,6 @@ export class SshChannelMultiplexer { private disposed = false private disposeReason: 'shutdown' | 'connection_lost' | null = null private decoderReadPaused = false - private writerSaturated = false // Track the oldest unacked outgoing message timestamp private unackedTimestamps = new Map() @@ -587,7 +594,12 @@ export class SshChannelMultiplexer { this.sendKeepAlive() - if (this.disposed || resumedAfterWake || this.decoderReadPaused || this.writerSaturated) { + // Why: a saturated writer used to suppress this check outright, which wedged a half-open + // link forever — no drain, so no frame ever left, and the writer's single-outstanding + // liveness guard silenced the one probe that could have noticed. The relay sends its own + // keepalive every KEEPALIVE_SEND_MS, so a slow-but-alive peer still refreshes + // lastReceivedAt; only a link that delivers nothing inbound is declared lost. + if (this.disposed || resumedAfterWake || this.decoderReadPaused) { return } @@ -650,7 +662,6 @@ export class SshChannelMultiplexer { } private handleWriterSaturationChange(saturated: boolean): void { - this.writerSaturated = saturated if (!saturated && !this.disposed) { this.rebaseHealthClocks(Date.now()) } diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index dda85057205..c5cb2dd552c 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -841,7 +841,7 @@ async function probeRequiredNativeDeps( hostPlatform, nodePath, remoteDir, - `try { & ${powerShellLiteral(nodePath)} -e ${powerShellNativeArg(probeJs)} } catch { 'MISSING' }` + `try { & ${powerShellLiteral(nodePath)} -e ${powerShellNativeArg(probeJs)}; if ($LASTEXITCODE -ne 0) { 'MISSING' } } catch { 'MISSING' }` ) : // Why: no `2>/dev/null` — it discarded the only line that says why node never reached the // script. stderr stays its own stream so it can't be mistaken for the verdict, mirroring diff --git a/src/main/ssh/ssh-relay-exec-command.ts b/src/main/ssh/ssh-relay-exec-command.ts index fb0a4fee012..bec41c8d43e 100644 --- a/src/main/ssh/ssh-relay-exec-command.ts +++ b/src/main/ssh/ssh-relay-exec-command.ts @@ -24,6 +24,16 @@ type SshCommandTerminationError = Error & { sshChannelCloseConfirmed: boolean } +// Why: callers must tell "the host answered no" from "the host never answered". Matching the +// message text is what let an unanswered probe be read as a definitive negative. +export const SSH_EXEC_TIMEOUT_CODE = 'SSH_EXEC_TIMEOUT' + +export function isSshExecTimeout(error: unknown): boolean { + return ( + error instanceof Error && (error as Partial<{ code: string }>).code === SSH_EXEC_TIMEOUT_CODE + ) +} + export function isUnconfirmedSshCommandTermination( error: unknown ): error is SshCommandTerminationError { @@ -164,8 +174,13 @@ export async function execCommand( } const timeout = setTimeout(() => { requestTermination( - new Error( - `Command "${redactRelayInstallMarkerTokens(command)}" timed out after ${timeoutMs / 1000}s` + Object.assign( + new Error( + `Command "${redactRelayInstallMarkerTokens(command)}" timed out after ${ + timeoutMs / 1000 + }s` + ), + { code: SSH_EXEC_TIMEOUT_CODE } ) ) }, timeoutMs) diff --git a/src/main/ssh/ssh-relay-native-deps-install.test.ts b/src/main/ssh/ssh-relay-native-deps-install.test.ts index 01625816170..edf19b1251b 100644 --- a/src/main/ssh/ssh-relay-native-deps-install.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-install.test.ts @@ -394,6 +394,31 @@ describe('installNativeDeps (via deployAndLaunchRelay)', () => { } }) + it('does not rewrite node_modules when the health probe never answered', async () => { + // Why (#14830): a wedged `require("node-pty")` makes the probe time out. Reading that silence + // as "every native dep is missing" sent a healthy install through npm install + rebuild that + // could not help, and the retry loop burned the whole deploy budget at "Deploying relay…". + const conn = makeMockConnection(sftpCapture) + vi.mocked(isRelayAlreadyInstalled).mockResolvedValue(true) + feed([ + '__ORCA_REMOTE_PLATFORM__ Linux x86_64', + '/home/u', + { reject: 'Command "node -e ..." timed out after 30s' } // health probe never answered + ]) + + await deployAndLaunchRelay(conn).catch(() => {}) + + const execCalls = vi.mocked(execCommand).mock.calls.map(([, c]) => c) + expect(execCalls.some((c) => c.includes('npm install'))).toBe(false) + expect(execCalls.some((c) => c.includes('npm rebuild'))).toBe(false) + + const warnMessages = warnSpy.mock.calls.map((args) => String(args[0] ?? '')) + expect(warnMessages.some((m) => m.includes('Repairing missing native deps'))).toBe(false) + // Why no log assertion: the behavioural claim above is the real one. Asserting on warn text + // pinned wording that main's landed probe verdict does not use, and #18000 adds its own. + expect(execCalls.some((c) => c.includes("rm -rf 'node_modules/node-pty'"))).toBe(false) + }) + it('lets a probe SSH-channel failure bubble up rather than silently mapping to MISSING', async () => { const conn = makeMockConnection(sftpCapture) feed( diff --git a/src/main/ssh/ssh-remote-platform-detection.test.ts b/src/main/ssh/ssh-remote-platform-detection.test.ts index aa3fb3a0916..a69f01d04b4 100644 --- a/src/main/ssh/ssh-remote-platform-detection.test.ts +++ b/src/main/ssh/ssh-remote-platform-detection.test.ts @@ -1,5 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { SshConnection } from './ssh-connection' +import { SSH_EXEC_TIMEOUT_CODE } from './ssh-relay-exec-command' const execCommandMock = vi.hoisted(() => vi.fn()) @@ -226,6 +227,32 @@ describe('detectRemoteHostPlatform failure reporting', () => { expect((error as Error).cause).toMatchObject({ reason: 2 }) }) + it('does not call an unmappable uname unsupported when the PowerShell probe timed out', async () => { + // The timed-out channel is the better explanation, and it is identified by execCommand's typed + // code — matching "timed out after Ns" in the message let any look-alike claim the branch. + execCommandMock + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ CYGWIN_NT-10.0 x86_64\n') + .mockRejectedValueOnce( + Object.assign(new Error('Command "pwsh" timed out after 30s'), { + code: SSH_EXEC_TIMEOUT_CODE + }) + ) + + const error = await detectRemoteHostPlatform(conn).catch((err: unknown) => err) + + expect(error).toBeInstanceOf(Error) + expect((error as Error).message).not.toMatch(/unsupported/iu) + expect((error as Error).cause).toMatchObject({ code: SSH_EXEC_TIMEOUT_CODE }) + }) + + it('does not read a look-alike timeout message as a timed-out channel', async () => { + execCommandMock + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ FreeBSD x86_64\n') + .mockRejectedValueOnce(new Error('the agent it launched timed out after 30s')) + + await expect(detectRemoteHostPlatform(conn)).resolves.toBeNull() + }) + it('falls through to PowerShell for a Cygwin uname it cannot map', async () => { execCommandMock .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ CYGWIN_NT-10.0 x86_64\n') diff --git a/src/main/ssh/ssh-remote-platform-detection.ts b/src/main/ssh/ssh-remote-platform-detection.ts index 5f830fb7449..6fd0f87c767 100644 --- a/src/main/ssh/ssh-remote-platform-detection.ts +++ b/src/main/ssh/ssh-remote-platform-detection.ts @@ -5,7 +5,7 @@ import { } from '../../shared/process-output-field-scanner' import { parseUnameToRelayPlatform, type RelayPlatform } from './relay-protocol' import { execCommand } from './ssh-relay-deploy-helpers' -import { isUnconfirmedSshCommandTermination } from './ssh-relay-exec-command' +import { isSshExecTimeout, isUnconfirmedSshCommandTermination } from './ssh-relay-exec-command' import { isSshSessionLimitError } from './ssh-session-limit-error' import { getRemoteHostPlatform, type RemoteHostPlatform } from './ssh-remote-platform' import { powerShellCommand } from './ssh-remote-powershell' @@ -14,7 +14,6 @@ const PLATFORM_PROBE_MARKER = '__ORCA_REMOTE_PLATFORM__' const MAX_UNAME_FIELD_CHARS = 64 const MAX_THROWN_OUTPUT_CHARS = 200 const MAX_LOGGED_OUTPUT_CHARS = 1000 -const EXEC_TIMEOUT_MESSAGE = /timed out after \d+s$/u type PlatformProbeOutcome = | { kind: 'detected'; platform: RelayPlatform } @@ -89,7 +88,7 @@ function isTransportShapedError(error: unknown): boolean { return ( isSshSessionLimitError(error) || isUnconfirmedSshCommandTermination(error) || - (error instanceof Error && EXEC_TIMEOUT_MESSAGE.test(error.message)) + isSshExecTimeout(error) ) } diff --git a/src/main/ssh/ssh-request-outcome-verdict.test.ts b/src/main/ssh/ssh-request-outcome-verdict.test.ts new file mode 100644 index 00000000000..e21368d2559 --- /dev/null +++ b/src/main/ssh/ssh-request-outcome-verdict.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from 'vitest' +import { + createSshDisposalError, + isSshRequestOutcomeUnverifiable, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from './ssh-channel-multiplexer' + +// docs/reference/ssh-execution-boundary.md: the vocabulary is live / unverifiable / exited, and +// loss of contact is never evidence of absence. Three call sites phrase this verdict to a user, so +// collapsing "unverifiable" into "could not be reached" is a user-visible lie. +describe('SSH request outcome verdict', () => { + it('treats a response deadline as unverifiable', () => { + const timedOut = Object.assign(new Error('Request "x" timed out after 30000ms'), { + code: SSH_MUX_REQUEST_TIMEOUT_CODE + }) + expect(isSshRequestOutcomeUnverifiable(timedOut)).toBe(true) + }) + + it('treats a link declared lost as unverifiable, not as absence', () => { + // The regression this exists for: declaring a wedged link lost at TIMEOUT_MS made those + // requests surface CONNECTION_LOST where they used to surface a timeout, silently downgrading + // the honest "may still be running on the remote host" to "could not be reached". + expect(isSshRequestOutcomeUnverifiable(createSshDisposalError('connection_lost'))).toBe(true) + }) + + it('does not claim unverifiable for a deliberate shutdown', () => { + expect(isSshRequestOutcomeUnverifiable(createSshDisposalError('shutdown'))).toBe(false) + }) + + it('does not claim unverifiable for an ordinary failure', () => { + expect(isSshRequestOutcomeUnverifiable(new Error('boom'))).toBe(false) + expect(isSshRequestOutcomeUnverifiable(undefined)).toBe(false) + }) +}) diff --git a/src/main/text-generation/commit-message-model-discovery.ts b/src/main/text-generation/commit-message-model-discovery.ts index 43873f01bf2..ec6be0f12bb 100644 --- a/src/main/text-generation/commit-message-model-discovery.ts +++ b/src/main/text-generation/commit-message-model-discovery.ts @@ -3,7 +3,7 @@ import type { CommitMessagePlan } from '../../shared/commit-message-plan' import { getAgentModelProbeSpec } from '../../shared/agent-model-probe-spec' import type { TuiAgent } from '../../shared/tui-agent' import { resolveCodexHomeProcessLockKeyForSpawnEnv } from '../codex-cli/codex-home-process-lock' -import { isSshMuxRequestTimeoutError } from '../ssh/ssh-channel-multiplexer' +import { isSshRequestOutcomeUnverifiable } from '../ssh/ssh-channel-multiplexer' import { WINDOWS_BATCH_UNSAFE_ARGUMENTS_ERROR } from '../win32-utils' import { finalizeModelDiscoveryOutput, @@ -206,7 +206,7 @@ export async function discoverModelsRemote(input: { console.error('[commit-message] Remote model discovery request failed:', error) return { success: false, - error: isSshMuxRequestTimeoutError(error) + error: isSshRequestOutcomeUnverifiable(error) ? `${spec.label} model discovery took longer than ${SOURCE_CONTROL_GENERATION_TIMEOUT_MS / 1000}s and may still be running on the remote host.` : `${spec.label} model discovery could not be reached on the remote PATH. Try again after the SSH connection recovers.` } diff --git a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts index 7ef438c06ae..4bc1a720f05 100644 --- a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts +++ b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts @@ -1,7 +1,10 @@ import { spawn } from 'node:child_process' import type * as ChildProcess from 'node:child_process' import { beforeEach, describe, expect, it, vi } from 'vitest' -import { SSH_MUX_REQUEST_TIMEOUT_CODE } from '../ssh/ssh-channel-multiplexer' +import { + createSshDisposalError, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from '../ssh/ssh-channel-multiplexer' import { discoverCommitMessageModelsLocal, discoverCommitMessageModelsRemote @@ -475,6 +478,26 @@ describe('generateCommitMessageFromContext', () => { }) }) + it('keeps the unverifiable wording when the link is declared lost instead of timing out', async () => { + // Same regression as the exec leg: a wedged link now disposes the mux before the response + // deadline, so this branch sees CONNECTION_LOST. Reporting "could not be reached" for it + // asserts absence the client never observed (docs/reference/ssh-execution-boundary.md). + const result = await discoverCommitMessageModelsRemote( + 'cursor', + '/remote/repo', + async () => { + throw createSshDisposalError('connection_lost') + }, + 'npx cursor-agent' + ) + + expect(result).toEqual({ + success: false, + error: + 'Cursor model discovery took longer than 60s and may still be running on the remote host.' + }) + }) + it('reports remote model discovery spawn failures with remote install guidance', async () => { const result = await discoverCommitMessageModelsRemote('cursor', '/remote/repo', async () => ({ stdout: '', diff --git a/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts b/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts index 10442b12f0e..26708149d51 100644 --- a/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts +++ b/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from 'vitest' -import { SSH_MUX_REQUEST_TIMEOUT_CODE } from '../ssh/ssh-channel-multiplexer' +import { + createSshDisposalError, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from '../ssh/ssh-channel-multiplexer' import { generateCommitMessageFromContext } from './commit-message-text-generation' describe('generateCommitMessageFromContext', () => { @@ -76,6 +79,40 @@ describe('generateCommitMessageFromContext', () => { }) }) + it('keeps the unverifiable wording when the link is declared lost instead of timing out', async () => { + // Declaring a wedged link lost disposes the mux before the 30s response deadline, so this leg + // now sees CONNECTION_LOST where it used to see SSH_MUX_REQUEST_TIMEOUT. Both mean the frame + // reached the wire and no answer came back, so both must keep "may still be running" — falling + // through to "could not be reached" asserts absence the client cannot observe + // (docs/reference/ssh-execution-boundary.md). + const result = await generateCommitMessageFromContext( + { + branch: 'main', + stagedSummary: 'M\tREADME.md', + stagedPatch: '+hello' + }, + { + agentId: 'custom', + model: '', + customAgentCommand: 'agent' + }, + { + kind: 'remote', + cwd: '/repo', + missingBinaryLocation: 'remote PATH', + execute: async () => { + throw createSshDisposalError('connection_lost') + } + } + ) + + expect(result).toEqual({ + success: false, + error: 'agent took longer than 60s to respond and may still be running on the remote host.', + canceled: undefined + }) + }) + it('sanitizes remote execution transport failures', async () => { const result = await generateCommitMessageFromContext( { diff --git a/src/main/text-generation/source-control-remote-generation.ts b/src/main/text-generation/source-control-remote-generation.ts index 5fe18d34718..ece1a6de1d8 100644 --- a/src/main/text-generation/source-control-remote-generation.ts +++ b/src/main/text-generation/source-control-remote-generation.ts @@ -1,5 +1,5 @@ import type { CommitMessagePlan } from '../../shared/commit-message-plan' -import { isSshMuxRequestTimeoutError } from '../ssh/ssh-channel-multiplexer' +import { isSshRequestOutcomeUnverifiable } from '../ssh/ssh-channel-multiplexer' import { WINDOWS_BATCH_UNSAFE_ARGUMENTS_ERROR } from '../win32-utils' import { finalizeFromAgentOutput, @@ -25,7 +25,7 @@ export async function runRemoteSourceControlPlan(input: { result = await target.execute(plan, target.cwd, SOURCE_CONTROL_GENERATION_TIMEOUT_MS, operation) } catch (error) { console.error('[commit-message] Remote generator request failed:', error) - if (isSshMuxRequestTimeoutError(error)) { + if (isSshRequestOutcomeUnverifiable(error)) { return { success: false, error: `${plan.label} took longer than ${SOURCE_CONTROL_GENERATION_TIMEOUT_MS / 1000}s to respond and may still be running on the remote host.` diff --git a/src/relay/dispatcher-client-lifecycle.ts b/src/relay/dispatcher-client-lifecycle.ts index 0aff8d977ac..672c51763e6 100644 --- a/src/relay/dispatcher-client-lifecycle.ts +++ b/src/relay/dispatcher-client-lifecycle.ts @@ -1,4 +1,4 @@ -import { FrameDecoder, KEEPALIVE_SEND_MS, encodeKeepAliveFrame } from './protocol' +import { FrameDecoder, KEEPALIVE_SEND_MS, TIMEOUT_MS, encodeKeepAliveFrame } from './protocol' import type { PtyConsumerCloseCause } from '../shared/pty-consumer-session-contract' import type { DispatcherClientWriter, @@ -12,6 +12,12 @@ import type { } from './dispatcher-contract' import { RelayDispatcherClientState } from './dispatcher-client-state' +// Why: a client that clears the endpoint handshake and then never frames anything is invisible to +// the silence window -- it has no lastReceivedAt to go stale -- so its socket, writer and client +// entry are held for the life of the relay. The bound is deliberately several windows wide: a real +// client frames immediately after the handshake, so only a peer that is already gone reaches it. +const SILENT_CONNECT_TIMEOUT_MS = TIMEOUT_MS * 6 + export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClientState { // Why: redirect outgoing frames to the reconnected socket without rebuilding the dispatcher + handler tree. // Why: a new multiplexer restarts at seq=1; reset state to avoid stalled acknowledgements. @@ -122,6 +128,9 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie bulkChain: Promise.resolve(), nextOutgoingSeq: 1, highestReceivedSeq: 0, + attachedAt: Date.now(), + lastReceivedAt: null, + keepaliveObserved: false, generation: 0, closed: false, droppedNotificationLog: null, @@ -144,6 +153,9 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie protected resetClient(client: RelayClient): void { client.nextOutgoingSeq = 1 client.highestReceivedSeq = 0 + client.attachedAt = Date.now() + client.lastReceivedAt = null + client.keepaliveObserved = false client.decoder.reset() client.generation++ client.closed = false @@ -163,14 +175,30 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie } protected startKeepalive(): void { + let lastTickAt = Date.now() this.keepaliveTimer = setInterval(() => { if (this.disposed) { return } + const now = Date.now() + // Why this threshold and not TIMEOUT_MS: a healthy client answers the PREVIOUS tick, so its + // lastReceivedAt is already up to KEEPALIVE_SEND_MS + RTT old. A tick gap beyond + // TIMEOUT_MS - KEEPALIVE_SEND_MS therefore pushes staleness past the window on its own, and a + // rebase armed at TIMEOUT_MS would not have fired -- reaping every client after a host + // suspend, a VM migration, or the relay's own event loop stalling. Mirrors the client's + // WAKE_GAP_MS guard (ssh-channel-multiplexer.ts). + const resumedAfterPause = now - lastTickAt >= TIMEOUT_MS - KEEPALIVE_SEND_MS + lastTickAt = now for (const client of this.clients.values()) { if (client.closed) { continue } + if (resumedAfterPause) { + client.attachedAt = now + if (client.lastReceivedAt !== null) { + client.lastReceivedAt = now + } + } client.writer.enqueue( 'liveness', () => { @@ -180,11 +208,53 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie 13 ) } + this.reapSilentClients(now) }, KEEPALIVE_SEND_MS) // Why: unref so the keepalive interval doesn't pin the event loop and block process exit. this.keepaliveTimer.unref() } + /** + * Drop the transport of a client that has gone silent. The relay's writer parks forever on a + * half-open link and nothing else ever notices, so an abandoned viewer kept its owner lease and + * left every PTY it held paused — the shape behind the "SSH degrades until I cannot connect at + * all" reports. Reaping is a statement about the TRANSPORT only: the cause stays the cautious + * 'local' default because silence is not evidence the peer died, and the PTYs stay live for the + * replacement client to reclaim (docs/reference/ssh-execution-boundary.md). + */ + private reapSilentClients(now: number): void { + for (const client of Array.from(this.clients.values())) { + // Why the primary is exempt: closing it tears down the relay's own stdin/stdout, and nothing + // in production revives it -- setWrite() has no non-test caller. The leak this exists for is + // a socket client holding an owner lease, and the launch channel's own liveness is already + // owned by the client-side dead-link check. + if (client === this.primaryClient) { + continue + } + if (client.closed) { + continue + } + // Why a client that has never spoken gets its own, much wider bound: a relay is launched + // before its client finishes handshaking, and on a slow link that can exceed the silence + // window, so judging it there would break the connect it is still completing. Leaving it + // unbounded instead held its socket and client entry forever. + if (client.lastReceivedAt === null) { + if (now - client.attachedAt > SILENT_CONNECT_TIMEOUT_MS) { + this.closeClient(client, new Error('Relay client never spoke'), true) + } + continue + } + // Why keepaliveObserved gates this: not every client speaks the keepalive protocol. The + // remote `orca` CLI sends one `orca.cli` request and waits for a result budgeted in minutes + // (src/relay/remote-cli-timeout.ts), so judging it on inbound silence would kill + // `terminal wait`, `--wait` and `orchestration ask` after 20s. + if (!client.keepaliveObserved || now - client.lastReceivedAt <= TIMEOUT_MS) { + continue + } + this.closeClient(client, new Error('Relay client stopped answering'), true) + } + } + protected closeClient( client: RelayClient, error: Error, diff --git a/src/relay/dispatcher-contract.ts b/src/relay/dispatcher-contract.ts index 15d4578c9bf..23c62431d2c 100644 --- a/src/relay/dispatcher-contract.ts +++ b/src/relay/dispatcher-contract.ts @@ -51,6 +51,18 @@ export type RelayClient = { bulkChain: Promise nextOutgoingSeq: number highestReceivedSeq: number + // Why: a client that never frames anything has no staleness to measure, so the silence window + // cannot see it. Its attach time is the only clock it has. + attachedAt: number + // Why: the relay had no inbound-liveness signal at all, so a half-open client was never reaped + // and kept its owner lease and paused PTYs indefinitely. + lastReceivedAt: number | null + // Why silence is only held against a client that sends keepalives: not every client speaks that + // protocol. The remote `orca` CLI opens the socket, sends one `orca.cli` request and then waits + // for a result that is deliberately budgeted in minutes (src/relay/remote-cli-timeout.ts), so + // judging it on inbound silence would kill `terminal wait`, `--wait` and `orchestration ask` + // after 20s. Only a client that has proven it participates is eligible. + keepaliveObserved: boolean generation: number closed: boolean droppedNotificationLog: DroppedProducerNotificationLog | null diff --git a/src/relay/dispatcher-frame-codec.ts b/src/relay/dispatcher-frame-codec.ts index b25a0d608ee..5a25fa1100f 100644 --- a/src/relay/dispatcher-frame-codec.ts +++ b/src/relay/dispatcher-frame-codec.ts @@ -15,11 +15,14 @@ import { RelayDispatcherCapacitySignals } from './dispatcher-capacity-signals' export abstract class RelayDispatcherFrameCodec extends RelayDispatcherCapacitySignals { protected handleFrame(client: RelayClient, frame: DecodedFrame): void { + // Before the KeepAlive early return: a keepalive is the only proof a quiet client is still there. + client.lastReceivedAt = Date.now() if (frame.id > client.highestReceivedSeq) { client.highestReceivedSeq = frame.id } if (frame.type === MessageType.KeepAlive) { + client.keepaliveObserved = true return } diff --git a/src/relay/dispatcher-silent-client-reaper.test.ts b/src/relay/dispatcher-silent-client-reaper.test.ts new file mode 100644 index 00000000000..b26fdd6bf4d --- /dev/null +++ b/src/relay/dispatcher-silent-client-reaper.test.ts @@ -0,0 +1,163 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { RelayDispatcher } from './dispatcher' +import { encodeJsonRpcFrame, encodeKeepAliveFrame, KEEPALIVE_SEND_MS, TIMEOUT_MS } from './protocol' + +// The relay had no inbound-liveness signal at all: its writer parks forever on a half-open link, so +// an abandoned viewer kept its owner lease and left the PTYs it held paused until the process died. +describe('RelayDispatcher silent-client reaper', () => { + let dispatcher: RelayDispatcher + + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(0) + }) + + afterEach(() => { + dispatcher.dispose() + vi.useRealTimers() + }) + + it('detaches a client that spoke once and then stopped answering', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + dispatcher.feedClient(clientId, encodeKeepAliveFrame(1, 0)) + + vi.advanceTimersByTime(TIMEOUT_MS + KEEPALIVE_SEND_MS * 2) + + // 'local', not a peer close: silence is not evidence the peer died, and a consumer that read it + // as one would shorten the owner grace on a session that is still there. + expect(detachListener).toHaveBeenCalledWith(clientId, 'local') + }) + + it('keeps a quiet but answering client attached', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + // A client with nothing to say still answers the keepalive; that is the only proof required. + for (let tick = 0; tick < 10; tick += 1) { + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + dispatcher.feedClient(clientId, encodeKeepAliveFrame(tick + 1, 0)) + } + + // Asserted against this client specifically: the unattached primary sink has no peer answering + // it in this harness, so it is expected to be reaped and says nothing about the case under test. + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('does not reap every client on the first tick after the host slept', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + dispatcher.attachClient(() => true) + + // One tick fires far late because the process was paused, not because the peers went away. + vi.setSystemTime(10 * 60_000) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + expect(detachListener).not.toHaveBeenCalled() + }) + + it('does not judge a client that has not spoken yet by the silence window', () => { + // A relay is launched before its client finishes handshaking, and on a slow link that can + // outlast the window. Reaping there would break the connect the client is still completing. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + vi.advanceTimersByTime(TIMEOUT_MS * 5) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('still bounds a client that never speaks at all', () => { + // Otherwise it is invisible to the silence window forever -- no lastReceivedAt to go stale -- + // and its socket, writer and client entry are held for the life of the relay. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + vi.advanceTimersByTime(TIMEOUT_MS * 7) + + expect(detachListener).toHaveBeenCalledWith(clientId, 'local') + }) + + it("does not spend a mute client's connect budget while the host was suspended", () => { + // Same rebase the silence window gets: a paused process is not a peer that went away. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + vi.setSystemTime(60 * 60_000) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('never reaps a client that does not send keepalives at all', () => { + // The remote `orca` CLI opens the socket, sends one `orca.cli` request and then waits for a + // result budgeted in minutes (remote-cli-timeout.ts: 5min default, 10min for wait, 11min for + // orchestration ask). It has no keepalive timer, so judging it on inbound silence would abort + // `terminal wait`, `--wait` and `orchestration ask` after 20s. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + dispatcher.feedClient( + clientId, + encodeJsonRpcFrame({ jsonrpc: '2.0', id: 1, method: 'orca.cli', params: {} }, 1, 0) + ) + + vi.advanceTimersByTime(TIMEOUT_MS * 20) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('does not reap a healthy client when the relay itself stalls for most of the window', () => { + // The dead band this exists for: a healthy client answers the PREVIOUS tick, so its + // lastReceivedAt is already ~KEEPALIVE_SEND_MS old. A tick gap short of TIMEOUT_MS still pushes + // staleness past the window, so a rebase armed at TIMEOUT_MS would never fire and every client + // would be reaped after a host suspend, VM migration, or an event-loop stall. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + // The client answers at t=5s, then the next tick at t=10s finds it already ~5s stale — normal. + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + dispatcher.feedClient(clientId, encodeKeepAliveFrame(1, 0)) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + // Now the relay stalls: the clock jumps but no tick runs, so the following tick lands 17s after + // the last one. Staleness is 22s (past the window) while the tick gap is under TIMEOUT_MS, so a + // rebase armed at TIMEOUT_MS would not fire and this healthy client would be reaped. + vi.setSystemTime(Date.now() + 12_000) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('never reaps the primary client, whose sink cannot be revived', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + + // Feed the primary first, or the test proves nothing: an unfed client is skipped by the + // never-spoken and no-keepalive guards, so it survives whether or not the exemption exists. + // A real primary answers keepalives, so the exemption is the only thing standing between it + // and the reaper. + dispatcher.feed(encodeKeepAliveFrame(1, 0)) + + vi.advanceTimersByTime(TIMEOUT_MS * 20) + + // Client id 1 is the primary sink; closing it would tear down the relay's own stdin/stdout and + // nothing in production calls setWrite() to bring it back. + expect(detachListener).not.toHaveBeenCalled() + }) +}) From 033a2a64e17f97d7a69bcaba7f1e21cf5fb82346 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:40 -0700 Subject: [PATCH 07/77] fix(remote-runtime): make every advertised recovery attempt reachable, and stop two recovery latches (#17822) * fix(remote-runtime): derive the recovery budget and stop faking a spent window #11305: RECOVERY_DELAYS_MS summed to 60,750ms against a hand-written REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS of 60,000ms, so the ladder's tail was unreachable. Derive the deadline from the schedule plus one RPC timeout per step so a half-open link can actually reach every backoff step, and pin the relation with a test that fails if the sum ever outgrows the budget. #12683: markDisconnected() is a UI latch, not proof the auto-recovery window ran out. Track deadline expiry on the recovery state and only let that license the same-handle reattach that bypasses require-replacement fencing. #12684: a recoverable connect() failure latched 'disconnected' with no armed retry, no parked retry and a Reconnect button that returned false. Schedule a bounded retry (which the deadline parks for online/resume) and let the button fire a parked retry. * fix(remote-runtime): stop a post-latch connect failure from re-arming the recovery window The last attempt's RPC budget expires at the same instant as the deadline, so a silently dropped link rejects after phase latched to 'disconnected'. begin() then started a fresh full-length window, so the budget never actually expired. Park the retry under the latched epoch instead, which keeps online/resume/Reconnect armed even when the deadline lands mid-attempt with nothing scheduled. Also fences the same-handle end-reuse window on its own 60s constant so the derived recovery budget no longer silently triples an unrelated stale-handle check. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- ...minalRemoteRuntimeReconnectBanner.test.tsx | 2 +- .../TerminalRemoteRuntimeReconnectBanner.tsx | 2 +- ...e-runtime-connect-failure-recovery.test.ts | 160 ++++++++++++++++++ ...ge-toast-flood-and-stuck-reconnect.test.ts | 5 +- ...mote-runtime-pty-deadline-reattach.test.ts | 67 +++++++- ...runtime-pty-latched-pane-retention.test.ts | 15 +- .../remote-runtime-pty-recovery-state.test.ts | 70 +++++++- .../remote-runtime-pty-recovery-state.ts | 47 ++++- ...-transport-create-outcome-recovery.test.ts | 11 +- ...ty-transport-stale-handle-recovery.test.ts | 5 +- ...ransport-sticky-replacement-policy.test.ts | 3 +- ...ime-pty-transport-stream-reconnect.test.ts | 3 +- ...-pty-transport-web-mirror-recovery.test.ts | 15 +- .../remote-runtime-pty-transport.ts | 57 ++++++- src/renderer/src/i18n/locales/es.json | 2 +- src/renderer/src/i18n/locales/ja.json | 2 +- src/renderer/src/i18n/locales/ko.json | 2 +- src/renderer/src/i18n/locales/zh.json | 2 +- 18 files changed, 426 insertions(+), 44 deletions(-) create mode 100644 src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts diff --git a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx index 70341d7a4f6..824a6a80744 100644 --- a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx @@ -17,7 +17,7 @@ describe('TerminalRemoteRuntimeReconnectBanner', () => { render() expect(screen.getByText('Reconnecting to remote runtime')).toBeInTheDocument() - expect(screen.getByText(/retry for up to one minute/)).toBeInTheDocument() + expect(screen.getByText(/retrying automatically/)).toBeInTheDocument() expect(screen.queryByRole('button')).not.toBeInTheDocument() }) diff --git a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx index c629fa71a36..1cf0fee0903 100644 --- a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx @@ -50,7 +50,7 @@ export function TerminalRemoteRuntimeReconnectBanner({ {retrying ? translate( 'auto.components.terminal.pane.TerminalRemoteRuntimeReconnectBanner.retryingBody', - 'Orca will retry for up to one minute. This terminal will resume if the connection returns.' + 'Orca is retrying automatically. This terminal will resume if the connection returns.' ) : translate( 'auto.components.terminal.pane.TerminalRemoteRuntimeReconnectBanner.disconnectedBody', diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts new file mode 100644 index 00000000000..6c12277f2c0 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts @@ -0,0 +1,160 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + createRemoteRuntimeTransportMocks, + type MultiplexSubscriptionCallbacks +} from './remote-runtime-pty-transport-test-harness' +import { + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS, + REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS +} from './remote-runtime-pty-recovery-state' + +let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null +let resolvedPaneHandle = 'terminal-1' + +const { runtimeCall, resetRemoteRuntimeTransport } = createRemoteRuntimeTransportMocks({ + getCallbacks: () => subscriptionCallbacks, + setCallbacks: (callbacks) => { + subscriptionCallbacks = callbacks + }, + getResolvedPaneHandle: () => resolvedPaneHandle, + setResolvedPaneHandle: (handle) => { + resolvedPaneHandle = handle + } +}) + +// #12684: connect() classified these failures as recoverable and then latched 'disconnected' with +// nothing armed — no backoff timer, no parked retry, and a Reconnect button that returned false. +describe('recoverable connect failures on a remote runtime pane', () => { + let resolvePaneCalls = 0 + + function installUnreachableRuntime(): void { + resolvePaneCalls = 0 + runtimeCall.mockImplementation(async (args: { method: string }) => { + if (args.method === 'terminal.resolvePane') { + resolvePaneCalls += 1 + } + throw Object.assign(new Error('Remote Orca runtime closed the connection.'), { + code: 'remote_runtime_unavailable' + }) + }) + } + + // Why: installUnreachableRuntime() rejects synchronously, so every failure lands during a backoff + // wait. A silently dropped link instead burns the whole RPC budget, so the rejection arrives while + // the attempt is still in flight — including after the auto-recovery deadline has already latched. + function installSilentlyDroppedRuntime(): void { + resolvePaneCalls = 0 + runtimeCall.mockImplementation(async (args: { method: string }) => { + if (args.method === 'terminal.resolvePane') { + resolvePaneCalls += 1 + } + await new Promise((resolve) => { + setTimeout(resolve, REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS) + }) + throw Object.assign(new Error('Remote Orca runtime closed the connection.'), { + code: 'remote_runtime_unavailable' + }) + }) + } + + beforeEach(() => { + resetRemoteRuntimeTransport() + }) + + it('keeps retrying a recoverable connect failure instead of latching immediately', async () => { + vi.useFakeTimers() + try { + installUnreachableRuntime() + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + const onError = vi.fn() + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'tab-1', + leafId: 'pane:1' + }) + + await transport.connect({ + url: '', + sessionId: 'remote:env-1@@', + callbacks: { onError } + }) + + // Loss of contact is unverifiable, not a dead terminal: automatic recovery must still be running. + expect(resolvePaneCalls).toBe(1) + expect(transport.getRecoveryState?.().phase).toBe('backoff') + expect(onError).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(1_000) + expect(resolvePaneCalls).toBeGreaterThan(1) + + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) + + it('leaves both revival paths armed once the recovery window is spent', async () => { + vi.useFakeTimers() + try { + installUnreachableRuntime() + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + // Why dynamic: resetRemoteRuntimeTransport() re-registers the module graph, and the retry + // registry only sees panes from the same instance the transport was loaded from. + const { retryAllRemoteRuntimePtyRecoveriesNow } = + await import('./remote-runtime-pty-recovery-state') + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'tab-1', + leafId: 'pane:1' + }) + + await transport.connect({ url: '', sessionId: 'remote:env-1@@', callbacks: {} }) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 1_000) + + expect(transport.getRecoveryState?.().phase).toBe('disconnected') + const callsAtCutoff = resolvePaneCalls + + // The cutoff stops self-initiated retries only; online/resume must still find a parked retry. + expect(retryAllRemoteRuntimePtyRecoveriesNow()).toBe(1) + await vi.advanceTimersByTimeAsync(1_000) + expect(resolvePaneCalls).toBeGreaterThan(callsAtCutoff) + + // ...and so must the Reconnect button, which returned false before #12684. + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 1_000) + expect(transport.getRecoveryState?.().phase).toBe('disconnected') + expect(transport.retryRecovery?.()).toBe(true) + + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) + it('keeps the window bounded when a silent drop fails after the deadline latched', async () => { + vi.useFakeTimers() + try { + installSilentlyDroppedRuntime() + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'tab-1', + leafId: 'pane:1' + }) + + void transport.connect({ url: '', sessionId: 'remote:env-1@@', callbacks: {} }) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS * 2) + + // The in-flight rejection must not begin a new epoch; that re-arms a full-length window forever. + expect(transport.getRecoveryState?.().phase).toBe('disconnected') + const callsAtCutoff = resolvePaneCalls + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS * 2) + expect(resolvePaneCalls).toBe(callsAtCutoff) + + // A deadline that lands mid-attempt parks nothing, so the latch must still stay revivable. + expect(transport.retryRecovery?.()).toBe(true) + + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts index b797a0f5877..9ac752b2b31 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts @@ -26,6 +26,7 @@ import { import { RuntimeRpcCallQueueOverloadError } from '../../../../shared/runtime-rpc-call-queue' import { withRemoteRuntimeTailscaleHint } from '../../../../shared/remote-runtime-tailscale-hint' import type { PtyTransportRecoveryState } from './pty-transport-types' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' const ELECTRON_IPC_PREFIX = "Error invoking remote method 'runtimeEnvironments:call': " @@ -409,7 +410,7 @@ describe('remote runtime outage: toast flood and stuck reconnect (issue3)', () = await vi.advanceTimersByTimeAsync(16_000) // Auto-recovery deadline latches the pane 'disconnected'. - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') // Connectivity restored; 'online'/system-resume trigger fires. @@ -417,7 +418,7 @@ describe('remote runtime outage: toast flood and stuck reconnect (issue3)', () = await vi.advanceTimersByTimeAsync(16_000) // Latch again, then the user clicks the Reconnect banner. - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) transport.retryRecovery?.() await vi.advanceTimersByTimeAsync(16_000) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts index 886516414ca..61758efb299 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts @@ -8,6 +8,7 @@ import { encodeTerminalStreamText } from '../../../../shared/terminal-stream-protocol' import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' describe('remote runtime pty reattach after the bounded recovery window', () => { const runtimeCall = vi.fn() @@ -198,7 +199,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => expect(handleEvents.getWebSessionTerminalHandleSubscriberCountForTests()).toBe(1) expect(transport.getRecoveryState?.().phase).not.toBe('disconnected') - await vi.advanceTimersByTimeAsync(50_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') // The cutoff must not tear down the accepted-snapshot listener; it is the only path back. expect(handleEvents.getWebSessionTerminalHandleSubscriberCountForTests()).toBe(1) @@ -227,7 +228,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => const { transport, onError } = await attachStalePane() const handleEvents = await import('../../runtime/web-session-terminal-handle-events') - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const listCallsAtCutoff = hostListCalls @@ -277,7 +278,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => const { transport, onError } = await attachStalePane() const handleEvents = await import('../../runtime/web-session-terminal-handle-events') - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') expect(handleEvents.getWebSessionTerminalHandleSubscriberCountForTests()).toBe(1) const listCallsAtCutoff = hostListCalls @@ -310,7 +311,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => const { retryAllRemoteRuntimePtyRecoveriesNow } = await import('./remote-runtime-pty-recovery-state') - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const listCallsAtCutoff = hostListCalls @@ -365,7 +366,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => callbacks: { onError: vi.fn() } }) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const callsBeforeRetry = runtimeCall.mock.calls.length @@ -378,4 +379,60 @@ describe('remote runtime pty reattach after the bounded recovery window', () => vi.useRealTimers() } }) + + // #12683: a fatal resubscribe latches the banner via markDisconnected(), but that latch is not + // evidence the auto-recovery window ran out, so it must not license reattaching a fenced handle. + it('does not reattach a fenced same handle when only a UI latch closed the window', async () => { + vi.useFakeTimers() + try { + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + const handleEvents = await import('../../runtime/web-session-terminal-handle-events') + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'web-terminal-tab-1', + leafId: 'pane:1', + onPtyExit: vi.fn(), + onPtyRebind: vi.fn() + }) + transport.attach({ + existingPtyId: 'remote:env-1@@terminal-stale', + cols: 80, + rows: 24, + callbacks: { onError: vi.fn() } + }) + await vi.waitFor(() => expect(subscriptionSendBinary).toHaveBeenCalled()) + emitSnapshot(latestSubscribePayload().streamId, 'live before the drop') + + // The host keeps publishing the same handle, so no replacement can ever arrive. + runtimeCall.mockImplementation(async (args: { method: string }) => { + if (args.method !== 'session.tabs.list') { + return { ok: true, result: {} } + } + hostListCalls += 1 + return { ok: true, result: hostSnapshot('terminal-stale', hostListCalls + 1, 'epoch-1') } + }) + // The stream drops and the resubscribe fails fatally with a stale handle: markDisconnected() + // latches the banner, then stale routing fences the handle with require-replacement. + // Only that one attempt fails, so a later reattach would succeed and be observable. + runtimeSubscribe.mockImplementationOnce(async () => { + throw new Error('terminal_handle_stale') + }) + subscriptionCallbacks?.onClose?.() + await vi.advanceTimersByTimeAsync(16_000) + expect(transport.getRecoveryState?.().phase).not.toBe('idle') + + const subscribesBeforeRepublish = subscribedTerminalHandles().length + handleEvents.queueAcceptedWebSessionTerminalSnapshot( + hostSnapshot('terminal-stale', 9, 'epoch-2'), + 'env-1' + ) + await vi.advanceTimersByTimeAsync(1_000) + + // The recovery deadline never fired, so the fenced handle stays fenced. + expect(subscribedTerminalHandles()).toHaveLength(subscribesBeforeRepublish) + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) }) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts index 5892ee2c19d..3c1ec0421a9 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts @@ -5,6 +5,7 @@ import { decodeTerminalStreamJson } from '../../../../shared/terminal-stream-protocol' import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' // Why: the recovery cutoff no longer tears down the retry registry entry or the accepted-snapshot // listener, so those two module-global collections are the only places a latched pane can accumulate. @@ -184,7 +185,7 @@ describe('remote runtime pty latched-pane retention', () => { for (let cycle = 0; cycle < 20; cycle += 1) { const transport = await attachStalePane(cycle) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') latched.push(await registries()) transport.destroy?.() @@ -208,7 +209,7 @@ describe('remote runtime pty latched-pane retention', () => { const settled: { subscribers: number; scheduled: number }[] = [] for (let cycle = 0; cycle < 20; cycle += 1) { const transport = await attachStalePane(cycle) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') transport.detach?.() await vi.advanceTimersByTimeAsync(1_000) @@ -227,7 +228,7 @@ describe('remote runtime pty latched-pane retention', () => { const transports: Awaited>[] = [] for (let pane = 0; pane < 8; pane += 1) { transports.push(await attachStalePane(pane)) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) } // Retention is per live pane, not per timeout: eight latched panes hold eight of each. expect(await registries()).toEqual({ subscribers: 8, scheduled: 8 }) @@ -247,7 +248,7 @@ describe('remote runtime pty latched-pane retention', () => { vi.useFakeTimers() try { const transport = await attachStalePane(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const baseline = await registries() @@ -277,7 +278,7 @@ describe('remote runtime pty latched-pane retention', () => { const { retryAllRemoteRuntimePtyRecoveriesNow } = await import('./remote-runtime-pty-recovery-state') const transport = await attachStalePane(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const baseline = await registries() @@ -295,7 +296,7 @@ describe('remote runtime pty latched-pane retention', () => { // A second trigger in the same window must find nothing to advance, so an online/resume // storm cannot stack fresh recovery epochs on one pane. expect(retryAllRemoteRuntimePtyRecoveriesNow()).toBe(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') observed.push({ ...(await registries()), timers: vi.getTimerCount(), revived }) } @@ -325,7 +326,7 @@ describe('remote runtime pty latched-pane retention', () => { try { const handleEvents = await import('../../runtime/web-session-terminal-handle-events') const transport = await attachStalePane(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const baseline = await registries() diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts index d71af0b9afb..c245c184436 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts @@ -1,6 +1,8 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS, + REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS, + REMOTE_RUNTIME_RECOVERY_DELAYS_MS, RemoteRuntimePtyRecoveryState, retryAllRemoteRuntimePtyRecoveriesNow } from './remote-runtime-pty-recovery-state' @@ -63,7 +65,7 @@ describe('RemoteRuntimePtyRecoveryState', () => { const epoch = state.begin() state.schedule(epoch, retry) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(state.currentPhase).toBe('disconnected') expect(state.isActive).toBe(false) @@ -112,7 +114,7 @@ describe('RemoteRuntimePtyRecoveryState', () => { const state = new RemoteRuntimePtyRecoveryState() const firstEpoch = state.begin() - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const manualEpoch = state.begin() expect(manualEpoch).toBe(firstEpoch + 1) @@ -227,4 +229,68 @@ describe('RemoteRuntimePtyRecoveryState', () => { expect(retryAllRemoteRuntimePtyRecoveriesNow()).toBe(0) state.dispose() }) + + // #11305: the schedule and the deadline lived as two independent literals and drifted apart. + it('keeps the backoff schedule inside the auto-recovery budget it arms', () => { + const scheduleSumMs = REMOTE_RUNTIME_RECOVERY_DELAYS_MS.reduce( + (total, delayMs) => total + delayMs, + 0 + ) + + expect(scheduleSumMs).toBeLessThanOrEqual(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + // Every step also needs room for the attempt it leads into, or the tail is dead code + // whenever a half-open link makes each attempt burn its full RPC timeout. + expect( + scheduleSumMs + + REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length * REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS + ).toBeLessThanOrEqual(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + }) + + it('reaches every backoff step when each attempt burns a full RPC timeout', async () => { + vi.useFakeTimers() + const state = new RemoteRuntimePtyRecoveryState() + const attemptStartsMs: number[] = [] + const epoch = state.begin() + const startedAt = Date.now() + + const failSlowly = (currentEpoch: number): void => { + attemptStartsMs.push(Date.now() - startedAt) + // Silent-drop reconnects do not fail instantly; they time out. + setTimeout(() => { + if (state.isCurrent(currentEpoch)) { + state.schedule(currentEpoch, failSlowly) + } + }, REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS) + } + state.schedule(epoch, failSlowly) + + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + + expect(attemptStartsMs.length).toBeGreaterThanOrEqual(REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length) + expect(state.currentPhase).toBe('disconnected') + state.dispose() + }) + + // #12683: markDisconnected() is a UI latch, not proof the window ran out. + it('only reports the auto-recovery window spent when the deadline actually fired', async () => { + vi.useFakeTimers() + const state = new RemoteRuntimePtyRecoveryState() + const epoch = state.begin() + state.schedule(epoch, vi.fn()) + + state.markDisconnected() + expect(state.currentPhase).toBe('disconnected') + expect(state.autoRecoveryDeadlineExpired).toBe(false) + + const secondEpoch = state.begin() + state.schedule(secondEpoch, vi.fn()) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + + expect(state.currentPhase).toBe('disconnected') + expect(state.autoRecoveryDeadlineExpired).toBe(true) + + state.markHealthy() + expect(state.autoRecoveryDeadlineExpired).toBe(false) + state.dispose() + }) }) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts index fbdd50e443f..002e1320398 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts @@ -1,5 +1,17 @@ -const RECOVERY_DELAYS_MS = [250, 500, 1000, 2000, 4000, 8000, 15_000, 30_000] as const -export const REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS = 60_000 +export const REMOTE_RUNTIME_RECOVERY_DELAYS_MS = [ + 250, 500, 1000, 2000, 4000, 8000, 15_000, 30_000 +] as const + +// Why: mirrors DEFAULT_REMOTE_RUNTIME_TIMEOUT_MS in the main-process runtime router; a silently +// dropped link burns the whole RPC timeout on the attempt each backoff step leads into. +export const REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS = 15_000 + +// Why derived, not hand-tuned: a literal deadline drifted below the ladder it arms, making the last +// backoff steps unreachable dead code (#11305). Loss of contact is never evidence of exit, so the +// window must outlast the schedule it advertises rather than the schedule being trimmed to fit. +export const REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS = + REMOTE_RUNTIME_RECOVERY_DELAYS_MS.reduce((total, delayMs) => total + delayMs, 0) + + REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length * REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS export type RemoteRuntimePtyRecoveryPhase = | 'idle' @@ -34,6 +46,9 @@ export class RemoteRuntimePtyRecoveryState { private deadlineTimer: ReturnType | null = null private pendingRetry: ((epoch: number) => void) | null = null private pendingEpoch: number | null = null + // Why: only the wall-clock deadline proves the auto-recovery window was actually spent; a UI latch + // via markDisconnected() must not forge that evidence (#12683). + private deadlineExpired = false constructor(private readonly onChange?: () => void) {} @@ -53,6 +68,10 @@ export class RemoteRuntimePtyRecoveryState { return this.attempt } + get autoRecoveryDeadlineExpired(): boolean { + return this.deadlineExpired + } + begin(): number { if (this.phase === 'disposed') { return this.epoch @@ -82,7 +101,10 @@ export class RemoteRuntimePtyRecoveryState { } this.clearRetryTimer() this.phase = 'backoff' - const delayMs = RECOVERY_DELAYS_MS[Math.min(this.attempt, RECOVERY_DELAYS_MS.length - 1)] + const delayMs = + REMOTE_RUNTIME_RECOVERY_DELAYS_MS[ + Math.min(this.attempt, REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length - 1) + ] this.attempt += 1 this.pendingRetry = retry this.pendingEpoch = epoch @@ -107,11 +129,22 @@ export class RemoteRuntimePtyRecoveryState { // Why: a wait that ends with no liveness evidence arms no timer, so park a retry or online/resume/reconnect find nothing to revive. parkRetryForExternalTrigger(epoch: number, retry: (epoch: number) => void): boolean { - if (!this.isCurrent(epoch) || this.pendingRetry !== null) { + return this.isCurrent(epoch) && this.parkRetry(retry) + } + + // Why: the deadline can latch while an attempt is still in flight, before schedule() parked anything, + // so the late failure has no live epoch to join and must not begin a new one — that would re-arm a + // full-length window and the budget would never actually expire. + parkRetryAfterDeadline(retry: (epoch: number) => void): boolean { + return this.phase === 'disconnected' && this.parkRetry(retry) + } + + private parkRetry(retry: (epoch: number) => void): boolean { + if (this.pendingRetry !== null) { return false } this.pendingRetry = retry - this.pendingEpoch = epoch + this.pendingEpoch = this.epoch scheduledRecoveries.add(this) return true } @@ -152,6 +185,7 @@ export class RemoteRuntimePtyRecoveryState { if (this.phase === 'disposed') { return } + this.deadlineExpired = false this.clearTimers() this.phase = 'idle' this.attempt = 0 @@ -175,6 +209,7 @@ export class RemoteRuntimePtyRecoveryState { if (this.phase === 'disposed') { return } + this.deadlineExpired = false this.epoch += 1 this.clearTimers() this.phase = 'idle' @@ -191,11 +226,13 @@ export class RemoteRuntimePtyRecoveryState { private armDeadline(epoch: number): void { this.clearDeadlineTimer() + this.deadlineExpired = false const timer = setTimeout(() => { if (this.deadlineTimer !== timer || !this.isCurrent(epoch)) { return } this.deadlineTimer = null + this.deadlineExpired = true // Why: the cutoff stops self-initiated retries but must keep the pane revivable by online/resume/reconnect. this.stopRetryTimer() this.phase = 'disconnected' diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts index b34ec4f3fea..59dc3612ed1 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts @@ -4,6 +4,7 @@ import { createRemoteRuntimeTransportMocks, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -81,7 +82,7 @@ describe('createRemoteRuntimePtyTransport', () => { let createCalls = 0 runtimeCall.mockImplementation(async (args: { method: string }) => { if (args.method === 'status.get') { - vi.setSystemTime(startedAt + 59_000) + vi.setSystemTime(startedAt + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS - 1_000) return { ok: true, result: { capabilities: [TERMINAL_CREATE_IDEMPOTENCY_RUNTIME_CAPABILITY] } @@ -163,7 +164,7 @@ describe('createRemoteRuntimePtyTransport', () => { transport.destroy?.() }) - it('stops unknown terminal-create recovery after one minute and remains manually retryable', async () => { + it('stops unknown terminal-create recovery at the cutoff and remains manually retryable', async () => { vi.useFakeTimers() try { let reachable = false @@ -205,7 +206,7 @@ describe('createRemoteRuntimePtyTransport', () => { onRecoveryStateChange: (state) => recoveryStates.push(state.phase) } }) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) await connect const callsAtCutoff = runtimeCall.mock.calls.length @@ -218,7 +219,7 @@ describe('createRemoteRuntimePtyTransport', () => { statusTimesOut = true expect(transport.retryRecovery?.()).toBe(true) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const callsAtManualCutoff = runtimeCall.mock.calls.length expect(transport.getRecoveryState?.().phase).toBe('disconnected') await vi.advanceTimersByTimeAsync(5 * 60_000) @@ -278,7 +279,7 @@ describe('createRemoteRuntimePtyTransport', () => { }) const connect = transport.connect({ url: '', callbacks: {} }) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) await connect expect(transport.getRecoveryState?.().phase).toBe('disconnected') diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts index 8817d5c23e4..53c1921fe74 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts @@ -4,6 +4,7 @@ import { readyHostSessionInventoryResponse, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -72,7 +73,7 @@ describe('createRemoteRuntimePtyTransport', () => { expect(hostListCalls).toBe(callsAfterTwoWindows) expect(transport.getRecoveryState?.().phase).toBe('recovering') - await vi.advanceTimersByTimeAsync(9_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') expect(subscribedTerminalHandles()).toEqual(['terminal-stable']) transport.destroy?.() @@ -180,7 +181,7 @@ describe('createRemoteRuntimePtyTransport', () => { 'terminal-flapping' ]) expect(transport.isConnected()).toBe(false) - await vi.advanceTimersByTimeAsync(45_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') transport.destroy?.() } finally { diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts index 48a24a2c084..5a61a5587e6 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts @@ -9,6 +9,7 @@ import { readyHostSessionInventoryResponse, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -160,7 +161,7 @@ describe('createRemoteRuntimePtyTransport', () => { expect(transport.isConnected()).toBe(false) expect(onPtyExit).not.toHaveBeenCalled() - await vi.advanceTimersByTimeAsync(44_001) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') expect(subscribedTerminalHandles()).toHaveLength(3) } finally { diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts index 4ef1075e3f7..e0b85414e3f 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts @@ -9,6 +9,7 @@ import { createRemoteRuntimeTransportMocks, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -534,7 +535,7 @@ describe('createRemoteRuntimePtyTransport', () => { partitioned = true callbacksByConnection[0].onClose?.() - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const disconnectedState = transport.getRecoveryState?.() const callsAtCutoff = runtimeSubscribe.mock.calls.length diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts index f8c07ea045d..784ed9d8768 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts @@ -3,6 +3,10 @@ import { createRemoteRuntimeTransportMocks, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS, + REMOTE_RUNTIME_RECOVERY_DELAYS_MS +} from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -140,10 +144,11 @@ describe('createRemoteRuntimePtyTransport', () => { existingPtyId: 'remote:env-1@@stale-client-handle', callbacks: {} }) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const attemptsAtCutoff = runtimeSubscribe.mock.calls.length - expect(attemptsAtCutoff).toBe(8) + // Why: every backoff step must be reachable inside the window it arms (#11305). + expect(attemptsAtCutoff).toBeGreaterThanOrEqual(REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length) expect(transport.getRecoveryState?.().phase).toBe('disconnected') await vi.advanceTimersByTimeAsync(5 * 60_000) @@ -235,7 +240,7 @@ describe('createRemoteRuntimePtyTransport', () => { await vi.advanceTimersByTimeAsync(250) expect(activateAttempts).toBe(2) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') rejectInFlight( @@ -292,7 +297,7 @@ describe('createRemoteRuntimePtyTransport', () => { await vi.advanceTimersByTimeAsync(250) expect(runtimeSubscribe).toHaveBeenCalledTimes(1) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') rejectSubscription( @@ -349,7 +354,7 @@ describe('createRemoteRuntimePtyTransport', () => { expect.objectContaining({ method: 'terminal.resolvePane' }) ) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') resolveMetadata({ diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts index 1e5782fe496..b1bd7336c86 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts @@ -90,6 +90,10 @@ const HOST_SESSION_POLL_MAX_MS = 1_000 const HOST_SESSION_ATTACH_TIMEOUT_MS = 15_000 const HOST_SESSION_INVENTORY_MAX_WINDOWS_PER_RECOVERY = 2 const HOST_SESSION_SAME_HANDLE_END_REUSE_LIMIT = 2 +// Why its own constant: this fences how long an end-then-reattach on the same handle still counts as +// one recovery, which is unrelated to how long auto-recovery keeps retrying. It read the recovery +// budget before that budget became a derived value, and must not drift with it. +const HOST_SESSION_SAME_HANDLE_END_REUSE_WINDOW_MS = 60_000 const MAX_SURFACED_TERMINAL_ERRORS = 8 const TERMINAL_CREATE_RETRY_DELAYS_MS = [250, 500, 1000, 2000, 4000, 8000, 15_000, 30_000] as const @@ -247,7 +251,9 @@ export function createRemoteRuntimePtyTransport( clearPublishedHandleWait() } if (recovery.currentPhase === 'disconnected') { - autoRecoveryWindowSpent = true + // Why: only the wall-clock deadline is evidence the window was spent; a UI latch from a fatal + // resubscribe must not license reattaching a fenced same handle (#12683). + autoRecoveryWindowSpent ||= recovery.autoRecoveryDeadlineExpired // Why: cached pixels may remain, but no stream from the exhausted epoch may keep delivering or accepting terminal traffic. subscriptionGeneration += 1 closeMultiplexedStream() @@ -326,7 +332,7 @@ export function createRemoteRuntimePtyTransport( if ( sameHandleEndReuseHandle !== targetHandle || sameHandleEndReuseAttachedAt === null || - Date.now() - sameHandleEndReuseAttachedAt >= REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + Date.now() - sameHandleEndReuseAttachedAt >= HOST_SESSION_SAME_HANDLE_END_REUSE_WINDOW_MS ) { resetSameHandleEndReuse() return 'prefer-replacement' @@ -870,6 +876,43 @@ export function createRemoteRuntimePtyTransport( return false } + // Why: a recoverable connect failure is unverifiable contact loss, not a dead terminal, so retry + // whichever path can still reach the pane instead of latching with nothing armed (#12684). + function retryAfterRecoverableConnectFailure(nextEpoch: number): void { + if (destroyed || terminalEnded) { + return + } + if (connected && handle) { + scheduleResubscribeAfterTransportClose(getRecoveryReplacementPolicy(handle), nextEpoch) + return + } + replayLastTransportEntryPoint() + } + + // Why: schedule() both auto-retries inside the window and leaves the retry parked when the deadline + // latches, so online/resume and the Reconnect button always find something to fire. + function scheduleConnectRetryAfterRecoverableFailure(): void { + if (destroyed) { + return + } + // Why: an ambiguous create already owns a reconciliation-gated retry that only Reconnect may + // re-enter; auto-replaying here would just re-probe a runtime that cannot reconcile. + if (terminalCreateNeedsReconciliation || agentSessionRequiresHostAuthorityReplay) { + recovery.markDisconnected() + return + } + // Why: the last attempt's RPC budget expires at the same instant as the deadline, so a silent drop + // rejects after the latch. Beginning a new epoch there re-arms the whole window, so park instead. + if (recovery.currentPhase === 'disconnected') { + recovery.parkRetryAfterDeadline(retryAfterRecoverableConnectFailure) + return + } + const recoveryEpoch = recovery.isActive ? recovery.currentEpoch : recovery.begin() + if (!recovery.schedule(recoveryEpoch, retryAfterRecoverableConnectFailure)) { + recovery.markDisconnected() + } + } + async function attachHostSessionMirror( options: { cols?: number; rows?: number }, notifySpawn = true, @@ -1033,6 +1076,8 @@ export function createRemoteRuntimePtyTransport( kind === 'agent-session' ? agentSessionRequiresHostAuthorityReplay : terminalCreateNeedsReconciliation + // Why the same budget: this loop calls recovery.begin(), so a shorter local deadline would abandon + // the create while the recovery state still reports 'recovering' with nothing in flight. let recoveryDeadlineAt: number | null = recovery.isActive ? Date.now() + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS : null @@ -2314,7 +2359,7 @@ export function createRemoteRuntimePtyTransport( } else if ( isRecoverableRemoteRuntimeConnectionError(toRemoteRuntimeClientErrorLike(error)) ) { - recovery.markDisconnected() + scheduleConnectRetryAfterRecoverableFailure() } else { recovery.cancel() emitRecoveryState() @@ -2616,6 +2661,12 @@ export function createRemoteRuntimePtyTransport( void transport.connect(lastConnectOptions) return true } + // Why: online/resume fires a parked retry; the button must not be weaker than an event (#12684). + if (!destroyed && !terminalEnded && recovery.currentPhase === 'disconnected') { + if (recovery.retryNow()) { + return true + } + } if ( destroyed || terminalEnded || diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index f4a4fd931de..0ac5d624ede 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -2881,7 +2881,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "Reconnecting to remote runtime", "disconnectedTitle": "Remote runtime disconnected", - "retryingBody": "Orca will retry for up to one minute. This terminal will resume if the connection returns.", + "retryingBody": "Orca is retrying automatically. This terminal will resume if the connection returns.", "disconnectedBody": "Automatic retries stopped. Reconnect to resume this terminal session.", "reconnectButton": "Reconnect" } diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index ee6e4440f79..d3af25ad697 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -2881,7 +2881,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "リモートランタイムに再接続中", "disconnectedTitle": "リモートランタイムが切断されました", - "retryingBody": "Orca は最大1分間再試行します。接続が復元されると、このターミナルが再開されます。", + "retryingBody": "Orca は自動的に再試行します。接続が復元されると、このターミナルが再開されます。", "disconnectedBody": "自動再試行が停止しました。このターミナルセッションを再開するには再接続してください。", "reconnectButton": "再接続" } diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index 76ed151ca2c..47d9e48fd26 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -2886,7 +2886,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "원격 런타임에 다시 연결 중", "disconnectedTitle": "원격 런타임 연결 끊김", - "retryingBody": "Orca는 최대 1분간 재시도합니다. 연결이 복원되면 이 터미널이 재개됩니다.", + "retryingBody": "Orca가 자동으로 재시도합니다. 연결이 복원되면 이 터미널이 재개됩니다.", "disconnectedBody": "자동 재시도가 중지되었습니다. 이 터미널 세션을 재개하려면 다시 연결하세요.", "reconnectButton": "다시 연결" } diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 5c1324afa12..b097b789571 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -2896,7 +2896,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "正在重新连接到远程运行时", "disconnectedTitle": "远程运行时已断开连接", - "retryingBody": "Orca 将重试最多一分钟。如果连接恢复,此终端将恢复。", + "retryingBody": "Orca 正在自动重试。如果连接恢复,此终端将恢复。", "disconnectedBody": "自动重试已停止。重新连接以恢复此终端会话。", "reconnectButton": "重新连接" } From 39330c5acab3529af627f6810f68575e3dc65a17 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:44 -0700 Subject: [PATCH 08/77] fix(relay): retire PTYs the host proves are gone, and stop two per-poll scan storms (#17832) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(relay): stop three CPU growth terms in a long-running remote session pty.resize gated only on `managed.disposed`, which is bookkeeping rather than liveness. A shell that exits without node-pty's `onExit` leaves an undisposed entry holding a closed master fd, and UnixTerminal.resize has no fd guard, so the ioctl threw `ioctl(2) failed, EBADF` into the dispatcher's generic parse-error catch. Nothing retired the entry, so it stayed advertised and kept activePtyCount above zero -- which is what stops a relay with an unlimited grace from reaching its idle-no-ptys exit (#12423). Probe liveness with the same helper attach/listProcesses use, retire a provably dead pid, and contain an ioctl failure over a live-or-unverifiable process. processHasChildren forked `pgrep -P` per pane per inspection poll, uncached. procps-ng opens six procfs files per process to resolve one ppid, so each call cost O(host process count). Answer from the TTL-cached `ps` table the same RPC already captured for the foreground lookup (#13537). The remote AI Vault scanner had no parse cache at all, so every forced rescan re-read and re-parsed the whole transcript corpus, including files untouched for a month. Give it the mtime+size keyed memo the local scanner has (#13753). * fix(pty): invalidate the descriptor when node-pty gives up the handle (#17930) Carried forward from PR #17930, which merged into this branch. Rebased onto current main; main's newer node-pty-fd-leak test is kept as-is. * fix(ai-vault): refresh codex titles on the remote parse-cache reuse path The remote cache keys on the transcript's (mtime, size, host), but codex titles live in $CODEX_HOME/session_index.jsonl and are written after the rollout — so a cache hit froze the fallback title forever. Mirrors the local scanner's existing reuse-path refresh via a shared core. * fix(relay): publish the exit a reap performs, and rescan for close decisions Two review findings on the CPU work. reapExitedPty told only the relay-internal exit listener, so a retirement left the client's pane mounted against a session the relay had already forgotten -- the next attach answered `PTY "" not found` with nothing before it to explain why. Pre-existing on three probe paths; resize made it user-triggered. Publish the same pending-exit the natural onExit path publishes, carrying -1 ("gone, status unrecoverable"), and skip it when onExit already reported the real code. processHasChildren now answers from a 500ms TTL-cached table. That is right for pty.inspectProcess, which every tracked pane polls, but pty.hasChildProcesses gates the window-close confirmation and workspace cleanup's idle evidence -- one destructive decision per answer, where a child started inside the window would be killed unasked. Give that RPC a fresh scan; pgrep used to. * fix(relay): publish a reap's exit only on proven-exited evidence The publication is a verdict the client acts on by retiring the pane, so it must not be reachable from the disposed-record sweep, which retires off our own bookkeeping rather than the host's process table. Only ESRCH earns it. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- config/patches/node-pty@1.1.0.patch | 78 ++++++- pnpm-lock.yaml | 6 +- .../remote-session-parse-cache.test.ts | 216 ++++++++++++++++++ .../ai-vault/remote-session-parse-cache.ts | 107 +++++++++ .../remote-session-scanner-codex-index.ts | 23 +- .../remote-session-scanner-sources.ts | 13 +- src/main/ai-vault/remote-session-scanner.ts | 43 +++- .../session-scanner-codex-cached-title.ts | 27 ++- .../ai-vault/session-scanner-parse-cache.ts | 2 + .../pty/node-pty-master-fd-retirement.test.ts | 194 ++++++++++++++++ .../pty-handler-resize-stale-pty.test.ts | 176 ++++++++++++++ src/relay/pty-handler-spawn-admission.test.ts | 15 ++ src/relay/pty-handler.ts | 117 +++++++++- src/relay/pty-shell-utils.test.ts | 85 +++++++ src/relay/pty-shell-utils.ts | 37 ++- 15 files changed, 1092 insertions(+), 47 deletions(-) create mode 100644 src/main/ai-vault/remote-session-parse-cache.test.ts create mode 100644 src/main/ai-vault/remote-session-parse-cache.ts create mode 100644 src/main/pty/node-pty-master-fd-retirement.test.ts create mode 100644 src/relay/pty-handler-resize-stale-pty.test.ts diff --git a/config/patches/node-pty@1.1.0.patch b/config/patches/node-pty@1.1.0.patch index 348ce6ef7ce..ff474f7d95e 100644 --- a/config/patches/node-pty@1.1.0.patch +++ b/config/patches/node-pty@1.1.0.patch @@ -138,8 +138,34 @@ index 8c4fca9022a6d6f015bca87f61625cde2278f428..0a01730616488119aa21ef441cf3c441 process.exit(0); //# sourceMappingURL=conpty_console_list_agent.js.map \ No newline at end of file +diff --git a/lib/terminal.js b/lib/terminal.js +index e2f9bc9131077b53ebc32d207207ad82804ff185..6c63bfaaf75128d88f9a2efece13476348780cfd 100644 +--- a/lib/terminal.js ++++ b/lib/terminal.js +@@ -172,6 +172,21 @@ var Terminal = /** @class */ (function () { + this.end = function () { }; + this._writable = false; + this._readable = false; ++ // Orca: libuv closes the master fd on EIO/EOF, and the kernel may hand ++ // that number straight to the next open(2). Retire it in the same block ++ // that gives up the handle so no later ioctl can address a reused fd. ++ // Inert on Windows, where `_fd` is written once and never read back. ++ // Upstream named this mechanism in microsoft/node-pty#220 ("fd number got ++ // reattached to something else"), closed 2025-12-19 as completed after ++ // only improving the error message; #827 is still open. Windows guards in ++ // windowsPtyAgent.ts, Unix does not. Orca tracking: #18109. ++ this._fd = -1; ++ // Orca: the write stream holds its own copy of that number, so retiring ++ // `_fd` alone leaves the queued and in-flight writes addressing it. ++ // Undefined on Windows and on `UnixTerminal.open()` handles. ++ if (this._writeStream) { ++ this._writeStream.dispose(); ++ } + }; + Terminal.prototype._parseEnv = function (env) { + var keys = Object.keys(env || {}); diff --git a/lib/unixTerminal.js b/lib/unixTerminal.js -index 1ec12f796a822c78fba9ad7f6448c3987e325c23..cec8b67aef02f8199e5606a0d257088bf1865877 100644 +index 1ec12f796a822c78fba9ad7f6448c3987e325c23..d838d795ecb9ea72e3bcc31113344947c006af7e 100644 --- a/lib/unixTerminal.js +++ b/lib/unixTerminal.js @@ -28,8 +28,12 @@ var native = utils_1.loadNativeModule('pty'); @@ -157,6 +183,56 @@ index 1ec12f796a822c78fba9ad7f6448c3987e325c23..cec8b67aef02f8199e5606a0d257088b var DEFAULT_FILE = 'sh'; var DEFAULT_NAME = 'xterm'; var DESTROY_SOCKET_TIMEOUT_MS = 200; +@@ -234,6 +238,11 @@ var UnixTerminal = /** @class */ (function (_super) { + * Gets the name of the process. + */ + get: function () { ++ // Orca: tcgetpgrp on a retired fd would name whatever process now ++ // owns that descriptor, so a closed master reports the spawn file. ++ if (this._fd < 0) { ++ return this._file; ++ } + if (process.platform === 'darwin') { + var title = pty.process(this._fd); + return (title !== 'kernel_task') ? title : this._file; +@@ -250,6 +259,11 @@ var UnixTerminal = /** @class */ (function (_super) { + if (cols <= 0 || rows <= 0 || isNaN(cols) || isNaN(rows) || cols === Infinity || rows === Infinity) { + throw new Error('resizing must be done using positive cols and rows'); + } ++ // Orca: a retired master is unreachable rather than EBADF-or-worse; cols ++ // and rows stay at the last size actually applied instead of a claim. ++ if (this._fd < 0) { ++ return; ++ } + pty.resize(this._fd, cols, rows); + this._cols = cols; + this._rows = rows; +@@ -287,8 +301,15 @@ var CustomWriteStream = /** @class */ (function () { + CustomWriteStream.prototype.dispose = function () { + clearImmediate(this._writeImmediate); + this._writeImmediate = undefined; ++ // Orca: retire this stream's own copy of the master fd and drop what has ++ // not shipped, so nothing queued here reaches a reused descriptor. ++ this._fd = -1; ++ this._writeQueue.length = 0; + }; + CustomWriteStream.prototype.write = function (data) { ++ if (this._fd < 0) { ++ return; ++ } + // Writes are put in a queue and processed asynchronously in order to handle + // backpressure from the kernel buffer. + var buffer = typeof data === 'string' +@@ -304,7 +325,8 @@ var CustomWriteStream = /** @class */ (function () { + CustomWriteStream.prototype._processWriteQueue = function () { + var _this = this; + this._writeImmediate = undefined; +- if (this._writeQueue.length === 0) { ++ // Orca: an in-flight fs.write can re-enter here after dispose(). ++ if (this._fd < 0 || this._writeQueue.length === 0) { + return; + } + var task = this._writeQueue[0]; diff --git a/src/conpty_console_list_agent.ts b/src/conpty_console_list_agent.ts index 181ccabbbe9c4948a9725fb1db907a68e9de01fc..67f31facf85562b67adbfbd04ce28ddd8eeb4a79 100644 --- a/src/conpty_console_list_agent.ts diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 1abec6c0590..eac56401344 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -115,7 +115,7 @@ patchedDependencies: '@xterm/addon-webgl@0.20.0-beta.299': 94687e89a0115e6e6aa102837f986debdc029c091527ee5eb4a4e17ceaf9473e '@xterm/xterm@6.1.0-beta.303': 98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d lint-staged@16.4.0: 7333b3837f80a7fbd045964db6d76ba4fc118e49134bdbabb00585b6b7b60673 - node-pty@1.1.0: 40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e + node-pty@1.1.0: e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa importers: @@ -156,7 +156,7 @@ importers: version: 3.3.1 node-pty: specifier: ^1.1.0 - version: 1.1.0(patch_hash=40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e) + version: 1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa) posthog-node: specifier: ^5.33.3 version: 5.33.3 @@ -12194,7 +12194,7 @@ snapshots: node-int64@0.4.0: {} - node-pty@1.1.0(patch_hash=40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e): + node-pty@1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa): dependencies: node-addon-api: 7.1.1 diff --git a/src/main/ai-vault/remote-session-parse-cache.test.ts b/src/main/ai-vault/remote-session-parse-cache.test.ts new file mode 100644 index 00000000000..446563c0652 --- /dev/null +++ b/src/main/ai-vault/remote-session-parse-cache.test.ts @@ -0,0 +1,216 @@ +import { beforeEach, describe, expect, it } from 'vitest' +import type { FileReadResult } from '../providers/types' +import { getRemoteHostPlatform } from '../ssh/ssh-remote-platform' +import { resetRemoteSessionParseCacheForTests } from './remote-session-parse-cache' +import { scanRemoteAiVaultSessions } from './remote-session-scanner' +import { MemoryRemoteProvider, jsonLines } from './remote-session-scanner-test-fixtures' + +/** + * Counts whole-transcript reads, which is the cost #13753 is about. Codex's + * per-scan `session_index.jsonl` title lookup is one small file and is not part + * of the corpus term, so it is excluded rather than asserted on. + */ +class CountingRemoteProvider extends MemoryRemoteProvider { + readonly readFilePaths: string[] = [] + + override async readFile(filePath: string): Promise { + if (filePath.includes('/sessions/')) { + this.readFilePaths.push(filePath) + } + return await super.readFile(filePath) + } +} + +function transcript(sessionId: string, title: string, timestamp: string): string { + return jsonLines([ + { + timestamp, + type: 'session_meta', + payload: { id: sessionId, cwd: '/home/ada/repo' } + }, + { + timestamp, + type: 'response_item', + payload: { type: 'message', role: 'user', content: [{ type: 'text', text: title }] } + } + ]) +} + +function scan(provider: CountingRemoteProvider): ReturnType { + return scanRemoteAiVaultSessions({ + provider, + executionHostId: 'ssh:dev-box', + remoteHome: '/home/ada', + hostPlatform: getRemoteHostPlatform('linux-x64') + }) +} + +describe('remote AI Vault transcript re-reads', () => { + beforeEach(() => { + resetRemoteSessionParseCacheForTests() + }) + + it('does not re-read an unchanged corpus on the next scan', async () => { + const provider = new CountingRemoteProvider() + for (const day of ['07/07', '07/25', '08/10']) { + provider.addFile( + `/home/ada/.codex/sessions/2026/${day}/rollout-${day.replace('/', '')}.jsonl`, + transcript( + `session-${day.replace('/', '')}`, + `Work from ${day}`, + '2026-07-07T01:00:00.000Z' + ), + 1_000 + ) + } + + const first = await scan(provider) + expect(first.sessions).toHaveLength(3) + expect(provider.readFilePaths).toHaveLength(3) + + provider.readFilePaths.length = 0 + const second = await scan(provider) + + // Historical transcripts are immutable; a second pass must cost zero reads. + expect(provider.readFilePaths).toEqual([]) + expect(second.sessions.map((session) => session.title)).toEqual( + first.sessions.map((session) => session.title) + ) + }) + + it('re-reads a transcript that actually changed', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-live.jsonl' + provider.addFile( + path, + transcript('live-session', 'First prompt', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + + await scan(provider) + provider.readFilePaths.length = 0 + + provider.addFile( + path, + transcript('live-session', 'Second prompt', '2026-08-31T02:00:00.000Z'), + 2_000 + ) + const result = await scan(provider) + + expect(provider.readFilePaths).toEqual([path]) + expect(result.sessions[0]?.title).toBe('Second prompt') + }) + + it('re-reads when only the size changed under an unchanged mtime', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-grown.jsonl' + provider.addFile(path, transcript('grown-session', 'Short', '2026-08-31T01:00:00.000Z'), 1_000) + + await scan(provider) + provider.readFilePaths.length = 0 + + provider.addFile( + path, + transcript( + 'grown-session', + 'A much longer first prompt than before', + '2026-08-31T01:00:00.000Z' + ), + 1_000 + ) + const result = await scan(provider) + + expect(provider.readFilePaths).toEqual([path]) + expect(result.sessions[0]?.title).toBe('A much longer first prompt than before') + }) + + // Codex names threads in $CODEX_HOME/session_index.jsonl asynchronously, after + // the rollout's last append — so the transcript's mtime+size never changes to + // signal it. That file sits outside `sessions/`, hence outside the read count. + it('picks up a session_index title written after the transcript was cached', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-named-later.jsonl' + provider.addFile( + path, + transcript('named-later-session', 'First prompt', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + provider.addFile( + '/home/ada/.codex/session_index.jsonl', + jsonLines([{ id: 'some-other-session', thread_name: 'Unrelated thread' }]), + 1_000 + ) + + const first = await scan(provider) + expect(first.sessions[0]?.title).toBe('First prompt') + + provider.readFilePaths.length = 0 + provider.addFile( + '/home/ada/.codex/session_index.jsonl', + jsonLines([ + { id: 'some-other-session', thread_name: 'Unrelated thread' }, + { id: 'named-later-session', thread_name: 'Named by Codex after the fact' } + ]), + 2_000 + ) + const second = await scan(provider) + + expect(second.sessions[0]?.title).toBe('Named by Codex after the fact') + // The #13753 win is preserved: the index is read, the transcript is not. + expect(provider.readFilePaths).toEqual([]) + }) + + it('does not serve a cached parse to a different execution host', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-host.jsonl' + provider.addFile( + path, + transcript('host-session', 'Host scoped', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + + await scan(provider) + provider.readFilePaths.length = 0 + + const other = await scanRemoteAiVaultSessions({ + provider, + executionHostId: 'ssh:other-box', + remoteHome: '/home/ada', + hostPlatform: getRemoteHostPlatform('linux-x64') + }) + + expect(provider.readFilePaths).toEqual([path]) + expect(other.sessions[0]?.executionHostId).toBe('ssh:other-box') + }) + + it('does not cache a read that failed', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-flaky.jsonl' + provider.addFile( + path, + transcript('flaky-session', 'Recovered', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + + let failNextRead = true + const originalReadFile = provider.readFile.bind(provider) + provider.readFile = async (filePath: string): Promise => { + if (failNextRead && filePath === path) { + failNextRead = false + provider.readFilePaths.push(filePath) + throw new Error('EIO: transient relay read failure') + } + return await originalReadFile(filePath) + } + + const failed = await scan(provider) + expect(failed.sessions).toEqual([]) + expect(failed.issues).toHaveLength(1) + + provider.readFilePaths.length = 0 + const recovered = await scan(provider) + + expect(provider.readFilePaths).toEqual([path]) + expect(recovered.sessions[0]?.title).toBe('Recovered') + }) +}) diff --git a/src/main/ai-vault/remote-session-parse-cache.ts b/src/main/ai-vault/remote-session-parse-cache.ts new file mode 100644 index 00000000000..7d8b0ebe2c7 --- /dev/null +++ b/src/main/ai-vault/remote-session-parse-cache.ts @@ -0,0 +1,107 @@ +import type { AiVaultSession } from '../../shared/ai-vault-types' +import type { RemoteScannerContext, RemoteSessionCandidate } from './remote-session-scanner-types' + +// Matches the local scanner's cap. The relay sidecar is forked with +// --max-old-space-size=384, and a retained session row is a title, a preview +// window and counters — orders of magnitude smaller than the transcript it was +// parsed from, which is what the cache stops us re-reading. +const MAX_CACHE_ENTRIES = 4096 + +type RemoteSessionParseCacheEntry = { + mtimeMs: number + sizeBytes: number | null + hostKey: string + session: AiVaultSession | null +} + +// Module scope so it outlives one scan: the sidecar is retired only after 10 +// idle minutes, so it spans many passes of a 30s cadence. +const cache = new Map() + +export type RemoteSessionParseStats = { reused: number; parsed: number } + +export function createRemoteSessionParseStats(): RemoteSessionParseStats { + return { reused: 0, parsed: 0 } +} + +export function resetRemoteSessionParseCacheForTests(): void { + cache.clear() +} + +/** Identity of the host a parse result belongs to; a result is not portable across either field. */ +export function remoteSessionParseHostKey(context: RemoteScannerContext): string { + return `${context.executionHostId}\u0000${context.hostPlatform.relayPlatform}` +} + +function storeEntry(path: string, entry: RemoteSessionParseCacheEntry): void { + cache.delete(path) + cache.set(path, entry) + if (cache.size > MAX_CACHE_ENTRIES) { + const oldest = cache.keys().next() + if (!oldest.done) { + cache.delete(oldest.value) + } + } +} + +/** + * Parse a remote transcript, reusing the previous result when the file is + * provably unchanged. + * + * Why this exists: the remote scanner had no cache of any kind, so every pass + * re-read and re-parsed the whole corpus — up to 3000 whole-file reads, GBs of + * JSONL, including July transcripts that had not changed in a month — which is + * what pegged the relay host on the renderer's 30s forced-rescan cadence + * (#13753). The local scanner has had `parseAgentSessionFileCached` for exactly + * this reason; this is its remote counterpart. + * + * `(mtimeMs, sizeBytes)` is a sound validity key here because discovery already + * folds a source's `contentDependencyPath` stat into both fields + * (remote-session-scanner-discovery.ts), so a metadata-only transcript whose + * companion file changed still looks changed. Sources whose parse reads a file + * discovery does not stat — Codex looks its title up in `session_index.jsonl` — + * are not covered by that key and pass `refreshReusedSession` to re-derive the + * uncovered part without touching the transcript. + * + * Only a completed parse is stored. A read that threw stays uncached so a + * transient filesystem failure cannot pin a wrong answer for the corpus's life. + */ +export async function parseRemoteSessionFileCached(args: { + candidate: RemoteSessionCandidate + hostKey: string + parse: () => Promise + // Applied to a reused session only; must not re-read the transcript. + refreshReusedSession?: (session: AiVaultSession) => Promise + stats?: RemoteSessionParseStats +}): Promise { + const { file } = args.candidate + const entry = cache.get(file.path) + const unchanged = + entry !== undefined && + entry.hostKey === args.hostKey && + entry.mtimeMs === file.mtimeMs && + (entry.sizeBytes === null || file.sizeBytes === undefined || entry.sizeBytes === file.sizeBytes) + if (unchanged) { + if (args.stats) { + args.stats.reused++ + } + if (entry.session && args.refreshReusedSession) { + entry.session = await args.refreshReusedSession(entry.session) + } + // Refresh recency without re-parsing so the LRU evicts cold paths first. + storeEntry(file.path, entry) + return entry.session + } + + const session = await args.parse() + if (args.stats) { + args.stats.parsed++ + } + storeEntry(file.path, { + mtimeMs: file.mtimeMs, + sizeBytes: file.sizeBytes ?? null, + hostKey: args.hostKey, + session + }) + return session +} diff --git a/src/main/ai-vault/remote-session-scanner-codex-index.ts b/src/main/ai-vault/remote-session-scanner-codex-index.ts index 8f5ec3c1bbb..89951110cdf 100644 --- a/src/main/ai-vault/remote-session-scanner-codex-index.ts +++ b/src/main/ai-vault/remote-session-scanner-codex-index.ts @@ -3,10 +3,31 @@ import { joinRemotePath } from '../ssh/ssh-remote-platform' import { extractString, normalizeTitleText, parseJsonObject } from './session-scanner-values' import { remoteSessionContentLines } from './remote-session-content-lines' import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation' -import type { RemoteSessionFilesystemProvider } from './remote-session-scanner-types' +import type { + RemoteScannerContext, + RemoteSessionFilesystemProvider +} from './remote-session-scanner-types' const CODEX_SESSION_INDEX_FILE = 'session_index.jsonl' +// One index read per CODEX_HOME per scan (`context.titleCaches` is scan-scoped); +// used both by the transcript parse and by the parse cache's reuse path. +export function remoteCodexIndexedTitleReader( + codexHome: string, + context: RemoteScannerContext +): (sessionId: string) => Promise { + return async (sessionId) => + ( + await remoteCodexIndexTitles({ + provider: context.provider, + codexHome, + hostPlatform: context.hostPlatform, + titleCaches: context.titleCaches, + signal: context.signal + }) + ).get(sessionId) ?? null +} + export async function remoteCodexIndexTitles(args: { provider: RemoteSessionFilesystemProvider codexHome: string diff --git a/src/main/ai-vault/remote-session-scanner-sources.ts b/src/main/ai-vault/remote-session-scanner-sources.ts index 1fd0ca30a61..2c21a93db9a 100644 --- a/src/main/ai-vault/remote-session-scanner-sources.ts +++ b/src/main/ai-vault/remote-session-scanner-sources.ts @@ -16,7 +16,7 @@ import { partitionSubagentTranscriptPaths } from './session-scanner-subagent-tra import { partitionOmpSubagentTranscriptPaths } from './session-scanner-omp-subagent-transcripts' import type { FileWithMtime } from './session-scanner-types' import { normalizeAgentSessionsDir } from './session-scanner-values' -import { remoteCodexIndexTitles } from './remote-session-scanner-codex-index' +import { remoteCodexIndexedTitleReader } from './remote-session-scanner-codex-index' import { remoteClineSource } from './remote-session-scanner-cline-source' import type { RemoteParserOptions, @@ -216,16 +216,7 @@ function remoteCodexSources( executionHostId: context.executionHostId, executionHostPlatform: context.hostPlatform.os, signal: context.signal, - readIndexedTitle: async (sessionId) => - ( - await remoteCodexIndexTitles({ - provider: context.provider, - codexHome, - hostPlatform, - titleCaches: context.titleCaches, - signal: context.signal - }) - ).get(sessionId) ?? null + readIndexedTitle: remoteCodexIndexedTitleReader(codexHome, context) }) })) } diff --git a/src/main/ai-vault/remote-session-scanner.ts b/src/main/ai-vault/remote-session-scanner.ts index 568b42c27cd..7b586eb9196 100644 --- a/src/main/ai-vault/remote-session-scanner.ts +++ b/src/main/ai-vault/remote-session-scanner.ts @@ -12,6 +12,11 @@ import { dedupeCodexRolloutFileAliases, dedupeCodexSessionsBySessionId } from './codex-session-root-dedup' +import { + parseRemoteSessionFileCached, + remoteSessionParseHostKey +} from './remote-session-parse-cache' +import { remoteCodexIndexedTitleReader } from './remote-session-scanner-codex-index' import { discoverRemoteSourceCandidates } from './remote-session-scanner-discovery' import { remoteSessionSources } from './remote-session-scanner-sources' import type { @@ -25,6 +30,7 @@ import { errorMessage } from './session-scanner-values' import { mapRemoteScanBatches } from './remote-session-scan-batching' import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation' import { recordSessionScanIssue } from './session-scan-issues' +import { refreshCodexTitleFromIndex } from './session-scanner-codex-cached-title' import { limitRemoteScanFilesystemConcurrency } from './remote-session-scan-concurrency' import { aiVaultScanLimit } from '../../shared/ai-vault-session-depth' @@ -224,12 +230,21 @@ async function parseRemoteSessionCandidate( ): Promise { try { throwIfAiVaultScanCancelled(context.signal) - const read = await context.provider.readFile(candidate.file.path) - throwIfAiVaultScanCancelled(context.signal) - if (read.isBinary) { - return null - } - const session = await candidate.source.parse(candidate.file, read.content, context) + // The read is inside the cached parse: an unchanged transcript must not be + // pulled off the remote disk at all, which is the whole cost of #13753. + const session = await parseRemoteSessionFileCached({ + candidate, + hostKey: remoteSessionParseHostKey(context), + parse: async () => { + const read = await context.provider.readFile(candidate.file.path) + throwIfAiVaultScanCancelled(context.signal) + if (read.isBinary) { + return null + } + return await candidate.source.parse(candidate.file, read.content, context) + }, + refreshReusedSession: reusedCodexTitleRefresh(candidate, context) + }) throwIfAiVaultScanCancelled(context.signal) // Mirror the local rule: every session carries its sibling subagent // transcript count (row badge; recoverable signal at zero turns). The @@ -251,6 +266,22 @@ async function parseRemoteSessionCandidate( } } +// Codex thread names live in `/session_index.jsonl`, not the +// rollout, and are written after it — so a transcript-keyed cache hit would +// pin the fallback title forever. Local counterpart: +// session-scanner-parse-cache.ts's reuse path. +function reusedCodexTitleRefresh( + candidate: RemoteSessionCandidate, + context: RemoteScannerContext +): ((session: AiVaultSession) => Promise) | undefined { + const codexHome = candidate.source.agent === 'codex' ? candidate.source.codexHome : undefined + if (!codexHome) { + return undefined + } + const readIndexedTitle = remoteCodexIndexedTitleReader(codexHome, context) + return (session) => refreshCodexTitleFromIndex(session, readIndexedTitle) +} + function mergeRemoteSessions( cappedSessions: AiVaultSession[], scopeSessions: AiVaultSession[] diff --git a/src/main/ai-vault/session-scanner-codex-cached-title.ts b/src/main/ai-vault/session-scanner-codex-cached-title.ts index f356e30b890..ed4045bd1da 100644 --- a/src/main/ai-vault/session-scanner-codex-cached-title.ts +++ b/src/main/ai-vault/session-scanner-codex-cached-title.ts @@ -2,14 +2,29 @@ import type { AiVaultSession } from '../../shared/ai-vault-types' import type { SessionFileCandidate } from './session-scanner-types' import { readCodexSessionIndexTitle } from './session-scanner-codex-title-index' -export async function refreshCachedCodexTitle( +/** + * Codex names a thread in /session_index.jsonl asynchronously, + * after the rollout exists — often after the rollout's last append. A parse + * cache keyed on the transcript's own mtime/size therefore freezes the fallback + * title forever, so every reuse path re-derives it through here. + * + * Both caches share this: `session-scanner-parse-cache.ts` (local disk, via + * `refreshCachedCodexTitle`) and `remote-session-parse-cache.ts` (relay + * provider, whose reader lives in `remote-session-scanner-codex-index.ts`). + */ +export async function refreshCodexTitleFromIndex( + session: AiVaultSession, + readIndexedTitle: (sessionId: string) => Promise +): Promise { + const title = await readIndexedTitle(session.sessionId) + return title && title !== session.title ? { ...session, title } : session +} + +export function refreshCachedCodexTitle( candidate: SessionFileCandidate, session: AiVaultSession ): Promise { - const title = await readCodexSessionIndexTitle( - candidate.file.path, - candidate.codexHome, - session.sessionId + return refreshCodexTitleFromIndex(session, (sessionId) => + readCodexSessionIndexTitle(candidate.file.path, candidate.codexHome, sessionId) ) - return title && title !== session.title ? { ...session, title } : session } diff --git a/src/main/ai-vault/session-scanner-parse-cache.ts b/src/main/ai-vault/session-scanner-parse-cache.ts index 27fc5e0f950..139940ba005 100644 --- a/src/main/ai-vault/session-scanner-parse-cache.ts +++ b/src/main/ai-vault/session-scanner-parse-cache.ts @@ -209,6 +209,8 @@ export async function parseAgentSessionFileCached( entry.session = { ...entry.session, subagentTranscriptCount } } } + // Codex titles come from session_index.jsonl, which mtime+size can't see. + // Remote counterpart: remote-session-scanner.ts's reusedCodexTitleRefresh. if (entry.session && candidate.agent === 'codex') { entry.session = await refreshCachedCodexTitle(candidate, entry.session) } diff --git a/src/main/pty/node-pty-master-fd-retirement.test.ts b/src/main/pty/node-pty-master-fd-retirement.test.ts new file mode 100644 index 00000000000..a0b537de1ca --- /dev/null +++ b/src/main/pty/node-pty-master-fd-retirement.test.ts @@ -0,0 +1,194 @@ +import * as pty from 'node-pty' +import { describe, expect, it } from 'vitest' + +/** + * node-pty hands the master fd to libuv, which closes it on EIO/EOF, but upstream + * never invalidated `_fd`, and none of the three fd-addressed surfaces consulted + * anything: `resize()`, the `process` getter, and `CustomWriteStream`, which holds + * its own plain-number copy of the fd taken at spawn. Orca's patch retires all + * three in the same block that gives up the handle + * (config/patches/node-pty@1.1.0.patch). + * + * Scope: this narrows the window, it does not close it. libuv closes the fd + * synchronously inside `uv_close`, before the JS `'close'` that runs `_close()`, + * so callers still need their own liveness verdict for that tick — and a relay + * host installs node-pty from npm, where this patch is not applied at all. + */ + +const POSIX_SHELL = '/bin/sh' + +function spawnPty(command: string, cols = 80, rows = 24): pty.IPty { + return pty.spawn(POSIX_SHELL, ['-c', command], { + name: 'xterm-256color', + cols, + rows, + cwd: process.cwd(), + env: { ...process.env } + }) +} + +function masterFd(term: pty.IPty): number { + return (term as unknown as { fd: number }).fd +} + +type CustomWriteStream = { _fd: number; _writeQueue: unknown[]; write(data: string): void } + +/** + * `Terminal._close()` already shadows `terminal.write` with a no-op, so the stream + * itself is the surface that still reached the fd: a residual `_writeQueue` and an + * in-flight `fs.write` both re-enter it after the close. + */ +function writeStream(term: pty.IPty): CustomWriteStream { + return (term as unknown as { _writeStream: CustomWriteStream })._writeStream +} + +/** + * Run `command` to completion and let node-pty finish giving up the master. + * + * `destroy: false` exercises only the EIO/EOF read-error path, which reaches + * `_close()` without ever calling `destroy()` — the path the exit of a shell + * actually takes, and the one the write stream was previously never told about. + */ +async function retiredPty( + command = 'exit 0', + { destroy = true }: { destroy?: boolean } = {} +): Promise<{ term: pty.IPty; spawnFd: number }> { + const term = spawnPty(command) + const spawnFd = masterFd(term) + await new Promise((resolve) => { + term.onExit(() => resolve()) + }) + if (destroy) { + ;(term as unknown as { destroy?: () => void }).destroy?.() + } + await new Promise((resolve) => setTimeout(resolve, 400)) + return { term, spawnFd } +} + +// Windows never reaches this code: WindowsTerminal.resize goes through the conpty +// agent and reads no fd, so the sentinel is written and never consulted there. +const describeOnPosix = process.platform === 'win32' ? describe.skip : describe + +describeOnPosix('node-pty master fd retirement', () => { + it('invalidates the descriptor once it gives up the handle', async () => { + const { term, spawnFd } = await retiredPty() + + expect(spawnFd).toBeGreaterThanOrEqual(0) + expect(masterFd(term)).toBe(-1) + }, 15000) + + it('answers a resize past retirement without issuing the ioctl', async () => { + const { term } = await retiredPty() + + // Pre-patch this threw `ioctl(2) failed, EBADF` out of whatever called it. + expect(() => term.resize(200, 50)).not.toThrow() + // Geometry stays at the last size actually applied rather than claiming one + // that no descriptor ever received. + expect([term.cols, term.rows]).toEqual([80, 24]) + }, 15000) + + it('retires the write stream fd on _close(), not only on destroy()', async () => { + const { term } = await retiredPty('exit 0', { destroy: false }) + + // The stream copied the fd number at spawn, so `Terminal._fd = -1` alone + // leaves it addressing a descriptor the kernel may already have reissued. + expect(writeStream(term)._fd).toBe(-1) + + writeStream(term).write('x') + expect(writeStream(term)._writeQueue).toHaveLength(0) + }, 15000) + + it('names the spawn file rather than tcgetpgrp on a retired descriptor', async () => { + const { term } = await retiredPty() + + expect(term.process).toBe(POSIX_SHELL) + }, 15000) +}) + +// Linux frees the master synchronously enough that the very next pty is handed the +// same descriptor number every time, which makes the reuse hazard directly +// observable rather than a race to reproduce. +// +// Which is also the constraint on writing a case here: the kernel hands out the +// lowest free number, so a case that returns while its live pty is still open +// leaks that descriptor into the next case's premise as an off-by-one. Await the +// exit, never a fixed sleep. +const describeOnLinux = process.platform === 'linux' ? describe : describe.skip + +describeOnLinux('node-pty master fd reuse', () => { + it('cannot resize a live pty handed the retired descriptor number', async () => { + const { term: retired, spawnFd } = await retiredPty() + + const live = spawnPty('sleep 1; stty size') + let output = '' + live.onData((data) => { + output += data + }) + try { + // The premise of this test: the kernel really did reissue the number. If it + // stops holding, the assertion below would pass for the wrong reason. + expect(masterFd(live)).toBe(spawnFd) + + // Pre-patch this reached TIOCSWINSZ on `live`'s master and silently resized + // a terminal it has no relationship to — no error, nothing for a liveness + // probe of the retired pid to observe. + retired.resize(200, 50) + + await new Promise((resolve) => { + live.onExit(() => resolve()) + }) + expect(output.trim()).toBe('24 80') + } finally { + live.kill() + } + }, 15000) + + it('cannot write into a live pty handed the retired descriptor number', async () => { + const { term: retired, spawnFd } = await retiredPty('exit 0', { destroy: false }) + + const live = spawnPty('sleep 1') + let output = '' + live.onData((data) => { + output += data + }) + try { + expect(masterFd(live)).toBe(spawnFd) + + // Pre-patch this fs.write reached `live`'s master, and the line discipline + // echoed it straight back: a retired pane's bytes landing in an unrelated + // terminal. Nothing has to read them for the leak to be observable. + writeStream(retired).write('leak\r') + + // Await the exit rather than sleeping: a fixed wait leaves this descriptor + // open into the next case, whose `expect(masterFd(live)).toBe(spawnFd)` + // premise then sees the kernel hand out the lower number this pty was still + // holding. That is what broke `node 24 1/8` on Linux, where alone among the + // platforms these cases actually run. + await new Promise((resolve) => { + live.onExit(() => resolve()) + }) + expect(output).not.toContain('leak') + } finally { + live.kill() + } + }, 15000) + + it('does not name a live pty foreground process off the retired descriptor', async () => { + const { term: retired, spawnFd } = await retiredPty() + + // `exec` replaces the shell, so the foreground pgrp's cmdline is distinct + // from the file this pty was spawned with. + const live = spawnPty('exec sleep 5') + try { + expect(masterFd(live)).toBe(spawnFd) + // Let the shell finish exec'ing, or its own cmdline is still the fallback. + await new Promise((resolve) => setTimeout(resolve, 300)) + + // Pre-patch this read tcgetpgrp off `live`'s master and reported `sleep`, + // attributing an unrelated pane's process to a pty that had already exited. + expect(retired.process).toBe(POSIX_SHELL) + } finally { + live.kill() + } + }, 15000) +}) diff --git a/src/relay/pty-handler-resize-stale-pty.test.ts b/src/relay/pty-handler-resize-stale-pty.test.ts new file mode 100644 index 00000000000..0dcb6b56497 --- /dev/null +++ b/src/relay/pty-handler-resize-stale-pty.test.ts @@ -0,0 +1,176 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { mockPtySpawn, mockPtyInstance, mockCreateShellPromptReadinessProbe } = vi.hoisted(() => ({ + mockPtySpawn: vi.fn(), + mockCreateShellPromptReadinessProbe: vi.fn(), + mockPtyInstance: { + pid: process.pid, + onData: vi.fn(), + onExit: vi.fn(), + write: vi.fn(), + resize: vi.fn(), + kill: vi.fn(), + clear: vi.fn(), + pause: vi.fn(), + resume: vi.fn() + } +})) + +vi.mock('node-pty', () => ({ + spawn: mockPtySpawn +})) + +vi.mock('../main/pty/posix-pty-process-groups', () => ({ + forceKillPosixPtyProcessGroups: vi.fn((_pid: number, fallback: () => void) => fallback()) +})) + +vi.mock('../main/shell-prompt-readiness-probe', () => ({ + createShellPromptReadinessProbe: mockCreateShellPromptReadinessProbe +})) + +import type { PtyHandler } from './pty-handler' +import * as ptyShellUtils from './pty-shell-utils' +import { beginPtyHandlerTest, endPtyHandlerTest, testPtyId } from './pty-handler-test-harness' +import type { MockDispatcher } from './pty-handler-test-harness' + +const PTY_1 = testPtyId(1) +const STALE_PID = 424_242 + +/** + * node-pty's native `pty.resize` error when the ioctl reaches a closed master. + * + * Orca's node-pty patch retires `_fd` when it gives up the master, so a patched + * handle answers a late resize with a no-op. That leaves this handler two cases + * it still has to contain: the tick between libuv closing the fd and node-pty's + * own handler observing it, and a relay host, which installs node-pty from npm + * and has no such guard. + */ +function ebadfResize(): never { + throw new Error('ioctl(2) failed, EBADF') +} + +describe('PtyHandler.resize against a stale PTY handle', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + let resize: ReturnType + + beforeEach(async () => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + resize = vi.fn() + // A shell that exited without node-pty producing `onExit`: the record is + // still in the pool and undisposed, but the master behind it is gone. + mockPtySpawn.mockReturnValue({ ...mockPtyInstance, pid: STALE_PID, resize }) + await dispatcher.callRequest('pty.spawn', {}) + expect(handler.activePtyCount).toBe(1) + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('retires an entry whose pid the host proves is gone, instead of issuing the ioctl', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(false) + resize.mockImplementation(ebadfResize) + + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + ).not.toThrow() + + expect(resize).not.toHaveBeenCalled() + // The record must leave the pool: while it stays, the relay keeps + // advertising a dead shell and `activePtyCount` never reaches zero, so a + // relay configured with an unlimited grace never reaches its idle exit. + expect(handler.activePtyCount).toBe(0) + }) + + it('contains an ioctl failure without re-classifying liveness, and keeps the record', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(true) + resize.mockImplementation(ebadfResize) + const stderr = vi.spyOn(process.stderr, 'write').mockReturnValue(true) + + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + ).not.toThrow() + // Repeats must stay contained too — this is the notification the client + // re-sends on every reconnect and every window resize. + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 90, rows: 30 }) + ).not.toThrow() + + // Loss of an fd observed the handle, not the host that owns the pid, so it is + // `unverifiable` and the claim is retained. Only the probe above retires. + expect(handler.activePtyCount).toBe(1) + expect(stderr.mock.calls.map(([line]) => String(line)).join('')).toContain( + 'ioctl(2) failed, EBADF' + ) + }) + + it('retires the entry when the pid goes absent between the pre-probe and the ioctl', () => { + // The race the catch-block re-probe exists for, and the one a constant + // liveness mock cannot express: alive when the pre-probe asks, gone by the + // time the ioctl fails. libuv closes the master synchronously inside + // `uv_close`, so this window opens before any JS guard can be set — and on + // a relay host, where node-pty comes from npm, there is no JS guard at all. + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValueOnce(true).mockReturnValueOnce(false) + resize.mockImplementation(ebadfResize) + const stderr = vi.spyOn(process.stderr, 'write').mockReturnValue(true) + + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + ).not.toThrow() + + expect(resize).toHaveBeenCalledTimes(1) + expect(handler.activePtyCount).toBe(0) + // Proven `exited` retires silently; only live-or-unverifiable is reported. + expect(stderr).not.toHaveBeenCalled() + }) + + it('publishes the exit so the client stops holding a pane on a retired session', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(false) + + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + + // Retiring the record without this leaves the pane mounted against a + // session the relay has already forgotten: the next attach answers + // `PTY "" not found` and nothing before it explained why. + expect(dispatcher._notifications).toContainEqual({ + method: 'pty.exit', + params: { id: PTY_1, code: -1, incarnationId: expect.any(String) } + }) + }) + + it('does not republish an exit node-pty already reported', async () => { + // The listing sweep also reaps entries the natural `onExit` left behind; a + // second `pty.exit` would hand the client a duplicate carrying -1 in place + // of the real status. That sweep retires off `managed.disposed`, which is + // bookkeeping rather than liveness, so it never publishes a verdict at all. + const onExit = mockPtyInstance.onExit.mock.calls.at(-1)?.[0] as (e: { + exitCode: number + }) => void + onExit({ exitCode: 7 }) + await vi.runAllTimersAsync() + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(false) + + await dispatcher.callRequest('pty.listProcesses', {}) + + const exits = dispatcher._notifications.filter( + (notification) => notification.method === 'pty.exit' + ) + expect(exits).toHaveLength(1) + expect(exits[0]?.params).toMatchObject({ id: PTY_1, code: 7 }) + }) + + it('still resizes a live PTY, with the clamped geometry', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(true) + + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 4_000, rows: 40 }) + + expect(resize).toHaveBeenCalledWith(500, 40) + expect(handler.activePtyCount).toBe(1) + }) +}) diff --git a/src/relay/pty-handler-spawn-admission.test.ts b/src/relay/pty-handler-spawn-admission.test.ts index 728f883d3b1..20b64d16b1c 100644 --- a/src/relay/pty-handler-spawn-admission.test.ts +++ b/src/relay/pty-handler-spawn-admission.test.ts @@ -87,6 +87,21 @@ describe('PtyHandler', () => { expect(notifMethods).not.toContain('pty.ackData') }) + it('rescans the process table for a close decision but not for a poll', async () => { + const hasChildren = vi.mocked(ptyShellUtils.processHasChildren) + const { id } = (await spawnPty({ cols: 80, rows: 24 })) as { id: string } + hasChildren.mockClear() + + await dispatcher.callRequest('pty.inspectProcess', { id }) + // The poll shares the TTL-cached table the foreground lookup already took. + expect(hasChildren).toHaveBeenLastCalledWith(mockPtyInstance.pid) + + await dispatcher.callRequest('pty.hasChildProcesses', { id }) + // This RPC only ever gates a destructive decision (window close, workspace + // cleanup), so it has to see a child started inside the 500ms window. + expect(hasChildren).toHaveBeenLastCalledWith(mockPtyInstance.pid, { fresh: true }) + }) + it('rejects strict process inspection for a missing relay PTY', async () => { await expect(dispatcher.callRequest('pty.inspectProcess', { id: 'missing' })).rejects.toThrow( 'terminal_gone' diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 25945203878..b7fe5bb160b 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -1990,8 +1990,7 @@ export class PtyHandler { } // Why: verify liveness because shells can exit without node-pty onExit. - if (managed.pty.pid && !isProcessAlive(managed.pty.pid)) { - this.reapExitedPty(managed) + if (this.reapPtyProvenExited(managed)) { throw new Error(`PTY "${id}" not found`) } @@ -2109,8 +2108,39 @@ export class PtyHandler { const cols = Math.max(1, Math.min(500, Math.floor(Number(params.cols) || 80))) const rows = Math.max(1, Math.min(500, Math.floor(Number(params.rows) || 24))) const managed = this.ptys.get(id) - if (managed && !managed.disposed) { + if (!managed || managed.disposed) { + return + } + // Why probe (same probe attach() and listProcesses() run): a shell that + // exited without node-pty's `onExit` leaves an undisposed entry behind, and + // while it stays the relay keeps advertising a dead shell and keeps holding + // `activePtyCount` above zero, which is what stops a relay with + // `relayGracePeriodSeconds: 0` from ever reaching its idle-no-ptys exit + // (#12423). This is retirement, not ioctl safety: only ESRCH from the host + // that owns the pid is evidence of `exited`. + if (this.reapPtyProvenExited(managed)) { + return + } + // The patched node-pty retires `_fd` in the same block that gives up the + // master (config/patches/node-pty@1.1.0.patch), which makes a resize past + // that point a no-op rather than a TIOCSWINSZ aimed at a reused descriptor. + // That covers only part of the window and does not cover this process at + // all: libuv closes the fd synchronously inside `uv_close`, before the JS + // `'close'` that runs `_close()`, and a relay host installs node-pty from + // npm, where the patch is not applied. So the catch below stays. + try { managed.pty.resize(cols, rows) + } catch (err) { + // A failed ioctl observed the handle, not the host's process table, so on + // its own it is `unverifiable`. Re-probe: a now-absent pid retires the + // entry, anything else keeps it and is contained here rather than + // escaping as a parse error on every later resize. + if (this.reapPtyProvenExited(managed)) { + return + } + process.stderr.write( + `[pty-handler] resize failed for PTY ${id} whose process is still live or unverifiable: ${err instanceof Error ? err.message : String(err)}\n` + ) } } @@ -2249,9 +2279,7 @@ export class PtyHandler { if (this.ptys.get(managed.id) !== managed || managed.disposed) { return } - const pid = managed.pty.pid - if (pid && !isProcessAlive(pid)) { - this.reapExitedPty(managed) + if (this.reapPtyProvenExited(managed)) { return } if (attemptsRemaining <= 0) { @@ -2278,12 +2306,22 @@ export class PtyHandler { managed.reapTimer = timer } - /** Retire every record for a PTY whose process is proven gone. Shared by the attach probe, the - * listing probe and the post-shutdown sweep so the three cannot drift on what "gone" retires. */ - private reapExitedPty(managed: ManagedPty): void { + /** + * Retire every record for a PTY whose process is proven gone. Shared by the attach probe, the + * listing probe and the post-shutdown sweep so the three cannot drift on what "gone" retires. + * + * `evidence` is not decoration: `exited` publishes a verdict to the client, and only ESRCH from + * the host that owns the pid earns it. The disposed-record sweep retires off our own + * bookkeeping, which says we tore the record down — not that the shell died — so it stays + * silent (docs/reference/ssh-execution-boundary.md). + */ + private reapExitedPty(managed: ManagedPty, evidence: 'exited' | 'record-torn-down'): void { managed.physicalExit?.markExited() this.releaseRelayIngress(managed) this.flushPtyOutput(managed.id) + if (evidence === 'exited') { + this.publishReapedExit(managed) + } this.notifyExitListener(managed) this.agentSessionOwners.release(managed.id) disposeManagedPty(managed) @@ -2291,6 +2329,53 @@ export class PtyHandler { this.clearPtyFlowState(managed.id) } + /** + * A reap is an exit the client has to hear about. `notifyExitListener` is + * relay-internal, so a retirement that stops there leaves the pane mounted + * against a session the relay has already forgotten — the next attach answers + * `PTY "" not found` and nothing before it said why. `resize` made that + * user-triggered. + * + * `-1` is this wire's "gone, status unrecoverable": the pid is proven absent + * (ESRCH from the host that owns it) but nothing waited on the shell, so no + * status exists. `ssh-relay-session` already publishes the same code for a + * dropped lease. Reached only from the proven-exited path — a client that acts + * on this retires the pane, so nothing weaker than ESRCH may reach it. + */ + private publishReapedExit(managed: ManagedPty): void { + // Why the guard: node-pty's own `onExit` already queued and published this + // pty's real exit code before reaching here, and the sweep also reaps + // entries that path left behind. + if (managed.exitListenerNotified || this.pendingExitByPty.has(managed.id)) { + return + } + this.pendingExitByPty.set(managed.id, { + id: managed.id, + code: -1, + incarnationId: managed.incarnationId + }) + this.publishPendingExit(managed.id) + } + + /** + * Retire this entry when the host proves its pid is gone; report whether it was. + * + * `managed.disposed` is bookkeeping, not liveness: it says we tore the record + * down, not that the shell died. A shell can exit without node-pty producing + * `onExit`, which leaves a non-disposed entry holding a handle whose master fd + * is already closed. Only `isProcessAlive` (ESRCH, from the host that owns the + * process) is positive evidence of absence; every other outcome is + * `unverifiable` and keeps its record and owner claim + * (docs/reference/ssh-execution-boundary.md). + */ + private reapPtyProvenExited(managed: ManagedPty): boolean { + if (!managed.pty.pid || isProcessAlive(managed.pty.pid)) { + return false + } + this.reapExitedPty(managed, 'exited') + return true + } + private async sendSignal(params: Record): Promise { const id = params.id as string const signal = params.signal as string @@ -2429,7 +2514,12 @@ export class PtyHandler { if (!managed || managed.disposed) { return false } - return await processHasChildren(managed.pty.pid) + // Fresh, not TTL-cached: this RPC exists to gate destructive decisions (the + // window-close confirmation, workspace cleanup's idle evidence), which act + // on the answer once. `pty.inspectProcess` below stays on the shared + // snapshot because it is the polled path, where a scan per pane per tick is + // the fork storm the cache removed. + return await processHasChildren(managed.pty.pid, { fresh: true }) } private async getForegroundProcess(params: Record): Promise { @@ -2494,8 +2584,11 @@ export class PtyHandler { } } for (const [entryIndex, [id, managed]] of managedEntries.entries()) { - if (managed.disposed || (managed.pty.pid && !isProcessAlive(managed.pty.pid))) { - this.reapExitedPty(managed) + if (managed.disposed) { + this.reapExitedPty(managed, 'record-torn-down') + continue + } + if (this.reapPtyProvenExited(managed)) { continue } // Reuse batched correlation; per-PTY tree scans recreate O(PTY × rows) work. diff --git a/src/relay/pty-shell-utils.test.ts b/src/relay/pty-shell-utils.test.ts index 6bfeab1a56a..0b4148a2a09 100644 --- a/src/relay/pty-shell-utils.test.ts +++ b/src/relay/pty-shell-utils.test.ts @@ -17,6 +17,7 @@ import { resetProcessTableSnapshotForTests } from '../shared/process-table-snaps import { getForegroundProcessName, isProcessAlive, + processHasChildren, resolveDefaultCwd, resolveWindowsDefaultShell } from './pty-shell-utils' @@ -525,3 +526,87 @@ describe('getForegroundProcessName', () => { await expect(getForegroundProcessName(100)).resolves.toBe('bash') }) }) + +describe('processHasChildren', () => { + // Why these assert on argv, not just the answer: the defect in #13537 was the + // cost of the answer. `pgrep -P` forks per pane per poll and opens six procfs + // files per host process to resolve one ppid, so the contract worth pinning is + // "no fork of its own, and share the foreground lookup's cached table". + const PS_TABLE = ['100 1 Ss bash', '101 100 S+ node /opt/codex', '200 1 Ss zsh'].join('\n') + + it('answers from the shared process table without forking pgrep', async () => { + await withProcessPlatform('linux', async () => { + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { stdout: PS_TABLE } + } + return new Error('unexpected command') + }) + + await expect(processHasChildren(100)).resolves.toBe(true) + await expect(processHasChildren(200)).resolves.toBe(false) + + expect(execFileMock.mock.calls.map((call) => call[0])).not.toContain('pgrep') + }) + }) + + it('shares one process-table capture across a burst of panes', async () => { + await withProcessPlatform('linux', async () => { + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { stdout: PS_TABLE } + } + return new Error('unexpected command') + }) + + const answers = await Promise.all([ + processHasChildren(100), + processHasChildren(100), + processHasChildren(200), + getForegroundProcessName(100, 'bash') + ]) + + expect(answers).toEqual([true, true, false, 'codex']) + expect(execFileMock).toHaveBeenCalledTimes(1) + }) + }) + + it('rescans for a close decision rather than serving a table from inside the TTL', async () => { + await withProcessPlatform('linux', async () => { + let table = PS_TABLE + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { stdout: table } + } + return new Error('unexpected command') + }) + + // The poll answers from the cache, which is the whole point of the memo. + await expect(processHasChildren(200)).resolves.toBe(false) + table = [PS_TABLE, '201 200 S+ npm run build'].join('\n') + await expect(processHasChildren(200)).resolves.toBe(false) + expect(execFileMock).toHaveBeenCalledTimes(1) + + // A close or cleanup acts on the answer once and destructively, so a + // child started inside the 500ms window has to be visible to it. + await expect(processHasChildren(200, { fresh: true })).resolves.toBe(true) + expect(execFileMock).toHaveBeenCalledTimes(2) + }) + }) + + it('reports no children when the process table is unreadable', async () => { + await withProcessPlatform('linux', async () => { + mockExecFile(() => new Error('ps table unavailable')) + + await expect(processHasChildren(100)).resolves.toBe(false) + }) + }) + + it('spawns nothing on Windows, where the answer was always false', async () => { + await withProcessPlatform('win32', async () => { + await expect(processHasChildren(100)).resolves.toBe(false) + + expect(execFileMock).not.toHaveBeenCalled() + }) + }) +}) diff --git a/src/relay/pty-shell-utils.ts b/src/relay/pty-shell-utils.ts index e89ac60a7ed..474416d3a41 100644 --- a/src/relay/pty-shell-utils.ts +++ b/src/relay/pty-shell-utils.ts @@ -10,6 +10,7 @@ import { } from '../shared/agent-process-recognition' import { getFirstCommandToken } from '../shared/command-token-scanner' import { + getFreshProcessTableSnapshot, getProcessTableSnapshot, type ProcessTableIndex, type ProcessTableRow @@ -170,15 +171,37 @@ export async function resolveProcessCwd(pid: number, fallbackCwd: string): Promi } /** - * Check whether a process has child processes (via pgrep). + * Check whether a process has child processes. + * + * Why the shared snapshot and not `pgrep -P`: this answers one field of + * `pty.inspectProcess`, which every tracked pane polls on a 750ms/2000ms + * cadence, and the fork was neither cached nor coalesced. procps-ng opens six + * procfs files per process to resolve a ppid — including a `/proc//ctty` + * that never exists on Linux — so one call cost O(host process count) syscalls, + * ~4k opens per pgrep on a 690-process host, at up to 8 forks/sec (#13537). + * `getForegroundProcessName` in the same RPC already captured the TTL-cached + * `ps` table, whose index carries the parent/child map, so the answer is free. + * + * `fresh` opts out of that TTL. A poll can read a 500ms-old table because its + * next tick corrects it, but a close or cleanup decision acts on the answer + * once and destructively — a child that started inside the TTL would be killed + * with no confirmation. `pgrep` scanned per call, so anything that decides + * has to keep scanning per call. */ -export async function processHasChildren(pid: number): Promise { +export async function processHasChildren( + pid: number, + options?: { fresh?: boolean } +): Promise { + // Windows has no `ps`; the previous `pgrep` fork always failed here too, so + // this keeps the same answer without spawning anything to reach it. + if (process.platform === 'win32') { + return false + } try { - const { stdout } = await execFile('pgrep', ['-P', String(pid)], { - encoding: 'utf-8', - timeout: 3000 - }) - return stdout.trim().length > 0 + const rows = options?.fresh + ? await getFreshProcessTableSnapshot() + : await getProcessTableSnapshot() + return (getProcessTableIndex(rows).childrenByPpid.get(pid)?.length ?? 0) > 0 } catch { return false } From 3a8f806d6c71ac2e6f1e82b7d04ffd9384c662e0 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:48 -0700 Subject: [PATCH 09/77] fix(remote-terminal): keep the stream stall deadline armed on unacknowledged credit (#17871) * fix(remote-terminal): keep the stream stall deadline armed on unacknowledged credit A paired-runtime terminal could stall silently with a live socket, a live PTY and no transport error (#11265). Two compounding defects on the read side: - The stream watchdog re-armed its 30s delivery deadline from zero on every settled delivery, so sibling traffic postponed the verdict indefinitely, and it cleared the timer entirely once renderer parse credit hit zero. Re-arming required inbound output -- the exact thing an exhausted host ACK window stops -- so once an ACK went missing nothing could ever detect the stall. The deadline is now anchored to the oldest unsettled delivery and stays armed while delivered bytes remain unacknowledged to the host. - flushOutputAcknowledgement zeroed pendingAckBytes before knowing the ACK frame was accepted, permanently shrinking the host's send window. Unsent bytes are re-charged and the flush timer re-armed. Recovery still reports onTransportClose({recoverable:true}); no path claims the PTY exited. * fix(remote-terminal): stop rearming the ack flush for a stream the failed send dropped A failing ACK send tears the stream down inside sendFrame, so the re-charge path armed a 4ms timer on an unregistered stream whose watchdog was already disposed; every later send returned false on !ready and rescheduled again. Also adds the missing integration coverage for the real ack -> watchdog flow. * fix(i18n): restore the activity-options key the rebase dropped The branch's en.json predates #18245, which added both the translate() call and its key. Rebasing took the branch copy wholesale, silently dropping the key and failing verify:localization-catalog. * fix(i18n): union en.json with main so the rebase cannot drop keys --- ...remote-runtime-terminal-flow-controller.ts | 35 ++++++--- ...untime-terminal-parse-backpressure.test.ts | 50 +++++++++++++ ...te-runtime-terminal-stall-recovery.test.ts | 72 +++++++++++++++++++ .../remote-terminal-stream-watchdog.test.ts | 58 +++++++++++++++ .../remote-terminal-stream-watchdog.ts | 40 ++++++++--- 5 files changed, 237 insertions(+), 18 deletions(-) create mode 100644 src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts diff --git a/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts b/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts index 640ec25933e..0488dd323cf 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts @@ -77,27 +77,46 @@ export abstract class RemoteRuntimeTerminalFlowController extends RemoteRuntimeT stream: RemoteRuntimeMultiplexedTerminalState, bytes: number ): boolean { - if (this.streams.get(stream.streamId) !== stream) { + if (!this.isRegisteredStream(stream)) { return true } stream.pendingAckBytes += bytes if (stream.pendingAckBytes >= TERMINAL_MULTIPLEX_ACK_BATCH_BYTES) { return this.flushOutputAcknowledgement(stream) } - if (stream.ackFlushTimer === null) { - stream.ackFlushTimer = setTimeout(() => { - stream.ackFlushTimer = null - this.flushOutputAcknowledgement(stream) - }, TERMINAL_MULTIPLEX_ACK_FLUSH_MS) - } + this.scheduleOutputAcknowledgementFlush(stream) return true } + private scheduleOutputAcknowledgementFlush(stream: RemoteRuntimeMultiplexedTerminalState): void { + if (stream.ackFlushTimer !== null) { + return + } + stream.ackFlushTimer = setTimeout(() => { + stream.ackFlushTimer = null + this.flushOutputAcknowledgement(stream) + }, TERMINAL_MULTIPLEX_ACK_FLUSH_MS) + } + private flushOutputAcknowledgement(stream: RemoteRuntimeMultiplexedTerminalState): boolean { clearAckFlushTimer(stream) const bytes = stream.pendingAckBytes + if (bytes <= 0) { + return true + } stream.pendingAckBytes = 0 - return bytes <= 0 || this.acknowledgeOutput(stream, bytes) + if (this.acknowledgeOutput(stream, bytes)) { + // Why: only a frame the transport took reopens the host's send window. + stream.watchdog.recordOutputAcknowledged(bytes) + return true + } + // Why guarded: a failed send may have torn the stream down, and re-charging a dropped stream reschedules itself forever. + if (this.isRegisteredStream(stream)) { + // Why re-charged: dropping an unsent ack shrinks the host window for the stream's life, and the only retry trigger is the output that shrunken window blocks. + stream.pendingAckBytes += bytes + this.scheduleOutputAcknowledgementFlush(stream) + } + return false } getStreamsForE2e(): Iterable { diff --git a/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts b/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts index ba49c7cfa26..0a004a0c564 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts @@ -427,6 +427,56 @@ describe('remote terminal renderer backpressure', () => { expect(sentAckBytes()).toEqual([1]) }) + it('stops rearming the ack flush after a failed ACK tears the stream down', async () => { + vi.useFakeTimers() + try { + const { getRemoteRuntimeTerminalMultiplexer } = + await import('./remote-runtime-terminal-multiplexer') + const { takeCurrentTerminalDeliveryCredit } = + await import('../lib/pane-manager/terminal-delivery-credit') + const parseCredits: (() => void)[] = [] + const stream = await getRemoteRuntimeTerminalMultiplexer('windows-test').subscribeTerminal({ + terminal: 'term-wedge', + client: { id: 'mac-viewer', type: 'desktop' }, + callbacks: { + onData: () => { + const credit = takeCurrentTerminalDeliveryCredit() + if (credit) { + parseCredits.push(credit) + } + }, + onSnapshot: vi.fn() + } + }) + sendBinary.mockClear() + sendBinary.mockImplementation((bytes) => { + if (decodeTerminalStreamFrame(bytes)?.opcode === TerminalStreamOpcode.Ack) { + throw new Error('socket closed') + } + }) + callbacks?.onBinary( + encodeTerminalStreamFrame({ + opcode: TerminalStreamOpcode.Output, + streamId: stream.streamId, + seq: 1, + payload: encodeTerminalStreamText('x') + }) + ) + parseCredits[0]?.() + await vi.advanceTimersByTimeAsync(50) + + expect(unsubscribe).toHaveBeenCalledOnce() + expect(vi.getTimerCount()).toBe(0) + await vi.advanceTimersByTimeAsync(10_000) + expect(vi.getTimerCount()).toBe(0) + await vi.advanceTimersByTimeAsync(70_000) + expect(vi.getTimerCount()).toBe(0) + expect(sentAckBytes()).toEqual([1]) + } finally { + vi.useRealTimers() + } + }) + function sentAckBytes(): number[] { return sendBinary.mock.calls.flatMap(([bytes]) => { const frame = decodeTerminalStreamFrame(bytes) diff --git a/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts b/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts index 1e8b239d1ce..c6e64e4e410 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts @@ -11,6 +11,7 @@ import { REMOTE_TERMINAL_COMMAND_RESPONSE_TIMEOUT_MS, REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS } from './remote-terminal-stream-watchdog' +import { TERMINAL_MULTIPLEX_ACK_FLUSH_MS } from '../../../shared/terminal-multiplex-flow-control' describe('remote terminal stalled stream recovery', () => { const sendBinary = vi.fn() @@ -101,6 +102,77 @@ describe('remote terminal stalled stream recovery', () => { healthy.close() }) + it('keeps a stream alive once the transport takes the ack for its parsed output', async () => { + const { getRemoteRuntimeTerminalMultiplexer } = + await import('./remote-runtime-terminal-multiplexer') + const { takeCurrentTerminalDeliveryCredit } = + await import('../lib/pane-manager/terminal-delivery-credit') + const credits: (() => void)[] = [] + const onTransportClose = vi.fn() + const stream = await getRemoteRuntimeTerminalMultiplexer('windows-test').subscribeTerminal({ + terminal: 'term-acked', + client: { id: 'mac-viewer', type: 'desktop' }, + callbacks: { + onData: () => { + const credit = takeCurrentTerminalDeliveryCredit() + if (credit) { + credits.push(credit) + } + }, + onSnapshot: vi.fn(), + onTransportClose + } + }) + sendBinary.mockClear() + + emitOutput(stream.streamId, 'host output the renderer parses') + credits[0]?.() + await vi.advanceTimersByTimeAsync(TERMINAL_MULTIPLEX_ACK_FLUSH_MS) + expect(sentFrames(TerminalStreamOpcode.Ack)).toHaveLength(1) + + await vi.advanceTimersByTimeAsync(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS) + + expect(onTransportClose).not.toHaveBeenCalled() + expect(sentUnsubscribeStreamIds()).toEqual([]) + stream.close() + }) + + it('keeps the delivery deadline anchored while sibling frames keep settling', async () => { + const { getRemoteRuntimeTerminalMultiplexer } = + await import('./remote-runtime-terminal-multiplexer') + const { takeCurrentTerminalDeliveryCredit } = + await import('../lib/pane-manager/terminal-delivery-credit') + const credits: (() => void)[] = [] + const onTransportClose = vi.fn() + const stream = await getRemoteRuntimeTerminalMultiplexer('windows-test').subscribeTerminal({ + terminal: 'term-anchored', + client: { id: 'mac-viewer', type: 'desktop' }, + callbacks: { + onData: () => { + const credit = takeCurrentTerminalDeliveryCredit() + if (credit) { + credits.push(credit) + } + }, + onSnapshot: vi.fn(), + onTransportClose + } + }) + sendBinary.mockClear() + + emitOutput(stream.streamId, 'frame the renderer never parses') + await vi.advanceTimersByTimeAsync(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS - 5_000) + emitOutput(stream.streamId, 'sibling frame that settles') + credits[1]?.() + await vi.advanceTimersByTimeAsync(TERMINAL_MULTIPLEX_ACK_FLUSH_MS) + + expect(onTransportClose).not.toHaveBeenCalled() + await vi.advanceTimersByTimeAsync(5_000) + + expect(onTransportClose).toHaveBeenCalledWith({ recoverable: true }) + expect(sentUnsubscribeStreamIds()).toEqual([stream.streamId]) + }) + it('probes then restarts a stream when an entered command receives no frames', async () => { const { getRemoteRuntimeTerminalMultiplexer, REMOTE_TERMINAL_SNAPSHOT_REQUEST_TIMEOUT_MS } = await import('./remote-runtime-terminal-multiplexer') diff --git a/src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts b/src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts new file mode 100644 index 00000000000..66d65f9bcbc --- /dev/null +++ b/src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts @@ -0,0 +1,58 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS, + createRemoteTerminalStreamWatchdog +} from './remote-terminal-stream-watchdog' + +describe('remote terminal stream watchdog delivery deadline', () => { + beforeEach(() => { + vi.useFakeTimers() + }) + + afterEach(() => { + vi.useRealTimers() + }) + + it('anchors the deadline to the oldest unsettled delivery instead of the last settled sibling', () => { + const onStall = vi.fn() + const watchdog = createRemoteTerminalStreamWatchdog(onStall) + + watchdog.beginOutputDelivery(100) + for (let tick = 0; tick < 3; tick += 1) { + vi.advanceTimersByTime(9_000) + const settle = watchdog.beginOutputDelivery(10) + settle() + } + expect(onStall).not.toHaveBeenCalled() + + vi.advanceTimersByTime(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS - 27_000) + + expect(onStall).toHaveBeenCalledTimes(1) + expect(onStall.mock.calls[0]?.[0]).toMatchObject({ reason: 'delivery-credit-timeout' }) + }) + + it('stays armed while parsed bytes remain unacknowledged to the host', () => { + const onStall = vi.fn() + const watchdog = createRemoteTerminalStreamWatchdog(onStall) + + watchdog.beginOutputDelivery(100)() + vi.advanceTimersByTime(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS) + + expect(onStall).toHaveBeenCalledTimes(1) + expect(onStall.mock.calls[0]?.[0]).toMatchObject({ + outstandingDeliveryBytes: 0, + reason: 'delivery-credit-timeout' + }) + }) + + it('disarms once the acknowledgement reaches the transport', () => { + const onStall = vi.fn() + const watchdog = createRemoteTerminalStreamWatchdog(onStall) + + watchdog.beginOutputDelivery(100)() + watchdog.recordOutputAcknowledged(100) + vi.advanceTimersByTime(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS * 2) + + expect(onStall).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts b/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts index 4cb6b081c5f..be9e1b9d404 100644 --- a/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts +++ b/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts @@ -9,6 +9,8 @@ export type RemoteTerminalStreamStall = { export type RemoteTerminalStreamWatchdog = { beginOutputDelivery: (bytes: number) => () => void + /** Bytes whose ACK frame reached the transport, releasing the host's window. */ + recordOutputAcknowledged: (bytes: number) => void completeCommandResponseProbe: () => void recordCommandInput: (text: string) => void recordInbound: () => void @@ -21,6 +23,9 @@ export function createRemoteTerminalStreamWatchdog( let responseTimer: ReturnType | null = null let deliveryTimer: ReturnType | null = null let outstandingDeliveryBytes = 0 + // Why separate from parse credit: the host reopens its window on ACK frames, so bytes parsed but not yet ACKed are still the credit whose loss stops output. + let unacknowledgedBytes = 0 + let deliveryPendingSinceMs: number | null = null let lastInboundAtMs = Date.now() let commandResponseProbePending = false let disposed = false @@ -54,23 +59,29 @@ export function createRemoteTerminalStreamWatchdog( reason }) } - const armDeliveryTimer = (): void => { - clearDeliveryTimer() - if (outstandingDeliveryBytes <= 0 || disposed) { + // Why anchored, never restarted: a deadline re-armed by sibling settles is postponed forever, and one cleared at zero parse credit can only re-arm from inbound output — which is what the stall stops. + const syncDeliveryTimer = (): void => { + if (disposed || outstandingDeliveryBytes + unacknowledgedBytes <= 0) { + clearDeliveryTimer() + deliveryPendingSinceMs = null return } - deliveryTimer = setTimeout( - () => trip('delivery-credit-timeout'), - REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS + deliveryPendingSinceMs ??= Date.now() + if (deliveryTimer) { + return + } + const remainingMs = Math.max( + 0, + REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS - (Date.now() - deliveryPendingSinceMs) ) + deliveryTimer = setTimeout(() => trip('delivery-credit-timeout'), remainingMs) } return { beginOutputDelivery(bytes) { outstandingDeliveryBytes += bytes - if (!deliveryTimer) { - armDeliveryTimer() - } + unacknowledgedBytes += bytes + syncDeliveryTimer() let settled = false return () => { if (settled || disposed) { @@ -78,9 +89,16 @@ export function createRemoteTerminalStreamWatchdog( } settled = true outstandingDeliveryBytes = Math.max(0, outstandingDeliveryBytes - bytes) - armDeliveryTimer() + syncDeliveryTimer() } }, + recordOutputAcknowledged(bytes) { + if (disposed) { + return + } + unacknowledgedBytes = Math.max(0, unacknowledgedBytes - bytes) + syncDeliveryTimer() + }, completeCommandResponseProbe() { commandResponseProbePending = false }, @@ -103,6 +121,8 @@ export function createRemoteTerminalStreamWatchdog( clearResponseTimer() clearDeliveryTimer() outstandingDeliveryBytes = 0 + unacknowledgedBytes = 0 + deliveryPendingSinceMs = null } } } From 278f9ee876bd0ddc9551032a3d823b04d45419fc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:53 -0700 Subject: [PATCH 10/77] fix(ssh): answer every MFA stage, stop dialling an unclaimed alias, and say where a clone failed (#17946) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): answer every MFA stage, not just the first ssh2 walks one flat auth-method list exactly once, so keyboard-interactive could only ever be offered a single time. A host running `AuthenticationMethods keyboard-interactive,keyboard-interactive` (or any ladder ending in a second challenge) partial-succeeds the first stage and then finds the list exhausted, which the user sees as "All configured authentication methods failed" — the reports in #8622 and #16820. Orca's own auth handler now runs for every target instead of only multi-key ones, and rebuilds its queue on each SSH_MSG_USERAUTH_FAILURE that carries partial success, narrowed to the methods the host still offers. Narrowing also stops keys being re-offered after the host has moved past publickey, which is what exhausts MaxAuthTries before the challenge is ever shown. Covered by a real ssh2 server fixture that stages partial success. * fix(git): say where a failing clone ran and why nothing could prompt Clones go through nonInteractiveGitEnv, so `ssh` runs with BatchMode=yes and an emptied SSH_ASKPASS. On a remote or paired-runtime clone that produces `fatal: Could not read from remote repository.` while the same `git clone` typed by hand on that box succeeds — the divergence in #14533. Nothing in the message said the clone ran on the other machine, under its keys, with the prompt deliberately disabled. getGitCloneFailureMessage now appends that fact, and names the two recognisable shapes: a publickey refusal (load the key into an agent there) and a host-key failure (record the key in that machine's known_hosts). Unrecognised SSH failures still get the where-it-ran note; non-SSH failures are untouched. One builder, so the SSH-target relay path and the runtime path both get it. * fix(ssh): stop dialling a bare alias no ssh_config block claims A wildcard `Host *` block supplies ProxyCommand/ProxyJump for every alias, so shouldUseSystemSshTransport picks the system transport for an alias whose own Host block was renamed or deleted, and buildSshArgs then dials that alias verbatim: no -l, no -p, no Hostname. Orca connects as the wildcard's user to the wildcard's host and discards the endpoint it stored (#11746). The signal #11746 assumed (hostBlockMatch, from the still-open #11707) does not exist, and `ssh -G` cannot supply it — it prints the merged config and answers for unknown aliases too. The config file is the only source of truth, so: - parseSshConfigAliasClaims retains raw Host patterns and flags Match blocks, which parseSshConfig discards because it mints importable targets. - sshConfigMayClaimAlias is sound in the negative direction only: an unreadable file, any Match block, or any non-catch-all pattern that might match all answer "claimed", so absence of evidence is never read as evidence of absence. Only a proven-unclaimed alias licenses an override. - buildSshArgs then states Hostname/Port/User, and only those: the wildcard is still the route, and -o Hostname does not change block selection, so the proxy keeps applying and %h expands to the host we mean. The verdict is injected rather than read inside buildSshArgs, so an arg builder does not answer differently per machine. Default is today's behaviour. Scoped to the system-SSH transport and the connection's own command/transport path. Port-forward processes and the ssh2 transport (#11707) are unchanged. * fix(ssh): read a negated Host group as uncertainty, and gate clone SSH guidance `Host * !prod` applies to every alias but `prod`, yet skipping both the catch-all and the `!` pattern answered "unclaimed" for `stage` — which licences overriding Hostname/Port/User against a block the user wrote. Any negation now makes the whole group uncertain; the function is only sound in the negative direction. Also require an ssh(1) diagnostic beside "could not read from remote repository" before appending the SSH clone note: git prints that same line for the HTTP remote helper, where advice about keys and agents is simply wrong. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ssh/ssh-config-alias-claim.test.ts | 80 ++++++ src/main/ssh/ssh-config-alias-claim.ts | 103 +++++++ src/main/ssh/ssh-config-parser.ts | 40 +++ src/main/ssh/ssh-connection.ts | 10 +- .../ssh-multi-factor-authentication.test.ts | 251 ++++++++++++++++++ .../ssh/ssh-multi-key-authentication.test.ts | 69 ++++- .../ssh/ssh-private-key-authentication.ts | 43 ++- src/main/ssh/ssh-system-fallback.test.ts | 46 ++++ src/main/ssh/system-ssh-args.ts | 41 +++ src/shared/git-clone-failure-message.test.ts | 55 ++++ src/shared/git-clone-failure-message.ts | 45 ++++ 11 files changed, 770 insertions(+), 13 deletions(-) create mode 100644 src/main/ssh/ssh-config-alias-claim.test.ts create mode 100644 src/main/ssh/ssh-config-alias-claim.ts create mode 100644 src/main/ssh/ssh-multi-factor-authentication.test.ts diff --git a/src/main/ssh/ssh-config-alias-claim.test.ts b/src/main/ssh/ssh-config-alias-claim.test.ts new file mode 100644 index 00000000000..ed6ec87971c --- /dev/null +++ b/src/main/ssh/ssh-config-alias-claim.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from 'vitest' +import { parseSshConfigAliasClaims } from './ssh-config-parser' +import { sshConfigMayClaimAlias } from './ssh-config-alias-claim' + +function mayClaim(config: string, alias: string): boolean { + return sshConfigMayClaimAlias(alias, parseSshConfigAliasClaims(config)) +} + +describe('sshConfigMayClaimAlias', () => { + it('treats a wildcard-only config as proof that nothing claims the alias', () => { + const config = ` +Host * + ProxyCommand nc -X connect -x proxy:8080 %h %p + ForwardAgent yes +` + expect(mayClaim(config, 'prod')).toBe(false) + }) + + it('keeps a Host block that names the alias authoritative', () => { + const config = ` +Host * + ProxyCommand nc %h %p +Host prod + HostName prod.internal +` + expect(mayClaim(config, 'prod')).toBe(true) + }) + + it('keeps a glob that reaches the alias authoritative', () => { + // parseSshConfig drops these, which is why the claim check cannot reuse it. + const config = ` +Host prod-* + HostName prod.internal +` + expect(mayClaim(config, 'prod-web')).toBe(true) + expect(mayClaim(config, 'stage-web')).toBe(false) + }) + + it('matches single-character wildcards the way OpenSSH does', () => { + expect(mayClaim('Host prod?\n User ops\n', 'prod1')).toBe(true) + expect(mayClaim('Host prod?\n User ops\n', 'prod12')).toBe(false) + }) + + it('refuses to answer once any Match block is present', () => { + const config = ` +Host * + ProxyCommand nc %h %p +Match host prod + User ops +` + expect(mayClaim(config, 'prod')).toBe(true) + }) + + it('reads any negated group as uncertainty, because OpenSSH still applies its positives', () => { + // `Host * !prod` routes stage; answering false there would licence overriding a block the user + // wrote. The exempted alias is not worth a second matching rule to recover. + expect(mayClaim('Host * !prod\n ForwardAgent yes\n', 'stage')).toBe(true) + expect(mayClaim('Host * !prod\n ForwardAgent yes\n', 'prod')).toBe(true) + }) + + it('still proves absence when no group negates', () => { + expect(mayClaim('Host *\n ForwardAgent yes\nHost prod\n User ops\n', 'stage')).toBe(false) + }) + + it('reads an unreadable config as uncertainty, not absence', () => { + expect(sshConfigMayClaimAlias('prod', null)).toBe(true) + }) + + it('reads an empty alias as uncertainty', () => { + expect(sshConfigMayClaimAlias('', parseSshConfigAliasClaims('Host *\n'))).toBe(true) + }) +}) + +describe('parseSshConfigAliasClaims', () => { + it('retains the raw pattern groups and flags Match blocks', () => { + expect( + parseSshConfigAliasClaims('Host a b* # comment\n User x\nMatch final\n User y\n') + ).toEqual({ hostPatternGroups: [['a', 'b*']], hasMatchBlock: true }) + }) +}) diff --git a/src/main/ssh/ssh-config-alias-claim.ts b/src/main/ssh/ssh-config-alias-claim.ts new file mode 100644 index 00000000000..74a3ea0d44c --- /dev/null +++ b/src/main/ssh/ssh-config-alias-claim.ts @@ -0,0 +1,103 @@ +import { existsSync, statSync } from 'node:fs' +import { homedir } from 'node:os' +import { join } from 'node:path' +import { normalizeSshConfigAlias } from '../../shared/ssh-config-alias' +import { expandSshConfigIncludes } from './ssh-config-include-expander' +import { parseSshConfigAliasClaims, type SshConfigAliasClaims } from './ssh-config-parser' + +/** + * Whether anything in the user's ssh_config could claim this alias — i.e. whether a `Host` or + * `Match` block other than a bare catch-all applies to it. + * + * Sound in the negative direction only. `false` means the parsed config proves nothing claims the + * alias; every uncertainty (unreadable file, any `Match` block, any negated `Host` group, any + * pattern that might match) + * answers `true`, because "we could not tell" must never be read as "no block exists". Callers use + * `false` as licence to override what OpenSSH would resolve, so a wrong `false` breaks a config the + * user explicitly wrote, which is worse than the routing bug it exists to fix. + */ +export function sshConfigMayClaimAlias( + alias: string, + claims: SshConfigAliasClaims | null +): boolean { + const normalizedAlias = normalizeSshConfigAlias(alias) + if (!normalizedAlias || claims === null) { + return true + } + // A Match block's criteria (exec, originalhost, user, …) are not modelled here, and one that + // routes this alias is indistinguishable from one that does not. + if (claims.hasMatchBlock) { + return true + } + return claims.hostPatternGroups.some((patterns) => + // A negation makes the whole group uncertain: `Host * !prod` still routes every other alias, + // so skipping both the catch-all and the `!` would answer "unclaimed" for one that is claimed. + patterns.some((pattern) => pattern.startsWith('!')) + ? true + : patterns.some( + (pattern) => + !isCatchAllHostPattern(pattern) && matchesHostPattern(pattern, normalizedAlias) + ) + ) +} + +/** `Host *` — the block every alias matches, which is exactly the one that proves nothing. */ +function isCatchAllHostPattern(pattern: string): boolean { + return pattern.length > 0 && /^\*+$/.test(pattern) +} + +function matchesHostPattern(pattern: string, normalizedAlias: string): boolean { + let expression = '' + for (const character of normalizeSshConfigAlias(pattern)) { + if (character === '*') { + expression += '.*' + } else if (character === '?') { + expression += '.' + } else { + expression += character.replace(/[.+^${}()|[\]\\]/, '\\$&') + } + } + return new RegExp(`^${expression}$`).test(normalizedAlias) +} + +// Bounds how long an edit to an Included file can go unnoticed; buildSshArgs runs per remote +// command, so re-expanding Includes every time is not an option. +const CLAIM_CACHE_TTL_MS = 5_000 + +let cachedClaims: { key: string; readAt: number; claims: SshConfigAliasClaims } | null = null + +export function invalidateSshConfigAliasClaimCache(): void { + cachedClaims = null +} + +/** + * Parse of `~/.ssh/config` (Includes expanded), or null when it cannot be read. + * + * Null and empty are different answers here: an absent or unreadable file is the uncertainty case, + * while a readable file with no matching block is the proof {@link sshConfigMayClaimAlias} needs. + */ +export function loadUserSshConfigAliasClaims(): SshConfigAliasClaims | null { + const configPath = join(homedir(), '.ssh', 'config') + try { + if (!existsSync(configPath)) { + return null + } + // Why key on the root file only: an edited Include can go unnoticed, so the cache also expires. + const stats = statSync(configPath) + const key = `${stats.mtimeMs}:${stats.size}` + const now = Date.now() + if (cachedClaims?.key === key && now - cachedClaims.readAt < CLAIM_CACHE_TTL_MS) { + return cachedClaims.claims + } + const claims = parseSshConfigAliasClaims(expandSshConfigIncludes(configPath)) + cachedClaims = { key, readAt: now, claims } + return claims + } catch { + return null + } +} + +/** Convenience wrapper over the two above; used where the caller has no claims to inject. */ +export function mayUserSshConfigClaimAlias(alias: string): boolean { + return sshConfigMayClaimAlias(alias, loadUserSshConfigAliasClaims()) +} diff --git a/src/main/ssh/ssh-config-parser.ts b/src/main/ssh/ssh-config-parser.ts index 1ac6d8ed4bf..bdad3b383ab 100644 --- a/src/main/ssh/ssh-config-parser.ts +++ b/src/main/ssh/ssh-config-parser.ts @@ -145,6 +145,46 @@ function appendHosts(target: SshConfigHost[], entries: SshConfigHost[]): void { } } +/** + * Every `Host` pattern list in a config, plus whether any `Match` block is present. + * + * Deliberately raw where {@link parseSshConfig} is not: that one keeps only concrete aliases + * because it mints importable targets, so a `Host prod.*` or a `Match host prod` route is + * invisible to it. Answering "does anything in this file claim this alias?" needs those back. + */ +export type SshConfigAliasClaims = { + hostPatternGroups: string[][] + hasMatchBlock: boolean +} + +export function parseSshConfigAliasClaims(content: string): SshConfigAliasClaims { + const hostPatternGroups: string[][] = [] + let hasMatchBlock = false + + for (const rawLine of content.split('\n')) { + const line = rawLine.trim() + if (!line || line.startsWith('#')) { + continue + } + const directive = parseConfigDirective(line) + if (!directive) { + continue + } + if (directive.key === 'host') { + const patterns = splitHostPatterns(directive.rawValue) + if (patterns.length > 0) { + hostPatternGroups.push(patterns) + } + continue + } + if (directive.key === 'match') { + hasMatchBlock = true + } + } + + return { hostPatternGroups, hasMatchBlock } +} + function parseConfigDirective(line: string): { key: string; rawValue: string } | null { const match = line.match(/^([^=\s]+)(?:\s*=\s*|\s+)(.*)$/) if (!match) { diff --git a/src/main/ssh/ssh-connection.ts b/src/main/ssh/ssh-connection.ts index f51f23a3d71..98d383c5c92 100644 --- a/src/main/ssh/ssh-connection.ts +++ b/src/main/ssh/ssh-connection.ts @@ -74,6 +74,7 @@ import { isTransientReconnectError } from './ssh-reconnect-error-classification' import { SshReconnectLadder } from './ssh-reconnect-ladder' +import { mayUserSshConfigClaimAlias } from './ssh-config-alias-claim' import { getPassphrasePrivateKeyPath } from './ssh-private-key-authentication' import { requiresSystemSshForSecurityKey, @@ -108,7 +109,9 @@ type SshRemoteFileOptions = { /** Bounds the trust-source reads that run before the handshake, which nothing else times out. */ const HOST_KEY_SOURCE_READ_TIMEOUT_MS = 5_000 -const SSH_KEYBOARD_INTERACTIVE_MAX_ROUNDS = 4 +// Counts every INFO_REQUEST of the handshake, so it must cover each partial-success stage the auth +// queue will answer (MAX_PARTIAL_SUCCESS_STAGES) times the rounds a PAM stack spends per stage. +const SSH_KEYBOARD_INTERACTIVE_MAX_ROUNDS = 8 const SSH_KEYBOARD_INTERACTIVE_READY_TIMEOUT_MS = SSH_CREDENTIAL_TIMEOUT_MS + 5_000 const SSH_KEYBOARD_INTERACTIVE_MAX_PROMPTS = 8 const SSH_KEYBOARD_INTERACTIVE_TEXT_MAX = 4_096 @@ -1299,6 +1302,11 @@ export class SshConnection { if (this.systemSshResolvedConfig) { options.resolvedConfig = this.systemSshResolvedConfig } + // Why here and not inside buildSshArgs: the verdict reads ~/.ssh/config, and an arg builder + // that consults the filesystem answers differently on every machine, tests included. + if (this.target.configHost && !mayUserSshConfigClaimAlias(this.target.configHost)) { + options.aliasClaimedByConfig = false + } if (this.systemSshControlMasterDisabledForSession) { options.disableControlMaster = true } diff --git a/src/main/ssh/ssh-multi-factor-authentication.test.ts b/src/main/ssh/ssh-multi-factor-authentication.test.ts new file mode 100644 index 00000000000..275ea2e3247 --- /dev/null +++ b/src/main/ssh/ssh-multi-factor-authentication.test.ts @@ -0,0 +1,251 @@ +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + Client, + Server as Ssh2Server, + utils, + type AuthContext, + type Connection, + type KeyboardAuthContext, + type PasswordAuthContext +} from 'ssh2' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { SshTarget } from '../../shared/ssh-types' +import type { SshResolvedConfig } from './ssh-config-parser' +import { buildConnectConfig } from './ssh-connection-utils' + +// OpenSSH's default; a host that burns it disconnects before the MFA stage is reached. +const MAX_AUTH_TRIES = 6 +const PASSWORD = 'stage-one-password' +const PASSCODE = '123456' + +type AuthStage = 'password' | 'keyboard-interactive' + +type MfaServer = { + port: number + attempts: string[] + close: () => Promise +} + +/** An OpenSSH-style `AuthenticationMethods a,b` host: each stage partial-succeeds into the next. */ +async function startMultiFactorServer(stages: AuthStage[]): Promise { + const attempts: string[] = [] + const connections = new Set() + // Ed25519 keygen can produce an invalid 31-byte key; ECDSA points always start with 0x04. + const hostKey = utils.generateKeyPairSync('ecdsa', { bits: 256 }).private + const server = new Ssh2Server({ hostKeys: [hostKey] }, (connection) => { + connections.add(connection) + connection.on('error', () => {}) + connection.on('close', () => connections.delete(connection)) + let stage = 0 + let failures = 0 + const remaining = (): AuthStage[] => [stages[stage]!] + const fail = (context: AuthContext): void => { + failures += 1 + if (failures >= MAX_AUTH_TRIES) { + connection.end() + return + } + context.reject(remaining(), false) + } + connection.on('authentication', (context) => { + attempts.push(context.method) + if (context.method === 'none') { + context.reject(remaining(), false) + return + } + if (context.method !== stages[stage]) { + fail(context) + return + } + if (context.method === 'password') { + if ((context as PasswordAuthContext).password !== PASSWORD) { + fail(context) + return + } + stage += 1 + if (stage === stages.length) { + context.accept() + return + } + context.reject(remaining(), true) + return + } + const keyboard = context as KeyboardAuthContext + keyboard.prompt( + [{ prompt: 'Duo passcode:', echo: false }], + 'Duo two-factor login', + 'Approve the push or enter a passcode.', + (answers) => { + if (answers?.[0] !== PASSCODE) { + fail(context) + return + } + stage += 1 + if (stage === stages.length) { + context.accept() + return + } + context.reject(remaining(), true) + } + ) + }) + }) + await new Promise((resolve, reject) => { + server.once('error', reject) + server.listen(0, '127.0.0.1', () => { + server.removeListener('error', reject) + resolve() + }) + }) + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('MFA fixture did not bind a TCP port') + } + return { + port: address.port, + attempts, + close: async () => { + for (const connection of connections) { + connection.end() + } + await new Promise((resolve, reject) => { + server.close((error) => (error ? reject(error) : resolve())) + }) + } + } +} + +function makeTarget(port: number, overrides: Partial = {}): SshTarget { + return { + id: 'mfa-target', + label: 'hpc', + source: 'manual', + host: '127.0.0.1', + port, + username: 'fixture', + ...overrides + } +} + +function makeResolved(port: number, identityFile: string[]): SshResolvedConfig { + return { + hostname: '127.0.0.1', + port, + user: 'fixture', + identityFile, + identitiesOnly: true, + forwardAgent: false, + proxyUseFdpass: false, + controlMaster: 'no', + controlPersist: 'no', + userKnownHostsFiles: [], + globalKnownHostsFiles: [], + strictHostKeyChecking: 'ask', + hashKnownHosts: false, + updateHostKeys: 'no' + } +} + +/** Drives ssh2 the way SshConnection does: one credential per keyboard-interactive prompt. */ +function connectWithOrcaConfig( + target: SshTarget, + resolved: SshResolvedConfig | null, + password: string | undefined, + answers: string[] +): { ready: Promise; prompts: string[] } { + const prompts: string[] = [] + const config = buildConnectConfig(target, resolved, { + includeAgent: false, + includePrivateKey: true + }) + if (password != null) { + config.password = password + } + const ready = new Promise((resolve, reject) => { + const client = new Client() + let answerIndex = 0 + client.on('keyboard-interactive', (_name, _instructions, _lang, requested, finish) => { + for (const requestedPrompt of requested) { + prompts.push(requestedPrompt.prompt) + } + finish(requested.map(() => answers[answerIndex++] ?? '')) + }) + client.once('ready', () => { + client.end() + resolve() + }) + client.once('error', reject) + client.once('close', () => reject(new Error('SSH connection closed during authentication'))) + client.connect({ ...config, hostVerifier: () => true, readyTimeout: 10_000 }) + }) + return { ready, prompts } +} + +describe('multi-stage SSH authentication', () => { + let tempDir: string + let keyPaths: string[] + + beforeEach(() => { + tempDir = mkdtempSync(join(tmpdir(), 'orca-mfa-')) + keyPaths = ['id_a', 'id_b'].map((name) => { + const path = join(tempDir, name) + writeFileSync(path, utils.generateKeyPairSync('ecdsa', { bits: 256 }).private) + return path + }) + }) + + afterEach(() => { + rmSync(tempDir, { recursive: true, force: true }) + }) + + it('answers a keyboard-interactive stage that follows a password partial success', async () => { + const server = await startMultiFactorServer(['password', 'keyboard-interactive']) + try { + const { ready, prompts } = connectWithOrcaConfig(makeTarget(server.port), null, PASSWORD, [ + PASSCODE + ]) + + await expect(ready).resolves.toBeUndefined() + expect(prompts).toEqual(['Duo passcode:']) + } finally { + await server.close() + } + }) + + it('answers a second keyboard-interactive stage after the first partially succeeds', async () => { + const server = await startMultiFactorServer(['keyboard-interactive', 'keyboard-interactive']) + try { + const { ready, prompts } = connectWithOrcaConfig(makeTarget(server.port), null, undefined, [ + PASSCODE, + PASSCODE + ]) + + await expect(ready).resolves.toBeUndefined() + expect(prompts).toEqual(['Duo passcode:', 'Duo passcode:']) + } finally { + await server.close() + } + }) + + it('reaches the MFA stage without burning the host auth-try budget on rejected keys', async () => { + const server = await startMultiFactorServer(['password', 'keyboard-interactive']) + try { + const target = makeTarget(server.port, { source: 'ssh-config', configHost: 'hpc' }) + const { ready } = connectWithOrcaConfig( + target, + makeResolved(server.port, keyPaths), + PASSWORD, + [PASSCODE] + ) + + await expect(ready).resolves.toBeUndefined() + // After the password stage partially succeeds the host only offers keyboard-interactive; + // re-offering keys there is what exhausts MaxAuthTries on real MFA hosts. + expect(server.attempts.filter((method) => method === 'publickey')).toHaveLength(0) + } finally { + await server.close() + } + }) +}) diff --git a/src/main/ssh/ssh-multi-key-authentication.test.ts b/src/main/ssh/ssh-multi-key-authentication.test.ts index 1f17c7dae66..7cee28a6fda 100644 --- a/src/main/ssh/ssh-multi-key-authentication.test.ts +++ b/src/main/ssh/ssh-multi-key-authentication.test.ts @@ -77,6 +77,18 @@ function nextAuth( return result ?? false } +function partialSuccessAuth( + config: ConnectConfig, + authsLeft: AuthenticationType[] +): AuthenticationType | AnyAuthMethod | false { + let result: AuthenticationType | AnyAuthMethod | false | undefined + const handler = config.authHandler as AuthHandlerMiddleware + handler(authsLeft, true, (attempt) => { + result = attempt + }) + return result ?? false +} + describe('ordered SSH private-key authentication', () => { beforeEach(() => { vi.stubEnv('SSH_AUTH_SOCK', '') @@ -111,6 +123,15 @@ describe('ordered SSH private-key authentication', () => { expect(mockReadFileSync).not.toHaveBeenCalledWith('/keys/stale-imported') }) + it('still offers the ssh-agent when no readable key precedes it', () => { + vi.stubEnv('SSH_AUTH_SOCK', '/tmp/agent.sock') + const config = buildConnectConfig(makeTarget({ identityFile: undefined }), null) + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + expect(nextAuth(config, false)).toMatchObject({ type: 'agent', agent: '/tmp/agent.sock' }) + expect(nextAuth(config, false)).toBe('keyboard-interactive') + }) + it('keeps explicit manual keys and unresolved imported keys as singular overrides', () => { const manual = buildConnectConfig( makeTarget({ @@ -128,9 +149,53 @@ describe('ordered SSH private-key authentication', () => { }) expect(manual.privateKey).toEqual(Buffer.from('/keys/manual')) - expect(manual.authHandler).toBeUndefined() expect(unresolvedImport.privateKey).toEqual(Buffer.from('/keys/stale-imported')) - expect(unresolvedImport.authHandler).toBeUndefined() + // Single-key targets are still ordered by Orca's handler so an MFA host reaches + // keyboard-interactive once per stage rather than once per connection. + expect(nextAuth(manual, true)).toMatchObject({ type: 'none' }) + expect(nextAuth(manual, false)).toMatchObject({ + type: 'publickey', + key: Buffer.from('/keys/manual') + }) + expect(nextAuth(manual, false)).toBe('keyboard-interactive') + }) + + it('re-offers keyboard-interactive for each partial-success MFA stage', () => { + const config = buildConnectConfig(makeTarget(), makeResolved(), { + includeAgent: false, + includePrivateKey: true + }) + config.password = 'stage-one' + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + expect(nextAuth(config, false)).toMatchObject({ type: 'password' }) + // Partial success: the host accepted the password and now offers only the challenge. + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + }) + + it('stops re-offering methods the host no longer accepts after a partial success', () => { + const config = buildConnectConfig(makeTarget(), makeResolved(), { + includeAgent: false, + includePrivateKey: true + }) + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + expect(nextAuth(config, false)).toBe(false) + }) + + it('bounds the number of partial-success stages it will answer', () => { + const config = buildConnectConfig(makeTarget(), makeResolved(), { + includeAgent: false, + includePrivateKey: true + }) + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + for (let stage = 0; stage < 4; stage += 1) { + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + } + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe(false) }) it('offers every resolved key for a manually owned config-picker target', () => { diff --git a/src/main/ssh/ssh-private-key-authentication.ts b/src/main/ssh/ssh-private-key-authentication.ts index a01a58bf9b6..3daa7a8c844 100644 --- a/src/main/ssh/ssh-private-key-authentication.ts +++ b/src/main/ssh/ssh-private-key-authentication.ts @@ -3,6 +3,16 @@ import type { PrivateKeyFile } from './ssh-auth-resolution' const passphraseKeyPaths = new WeakMap() +// Bounds an `AuthenticationMethods a,b,c` ladder so a host that keeps replying +// "partial success" cannot keep the client prompting forever. +const MAX_PARTIAL_SUCCESS_STAGES = 4 + +function authMethodName(attempt: AuthenticationType | AnyAuthMethod): AuthenticationType { + const type = typeof attempt === 'string' ? attempt : attempt.type + // Agent identities are signed as publickey; a server's method list never names 'agent'. + return type === 'agent' ? 'publickey' : type +} + function buildAuthQueue( config: ConnectConfig, keys: PrivateKeyFile[] @@ -35,21 +45,34 @@ export function configurePrivateKeyAuthentication( passphraseKeyPath?: string ): void { const firstKey = keys[0] - if (!firstKey) { - return - } - config.privateKey = firstKey.contents - if (passphraseKeyPath) { - passphraseKeyPaths.set(config, passphraseKeyPath) - } - if (keys.length === 1) { - return + if (firstKey) { + config.privateKey = firstKey.contents + if (passphraseKeyPath) { + passphraseKeyPaths.set(config, passphraseKeyPath) + } } + // Why this replaces ssh2's own handler for every target, not just multi-key ones: ssh2 walks one + // flat method list exactly once, so keyboard-interactive can only ever be offered a single time. + // An MFA host running `AuthenticationMethods keyboard-interactive,keyboard-interactive` (or any + // ladder whose last stage is a second challenge) partial-succeeds the first stage and then finds + // the list exhausted — reported to the user as "All configured authentication methods failed". let queue: (AuthenticationType | AnyAuthMethod)[] = [] - config.authHandler = (authsLeft, _partialSuccess, next) => { + let partialSuccessStagesLeft = MAX_PARTIAL_SUCCESS_STAGES + config.authHandler = (authsLeft, partialSuccess, next) => { if (authsLeft == null) { queue = buildAuthQueue(config, keys) + partialSuccessStagesLeft = MAX_PARTIAL_SUCCESS_STAGES + } else if (partialSuccess && partialSuccessStagesLeft > 0) { + // A stage was accepted and the host now demands another method. Restart from a fresh queue + // narrowed to what it still offers: re-offering keys it has stopped accepting is what + // exhausts MaxAuthTries before the challenge is ever shown. + partialSuccessStagesLeft -= 1 + const offered = Array.isArray(authsLeft) ? authsLeft : [] + queue = buildAuthQueue(config, keys).filter((attempt) => { + const method = authMethodName(attempt) + return method !== 'none' && offered.includes(method) + }) } const attempt = queue.shift() next((attempt ?? false) as Parameters[0]) diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index 619efcd111c..aa59a1f2eb5 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -268,6 +268,52 @@ describe('spawnSystemSsh', () => { expect(args).not.toContain('ProxyCommand=ignored') }) + it('states the stored endpoint when no Host block claims the alias', () => { + // A wildcard `Host *` supplies the proxy for every alias, so an alias whose own block is gone + // still reads as config-backed and gets dialled bare - the #11746 P1. + const args = buildSshArgs( + createTarget({ + source: 'ssh-config', + configHost: 'prod', + host: '10.0.0.5', + port: 2222, + username: 'deploy' + }), + { aliasClaimedByConfig: false } + ) + + expect(args).toContain('Hostname=10.0.0.5') + expect(args.slice(args.indexOf('-p'))).toContain('2222') + expect(args.slice(args.indexOf('-l'))).toContain('deploy') + // The alias is still the destination so OpenSSH keeps applying the wildcard's proxy. + expect(args.at(-1)).toBe('prod') + expect(args).not.toContain('deploy@prod') + expect(args).not.toContain('-i') + expect(args).not.toContain('-J') + }) + + it('stays a no-op when the stored endpoint matches the unclaimed alias', () => { + const args = buildSshArgs( + createTarget({ source: 'ssh-config', configHost: 'prod', host: 'prod', username: '' }), + { aliasClaimedByConfig: false } + ) + + expect(args.some((arg) => arg.startsWith('Hostname='))).toBe(false) + expect(args).not.toContain('-p') + expect(args).not.toContain('-l') + expect(args.at(-1)).toBe('prod') + }) + + it('leaves a manual target alone even when nothing claims its alias', () => { + const args = buildSshArgs( + createTarget({ source: 'manual', configHost: 'prod', host: '10.0.0.5', port: 2222 }), + { aliasClaimedByConfig: false } + ) + + expect(args.some((arg) => arg.startsWith('Hostname='))).toBe(false) + expect(args).toContain('deploy@prod') + }) + it('passes an explicit main-owned OpenSSH config as one argument', () => { const args = buildSshArgs(createTarget({ configHost: 'isolated-host', source: 'ssh-config' }), { configFile: '/tmp/orca isolated/ssh_config' diff --git a/src/main/ssh/system-ssh-args.ts b/src/main/ssh/system-ssh-args.ts index 82e29d0a494..70fc4318ab6 100644 --- a/src/main/ssh/system-ssh-args.ts +++ b/src/main/ssh/system-ssh-args.ts @@ -8,6 +8,14 @@ export type SystemSshBuildArgsOptions = { suppressOrcaControlMaster?: boolean gssapiOnly?: boolean nonInteractive?: boolean + /** + * `false` only when the parsed ssh_config proves no `Host`/`Match` block claims `configHost`. + * + * Absent or `true` keeps the config alias fully authoritative, which is right whenever a block + * really does name it — and is the only safe default, since a caller that cannot answer must not + * be read as having answered "nothing claims it". See `sshConfigMayClaimAlias`. + */ + aliasClaimedByConfig?: boolean } export function buildSshArgs(target: SshTarget, options?: SystemSshBuildArgsOptions): string[] { @@ -51,6 +59,15 @@ export function buildSshArgs(target: SshTarget, options?: SystemSshBuildArgsOpti const useConfigHost = shouldUseOpenSshConfigHost(target) + // Why: a wildcard `Host *` block supplies ProxyCommand/ProxyJump for every alias, so an alias + // whose own Host block was renamed or deleted still looks config-backed and gets dialled bare — + // as the wildcard's user, at the wildcard's host, discarding the endpoint Orca stored. Keep the + // system transport (OpenSSH must still apply that proxy) but state the stored endpoint, and only + // where the config proves no block claims the alias. + if (useConfigHost && options?.aliasClaimedByConfig === false) { + appendUnclaimedAliasEndpoint(args, target) + } + if (!useConfigHost && target.port !== 22) { args.push('-p', String(target.port)) } @@ -122,6 +139,9 @@ export function getSystemSshBuildArgsFromOperationOptions( if (options?.nonInteractive === true) { buildArgsOptions.nonInteractive = true } + if (options?.aliasClaimedByConfig === false) { + buildArgsOptions.aliasClaimedByConfig = false + } return Object.keys(buildArgsOptions).length === 0 ? undefined : buildArgsOptions } @@ -172,6 +192,27 @@ function hasEnabledControlPath(value: string | undefined): boolean { return normalized != null && normalized !== '' && normalized !== 'none' } +/** + * Restore only Hostname/Port/User, and only where they diverge from the alias. + * + * Not `-i`/`-J`/ProxyCommand: the wildcard block is still the route to this network, and `-o + * Hostname=` does not change which blocks OpenSSH selects (matching uses the original destination), + * so the proxy keeps applying and `%h` now expands to the host we actually mean. + */ +function appendUnclaimedAliasEndpoint(args: string[], target: SshTarget): void { + const alias = target.configHost + const storedHost = target.host.trim() + if (storedHost && alias && storedHost !== alias) { + args.push('-o', `Hostname=${storedHost}`) + } + if (target.port && target.port !== 22) { + args.push('-p', String(target.port)) + } + if (target.username) { + args.push('-l', target.username) + } +} + function shouldUseOpenSshConfigHost(target: SshTarget): boolean { if (!target.configHost) { return false diff --git a/src/shared/git-clone-failure-message.test.ts b/src/shared/git-clone-failure-message.test.ts index 71ca28617e7..9b32b25d714 100644 --- a/src/shared/git-clone-failure-message.test.ts +++ b/src/shared/git-clone-failure-message.test.ts @@ -71,4 +71,59 @@ describe('getGitCloneFailureMessage', () => { ) expect(usedLineSplit).toBe(false) }) + + it('names the non-interactive remote clone when a key could not be offered', () => { + const stderr = + "Cloning into 'repo'...\n" + + 'git@github.com: Permission denied (publickey).\n' + + 'fatal: Could not read from remote repository.\n' + + const message = getGitCloneFailureMessage(stderr) + + expect(message).toContain('fatal: Could not read from remote repository.') + expect(message).toContain('BatchMode=yes') + expect(message).toContain('ssh-add') + }) + + it('points host key failures at the machine that runs the clone', () => { + const stderr = 'Host key verification failed.\nfatal: Could not read from remote repository.\n' + + const message = getGitCloneFailureMessage(stderr) + + expect(message).toContain('known_hosts') + expect(message).not.toContain('ssh-add') + }) + + it('still explains where an unrecognised SSH clone failure ran', () => { + const stderr = + 'kex_exchange_identification: read: Connection reset by peer\nfatal: Could not read from remote repository.\n' + + expect(getGitCloneFailureMessage(stderr)).toContain('BatchMode=yes') + }) + + it('leaves non-SSH clone failures untouched', () => { + expect( + getGitCloneFailureMessage("fatal: repository 'https://github.com/org/repo.git/' not found") + ).toBe("fatal: repository 'https://github.com/org/repo.git/' not found") + }) + + it('withholds SSH guidance when the failing transport was HTTPS', () => { + // Git reuses this line for the HTTP remote helper, so the string alone does not prove SSH. + const stderr = + 'remote: Invalid username or token.\n' + + "fatal: Authentication failed for 'https://github.com/org/repo.git/'\n" + + 'fatal: Could not read from remote repository.\n' + + expect(getGitCloneFailureMessage(stderr)).not.toContain('BatchMode=yes') + }) + + it('does not repeat the guidance when a relay message is re-parsed', () => { + const relayMessage = `Clone failed: ${getGitCloneFailureMessage( + 'git@github.com: Permission denied (publickey).\nfatal: Could not read from remote repository.\n' + )}` + + const reparsed = getGitCloneFailureMessage(relayMessage) + + expect(reparsed.match(/BatchMode=yes/g)).toHaveLength(1) + }) }) diff --git a/src/shared/git-clone-failure-message.ts b/src/shared/git-clone-failure-message.ts index 44223dad59f..e706d523e5c 100644 --- a/src/shared/git-clone-failure-message.ts +++ b/src/shared/git-clone-failure-message.ts @@ -1,8 +1,53 @@ import { stripCredentialsFromMessage } from './git-remote-error' +// Why: clones run under nonInteractiveGitEnv (GIT_TERMINAL_PROMPT=0, empty SSH_ASKPASS, +// `ssh -o BatchMode=yes`) so a background clone cannot hang on a prompt nobody sees. The cost is +// that git's own SSH errors read identically to an ordinary permission problem, and on a remote +// or paired-runtime clone the user is looking at their own working local `git clone` while Orca +// fails — with nothing in the message saying the clone ran somewhere else, without their agent. +const CLONE_HOST_NOTE = + 'The clone runs non-interactively (BatchMode=yes) on the machine that will hold the repository, using the SSH keys and agent on that machine rather than the ones on this computer.' +const CLONE_KEY_HINT = `${CLONE_HOST_NOTE} A passphrase-protected key cannot prompt there, so load it into an agent on that machine (ssh-add) and retry.` +const CLONE_HOST_KEY_HINT = `${CLONE_HOST_NOTE} It has not trusted this host key yet — connect once from a shell on that machine to record it in its known_hosts.` + +/** An ssh(1) diagnostic, i.e. a line only the SSH transport can have produced. */ +const SSH_TRANSPORT_DIAGNOSTIC = + /\bssh|permission denied \(|connection (?:closed|reset|refused|timed out) by/i + export function getGitCloneFailureMessage( stderr: string, options: { clonePath?: string | null } = {} +): string { + return appendCloneTransportGuidance( + getGitCloneFailureLine(stderr, options), + stripCredentialsFromMessage(stderr) + ) +} + +/** Guidance the raw git error omits: where the clone ran, and why nothing could prompt there. */ +function appendCloneTransportGuidance(message: string, scrubbedStderr: string): string { + // Re-entrant: remote-repo-clone re-parses a message the relay already built. + if (message.includes(CLONE_HOST_NOTE)) { + return message + } + if (/host key verification failed/i.test(scrubbedStderr)) { + return `${message} ${CLONE_HOST_KEY_HINT}` + } + if (/permission denied \(([^)]*publickey[^)]*)\)/i.test(scrubbedStderr)) { + return `${message} ${CLONE_KEY_HINT}` + } + // Every other SSH-transport failure still needs the one fact the reporter was missing — but only + // once something proves the transport was SSH: git prints this same line for the HTTP remote + // helper, where a note about keys and agents is simply wrong. + return /could not read from remote repository/i.test(scrubbedStderr) && + SSH_TRANSPORT_DIAGNOSTIC.test(scrubbedStderr) + ? `${message} ${CLONE_HOST_NOTE}` + : message +} + +function getGitCloneFailureLine( + stderr: string, + options: { clonePath?: string | null } = {} ): string { let fallbackLine: string | null = null From 074a2366bf1e50d8c6b6e45656ac6b7201fc36de Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:58 -0700 Subject: [PATCH 11/77] test(e2e): pass testInfo to startDockerSshRelayTarget in the freeze repro (#18257) The spec called startDockerSshRelayTarget() with no argument while the helper signature is (testInfo: TestInfo) and dereferences testInfo.workerIndex, so it threw before any Orca code ran and took the Docker SSH lane red on every PR. Fixes #16764 --- tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts index ee9f8c3845a..a7754a02507 100644 --- a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts +++ b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts @@ -55,11 +55,11 @@ test.describe('R2 Docker SSH bulk-open freeze', () => { test('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({ orcaPage, registerPostElectronShutdownCleanup - }) => { + }, testInfo) => { test.setTimeout(420_000) let target: DockerSshRelayTarget | null = null try { - target = startDockerSshRelayTarget() + target = startDockerSshRelayTarget(testInfo) registerPostElectronShutdownCleanup(async () => { if (target) { cleanupDockerSshRelayTarget(target) From 7f8eb90ac3a7fa00102015f16f35dfbe923721cf Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:32:06 -0700 Subject: [PATCH 12/77] Align worktree host labels across desktop and mobile (#18237) * refactor: align worktree host labels across clients * fix(mobile): expose safe host display labels * fix(mobile): preserve legacy mixed-host labels --------- Co-authored-by: Merge Sim --- mobile/src/components/WorktreeListRow.test.ts | 41 ++++- mobile/src/components/WorktreeListRow.tsx | 37 +++- .../src/host-screen/use-host-repo-metadata.ts | 66 +++++++ .../host-screen/use-host-screen-controller.ts | 17 +- .../host-screen/use-host-screen-identity.ts | 6 + .../src/host-screen/use-host-screen-state.ts | 13 ++ mobile/src/worktree/workspace-list-types.ts | 4 + .../worktree-host-context-labels.test.ts | 164 ++++++++++++++++++ .../worktree/worktree-host-context-labels.ts | 92 ++++++++++ .../paired-settings.spec.ts | 6 + src/main/runtime/runtime-client-settings.ts | 16 +- src/main/runtime/runtime-store-contract.ts | 1 + .../worktree-list-groups-host-labels.test.ts | 47 +++++ .../worktree-list/grouping/host-labels.ts | 40 +++-- src/shared/worktree/host-context-labels.ts | 94 ++++++++++ 15 files changed, 618 insertions(+), 26 deletions(-) create mode 100644 mobile/src/worktree/worktree-host-context-labels.test.ts create mode 100644 mobile/src/worktree/worktree-host-context-labels.ts create mode 100644 src/shared/worktree/host-context-labels.ts diff --git a/mobile/src/components/WorktreeListRow.test.ts b/mobile/src/components/WorktreeListRow.test.ts index 01e0128cdc1..7b6d222d29e 100644 --- a/mobile/src/components/WorktreeListRow.test.ts +++ b/mobile/src/components/WorktreeListRow.test.ts @@ -30,7 +30,9 @@ vi.mock('lucide-react-native', () => ({ ChevronDown: 'ChevronDown', ChevronRight: 'ChevronRight', GitBranch: 'GitBranch', - GitPullRequest: 'GitPullRequest' + GitPullRequest: 'GitPullRequest', + Monitor: 'Monitor', + Server: 'Server' })) vi.mock('../platform/haptics', () => ({ triggerMediumImpact: vi.fn() })) @@ -225,4 +227,41 @@ describe('memoized worktree rows', () => { workingMode: 'monitoring' }) }) + + it('names the host with a glyph that matches the host kind', async () => { + const textNodes = (): string[] => + renderer!.root + .findAllByType('Text' as never) + .flatMap((node) => node.props.children) + .filter((child): child is string => typeof child === 'string') + + await act(async () => { + renderer = create( + createElement(ListRowHarness, { + item: { ...baseItem, hostId: 'ssh:ssh-1', hostContextLabel: 'openclaw' }, + now: 2_000 + }) + ) + }) + expect(textNodes()).toContain('openclaw') + expect(renderer!.root.findAllByType('Server' as never)).toHaveLength(1) + expect(renderer!.root.findAllByType('Monitor' as never)).toHaveLength(0) + + await act(async () => + renderer!.update( + createElement(ListRowHarness, { + item: { ...baseItem, hostContextLabel: 'Local Mac' }, + now: 2_000 + }) + ) + ) + expect(textNodes()).toContain('Local Mac') + expect(renderer!.root.findAllByType('Monitor' as never)).toHaveLength(1) + + await act(async () => + renderer!.update(createElement(ListRowHarness, { item: baseItem, now: 2_000 })) + ) + expect(textNodes()).not.toContain('Local Mac') + expect(renderer!.root.findAllByType('Monitor' as never)).toHaveLength(0) + }) }) diff --git a/mobile/src/components/WorktreeListRow.tsx b/mobile/src/components/WorktreeListRow.tsx index ba1c3865fc7..1de862e630c 100644 --- a/mobile/src/components/WorktreeListRow.tsx +++ b/mobile/src/components/WorktreeListRow.tsx @@ -1,6 +1,15 @@ import { memo } from 'react' -import { Bell, ChevronDown, ChevronRight, GitBranch, GitPullRequest } from 'lucide-react-native' +import { + Bell, + ChevronDown, + ChevronRight, + GitBranch, + GitPullRequest, + Monitor, + Server +} from 'lucide-react-native' import { Pressable, StyleSheet, Text, View } from 'react-native' +import { parseExecutionHostId, type ExecutionHostId } from '../../../src/shared/execution-host' import type { RepoIcon } from '../../../src/shared/repo-icon' import type { AgentWorkingMode } from '../../../src/shared/agent-status-types' import type { RuntimeWorktreeAgentRow } from '../../../src/shared/runtime-types' @@ -22,6 +31,11 @@ function displayBranch(branch: string): string { export type WorktreeListRowItem = { workspaceKind?: 'git' | 'folder-workspace' worktreeId: string + hostId?: ExecutionHostId + /** Present only when the list spans hosts; names the host this row runs on. */ + hostContextLabel?: string + /** Resolved host for the display label; present when legacy rows omit hostId. */ + hostContextHostId?: ExecutionHostId repo: string branch: string displayName: string @@ -150,6 +164,20 @@ function WorktreeListRowComponent({ Child )} + {item.hostContextLabel ? ( + + {/* Rows from hosts that predate hostId stamping are local: a remote row always carries one. */} + {(parseExecutionHostId(item.hostContextHostId ?? item.hostId)?.kind ?? 'local') === + 'local' ? ( + + ) : ( + + )} + + {item.hostContextLabel} + + + ) : null} {/* Repo glyph+name only when not already grouped under this repo; MobileRepoIcon falls back to a Folder (matching desktop's default) rather than a bare colored dot. */} @@ -306,6 +334,13 @@ const styles = StyleSheet.create({ fontSize: 10, color: colors.textMuted }, + hostBadge: { + flexShrink: 1, + maxWidth: 140 + }, + hostBadgeText: { + flexShrink: 1 + }, lineageToggle: { alignSelf: 'flex-start', flexDirection: 'row', diff --git a/mobile/src/host-screen/use-host-repo-metadata.ts b/mobile/src/host-screen/use-host-repo-metadata.ts index a5367a971ad..198efbaf3ea 100644 --- a/mobile/src/host-screen/use-host-repo-metadata.ts +++ b/mobile/src/host-screen/use-host-repo-metadata.ts @@ -1,13 +1,54 @@ import { useCallback } from 'react' +import { getRepoExecutionHostId } from '../../../src/shared/execution-host' import { setCachedRepos } from '../cache/repo-cache' import type { RpcClient } from '../transport/rpc-client' import type { ConnectionState, RpcSuccess } from '../transport/types' import type { RepoSummary } from '../worktree/host-worktree-rpc-types' import { repoColor } from '../worktree/repo-color' +import { + buildHostLabelById, + buildRepoHostIdByRepoId +} from '../worktree/worktree-host-context-labels' import type { HostScreenState } from './use-host-screen-state' const REPO_METADATA_REFRESH_MS = 60_000 +type SshTargetSummaryRow = { id: string; label: string } + +async function requestResult(client: RpcClient, method: string): Promise { + try { + const response = await client.sendRequest(method) + return response.ok ? (response as RpcSuccess).result : null + } catch { + // Best-effort: hosts that predate a method still list repos; labels degrade to host ids. + return null + } +} + +function readSshTargets(result: unknown): SshTargetSummaryRow[] { + const targets = (result as { targets?: unknown } | null)?.targets + if (!Array.isArray(targets)) { + return [] + } + return targets.filter( + (target): target is SshTargetSummaryRow => + typeof target === 'object' && + target !== null && + typeof (target as SshTargetSummaryRow).id === 'string' && + typeof (target as SshTargetSummaryRow).label === 'string' + ) +} + +function readHostPlatform(result: unknown): NodeJS.Platform | null { + const platform = (result as { platform?: unknown } | null)?.platform + return typeof platform === 'string' && platform ? (platform as NodeJS.Platform) : null +} + +function readHostSettingOverrides(result: unknown): unknown { + return (result as { settings?: { hostSettingOverrides?: unknown } } | null)?.settings + ?.hostSettingOverrides +} + export function useHostRepoMetadata(args: { client: RpcClient | null connState: ConnectionState @@ -20,7 +61,10 @@ export function useHostRepoMetadata(args: { fetchRepoMetadataInFlightRef, fetchRepoMetadataPendingRef, repoMetadataFetchedAtRef, + setHostLabelById, + setHostPlatform, setRepoColorsByName, + setRepoHostIdByRepoId, setRepoIconsByName, setRepoIdsByName } = state @@ -69,6 +113,28 @@ export function useHostRepoMetadata(args: { ) ) setRepoIdsByName(new Map(repoResult.repos.map((repo) => [repo.displayName, repo.id]))) + setRepoHostIdByRepoId(buildRepoHostIdByRepoId(repoResult.repos)) + // Why: rows only name their host when the list spans hosts, so a single-host + // catalog never pays for the label lookups. Counted over repos, not the id-keyed + // map: one repo id registered on two hosts is two hosts. + const hostIds = new Set(repoResult.repos.map((repo) => getRepoExecutionHostId(repo))) + if (hostIds.size > 1) { + const [sshTargets, hostSettings, hostPlatform] = await Promise.all([ + requestResult(requestClient, 'ssh.listTargetSummaries'), + requestResult(requestClient, 'settings.get'), + requestResult(requestClient, 'host.platform') + ]) + if (clientRef.current !== requestClient || hostId !== requestHostId) { + return + } + setHostLabelById( + buildHostLabelById({ + sshTargets: readSshTargets(sshTargets), + hostSettingOverrides: readHostSettingOverrides(hostSettings) + }) + ) + setHostPlatform(readHostPlatform(hostPlatform)) + } } while (fetchRepoMetadataPendingRef.current.has(requestClient)) } catch { // Repo metadata is decorative; the next refresh can retry. diff --git a/mobile/src/host-screen/use-host-screen-controller.ts b/mobile/src/host-screen/use-host-screen-controller.ts index 347fd36a72e..ad2bf9a5d77 100644 --- a/mobile/src/host-screen/use-host-screen-controller.ts +++ b/mobile/src/host-screen/use-host-screen-controller.ts @@ -14,6 +14,7 @@ import { useRelayRecoveryStatus } from '../transport/client-context-connection-metrics' import { applyWorktreeRowDisplayState } from '../worktree/worktree-host-row-identity' +import { applyWorktreeHostContextLabels } from '../worktree/worktree-host-context-labels' import { useWorkspaceSections } from '../worktree/use-workspace-sections' import { useHostRepoMetadata } from './use-host-repo-metadata' import { useHostScreenIdentity } from './use-host-screen-identity' @@ -95,17 +96,23 @@ export function useHostScreenController({ // Why: live `worktrees` is authoritative only while connected; under the amber // mount default, connecting/handshaking must keep the pre-reconnect list too. const base = connState === 'connected' ? state.worktrees : state.lastKnownWorktrees - return applyWorktreeRowDisplayState( - base, - state.sleptIds, - state.optimisticActiveWorktreeIdentity + return applyWorktreeHostContextLabels( + applyWorktreeRowDisplayState(base, state.sleptIds, state.optimisticActiveWorktreeIdentity), + { + repoHostIdByRepoId: state.repoHostIdByRepoId, + hostLabelById: state.hostLabelById, + hostPlatform: state.hostPlatform + } ) }, [ connState, state.worktrees, state.lastKnownWorktrees, state.sleptIds, - state.optimisticActiveWorktreeIdentity + state.optimisticActiveWorktreeIdentity, + state.repoHostIdByRepoId, + state.hostLabelById, + state.hostPlatform ]) const sectionsResult = useWorkspaceSections({ displayWorktrees, diff --git a/mobile/src/host-screen/use-host-screen-identity.ts b/mobile/src/host-screen/use-host-screen-identity.ts index c0b09f61714..0a45502090f 100644 --- a/mobile/src/host-screen/use-host-screen-identity.ts +++ b/mobile/src/host-screen/use-host-screen-identity.ts @@ -17,10 +17,13 @@ export function useHostScreenIdentity(args: { repoMetadataFetchedAtRef, setCatalogError, setError, + setHostLabelById, setHostName, + setHostPlatform, setLastKnownWorktrees, setPinnedIds, setRepoColorsByName, + setRepoHostIdByRepoId, setRepoIconsByName, setWorktrees, setWorktreesLoaded @@ -54,6 +57,9 @@ export function useHostScreenIdentity(args: { setError('') setRepoColorsByName(new Map()) setRepoIconsByName(new Map()) + setRepoHostIdByRepoId(new Map()) + setHostLabelById(new Map()) + setHostPlatform(null) repoMetadataFetchedAtRef.current = 0 // Why: useState initializer runs only on first mount, so re-seed the cache when Expo Router reuses this screen for a new hostId. const freshCache = hostId ? (getCachedWorktrees(hostId) as Worktree[] | null) : null diff --git a/mobile/src/host-screen/use-host-screen-state.ts b/mobile/src/host-screen/use-host-screen-state.ts index b27e15ffb91..ca6bd0d85e7 100644 --- a/mobile/src/host-screen/use-host-screen-state.ts +++ b/mobile/src/host-screen/use-host-screen-state.ts @@ -1,4 +1,5 @@ import { useRef, useState } from 'react' +import type { ExecutionHostId } from '../../../src/shared/execution-host' import type { RepoIcon } from '../../../src/shared/repo-icon' import type { WorkspaceStatusDefinition } from '../../../src/shared/worktree/types' import { getCachedWorktrees } from '../cache/worktree-cache' @@ -56,6 +57,12 @@ export function useHostScreenState(hostId: string | undefined, action: string | ) // displayName → repo id: filters key on repo id, but section headers/rows key on displayName, so bridge the two. const [repoIdsByName, setRepoIdsByName] = useState>(new Map()) + // Host-label inputs for rows: repo → host, SSH/override labels, and the host's own platform. + const [repoHostIdByRepoId, setRepoHostIdByRepoId] = useState>( + new Map() + ) + const [hostLabelById, setHostLabelById] = useState>(new Map()) + const [hostPlatform, setHostPlatform] = useState(null) const [showSortPicker, setShowSortPicker] = useState(false) const [showGroupPicker, setShowGroupPicker] = useState(false) const [showFilterModal, setShowFilterModal] = useState(false) @@ -93,13 +100,16 @@ export function useHostScreenState(hostId: string | undefined, action: string | fetchWorktreesInFlightRef, filters, groupMode, + hostLabelById, hostName, + hostPlatform, lastKnownWorktrees, newWorktreeModalRef, newWorktreeModalVisibleRef, optimisticActiveWorktreeIdentity, pinnedIds, repoColorsByName, + repoHostIdByRepoId, repoIconsByName, repoIdsByName, repoMetadataFetchedAtRef, @@ -113,11 +123,14 @@ export function useHostScreenState(hostId: string | undefined, action: string | setError, setFilters, setGroupMode, + setHostLabelById, setHostName, + setHostPlatform, setLastKnownWorktrees, setOptimisticActiveWorktreeIdentity, setPinnedIds, setRepoColorsByName, + setRepoHostIdByRepoId, setRepoIconsByName, setRepoIdsByName, setRouteActionState, diff --git a/mobile/src/worktree/workspace-list-types.ts b/mobile/src/worktree/workspace-list-types.ts index afa937f8baa..255e429e892 100644 --- a/mobile/src/worktree/workspace-list-types.ts +++ b/mobile/src/worktree/workspace-list-types.ts @@ -9,6 +9,10 @@ export type Worktree = { repoId: string hostId?: ExecutionHostId terminalPlatform?: NodeJS.Platform + /** Display-only; set when the list spans hosts, so rows say which host they run on. */ + hostContextLabel?: string + /** Resolved host for the display label; present when legacy rows omit hostId. */ + hostContextHostId?: ExecutionHostId repo: string branch: string displayName: string diff --git a/mobile/src/worktree/worktree-host-context-labels.test.ts b/mobile/src/worktree/worktree-host-context-labels.test.ts new file mode 100644 index 00000000000..a921ceebb90 --- /dev/null +++ b/mobile/src/worktree/worktree-host-context-labels.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it } from 'vitest' +import type { Worktree } from './workspace-list-types' +import { + applyWorktreeHostContextLabels, + buildHostLabelById, + buildRepoHostIdByRepoId, + getWorktreeHostContextLabels, + resolveWorktreeHostId +} from './worktree-host-context-labels' + +function worktree(overrides: Partial = {}): Worktree { + return { + workspaceKind: 'git', + worktreeId: 'repo-1::/home/me/orca', + repoId: 'repo-1', + repo: 'orca', + branch: 'main', + displayName: 'main', + path: '/home/me/orca', + liveTerminalCount: 0, + hasAttachedPty: false, + preview: '', + unread: false, + isPinned: false, + linkedPR: null, + ...overrides + } +} + +const sshHostId = 'ssh:ssh-1785104650217-eduhep' as const + +describe('buildHostLabelById', () => { + it('labels SSH targets by their registered label and lets a display override win', () => { + const labels = buildHostLabelById({ + sshTargets: [ + { id: 'ssh-1785104650217-eduhep', label: 'openclaw' }, + { id: 'ssh-blank', label: ' ' } + ], + hostSettingOverrides: { [sshHostId]: { displayLabel: 'openclaw (renamed)' } } + }) + expect(labels.get(sshHostId)).toBe('openclaw (renamed)') + expect(labels.has('ssh:ssh-blank')).toBe(false) + }) + + it('normalizes legacy raw SSH ids used by persisted display overrides', () => { + const labels = buildHostLabelById({ + sshTargets: [], + hostSettingOverrides: { 'ssh-1785104650217-eduhep': { displayLabel: 'openclaw' } } + }) + expect(labels.get(sshHostId)).toBe('openclaw') + }) + + it('accepts canonical SSH host ids from newer target-summary payloads', () => { + const labels = buildHostLabelById({ + sshTargets: [{ id: sshHostId, label: 'openclaw' }], + hostSettingOverrides: undefined + }) + expect(labels.get(sshHostId)).toBe('openclaw') + expect(labels.has('ssh:ssh:ssh-1785104650217-eduhep')).toBe(false) + }) + + it('tolerates a malformed settings payload', () => { + expect(buildHostLabelById({ sshTargets: [], hostSettingOverrides: 'nope' }).size).toBe(0) + expect(buildHostLabelById({ sshTargets: [], hostSettingOverrides: undefined }).size).toBe(0) + }) +}) + +describe('resolveWorktreeHostId', () => { + it('prefers the row host, then the repo host, then local', () => { + const repoHosts = buildRepoHostIdByRepoId([ + { id: 'repo-1', connectionId: 'ssh-1785104650217-eduhep' }, + { id: 'repo-2', executionHostId: 'runtime:env-1' }, + { id: 'repo-3' } + ]) + expect(resolveWorktreeHostId(worktree({ hostId: 'local', repoId: 'repo-1' }), repoHosts)).toBe( + 'local' + ) + expect(resolveWorktreeHostId(worktree({ repoId: 'repo-1' }), repoHosts)).toBe(sshHostId) + expect(resolveWorktreeHostId(worktree({ repoId: 'repo-2' }), repoHosts)).toBe('runtime:env-1') + expect(resolveWorktreeHostId(worktree({ repoId: 'repo-3' }), repoHosts)).toBe('local') + expect(resolveWorktreeHostId(worktree({ repoId: 'unknown' }), repoHosts)).toBe('local') + }) +}) + +describe('getWorktreeHostContextLabels', () => { + const sources = { + repoHostIdByRepoId: new Map(), + hostLabelById: new Map([[sshHostId, 'openclaw']]), + hostPlatform: 'darwin' as const + } + + it('returns nothing for a single-host list', () => { + const rows = [worktree({ hostId: 'local' }), worktree({ hostId: 'local', worktreeId: 'b' })] + expect(getWorktreeHostContextLabels(rows, sources)).toBeUndefined() + expect(applyWorktreeHostContextLabels(rows, sources)).toBe(rows) + }) + + it('names every row by host once the list spans hosts', () => { + const rows = [ + worktree({ hostId: 'local', worktreeId: 'a' }), + worktree({ hostId: sshHostId, worktreeId: 'b' }), + worktree({ hostId: 'ssh:unlabeled', worktreeId: 'c' }), + worktree({ hostId: 'runtime:env-1', worktreeId: 'd' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, sources) + expect(labeled.map((row) => row.hostContextLabel)).toEqual([ + 'Local Mac', + 'openclaw', + 'unlabeled', + 'env-1' + ]) + }) + + it('names the local host from the paired host platform, not the phone', () => { + const rows = [ + worktree({ hostId: 'local', worktreeId: 'a' }), + worktree({ hostId: sshHostId, worktreeId: 'b' }) + ] + const linux = applyWorktreeHostContextLabels(rows, { ...sources, hostPlatform: 'linux' }) + expect(linux[0].hostContextLabel).toBe('Local Linux') + const unknown = applyWorktreeHostContextLabels(rows, { ...sources, hostPlatform: null }) + expect(unknown[0].hostContextLabel).toBe('This computer') + }) + + it('keys labels by host-qualified identity so a shared id on two hosts gets two labels', () => { + const rows = [ + worktree({ hostId: 'local', worktreeId: 'same' }), + worktree({ hostId: sshHostId, worktreeId: 'same' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, sources) + expect(labeled.map((row) => row.hostContextLabel)).toEqual(['Local Mac', 'openclaw']) + }) + + it('falls back to the repo host for rows from hosts that predate hostId stamping', () => { + const rows = [ + worktree({ repoId: 'repo-local', worktreeId: 'a' }), + worktree({ repoId: 'repo-ssh', worktreeId: 'b' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, { + ...sources, + repoHostIdByRepoId: buildRepoHostIdByRepoId([ + { id: 'repo-local' }, + { id: 'repo-ssh', connectionId: 'ssh-1785104650217-eduhep' } + ]) + }) + expect(labeled.map((row) => row.hostContextLabel)).toEqual(['Local Mac', 'openclaw']) + expect(labeled.map((row) => row.hostContextHostId)).toEqual(['local', sshHostId]) + }) + + it('keeps labels distinct when legacy rows reuse an id across hosts', () => { + const rows = [ + worktree({ repoId: 'repo-local', worktreeId: 'same' }), + worktree({ repoId: 'repo-ssh', worktreeId: 'same' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, { + ...sources, + repoHostIdByRepoId: buildRepoHostIdByRepoId([ + { id: 'repo-local' }, + { id: 'repo-ssh', connectionId: 'ssh-1785104650217-eduhep' } + ]) + }) + expect(labeled.map((row) => row.hostContextLabel)).toEqual(['Local Mac', 'openclaw']) + }) +}) diff --git a/mobile/src/worktree/worktree-host-context-labels.ts b/mobile/src/worktree/worktree-host-context-labels.ts new file mode 100644 index 00000000000..de33c62ca5d --- /dev/null +++ b/mobile/src/worktree/worktree-host-context-labels.ts @@ -0,0 +1,92 @@ +import { + LOCAL_EXECUTION_HOST_ID, + getRepoExecutionHostId, + normalizeExecutionHostId, + type ExecutionHostId +} from '../../../src/shared/execution-host' +import { getMixedHostContextLabels as getSharedMixedHostContextLabels } from '../../../src/shared/worktree/host-context-labels' +import { composeWorktreeHostIdentity } from '../../../src/shared/worktree/host-qualified-identity' +export { + buildHostLabelById, + getHostContextLabel +} from '../../../src/shared/worktree/host-context-labels' +import type { RepoSummary } from './host-worktree-rpc-types' +import type { Worktree } from './workspace-list-types' + +export type HostLabelSources = { + /** Host id per repo id from repo.list; rows from hosts that predate `hostId` fall back to it. */ + repoHostIdByRepoId: ReadonlyMap + /** User-facing labels for non-local hosts: SSH target labels, then per-host display overrides. */ + hostLabelById: ReadonlyMap + /** The paired host's own platform; the phone's platform must never name the desktop. */ + hostPlatform: NodeJS.Platform | null +} + +export function buildRepoHostIdByRepoId( + repos: readonly Pick[] +): Map { + return new Map(repos.map((repo) => [repo.id, getRepoExecutionHostId(repo)])) +} + +export function resolveWorktreeHostId( + worktree: Pick, + repoHostIdByRepoId: ReadonlyMap +): ExecutionHostId { + return ( + normalizeExecutionHostId(worktree.hostId) ?? + repoHostIdByRepoId.get(worktree.repoId) ?? + LOCAL_EXECUTION_HOST_ID + ) +} + +function getResolvedWorktreeRowIdentity( + worktree: Pick, + repoHostIdByRepoId: ReadonlyMap +): string { + return composeWorktreeHostIdentity( + resolveWorktreeHostId(worktree, repoHostIdByRepoId), + worktree.worktreeId + ) +} + +// Kept as a local adapter so existing mobile imports remain stable. + +/** + * Host label per row identity, only when the list spans more than one host — a single-host + * list gains nothing from a badge on every row. Mirrors the desktop sidebar's mixed-host rule. + */ +export function getWorktreeHostContextLabels( + worktrees: readonly Worktree[], + sources: HostLabelSources +): Map | undefined { + return getSharedMixedHostContextLabels(worktrees, { + getHostId: (worktree) => resolveWorktreeHostId(worktree, sources.repoHostIdByRepoId), + // Legacy hosts omit row.hostId; key by the resolved repo owner so duplicate + // worktree ids from different hosts do not overwrite each other's label. + getIdentity: (worktree) => getResolvedWorktreeRowIdentity(worktree, sources.repoHostIdByRepoId), + sources + }) +} + +export function applyWorktreeHostContextLabels( + worktrees: Worktree[], + sources: HostLabelSources +): Worktree[] { + const labels = getWorktreeHostContextLabels(worktrees, sources) + if (!labels) { + return worktrees + } + return worktrees.map((worktree) => { + const hostContextLabel = labels.get( + getResolvedWorktreeRowIdentity(worktree, sources.repoHostIdByRepoId) + ) + if (!hostContextLabel) { + return worktree + } + return { + ...worktree, + hostContextLabel, + hostContextHostId: resolveWorktreeHostId(worktree, sources.repoHostIdByRepoId) + } + }) +} diff --git a/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts b/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts index 85a6099a8b2..ac3fdd4154d 100644 --- a/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts +++ b/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts @@ -23,6 +23,9 @@ describe('OrcaRuntimeService', () => { ...store, getSettings: () => ({ ...store.getSettings(), + hostSettingOverrides: { + 'ssh:target-1': { displayLabel: 'Build host', defaultWorktreeLocation: '/srv/worktrees' } + }, experimentalNewWorktreeCardStyle: true, compactWorktreeCards: true, minimaxGroupId: 'group-42', @@ -39,6 +42,9 @@ describe('OrcaRuntimeService', () => { minimaxUsageModels: 'general,abab6.5' }) expect(runtime.getClientSettings()).not.toHaveProperty('terminalQuickCommands') + expect(runtime.getClientSettings().hostSettingOverrides).toEqual({ + 'ssh:target-1': { displayLabel: 'Build host' } + }) expect(runtime.getClientTerminalQuickCommands()).toEqual(terminalQuickCommands) }) diff --git a/src/main/runtime/runtime-client-settings.ts b/src/main/runtime/runtime-client-settings.ts index 1fb815936e6..41a251ce649 100644 --- a/src/main/runtime/runtime-client-settings.ts +++ b/src/main/runtime/runtime-client-settings.ts @@ -10,6 +10,8 @@ import { } from '../../shared/terminal-quick-commands' import { haveSameDisabledTuiAgents } from '../../shared/tui-agent-selection' import type { GlobalSettings } from '../../shared/global-settings-types' +import { getHostDisplayLabelOverrides } from '../../shared/host-setting-overrides' +import type { ExecutionHostId } from '../../shared/execution-host' import type { TerminalQuickCommand } from '../../shared/terminal-quick-command-types' import { recordManagedHookInstallFailure } from '../agent-hooks/install-telemetry' import { applyAgentStatusHooksEnabled } from '../agent-hooks/managed-agent-hook-controls' @@ -37,6 +39,13 @@ export type RuntimeClientSettings = Pick< | 'artifactSharingEnabled' | 'worktreeVisibilityDefaults' | 'agentSkillSharingEnabled' +> & { + hostSettingOverrides: RuntimeHostDisplayLabelOverrides +} + +/** Safe paired projection: host labels only; filesystem defaults stay host-private. */ +export type RuntimeHostDisplayLabelOverrides = Partial< + Record > export type RuntimeClientSettingsUpdate = Pick< @@ -94,7 +103,12 @@ export class RuntimeClientSettingsController { prBotAuthorOverrides: settings.prBotAuthorOverrides ?? [], artifactSharingEnabled: isArtifactSharingEnabled(settings), worktreeVisibilityDefaults: settings.worktreeVisibilityDefaults ?? { external: 'hide' }, - agentSkillSharingEnabled: isAgentSkillSharingEnabled(settings) + agentSkillSharingEnabled: isAgentSkillSharingEnabled(settings), + hostSettingOverrides: Object.fromEntries( + [ + ...getHostDisplayLabelOverrides({ hostSettingOverrides: settings.hostSettingOverrides }) + ].map(([hostId, displayLabel]) => [hostId, { displayLabel }]) + ) as RuntimeHostDisplayLabelOverrides } } diff --git a/src/main/runtime/runtime-store-contract.ts b/src/main/runtime/runtime-store-contract.ts index aece6a8e7ad..649a8ac49b7 100644 --- a/src/main/runtime/runtime-store-contract.ts +++ b/src/main/runtime/runtime-store-contract.ts @@ -112,6 +112,7 @@ export type RuntimeStore = { terminalHiddenDeliveryGate?: GlobalSettings['terminalHiddenDeliveryGate'] terminalModelQueryAuthority?: GlobalSettings['terminalModelQueryAuthority'] worktreeVisibilityDefaults?: GlobalSettings['worktreeVisibilityDefaults'] + hostSettingOverrides?: GlobalSettings['hostSettingOverrides'] agentSkillSharingEnabled?: GlobalSettings['agentSkillSharingEnabled'] } // Why: narrow to `unknown` return so test mocks can return void without diff --git a/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts b/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts index a5f415c286a..e8cc793eeea 100644 --- a/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts +++ b/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts @@ -155,6 +155,53 @@ describe('buildRows with pinned worktrees', () => { ]) }) + it('uses the registered SSH target label for openclaw rows', () => { + const sshRepo: Repo = { + ...remoteRepo, + id: 'repo-openclaw', + connectionId: 'openclaw', + executionHostId: 'ssh:openclaw' + } + const sshWorktree: Worktree = { + ...remoteWorktree, + id: 'wt-openclaw', + repoId: sshRepo.id + } + const rows = buildRows( + 'workspace-status', + [worktree, sshWorktree], + new Map([ + [repo.id, repo], + [sshRepo.id, sshRepo] + ]), + null, + new Set(), + undefined, + undefined, + undefined, + {}, + new Map([ + [worktree.id, worktree], + [sshWorktree.id, sshWorktree] + ]), + false, + undefined, + [], + new Set(), + new Map(), + new Map(), + [], + undefined, + [], + new Map([['ssh:openclaw', 'openclaw']]) + ) + + expect(rows.filter((row) => row.type === 'item')).toMatchObject([ + { worktree: { id: worktree.id }, hostContextLabel: LOCAL_HOST_LABEL }, + { worktree: { id: sshWorktree.id }, hostContextLabel: 'openclaw' } + ]) + }) + it('shows distinct Orca server names when status grouping mixes runtime hosts', () => { const firstRepo: Repo = { ...repo, diff --git a/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts index e8265667cbb..d1c5c8be2d4 100644 --- a/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts +++ b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts @@ -1,11 +1,14 @@ import type { Repo } from '../../../../../../shared/repo-types' import type { Worktree } from '../../../../../../shared/worktree/types' import { - getExecutionHostLabel, getRepoExecutionHostId, getWorktreeExecutionHostId } from '../../../../../../shared/execution-host' import type { ExecutionHostId } from '../../../../../../shared/execution-host' +import { + getHostContextLabel, + getMixedHostContextLabels as getSharedMixedHostContextLabels +} from '../../../../../../shared/worktree/host-context-labels' import { getWorktreeHostIdentity } from '../../../../../../shared/worktree/host-qualified-identity' import { getProjectGroupingForRepo, @@ -15,7 +18,7 @@ import { import { getFolderWorkspaceHostId } from '../../folder-workspace-host-id' import type { RenderableFolderWorkspace } from './folder-workspace-lanes' -function getRepoHostId(repoId: string, repoMap: Map): string | null { +function getRepoHostId(repoId: string, repoMap: Map): ExecutionHostId | null { const repo = repoMap.get(repoId) return repo ? getRepoExecutionHostId(repo) : null } @@ -28,14 +31,14 @@ function getRepoHostLabel( ): string | null { const setup = projectIndex?.setupByRepoId.get(repoId) if (setup) { - return hostLabelById?.get(setup.hostId) ?? getExecutionHostLabel(setup.hostId) + return getHostContextLabel(setup.hostId, { hostLabelById }) } const repo = repoMap.get(repoId) if (!repo) { return null } const hostId = getRepoExecutionHostId(repo) - return hostLabelById?.get(hostId) ?? getExecutionHostLabel(hostId) + return getHostContextLabel(hostId, { hostLabelById }) } export function getMixedHostContextLabels( @@ -45,16 +48,22 @@ export function getMixedHostContextLabels( hostLabelById: ReadonlyMap | undefined ): Map | undefined { const labelsByRepoId = new Map() - const uniqueLabels = new Set() + // Host identity, not the rendered label, determines whether rows are ambiguous: + // two hosts can intentionally share a user-facing label. + const uniqueHostIds = new Set() for (const repoId of group.repoIds) { const label = getRepoHostLabel(repoId, repoMap, projectIndex, hostLabelById) if (!label) { continue } labelsByRepoId.set(repoId, label) - uniqueLabels.add(label) + const setup = projectIndex?.setupByRepoId.get(repoId) + const hostId = setup?.hostId ?? getRepoHostId(repoId, repoMap) + if (hostId) { + uniqueHostIds.add(hostId) + } } - return uniqueLabels.size > 1 ? labelsByRepoId : undefined + return uniqueHostIds.size > 1 ? labelsByRepoId : undefined } /** @@ -128,17 +137,12 @@ export function getMixedWorktreeHostContextLabels( hostLabelById: ReadonlyMap | undefined, defaultHostId: ExecutionHostId ): Map | undefined { - const labelsByIdentity = new Map() - const uniqueHostIds = new Set() - for (const worktree of worktrees) { - const hostId = getWorktreeExecutionHostId(worktree, repoMap.get(worktree.repoId), defaultHostId) - uniqueHostIds.add(hostId) - labelsByIdentity.set( - getWorktreeHostIdentity(worktree), - hostLabelById?.get(hostId) ?? getExecutionHostLabel(hostId) - ) - } - return uniqueHostIds.size > 1 ? labelsByIdentity : undefined + return getSharedMixedHostContextLabels(worktrees, { + getHostId: (worktree) => + getWorktreeExecutionHostId(worktree, repoMap.get(worktree.repoId), defaultHostId), + getIdentity: getWorktreeHostIdentity, + sources: { hostLabelById } + }) } export function getHostWorktreeCounts( diff --git a/src/shared/worktree/host-context-labels.ts b/src/shared/worktree/host-context-labels.ts new file mode 100644 index 00000000000..4b9481c441b --- /dev/null +++ b/src/shared/worktree/host-context-labels.ts @@ -0,0 +1,94 @@ +import { + getExecutionHostLabel, + getLocalExecutionHostLabel, + normalizeExecutionHostId, + parseExecutionHostId, + toSshExecutionHostId, + type ExecutionHostId +} from '../execution-host' +import type { GlobalSettings } from '../global-settings-types' +import { getHostDisplayLabelOverrides } from '../host-setting-overrides' + +/** Inputs used by every client when spelling a host in a workspace row. */ +export type HostContextLabelSources = { + /** Explicit labels (SSH target names and per-host display overrides). */ + hostLabelById?: ReadonlyMap + /** The execution host's platform; clients must not use the device platform. */ + hostPlatform?: NodeJS.Platform | null +} + +/** Canonical user-facing host label used by desktop and mobile workspace rows. */ +export function getHostContextLabel( + hostId: ExecutionHostId, + sources: HostContextLabelSources = {} +): string { + const override = sources.hostLabelById?.get(hostId)?.trim() + if (override) { + return override + } + if (parseExecutionHostId(hostId)?.kind === 'local') { + // An explicit null means the paired host platform is unknown (mobile); an + // omitted platform means use the current process (desktop). + if (sources.hostPlatform === null) { + return 'This computer' + } + return sources.hostPlatform !== undefined + ? getLocalExecutionHostLabel(sources.hostPlatform) + : getExecutionHostLabel(hostId) + } + return getExecutionHostLabel(hostId) +} + +/** + * Build labels for SSH targets and apply persisted display-name overrides. + * Target summaries have appeared both as raw target ids and canonical `ssh:` ids + * across protocol versions, so accept either representation. + */ +export function buildHostLabelById(args: { + sshTargets: readonly { id: string; label: string }[] + hostSettingOverrides: unknown +}): Map { + const labels = new Map() + for (const target of args.sshTargets) { + const label = target.label.trim() + if (!target.id.trim() || !label) { + continue + } + const hostId = normalizeExecutionHostId(target.id) ?? toSshExecutionHostId(target.id) + if (hostId) { + labels.set(hostId, label) + } + } + const overrides = + args.hostSettingOverrides && typeof args.hostSettingOverrides === 'object' + ? getHostDisplayLabelOverrides({ + hostSettingOverrides: args.hostSettingOverrides as GlobalSettings['hostSettingOverrides'] + }) + : new Map() + for (const [hostId, label] of overrides) { + const normalized = normalizeExecutionHostId(hostId) ?? toSshExecutionHostId(hostId) + if (normalized) { + labels.set(normalized, label) + } + } + return labels +} + +/** Generic mixed-host projection shared by desktop grouping and mobile sections. */ +export function getMixedHostContextLabels( + items: readonly T[], + args: { + getHostId: (item: T) => ExecutionHostId + getIdentity: (item: T) => string + sources?: HostContextLabelSources + } +): Map | undefined { + const labelsByIdentity = new Map() + const hostIds = new Set() + for (const item of items) { + const hostId = args.getHostId(item) + hostIds.add(hostId) + labelsByIdentity.set(args.getIdentity(item), getHostContextLabel(hostId, args.sources)) + } + return hostIds.size > 1 ? labelsByIdentity : undefined +} From 510305e57434b65f69966645073552ed5985668b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:42:08 -0700 Subject: [PATCH 13/77] fix(relay): signal capacity loss instead of dropping, hanging, or truncating (#17870) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three failures with one shape: a payload past a fixed capacity was met with silence, with a wait that never ends, or with a prefix presented as a whole. **The workspace snapshot was silently dropped.** `workspace.changed` carries the tab/session list, and a snapshot past the producer frame capacity (12288 B on a Node <=21 remote) was dropped with only a relay stderr line, so the client kept a stale list forever. The relay now publishes per client and, for a client whose sink refused the frame, sends a compact `workspace.stale` marker on the control lane; the client re-reads through `workspace.get`, whose lane is budgeted in megabytes rather than in one producer frame. A new JSON-RPC notification rather than a new field on `workspace.changed`: `normalizeSnapshot(undefined, ns)` yields revision 0 and an empty session, so a Rule-1 field would make an old client replace its tab list with nothing — worse than the drop. An old client ignores the unknown method and is exactly where it is today. The marker retention/retry machinery is extracted from the `fs.changed` overflow path and shared by both. **The Windows upload hung, and the fix for it could truncate.** `#16432` was attributed to `[Console]::In.ReadToEnd()` materializing the base64 bundle. That is not what the reporter measured: he also measured `new IO.StreamReader([Console]::OpenStandardInput())` — an incremental reader — hanging at 1 MB. The limit is in the stdin the host hands PowerShell over a non-pty ssh exec, not in the string the script builds. - `uploadFileViaSystemSsh` — the user file-import path — was piping a whole file into one Windows stdin, unchunked and untimed. That is the path large files take; it now chunks into 32 KB writes and bounds each wait. - The Windows directory upload reuses that single-file path rather than repeating a weaker copy of chunk-read + write-buffer; the `ino`/`dev` TOCTOU verification comes with it. - A Windows write needing more than one exec lands on a `.orca-partial` staging path and is published by rename, so a failed chunk cannot leave a truncated artifact under the real name. `exclusive` is enforced once at the rename, not on the first chunk, where a retry met its own leftovers. - The mkdir batch reads stdin through the stream reader the reporter measured surviving 50 KB, not `[Console]::In`, which he measured wedging at that size. - `waitForChannelClose` takes an optional bound. A wedged PowerShell stays alive at idle CPU and never closes, so without one the promise is simply never settled and the caller waits forever with no error to show. **Quick Open showed a prefix as the whole workspace.** The mechanism "a full page means there is more" only works if the caller named the cap, and the failing UI named none — it hardcoded `truncated: false`. Quick Open now names `QUICK_OPEN_LISTING_MAX_RESULTS` on both the Electron IPC hop and the runtime-RPC hop (the field #17954 added to `files.listAll`), and reads a full page as truncation. The local hop honours the cap too, which it previously ignored. Rebase note on `fs.listFiles`: an earlier revision of this work also clamped the host unconditionally, and #17934 escalated an uncapped request to an explicit error. #17954 has since landed and made an oversized reply streamable, which removes the premise — the host no longer has to choose between a prefix and a refusal, so it returns the whole listing when no limit is named and only clamps a limit it was given. Keeping either would have regressed #17954 and hard-failed three in-tree callers that deliberately pass no options (`runtime-file-commands-search-runtime-files.ts:81`, `filesystem-read-handlers.ts:125`, `runtime-file-commands-constructor.ts:41`). --- .../filesystem/filesystem-search-handlers.ts | 8 +- .../ipc/remote-workspace-stale-resync.test.ts | 150 +++++++++ src/main/ipc/remote-workspace-stale-resync.ts | 62 ++++ src/main/ipc/remote-workspace.ts | 56 +++- src/main/ssh/ssh-system-fallback.test.ts | 41 ++- .../ssh/system-ssh-file-binary-transfer.ts | 158 +++++++++- src/main/ssh/system-ssh-file-transfer.ts | 154 +++++----- .../ssh/system-ssh-operation-lifecycle.ts | 28 +- .../ssh/system-ssh-windows-upload.test.ts | 288 ++++++++++++++++++ ...fs-handler-list-files-result-limit.test.ts | 116 +++++++ src/relay/fs-handler.ts | 6 +- src/relay/relay-client-resync-marker.ts | 168 ++++++++++ src/relay/relay-watcher-event-emitter.ts | 147 +-------- src/relay/workspace-session-handler.ts | 15 +- .../workspace-snapshot-publication.test.ts | 166 ++++++++++ src/relay/workspace-snapshot-publication.ts | 49 +++ .../quick-open-file-list.react.test.tsx | 31 ++ .../src/components/quick-open-file-list.ts | 10 +- .../src/runtime/runtime-file-search-client.ts | 11 +- src/shared/remote-workspace-types.ts | 13 + 20 files changed, 1404 insertions(+), 273 deletions(-) create mode 100644 src/main/ipc/remote-workspace-stale-resync.test.ts create mode 100644 src/main/ipc/remote-workspace-stale-resync.ts create mode 100644 src/main/ssh/system-ssh-windows-upload.test.ts create mode 100644 src/relay/fs-handler-list-files-result-limit.test.ts create mode 100644 src/relay/relay-client-resync-marker.ts create mode 100644 src/relay/workspace-snapshot-publication.test.ts create mode 100644 src/relay/workspace-snapshot-publication.ts diff --git a/src/main/ipc/filesystem/filesystem-search-handlers.ts b/src/main/ipc/filesystem/filesystem-search-handlers.ts index f6dd8d57284..77a26c1e58d 100644 --- a/src/main/ipc/filesystem/filesystem-search-handlers.ts +++ b/src/main/ipc/filesystem/filesystem-search-handlers.ts @@ -223,7 +223,13 @@ export function registerFilesystemSearchHandlers(context: FilesystemHandlerConte signal: controller?.signal }) } - return await listQuickOpenFiles(args.rootPath, store, args.excludePaths, controller?.signal) + return await listQuickOpenFiles( + args.rootPath, + store, + args.excludePaths, + controller?.signal, + args.maxResults + ) } finally { listFilesCancellations.finish(event, args.requestToken, controller) } diff --git a/src/main/ipc/remote-workspace-stale-resync.test.ts b/src/main/ipc/remote-workspace-stale-resync.test.ts new file mode 100644 index 00000000000..a75e6d81971 --- /dev/null +++ b/src/main/ipc/remote-workspace-stale-resync.test.ts @@ -0,0 +1,150 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import { + REMOTE_WORKSPACE_STALE_NOTIFICATION, + type RemoteWorkspaceChangedEvent, + type RemoteWorkspaceSession +} from '../../shared/remote-workspace-types' + +const { getActiveMultiplexerMock, getSshConnectionStoreMock } = vi.hoisted(() => ({ + getActiveMultiplexerMock: vi.fn(), + getSshConnectionStoreMock: vi.fn() +})) + +vi.mock('electron', () => ({ + ipcMain: { handle: vi.fn(), removeHandler: vi.fn() } +})) + +vi.mock('./ssh', () => ({ + getActiveMultiplexer: getActiveMultiplexerMock, + getSshConnectionStore: getSshConnectionStoreMock +})) + +vi.mock('./remote-workspace-events', () => ({ + registerRemoteWorkspaceNotificationHandler: vi.fn(() => vi.fn()) +})) + +import { + _resetRemoteWorkspaceCachesForTests, + handleRemoteWorkspaceNotification, + registerRemoteWorkspaceHandlers +} from './remote-workspace' + +function session(activeTabId: string): RemoteWorkspaceSession { + return { + activeWorktreePath: '/remote/worktree', + activeTabId, + tabsByWorktreePath: { + '/remote/worktree': [{ id: activeTabId, worktreePath: '/remote/worktree' } as never] + }, + terminalLayoutsByTabId: {} + } +} + +describe('workspace.stale resync', () => { + const sent: RemoteWorkspaceChangedEvent[] = [] + const request = vi.fn() + const store = { getRepo: vi.fn(), getWorkspaceSession: vi.fn() } as unknown as Store + + beforeEach(() => { + sent.length = 0 + request.mockReset() + _resetRemoteWorkspaceCachesForTests() + getActiveMultiplexerMock.mockReset() + getActiveMultiplexerMock.mockImplementation(() => ({ request })) + getSshConnectionStoreMock.mockReset() + getSshConnectionStoreMock.mockImplementation(() => ({ + getTarget: (id: string) => ({ id, host: 'example.test', username: 'dev' }), + listTargets: () => [] + })) + const win = { + isDestroyed: () => false, + webContents: { + send: (_channel: string, event: RemoteWorkspaceChangedEvent) => sent.push(event) + } + } + registerRemoteWorkspaceHandlers(store, () => win as never) + }) + + it('re-reads the snapshot through workspace.get and publishes it to the renderer', async () => { + request.mockResolvedValue({ + namespace: 'target-1', + revision: 12, + updatedAt: 5, + schemaVersion: 1, + session: session('tab-from-other-device') + }) + + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + await vi.waitFor(() => expect(sent).toHaveLength(1)) + + expect(request).toHaveBeenCalledWith('workspace.get', { namespace: expect.any(String) }) + expect(sent[0].targetId).toBe('target-1') + expect(sent[0].snapshot.revision).toBe(12) + expect(sent[0].snapshot.session.activeTabId).toBe('tab-from-other-device') + // The marker names no author, so the renderer's own-echo filter must not discard the resync. + expect(sent[0].sourceClientId).toBeUndefined() + }) + + it('collapses a burst of markers into one extra read rather than one read per marker', async () => { + const released: ((value: unknown) => void)[] = [] + request.mockImplementation( + () => + new Promise((resolve) => { + released.push(resolve) + }) + ) + + for (let i = 0; i < 4; i++) { + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + } + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(1)) + + request.mockResolvedValue({ + namespace: 'target-1', + revision: 3, + updatedAt: 1, + schemaVersion: 1, + session: session('tab-a') + }) + released[0]?.({ + namespace: 'target-1', + revision: 2, + updatedAt: 1, + schemaVersion: 1, + session: session('tab-a') + }) + + // Exactly one follow-up read for the markers that landed mid-flight: never zero, never four. + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(2)) + await Promise.resolve() + expect(request).toHaveBeenCalledTimes(2) + }) + + it('stays silent when the re-read finds the session it already had', async () => { + request.mockResolvedValue({ + namespace: 'target-1', + revision: 4, + updatedAt: 1, + schemaVersion: 1, + session: session('tab-a') + }) + + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(1)) + await vi.waitFor(() => expect(sent).toHaveLength(1)) + + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(2)) + await Promise.resolve() + expect(sent).toHaveLength(1) + }) +}) diff --git a/src/main/ipc/remote-workspace-stale-resync.ts b/src/main/ipc/remote-workspace-stale-resync.ts new file mode 100644 index 00000000000..38c96977e97 --- /dev/null +++ b/src/main/ipc/remote-workspace-stale-resync.ts @@ -0,0 +1,62 @@ +import type { RemoteWorkspaceObservedSnapshot } from '../../shared/remote-workspace-types' +import type { SshTarget } from '../../shared/ssh-types' +import { getRemoteSnapshot } from './remote-workspace-relay-sync' +import { getCachedRemoteWorkspaceSnapshot } from './remote-workspace-snapshot-cache' +import { remoteWorkspaceSessionMatchesSnapshot } from './remote-workspace-snapshot-normalization' + +type PendingResync = { promise: Promise; requeued: boolean } + +const pendingByTargetId = new Map() + +export function _resetRemoteWorkspaceStaleResyncForTests(): void { + pendingByTargetId.clear() +} + +export function isRemoteWorkspaceResyncInFlight(targetId: string): boolean { + return pendingByTargetId.has(targetId) +} + +/** + * The relay told us it could not deliver a snapshot, so pull it. `workspace.get` is a response, and + * responses are admitted against the megabyte-scale control/legacy-response budget rather than the + * single ~12KB producer frame that refused the broadcast — the payload was never too big for the + * link, only for that one lane. + */ +export function resyncStaleRemoteWorkspace( + target: SshTarget, + deliver: (snapshot: RemoteWorkspaceObservedSnapshot) => void, + onError: (error: unknown) => void = () => {} +): Promise { + const existing = pendingByTargetId.get(target.id) + if (existing) { + // Why: a burst of markers must collapse to one extra read, but never to zero — a marker that + // arrived while a read was already in flight may describe a revision that read did not see. + existing.requeued = true + return existing.promise + } + const pending: PendingResync = { requeued: false, promise: Promise.resolve() } + pending.promise = (async () => { + try { + do { + pending.requeued = false + const previous = getCachedRemoteWorkspaceSnapshot(target.id) + const snapshot = await getRemoteSnapshot(target) + if (!snapshot) { + return + } + // Suppress the echo: our own patch response already cached this session, and re-publishing it + // makes the renderer rehydrate a state it authored. + if (remoteWorkspaceSessionMatchesSnapshot(previous, snapshot.session)) { + continue + } + deliver(snapshot) + } while (pending.requeued) + } catch (error) { + onError(error) + } finally { + pendingByTargetId.delete(target.id) + } + })() + pendingByTargetId.set(target.id, pending) + return pending.promise +} diff --git a/src/main/ipc/remote-workspace.ts b/src/main/ipc/remote-workspace.ts index 6a6acec6adc..74339e8f694 100644 --- a/src/main/ipc/remote-workspace.ts +++ b/src/main/ipc/remote-workspace.ts @@ -2,10 +2,13 @@ import { ipcMain, type BrowserWindow } from 'electron' import type { Store } from '../persistence' import { getActiveMultiplexer, getSshConnectionStore } from './ssh' import { exportRemoteWorkspaceSession } from '../../shared/remote-workspace-session-projection' -import type { - RemoteWorkspaceChangedEvent, - RemoteWorkspaceObservedPatchResult, - RemoteWorkspaceSession +import { + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + REMOTE_WORKSPACE_STALE_NOTIFICATION, + type RemoteWorkspaceChangedEvent, + type RemoteWorkspaceObservedPatchResult, + type RemoteWorkspaceObservedSnapshot, + type RemoteWorkspaceSession } from '../../shared/remote-workspace-types' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { getRepoIdFromWorktreeId } from '../../shared/worktree/id' @@ -29,6 +32,10 @@ import { rememberRemoteWorkspaceSnapshot } from './remote-workspace-snapshot-cache' import { normalizeSnapshot } from './remote-workspace-snapshot-normalization' +import { + _resetRemoteWorkspaceStaleResyncForTests, + resyncStaleRemoteWorkspace +} from './remote-workspace-stale-resync' let mainWindowGetter: (() => BrowserWindow | null) | null = null let unregisterRemoteWorkspaceNotifications: (() => void) | null = null @@ -36,6 +43,7 @@ let unregisterRemoteWorkspaceNotifications: (() => void) | null = null export function _resetRemoteWorkspaceCachesForTests(): void { clearRemoteWorkspaceSnapshotCache() clearRemoteWorkspacePatchTails() + _resetRemoteWorkspaceStaleResyncForTests() } export function _getRemoteWorkspaceCacheSizesForTests(): { @@ -119,12 +127,40 @@ function exportSessionForTarget( }) } +function sendRemoteWorkspaceChanged( + targetId: string, + snapshot: RemoteWorkspaceObservedSnapshot, + sourceClientId: string | undefined +): void { + const event: RemoteWorkspaceChangedEvent = { + targetId, + snapshot, + ...(sourceClientId !== undefined ? { sourceClientId } : {}) + } + const win = mainWindowGetter?.() + if (win && !win.isDestroyed()) { + win.webContents.send('remoteWorkspace:changed', event) + } +} + export function handleRemoteWorkspaceNotification( targetId: string, method: string, params: Record ): void { - if (method !== 'workspace.changed') { + if (method === REMOTE_WORKSPACE_STALE_NOTIFICATION) { + const target = getSshConnectionStore()?.getTarget(targetId) + if (!target) { + return + } + // No sourceClientId on the resynced event: the marker names no author, and guessing one would + // let the renderer's own-echo filter discard another device's change. + void resyncStaleRemoteWorkspace(target, (snapshot) => + sendRemoteWorkspaceChanged(targetId, snapshot, undefined) + ) + return + } + if (method !== REMOTE_WORKSPACE_CHANGED_NOTIFICATION) { return } const target = getSshConnectionStore()?.getTarget(targetId) @@ -139,15 +175,7 @@ export function handleRemoteWorkspaceNotification( sourceClientId === CLIENT_ID ? rememberLocallyPatchedRemoteWorkspaceSnapshot(targetId, snapshot) : rememberRemoteWorkspaceSnapshot(targetId, snapshot) - const event: RemoteWorkspaceChangedEvent = { - targetId, - snapshot: observedSnapshot, - sourceClientId - } - const win = mainWindowGetter?.() - if (win && !win.isDestroyed()) { - win.webContents.send('remoteWorkspace:changed', event) - } + sendRemoteWorkspaceChanged(targetId, observedSnapshot, sourceClientId) } export function registerRemoteWorkspaceHandlers( diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index aa59a1f2eb5..c366e899bf6 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -105,6 +105,13 @@ type EventedProcess = EventEmitter & { killed: boolean } +// Windows writes read their source asynchronously before spawning, so a close emitted straight +// after the call can beat the listener. Emit it from the spawn instead. +function closeOnceSpawned(proc: EventedProcess): EventedProcess { + setImmediate(() => proc.emit('close', 0, null)) + return proc +} + function createEventedProcess(): EventedProcess { const proc = new EventEmitter() as EventedProcess proc.stdin = Object.assign(new EventEmitter(), { @@ -704,7 +711,7 @@ describe('spawnSystemSsh', () => { it('writes files to Windows system SSH targets with PowerShell stdin bytes', async () => { const proc = createEventedProcess() - spawnMock.mockReturnValue(proc) + spawnMock.mockImplementation(() => closeOnceSpawned(proc)) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -713,7 +720,6 @@ describe('spawnSystemSsh', () => { '0.1.0', { hostPlatform } ) - proc.emit('close', 0, null) await expect(promise).resolves.toBeUndefined() const args = spawnMock.mock.calls[0][1] as string[] @@ -725,7 +731,7 @@ describe('spawnSystemSsh', () => { it('writes binary buffers to Windows system SSH targets with CreateNew mode', async () => { const proc = createEventedProcess() - spawnMock.mockReturnValue(proc) + spawnMock.mockImplementation(() => closeOnceSpawned(proc)) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeBufferViaSystemSsh( @@ -734,7 +740,6 @@ describe('spawnSystemSsh', () => { Buffer.from('png'), { hostPlatform, exclusive: true } ) - proc.emit('close', 0, null) await expect(promise).resolves.toBeUndefined() const args = spawnMock.mock.calls[0][1] as string[] @@ -775,7 +780,7 @@ describe('spawnSystemSsh', () => { it('forces standalone SSH for Windows file writes when requested', async () => { const proc = createEventedProcess() - spawnMock.mockReturnValue(proc) + spawnMock.mockImplementation(() => closeOnceSpawned(proc)) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -784,7 +789,6 @@ describe('spawnSystemSsh', () => { '0.1.0', { hostPlatform, disableControlMaster: true } ) - proc.emit('close', 0, null) await expect(promise).resolves.toBeUndefined() const args = spawnMock.mock.calls[0][1] as string[] @@ -793,7 +797,7 @@ describe('spawnSystemSsh', () => { expect(args[standaloneControlIdx + 1]).toBe('none') }) - it('uploads directories to Windows system SSH targets in one PowerShell batch', async () => { + it('uploads a Windows directory as a mkdir batch plus per-file writes, never one blob', async () => { const localDir = mkdtempSync(join(tmpdir(), 'orca-system-ssh-upload-')) writeFileSync(join(localDir, 'relay.js'), 'console.log("relay")') const spawned: EventedProcess[] = [] @@ -816,24 +820,17 @@ describe('spawnSystemSsh', () => { } const commands = spawnMock.mock.calls.map((call) => (call[1] as string[]).at(-1) ?? '') - expect(commands).toHaveLength(1) + // #16432: directories first (metadata only), then the file bytes on their own stdin. One batch + // meant base64-ing the whole bundle into a single PowerShell string, which the remote never read. + expect(commands).toHaveLength(2) expect(commands.every((command) => command.includes('powershell.exe'))).toBe(true) expect(commands.every((command) => !command.includes('/bin/sh'))).toBe(true) expect(commands.join('\n')).not.toContain('tar -xzf') - const payload = JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string) as { - kind: string - path: string - contentsBase64?: string - }[] - expect(payload).toEqual( - expect.arrayContaining([ - { kind: 'directory', path: 'C:/Users/me/.orca-remote/relay' }, - { - kind: 'file', - path: 'C:/Users/me/.orca-remote/relay/relay.js', - contentsBase64: Buffer.from('console.log("relay")').toString('base64') - } - ]) + expect(JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string)).toEqual([ + 'C:/Users/me/.orca-remote/relay' + ]) + expect(Buffer.from(spawned[1].stdin.end.mock.calls[0]?.[0] as Buffer).toString('utf-8')).toBe( + 'console.log("relay")' ) }) diff --git a/src/main/ssh/system-ssh-file-binary-transfer.ts b/src/main/ssh/system-ssh-file-binary-transfer.ts index e86aa3c9ae0..b0c5b662ed1 100644 --- a/src/main/ssh/system-ssh-file-binary-transfer.ts +++ b/src/main/ssh/system-ssh-file-binary-transfer.ts @@ -74,7 +74,14 @@ export async function writeBufferViaSystemSsh( ): Promise { throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - await writeBufferViaSystemSshWindows(target, remotePath, contents, options) + await writeWindowsBytesViaSystemSsh( + target, + remotePath, + contents.length, + (offset, maxBytes) => + Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), + options + ) return } @@ -119,16 +126,28 @@ export async function uploadFileViaSystemSsh( } throwIfAborted(options?.signal) - const isWindows = options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform) + if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { + // #16432: a Windows host cannot take a whole file through one stdin, however the local side + // paces it — see WINDOWS_STDIN_WRITE_CHUNK_BYTES. This is the path that carries the large + // files, so it is the one that has to be chunked and bounded. + await writeWindowsBytesViaSystemSsh( + target, + remotePath, + openedStat.size, + async (offset, maxBytes) => { + const buffer = Buffer.allocUnsafe(Math.min(maxBytes, openedStat.size - offset)) + const { bytesRead } = await handle.read(buffer, 0, buffer.length, offset) + return buffer.subarray(0, bytesRead) + }, + options + ) + return + } + const channel = spawnSystemSshCommand( target, - isWindows - ? makeWindowsWriteFileCommand(remotePath, options) - : makePosixWriteFileCommand(remotePath, options), - { - wrapCommand: !isWindows, - ...getSystemSshBuildArgsFromOperationOptions(options) - } + makePosixWriteFileCommand(remotePath, options), + getSystemSshBuildArgsFromOperationOptions(options) ) const input = handle.createReadStream({ autoClose: false }) try { @@ -153,11 +172,79 @@ export async function uploadFileViaSystemSsh( } } -async function writeBufferViaSystemSshWindows( +/** + * #16432: Windows PowerShell 5.1 stops draining a redirected stdin over a non-pty ssh exec + * somewhere between 50KB and 1MB, depending on the host's `DefaultShell`, and it hangs rather than + * failing. The reporter measured that on both constructs he tried — `[Console]::In.ReadToEnd()` and + * `new IO.StreamReader([Console]::OpenStandardInput())`, the latter reading incrementally, which is + * why the limit cannot be attributed to materializing the payload. `Stream.CopyTo` reads the same + * `[Console]::OpenStandardInput()` object with the same incremental `Read` loop, so nothing in it + * escapes that limit either: no single write may exceed what one stdin is known to carry. + * + * 32KB is an order of magnitude under the low end of the measured range, and under 50KB, which the + * reporter measured succeeding against a stream reader on the worse of the two `DefaultShell` + * settings. + */ +export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 + +/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ +export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 + +/** Suffix for the path a multi-exec Windows write lands on before it is published by rename. */ +export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' + +/** + * Splits one logical Windows write into stdin-sized execs. + * + * A write that needs more than one exec cannot land on the destination directly: a chunk failing + * mid-file would leave a truncated artifact under the real name with nothing marking it incomplete, + * and the retry would then meet its own leftovers — under `exclusive` the retry's `CreateNew` fails + * on them. Multi-exec creates therefore land on a staging path and are published by a rename, which + * is also where `exclusive` is enforced: once, at the destination, instead of smeared across the + * first chunk. A caller-requested append cannot be staged without reading the remote file back, so + * it keeps writing straight through, as its own protocol already implies. + */ +async function writeWindowsBytesViaSystemSsh( target: SshTarget, remotePath: string, - contents: Buffer, + totalBytes: number, + readChunk: (offset: number, maxBytes: number) => Promise, options: SystemSshWriteBufferOptions +): Promise { + throwIfAborted(options.signal) + const staged = !options.append && totalBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES + const writePath = staged ? `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` : remotePath + let offset = 0 + // An empty write still has to run: it is what creates (or truncates) the file. + do { + const chunk = await readChunk(offset, WINDOWS_STDIN_WRITE_CHUNK_BYTES) + if (chunk.length === 0 && offset < totalBytes) { + throw new Error(`Source ran short during upload of ${remotePath}`) + } + await writeWindowsChunkViaSystemSsh( + target, + writePath, + chunk, + { + ...options, + append: staged ? offset > 0 : options.append === true || offset > 0, + exclusive: staged ? false : options.exclusive === true && offset === 0 + }, + offset + ) + offset += chunk.length + } while (offset < totalBytes) + if (staged) { + await publishWindowsStagedWrite(target, writePath, remotePath, options) + } +} + +async function writeWindowsChunkViaSystemSsh( + target: SshTarget, + remotePath: string, + chunk: Buffer, + options: SystemSshWriteBufferOptions, + offset: number ): Promise { throwIfAborted(options.signal) const channel = spawnSystemSshCommand(target, makeWindowsWriteFileCommand(remotePath, options), { @@ -167,10 +254,37 @@ async function writeBufferViaSystemSshWindows( const closePromise = awaitWithSystemSshAbort( options.signal, () => channel.close(), - waitForChannelClose(channel, `write ${remotePath}`) + waitForChannelClose( + channel, + `write ${remotePath} at offset ${offset}`, + WINDOWS_STDIN_WRITE_TIMEOUT_MS + ) ) if (!options.signal?.aborted) { - channel.stdin.end(contents) + channel.stdin.end(chunk) + } + await closePromise +} + +async function publishWindowsStagedWrite( + target: SshTarget, + stagingPath: string, + remotePath: string, + options: SystemSshWriteBufferOptions +): Promise { + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand( + target, + makeWindowsPublishStagedFileCommand(stagingPath, remotePath, options.exclusive === true), + { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } + ) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, `publish ${remotePath}`, WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end() } await closePromise } @@ -193,6 +307,24 @@ function makeWindowsWriteFileCommand( ) } +// `File::Move` throws when the destination exists, which is exactly the exclusive contract; the +// non-exclusive caller asked to replace, so it deletes first (a no-op on an absent path). +function makeWindowsPublishStagedFileCommand( + stagingPath: string, + remotePath: string, + exclusive: boolean +): string { + return powerShellCommand( + [ + '$ErrorActionPreference = "Stop"', + `$staging = ${powerShellLiteral(stagingPath)}`, + `$path = ${powerShellLiteral(remotePath)}`, + ...(exclusive ? [] : ['[System.IO.File]::Delete($path)']), + '[System.IO.File]::Move($staging, $path)' + ].join('; ') + ) +} + function makePosixWriteFileCommand( remotePath: string, options?: { append?: boolean; exclusive?: boolean } diff --git a/src/main/ssh/system-ssh-file-transfer.ts b/src/main/ssh/system-ssh-file-transfer.ts index 72cb4425d11..f728c0eab02 100644 --- a/src/main/ssh/system-ssh-file-transfer.ts +++ b/src/main/ssh/system-ssh-file-transfer.ts @@ -1,6 +1,5 @@ import { spawn } from 'node:child_process' -import { constants } from 'node:fs' -import { lstat, open, readdir } from 'node:fs/promises' +import { lstat, readdir } from 'node:fs/promises' import { join as pathJoin } from 'node:path' import { pipeline } from 'node:stream/promises' import type { SshTarget } from '../../shared/ssh-types' @@ -22,7 +21,12 @@ import { waitForProcess, type ProcessResult } from './system-ssh-operation-lifecycle' -import { writeBufferViaSystemSsh } from './system-ssh-file-binary-transfer' +import { + uploadFileViaSystemSsh, + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS, + writeBufferViaSystemSsh +} from './system-ssh-file-binary-transfer' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -109,33 +113,36 @@ async function uploadDirectoryViaSystemSshWindows( if (!hostPlatform) { throw new Error('Windows system SSH upload requires a remote host platform') } - const entries = await collectWindowsUploadEntries( - localDir, - remoteDir, - hostPlatform, - options.signal - ) - await writeWindowsUploadPackageViaSystemSsh(target, entries, options) + const plan = await collectWindowsUploadPlan(localDir, remoteDir, hostPlatform, options.signal) + await createWindowsUploadDirectories(target, plan.directories, options) + for (const file of plan.files) { + throwIfAborted(options.signal) + // Reuses the single-file upload: it already opens O_NOFOLLOW, verifies the source did not + // change under it, and splits the bytes into stdin-sized writes staged under a partial name. + await uploadFileViaSystemSsh(target, file.localPath, file.remotePath, options) + } } -type WindowsUploadEntry = - | { - kind: 'directory' - path: string - } - | { - kind: 'file' - path: string - contentsBase64: string - } +type WindowsUploadPlan = { + directories: string[] + files: { localPath: string; remotePath: string }[] +} -async function collectWindowsUploadEntries( +/** + * #16432: this used to base64 every artifact into one JSON array and push the whole ~1.9MB string + * into one PowerShell stdin. Base64 inflates the payload 1.33x, and Windows PowerShell 5.1 cannot + * read a stdin that large over a non-pty ssh exec — it blocks forever instead of failing. Nothing + * about a directory upload requires one frame: the plan carries paths only, and the bytes go per + * file, in writes bounded by WINDOWS_STDIN_WRITE_CHUNK_BYTES. + */ +async function collectWindowsUploadPlan( localDir: string, remoteDir: string, hostPlatform: RemoteHostPlatform, - signal: AbortSignal | undefined -): Promise { - const entries: WindowsUploadEntry[] = [{ kind: 'directory', path: remoteDir }] + signal: AbortSignal | undefined, + plan: WindowsUploadPlan = { directories: [], files: [] } +): Promise { + plan.directories.push(remoteDir) const dirEntries = await readdir(localDir, { withFileTypes: true }) for (const entry of dirEntries) { throwIfAborted(signal) @@ -146,75 +153,68 @@ async function collectWindowsUploadEntries( continue } if (statResult.isDirectory()) { - entries.push( - ...(await collectWindowsUploadEntries(localPath, remotePath, hostPlatform, signal)) - ) + await collectWindowsUploadPlan(localPath, remotePath, hostPlatform, signal, plan) continue } - const buffer = await readLocalUploadFile(localPath, statResult) - entries.push({ kind: 'file', path: remotePath, contentsBase64: buffer.toString('base64') }) + plan.files.push({ localPath, remotePath }) } - return entries + return plan } -async function writeWindowsUploadPackageViaSystemSsh( +// Why the JSON envelope survives here: a path list is metadata, so this payload stays in the +// hundreds of bytes even for a deep tree. Batched anyway, so a pathological tree cannot walk back +// into the same stdin size that wedges PowerShell. +async function createWindowsUploadDirectories( target: SshTarget, - entries: WindowsUploadEntry[], + directories: readonly string[], options: SystemSshOperationOptions ): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsUploadPackageCommand(), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, 'windows relay upload') - ) - if (!options.signal?.aborted) { - channel.stdin.end(JSON.stringify(entries)) - } - await closePromise -} - -async function readLocalUploadFile( - localPath: string, - statResult: Awaited> -): Promise { - const handle = await open(localPath, constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0)) - try { - const openedStat = await handle.stat() - if ( - !openedStat.isFile() || - openedStat.size !== statResult.size || - (statResult.ino !== 0 && openedStat.ino !== 0 && openedStat.ino !== statResult.ino) || - (statResult.dev !== 0 && openedStat.dev !== 0 && openedStat.dev !== statResult.dev) - ) { - throw new Error(`File changed during upload: ${localPath}`) + let batch: string[] = [] + let batchBytes = 0 + const flush = async (): Promise => { + if (batch.length === 0) { + return } - return await handle.readFile() - } finally { - await handle.close() + const payload = JSON.stringify(batch) + batch = [] + batchBytes = 0 + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end(payload) + } + await closePromise } + for (const directory of directories) { + const entryBytes = Buffer.byteLength(directory) + 4 + if (batch.length > 0 && batchBytes + entryBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES) { + await flush() + } + batch.push(directory) + batchBytes += entryBytes + } + await flush() } -function makeWindowsUploadPackageCommand(): string { +function makeWindowsCreateDirectoriesCommand(): string { return powerShellCommand( [ '$ErrorActionPreference = "Stop"', - '$json = [Console]::In.ReadToEnd()', + // The reporter measured this reader surviving 50KB where `[Console]::In` wedged at the same + // size (#16432); the batch above stays under that. + '$reader = New-Object System.IO.StreamReader([Console]::OpenStandardInput())', + 'try { $json = $reader.ReadToEnd() } finally { $reader.Dispose() }', 'if ([string]::IsNullOrWhiteSpace($json)) { return }', - '$items = $json | ConvertFrom-Json', - 'foreach ($item in @($items)) {', - ' $path = [string]$item.path', - ' if ($item.kind -eq "directory") {', - ' $null = [System.IO.Directory]::CreateDirectory($path)', - ' continue', - ' }', - ' $parent = [System.IO.Path]::GetDirectoryName($path)', - ' if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', - ' [System.IO.File]::WriteAllBytes($path, [Convert]::FromBase64String([string]$item.contentsBase64))', + 'foreach ($path in @($json | ConvertFrom-Json)) {', + ' $null = [System.IO.Directory]::CreateDirectory([string]$path)', '}' ].join('; ') ) diff --git a/src/main/ssh/system-ssh-operation-lifecycle.ts b/src/main/ssh/system-ssh-operation-lifecycle.ts index c9c4424325b..5ebfb17ced1 100644 --- a/src/main/ssh/system-ssh-operation-lifecycle.ts +++ b/src/main/ssh/system-ssh-operation-lifecycle.ts @@ -3,13 +3,25 @@ import type { SystemSshCommandChannel } from './system-ssh-command' export type ProcessResult = { label: string; stderr: string } +/** + * `timeoutMs` bounds a remote consumer that never returns. Windows PowerShell 5.1 cannot drain a + * large redirected stdin over a non-pty ssh exec (#16432): the remote process stays alive at idle + * CPU, writes nothing, and never closes — so without a bound this promise is simply never settled + * and the caller waits forever with no error to show. + */ export function waitForChannelClose( channel: SystemSshCommandChannel, - label: string + label: string, + timeoutMs?: number ): Promise { return new Promise((resolve, reject) => { let stderr = '' + let timer: ReturnType | null = null const cleanup = (): void => { + if (timer) { + clearTimeout(timer) + timer = null + } channel.stderr.off('data', onStderrData) channel.off('error', onError) channel.off('close', onClose) @@ -18,6 +30,20 @@ export function waitForChannelClose( cleanup() fn(val as never) } + if (timeoutMs !== undefined) { + timer = setTimeout(() => { + // Settle before closing: the close we request would otherwise come back as a SIGTERM + // failure and mask the timeout, which is the only diagnosis a wedged remote gives. + settle( + reject, + new Error( + `${label} timed out after ${timeoutMs}ms with no response from the remote host: ${stderr.trim()}` + ) + ) + channel.close() + }, timeoutMs) + timer.unref?.() + } const onStderrData = (data: Buffer): void => { stderr += data.toString('utf-8') } diff --git a/src/main/ssh/system-ssh-windows-upload.test.ts b/src/main/ssh/system-ssh-windows-upload.test.ts new file mode 100644 index 00000000000..207c3e2df8e --- /dev/null +++ b/src/main/ssh/system-ssh-windows-upload.test.ts @@ -0,0 +1,288 @@ +/** + * #16432: the Windows relay upload pushed the whole bundle into one PowerShell stdin, which + * Windows PowerShell 5.1 cannot drain over a non-pty ssh exec — the remote blocks forever, and + * `waitForChannelClose()` had no timeout, so the UI sat at "Connecting…" with no error. Covered + * here: no write exceeds one stdin's worth on any Windows path (bundle upload *and* single-file + * upload, which is the one that carries large files), a partial write never lands under the real + * name, and a remote that never closes fails instead of hanging. + */ +import { EventEmitter } from 'node:events' +import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { PassThrough, Writable } from 'node:stream' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as SystemSshOperationLifecycle from './system-ssh-operation-lifecycle' + +const { spawnSystemSshCommandMock, waitForChannelCloseSpy } = vi.hoisted(() => ({ + spawnSystemSshCommandMock: vi.fn(), + waitForChannelCloseSpy: vi.fn() +})) + +vi.mock('./system-ssh-command', () => ({ + spawnSystemSshCommand: spawnSystemSshCommandMock +})) + +// Delegates to the real implementation; the spy only records whether each wait was given a bound. +vi.mock('./system-ssh-operation-lifecycle', async (importActual) => { + const actual = (await importActual()) as typeof SystemSshOperationLifecycle + waitForChannelCloseSpy.mockImplementation(actual.waitForChannelClose) + return { ...actual, waitForChannelClose: waitForChannelCloseSpy } +}) + +import { uploadDirectoryViaSystemSsh } from './system-ssh-file-transfer' +import { + uploadFileViaSystemSsh, + WINDOWS_STAGED_WRITE_SUFFIX, + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS, + writeBufferViaSystemSsh +} from './system-ssh-file-binary-transfer' +import { waitForChannelClose } from './system-ssh-operation-lifecycle' +import { getRemoteHostPlatform } from './ssh-remote-platform' +import type { SshTarget } from '../../shared/ssh-types' + +type FakeChannel = EventEmitter & { + stdin: Writable + stderr: PassThrough + close: () => void + written: Buffer +} + +const target = { id: 'win-1', host: 'win.example', username: 'dev' } as unknown as SshTarget +const hostPlatform = getRemoteHostPlatform('win32-x64') +const remoteRoot = 'C:/Users/dev/.orca-remote' + +/** Recover the script from `powershell.exe ... -EncodedCommand `. */ +function decodePowerShellCommand(command: string): string { + const encoded = /-EncodedCommand (\S+)/.exec(command)?.[1] + return encoded === undefined ? command : Buffer.from(encoded, 'base64').toString('utf16le') +} + +function createFakeChannel(onEnd: (channel: FakeChannel) => void): FakeChannel { + const channel = new EventEmitter() as FakeChannel + channel.written = Buffer.alloc(0) + channel.stderr = new PassThrough() + channel.stdin = new Writable({ + write(chunk, _encoding, callback) { + channel.written = Buffer.concat([channel.written, Buffer.from(chunk)]) + callback() + }, + final(callback) { + callback() + onEnd(channel) + } + }) + channel.close = () => channel.emit('close', null, 'SIGTERM') + return channel +} + +type RecordedCommand = { script: string; stdin: Buffer } + +describe('Windows upload stdin framing', () => { + let localDir: string + const commands: RecordedCommand[] = [] + /** Index of the spawn that should report a non-zero exit, to model a chunk failing mid-file. */ + let failAtSpawn = -1 + + const fileWrites = (): RecordedCommand[] => + commands.filter((command) => command.script.includes('FileMode]::')) + const writtenPath = (command: RecordedCommand): string => + /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1].replace(/''/g, "'") ?? '' + const fileMode = (command: RecordedCommand): string | undefined => + /FileMode\]::(\w+)/.exec(command.script)?.[1] + + beforeEach(() => { + commands.length = 0 + failAtSpawn = -1 + waitForChannelCloseSpy.mockClear() + localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) + spawnSystemSshCommandMock.mockReset() + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + return createFakeChannel((channel) => { + commands.push({ script: decodePowerShellCommand(command), stdin: channel.written }) + setImmediate(() => + spawnIndex === failAtSpawn + ? channel.emit('close', 1, null) + : channel.emit('close', 0, null) + ) + }) + }) + }) + + afterEach(async () => { + await rm(localDir, { recursive: true, force: true }) + }) + + it('never pushes a whole artifact bundle into one PowerShell stdin', async () => { + mkdirSync(join(localDir, 'node'), { recursive: true }) + // Comfortably past the ~50KB point at which the reporter measured PowerShell 5.1 wedging. + writeFileSync(join(localDir, 'node', 'relay.js'), Buffer.alloc(600 * 1024, 0x61)) + writeFileSync(join(localDir, 'index.js'), Buffer.alloc(300 * 1024, 0x62)) + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const largest = Math.max(...commands.map((command) => command.stdin.length)) + expect(largest).toBeLessThanOrEqual(WINDOWS_STDIN_WRITE_CHUNK_BYTES) + // The base64 + JSON envelope is gone entirely: nothing reads the bundle as one string. + expect(commands.some((command) => command.script.includes('FromBase64String'))).toBe(false) + // `[Console]::In` wedged at 50KB where the stream reader did not, so the mkdir batch — the one + // payload still read as a string — must use the reader the reporter measured surviving. + expect(commands.some((command) => command.script.includes('[Console]::In.ReadToEnd()'))).toBe( + false + ) + expect( + commands.filter((command) => command.script.includes('StreamReader([Console]::')) + ).toHaveLength(1) + }) + + it('bounds the single-file upload too, which is the path large files take', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) + const localPath = join(localDir, 'big.node') + writeFileSync(localPath, contents) + + await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/big.node`, { hostPlatform }) + + const writes = fileWrites() + expect(writes).toHaveLength(4) + expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( + WINDOWS_STDIN_WRITE_CHUNK_BYTES + ) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. + expect( + waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ).toBe(true) + }) + + it('writes every byte of every artifact across the chunked writes', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 2 + 17, 0x63) + writeFileSync(join(localDir, 'relay.js'), contents) + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const writes = fileWrites() + expect(writes).toHaveLength(3) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + // Only the first write creates the staging file; the rest must extend it or it is truncated. + expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append']) + }) + + it('still creates an empty artifact on the host', async () => { + writeFileSync(join(localDir, 'empty.txt'), '') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + expect(fileWrites().map(writtenPath)).toEqual([`${remoteRoot}/empty.txt`]) + expect(fileWrites()[0].stdin).toHaveLength(0) + expect(fileMode(fileWrites()[0])).toBe('Create') + }) + + it('lands a multi-chunk write on a staging path and publishes it by rename', async () => { + const remotePath = `${remoteRoot}/relay.js` + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + // Nothing touches the real name until every byte is on the host. + expect(fileWrites().map(writtenPath)).toEqual([ + `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`, + `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` + ]) + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).toContain('[System.IO.File]::Delete($path)') + }) + + it('leaves no truncated file under the real name when a chunk fails mid-file', async () => { + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + // Spawns: 0 = mkdir batch, 1..3 = chunk writes. Fail the second chunk. + failAtSpawn = 2 + + await expect( + uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + ).rejects.toThrow() + + expect(fileWrites().map(writtenPath)).not.toContain(`${remoteRoot}/relay.js`) + expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) + }) + + it('enforces exclusive once at the rename, so a retry is not blocked by its own leftovers', async () => { + const localPath = join(localDir, 'import.bin') + writeFileSync(localPath, Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + + await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + + // CreateNew on chunk one would fail against a leftover staging file from a failed attempt; + // `File::Move` raising on an existing destination is what carries the exclusive contract. + expect(fileWrites().map(fileMode)).toEqual(['Create', 'Append']) + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + }) + + it('keeps a single-chunk write on the destination, with the caller mode intact', async () => { + await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { + hostPlatform, + exclusive: true + }) + + expect(fileWrites()).toHaveLength(1) + expect(writtenPath(fileWrites()[0])).toBe(`${remoteRoot}/version`) + expect(fileMode(fileWrites()[0])).toBe('CreateNew') + expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) + }) + + it('appends onto the destination rather than staging, since append cannot be staged', async () => { + const remotePath = `${remoteRoot}/log.bin` + await writeBufferViaSystemSsh( + target, + remotePath, + Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1), + { hostPlatform, append: true } + ) + + expect(fileWrites().map(writtenPath)).toEqual([remotePath, remotePath]) + expect(fileWrites().map(fileMode)).toEqual(['Append', 'Append']) + }) +}) + +describe('waitForChannelClose bounding', () => { + it('fails a remote that accepts stdin and never closes, instead of waiting forever', async () => { + vi.useFakeTimers() + try { + const channel = createFakeChannel(() => {}) + const settled = vi.fn() + const promise = waitForChannelClose(channel as never, 'windows relay upload', 1_000) + promise.then(settled, settled) + + await vi.advanceTimersByTimeAsync(999) + expect(settled).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(2) + await expect(promise).rejects.toThrow(/timed out after 1000ms with no response/) + } finally { + vi.useRealTimers() + } + }) + + it('leaves an unbounded wait unbounded when no timeout is asked for', async () => { + vi.useFakeTimers() + try { + const channel = createFakeChannel(() => {}) + const settled = vi.fn() + // POSIX `cat` drains its stdin; only the Windows writes need the bound. + void waitForChannelClose(channel as never, 'posix write').then(settled, settled) + + await vi.advanceTimersByTimeAsync(60 * 60 * 1000) + expect(settled).not.toHaveBeenCalled() + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/src/relay/fs-handler-list-files-result-limit.test.ts b/src/relay/fs-handler-list-files-result-limit.test.ts new file mode 100644 index 00000000000..17876d76020 --- /dev/null +++ b/src/relay/fs-handler-list-files-result-limit.test.ts @@ -0,0 +1,116 @@ +/** + * #12547: the relay used to serialize an unbounded `fs.listFiles` reply into one response frame, + * which died as "Message too large" or over-capacity. #17954 fixed that by streaming the reply, so + * the size of a listing is no longer a correctness question and the host must NOT quietly impose a + * cap of its own — a caller that named no limit reads the array as the whole listing, and clients + * that predate `maxResults` on this call hardcode `truncated: false`, so a prefix would reach them + * as a complete tree with nothing on the wire to notice. The cap belongs to the caller; the host + * only clamps it to the ceiling the scan's retention budget assumes. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const { runListFilesScanMock } = vi.hoisted(() => ({ + runListFilesScanMock: vi.fn() +})) + +vi.mock('./fs-list-files-fallback-chain', () => ({ + runListFilesScan: runListFilesScanMock +})) + +vi.mock('@parcel/watcher', () => ({ subscribe: vi.fn() })) + +import { FsHandler } from './fs-handler' +import { RelayContext } from './context' +import type { RelayDispatcher } from './dispatcher' +import { QUICK_OPEN_LISTING_MAX_RESULTS } from '../shared/quick-open-listing-limits' + +type ListFilesHandler = ( + params: Record, + context?: { clientId: number } +) => Promise + +function createHandler(): { listFiles: ListFilesHandler; dispose: () => void } { + const requestHandlers = new Map() + const dispatcher = { + onRequest: (method: string, handler: ListFilesHandler) => requestHandlers.set(method, handler), + onNotification: vi.fn(), + onClientDetached: vi.fn(), + notify: vi.fn(), + notifyBulk: vi.fn(), + publishProducerNotification: vi.fn(() => true), + activeClientIds: () => [], + producerEnvelopeBudget: () => Number.MAX_SAFE_INTEGER + } as unknown as RelayDispatcher + const handler = new FsHandler(dispatcher, new RelayContext(), { + dispose: vi.fn(), + forgetRoot: vi.fn(), + subscribe: vi.fn() + }) + return { listFiles: requestHandlers.get('fs.listFiles')!, dispose: () => handler.dispose() } +} + +describe('fs.listFiles result limit', () => { + let listFiles: ListFilesHandler + let dispose: () => void + + beforeEach(() => { + runListFilesScanMock.mockReset() + runListFilesScanMock.mockResolvedValue([]) + const created = createHandler() + listFiles = created.listFiles + dispose = created.dispose + return () => dispose() + }) + + function scanMaxResults(): unknown { + // runListFilesScan(rootPath, excludePathPrefixes, signal, maxResults, searchQuery) + return runListFilesScanMock.mock.calls[0][3] + } + + it('leaves a request that omitted maxResults unbounded rather than silently prefixing it', async () => { + await listFiles({ rootPath: '/remote/root' }, { clientId: 1 }) + + expect(scanMaxResults()).toBeUndefined() + }) + + it('ignores a malformed maxResults rather than treating it as a cap', async () => { + await listFiles({ rootPath: '/remote/root', maxResults: 'all' }, { clientId: 1 }) + + expect(scanMaxResults()).toBeUndefined() + }) + + it('answers an uncapped request in full, however large the tree is', async () => { + const files = Array.from( + { length: QUICK_OPEN_LISTING_MAX_RESULTS + 500 }, + (_, index) => `f${index}` + ) + runListFilesScanMock.mockResolvedValue(files) + + await expect(listFiles({ rootPath: '/remote/root' }, { clientId: 1 })).resolves.toEqual(files) + }) + + it('hands a client that named a cap the prefix it asked for', async () => { + runListFilesScanMock.mockResolvedValue( + Array.from({ length: QUICK_OPEN_LISTING_MAX_RESULTS }, (_, index) => `f${index}`) + ) + + const files = await listFiles( + { rootPath: '/remote/root', maxResults: QUICK_OPEN_LISTING_MAX_RESULTS }, + { clientId: 1 } + ) + + expect(files).toHaveLength(QUICK_OPEN_LISTING_MAX_RESULTS) + }) + + it('keeps a smaller client limit and clamps a larger one', async () => { + await listFiles({ rootPath: '/remote/root', maxResults: 33 }, { clientId: 1 }) + expect(scanMaxResults()).toBe(33) + + runListFilesScanMock.mockClear() + await listFiles( + { rootPath: '/remote/root', maxResults: QUICK_OPEN_LISTING_MAX_RESULTS * 10 }, + { clientId: 2 } + ) + expect(scanMaxResults()).toBe(QUICK_OPEN_LISTING_MAX_RESULTS) + }) +}) diff --git a/src/relay/fs-handler.ts b/src/relay/fs-handler.ts index ec11a806bb8..d20d0d073d6 100644 --- a/src/relay/fs-handler.ts +++ b/src/relay/fs-handler.ts @@ -25,6 +25,7 @@ import { writeRelayFile } from './fs-path-mutation-requests' import { buildExcludePathPrefixes } from '../shared/quick-open-filter' +import { resolveQuickOpenResultLimit } from '../shared/quick-open-listing-limits' import { maybeStreamRpcResponse, type GitResponseStreamRegistry } from './git-response-stream' import { readRelayFileContent, readRelayFileStreamMetadata } from './fs-handler-file-read' import { readRelayFileRange } from './fs-handler-file-range' @@ -217,11 +218,14 @@ export class FsHandler { context?: RequestContext ): Promise { const rootPath = expandTilde(params.rootPath as string) + // Why no host-side default: #17954 made an oversized reply streamable, so a caller that names no + // limit gets its whole listing instead of an unannounced prefix it would report as complete. + // A requested limit is still clamped to the shared ceiling the scan's retention budget assumes. const maxResults = typeof params.maxResults === 'number' && Number.isInteger(params.maxResults) && params.maxResults > 0 - ? Math.min(params.maxResults, 20_001) + ? resolveQuickOpenResultLimit(params.maxResults) : undefined const searchQuery = typeof params.searchQuery === 'string' && params.searchQuery.trim().length > 0 diff --git a/src/relay/relay-client-resync-marker.ts b/src/relay/relay-client-resync-marker.ts new file mode 100644 index 00000000000..4d6353ea59b --- /dev/null +++ b/src/relay/relay-client-resync-marker.ts @@ -0,0 +1,168 @@ +import type { RelayDispatcher } from './dispatcher' + +type RetainedMarker = { params: Record; estimatedBytes: number } + +type MarkerState = { + method: string + // Why: one outstanding marker per (client, key) keeps sustained backpressure bounded. + inFlight: Set + // Rejected markers, retained per client so they can be republished when the control lane frees up. + pending: Map> + capacityUnsubscribes: Map void> +} + +/** + * Publishes a small "your view is stale, re-read it" notification on the control lane after a + * producer-lane payload was refused. Shared by every producer whose oversized frame would otherwise + * desync a client silently; the marker is coalesced per key and retried on capacity, never dropped. + */ +export type RelayClientResyncMarkerPublisher = { + emit(clientId: number, markerKey: string, params: Record): void + forgetClient(clientId: number): void +} + +export function createRelayClientResyncMarkerPublisher( + dispatcher: RelayDispatcher, + method: string +): RelayClientResyncMarkerPublisher { + const state: MarkerState = { + method, + inFlight: new Set(), + pending: new Map(), + capacityUnsubscribes: new Map() + } + // In-flight keys need no sweep here: closing a client settles every queued and written frame first. + dispatcher.onClientDetached((clientId) => { + // Not every detach retires the id: invalidateClient() detaches the primary without removing it and + // setWrite() revives it, so dropping the markers here would desync the state the reconnect restores. + if (dispatcher.isClientAttached(clientId)) { + return + } + forgetClientMarkers(state, clientId) + }) + return { + emit(clientId, markerKey, params) { + const retained = state.pending.get(clientId)?.get(markerKey) + if (retained) { + // Latest generation wins: a retained marker that has not been sent yet must not replay a + // projection the producer has already moved past. + retained.params = params + retained.estimatedBytes = dispatcher.notificationFrameBytes(method, params) + return + } + // Per key, never per client alone: an outstanding marker for one subject must not suppress another's resync. + if (state.inFlight.has(markerId(clientId, markerKey))) { + return + } + publishMarker(dispatcher, state, clientId, markerKey, params) + }, + forgetClient(clientId) { + forgetClientMarkers(state, clientId) + } + } +} + +function markerId(clientId: number, markerKey: string): string { + return `${clientId} ${markerKey}` +} + +function forgetClientMarkers(state: MarkerState, clientId: number): void { + // Unsubscribe first so no re-entrant flush can observe a half-cleared client. + state.capacityUnsubscribes.get(clientId)?.() + state.capacityUnsubscribes.delete(clientId) + state.pending.delete(clientId) +} + +// Why: the control lane — on the producer lane the marker would hit the same full queue that just +// rejected the payload and be dropped, silently desyncing the client. +function publishMarker( + dispatcher: RelayDispatcher, + state: MarkerState, + clientId: number, + markerKey: string, + params: Record, + estimatedBytes?: number +): void { + const key = markerId(clientId, markerKey) + const frameBytes = estimatedBytes ?? dispatcher.notificationFrameBytes(state.method, params) + state.inFlight.add(key) + let settled = false + const accepted = dispatcher.tryNotifyClient( + clientId, + state.method, + params, + (result) => { + // Settles on write, drop, or client close, so the slot can never leak. + settled = true + state.inFlight.delete(key) + if (result.ok) { + return + } + // A frame the sink never wrote leaves the client just as desynced as a rejected one — setWrite + // fails every queued and in-flight frame this way. Retain unconditionally: a real detach clears + // it through onClientDetached, which fires after this settlement. + retainMarker(dispatcher, state, clientId, markerKey, params, frameBytes) + }, + { controlOverflow: 'reject' } + ) + if (accepted || settled) { + return + } + // Admission rejection has no settlement callback: retain the marker instead of desyncing the client. + state.inFlight.delete(key) + retainMarker(dispatcher, state, clientId, markerKey, params, frameBytes) +} + +function retainMarker( + dispatcher: RelayDispatcher, + state: MarkerState, + clientId: number, + markerKey: string, + params: Record, + estimatedBytes: number +): void { + if (!state.capacityUnsubscribes.has(clientId)) { + const unsubscribe = dispatcher.onClientCapacity(clientId, () => + flushPendingMarkers(dispatcher, state, clientId) + ) + if (!unsubscribe) { + // The client went away between admission and arming, so there is nothing left to resync. + return + } + state.capacityUnsubscribes.set(clientId, unsubscribe) + } + const retained = state.pending.get(clientId) + if (retained) { + retained.set(markerKey, { params, estimatedBytes }) + return + } + state.pending.set(clientId, new Map([[markerKey, { params, estimatedBytes }]])) +} + +function flushPendingMarkers( + dispatcher: RelayDispatcher, + state: MarkerState, + clientId: number +): void { + const retained = state.pending.get(clientId) + if (!retained) { + return + } + for (const [markerKey, marker] of Array.from(retained)) { + // Capacity fires on every lane; retry only frames the control queue can admit now. + if (!dispatcher.canAdmitControlFrame(clientId, marker.estimatedBytes)) { + continue + } + // Drop before republishing so a synchronous settlement cannot see the marker as still pending — + // and skip keys a re-entrant flush already took, which would otherwise send the marker twice. + if (!retained.delete(markerKey)) { + continue + } + publishMarker(dispatcher, state, clientId, markerKey, marker.params, marker.estimatedBytes) + } + // Identity check: a re-entrant flush may have retired this set and armed a fresh one to keep. + if (retained.size > 0 || state.pending.get(clientId) !== retained) { + return + } + forgetClientMarkers(state, clientId) +} diff --git a/src/relay/relay-watcher-event-emitter.ts b/src/relay/relay-watcher-event-emitter.ts index 739ae89b2f8..d42c00824bb 100644 --- a/src/relay/relay-watcher-event-emitter.ts +++ b/src/relay/relay-watcher-event-emitter.ts @@ -1,6 +1,10 @@ import type { WatcherProcessEvent } from '../main/ipc/parcel-watcher-process' import { resolveRuntimePath } from '../shared/cross-platform-path' import type { RelayDispatcher } from './dispatcher' +import { + createRelayClientResyncMarkerPublisher, + type RelayClientResyncMarkerPublisher +} from './relay-client-resync-marker' type MappedWatcherEvent = { kind: string @@ -13,15 +17,7 @@ type WatcherBatchSizing = { batchBytes: number } -type OverflowMarkerState = { - // Why: one outstanding marker per (client, root) keeps sustained backpressure bounded. - inFlight: Set - // Rejected markers, retained per client so they can be republished when the control lane frees up. - pending: Map> - capacityUnsubscribes: Map void> -} - -const overflowMarkerStates = new WeakMap() +const overflowMarkerPublishers = new WeakMap() export function emitRelayWatcherEvents( dispatcher: RelayDispatcher, @@ -164,147 +160,26 @@ function publishWatcherBatchToClient( } } -function overflowMarkerState(dispatcher: RelayDispatcher): OverflowMarkerState { - const existing = overflowMarkerStates.get(dispatcher) +function overflowMarkerPublisher(dispatcher: RelayDispatcher): RelayClientResyncMarkerPublisher { + const existing = overflowMarkerPublishers.get(dispatcher) if (existing) { return existing } - const state: OverflowMarkerState = { - inFlight: new Set(), - pending: new Map(), - capacityUnsubscribes: new Map() - } - overflowMarkerStates.set(dispatcher, state) - // In-flight keys need no sweep here: closing a client settles every queued and written frame first. - dispatcher.onClientDetached((clientId) => { - // Not every detach retires the id: invalidateClient() detaches the primary without removing it and - // setWrite() revives it, so dropping the markers here would desync the tree the reconnect restores. - if (dispatcher.isClientAttached(clientId)) { - return - } - forgetClientMarkers(state, clientId) - }) - return state -} - -function forgetClientMarkers(state: OverflowMarkerState, clientId: number): void { - // Unsubscribe first so no re-entrant flush can observe a half-cleared client. - state.capacityUnsubscribes.get(clientId)?.() - state.capacityUnsubscribes.delete(clientId) - state.pending.delete(clientId) + const publisher = createRelayClientResyncMarkerPublisher(dispatcher, 'fs.changed') + overflowMarkerPublishers.set(dispatcher, publisher) + return publisher } function overflowMarkerParams(rootPath: string): Record { return { events: [{ kind: 'overflow', absolutePath: rootPath }] } } -// Why: the control lane — on the producer lane the marker would hit the same full queue that just -// rejected the batch and be dropped, silently desyncing the remote file tree. function emitWatcherOverflowToClient( dispatcher: RelayDispatcher, clientId: number, rootPath: string ): void { - const state = overflowMarkerState(dispatcher) - // Per root, never per client alone: an outstanding marker for one tree must not suppress another's resync. - if ( - state.inFlight.has(`${clientId} ${rootPath}`) || - state.pending.get(clientId)?.has(rootPath) === true - ) { - return - } - publishOverflowMarker(dispatcher, state, clientId, rootPath) -} - -function publishOverflowMarker( - dispatcher: RelayDispatcher, - state: OverflowMarkerState, - clientId: number, - rootPath: string, - estimatedBytes?: number -): void { - const key = `${clientId} ${rootPath}` - const params = overflowMarkerParams(rootPath) - const frameBytes = estimatedBytes ?? dispatcher.notificationFrameBytes('fs.changed', params) - state.inFlight.add(key) - let settled = false - const accepted = dispatcher.tryNotifyClient( - clientId, - 'fs.changed', - params, - (result) => { - // Settles on write, drop, or client close, so the slot can never leak. - settled = true - state.inFlight.delete(key) - if (result.ok) { - return - } - // A frame the sink never wrote leaves the tree just as desynced as a rejected one — setWrite - // fails every queued and in-flight frame this way. Retain unconditionally: a real detach clears - // it through onClientDetached, which fires after this settlement. - retainOverflowMarker(dispatcher, state, clientId, rootPath, frameBytes) - }, - { controlOverflow: 'reject' } - ) - if (accepted || settled) { - return - } - // Admission rejection has no settlement callback: retain the marker instead of desyncing the tree. - state.inFlight.delete(key) - retainOverflowMarker(dispatcher, state, clientId, rootPath, frameBytes) -} - -function retainOverflowMarker( - dispatcher: RelayDispatcher, - state: OverflowMarkerState, - clientId: number, - rootPath: string, - estimatedBytes: number -): void { - if (!state.capacityUnsubscribes.has(clientId)) { - const unsubscribe = dispatcher.onClientCapacity(clientId, () => - flushPendingOverflowMarkers(dispatcher, state, clientId) - ) - if (!unsubscribe) { - // The client went away between admission and arming, so there is nothing left to resync. - return - } - state.capacityUnsubscribes.set(clientId, unsubscribe) - } - const roots = state.pending.get(clientId) - if (roots) { - roots.set(rootPath, estimatedBytes) - return - } - state.pending.set(clientId, new Map([[rootPath, estimatedBytes]])) -} - -function flushPendingOverflowMarkers( - dispatcher: RelayDispatcher, - state: OverflowMarkerState, - clientId: number -): void { - const roots = state.pending.get(clientId) - if (!roots) { - return - } - for (const [rootPath, estimatedBytes] of Array.from(roots)) { - // Capacity fires on every lane; retry only frames the control queue can admit now. - if (!dispatcher.canAdmitControlFrame(clientId, estimatedBytes)) { - continue - } - // Drop before republishing so a synchronous settlement cannot see the marker as still pending — - // and skip roots a re-entrant flush already took, which would otherwise send the marker twice. - if (!roots.delete(rootPath)) { - continue - } - publishOverflowMarker(dispatcher, state, clientId, rootPath, estimatedBytes) - } - // Identity check: a re-entrant flush may have retired this set and armed a fresh one to keep. - if (roots.size > 0 || state.pending.get(clientId) !== roots) { - return - } - forgetClientMarkers(state, clientId) + overflowMarkerPublisher(dispatcher).emit(clientId, rootPath, overflowMarkerParams(rootPath)) } export function emitRelayWatcherOverflow( diff --git a/src/relay/workspace-session-handler.ts b/src/relay/workspace-session-handler.ts index c87a8b6076f..f9b3ef2e507 100644 --- a/src/relay/workspace-session-handler.ts +++ b/src/relay/workspace-session-handler.ts @@ -2,6 +2,7 @@ import { existsSync, mkdirSync, readFileSync, renameSync, writeFileSync } from ' import { homedir } from 'node:os' import { dirname, join } from 'node:path' import type { RelayDispatcher } from './dispatcher' +import { publishWorkspaceSnapshotChange } from './workspace-snapshot-publication' type RemoteWorkspaceSnapshot = { namespace: string @@ -151,11 +152,15 @@ export class WorkspaceSessionHandler { session: patch.session as Record } this.write(snapshot) - this.dispatcher.notify('workspace.changed', { - namespace, - snapshot, - sourceClientId: typeof params.clientId === 'string' ? params.clientId : undefined - }) + publishWorkspaceSnapshotChange( + this.dispatcher, + { + namespace, + snapshot, + sourceClientId: typeof params.clientId === 'string' ? params.clientId : undefined + }, + namespace + ) return { ok: true, snapshot } } diff --git a/src/relay/workspace-snapshot-publication.test.ts b/src/relay/workspace-snapshot-publication.test.ts new file mode 100644 index 00000000000..804a13f452b --- /dev/null +++ b/src/relay/workspace-snapshot-publication.test.ts @@ -0,0 +1,166 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { RelayDispatcher } from './dispatcher' +import { relayWriterControlReserve } from './dispatcher-writer-admission' +import { encodeJsonRpcFrame, MessageType, type JsonRpcRequest } from './protocol' +import { WorkspaceSessionHandler } from './workspace-session-handler' +import { + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + REMOTE_WORKSPACE_STALE_NOTIFICATION +} from '../shared/remote-workspace-types' + +// The relay runs on the REMOTE host, so the sink default is that host's Node major. +// Node <= 21 defaults to a 16KB high-water mark, which is the 12288B capacity issue #15238 reports. +const NODE21_HWM = 16 * 1024 + +function decodeNotifications( + written: Buffer[] +): { method: string; params: Record }[] { + return written + .filter((buf) => buf[0] === MessageType.Regular) + .map((buf) => { + const len = buf.readUInt32BE(9) + return JSON.parse(buf.subarray(13, 13 + len).toString('utf-8')) as { + method?: string + params?: Record + } + }) + .filter( + (msg): msg is { method: string; params: Record } => + typeof msg.method === 'string' + ) + .map((msg) => ({ method: msg.method, params: msg.params ?? {} })) +} + +/** A session shaped like the report: several worktrees, each with a handful of tabs. */ +function oversizedSession(worktrees: number, tabsPerWorktree: number): Record { + const tabsByWorktreePath: Record = {} + const terminalLayoutsByTabId: Record = {} + for (let w = 0; w < worktrees; w++) { + const worktreePath = `/home/dev/orca/workspaces/project/feature-branch-${w}` + tabsByWorktreePath[worktreePath] = Array.from({ length: tabsPerWorktree }, (_, t) => ({ + id: `tab-${w}-${t}-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a`, + title: `claude — feature-branch-${w} — pane ${t}`, + worktreePath, + kind: 'terminal', + startupCommand: 'claude --dangerously-skip-permissions', + cwd: worktreePath + })) + for (let t = 0; t < tabsPerWorktree; t++) { + terminalLayoutsByTabId[`tab-${w}-${t}-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a`] = { + direction: 'row', + panes: [ + { id: `pane-${w}-${t}-a`, size: 50, remoteSessionId: `orca-remote-${w}-${t}-a` }, + { id: `pane-${w}-${t}-b`, size: 50, remoteSessionId: `orca-remote-${w}-${t}-b` } + ] + } + } + } + return { + activeWorktreePath: '/home/dev/orca/workspaces/project/feature-branch-0', + activeTabId: 'tab-0-0-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a', + tabsByWorktreePath, + terminalLayoutsByTabId + } +} + +describe('workspace snapshot publication over a bounded producer frame', () => { + let baseDir: string + let dispatcher: RelayDispatcher + let written: Buffer[] + + beforeEach(() => { + baseDir = mkdtempSync(join(tmpdir(), 'orca-workspace-publication-')) + written = [] + dispatcher = new RelayDispatcher( + (data) => { + written.push(Buffer.from(data)) + return true + }, + { + writableHighWaterMark: () => NODE21_HWM, + writableLength: () => 0, + supportsWriteCallback: false + } + ) + new WorkspaceSessionHandler(dispatcher, baseDir) + }) + + afterEach(() => { + dispatcher.dispose() + rmSync(baseDir, { recursive: true, force: true }) + }) + + async function patch(session: Record, id: number): Promise { + const req: JsonRpcRequest = { + jsonrpc: '2.0', + id, + method: 'workspace.patch', + params: { + namespace: 'ssh_host_project', + baseRevision: id - 1, + clientId: 'client-a', + patch: { kind: 'replace-session', session } + } + } + dispatcher.feed(encodeJsonRpcFrame(req, id, 0)) + await Promise.resolve() + await Promise.resolve() + } + + it('publishes the snapshot inline while it fits the producer frame', async () => { + await patch(oversizedSession(1, 1), 1) + + const methods = decodeNotifications(written).map((msg) => msg.method) + expect(methods).toContain(REMOTE_WORKSPACE_CHANGED_NOTIFICATION) + expect(methods).not.toContain(REMOTE_WORKSPACE_STALE_NOTIFICATION) + }) + + it('tells the client its view is stale instead of dropping an oversized snapshot', async () => { + const session = oversizedSession(3, 8) + // Pin the premise: this really is over the 12288B capacity the issue reports, so the assertion + // below measures the drop path and not a payload that happened to fit. + const capacity = NODE21_HWM - relayWriterControlReserve(NODE21_HWM) + expect(capacity).toBe(12288) + expect( + dispatcher.notificationFrameBytes(REMOTE_WORKSPACE_CHANGED_NOTIFICATION, { + namespace: 'ssh_host_project', + snapshot: { + namespace: 'ssh_host_project', + revision: 1, + updatedAt: 0, + schemaVersion: 1, + session + }, + sourceClientId: 'client-a' + }) + ).toBeGreaterThan(capacity) + + await patch(session, 1) + + const notifications = decodeNotifications(written) + expect(notifications.map((msg) => msg.method)).not.toContain( + REMOTE_WORKSPACE_CHANGED_NOTIFICATION + ) + const stale = notifications.filter((msg) => msg.method === REMOTE_WORKSPACE_STALE_NOTIFICATION) + expect(stale).toHaveLength(1) + expect(stale[0].params).toEqual({ namespace: 'ssh_host_project' }) + }) + + it('carries no revision or author, so a coalesced marker cannot replay a superseded generation', async () => { + const session = oversizedSession(3, 8) + await patch(session, 1) + written = [] + await patch({ ...session, activeTabId: 'tab-1-1-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a' }, 2) + + for (const marker of decodeNotifications(written).filter( + (msg) => msg.method === REMOTE_WORKSPACE_STALE_NOTIFICATION + )) { + expect(marker.params).not.toHaveProperty('revision') + expect(marker.params).not.toHaveProperty('sourceClientId') + expect(marker.params).not.toHaveProperty('snapshot') + } + }) +}) diff --git a/src/relay/workspace-snapshot-publication.ts b/src/relay/workspace-snapshot-publication.ts new file mode 100644 index 00000000000..3a621cb1172 --- /dev/null +++ b/src/relay/workspace-snapshot-publication.ts @@ -0,0 +1,49 @@ +import { + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + REMOTE_WORKSPACE_STALE_NOTIFICATION +} from '../shared/remote-workspace-types' +import type { RelayDispatcher } from './dispatcher' +import { + createRelayClientResyncMarkerPublisher, + type RelayClientResyncMarkerPublisher +} from './relay-client-resync-marker' + +const stalePublishers = new WeakMap() + +function stalePublisher(dispatcher: RelayDispatcher): RelayClientResyncMarkerPublisher { + const existing = stalePublishers.get(dispatcher) + if (existing) { + return existing + } + const publisher = createRelayClientResyncMarkerPublisher( + dispatcher, + REMOTE_WORKSPACE_STALE_NOTIFICATION + ) + stalePublishers.set(dispatcher, publisher) + return publisher +} + +/** + * Per client, because frame capacity is per client: one peer on a small sink must not cost the + * others their snapshot, and a peer that cannot take the snapshot still learns it is behind. + * The marker deliberately carries no revision or author — a coalesced marker would then replay a + * generation the producer has moved past, which is the same silent staleness this exists to remove. + */ +export function publishWorkspaceSnapshotChange( + dispatcher: RelayDispatcher, + params: Record, + namespace: string +): void { + for (const clientId of dispatcher.activeClientIds()) { + if ( + dispatcher.publishProducerNotification( + clientId, + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + params + ) + ) { + continue + } + stalePublisher(dispatcher).emit(clientId, namespace, { namespace }) + } +} diff --git a/src/renderer/src/components/quick-open-file-list.react.test.tsx b/src/renderer/src/components/quick-open-file-list.react.test.tsx index b59f08fae93..2673e90af76 100644 --- a/src/renderer/src/components/quick-open-file-list.react.test.tsx +++ b/src/renderer/src/components/quick-open-file-list.react.test.tsx @@ -7,6 +7,7 @@ import type { FolderWorkspace } from '../../../shared/folder-workspace-types' import type { ProjectGroup } from '../../../shared/project-group-types' import type { Worktree } from '../../../shared/worktree/types' import { folderWorkspaceKey } from '../../../shared/workspace-scope' +import { QUICK_OPEN_LISTING_MAX_RESULTS } from '../../../shared/quick-open-listing-limits' import { useAppStore } from '@/store' import type { AppState } from '@/store/types' import { useRuntimeFileListForWorktree, type RuntimeFileListState } from './quick-open-file-list' @@ -201,10 +202,40 @@ describe('useRuntimeFileListForWorktree', () => { rootPath: '/srv/platform', excludePaths: undefined, requestToken: expect.any(String), + // #12547: the caller names the cap so a full page is readable as truncation. + maxResults: QUICK_OPEN_LISTING_MAX_RESULTS, signal: expect.any(AbortSignal) } ) expect(states.at(-1)?.files).toEqual(['packages/app/package.json']) + expect(states.at(-1)?.truncated).toBe(false) + }) + + // #12547: the host stops at the cap the caller names, so a full page is a prefix. Reporting + // truncated:false unconditionally is what left the user with a silent partial list. + it('reports a capped listing as truncated instead of as the whole workspace', async () => { + const states: RuntimeFileListState[] = [] + const workspaceKey = folderWorkspaceKey('folder-workspace-1') + listRuntimeFilesMock.mockResolvedValue( + Array.from({ length: QUICK_OPEN_LISTING_MAX_RESULTS }, (_, i) => `src/file-${i}.ts`) + ) + + useAppStore.setState({ + folderWorkspaces: [makeFolderWorkspace({ connectionId: 'ssh-1' })], + projectGroups: [makeProjectGroup({ connectionId: 'ssh-1' })], + repos: [], + worktreesByRepo: {} + } as Partial) + + await renderProbe({ + enabled: true, + onState: (state) => states.push(state), + worktreeId: workspaceKey + }) + await waitForListRuntimeFilesCall() + + expect(states.at(-1)?.files).toHaveLength(QUICK_OPEN_LISTING_MAX_RESULTS) + expect(states.at(-1)?.truncated).toBe(true) }) it('routes paired folder workspace queries to the owning runtime', async () => { diff --git a/src/renderer/src/components/quick-open-file-list.ts b/src/renderer/src/components/quick-open-file-list.ts index 04490ed7af6..7f605ce2e41 100644 --- a/src/renderer/src/components/quick-open-file-list.ts +++ b/src/renderer/src/components/quick-open-file-list.ts @@ -5,6 +5,7 @@ import type { Worktree } from '../../../shared/worktree/types' import { isWindowsAbsolutePathLike } from '../../../shared/cross-platform-path' import { createBrowserUuid } from '@/lib/browser-uuid' import { isQuickOpenRemoteQueryTooLarge } from '@/components/quick-open-search' +import { QUICK_OPEN_LISTING_MAX_RESULTS } from '../../../shared/quick-open-listing-limits' import { cancelRuntimeFileList, listRuntimeFiles, @@ -261,8 +262,15 @@ export function useRuntimeFileListForWorktree({ rootPath: worktreePath, excludePaths, requestToken, + maxResults: QUICK_OPEN_LISTING_MAX_RESULTS, signal: requestAbortController.signal - }).then((files) => ({ files, truncated: false })) + }).then((files) => ({ + // #12547: naming the cap is what makes a full page readable as "there is more". Reporting + // false unconditionally is what made the truncation silent — the host bounds the scan to + // the cap it is given, so a full page means there are more paths behind it. + files, + truncated: files.length >= QUICK_OPEN_LISTING_MAX_RESULTS + })) void request .then((result) => { diff --git a/src/renderer/src/runtime/runtime-file-search-client.ts b/src/renderer/src/runtime/runtime-file-search-client.ts index 31a91b2f150..8168b24bd54 100644 --- a/src/renderer/src/runtime/runtime-file-search-client.ts +++ b/src/renderer/src/runtime/runtime-file-search-client.ts @@ -48,6 +48,10 @@ export async function listRuntimeFiles( rootPath: string excludePaths?: string[] requestToken?: string + // Why: naming the cap is what makes a full page readable as "there is more". The host returns + // the whole listing when no limit is named, so a caller that never states one cannot tell a + // bound from a total. + maxResults?: number signal?: AbortSignal } ): Promise { @@ -57,7 +61,8 @@ export async function listRuntimeFiles( rootPath: args.rootPath, connectionId: context.connectionId, excludePaths: args.excludePaths, - requestToken: args.requestToken + requestToken: args.requestToken, + ...(args.maxResults === undefined ? {} : { maxResults: args.maxResults }) }) } return callRuntimeRpc( @@ -65,7 +70,9 @@ export async function listRuntimeFiles( 'files.listAll', { worktree: toRuntimeWorktreeSelector(context.worktreeId), - excludePaths: args.excludePaths + excludePaths: args.excludePaths, + // Optional on the host schema since #17954; an older host strips it and keeps its own default. + ...(args.maxResults === undefined ? {} : { maxResults: args.maxResults }) }, { timeoutMs: 15_000, ...(args.signal === undefined ? {} : { signal: args.signal }) } ) diff --git a/src/shared/remote-workspace-types.ts b/src/shared/remote-workspace-types.ts index f7beea81e8a..4d6eec6021e 100644 --- a/src/shared/remote-workspace-types.ts +++ b/src/shared/remote-workspace-types.ts @@ -59,6 +59,19 @@ export type RemoteWorkspaceObservedPatchResult = message?: string } +export const REMOTE_WORKSPACE_CHANGED_NOTIFICATION = 'workspace.changed' + +/** + * Sent instead of `workspace.changed` when the snapshot frame did not fit the client's producer + * frame capacity. Carries no session: the client re-reads through `workspace.get`, whose response + * lane is budgeted in megabytes rather than in one ~12KB producer frame. + * + * Wire contract: a relay that predates this never sends it, and a client that predates it drops it + * the same way it drops any unknown notification method — which is exactly the silent drop this + * replaces, so an un-negotiated pairing is never worse than before. + */ +export const REMOTE_WORKSPACE_STALE_NOTIFICATION = 'workspace.stale' + export type RemoteWorkspaceChangedEvent = { targetId: string snapshot: RemoteWorkspaceObservedSnapshot From 6cd477a2f14d5efb2890023d60b65fb495949d9a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:58:42 -0700 Subject: [PATCH 14/77] test(e2e): un-rot the SSH freeze repro and probe two failure modes nothing covered (#17940) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Test-only. No production code. ## The freeze repro was rotted in three ways, not one #16764 tracks four stale call sites. There were three separate problems: 1. **Stale call sites** — `execInTerminal` gained a `ptyId` and `splitActiveTerminalPane` gained a direction. (`startDockerSshRelayTarget`'s missing `testInfo` was the third; #18257 has since landed it on main.) 2. **It connected before session restore settled**, so the seeded tab never bound to a remote PTY and the terminal sat on "Connecting…" forever. 3. **It could never have passed, even once.** It waited for a one-shot `READY:` line through a 4000-char terminal window while its own 2 KB-every-8 ms flood buries that line within ~16 ms. Readiness is now keyed on the repeating `BG:` flood marker, which is strictly stronger — it proves the pane is streaming rather than merely started. It now runs end to end and prints a measurement instead of dying on a call site: ``` [freeze-repro R2] hiddenFloodMaxLagMs 2.1 bulkOpenMaxLagMs 41.5 interactionProbeMs 53.6 softFreeze false hardFreeze false ``` **It is still not CI-gateable, and the exclusion comment now says so.** The same spec on the same commit measured `bulkOpen 2575.6ms / interaction 3464.2ms` on a GitHub ubuntu runner against a 2500 ms soft budget — a ~60x spread on the number the budget reads, with the relay still streaming. That is the budget failing, not the product. The earlier draft of this comment claimed "repaired and passing", which was true only of the host it was measured on; gating this needs a host-relative oracle, not a bigger constant. ## New: a half-open link is judged, not wedged The fixture image has no `iptables` and the container has no `NET_ADMIN`, so `docker pause` is used instead — a harder case, because the container's TCP stack keeps ACKing: no FIN, no RST, and the socket looks perfectly healthy. Only an application-level probe can detect it. ``` [half-open] {"verdict":"reconnecting","verdictMs":25135,"budgetMs":90000} ``` Nothing in the suite covered the failure mode behind the "SSH hangs until I restart Orca" reports. ## New: resource accumulation measured on the remote host 6 terminals, then 5 reconnect cycles, counted on the container itself: ``` open: pts 1->6 (exactly 1/terminal), relay fds 25->30 (exactly 1/terminal) reconnect: pts flat at 6, relay procs flat at 1, node procs flat at 3 ``` `leakedMasterFdCount` is now **asserted**, not merely recorded. It counts PTY master fds held by non-relay processes: without `FD_CLOEXEC` a master is inherited by every later child, so terminal k adds k of them — the triangular signature measured as 15 across 5 terminals before the fix. #17914 patched the app and daemon and #17920 shipped the same patch to the relay host, and both are now on main, so the correct value is 0 and the probe holds it there: ``` baseline leakedMasterFdCount 0 6 terminals leakedMasterFdCount 0 (holders: only relay.js, n=6) reconnects leakedMasterFdCount 0 across all 5 cycles ``` Any growth here means the relay's node-pty rebuild did not take on that host, which is exactly what a remote-host probe exists to catch — and it is the half of #17914's claim that no unit test can reach. ## Routing Both new probes are claimed by `run-ssh-docker-e2e.mjs` (a Docker-gated spec no runner names self-skips everywhere and still reports green) **and** by the `ssh-terminal-source` route in `pr-e2e-source-routing.mjs`, so they run when the relay and SSH code they guard changes rather than only on a scheduled lane. --- config/scripts/pr-e2e-source-routing.mjs | 2 + config/scripts/run-ssh-docker-e2e.mjs | 36 +++- .../ssh-docker-bulk-open-freeze-repro.spec.ts | 60 +++++- tests/e2e/ssh-docker-half-open-link.spec.ts | 123 +++++++++++ .../ssh-docker-resource-accumulation.spec.ts | 204 ++++++++++++++++++ 5 files changed, 405 insertions(+), 20 deletions(-) create mode 100644 tests/e2e/ssh-docker-half-open-link.spec.ts create mode 100644 tests/e2e/ssh-docker-resource-accumulation.spec.ts diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs index d81f4c040fe..78814b663cb 100644 --- a/config/scripts/pr-e2e-source-routing.mjs +++ b/config/scripts/pr-e2e-source-routing.mjs @@ -26,7 +26,9 @@ export const PR_E2E_SOURCE_ROUTES = [ specs: [ 'tests/e2e/pty-input-write-queue-ssh.spec.ts', 'tests/e2e/ssh-cold-activation-restore.spec.ts', + 'tests/e2e/ssh-docker-half-open-link.spec.ts', 'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts', + 'tests/e2e/ssh-docker-resource-accumulation.spec.ts', 'tests/e2e/ssh-docker-transport-drop-recovery.spec.ts', 'tests/e2e/ssh-port-forward-lifecycle.spec.ts', 'tests/e2e/ssh-reconnect-tab-destruction.spec.ts', diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs index dddfa3e0148..9ab44b8457e 100644 --- a/config/scripts/run-ssh-docker-e2e.mjs +++ b/config/scripts/run-ssh-docker-e2e.mjs @@ -33,15 +33,31 @@ if (runtime.status !== 0) { // all. Recorded as a real gap, not as coverage living somewhere else. // ssh-codex-display-artifacts-repro.spec.ts — installs a real remote codex binary that CI // runners do not have (observed as `spawn codex ENOENT`). Runs in no CI lane at all. -// ssh-docker-bulk-open-freeze-repro.spec.ts — two reasons, both disqualifying: -// (a) it is a perf oracle, not a correctness one: SOFT_FREEZE_LAG_MS=2500 / -// HARD_FREEZE_LAG_MS=5000 measured by a renderer lag probe under a deliberate -// 5-pane output flood on a 420s budget. Same rule as ssh-docker-relay-perf above. -// (b) it is ROTTED: four call sites are out of date against terminal.ts's current -// helpers — execInTerminal gained a ptyId parameter and splitActiveTerminalPane -// gained a direction, so it cannot compile, let alone pass. Repairing it needs two -// semantic decisions (which ptyId to capture, which split direction) that change -// what the repro measures. Tracked in stablyai/orca#16764. +// ssh-docker-bulk-open-freeze-repro.spec.ts — un-rotted and now measurable, and marked +// `test.fixme` because its oracle cannot gate. Absent from this list AND skipped, so the +// two cannot drift: it is also reachable from the changed-specs lane whenever the spec +// itself is edited, and a wall-clock oracle that fails there is worth no more than one +// that fails here. +// The rot (#16764) is fixed: the stale call sites are repaired, it connects after session +// restore instead of before, and readiness keys on the repeating flood marker rather than +// a one-shot READY line the flood buries within ~16ms. It runs end to end and prints a +// measurement instead of dying on a call site. +// What it is NOT is portable. Three runs of the same measurement path: +// developer workstation: hiddenFlood 2.1ms bulkOpen 41.5ms interaction 53.6ms +// GitHub ubuntu runner A: hiddenFlood 1.5ms bulkOpen 2575.6ms interaction 3464.2ms +// GitHub ubuntu runner B: hiddenFlood 0.2ms bulkOpen 397.4ms interaction 3386.7ms +// bulkOpen swings 6.5x between two CI runs of the same code, so a fixed threshold on it is +// a coin flip; interaction sits stably ~64x over the workstation figure because it times a +// view remount, not the renderer freeze the issue reports, and only shares the budget +// constant because both are milliseconds. Every failure so far is the soft budget; hard +// has never tripped, and the relay was still streaming each time — the budget failed, not +// the product. Same rule as ssh-docker-relay-perf above. Gating needs a distribution +// first, then a host-relative oracle; a bigger constant, or a ratio picked from three +// samples, is the same arbitrary number in different clothes. +// COVERAGE GAP, recorded as such: 5 simultaneously flooding SSH panes exercise writer +// saturation, ACK/credit accounting and per-pane polling together, and nothing else covers +// that combination. Flip `test.fixme` back to `test` to run it. Tracked in +// stablyai/orca#16764. // // Why both projects: ssh-port-forward-lifecycle is @headful, which the headless project // grep-inverts away. @@ -71,8 +87,10 @@ const result = spawnSync( 'tests/e2e/ssh-ai-vault-session-history.spec.ts', 'tests/e2e/ssh-cold-activation-restore.spec.ts', 'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts', + 'tests/e2e/ssh-docker-half-open-link.spec.ts', 'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts', 'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts', + 'tests/e2e/ssh-docker-resource-accumulation.spec.ts', 'tests/e2e/ssh-docker-transport-drop-recovery.spec.ts', 'tests/e2e/ssh-external-image-preview.spec.ts', 'tests/e2e/ssh-lost-kill-tab-resurrection.spec.ts', diff --git a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts index a7754a02507..2de70c199d3 100644 --- a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts +++ b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts @@ -16,7 +16,7 @@ import { type DockerSshRelayTarget } from './helpers/docker-ssh-relay-target' import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' -import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { execInTerminal, focusLastTerminalPane, @@ -31,6 +31,7 @@ import { HARD_FREEZE_LAG_MS, SOFT_FREEZE_LAG_MS } from './helpers/remote-session const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' const REPORT_DIR = path.join(process.cwd(), 'test-results', 'freeze-repro') const SESSION_SPLITS = 5 +const FLOOD_READ_CHARS = 80_000 function shellQuote(value: string): string { return `'${value.replaceAll("'", "'\\''")}'` @@ -52,7 +53,32 @@ function continuousFloodCommand(runId: string, index: number): string { test.describe('R2 Docker SSH bulk-open freeze', () => { test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker SSH freeze repro') - test('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({ + // Fixme: un-rotted and measurable, but its oracle is wall-clock and does not survive a change of + // host, so it cannot gate. Three runs of the same measurement path: + // + // host hiddenFlood bulkOpen interaction + // developer workstation 2.1ms 41.5ms 53.6ms + // GitHub ubuntu runner A 1.5ms 2575.6ms 3464.2ms + // GitHub ubuntu runner B 0.2ms 397.4ms 3386.7ms + // + // Two separate problems, and neither is the product. `bulkOpenMaxLagMs` swings 6.5x between two + // CI runs of the same code, so a fixed threshold on it is a coin flip; `interactionProbeMs` sits + // stably ~64x over the workstation figure, because it times two `setActiveView` round trips + // through a double rAF — a view remount cost, not the renderer freeze #16764 reports. It shares + // SOFT/HARD_FREEZE_LAG_MS with the lag probe only because both are milliseconds. `hardFreeze` + // has never tripped on any host; the failure is always the soft budget. + // + // Not converted to a ratio against a calibration run: with a 6.5x within-host swing on the very + // quantity that would be normalized, a threshold picked from three samples is the same arbitrary + // constant in dimensionless clothing. Gating needs a distribution first. + // + // Kept executable rather than deleted: flip `test.fixme` back to `test` to run it, which is how + // the numbers above were taken. Tracked in stablyai/orca#16764. + // + // The cost is real and is recorded in run-ssh-docker-e2e.mjs: 5 simultaneously flooding SSH panes + // exercise writer saturation, ACK/credit accounting and per-pane polling together, and nothing + // else covers that combination. It is a gap, not coverage living somewhere else. + test.fixme('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({ orcaPage, registerPostElectronShutdownCleanup }, testInfo) => { @@ -66,24 +92,36 @@ test.describe('R2 Docker SSH bulk-open freeze', () => { } }) + // Why: session restore must settle before the remote worktree is added, or the + // seeded terminal tab races tab hydration and never binds to the remote PTY. + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) await connectDockerSshRelayTarget(orcaPage, target, { remotePath: DOCKER_SSH_RELAY_REMOTE_REPO_PATH }) - await waitForSessionReady(orcaPage) - await waitForActiveWorktree(orcaPage) const runId = `${Date.now()}` // First terminal on the SSH worktree. - await waitForActiveTerminalManager(orcaPage) - await execInTerminal(orcaPage, continuousFloodCommand(runId, 0)) - await waitForTerminalOutput(orcaPage, `READY:SSH_BULK_${runId}_0`, 60_000) + await ensureTerminalVisible(orcaPage, 45_000) + await waitForActiveTerminalManager(orcaPage, 60_000) + const firstPtyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, firstPtyId, continuousFloodCommand(runId, 0)) + // Why: the one-shot READY line is buried by the 2KB/8ms flood within ~16ms, so it is + // unobservable through the terminal read window. The repeating BG marker is the only + // stable readiness signal, and it also proves the pane is actually flooding. + await waitForTerminalOutput(orcaPage, `BG:SSH_BULK_${runId}_0:`, 60_000, FLOOD_READ_CHARS) for (let i = 1; i < SESSION_SPLITS; i += 1) { - await splitActiveTerminalPane(orcaPage) + await splitActiveTerminalPane(orcaPage, 'vertical') await focusLastTerminalPane(orcaPage) - await waitForActivePanePtyId(orcaPage, 30_000) - await execInTerminal(orcaPage, continuousFloodCommand(runId, i)) - await waitForTerminalOutput(orcaPage, `READY:SSH_BULK_${runId}_${i}`, 60_000) + const panePtyId = await waitForActivePanePtyId(orcaPage, 30_000) + await execInTerminal(orcaPage, panePtyId, continuousFloodCommand(runId, i)) + await waitForTerminalOutput( + orcaPage, + `BG:SSH_BULK_${runId}_${i}:`, + 60_000, + FLOOD_READ_CHARS + ) } // Leave the workspace view so panes go inactive while flooding. diff --git a/tests/e2e/ssh-docker-half-open-link.spec.ts b/tests/e2e/ssh-docker-half-open-link.spec.ts new file mode 100644 index 00000000000..c5b6f715dc9 --- /dev/null +++ b/tests/e2e/ssh-docker-half-open-link.spec.ts @@ -0,0 +1,123 @@ +/** + * Half-open SSH link probe. + * + * Freezes the remote host with `docker pause`. The container's TCP stack keeps + * ACKing, so the socket never sees a FIN or an RST — only the application stops + * answering. That is the wedge shape #17817 and #17838 are about: a link that + * looks perfectly healthy to TCP and can only be judged by an application probe. + * + * Requires: ORCA_E2E_SSH_DOCKER=1 and Docker available. + */ +import { execFileSync } from 'node:child_process' +import { expect, test } from './helpers/orca-app' +import { + cleanupDockerSshRelayTarget, + startDockerSshRelayTarget, + type DockerSshRelayTarget +} from './helpers/docker-ssh-relay-target' +import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { + execInTerminal, + waitForActivePanePtyId, + waitForActiveTerminalManager, + waitForTerminalOutput +} from './helpers/terminal' + +const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' +/** Generous: the point is that a verdict arrives at all, not its exact latency. */ +const LOST_VERDICT_BUDGET_MS = 90_000 + +function docker(args: string[]): void { + execFileSync('docker', args, { timeout: 30_000 }) +} + +async function readSshStatus( + page: Parameters[0], + targetId: string +): Promise { + return page.evaluate( + (id) => window.__store?.getState().sshConnectionStates.get(id)?.status ?? null, + targetId + ) +} + +test.describe('Docker SSH half-open link', () => { + test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker-backed SSH tests.') + test.skip(process.platform === 'win32', 'Uses docker pause against a Linux container.') + + test('declares a frozen host lost instead of wedging, and recovers @half-open', async ({ + orcaPage, + registerPostElectronShutdownCleanup + }, testInfo) => { + test.setTimeout(420_000) + let target: DockerSshRelayTarget | null = null + let paused = false + try { + target = startDockerSshRelayTarget(testInfo) + const captured = target + registerPostElectronShutdownCleanup(async () => { + cleanupDockerSshRelayTarget(captured) + }) + + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const remote = await connectDockerSshRelayTarget(orcaPage, target) + await ensureTerminalVisible(orcaPage, 45_000) + await waitForActiveTerminalManager(orcaPage, 60_000) + const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) + + const runId = String(Date.now()) + await execInTerminal(orcaPage, ptyId, `echo LIVE_${runId}`) + await waitForTerminalOutput(orcaPage, `LIVE_${runId}`, 60_000) + expect(await readSshStatus(orcaPage, remote.targetId)).toBe('connected') + + // Freeze the host: TCP keeps ACKing, the application stops answering. + docker(['pause', target.containerName]) + paused = true + const frozenAt = Date.now() + + let verdict: string | null = 'connected' + while (Date.now() - frozenAt < LOST_VERDICT_BUDGET_MS) { + verdict = await readSshStatus(orcaPage, remote.targetId) + if (verdict !== 'connected') { + break + } + await orcaPage.waitForTimeout(1_000) + } + const verdictMs = Date.now() - frozenAt + console.log( + `[half-open] ${JSON.stringify({ verdict, verdictMs, budgetMs: LOST_VERDICT_BUDGET_MS })}` + ) + + docker(['unpause', target.containerName]) + paused = false + + // Why this is the assertion: a wedged client sits on `connected` forever and + // never offers the user a reconnect. Any non-connected verdict is a pass. + expect( + verdict, + `client never left "connected" ${verdictMs}ms after the host was frozen` + ).not.toBe('connected') + + // The link must be usable again once the host thaws. + await expect + .poll(() => readSshStatus(orcaPage, remote.targetId), { timeout: 120_000 }) + .toBe('connected') + const recoveredPtyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, recoveredPtyId, `echo RECOVERED_${runId}`) + await waitForTerminalOutput(orcaPage, `RECOVERED_${runId}`, 90_000) + } finally { + if (target && paused) { + try { + docker(['unpause', target.containerName]) + } catch { + // The container may already be gone; cleanup below is authoritative. + } + } + if (target) { + cleanupDockerSshRelayTarget(target) + } + } + }) +}) diff --git a/tests/e2e/ssh-docker-resource-accumulation.spec.ts b/tests/e2e/ssh-docker-resource-accumulation.spec.ts new file mode 100644 index 00000000000..f028ef0d2ab --- /dev/null +++ b/tests/e2e/ssh-docker-resource-accumulation.spec.ts @@ -0,0 +1,204 @@ +/** + * Adversarial resource-accumulation probe for the SSH relay. + * + * Covers claims no unit test can reach, measured on the remote host itself: + * - #17914 / #17920: PTY master fds are close-on-exec, so /dev/pts and the + * relay's fd table must not grow per terminal beyond the terminals + * themselves. #17914 patches the app and terminal daemon; the relay installs + * node-pty from npm on the remote host, so #17920 ships the same patch as a + * relay asset and rebuilds there. Only a remote host can judge that second + * half, which is why leakedMasterFdCount is measured on the container. + * - #17817/#17821/#17831: repeated disconnect/reconnect must not accumulate + * relay processes, orphan PTYs, or fds. + * + * Requires: ORCA_E2E_SSH_DOCKER=1 and Docker available. + */ +import { expect, test } from './helpers/orca-app' +import { + cleanupDockerSshRelayTarget, + execDockerSshRelayTargetCommand, + startDockerSshRelayTarget, + type DockerSshRelayTarget +} from './helpers/docker-ssh-relay-target' +import { + connectDockerSshRelayTarget, + reconnectDockerSshRelayTarget +} from './helpers/docker-ssh-relay-connection' +import { readDockerSshRelayProcessSnapshots } from './helpers/docker-ssh-relay-processes' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { + execInTerminal, + focusLastTerminalPane, + splitActiveTerminalPane, + waitForActivePanePtyId, + waitForActiveTerminalManager, + waitForTerminalOutput +} from './helpers/terminal' + +const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' +const TERMINAL_COUNT = 6 +const RECONNECT_CYCLES = 5 + +type RemoteResourceSample = { + ptsCount: number + relayFdCount: number + relayProcessCount: number + nodeProcessCount: number + /** + * PTY master fds held by processes other than the relay. A master fd without + * FD_CLOEXEC is inherited by every later-spawned child, so this grows ~N^2/2 + * across N terminals when the close-on-exec fix is absent (#17914). + */ + leakedMasterFdCount: number +} + +const COUNT_LEAKED_MASTER_FDS = [ + 'total=0', + 'for p in $(ls /proc | grep -E "^[0-9]+$"); do', + ' cmd=$(tr "\\0" " " < /proc/$p/cmdline 2>/dev/null || true)', + ' case "$cmd" in *relay.js*) continue;; esac', + ' n=$(ls -l /proc/$p/fd 2>/dev/null | grep -c "ptmx" || true)', + ' total=$((total+n))', + 'done', + 'echo $total' +].join('\n') + +const DESCRIBE_MASTER_FD_HOLDERS = [ + 'for p in $(ls /proc | grep -E "^[0-9]+$"); do', + ' cmd=$(tr "\\0" " " < /proc/$p/cmdline 2>/dev/null || true)', + ' n=$(ls -l /proc/$p/fd 2>/dev/null | grep -c "ptmx" || true)', + ' if [ "$n" != "0" ]; then echo "$p n=$n cmd=$cmd"; fi', + 'done' +].join('\n') + +function sampleRemoteResources(target: DockerSshRelayTarget): RemoteResourceSample { + const groups = readDockerSshRelayProcessSnapshots(target) + // Why: fd growth is only meaningful against the relay that owns the PTYs, so read + // the table of every relay group and sum, rather than assuming a single relay. + const relayFdCount = groups.reduce((total, group) => { + const raw = execDockerSshRelayTargetCommand( + target, + `ls /proc/${group.relayPid}/fd 2>/dev/null | wc -l` + ) + return total + Number(raw.trim() || '0') + }, 0) + const ptsCount = Number( + execDockerSshRelayTargetCommand(target, 'ls /dev/pts | grep -c "^[0-9]" || true').trim() || '0' + ) + const nodeProcessCount = Number( + execDockerSshRelayTargetCommand(target, 'pgrep -c node || true').trim() || '0' + ) + const leakedMasterFdCount = Number( + execDockerSshRelayTargetCommand(target, COUNT_LEAKED_MASTER_FDS).trim() || '0' + ) + return { + ptsCount, + relayFdCount, + relayProcessCount: groups.length, + nodeProcessCount, + leakedMasterFdCount + } +} + +test.describe('Docker SSH relay resource accumulation', () => { + test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker-backed SSH tests.') + test.skip(process.platform === 'win32', 'Uses POSIX /proc and /dev/pts probes.') + + test('does not accumulate pts devices, relay fds, or relay processes @resource-accumulation', async ({ + orcaPage, + registerPostElectronShutdownCleanup + }, testInfo) => { + test.setTimeout(420_000) + let target: DockerSshRelayTarget | null = null + try { + target = startDockerSshRelayTarget(testInfo) + const captured = target + registerPostElectronShutdownCleanup(async () => { + cleanupDockerSshRelayTarget(captured) + }) + + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const remote = await connectDockerSshRelayTarget(orcaPage, target) + await ensureTerminalVisible(orcaPage, 45_000) + await waitForActiveTerminalManager(orcaPage, 60_000) + + const runId = String(Date.now()) + const firstPtyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, firstPtyId, `echo PANE_READY_${runId}_0`) + await waitForTerminalOutput(orcaPage, `PANE_READY_${runId}_0`, 60_000) + + const baseline = sampleRemoteResources(target) + const samples: RemoteResourceSample[] = [] + + // Open N more terminals; each must cost a bounded, roughly constant amount. + for (let index = 1; index < TERMINAL_COUNT; index += 1) { + await splitActiveTerminalPane(orcaPage, 'vertical') + await focusLastTerminalPane(orcaPage) + const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, ptyId, `echo PANE_READY_${runId}_${index}`) + await waitForTerminalOutput(orcaPage, `PANE_READY_${runId}_${index}`, 60_000) + samples.push(sampleRemoteResources(target)) + } + + const withTerminals = samples.at(-1)! + const openedTerminals = TERMINAL_COUNT - 1 + const ptsGrowth = withTerminals.ptsCount - baseline.ptsCount + const fdGrowth = withTerminals.relayFdCount - baseline.relayFdCount + const fdPerTerminal = fdGrowth / openedTerminals + + console.log( + `[resource-accumulation] open ${JSON.stringify({ + baseline, + withTerminals, + openedTerminals, + ptsGrowth, + fdGrowth, + fdPerTerminal + })}` + ) + + console.log( + `[resource-accumulation] master-fd holders\n${execDockerSshRelayTargetCommand( + target, + DESCRIBE_MASTER_FD_HOLDERS + )}` + ) + + // Each remote terminal legitimately costs one pts device. + expect(ptsGrowth).toBeLessThanOrEqual(openedTerminals) + // Why: a master fd that leaks into every child would push this well past a + // small constant per terminal. Allow slack for the relay's own bookkeeping. + expect(fdPerTerminal).toBeLessThanOrEqual(4) + // Why an equality-shaped bound rather than slack: a master fd that is not close-on-exec is + // inherited by every later child, so terminal k adds k of them (1+2+3+4+5 = 15 was the + // observed pre-fix signature). With #17914's patch reaching the relay host through #17920 + // no non-relay process holds a master at all, so any growth here means the relay's node-pty + // rebuild did not take on this host — which is exactly what this probe exists to catch. + expect(withTerminals.leakedMasterFdCount).toBeLessThanOrEqual(baseline.leakedMasterFdCount) + expect(withTerminals.relayProcessCount).toBe(1) + + // Repeated reconnects must not accumulate anything on the host. + const reconnectSamples: RemoteResourceSample[] = [] + for (let cycle = 0; cycle < RECONNECT_CYCLES; cycle += 1) { + await reconnectDockerSshRelayTarget(orcaPage, remote.targetId) + reconnectSamples.push(sampleRemoteResources(target)) + } + console.log(`[resource-accumulation] reconnects ${JSON.stringify(reconnectSamples)}`) + + const first = reconnectSamples[0] + const last = reconnectSamples.at(-1)! + expect(last.relayProcessCount).toBe(1) + // Why: the interesting failure is monotonic growth across cycles, not the + // absolute count, so compare the last cycle against the first. + expect(last.ptsCount).toBeLessThanOrEqual(first.ptsCount) + expect(last.relayFdCount).toBeLessThanOrEqual(first.relayFdCount + 4) + expect(last.nodeProcessCount).toBeLessThanOrEqual(first.nodeProcessCount) + expect(last.leakedMasterFdCount).toBeLessThanOrEqual(first.leakedMasterFdCount) + } finally { + if (target) { + cleanupDockerSshRelayTarget(target) + } + } + }) +}) From b8b7a6be9d4089b8c101843f4430df5d6dbe5497 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:04:11 -0700 Subject: [PATCH 15/77] fix(activity): persist the agents unread filter and grouping (#18255) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(activity): persist the agents unread filter and grouping The Agents view's "Show unread threads only" toggle and Group-by select were plain component state in the sidebar and the Activity page, so both reset on every mount — including app restart — while their neighbours in the same toolbar (compact rows, show child agents) survived via the persisted UI store. Promote both to `agentsReadFilter` / `agentsGroupBy` persisted UI preferences, wired through the same seams as `agentsCompactMode`: shared type, default, strict client RPC schema, pairing-local field census, web read pin, store contract/actions, and hydration normalizers that reject unknown values. Both consumers now read the store, so the sidebar and the Activity page share one filter the way they already share compact mode. * refactor: centralize thread filter value domains Establish filter and groupby value domains as the single source of truth, with types derived from them to prevent drift between valid values and their normalizers. Extract common validation logic into a shared isMember helper to keep the two normalization functions in sync. * refactor: centralize thread filter value domains Consolidate filter value definitions in agents-view-thread-filters and use them in Zod schema validation to ensure consistent, persistent serialization of filter state. --- .../client-ui-pairing-local-fields.test.ts | 2 ++ .../runtime/rpc/methods/client-ui-schemas.ts | 6 +++++ .../activity/ActivityPrototypePage.tsx | 12 ++++------ .../activity/activity-thread-types.ts | 4 ++-- .../components/sidebar/SidebarAgentsList.tsx | 2 +- src/renderer/src/components/sidebar/index.tsx | 7 +++--- ...ui-hydration-workspace-preferences.test.ts | 21 +++++++++++++++++ .../ui/ui-slice-contract-preferences.ts | 6 +++++ .../slices/ui/ui-slice-hydration-actions.ts | 6 +++++ .../slices/ui/ui-slice-preference-actions.ts | 14 +++++++++++ .../web-preference-normalization.ts | 2 ++ .../src/web/web-preload-api-ui.test.ts | 4 ++++ src/shared/agents-view-thread-filters.ts | 23 +++++++++++++++++++ src/shared/constants.ts | 3 +++ src/shared/pairing-local-ui-fields.test.ts | 2 ++ src/shared/pairing-local-ui-fields.ts | 2 ++ src/shared/persisted-ui-state-types.ts | 6 +++++ src/shared/ui-chrome-types.ts | 4 ++++ 18 files changed, 113 insertions(+), 13 deletions(-) create mode 100644 src/shared/agents-view-thread-filters.ts diff --git a/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts b/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts index d6c08ad44cf..2e7ac3e93b1 100644 --- a/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts +++ b/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts @@ -49,6 +49,8 @@ describe('client UI RPC pairing-local field seams', () => { agentsFilterRepoIds: ['repo-a'], agentsShowChildAgents: true, agentsCompactMode: false, + agentsReadFilter: 'unread', + agentsGroupBy: 'project', activityClearedAtByPaneKey: { 'tab-1:leaf-1': 123 }, manuallyUnreadTurnsByPaneKey: { 'tab-1:leaf-1': 321 } } diff --git a/src/main/runtime/rpc/methods/client-ui-schemas.ts b/src/main/runtime/rpc/methods/client-ui-schemas.ts index edb3f651c32..c772ba2fe67 100644 --- a/src/main/runtime/rpc/methods/client-ui-schemas.ts +++ b/src/main/runtime/rpc/methods/client-ui-schemas.ts @@ -3,6 +3,10 @@ import { isFeatureInteractionId, type FeatureInteractionId } from '../../../../shared/feature-interactions' +import { + ACTIVITY_GROUP_BY_VALUES, + THREAD_READ_FILTER_VALUES +} from '../../../../shared/agents-view-thread-filters' import { isFeatureTipId } from '../../../../shared/feature-tips' import { isReleaseChannel, type ReleaseChannel } from '../../../../shared/release-channel' import { @@ -126,6 +130,8 @@ const UiUpdateFields = z agentsFilterRepoIds: StringArray.optional(), agentsShowChildAgents: z.boolean().optional(), agentsCompactMode: z.boolean().optional(), + agentsReadFilter: z.enum(THREAD_READ_FILTER_VALUES).optional(), + agentsGroupBy: z.enum(ACTIVITY_GROUP_BY_VALUES).optional(), workspaceHostOrder: z.array(z.string()).optional(), automationHostFilter: z .union([ diff --git a/src/renderer/src/components/activity/ActivityPrototypePage.tsx b/src/renderer/src/components/activity/ActivityPrototypePage.tsx index 83581d4cf10..44c1f398122 100644 --- a/src/renderer/src/components/activity/ActivityPrototypePage.tsx +++ b/src/renderer/src/components/activity/ActivityPrototypePage.tsx @@ -21,22 +21,20 @@ import { useActivityTerminalLoadingLabel, useActivityTerminalPortalStatus } from './activity-terminal-portal-status' -import type { - ActivityGroupBy, - ActivityTerminalPortalSlotId, - ThreadReadFilter -} from './activity-thread-types' +import type { ActivityTerminalPortalSlotId } from './activity-thread-types' export * from './activity-prototype-page-exports' export default function ActivityPrototypePage(): React.JSX.Element { - const [readFilter, setReadFilter] = useState('all') - const [groupBy, setGroupBy] = useState('status') const [query, setQuery] = useState('') const activityFilterInputRef = useRef(null) // Why: bounds auto mark-read to one acknowledgement per selected thread turn. const autoAcknowledgedTurnRef = useRef(null) // Why store-backed: persisted preferences shared with the sidebar agents list. + const readFilter = useAppStore((s) => s.agentsReadFilter) + const setReadFilter = useAppStore((s) => s.setAgentsReadFilter) + const groupBy = useAppStore((s) => s.agentsGroupBy) + const setGroupBy = useAppStore((s) => s.setAgentsGroupBy) const compactMode = useAppStore((s) => s.agentsCompactMode) const setCompactMode = useAppStore((s) => s.setAgentsCompactMode) const showChildAgents = useAppStore((s) => s.agentsShowChildAgents) diff --git a/src/renderer/src/components/activity/activity-thread-types.ts b/src/renderer/src/components/activity/activity-thread-types.ts index 7225a71b935..ed73642c8eb 100644 --- a/src/renderer/src/components/activity/activity-thread-types.ts +++ b/src/renderer/src/components/activity/activity-thread-types.ts @@ -9,8 +9,8 @@ import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { Worktree } from '../../../../shared/worktree/types' import type { ActivityPortalReadinessStatus } from './activity-portal-readiness-oscillation' -export type ThreadReadFilter = 'all' | 'unread' -export type ActivityGroupBy = 'none' | 'status' | 'project' | 'worktree' | 'agent' +export type { ActivityGroupBy, ThreadReadFilter } from '../../../../shared/ui-chrome-types' + export type ActivityEventState = Extract export type ActivityHookLiveAgentState = Extract< AgentStatusState, diff --git a/src/renderer/src/components/sidebar/SidebarAgentsList.tsx b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx index 5d82b56253a..ca5a8c5bce1 100644 --- a/src/renderer/src/components/sidebar/SidebarAgentsList.tsx +++ b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx @@ -41,7 +41,7 @@ export default function SidebarAgentsList({ }: SidebarAgentsListProps): React.JSX.Element { // The search row is owned here and mounts conditionally, so subscribe this host to locale changes. useTranslation() - // Why store-backed: these are persisted preferences (agents* UI fields), unlike the momentary read filter/search. + // Why store-backed: these are persisted preferences (agents* UI fields), unlike the momentary search. const compactMode = useAppStore((s) => s.agentsCompactMode) const setCompactMode = useAppStore((s) => s.setAgentsCompactMode) const showChildAgents = useAppStore((s) => s.agentsShowChildAgents) diff --git a/src/renderer/src/components/sidebar/index.tsx b/src/renderer/src/components/sidebar/index.tsx index 90c31999bd3..8b837b5ef39 100644 --- a/src/renderer/src/components/sidebar/index.tsx +++ b/src/renderer/src/components/sidebar/index.tsx @@ -11,7 +11,6 @@ import WorkspaceKanbanDrawer from './WorkspaceKanbanDrawer' import type { VirtualizedScrollAnchor } from '@/hooks/useVirtualizedScrollAnchor' import { cn } from '@/lib/utils' import { FolderPlus, Loader2 } from 'lucide-react' -import type { ActivityGroupBy, ThreadReadFilter } from '@/components/activity/activity-thread-types' import { ActivityThreadCollapseContext } from '@/components/activity/activity-thread-collapse-context' import { useSidebarProjectDrop } from './useSidebarProjectDrop' import { useWorkspaceBoardPanel } from './useWorkspaceBoardPanel' @@ -59,8 +58,10 @@ function Sidebar({ const showAgentDashboard = settings?.experimentalAgentDashboardPopout === true const agentDashboardDrawerOpen = useAppStore((s) => s.agentDashboardDrawerOpen) const setAgentDashboardDrawerOpen = useAppStore((s) => s.setAgentDashboardDrawerOpen) - const [agentReadFilter, setAgentReadFilter] = React.useState('all') - const [agentGroupBy, setAgentGroupBy] = React.useState('status') + const agentReadFilter = useAppStore((s) => s.agentsReadFilter) + const setAgentReadFilter = useAppStore((s) => s.setAgentsReadFilter) + const agentGroupBy = useAppStore((s) => s.agentsGroupBy) + const setAgentGroupBy = useAppStore((s) => s.setAgentsGroupBy) const [agentQuery, setAgentQuery] = React.useState('') const [agentOptionsTarget, setAgentOptionsTarget] = React.useState(null) const agentsScrollTopRef = React.useRef(0) diff --git a/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts b/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts index 642297bb109..6d3a559c582 100644 --- a/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts +++ b/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts @@ -551,6 +551,27 @@ describe('createUISlice hydratePersistedUI', () => { expect(store.getState().agentsFilterRepoIds).toEqual([]) expect(store.getState().agentsShowChildAgents).toBe(false) expect(store.getState().agentsCompactMode).toBe(true) + expect(store.getState().agentsReadFilter).toBe('all') + expect(store.getState().agentsGroupBy).toBe('status') + }) + + it('restores the persisted agents read filter and grouping, rejecting unknown values', () => { + const store = createUIStore() + + store + .getState() + .hydratePersistedUI(makePersistedUI({ agentsReadFilter: 'unread', agentsGroupBy: 'project' })) + expect(store.getState().agentsReadFilter).toBe('unread') + expect(store.getState().agentsGroupBy).toBe('project') + + store.getState().hydratePersistedUI( + makePersistedUI({ + agentsReadFilter: 'bogus' as unknown as PersistedUIState['agentsReadFilter'], + agentsGroupBy: 'bogus' as unknown as PersistedUIState['agentsGroupBy'] + }) + ) + expect(store.getState().agentsReadFilter).toBe('all') + expect(store.getState().agentsGroupBy).toBe('status') }) it('sanitizes malformed agents repo filters before the repo catalog loads', () => { diff --git a/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts b/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts index 2e764e5cd79..1162f399cd5 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts @@ -1,9 +1,11 @@ import type { PersistedUIState } from '../../../../../shared/persisted-ui-state-types' import type { + ActivityGroupBy, AgentActivityDisplayMode, ManualRepoOrderEntry, ProjectOrderBy, StatusBarItem, + ThreadReadFilter, WorktreeCardMode, WorktreeCardProperty, WorkspaceHostOrder, @@ -71,6 +73,10 @@ export type UISlicePreferences = { setAgentsShowChildAgents: (v: boolean) => void agentsCompactMode: boolean setAgentsCompactMode: (v: boolean) => void + agentsReadFilter: ThreadReadFilter + setAgentsReadFilter: (v: ThreadReadFilter) => void + agentsGroupBy: ActivityGroupBy + setAgentsGroupBy: (v: ActivityGroupBy) => void collapsedGroups: Set toggleCollapsedGroup: (key: string) => void worktreeCardProperties: WorktreeCardProperty[] diff --git a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts index f44774b3770..08adff8e2a3 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts @@ -20,6 +20,10 @@ import { normalizeWorktreeCardProperties, normalizeAgentActivityDisplayMode } from '../../../../../shared/constants' +import { + normalizeActivityGroupBy, + normalizeThreadReadFilter +} from '../../../../../shared/agents-view-thread-filters' import { clampWorkspaceBoardColumnWidth, clampWorkspaceBoardOpacity, @@ -182,6 +186,8 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par ), agentsShowChildAgents: ui.agentsShowChildAgents === true, agentsCompactMode: ui.agentsCompactMode !== false, + agentsReadFilter: normalizeThreadReadFilter(ui.agentsReadFilter), + agentsGroupBy: normalizeActivityGroupBy(ui.agentsGroupBy), collapsedGroups: new Set(ui.collapsedGroups ?? []), uiZoomLevel: ui.uiZoomLevel ?? 0, editorFontZoomLevel: ui.editorFontZoomLevel ?? 0, diff --git a/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts index 971c64dfa52..9a9dc5947b1 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts @@ -1,4 +1,8 @@ import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' +import { + DEFAULT_AGENTS_GROUP_BY, + DEFAULT_AGENTS_READ_FILTER +} from '../../../../../shared/agents-view-thread-filters' import { DEFAULT_AGENT_ACTIVITY_DISPLAY_MODE, DEFAULT_SHOW_SLEEPING_WORKSPACES, @@ -170,6 +174,16 @@ export function createUiPreferenceActions(set: UISliceSet, get: UISliceGet): Par set({ agentsCompactMode: v }) window.api.ui.set({ agentsCompactMode: v }).catch(console.error) }, + agentsReadFilter: DEFAULT_AGENTS_READ_FILTER, + setAgentsReadFilter: (v) => { + set({ agentsReadFilter: v }) + window.api.ui.set({ agentsReadFilter: v }).catch(console.error) + }, + agentsGroupBy: DEFAULT_AGENTS_GROUP_BY, + setAgentsGroupBy: (v) => { + set({ agentsGroupBy: v }) + window.api.ui.set({ agentsGroupBy: v }).catch(console.error) + }, collapsedGroups: new Set(), toggleCollapsedGroup: (key) => diff --git a/src/renderer/src/web/preload-api/web-preference-normalization.ts b/src/renderer/src/web/preload-api/web-preference-normalization.ts index 9ae3ad00d4a..be7e02f029a 100644 --- a/src/renderer/src/web/preload-api/web-preference-normalization.ts +++ b/src/renderer/src/web/preload-api/web-preference-normalization.ts @@ -73,6 +73,8 @@ export function mergeHostWebUIState( agentsFilterRepoIds: local.agentsFilterRepoIds, agentsShowChildAgents: local.agentsShowChildAgents, agentsCompactMode: local.agentsCompactMode, + agentsReadFilter: local.agentsReadFilter, + agentsGroupBy: local.agentsGroupBy, activityClearedAtByPaneKey: local.activityClearedAtByPaneKey, manuallyUnreadTurnsByPaneKey: local.manuallyUnreadTurnsByPaneKey } satisfies Record & Partial diff --git a/src/renderer/src/web/web-preload-api-ui.test.ts b/src/renderer/src/web/web-preload-api-ui.test.ts index 94e9ba8e1eb..a01ffbad567 100644 --- a/src/renderer/src/web/web-preload-api-ui.test.ts +++ b/src/renderer/src/web/web-preload-api-ui.test.ts @@ -469,6 +469,8 @@ describe('web UI preload API', () => { agentsFilterRepoIds: ['repo-b'], agentsShowChildAgents: true, agentsCompactMode: false, + agentsReadFilter: 'unread', + agentsGroupBy: 'project', activityClearedAtByPaneKey: { 'tab-1:leaf-1': 123 }, manuallyUnreadTurnsByPaneKey: { 'tab-1:leaf-1': 321 } } @@ -481,6 +483,8 @@ describe('web UI preload API', () => { agentsFilterRepoIds: ['repo-a'], agentsShowChildAgents: false, agentsCompactMode: true, + agentsReadFilter: 'all', + agentsGroupBy: 'status', activityClearedAtByPaneKey: { 'tab-2:leaf-2': 456 }, manuallyUnreadTurnsByPaneKey: { 'tab-2:leaf-2': 654 } } diff --git a/src/shared/agents-view-thread-filters.ts b/src/shared/agents-view-thread-filters.ts new file mode 100644 index 00000000000..7f77015078b --- /dev/null +++ b/src/shared/agents-view-thread-filters.ts @@ -0,0 +1,23 @@ +/** The two filter value domains, in menu order. The types below, the client + * schema's `z.enum`s and the normalizers all derive from these, so a new value + * cannot drift out of any of them. */ +export const THREAD_READ_FILTER_VALUES = ['all', 'unread'] as const +export const ACTIVITY_GROUP_BY_VALUES = ['none', 'status', 'project', 'worktree', 'agent'] as const + +export type ThreadReadFilter = (typeof THREAD_READ_FILTER_VALUES)[number] +export type ActivityGroupBy = (typeof ACTIVITY_GROUP_BY_VALUES)[number] + +export const DEFAULT_AGENTS_READ_FILTER: ThreadReadFilter = 'all' +export const DEFAULT_AGENTS_GROUP_BY: ActivityGroupBy = 'status' + +export function normalizeThreadReadFilter(value: unknown): ThreadReadFilter { + return isMember(THREAD_READ_FILTER_VALUES, value) ? value : DEFAULT_AGENTS_READ_FILTER +} + +export function normalizeActivityGroupBy(value: unknown): ActivityGroupBy { + return isMember(ACTIVITY_GROUP_BY_VALUES, value) ? value : DEFAULT_AGENTS_GROUP_BY +} + +function isMember(catalog: readonly T[], value: unknown): value is T { + return typeof value === 'string' && (catalog as readonly string[]).includes(value) +} diff --git a/src/shared/constants.ts b/src/shared/constants.ts index 9d26b140dec..bb2f5940f0e 100644 --- a/src/shared/constants.ts +++ b/src/shared/constants.ts @@ -11,6 +11,7 @@ import { DEFAULT_STATUS_BAR_ITEMS } from './status-bar-defaults' import type { VoiceSettings } from './speech-types' import { cloneDefaultWorkspaceStatuses } from './workspace-statuses' import { DEFAULT_WORKTREE_CARD_PROPERTIES } from './worktree/card-properties' +import { DEFAULT_AGENTS_GROUP_BY, DEFAULT_AGENTS_READ_FILTER } from './agents-view-thread-filters' import { DEFAULT_USAGE_PERCENTAGE_DISPLAY } from './usage-percentage-display' import { DEFAULT_STATUS_BAR_USAGE_MODE } from './status-bar-usage-mode' import { buildDefaultSettings } from './default-global-settings' @@ -273,6 +274,8 @@ export function getDefaultUIState(): PersistedUIState { agentsFilterRepoIds: [], agentsShowChildAgents: false, agentsCompactMode: true, + agentsReadFilter: DEFAULT_AGENTS_READ_FILTER, + agentsGroupBy: DEFAULT_AGENTS_GROUP_BY, collapsedGroups: [], uiZoomLevel: 0, editorFontZoomLevel: 0, diff --git a/src/shared/pairing-local-ui-fields.test.ts b/src/shared/pairing-local-ui-fields.test.ts index ffc7baafe07..35bd5f33331 100644 --- a/src/shared/pairing-local-ui-fields.test.ts +++ b/src/shared/pairing-local-ui-fields.test.ts @@ -14,6 +14,8 @@ describe('pairing-local UI fields', () => { 'agentsFilterRepoIds', 'agentsShowChildAgents', 'agentsCompactMode', + 'agentsReadFilter', + 'agentsGroupBy', 'activityClearedAtByPaneKey', 'manuallyUnreadTurnsByPaneKey' ]) diff --git a/src/shared/pairing-local-ui-fields.ts b/src/shared/pairing-local-ui-fields.ts index d925c3094d8..f642bb2b821 100644 --- a/src/shared/pairing-local-ui-fields.ts +++ b/src/shared/pairing-local-ui-fields.ts @@ -17,6 +17,8 @@ export const PAIRING_LOCAL_UI_FIELDS = [ 'agentsFilterRepoIds', 'agentsShowChildAgents', 'agentsCompactMode', + 'agentsReadFilter', + 'agentsGroupBy', 'activityClearedAtByPaneKey', 'manuallyUnreadTurnsByPaneKey' ] as const satisfies readonly (keyof PersistedUIState)[] diff --git a/src/shared/persisted-ui-state-types.ts b/src/shared/persisted-ui-state-types.ts index 14e97bc34fc..b6d40480017 100644 --- a/src/shared/persisted-ui-state-types.ts +++ b/src/shared/persisted-ui-state-types.ts @@ -8,6 +8,7 @@ import type { StatusBarUsageMode } from './status-bar-usage-mode' import type { PersistedTrustedOrcaHooks } from './orca-yaml-hook-types' import type { CustomPet } from './pet-types' import type { + ActivityGroupBy, AgentActivityDisplayMode, ManualRepoOrderEntry, ProjectOrderBy, @@ -15,6 +16,7 @@ import type { RightSidebarTab, StatusBarItem, TaskResumeState, + ThreadReadFilter, TopLevelView, VisibleWorkspaceHostIds, WorkspaceHostOrder, @@ -81,6 +83,10 @@ export type PersistedUIState = { agentsShowChildAgents?: boolean /** Agents-view compact thread rows. Absent means on. */ agentsCompactMode?: boolean + /** Agents-view unread-only thread filter. Absent means 'all'. */ + agentsReadFilter?: ThreadReadFilter + /** Agents-view thread grouping. Absent means 'status'. */ + agentsGroupBy?: ActivityGroupBy collapsedGroups: string[] uiZoomLevel: number editorFontZoomLevel: number diff --git a/src/shared/ui-chrome-types.ts b/src/shared/ui-chrome-types.ts index 90a12cf5b02..fe6157d870b 100644 --- a/src/shared/ui-chrome-types.ts +++ b/src/shared/ui-chrome-types.ts @@ -49,6 +49,10 @@ export type WorktreeCardMode = 'Default' | 'Compact' export type AgentActivityDisplayMode = 'compact' | 'full' +// Re-exported so existing importers keep one home for UI chrome types; the +// value domain lives with the normalizers that police it. +export type { ActivityGroupBy, ThreadReadFilter } from './agents-view-thread-filters' + export type StatusBarItem = | 'claude' | 'codex' From 4b2d9b5aac16c87304068a4adfa5933d005ffbac Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:34:44 -0700 Subject: [PATCH 16/77] fix(path): stop seeded user bin dirs from outranking the inherited PATH (#18265) `patchPackagedProcessPath` prepends every seeded directory, so `~/bin` and `~/.local/bin` land ahead of the PATH a GUI-launched Electron inherited. That does more than make a tool findable, which is what seeding is for -- it re-ranks binaries the user already has, and those two directories are user-writable and can hold a wrapper for any system tool. On the #18234 reporter's box `~/.local/bin/gh` wraps `mise x gh -- gh`. Seeded ahead of /usr/bin we ran the wrapper where their own shell ran the real binary, and the wrapper's inner bare `gh` resolved back to itself. Measured in an Ubuntu 24.04 container: with their shell's ordering the chain exits in 22ms; with ours it never terminates and creates ~1,500 processes/second. Seed order now follows the rule the WSL twin already documents in posix-version-manager-bin-dirs.ts -- append, never prepend, because a login PATH that did resolve is authoritative. Version-manager shim dirs keep leading, since an nvm/mise/asdf user's runtime must still beat a system install; the generic user bin dirs move behind the inherited PATH. `getVersionManagerBinPaths` carries `~/bin` and `~/.local/bin` too (bun and pnpm install there), so they are filtered out of the leading group by name rather than by which list produced them. --- src/main/startup/configure-process.test.ts | 60 ++++++++++++++++++++++ src/main/startup/configure-process.ts | 47 ++++++++++++----- 2 files changed, 95 insertions(+), 12 deletions(-) diff --git a/src/main/startup/configure-process.test.ts b/src/main/startup/configure-process.test.ts index 4747f797cf9..ef7ec6a9b69 100644 --- a/src/main/startup/configure-process.test.ts +++ b/src/main/startup/configure-process.test.ts @@ -137,6 +137,66 @@ describe('patchPackagedProcessPath', () => { expect(segments).toContain('/usr/local/bin') }) + // Why this ordering is load-bearing (#18234): a seed exists so a GUI-launched + // Electron can *find* a tool, not to re-rank tools the user already has. + // `~/.local/bin` is user-writable and can hold a wrapper for any system tool. + // The reporter's `~/.local/bin/gh` wrapped `mise x gh -- gh`; seeded ahead of + // /usr/bin it ran instead of the real gh, and the wrapper's inner bare `gh` + // resolved back to itself. Measured in a container: with the login shell's + // ordering that chain exits in 22ms, with the seeded ordering it never + // terminates and creates ~1,300 processes/second. + it('never lets a seeded user dir overtake a system dir already on PATH', async () => { + const { app } = await import('electron') + const { patchPackagedProcessPath } = await import('./configure-process') + + setPlatform('linux') + Object.defineProperty(app, 'isPackaged', { configurable: true, value: true }) + process.env.HOME = '/home/tester' + process.env.PATH = '/usr/local/bin:/usr/bin:/bin' + + patchPackagedProcessPath() + + const segments = (process.env.PATH ?? '').split(':') + const localBin = segments.indexOf(join('/home/tester', '.local/bin')) + // Still reachable — that is what the seeding is for (#829). + expect(localBin).toBeGreaterThan(-1) + for (const systemDir of ['/usr/bin', '/bin', '/usr/local/bin']) { + expect(segments.indexOf(systemDir)).toBeLessThan(localBin) + } + expect(segments.indexOf(join('/home/tester', 'bin'))).toBeGreaterThan( + segments.indexOf('/usr/bin') + ) + }) + + it('keeps version-manager shims ahead of the inherited PATH', async () => { + const { app } = await import('electron') + const { patchPackagedProcessPath } = await import('./configure-process') + const { getVersionManagerBinPaths } = await import('../../shared/node-cli-command-resolution') + + setPlatform('linux') + Object.defineProperty(app, 'isPackaged', { configurable: true, value: true }) + process.env.HOME = '/home/tester' + process.env.PATH = '/usr/bin:/bin' + + patchPackagedProcessPath() + + const segments = (process.env.PATH ?? '').split(':') + const genericUserBinDirs = [join('/home/tester', 'bin'), join('/home/tester', '.local/bin')] + const seeded = getVersionManagerBinPaths({ platform: 'linux', homePath: '/home/tester' }) + const shimDirs = seeded.filter((dir) => !genericUserBinDirs.includes(dir)) + expect(shimDirs).not.toHaveLength(0) + // Why these keep leading: an nvm/mise/asdf user's runtime must beat a + // system install, which is the reason this seeding is ordered at all. + for (const dir of shimDirs) { + expect(segments.indexOf(dir)).toBeLessThan(segments.indexOf('/usr/bin')) + } + // Why these do not: the same list carries the generic user bin dirs, which + // hold whatever was last installed there rather than a managed toolchain. + for (const dir of genericUserBinDirs) { + expect(segments.indexOf(dir)).toBeGreaterThan(segments.indexOf('/usr/bin')) + } + }) + it('leaves PATH untouched when the app is not packaged', async () => { const { app } = await import('electron') const { patchPackagedProcessPath } = await import('./configure-process') diff --git a/src/main/startup/configure-process.ts b/src/main/startup/configure-process.ts index e49b7e04a03..6b167d7fe14 100644 --- a/src/main/startup/configure-process.ts +++ b/src/main/startup/configure-process.ts @@ -117,20 +117,37 @@ export function patchPackagedProcessPath(): void { } const home = process.env.HOME ?? '' - const extraPaths: string[] = [] + // Why two lists: a seed exists so a GUI-launched Electron can *find* a tool + // its minimal PATH omits. Putting one ahead of the inherited PATH does more + // than that — it re-ranks binaries the user already has, and `~/bin` and + // `~/.local/bin` are arbitrary user-writable directories that can shadow any + // system tool. On the #18234 reporter's box `~/.local/bin/gh` is a wrapper + // around `mise x gh -- gh`; hoisting it over /usr/bin/gh made us run the + // wrapper where their own shell ran the real binary, and the inner bare `gh` + // then resolved back to the wrapper. So: append these, and let a real + // ordering opinion come from the login shell via mergePathSegments. + const isGenericUserBinDir = (path: string): boolean => + process.platform !== 'win32' && + home !== '' && + (path === join(home, 'bin') || path === join(home, '.local/bin')) + const appendPaths: string[] = [] + // Why these still lead: version-manager shims must beat a system install or + // an nvm/mise/asdf user gets the wrong runtime, which is the whole reason + // this seeding is ordered rather than appended (see hydrate-shell-path.ts). + const prependPaths: string[] = [] if (process.platform !== 'win32') { - extraPaths.push('/opt/homebrew/bin', '/opt/homebrew/sbin', '/usr/local/bin', '/usr/local/sbin') + appendPaths.push('/opt/homebrew/bin', '/opt/homebrew/sbin', '/usr/local/bin', '/usr/local/sbin') if (process.platform === 'linux') { // Why: snap and Linuxbrew ship on Linux only, so seeding them elsewhere adds phantom PATH entries every spawn must stat. - extraPaths.push('/snap/bin', '/home/linuxbrew/.linuxbrew/bin') + appendPaths.push('/snap/bin', '/home/linuxbrew/.linuxbrew/bin') } - extraPaths.push('/nix/var/nix/profiles/default/bin') + appendPaths.push('/nix/var/nix/profiles/default/bin') if (home) { - extraPaths.push( + appendPaths.push( join(home, 'bin'), join(home, '.local/bin'), join(home, '.nix-profile/bin'), @@ -142,18 +159,24 @@ export function patchPackagedProcessPath(): void { } // Why: version-manager CLIs use env-node shebangs, so node must be on PATH or spawns fail (also seeds Windows user-local dirs). - extraPaths.push(...getVersionManagerBinPaths()) + // Why the filter: that list carries `~/bin` and `~/.local/bin` too, because + // bun/pnpm/npm --user also install there. Those two are generic user bin + // directories, not a version manager's own shim directory, so they hold + // whatever the user last dropped in them and must not outrank a system dir. + // The specific dirs (.volta/bin, .asdf/shims, mise shims, .bun/bin, …) keep + // leading, which is what the ordering was actually for. + prependPaths.push(...getVersionManagerBinPaths().filter((path) => !isGenericUserBinDir(path))) const pathKey = process.platform === 'win32' && process.env.Path !== undefined ? 'Path' : 'PATH' const currentPath = process.env[pathKey] ?? '' const pathDelimiter = getProcessPathDelimiter() - const existing = new Set(currentPath.split(pathDelimiter)) - const missing = extraPaths.filter((path) => !existing.has(path)) + const currentSegments = currentPath.split(pathDelimiter).filter(Boolean) + const existing = new Set(currentSegments) + const prepend = prependPaths.filter((path) => !existing.has(path)) + const append = appendPaths.filter((path) => !existing.has(path) && !prepend.includes(path)) - if (missing.length > 0) { - process.env[pathKey] = [...missing, ...currentPath.split(pathDelimiter).filter(Boolean)].join( - pathDelimiter - ) + if (prepend.length > 0 || append.length > 0) { + process.env[pathKey] = [...prepend, ...currentSegments, ...append].join(pathDelimiter) } } From c61ca56a9b0970a8ac68706db463f919e3acacdc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:41:13 -0700 Subject: [PATCH 17/77] fix(ssh): resolve the worktree's execution host instead of guessing from one repo row (#17909) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(host-routing): resolve the execution host before reading a connection Three issues in one defect class: a resolver reads one spelling of one arbitrarily chosen row instead of resolving the worktree's execution host, so something local answers a question about a remote. returned that row's connectionId. With duplicate repo rows for one repo id it could pair a runtime owner with a client-owned SSH connection. It now resolves through the same ambiguity-aware index getRuntimeEnvironmentIdForWorktree uses, prefers the repo row for the host the worktree names, and derives the connection from the resolved host. Conflicting rows return `undefined` (this module's documented "cannot determine the host"), never `null`. `store.getRepo(worktree.repoId)?.connectionId ?? null`. `getRepo` is host-blind and the same repo id can exist on local, SSH and runtime hosts, so a remote worktree could spawn its PTY on the client with the remote cwd. resolveWorktreeLaunchHost picks the row for the worktree's host and reads the connection off that host; conflicting rows are unresolved, not local. session-partition owner maps that contradict each other. Both now compute through one shared function whose argument records the divergence. No behaviour change on either side: converging needs a read-both migration, since both partitions hold real data written by shipping builds. * fix(host-routing): keep nested SSH connections resolvable under a runtime host getRepoSshConnectionId read only the resolved execution host, so a repo row owned by a runtime that reaches a nested SSH target (connectionId: ssh-*, executionHostId: runtime:*) resolved to no connection — answering 'local' for a remote worktree, the same defect #17909 fixed in the other direction. * fix(host-routing): resolve both sides of the execution host through one rule The renderer resolver leaked between two different SSH hosts: a worktree on `ssh:m4air` whose only indexed repo row belonged to `openclaw` answered 'openclaw', because the host-scoped lookup missing fell through to an id-only one. Main's resolver, in the same change, answered 'm4air' — two resolvers, one right and one wrong, on identical input. Both sides now adapt one shared rule (`worktree-execution-host-resolution.ts`): the worktree's own host outranks every repo row, and a row on a different host is never evidence about this one. The renderer's WeakMap index becomes the memoizing adapter it always was; `resolveWorktreeLaunchHost` becomes main's mapping of unresolved onto its throw. Settles the rule the change previously answered two ways. `getRepoSshConnectionId` and `getSshTargetIdForExecutionHost` disagreed for a runtime host carrying a nested `connectionId`; they now compose, so the execution host is the single authority. On a `runtime:*` row that field is a paired HUB's private SSH target, spread through by `repoWithFetchedOwner` and unaddressable from this client — the project-first successor of the row nulls it for exactly that reason. That also fixes the `kind !== 'ssh'` fallback, which fired for `local`: a row declaring itself local handed out an SSH connection. --- ...ser-network-execution-host-for-worktree.ts | 14 +- .../runtime-workspace-session-controller.ts | 7 +- .../runtime/worktree-launch-host-repo.test.ts | 94 ++++++++++ src/main/runtime/worktree-launch-host-repo.ts | 43 +++++ .../src/lib/connection-context.test.ts | 147 +++++++++++++++ .../src/lib/connection-owner-resolution.ts | 28 ++- .../lib/workspace-session-host-persistence.ts | 6 +- src/shared/execution-host.test.ts | 37 ++++ src/shared/execution-host.ts | 38 ++++ .../workspace-session-partition-owner.test.ts | 29 +++ .../workspace-session-partition-owner.ts | 39 ++++ ...worktree-execution-host-resolution.test.ts | 167 ++++++++++++++++++ .../worktree-execution-host-resolution.ts | 109 ++++++++++++ 13 files changed, 749 insertions(+), 9 deletions(-) create mode 100644 src/main/runtime/worktree-launch-host-repo.test.ts create mode 100644 src/main/runtime/worktree-launch-host-repo.ts create mode 100644 src/shared/workspace-session-partition-owner.test.ts create mode 100644 src/shared/workspace-session-partition-owner.ts create mode 100644 src/shared/worktree-execution-host-resolution.test.ts create mode 100644 src/shared/worktree-execution-host-resolution.ts diff --git a/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts b/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts index 87411ad55d7..decc28aa45b 100644 --- a/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts +++ b/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts @@ -10,6 +10,7 @@ import { import { resolveRuntimeBrowserNetworkExecutionHost } from './runtime-browser-network-execution-host' import { resolveLocalProjectRuntimeForWorktreeId } from '../local-project-runtime-resolution' import { getRegisteredSshState } from '../ssh/ssh-target-registry' +import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' import { folderWorkspaceKey, parseWorkspaceKey } from '../../shared/workspace-scope' import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { ResolvedWorktree } from './runtime-worktree-path-identity' @@ -124,7 +125,16 @@ export class OrcaRuntimeWithResolveBrowserNetworkExecutionHostForWorktree extend const parsed = parseWorkspaceKey(workspaceSelector) const worktreeSelector = parsed?.type === 'worktree' ? `id:${parsed.worktreeId}` : selector const worktree = await this.resolveWorktreeSelector(worktreeSelector) - const repo = this.store?.getRepo(worktree.repoId) ?? null + // Why: `getRepo(id)` is host-blind and the same repo id can exist on local, SSH and runtime + // hosts. Reading `connectionId` off an arbitrary row reports "local" for a remote worktree and + // spawns its PTY on the client with the remote cwd (#11163). Loss of a usable answer is + // `unresolved`, never `local`. + const resolution = resolveWorktreeLaunchHost(this.store?.getRepos() ?? [], worktree) + if (resolution.kind === 'ambiguous') { + throw new Error('worktree_execution_host_unresolved') + } + // Metadata only (display name, hook settings); the routing decision is `resolution.connectionId`. + const repo = resolution.repo ?? this.store?.getRepo(worktree.repoId) ?? null triggerTerminalSpawnPushTargetMaterialization( worktree.path, worktree.pushTarget, @@ -137,7 +147,7 @@ export class OrcaRuntimeWithResolveBrowserNetworkExecutionHostForWorktree extend scope: { id: worktree.id, path: worktree.path, - connectionId: repo?.connectionId ?? null, + connectionId: resolution.connectionId, repo, folderWorkspace: null }, diff --git a/src/main/runtime/runtime-workspace-session-controller.ts b/src/main/runtime/runtime-workspace-session-controller.ts index 3c252a06b60..168f76e256c 100644 --- a/src/main/runtime/runtime-workspace-session-controller.ts +++ b/src/main/runtime/runtime-workspace-session-controller.ts @@ -8,6 +8,7 @@ import { import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { getRepoIdFromWorktreeId } from '../../shared/worktree/id' +import { workspaceSessionPartitionHostId } from '../../shared/workspace-session-partition-owner' import { parseWorkspaceKey } from '../../shared/workspace-scope' import type { RuntimeStore } from './runtime-store-contract' @@ -44,7 +45,11 @@ export class RuntimeWorkspaceSessionController { } const resolvedWorktreeId = scope?.type === 'worktree' ? scope.worktreeId : worktreeId const repo = store?.getRepo?.(getRepoIdFromWorktreeId(resolvedWorktreeId)) - return repo ? getRepoExecutionHostId(repo) : LOCAL_EXECUTION_HOST_ID + // Why: SSH worktrees keep their own `ssh:` partition here while the renderer writes + // them to 'local'; the shared owner map records that divergence (#12723). + return repo + ? workspaceSessionPartitionHostId(getRepoExecutionHostId(repo), 'host-partition') + : LOCAL_EXECUTION_HOST_ID } getHostId(worktreeId: string): ExecutionHostId { diff --git a/src/main/runtime/worktree-launch-host-repo.test.ts b/src/main/runtime/worktree-launch-host-repo.test.ts new file mode 100644 index 00000000000..8a81f2424d1 --- /dev/null +++ b/src/main/runtime/worktree-launch-host-repo.test.ts @@ -0,0 +1,94 @@ +import { describe, expect, it } from 'vitest' +import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' + +// Why (#11163): the terminal launch scope read +// `store.getRepo(worktree.repoId)?.connectionId ?? null` — one spelling of one arbitrarily chosen +// row instead of the worktree's execution host. A remote worktree then spawns its PTY on the +// client with the remote cwd (`DaemonProtocolError: Working directory "…" does not exist`). +describe('resolveWorktreeLaunchHost', () => { + const localRow = { id: 'shared', path: '/local/repo' } + const sshRow = { id: 'shared', path: '/remote/repo', connectionId: 'ssh-b' } + + it('reports ambiguous when duplicate repo rows disagree about the owning host', () => { + expect(resolveWorktreeLaunchHost([localRow, sshRow], { repoId: 'shared' })).toEqual({ + kind: 'ambiguous' + }) + }) + + it('resolves the row for the host the worktree names', () => { + expect( + resolveWorktreeLaunchHost([localRow, sshRow], { repoId: 'shared', hostId: 'ssh:ssh-b' }) + ).toEqual({ kind: 'resolved', repo: sshRow, connectionId: 'ssh-b' }) + expect( + resolveWorktreeLaunchHost([localRow, sshRow], { repoId: 'shared', hostId: 'local' }) + ).toEqual({ kind: 'resolved', repo: localRow, connectionId: null }) + }) + + // The settled rule: the execution host is authoritative, and a row on some *other* host is never + // evidence about this one — not for the connection, and not for the metadata row either. This is + // the question `getRepoSshConnectionId` and `getSshTargetIdForExecutionHost` once answered two + // ways; they now compose, and `execution-host.test.ts` pins the composition. + it('never hands a worktree a connection belonging to a different host', () => { + const clientOwnedRow = { id: 'r', path: '/p', connectionId: 'ssh-client' } + expect( + resolveWorktreeLaunchHost([clientOwnedRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', repo: null, connectionId: null }) + // Even the runtime host's *own* row contributes no PTY route: its nested target lives in that + // machine's namespace, so spawning against it here would dial the wrong box. The renderer + // reads the same resolution and does want that id — see execution-host.test.ts. + const nestedRow = { + id: 'r', + path: '/p', + connectionId: 'ssh-nested', + executionHostId: 'runtime:env-a' as const + } + expect( + resolveWorktreeLaunchHost([nestedRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', repo: nestedRow, connectionId: null }) + // Two SSH hosts, one shared repo id: the worktree's own host wins outright. + expect( + resolveWorktreeLaunchHost([{ id: 'r', path: '/p', connectionId: 'openclaw' }], { + repoId: 'r', + hostId: 'ssh:m4air' + }) + ).toEqual({ kind: 'resolved', repo: null, connectionId: 'm4air' }) + expect( + resolveWorktreeLaunchHost( + [ + { id: 'r', path: '/p', connectionId: 'openclaw' }, + { id: 'r', path: '/q', connectionId: 'm4air' } + ], + { repoId: 'r', hostId: 'ssh:m4air' } + ) + ).toEqual({ + kind: 'resolved', + repo: { id: 'r', path: '/q', connectionId: 'm4air' }, + connectionId: 'm4air' + }) + // A row declaring itself local hands out no SSH connection, whatever `connectionId` says. + expect( + resolveWorktreeLaunchHost([{ id: 'r', path: '/p', connectionId: 'openclaw' }], { + repoId: 'r', + hostId: 'local' + }) + ).toEqual({ kind: 'resolved', repo: null, connectionId: null }) + }) + + it('leaves a single unambiguous row alone', () => { + expect(resolveWorktreeLaunchHost([sshRow], { repoId: 'shared' })).toEqual({ + kind: 'resolved', + repo: sshRow, + connectionId: 'ssh-b' + }) + expect(resolveWorktreeLaunchHost([localRow], { repoId: 'shared' })).toEqual({ + kind: 'resolved', + repo: localRow, + connectionId: null + }) + expect(resolveWorktreeLaunchHost([], { repoId: 'shared' })).toEqual({ + kind: 'resolved', + repo: null, + connectionId: null + }) + }) +}) diff --git a/src/main/runtime/worktree-launch-host-repo.ts b/src/main/runtime/worktree-launch-host-repo.ts new file mode 100644 index 00000000000..fa2641655b3 --- /dev/null +++ b/src/main/runtime/worktree-launch-host-repo.ts @@ -0,0 +1,43 @@ +import { + createRepoRowExecutionHostLookup, + resolveWorktreeExecutionHost, + type ExecutionHostOwnerRow +} from '../../shared/worktree-execution-host-resolution' +import { getSshTargetIdForExecutionHost } from '../../shared/execution-host' +import type { Repo } from '../../shared/repo-types' + +export type LaunchHostRepo = Pick + +export type WorktreeLaunchHostResolution = + | { kind: 'resolved'; repo: T | null; connectionId: string | null } + | { kind: 'ambiguous' } + +/** + * Main-side adapter over the shared execution-host rule + * (`src/shared/worktree-execution-host-resolution.ts`), which the renderer's owner index answers + * with too. Two things are local to this side: + * + * - rival rows that disagree about the host are `ambiguous` and the launch scope throws, while an + * id nobody carries stays "no repo, no connection" — the launch path's long-standing behaviour + * for a worktree whose repo row has gone; + * - the connection comes off the *host*, not the resolved row. This is a client-dialable PTY + * route, so a `runtime:` host contributes nothing: its nested SSH target belongs to that + * machine's namespace and spawning against it here would dial the wrong box. The renderer wants + * the opposite answer from the same resolution, which is why the shared type carries both. + */ +export function resolveWorktreeLaunchHost( + repos: readonly T[], + worktree: { repoId: string; hostId?: string | null } +): WorktreeLaunchHostResolution { + const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) + if (resolution.kind === 'unresolved') { + return resolution.reason === 'ambiguous' + ? { kind: 'ambiguous' } + : { kind: 'resolved', repo: null, connectionId: null } + } + return { + kind: 'resolved', + repo: resolution.owner, + connectionId: getSshTargetIdForExecutionHost(resolution.hostId) + } +} diff --git a/src/renderer/src/lib/connection-context.test.ts b/src/renderer/src/lib/connection-context.test.ts index 2ef70ba95e2..fae6c74cdd8 100644 --- a/src/renderer/src/lib/connection-context.test.ts +++ b/src/renderer/src/lib/connection-context.test.ts @@ -27,6 +27,27 @@ function makeRepo(overrides: Partial & { id: string }): Repo { } } +function makeWorktree(overrides: Partial & { id: string; repoId: string }): Worktree { + return { + path: '/srv/repo', + head: 'abc123', + branch: 'refs/heads/main', + isBare: false, + isMainWorktree: false, + displayName: 'Workspace', + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 0, + ...overrides + } +} + describe('getConnectionId', () => { afterEach(() => { useAppStore.setState(initialState, true) @@ -525,6 +546,132 @@ describe('getConnectionIdFromState', () => { expect(getConnectionIdFromState(state, 'repo-ssh::/home/neil/repo-feature')).toBe('ssh-2') }) + it('refuses to resolve a connection when duplicate repo rows disagree about the owning host', () => { + // Why (#17799): a repo id carried by two rows — one runtime-owned, one holding a + // client-owned SSH connection — must not hand the client's connection to the runtime. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ id: 'repo-dup', executionHostId: 'runtime:env-a' }), + makeRepo({ id: 'repo-dup', connectionId: 'ssh-client' }) + ], + worktreesByRepo: {} + } + + expect(getConnectionIdFromState(state, 'repo-dup::/home/neil/repo-feature')).toBeUndefined() + }) + + it('still resolves duplicate repo rows that agree about the owning host', () => { + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ id: 'repo-dup', connectionId: 'ssh-same' }), + makeRepo({ id: 'repo-dup', connectionId: 'ssh-same', path: '/home/neil/other' }) + ], + worktreesByRepo: {} + } + + expect(getConnectionIdFromState(state, 'repo-dup::/home/neil/repo-feature')).toBe('ssh-same') + }) + + it('never hands a worktree the SSH connection of a different host', () => { + // Why (#11163): two SSH hosts, one shared repo id. The worktree names `ssh:m4air`; the only + // indexed row belongs to `openclaw`. An id-only fallback after the host lookup misses answers + // with the wrong host's connection — "Reconnect openclaw" on an m4air pane, and file reads + // routed to a machine that never held the path. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [makeRepo({ id: 'repo-shared', connectionId: 'openclaw' })], + worktreesByRepo: { + 'repo-shared': [ + makeWorktree({ + id: 'repo-shared::/srv/repo', + repoId: 'repo-shared', + hostId: 'ssh:m4air' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-shared::/srv/repo')).toBe('m4air') + }) + + it('never hands a runtime-hosted worktree a client-owned SSH connection', () => { + // The row is on `ssh:openclaw`, not on the runtime host, so it says nothing about this + // worktree. This is the cross-host case, not the nested-SSH one below. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [makeRepo({ id: 'repo-shared', connectionId: 'openclaw' })], + worktreesByRepo: { + 'repo-shared': [ + makeWorktree({ + id: 'repo-shared::/srv/repo', + repoId: 'repo-shared', + hostId: 'runtime:awin' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-shared::/srv/repo')).toBeNull() + }) + + it('keeps a runtime host nested SSH target, which decides local readability', () => { + // `repoWithFetchedOwner` stamps the runtime host and spreads the nested target through. The + // pane pairs it with the environment (`selectRuntimeAwareSshStatus`) for reconnect state, and + // `isNativeChatTranscriptLocalReadable` treats a null here as "this client can read it" — so + // dropping it would send a transcript read to the wrong machine. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ + id: 'repo-runtime', + connectionId: 'ssh-nested', + executionHostId: 'runtime:env-a' + }) + ], + worktreesByRepo: { + 'repo-runtime': [ + makeWorktree({ + id: 'repo-runtime::/srv/repo', + repoId: 'repo-runtime', + hostId: 'runtime:env-a', + runtimeOwnerEnvironmentId: 'env-a' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-runtime::/srv/repo')).toBe('ssh-nested') + }) + + it('resolves the row on the SSH host the worktree names when both hosts carry the id', () => { + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ id: 'repo-shared', connectionId: 'openclaw' }), + makeRepo({ id: 'repo-shared', connectionId: 'm4air', path: '/srv/repo' }) + ], + worktreesByRepo: { + 'repo-shared': [ + makeWorktree({ + id: 'repo-shared::/srv/repo', + repoId: 'repo-shared', + hostId: 'ssh:m4air' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-shared::/srv/repo')).toBe('m4air') + }) + it('indexes immutable worktree and repo snapshots once across repeated selector calls', () => { let worktreeIdReads = 0 let repoIdReads = 0 diff --git a/src/renderer/src/lib/connection-owner-resolution.ts b/src/renderer/src/lib/connection-owner-resolution.ts index 91169e939c6..f1b52607d46 100644 --- a/src/renderer/src/lib/connection-owner-resolution.ts +++ b/src/renderer/src/lib/connection-owner-resolution.ts @@ -1,5 +1,10 @@ import type { AppState } from '@/store/types' -import { getIndexedRepoMap, getIndexedWorktreeMap } from '@/store/worktree-repo-index' +import { + findIndexedRepoOwnerForHost, + resolveIndexedRepoOwner, + resolveIndexedWorktreeOwner +} from './worktree-runtime-owner-index' +import { resolveWorktreeExecutionHost } from '../../../shared/worktree-execution-host-resolution' import { FLOATING_TERMINAL_WORKTREE_ID } from '../../../shared/constants' import { getRepoIdFromWorktreeId } from '../../../shared/worktree/id' import { parseWorkspaceKey } from '../../../shared/workspace-scope' @@ -60,10 +65,25 @@ export function getConnectionIdFromState( } // Why: owner resolution runs from retained Zustand selectors, so unrelated // store writes must not flatten every worktree or scan every repository. - const worktree = getIndexedWorktreeMap(state.worktreesByRepo).get(worktreeId) + const worktreeResolution = resolveIndexedWorktreeOwner(state.worktreesByRepo, worktreeId) + if (worktreeResolution.kind === 'ambiguous') { + // Why (#17799): rows that disagree about the owner cannot name a connection. + // `undefined` is this module's documented "cannot determine the host" answer; + // collapsing it to `null` would authorize a local read of a remote path. + return undefined + } + const worktree = worktreeResolution.kind === 'resolved' ? worktreeResolution.owner : undefined const repoId = worktree?.repoId ?? getRepoIdFromWorktreeId(worktreeId) - const repo = getIndexedRepoMap(state.repos).get(repoId) - return repo ? (repo.connectionId ?? null) : undefined + // Why (#17799, #11163): one rule, shared with main's launch scope. The renderer's contribution is + // only the memoized index — unrelated store writes must not rescan every repository. + const resolution = resolveWorktreeExecutionHost( + { + byId: (id) => resolveIndexedRepoOwner(state.repos, id), + byHost: (id, hostId) => findIndexedRepoOwnerForHost(state.repos, id, hostId) + }, + { repoId, hostId: worktree?.hostId ?? null } + ) + return resolution.kind === 'resolved' ? resolution.connectionId : undefined } export function getConnectionIdForFileFromState( diff --git a/src/renderer/src/lib/workspace-session-host-persistence.ts b/src/renderer/src/lib/workspace-session-host-persistence.ts index 1ee386883d5..07dd0b2f775 100644 --- a/src/renderer/src/lib/workspace-session-host-persistence.ts +++ b/src/renderer/src/lib/workspace-session-host-persistence.ts @@ -9,6 +9,7 @@ import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host' +import { workspaceSessionPartitionHostId } from '../../../shared/workspace-session-partition-owner' import { parseWorkspaceKey } from '../../../shared/workspace-scope' import { getRepoIdFromWorktreeId } from '../../../shared/worktree/id' import { @@ -187,8 +188,9 @@ export function buildHostSessionRouting(state: HostPersistenceState): HostSessio if (!repoHostId) { return LOCAL_EXECUTION_HOST_ID } - const parsed = parseExecutionHostId(repoHostId) - return parsed?.kind === 'runtime' ? parsed.id : LOCAL_EXECUTION_HOST_ID + // Why: SSH-owned worktrees stay in the 'local' partition here while the runtime writes them to + // `ssh:`; the shared owner map records that divergence (#12723). + return workspaceSessionPartitionHostId(repoHostId, 'local-partition') } return { hostIdByWorktreeId, claims } } diff --git a/src/shared/execution-host.test.ts b/src/shared/execution-host.test.ts index de895978f3b..9905fc5fe2a 100644 --- a/src/shared/execution-host.test.ts +++ b/src/shared/execution-host.test.ts @@ -4,7 +4,9 @@ import { LOCAL_EXECUTION_HOST_ID, getLocalExecutionHostLabel, getRepoExecutionHostId, + getRepoSshConnectionId, getSettingsFocusedExecutionHostId, + getSshTargetIdForExecutionHost, getWorktreeExecutionHostId, normalizeExecutionHostOrder, normalizeExecutionHostScope, @@ -113,6 +115,41 @@ describe('execution host identity', () => { expect(getWorktreeExecutionHostId({}, {}, 'runtime:focused-host')).toBe('runtime:focused-host') }) + // These two look interchangeable and are not: one answers "which SSH target holds this row's + // files", the other "which connection may this client dial". They agree except on a runtime + // host, where a nested target exists but is not dialable from here — so the pane that reads it + // needs one answer and the PTY route needs the other. + it('distinguishes the SSH target holding a row from the connection this client may dial', () => { + // Legacy spelling: `connectionId` alone *is* the host, so both answers agree. + expect(getRepoSshConnectionId({ connectionId: 'openclaw' })).toBe('openclaw') + expect(getSshTargetIdForExecutionHost('ssh:openclaw')).toBe('openclaw') + // Unified spelling, no legacy field. + expect(getRepoSshConnectionId({ executionHostId: 'ssh:m4air' })).toBe('m4air') + + // A row declaring itself local hands out no SSH connection, whatever the legacy field says: + // `local` has no SSH namespace to nest in, so the two spellings are contradicting each other. + expect( + getRepoSshConnectionId({ executionHostId: 'local', connectionId: 'openclaw' }) + ).toBeNull() + + // A runtime host does have its own namespace, and a nested target appears only in this field. + // Dropping it would make a nested-SSH workspace read as local — which is what decides whether + // this client tries to read the transcript itself. + expect( + getRepoSshConnectionId({ executionHostId: 'runtime:env-a', connectionId: 'ssh-nested' }) + ).toBe('ssh-nested') + // ...but that id is not dialable from this client alone, so the routing answer stays null. + expect(getSshTargetIdForExecutionHost('runtime:env-a')).toBeNull() + // A runtime host with no nested target is simply not on SSH. + expect(getRepoSshConnectionId({ executionHostId: 'runtime:env-a' })).toBeNull() + + // An ephemeral-VM target is an ordinary client-dialable target and stays an `ssh:` host. + expect(getRepoSshConnectionId({ connectionId: 'runtime-ssh-vm-1' })).toBe('runtime-ssh-vm-1') + expect(getRepoExecutionHostId({ connectionId: 'runtime-ssh-vm-1' })).toBe( + 'ssh:runtime-ssh-vm-1' + ) + }) + it('derives focused host compatibility from active runtime settings', () => { expect(getSettingsFocusedExecutionHostId(null)).toBe(LOCAL_EXECUTION_HOST_ID) expect(getSettingsFocusedExecutionHostId({ activeRuntimeEnvironmentId: 'runtime-1' })).toBe( diff --git a/src/shared/execution-host.ts b/src/shared/execution-host.ts index aacbc695cdc..a77d02b3882 100644 --- a/src/shared/execution-host.ts +++ b/src/shared/execution-host.ts @@ -166,6 +166,44 @@ export function getRepoExecutionHostId( return connectionId ? toSshExecutionHostId(connectionId) : LOCAL_EXECUTION_HOST_ID } +export function getSshTargetIdForExecutionHost( + executionHostId: string | null | undefined +): string | null { + const parsed = parseExecutionHostId(executionHostId) + return parsed?.kind === 'ssh' ? parsed.targetId : null +} + +// Why: SSH ownership has two spellings on a repo row — the legacy `connectionId` +// field and the unified `executionHostId`. Routing that reads the raw field answers +// "local" for a row that only carries `ssh:`, which runs a remote operation +// on the client. Resolve the host first, then read the connection off it. +// +// The two hosts that are not themselves SSH are not the same case: +// +// - `local` has no SSH namespace to nest in, so a surviving `connectionId` is a row +// contradicting itself — the shape main's `resolveRepoOwnershipEvidence` calls +// `contradictory`. Answering with it hands out an SSH connection for a row that declares +// itself local. +// - `runtime:` is a different machine with its own SSH targets, and a nested one appears +// only in this field (`repoWithFetchedOwner` spreads it through). It is not dialable on its +// own, but it is addressable as the pair (environmentId, targetId) — which is how the +// renderer reads it, recovering the environment from the worktree and looking the target up +// inside it (`selectRuntimeAwareSshStatus`). Dropping it makes a nested-SSH workspace read +// as local, which is what decides whether a transcript is read on this client. +// +// So this answers "which SSH target holds this row's files", not "which connection may this +// client dial". `getSshTargetIdForExecutionHost` answers the latter; callers routing a +// client-local PTY or Git provider want that one instead. +export function getRepoSshConnectionId( + repo: Pick +): string | null { + const host = parseExecutionHostId(getRepoExecutionHostId(repo)) + if (host?.kind === 'ssh') { + return host.targetId + } + return host?.kind === 'runtime' ? normalizeHostPart(repo.connectionId) : null +} + export function getWorktreeExecutionHostId( worktree: Pick, repo: Pick | undefined, diff --git a/src/shared/workspace-session-partition-owner.test.ts b/src/shared/workspace-session-partition-owner.test.ts new file mode 100644 index 00000000000..f4362471c66 --- /dev/null +++ b/src/shared/workspace-session-partition-owner.test.ts @@ -0,0 +1,29 @@ +import { describe, expect, it } from 'vitest' +import { workspaceSessionPartitionHostId } from './workspace-session-partition-owner' + +// Why (#12723): the renderer and the runtime used two independent owner maps for the same +// worktree's session state. They now share one function, so the divergence is a single argument +// and cannot drift further. Behaviour on both sides is unchanged. +describe('workspaceSessionPartitionHostId', () => { + it('keeps runtime worktrees in their own partition on both sides', () => { + expect(workspaceSessionPartitionHostId('runtime:env-a', 'local-partition')).toBe( + 'runtime:env-a' + ) + expect(workspaceSessionPartitionHostId('runtime:env-a', 'host-partition')).toBe('runtime:env-a') + }) + + it('keeps local worktrees local on both sides', () => { + expect(workspaceSessionPartitionHostId('local', 'local-partition')).toBe('local') + expect(workspaceSessionPartitionHostId('local', 'host-partition')).toBe('local') + }) + + it('records the SSH divergence as the only difference between the two models', () => { + expect(workspaceSessionPartitionHostId('ssh:devbox', 'local-partition')).toBe('local') + expect(workspaceSessionPartitionHostId('ssh:devbox', 'host-partition')).toBe('ssh:devbox') + }) + + it('falls back to the local partition for unparseable host ids', () => { + expect(workspaceSessionPartitionHostId(null, 'host-partition')).toBe('local') + expect(workspaceSessionPartitionHostId('nonsense', 'host-partition')).toBe('local') + }) +}) diff --git a/src/shared/workspace-session-partition-owner.ts b/src/shared/workspace-session-partition-owner.ts new file mode 100644 index 00000000000..83166fb76f3 --- /dev/null +++ b/src/shared/workspace-session-partition-owner.ts @@ -0,0 +1,39 @@ +import { + LOCAL_EXECUTION_HOST_ID, + parseExecutionHostId, + type ExecutionHostId +} from './execution-host' + +/** + * Where an SSH-owned worktree's durable session state lives. + * + * This is the single axis on which the renderer and the main-process runtime disagree today + * (stablyai/orca#12723). Both sides now compute their partition through this function so the + * divergence is one argument in one place instead of two independently drifting owner maps: + * + * - `local-partition` — the renderer's shipping model. SSH worktrees keep their session state in + * the `local` partition; partitioning them would double-own the data. + * - `host-partition` — the runtime's shipping model (#12671). Pane retirement, windowless PTY + * handoff and orchestration fences read-modify-write `ssh:`. + * + * Both partitions hold real data written by shipping builds, so neither side can simply adopt the + * other's answer: flipping a resolver orphans whichever store it stops reading. Converging needs a + * read-both transition (generalize `workspaceSessionPartitionIdsForHost`) and should converge on + * `host-partition`, since Orca Remote — SSH's successor — is already partitioned as `runtime:*`. + * Until then this function preserves today's behaviour exactly on both sides. + */ +export type WorkspaceSessionSshOwnership = 'local-partition' | 'host-partition' + +export function workspaceSessionPartitionHostId( + executionHostId: string | null | undefined, + sshOwnership: WorkspaceSessionSshOwnership +): ExecutionHostId { + const parsed = parseExecutionHostId(executionHostId) + if (parsed?.kind === 'runtime') { + return parsed.id + } + if (parsed?.kind === 'ssh') { + return sshOwnership === 'host-partition' ? parsed.id : LOCAL_EXECUTION_HOST_ID + } + return LOCAL_EXECUTION_HOST_ID +} diff --git a/src/shared/worktree-execution-host-resolution.test.ts b/src/shared/worktree-execution-host-resolution.test.ts new file mode 100644 index 00000000000..70ee6d304b6 --- /dev/null +++ b/src/shared/worktree-execution-host-resolution.test.ts @@ -0,0 +1,167 @@ +import { describe, expect, it } from 'vitest' +import { + createRepoRowExecutionHostLookup, + resolveWorktreeExecutionHost +} from './worktree-execution-host-resolution' + +// Why (#11163, #17799): main's terminal launch scope and the renderer's owner index both answer +// "which host does this worktree execute on". They used to answer it separately, and disagreed — +// main derived the host from the worktree while the renderer fell back to an id-only repo lookup, +// so a pane on one SSH host was routed to another. One rule now, exercised here directly. +const resolve = ( + repos: readonly { id: string; connectionId?: string; executionHostId?: string }[], + worktree: { repoId: string; hostId?: string | null } +): ReturnType => + resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos as never), worktree) as never + +describe('resolveWorktreeExecutionHost', () => { + describe('the worktree names its own host', () => { + it('routes to that host even when the only row belongs to a different SSH host', () => { + // The reproduced defect: `ssh:m4air` worktree, sole row on `openclaw`. + expect( + resolve([{ id: 'r', connectionId: 'openclaw' }], { repoId: 'r', hostId: 'ssh:m4air' }) + ).toEqual({ kind: 'resolved', hostId: 'ssh:m4air', connectionId: 'm4air', owner: null }) + }) + + it('answers before the repo row hydrates, because the host is not a guess', () => { + // Deliberate change from "unresolved": #6648 blocks destructive ops while the *host* is + // unknown. A worktree naming `ssh:m4air` is not that case — the repo row adds nothing the + // host id has not already settled, and refusing here stalls a remote pane on hydration. + expect(resolve([], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: null + }) + }) + + it('picks the row on that host when both SSH hosts carry the id', () => { + const openclaw = { id: 'r', connectionId: 'openclaw' } + const m4air = { id: 'r', connectionId: 'm4air' } + expect(resolve([openclaw, m4air], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: m4air + }) + expect(resolve([openclaw, m4air], { repoId: 'r', hostId: 'ssh:openclaw' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:openclaw', + connectionId: 'openclaw', + owner: openclaw + }) + }) + + it('matches a row that names the host in either spelling', () => { + const stamped = { id: 'r', executionHostId: 'ssh:m4air' } + expect(resolve([stamped], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: stamped + }) + }) + + it('takes no connection from a row on a different host, whatever this host is', () => { + // The row lives on `ssh:openclaw`; neither a local nor a runtime worktree may borrow it. + for (const hostId of ['local', 'runtime:env-a']) { + expect(resolve([{ id: 'r', connectionId: 'openclaw' }], { repoId: 'r', hostId })).toEqual({ + kind: 'resolved', + hostId, + connectionId: null, + owner: null + }) + } + }) + + it('reads a runtime host nested SSH target off the row on that same host', () => { + // Not a cross-host borrow: this row *is* the runtime host's row, and the nested target + // appears nowhere else. Nulling it makes the workspace read as local, which decides whether + // this client tries to read a transcript that lives on the nested host. + const nested = { id: 'r', connectionId: 'ssh-nested', executionHostId: 'runtime:env-a' } + expect(resolve([nested], { repoId: 'r', hostId: 'runtime:env-a' })).toEqual({ + kind: 'resolved', + hostId: 'runtime:env-a', + connectionId: 'ssh-nested', + owner: nested + }) + }) + + it('gives a local row no SSH connection even when it carries a stale one', () => { + const contradictory = { id: 'r', connectionId: 'openclaw', executionHostId: 'local' } + expect(resolve([contradictory], { repoId: 'r', hostId: 'local' })).toEqual({ + kind: 'resolved', + hostId: 'local', + connectionId: null, + owner: contradictory + }) + }) + }) + + describe('the worktree names no host', () => { + it('resolves from the sole row, in either spelling', () => { + const legacy = { id: 'r', connectionId: 'openclaw' } + expect(resolve([legacy], { repoId: 'r' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:openclaw', + connectionId: 'openclaw', + owner: legacy + }) + const stamped = { id: 'r', executionHostId: 'ssh:m4air' } + expect(resolve([stamped], { repoId: 'r' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: stamped + }) + const local = { id: 'r' } + expect(resolve([local], { repoId: 'r' })).toEqual({ + kind: 'resolved', + hostId: 'local', + connectionId: null, + owner: local + }) + }) + + it('refuses when rival rows disagree about the host, including two SSH hosts', () => { + expect( + resolve( + [ + { id: 'r', connectionId: 'openclaw' }, + { id: 'r', connectionId: 'm4air' } + ], + { + repoId: 'r' + } + ) + ).toEqual({ kind: 'unresolved', reason: 'ambiguous' }) + expect( + resolve([{ id: 'r', connectionId: 'openclaw' }, { id: 'r' }], { repoId: 'r' }) + ).toEqual({ kind: 'unresolved', reason: 'ambiguous' }) + }) + + it('treats the two spellings of one host as agreement, not conflict', () => { + expect( + resolve( + [ + { id: 'r', connectionId: 'm4air' }, + { id: 'r', executionHostId: 'ssh:m4air' } + ], + { repoId: 'r' } + ) + ).toMatchObject({ kind: 'resolved', hostId: 'ssh:m4air', connectionId: 'm4air' }) + }) + + it('reports an unknown owner distinctly from a conflicting one', () => { + expect(resolve([], { repoId: 'r' })).toEqual({ kind: 'unresolved', reason: 'unknown' }) + }) + }) + + it('ignores an unparseable host id rather than treating it as a host', () => { + const row = { id: 'r', connectionId: 'openclaw' } + expect(resolve([row], { repoId: 'r', hostId: 'ssh:' })).toMatchObject({ + kind: 'resolved', + connectionId: 'openclaw' + }) + }) +}) diff --git a/src/shared/worktree-execution-host-resolution.ts b/src/shared/worktree-execution-host-resolution.ts new file mode 100644 index 00000000000..8abe5de72cc --- /dev/null +++ b/src/shared/worktree-execution-host-resolution.ts @@ -0,0 +1,109 @@ +/** + * One rule for "which host does this worktree execute on, and what connection routes there". + * + * Main and the renderer both have to answer it — the terminal launch scope picks a PTY route from + * it, the renderer picks a file-read route and the reconnect affordance from it — so the rule lives + * here instead of being re-derived per side. Two re-derivations already disagreed: main answered + * from the worktree's own host while the renderer fell back to an id-only repo lookup, so a pane on + * `ssh:m4air` was offered "Reconnect openclaw" and read its files off openclaw (#11163). + * + * `unresolved` is a distinct answer, never "local": the same repo id can exist on a local, an SSH + * and a runtime host at once, and loss of a usable answer must fail closed rather than authorize a + * client-side read of a remote path (#6648, #17799). + */ + +import type { Repo } from './repo-types' +import { + getRepoExecutionHostId, + getRepoSshConnectionId, + getSshTargetIdForExecutionHost, + normalizeExecutionHostId, + type ExecutionHostId +} from './execution-host' + +export type ExecutionHostOwnerRow = Pick + +export type ExecutionHostOwnerMatch = + | { kind: 'resolved'; owner: T } + | { kind: 'missing' } + | { kind: 'ambiguous' } + +/** + * How a caller finds repo rows. Main scans the store array; the renderer answers from a + * WeakMap-memoized index because owner resolution runs inside retained selectors. That is a + * performance difference, not a different rule. + */ +export type ExecutionHostOwnerLookup = { + /** The row for `repoId`, or `ambiguous` when rival rows disagree about the owning host. */ + byId: (repoId: string) => ExecutionHostOwnerMatch + /** The row for `repoId` on exactly `hostId`, or null when that host carries no row. */ + byHost: (repoId: string, hostId: ExecutionHostId) => T | null +} + +export type WorktreeExecutionHostResolution = + | { + kind: 'resolved' + hostId: ExecutionHostId + /** + * The SSH target whose filesystem holds this workspace — for a `runtime:` host, its nested + * target, addressable only as the pair with `hostId`. Callers deciding what *this client* + * may dial (a PTY route, a Git provider) must use `getSshTargetIdForExecutionHost(hostId)` + * instead; this field can name a host the client cannot reach on its own. + */ + connectionId: string | null + /** Display metadata only. The decisions are `hostId` / `connectionId`. */ + owner: T | null + } + | { kind: 'unresolved'; reason: 'ambiguous' | 'unknown' } + +export function resolveWorktreeExecutionHost( + lookup: ExecutionHostOwnerLookup, + worktree: { repoId: string; hostId?: string | null } +): WorktreeExecutionHostResolution { + const worktreeHostId = normalizeExecutionHostId(worktree.hostId) + if (worktreeHostId) { + // The worktree names its own host, which outranks every repo row. A row on a *different* host + // is not evidence about this one — falling back to it is the cross-host leak: one SSH host's + // pane routed to another. A row on *this* host still is evidence, and is the only place a + // runtime's nested SSH target appears. + const owner = lookup.byHost(worktree.repoId, worktreeHostId) + return { + kind: 'resolved', + hostId: worktreeHostId, + connectionId: + getSshTargetIdForExecutionHost(worktreeHostId) ?? + (owner ? getRepoSshConnectionId(owner) : null), + owner + } + } + const match = lookup.byId(worktree.repoId) + if (match.kind !== 'resolved') { + return { kind: 'unresolved', reason: match.kind === 'ambiguous' ? 'ambiguous' : 'unknown' } + } + return { + kind: 'resolved', + hostId: getRepoExecutionHostId(match.owner), + connectionId: getRepoSshConnectionId(match.owner), + owner: match.owner + } +} + +/** Array-backed lookup for callers holding the whole repo list (main's store). */ +export function createRepoRowExecutionHostLookup( + repos: readonly T[] +): ExecutionHostOwnerLookup { + const rowsFor = (repoId: string): T[] => repos.filter((repo) => repo.id === repoId) + return { + byId: (repoId) => { + const rows = rowsFor(repoId) + if (rows.length === 0) { + return { kind: 'missing' } + } + const hostIds = new Set(rows.map((repo) => getRepoExecutionHostId(repo))) + const owner = rows[0] + return hostIds.size > 1 || !owner ? { kind: 'ambiguous' } : { kind: 'resolved', owner } + }, + byHost: (repoId, hostId) => + rowsFor(repoId).find((repo) => getRepoExecutionHostId(repo) === hostId) ?? null + } +} From 7c94d12190ed26181e8defa69cfd9e8c80202e07 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:59:29 -0700 Subject: [PATCH 18/77] fix(ssh): route four host-blind seams through the resolved execution host (#17919) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(host-routing): resolve the execution host before reading a connection Three issues in one defect class: a resolver reads one spelling of one arbitrarily chosen row instead of resolving the worktree's execution host, so something local answers a question about a remote. returned that row's connectionId. With duplicate repo rows for one repo id it could pair a runtime owner with a client-owned SSH connection. It now resolves through the same ambiguity-aware index getRuntimeEnvironmentIdForWorktree uses, prefers the repo row for the host the worktree names, and derives the connection from the resolved host. Conflicting rows return `undefined` (this module's documented "cannot determine the host"), never `null`. `store.getRepo(worktree.repoId)?.connectionId ?? null`. `getRepo` is host-blind and the same repo id can exist on local, SSH and runtime hosts, so a remote worktree could spawn its PTY on the client with the remote cwd. resolveWorktreeLaunchHost picks the row for the worktree's host and reads the connection off that host; conflicting rows are unresolved, not local. session-partition owner maps that contradict each other. Both now compute through one shared function whose argument records the divergence. No behaviour change on either side: converging needs a read-both migration, since both partitions hold real data written by shipping builds. * fix(host-routing): keep nested SSH connections resolvable under a runtime host getRepoSshConnectionId read only the resolved execution host, so a repo row owned by a runtime that reaches a nested SSH target (connectionId: ssh-*, executionHostId: runtime:*) resolved to no connection — answering 'local' for a remote worktree, the same defect #17909 fixed in the other direction. * fix(host-routing): resolve both sides of the execution host through one rule The renderer resolver leaked between two different SSH hosts: a worktree on `ssh:m4air` whose only indexed repo row belonged to `openclaw` answered 'openclaw', because the host-scoped lookup missing fell through to an id-only one. Main's resolver, in the same change, answered 'm4air' — two resolvers, one right and one wrong, on identical input. Both sides now adapt one shared rule (`worktree-execution-host-resolution.ts`): the worktree's own host outranks every repo row, and a row on a different host is never evidence about this one. The renderer's WeakMap index becomes the memoizing adapter it always was; `resolveWorktreeLaunchHost` becomes main's mapping of unresolved onto its throw. Settles the rule the change previously answered two ways. `getRepoSshConnectionId` and `getSshTargetIdForExecutionHost` disagreed for a runtime host carrying a nested `connectionId`; they now compose, so the execution host is the single authority. On a `runtime:*` row that field is a paired HUB's private SSH target, spread through by `repoWithFetchedOwner` and unaddressable from this client — the project-first successor of the row nulls it for exactly that reason. That also fixes the `kind !== 'ssh'` fallback, which fired for `local`: a row declaring itself local handed out an SSH connection. * fix(ssh): resolve the execution host in the worktree scan and managed create The worktree scan and createManagedWorktree both picked remote-vs-local from repo.connectionId, so a row stamped only executionHostId: 'ssh:*' was scanned and created on the client against a remote path. The folder branch returns before the check, so its agent-trust write landed locally too. Refs #11163 * fix(ssh): stop over-rejecting and refusing SSH hosts the process owns runtimeRepoMatchesExecutionHost rejected an unstamped SSH repo against its own ssh:, so repo-add/clone dedupe could register a second row for a path the host already owns. assertHostIsSupported made the CLI/runtime RPC refuse --host ssh:* while the same process's IPC handler routed it correctly; setupExistingFolder now shares that registration. Clone still refuses, because nothing in this process clones onto an SSH host. Refs #11163 * test(ssh): retarget the SSH host-setup guard spec at the substitution it prevents setupProjectExistingFolder now registers the remote path through the same addRemoteRepoFromPath the desktop IPC uses, so it fails on the host's terms (connection not registered) rather than a categorical refusal. The local clone/probe side effects it exists to catch are still asserted absent. Refs #11163 * fix(cli): require an absolute path when setting a project up on an SSH host Routing --host ssh:* to the remote registration made relative paths newly reachable there, and they were resolved against the client cwd — registering a path that names the wrong machine. Refs #11163 * fix(repos): read the SSH registry directly so the runtime stays Node-bootable Routing runtime project setup through addRemoteRepoFromPath dragged ipc/ssh -- and its 25-module electron graph -- into the runtime bundle. ssh-target-registry already exists for exactly this; ipc/ssh only re-exports it. * fix(ssh): close the agent-launch and session-export host-blind twins Three sites left on the legacy spelling, all the same shape as the ones this branch already fixed: - `launchAgentTerminal` did `getRepo(worktree.repoId)` then wrote agent trust with that row's `connectionId`. Host-blind, so a repo id carried by two SSH hosts wrote a remote path into the *client's* Codex/Cursor/Copilot config and the agent on the host never saw the trust. Every sibling call site already passes the resolved `workspace.connectionId`; this was the last that did not. - `targetForWorktree` (workspace-session export) fell back to the same host-blind read, so a session could be published to a machine that never owned the worktree. Unresolvable ownership now exports to nobody. - `addRemoteRepoFromPath` minted `connectionId`-only rows while being the routing path this branch adds, so it kept creating rows in exactly the spelling the branch works around. It now stamps `toSshExecutionHostId(connectionId)` at creation; `reassignSshTargetId` already migrates both spellings, so target rename stays correct. Tests cover two *different* SSH hosts throughout — the case none of the earlier duplicate-row tests had, all of which were local-vs-ssh or runtime-vs-ssh. --- src/cli/handlers/project.ts | 10 +- src/cli/index-project-setup.test.ts | 31 +++ .../ipc/remote-workspace-patch-queue.test.ts | 5 +- src/main/ipc/remote-workspace.test.ts | 2 + src/main/ipc/remote-workspace.ts | 20 +- src/main/ipc/repos/remote-home-path.ts | 2 +- .../repos/remote-repo-registration.test.ts | 124 ++++++++++ .../ipc/repos/remote-repo-registration.ts | 12 +- .../agent-terminal-launch-trust-host.test.ts | 107 +++++++++ ...ged-worktree-create-execution-host.test.ts | 155 +++++++++++++ .../orca-runtime-create-managed-worktree.ts | 33 ++- ...-resolved-worktrees-for-explicit-target.ts | 14 +- .../orca-runtime-preserved-branch-cleanup.ts | 13 ++ ...orca-runtime-refresh-repo-worktree-scan.ts | 16 +- ...a-runtime-terminal-create-deduplication.ts | 13 +- .../repository-project-operations.spec.ts | 9 +- ...time-project-host-setup-controller.test.ts | 120 ++++++++++ .../runtime-project-host-setup-controller.ts | 42 +++- .../runtime-worktree-selection.test.ts | 55 +++++ .../runtime/runtime-worktree-selection.ts | 13 +- ...rktree-scan-execution-host-routing.test.ts | 214 ++++++++++++++++++ 21 files changed, 954 insertions(+), 56 deletions(-) create mode 100644 src/main/ipc/repos/remote-repo-registration.test.ts create mode 100644 src/main/runtime/agent-terminal-launch-trust-host.test.ts create mode 100644 src/main/runtime/managed-worktree-create-execution-host.test.ts create mode 100644 src/main/runtime/runtime-project-host-setup-controller.test.ts create mode 100644 src/main/runtime/runtime-worktree-selection.test.ts create mode 100644 src/main/runtime/worktree-scan-execution-host-routing.test.ts diff --git a/src/cli/handlers/project.ts b/src/cli/handlers/project.ts index 2b0a8d8ada2..a7bf9ee2abe 100644 --- a/src/cli/handlers/project.ts +++ b/src/cli/handlers/project.ts @@ -10,7 +10,7 @@ import type { ProjectHostSetupUpdateArgs, ProjectHostSetupUpdateResult } from '../../shared/project-types' -import type { ExecutionHostId } from '../../shared/execution-host' +import { getSshTargetIdForExecutionHost, type ExecutionHostId } from '../../shared/execution-host' import type { RepoKind } from '../../shared/repo-types' import type { CommandHandler, HandlerContext } from '../dispatch' import { @@ -111,10 +111,14 @@ export const PROJECT_HANDLERS: Record = { }, 'project setup-existing-folder': async ({ flags, client, cwd, json }) => { const rawPath = getRequiredStringFlag(flags, 'path') + const hostId = getRequiredHostId(flags) + // An SSH host's filesystem is not the CLI's, so resolving a relative path against the client + // cwd would register a path that names the wrong machine. + const pathIsOffClient = client.isRemote || getSshTargetIdForExecutionHost(hostId) !== null const args: ProjectHostSetupExistingFolderArgs = { projectId: getRequiredStringFlag(flags, 'project'), - hostId: getRequiredHostId(flags), - path: resolveRepoPathArgument(rawPath, cwd, client.isRemote, 'Remote project setup'), + hostId, + path: resolveRepoPathArgument(rawPath, cwd, pathIsOffClient, 'Remote project setup'), kind: getOptionalRepoKind(flags), displayName: getOptionalStringFlag(flags, 'display-name') } diff --git a/src/cli/index-project-setup.test.ts b/src/cli/index-project-setup.test.ts index 068c246ba7c..f104e45c39a 100644 --- a/src/cli/index-project-setup.test.ts +++ b/src/cli/index-project-setup.test.ts @@ -481,6 +481,37 @@ describe('orca cli worktree awareness', () => { process.exitCode = priorExitCode }) + it('rejects SSH project setup relative paths, which name the client filesystem', async () => { + // A local CLI reaching an `ssh:*` host is still off-client: resolving `./orca` against the + // CLI cwd would register a path that exists on the wrong machine. + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + const errSpy = vi.spyOn(console, 'error').mockImplementation(() => {}) + const priorExitCode = process.exitCode + + await main( + [ + 'project', + 'setup-existing-folder', + '--project', + 'github:stablyai/orca', + '--host', + 'ssh:openclaw', + '--path', + './orca', + '--json' + ], + '/tmp/repo' + ) + + expect(callMock).not.toHaveBeenCalled() + expect([...logSpy.mock.calls, ...errSpy.mock.calls].flat().join('\n')).toContain( + 'Remote project setup requires --path to be an absolute path on the remote server.' + ) + expect(process.exitCode).toBe(1) + + process.exitCode = priorExitCode + }) + it('rejects remote repo.add relative paths instead of resolving against client cwd', async () => { const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) const errSpy = vi.spyOn(console, 'error').mockImplementation(() => {}) diff --git a/src/main/ipc/remote-workspace-patch-queue.test.ts b/src/main/ipc/remote-workspace-patch-queue.test.ts index 30ff0af7761..6b5d342b3f6 100644 --- a/src/main/ipc/remote-workspace-patch-queue.test.ts +++ b/src/main/ipc/remote-workspace-patch-queue.test.ts @@ -77,8 +77,11 @@ describe('remoteWorkspace:setForConnectedTargets patch queue', () => { const handlers = new Map unknown>() const muxByTargetId = new Map }>() const getRepoMock = vi.fn() + // Ownership resolution reads the catalog, not one id-keyed row, so the fake has to project one. + const KNOWN_REPO_IDS = ['repo-target-1', 'repo-target-2', 'repo-reset', 'repo-newer'] const store = { - getRepo: getRepoMock + getRepo: getRepoMock, + getRepos: () => KNOWN_REPO_IDS.map((repoId) => getRepoMock(repoId)).filter(Boolean) } as unknown as Store const target: SshTarget = { diff --git a/src/main/ipc/remote-workspace.test.ts b/src/main/ipc/remote-workspace.test.ts index 434f8c7dec4..56eb4804a82 100644 --- a/src/main/ipc/remote-workspace.test.ts +++ b/src/main/ipc/remote-workspace.test.ts @@ -153,8 +153,10 @@ describe('remoteWorkspace:setForConnectedTargets', () => { const muxByTargetId = new Map }>() const getRepoMock = vi.fn() const getWorkspaceSessionMock = vi.fn() + // Ownership resolution reads the catalog, not one id-keyed row, so the fake has to project one. const store = { getRepo: getRepoMock, + getRepos: () => [getRepoMock('repo-target-1')].filter(Boolean), getWorkspaceSession: getWorkspaceSessionMock } as unknown as Store diff --git a/src/main/ipc/remote-workspace.ts b/src/main/ipc/remote-workspace.ts index 74339e8f694..f9479f15ce9 100644 --- a/src/main/ipc/remote-workspace.ts +++ b/src/main/ipc/remote-workspace.ts @@ -12,7 +12,10 @@ import { } from '../../shared/remote-workspace-types' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { getRepoIdFromWorktreeId } from '../../shared/worktree/id' -import { parseExecutionHostId } from '../../shared/execution-host' +import { + createRepoRowExecutionHostLookup, + resolveWorktreeExecutionHost +} from '../../shared/worktree-execution-host-resolution' import { getRemoteWorkspaceNamespace } from './remote-workspace-namespace' import { registerRemoteWorkspaceNotificationHandler } from './remote-workspace-events' import { CLIENT_ID } from './remote-workspace-client-identity' @@ -108,12 +111,15 @@ function targetForWorktree( worktreeId: string, executionHostId?: string ): string | null { - const parsedHostId = parseExecutionHostId(executionHostId) - if (parsedHostId?.kind === 'ssh') { - return parsedHostId.targetId - } - const repoId = getRepoIdFromWorktreeId(worktreeId) - return store.getRepo(repoId)?.connectionId ?? null + // Why: this decides which SSH target a workspace session is exported to. The old fallback read + // `getRepo(id)?.connectionId`, which is host-blind — the same repo id can name rows on several + // hosts, so a session could be published to a machine that never owned the worktree (#11163). + // Unresolvable ownership exports to nobody rather than guessing. + const resolution = resolveWorktreeExecutionHost( + createRepoRowExecutionHostLookup(store.getRepos()), + { repoId: getRepoIdFromWorktreeId(worktreeId), hostId: executionHostId ?? null } + ) + return resolution.kind === 'resolved' ? resolution.connectionId : null } function exportSessionForTarget( diff --git a/src/main/ipc/repos/remote-home-path.ts b/src/main/ipc/repos/remote-home-path.ts index 2952764f872..602eed25684 100644 --- a/src/main/ipc/repos/remote-home-path.ts +++ b/src/main/ipc/repos/remote-home-path.ts @@ -1,4 +1,4 @@ -import { getActiveMultiplexer } from '../ssh' +import { getActiveMultiplexer } from '../../ssh/ssh-target-registry' export async function resolveRemoteHomePath(connectionId: string, path: string): Promise { if (path !== '~' && path !== '~/' && !path.startsWith('~/')) { diff --git a/src/main/ipc/repos/remote-repo-registration.test.ts b/src/main/ipc/repos/remote-repo-registration.test.ts new file mode 100644 index 00000000000..0a7069b1f63 --- /dev/null +++ b/src/main/ipc/repos/remote-repo-registration.test.ts @@ -0,0 +1,124 @@ +// Registration is now the runtime's SSH path too (`projectHostSetup.setupExistingFolder --host +// ssh:*`), so what it stamps decides what every downstream host resolver can read. It minted +// `connectionId`-only rows, leaving the unified spelling permanently empty, and deduped by raw +// `connectionId`, which cannot see a row stamped `executionHostId: 'ssh:*'`. +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../shared/repo-types' + +const getSshGitProviderMock = vi.hoisted(() => vi.fn()) +vi.mock('../../providers/ssh-git-dispatch', () => ({ + getSshGitProvider: getSshGitProviderMock +})) + +vi.mock('../../repo-icon-autodetect', () => ({ + detectRepoIconAndUpstream: vi.fn(async () => ({})) +})) + +vi.mock('../../ssh/ssh-target-registry', () => ({ + getActiveMultiplexer: vi.fn(() => null) +})) + +vi.mock('./remote-home-path', () => ({ + resolveRemoteHomePath: vi.fn(async (_connectionId: string, path: string) => path) +})) + +import { addRemoteRepoFromPath } from './remote-repo-registration' + +function makeStore(repos: Repo[]) { + return { + getRepos: () => repos, + getSshTarget: () => undefined, + addRepo: (repo: Repo) => { + repos.push(repo) + } + } +} + +describe('addRemoteRepoFromPath', () => { + beforeEach(() => { + getSshGitProviderMock.mockReset() + getSshGitProviderMock.mockReturnValue({ + isGitRepoAsync: vi.fn(async () => ({ isRepo: true, rootPath: '/srv/app' })) + }) + }) + + it('stamps the unified execution-host spelling alongside the legacy connection id', async () => { + const repos: Repo[] = [] + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'm4air', + remotePath: '/srv/app' + }) + + expect('error' in result).toBe(false) + const repo = (result as { repo: Repo }).repo + expect(repo.connectionId).toBe('m4air') + expect(repo.executionHostId).toBe('ssh:m4air') + }) + + it('dedupes against a row that names the host in the unified spelling only', async () => { + const existing = { + id: 'existing', + path: '/srv/app', + displayName: 'app', + badgeColor: '#000', + addedAt: 0, + executionHostId: 'ssh:m4air' + } as Repo + const repos: Repo[] = [existing] + + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'm4air', + remotePath: '/srv/app' + }) + + expect(result).toEqual({ repo: existing, alreadyExisted: true }) + expect(repos).toHaveLength(1) + }) + + it('does not dedupe onto a row on a different SSH host at the same path', async () => { + // Two hosts can both hold /srv/app. Matching on path alone registers one host's repo as the + // other's — the mirror image of the id-only lookup this change removes. + const repos: Repo[] = [ + { + id: 'openclaw-row', + path: '/srv/app', + displayName: 'app', + badgeColor: '#000', + addedAt: 0, + connectionId: 'openclaw' + } as Repo + ] + + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'm4air', + remotePath: '/srv/app' + }) + + expect((result as { alreadyExisted: boolean }).alreadyExisted).toBe(false) + expect((result as { repo: Repo }).repo.executionHostId).toBe('ssh:m4air') + expect(repos).toHaveLength(2) + }) + + it('does not dedupe onto a local row that carries a stale connection id', async () => { + // The pullfrog case: a row declaring itself local must not answer as an SSH host. + const repos: Repo[] = [ + { + id: 'local-row', + path: '/srv/app', + displayName: 'app', + badgeColor: '#000', + addedAt: 0, + executionHostId: 'local', + connectionId: 'develop' + } as Repo + ] + + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'develop', + remotePath: '/srv/app' + }) + + expect((result as { alreadyExisted: boolean }).alreadyExisted).toBe(false) + expect(repos).toHaveLength(2) + }) +}) diff --git a/src/main/ipc/repos/remote-repo-registration.ts b/src/main/ipc/repos/remote-repo-registration.ts index 3f3b94ff58b..35ab72056ae 100644 --- a/src/main/ipc/repos/remote-repo-registration.ts +++ b/src/main/ipc/repos/remote-repo-registration.ts @@ -3,9 +3,10 @@ import type { Store } from '../../persistence' import type { Repo } from '../../../shared/repo-types' import { DEFAULT_REPO_BADGE_COLOR } from '../../../shared/constants' import { normalizeRuntimePathForComparison } from '../../../shared/cross-platform-path' +import { getRepoSshConnectionId, toSshExecutionHostId } from '../../../shared/execution-host' import { getSshGitProvider } from '../../providers/ssh-git-dispatch' import { detectRepoIconAndUpstream } from '../../repo-icon-autodetect' -import { getActiveMultiplexer } from '../ssh' +import { getActiveMultiplexer } from '../../ssh/ssh-target-registry' import { resolveRemoteHomePath } from './remote-home-path' export async function addRemoteRepoFromPath( @@ -26,11 +27,13 @@ export async function addRemoteRepoFromPath( let repoKind: 'git' | 'folder' = args.kind ?? 'git' let resolvedPath = await resolveRemoteHomePath(args.connectionId, args.remotePath) + // Resolve the host: a row stamped only `executionHostId: 'ssh:*'` is the same registration, and + // missing it here registers a duplicate repo for a path the host already owns. const existing = store .getRepos() .find( (repo) => - repo.connectionId === args.connectionId && + getRepoSshConnectionId(repo) === args.connectionId && normalizeRuntimePathForComparison(repo.path) === normalizeRuntimePathForComparison(resolvedPath) ) @@ -61,7 +64,7 @@ export async function addRemoteRepoFromPath( .getRepos() .find( (repo) => - repo.connectionId === args.connectionId && + getRepoSshConnectionId(repo) === args.connectionId && normalizeRuntimePathForComparison(repo.path) === normalizeRuntimePathForComparison(resolvedPath) ) @@ -92,6 +95,9 @@ export async function addRemoteRepoFromPath( addedAt: Date.now(), kind: repoKind, connectionId: args.connectionId, + // Stamp the unified spelling at creation: this is now the runtime's SSH registration path too, + // and minting `connectionId`-only rows leaves every host-resolving reader on the legacy field. + executionHostId: toSshExecutionHostId(args.connectionId), ...(repoKind === 'git' ? { externalWorktreeVisibilityLegacy: false, diff --git a/src/main/runtime/agent-terminal-launch-trust-host.test.ts b/src/main/runtime/agent-terminal-launch-trust-host.test.ts new file mode 100644 index 00000000000..949b0c545fc --- /dev/null +++ b/src/main/runtime/agent-terminal-launch-trust-host.test.ts @@ -0,0 +1,107 @@ +// launchAgentTerminal read `store.getRepo(worktree.repoId)?.connectionId` for the trust write — +// host-blind, so the same repo id on two hosts wrote a remote path into the client's agent config +// and the agent on the host never saw the trust (#11163). Every sibling call site already passes +// the resolved `workspace.connectionId`; this was the last one that did not. +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } +})) + +import { OrcaRuntimeService } from './orca-runtime' + +const REMOTE_PATH = '/srv/app-feature' + +type RuntimeInternals = { + resolveWorktreeSelector: (selector: string) => Promise + buildStartupForAgent: (repo: unknown, agent: unknown, prompt: string) => unknown + markWorkspaceTrustedForAgent: ( + agent: unknown, + connectionId: string | null | undefined, + path: string + ) => Promise + createTerminal: (selector: string, opts: unknown) => Promise +} + +function makeRuntime(repos: readonly Record[], hostId?: string) { + const store = { + getSettings: () => ({ disabledTuiAgents: [], workspaceDir: '/tmp/workspaces' }), + getProjectHostSetups: () => [], + getRepos: () => repos, + getRepo: (id: string) => repos.find((repo) => repo.id === id) + } + const runtime = new OrcaRuntimeService(store as never) + const internals = runtime as unknown as RuntimeInternals + vi.spyOn(internals, 'resolveWorktreeSelector').mockResolvedValue({ + id: 'repo-shared::/srv/app-feature', + repoId: 'repo-shared', + path: REMOTE_PATH, + ...(hostId ? { hostId } : {}) + }) + vi.spyOn(internals, 'buildStartupForAgent').mockReturnValue({ + agent: 'codex', + startup: { command: 'codex', env: {}, startupCommandDelivery: 'none', telemetry: {} } + }) + const markTrusted = vi.fn(async () => {}) + vi.spyOn(internals, 'markWorkspaceTrustedForAgent').mockImplementation(markTrusted) + vi.spyOn(internals, 'createTerminal').mockResolvedValue({ id: 'pty-1' }) + return { runtime, markTrusted } +} + +describe('launchAgentTerminal trust write', () => { + beforeEach(() => { + vi.restoreAllMocks() + }) + + it('writes trust on the host the worktree names, not on a rival row', async () => { + // Two SSH hosts publish the same repo id; the worktree is on m4air. + const { runtime, markTrusted } = makeRuntime( + [ + { id: 'repo-shared', path: '/home/me/app', connectionId: 'openclaw' }, + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' } + ], + 'ssh:m4air' + ) + + await runtime.launchAgentTerminal('id:repo-shared::/srv/app-feature', { + agent: 'codex', + prompt: 'go' + } as never) + + expect(markTrusted).toHaveBeenCalledWith('codex', 'm4air', REMOTE_PATH) + }) + + it('writes trust locally for a local worktree even when a remote row shares the id', async () => { + const { runtime, markTrusted } = makeRuntime( + [ + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }, + { id: 'repo-shared', path: '/home/me/app' } + ], + 'local' + ) + + await runtime.launchAgentTerminal('id:repo-shared::/srv/app-feature', { + agent: 'codex', + prompt: 'go' + } as never) + + expect(markTrusted).toHaveBeenCalledWith('codex', null, REMOTE_PATH) + }) + + it('refuses rather than guessing when rival rows disagree and the worktree names no host', async () => { + const { runtime } = makeRuntime([ + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }, + { id: 'repo-shared', path: '/home/me/app' } + ]) + + await expect( + runtime.launchAgentTerminal('id:repo-shared::/srv/app-feature', { + agent: 'codex', + prompt: 'go' + } as never) + ).rejects.toThrow('worktree_execution_host_unresolved') + }) +}) diff --git a/src/main/runtime/managed-worktree-create-execution-host.test.ts b/src/main/runtime/managed-worktree-create-execution-host.test.ts new file mode 100644 index 00000000000..af1a73dc9cc --- /dev/null +++ b/src/main/runtime/managed-worktree-create-execution-host.test.ts @@ -0,0 +1,155 @@ +// createManagedWorktree used to pick remote-vs-local from the raw `connectionId` field, so a repo +// stamped only `executionHostId: 'ssh:*'` ran `git worktree add` on the client against a remote +// path — and the folder branch, which returns before that check, wrote agent trust locally for a +// remote workspace. Both are the #11163 shape: read the execution host, never one spelling of it. +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } +})) + +const createRuntimeFolderWorktreeMock = vi.hoisted(() => vi.fn()) +vi.mock('./runtime-folder-worktree-create', () => ({ + createRuntimeFolderWorktree: createRuntimeFolderWorktreeMock +})) + +const createRuntimeLocalManagedWorktreeMock = vi.hoisted(() => vi.fn()) +vi.mock('./runtime-local-worktree-create', () => ({ + createRuntimeLocalManagedWorktree: createRuntimeLocalManagedWorktreeMock +})) + +const trustMocks = vi.hoisted(() => ({ + local: vi.fn(async () => {}), + remote: vi.fn(async () => {}) +})) +vi.mock('./runtime-worktree-agent-startup', async (importOriginal) => ({ + ...(await importOriginal>()), + markLocalWorktreeTrusted: trustMocks.local, + markRemoteWorktreeTrusted: trustMocks.remote +})) + +import { OrcaRuntimeService } from './orca-runtime' + +const TARGET_ID = 'remote-1' +const REMOTE_PATH = '/srv/app' + +type RuntimeInternals = { + resolveRepoSelector: (selector: string) => Promise + createManagedRemoteWorktree: (repo: unknown, args: unknown) => Promise + resolveLineageForWorktreeCreate: (input: unknown) => Promise + recordCreatedWorktreeLineage: (worktree: unknown, resolution: unknown) => unknown +} + +function makeRuntime(repo: Record): { + runtime: OrcaRuntimeService + createRemote: ReturnType +} { + const store = { + getSettings: () => ({ disabledTuiAgents: [], workspaceDir: '/tmp/workspaces' }), + getProjectHostSetups: () => [] + } + const runtime = new OrcaRuntimeService(store as never) + const internals = runtime as unknown as RuntimeInternals + vi.spyOn(internals, 'resolveRepoSelector').mockResolvedValue(repo) + vi.spyOn(internals, 'resolveLineageForWorktreeCreate').mockResolvedValue(null) + vi.spyOn(internals, 'recordCreatedWorktreeLineage').mockReturnValue({ + lineage: null, + workspaceLineage: null, + warnings: [] + }) + const createRemote = vi.fn().mockResolvedValue({ + worktree: { id: 'wt-1', path: '/srv/app-feature', branch: 'feature' } + }) + vi.spyOn(internals, 'createManagedRemoteWorktree').mockImplementation(createRemote) + return { runtime, createRemote } +} + +describe('createManagedWorktree execution-host routing', () => { + beforeEach(() => { + createRuntimeFolderWorktreeMock.mockReset() + createRuntimeFolderWorktreeMock.mockResolvedValue({ worktree: { id: 'folder-1' } }) + createRuntimeLocalManagedWorktreeMock.mockReset() + // Name the defect in the failure output: reaching this mock means a remote repo was routed + // into a client-side `git worktree add`. + createRuntimeLocalManagedWorktreeMock.mockRejectedValue( + new Error('local_worktree_create_ran_for_remote_repo') + ) + trustMocks.local.mockClear() + trustMocks.remote.mockClear() + }) + + it('creates on the SSH host for a repo stamped executionHostId only', async () => { + const { runtime, createRemote } = makeRuntime({ + id: 'repo-remote', + path: REMOTE_PATH, + kind: 'git', + executionHostId: `ssh:${TARGET_ID}` + }) + + await runtime.createManagedWorktree({ repoSelector: 'repo-remote', name: 'feature' } as never) + + // A local `git worktree add` against a remote path is the silent-substitution failure. + expect(createRuntimeLocalManagedWorktreeMock).not.toHaveBeenCalled() + expect(createRemote).toHaveBeenCalledWith( + expect.objectContaining({ id: 'repo-remote', connectionId: TARGET_ID }), + expect.anything() + ) + }) + + it('still creates on the SSH host for a legacy connectionId-only repo', async () => { + const { runtime, createRemote } = makeRuntime({ + id: 'repo-remote', + path: REMOTE_PATH, + kind: 'git', + connectionId: TARGET_ID + }) + + await runtime.createManagedWorktree({ repoSelector: 'repo-remote', name: 'feature' } as never) + + expect(createRuntimeLocalManagedWorktreeMock).not.toHaveBeenCalled() + expect(createRemote).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: TARGET_ID }), + expect.anything() + ) + }) + + it('marks a folder workspace trusted on its SSH host, not on the client', async () => { + const { runtime } = makeRuntime({ + id: 'repo-folder', + path: REMOTE_PATH, + kind: 'folder', + connectionId: TARGET_ID, + executionHostId: `ssh:${TARGET_ID}` + }) + + await runtime.createManagedWorktree({ repoSelector: 'repo-folder', name: 'notes' } as never) + + const deps = createRuntimeFolderWorktreeMock.mock.calls[0]?.[0]?.deps + await deps.markTrusted('codex', '/srv/app') + + expect(trustMocks.remote).toHaveBeenCalledWith('codex', TARGET_ID, '/srv/app') + expect(trustMocks.local).not.toHaveBeenCalled() + }) + + it('keeps a local folder workspace trusted on the client', async () => { + const { runtime } = makeRuntime({ + id: 'repo-folder-local', + path: '/Users/me/notes', + kind: 'folder' + }) + + await runtime.createManagedWorktree({ + repoSelector: 'repo-folder-local', + name: 'notes' + } as never) + + const deps = createRuntimeFolderWorktreeMock.mock.calls[0]?.[0]?.deps + await deps.markTrusted('codex', '/Users/me/notes') + + expect(trustMocks.local).toHaveBeenCalledWith('codex', '/Users/me/notes') + expect(trustMocks.remote).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/runtime/orca-runtime-create-managed-worktree.ts b/src/main/runtime/orca-runtime-create-managed-worktree.ts index a6f8ebfbd79..f9116f73405 100644 --- a/src/main/runtime/orca-runtime-create-managed-worktree.ts +++ b/src/main/runtime/orca-runtime-create-managed-worktree.ts @@ -4,6 +4,7 @@ import type { RuntimeManagedWorktreeCreateArgs } from './runtime-managed-worktre import type { CreateWorktreeResult } from '../../shared/worktree/create-types' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' import { isFolderRepo } from '../../shared/repo-kind' +import { getRepoSshConnectionId } from '../../shared/execution-host' import { createRuntimeFolderWorktree } from './runtime-folder-worktree-create' import { createRuntimeLocalManagedWorktree } from './runtime-local-worktree-create' import { prepareRuntimeLocalWorktreeSetup } from './runtime-local-worktree-setup' @@ -56,7 +57,13 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork draftStartup?.agent ?? (requestedAgentEnabled ? requestedAgent : undefined)) const effectiveDraftPaste = args.startupDraftPaste ?? draftStartup?.draftPaste + // Resolve the execution host once: SSH ownership has two spellings, and reading the raw + // `connectionId` field routes an `executionHostId: 'ssh:*'`-only repo down the local path, + // which runs `git worktree add` on the client against a remote path. + const sshConnectionId = getRepoSshConnectionId(repo) if (isFolderRepo(repo)) { + // A folder workspace is a registration, not a filesystem create, so it is host-agnostic — + // except for the agent trust write, which must land on the host that will run the agent. return createRuntimeFolderWorktree({ request: args, repo, @@ -68,7 +75,8 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork store: this.store, ptySpawnAvailable: Boolean(this.ptyController?.spawn), createTerminal: (selector, options) => this.createTerminal(selector, options), - markTrusted: (agent, path) => this.markLocalWorkspaceTrustedForAgent(agent, path), + markTrusted: (agent, path) => + this.markWorkspaceTrustedForAgent(agent, sshConnectionId, path), pasteDraft: (handle, draft) => this.pasteStartupDraftWhenReady(handle, draft), sendFollowup: (handle, followup) => this.sendStartupFollowupWhenReady(handle, followup), invalidateResolvedWorktrees: () => this.invalidateResolvedWorktreeCache(), @@ -89,15 +97,20 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork const lineageInput = args.lineage || args.comment ? { ...args.lineage, comment: args.comment } : undefined const lineageResolution = await this.resolveLineageForWorktreeCreate(lineageInput) - if (repo.connectionId) { - const result = await this.createManagedRemoteWorktree(repo, { - ...args, - activate: args.activate, - ...(effectiveStartup ? { startup: effectiveStartup } : {}), - ...(effectiveStartupFollowup ? { startupFollowup: effectiveStartupFollowup } : {}), - ...(effectiveCreatedWithAgent ? { createdWithAgent: effectiveCreatedWithAgent } : {}), - ...(effectiveDraftPaste ? { startupDraftPaste: effectiveDraftPaste } : {}) - }) + if (sshConnectionId) { + // Why normalize the row: the remote-create pipeline reads `repo.connectionId!` at every + // depth, so hand it the connection the resolved host actually names. + const result = await this.createManagedRemoteWorktree( + { ...repo, connectionId: sshConnectionId }, + { + ...args, + activate: args.activate, + ...(effectiveStartup ? { startup: effectiveStartup } : {}), + ...(effectiveStartupFollowup ? { startupFollowup: effectiveStartupFollowup } : {}), + ...(effectiveCreatedWithAgent ? { createdWithAgent: effectiveCreatedWithAgent } : {}), + ...(effectiveDraftPaste ? { startupDraftPaste: effectiveDraftPaste } : {}) + } + ) const recordedLineage = this.recordCreatedWorktreeLineage(result.worktree, lineageResolution) this.emitWorktreeLifecycle({ kind: 'created', diff --git a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts index f8632bb2643..1f0dbe591a3 100644 --- a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts +++ b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts @@ -19,7 +19,7 @@ import type { Repo } from '../../shared/repo-types' import type { ProjectExecutionRuntimeResolution } from '../../shared/project-execution-runtime' import type { RuntimeWorktreeScanResult } from './repo-worktree-resolution-scan' import { getSshGitProviderGeneration } from '../providers/ssh-git-dispatch' -import { getRepoExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, getRepoSshConnectionId } from '../../shared/execution-host' import type { RuntimeWorktreeScanCache } from './orca-runtime-core' import { resolveWorktreeScanCacheTtlMs } from './runtime-worktree-scan-cache' @@ -135,17 +135,21 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends repo: Repo, projectRuntimeByRepoId?: ReadonlyMap ): Promise { + // Resolve the execution host, not the raw field: an `executionHostId: 'ssh:*'` row with no + // `connectionId` would otherwise get a local project runtime and a `local:default` cache key, + // so its scan neither routes remotely nor re-runs when the SSH provider is replaced. + const sshConnectionId = getRepoSshConnectionId(repo) const projectRuntime = projectRuntimeByRepoId ? projectRuntimeByRepoId.get(repo.id) - : !repo.connectionId + : !sshConnectionId ? resolveLocalProjectRuntimeForRepo(this.requireStore(), repo) : undefined const runtimeKey = projectRuntime ? projectRuntime.status === 'resolved' ? projectRuntime.runtime.cacheKey : projectRuntime.repair.cacheKey - : repo.connectionId - ? `ssh:${repo.connectionId}:${getSshGitProviderGeneration(repo.connectionId)}` + : sshConnectionId + ? `ssh:${sshConnectionId}:${getSshGitProviderGeneration(sshConnectionId)}` : 'local:default' const now = Date.now() const scanScopeKey = `${repo.id}\0${getRepoExecutionHostId(repo)}` @@ -176,7 +180,7 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends return this.listRepoWorktreesForResolution(repo, projectRuntimeByRepoId) } if ( - (refresh.result.ok || !repo.connectionId) && + (refresh.result.ok || !sshConnectionId) && this.worktreeScanInFlight.get(scanScopeKey)?.promise === promise ) { const entry: RuntimeWorktreeScanCache = { diff --git a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts index 43853f0f9c0..810ffeca3be 100644 --- a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts +++ b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts @@ -26,6 +26,8 @@ import { RuntimeAccountController } from './runtime-account-controller' import { RuntimeMobileSpeechCatalog } from './runtime-mobile-speech-catalog' import { RuntimeMobileDictationController } from './runtime-mobile-dictation-controller' import { RuntimeProjectHostSetupController } from './runtime-project-host-setup-controller' +import { addRemoteRepoFromPath } from '../ipc/repos/remote-repo-registration' +import type { Store } from '../persistence' import { RuntimeProjectGroupController } from './runtime-project-group-controller' import { RuntimeNestedRepoImport } from './runtime-nested-repo-import' import { RuntimeRepositoryRegistrationController } from './runtime-repository-registration-controller' @@ -194,6 +196,17 @@ export class OrcaRuntimeWithPreservedBranchCleanup extends OrcaRuntimeWithTermin listRepos: () => this.listRepos(), addRepo: (path, kind, hostId) => (this as RuntimeCommandSurfaceHost).addRepo(path, kind, hostId), + addRemoteRepo: async (remote) => { + // The same registration the desktop IPC handler uses, so both surfaces agree on SSH hosts. + const result = await addRemoteRepoFromPath(this.requireStore() as unknown as Store, remote) + if ('error' in result) { + throw new Error(result.error) + } + this.invalidateResolvedWorktreeCache() + this.invalidateWorktreeScanCacheForRepo(result.repo.id) + this.notifyReposChanged() + return result.repo + }, cloneRepo: (url, destination, hostId) => (this as RuntimeCommandSurfaceHost).cloneRepo(url, destination, hostId), invalidateResolvedWorktrees: () => this.invalidateResolvedWorktreeCache(), diff --git a/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts b/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts index 5e3282c5e53..1964b5fdd2f 100644 --- a/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts +++ b/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts @@ -17,7 +17,7 @@ import { getSshGitProvider } from '../providers/ssh-git-dispatch' import type { GitWorktreeInfo } from '../../shared/worktree/types' import { listStoredWorktreeRowsForRepo } from './repo-worktree-row-resolution' import type { ResolvedWorktree } from './runtime-worktree-path-identity' -import { getRepoExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, getRepoSshConnectionId } from '../../shared/execution-host' export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget { /** @@ -31,8 +31,10 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK ): Promise { const scannedAt = Date.now() // SSH and WSL-routed repos run Git off-host, so a local admin-dir read cannot describe them. + // Resolve the execution host rather than reading `connectionId`: a row stamped only + // `executionHostId: 'ssh:*'` is just as off-host, and fingerprinting it stats client paths. const fingerprintCapable = - !repo.connectionId && + !getRepoSshConnectionId(repo) && // Why: a repo whose scan TTL already reaches the reconciliation interval can never reuse a // fingerprint, so reading one would be pure work. Agent-scratch roots are that case today. resolveWorktreeScanCacheTtlMs(repo) < WORKTREE_SCAN_ADMIN_RECONCILE_INTERVAL_MS && @@ -92,13 +94,17 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK repo: Repo, projectRuntime: ProjectExecutionRuntimeResolution | undefined ): Promise { - if (!repo.connectionId) { + // Why not `repo.connectionId`: SSH ownership has two spellings, and a repo carrying only + // `executionHostId: 'ssh:*'` would otherwise be scanned on the client against a remote path — + // `git worktree list` then reports nothing, so the remote worktrees never resolve at all. + const sshConnectionId = getRepoSshConnectionId(repo) + if (!sshConnectionId) { return await scanLocalRepoWorktreesForResolution( repo.path, getLocalProjectWorktreeGitOptionsForRuntime(repo, projectRuntime) ) } - const provider = getSshGitProvider(repo.connectionId) + const provider = getSshGitProvider(sshConnectionId) if (!provider) { return { ok: false, worktrees: this.listStoredWorktreesForResolution(repo) } } @@ -149,7 +155,7 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK protected invalidateSshWorktreeScanCacheInternal(targetId: string): void { const repos = this.store?.getRepos() ?? [] - const affectedRepos = repos.filter((repo) => repo.connectionId === targetId) + const affectedRepos = repos.filter((repo) => getRepoSshConnectionId(repo) === targetId) const affectedScopeKeys = new Set( affectedRepos.map((repo) => `${repo.id}\0${getRepoExecutionHostId(repo)}`) ) diff --git a/src/main/runtime/orca-runtime-terminal-create-deduplication.ts b/src/main/runtime/orca-runtime-terminal-create-deduplication.ts index a1743a855ce..cdb24994fce 100644 --- a/src/main/runtime/orca-runtime-terminal-create-deduplication.ts +++ b/src/main/runtime/orca-runtime-terminal-create-deduplication.ts @@ -8,6 +8,7 @@ import { PTY_CONTROLLER_LIST_TIMEOUT_MS } from './orca-runtime-postlude' import { inferWorktreeIdFromPtyId } from './runtime-worktree-path-identity' import { getRegisteredSshState } from '../ssh/ssh-target-registry' import { LOCAL_EXECUTION_HOST_ID, toSshExecutionHostId } from '../../shared/execution-host' +import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' import type { TuiAgent } from '../../shared/tui-agent' export class OrcaRuntimeWithTerminalCreateDeduplication extends OrcaRuntimeWithCreateAgentSession { @@ -143,12 +144,20 @@ export class OrcaRuntimeWithTerminalCreateDeduplication extends OrcaRuntimeWithC opts: { agent: TuiAgent; prompt: string; title?: string } ): Promise { const worktree = await this.resolveWorktreeSelector(worktreeSelector) - const repo = this.store?.getRepo(worktree.repoId) + // Why: the trust write lands in an agent's config on the machine that runs it, keyed by the + // workspace path. `getRepo(id)` is host-blind, so reading `connectionId` off it wrote a remote + // path into the *client's* config — the agent on the host never sees the trust (#11163). + // Same shape as the folder-create trust write fixed alongside this; the agent-launch half. + const resolution = resolveWorktreeLaunchHost(this.store?.getRepos() ?? [], worktree) + if (resolution.kind === 'ambiguous') { + throw new Error('worktree_execution_host_unresolved') + } + const repo = resolution.repo ?? this.store?.getRepo(worktree.repoId) if (!repo) { throw new Error('Repository for the selected workspace is no longer available.') } const startup = this.buildStartupForAgent(repo, opts.agent, opts.prompt) - await this.markWorkspaceTrustedForAgent(opts.agent, repo.connectionId, worktree.path) + await this.markWorkspaceTrustedForAgent(opts.agent, resolution.connectionId, worktree.path) return await this.createTerminal(`id:${worktree.id}`, { command: startup.startup.command, env: startup.startup.env, diff --git a/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts b/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts index afe3dded381..c9879bf3d7c 100644 --- a/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts +++ b/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts @@ -588,7 +588,7 @@ describe('OrcaRuntimeService', () => { } }) - it('refuses SSH hosts instead of setting the project up on the local machine', async () => { + it('never sets an SSH-hosted project up on the local machine', async () => { // Why: both inputs must be paths the pre-guard code would have accepted. An unwritable // destination fails at mkdir and a non-repo path fails at isGitRepo, which would leave the // side-effect assertions below unable to observe the local clone/probe they exist to catch. @@ -639,11 +639,14 @@ describe('OrcaRuntimeService', () => { // first so a regression reports the corruption rather than stopping at the first throw. expect(spawnSpy).not.toHaveBeenCalled() expect(repos).toHaveLength(0) + // Cloning onto an SSH host has no implementation here, so it still refuses outright. expect(cloneError).toMatchObject({ - message: expect.stringMatching(/SSH hosts are not supported/) + message: expect.stringMatching(/Cloning onto an SSH host is not supported/) }) + // Registering an existing remote path does have one, so this now fails on the host's own + // terms — the SSH connection is not registered — rather than on a categorical refusal. expect(existingFolderError).toMatchObject({ - message: expect.stringMatching(/SSH hosts are not supported/) + message: expect.stringMatching(/SSH connection "openclaw" not found or not connected/) }) } finally { spawnSpy.mockRestore() diff --git a/src/main/runtime/runtime-project-host-setup-controller.test.ts b/src/main/runtime/runtime-project-host-setup-controller.test.ts new file mode 100644 index 00000000000..bb3d928d406 --- /dev/null +++ b/src/main/runtime/runtime-project-host-setup-controller.test.ts @@ -0,0 +1,120 @@ +// The CLI/runtime RPC used to refuse `--host ssh:*` with "set the project up from the Orca desktop +// app" — while the desktop IPC handler in the *same process* routed it correctly through +// addRemoteRepoFromPath. Safe but wrong: the process refusing is the one that owns the connection. +import { describe, expect, it, vi } from 'vitest' +import { RuntimeProjectHostSetupController } from './runtime-project-host-setup-controller' +import { getProjectHostSetupForRepo } from '../../shared/project-host-setup-lookup' +import { projectHostSetupProjectionFromRepos } from '../../shared/project-host-setup-projection' +import type { Repo } from '../../shared/repo-types' + +const TARGET_ID = 'target-1' +const REMOTE_PATH = '/srv/app' + +const remoteRepo = { + id: 'repo-remote', + path: REMOTE_PATH, + displayName: 'app', + badgeColor: 'blue', + addedAt: 1, + kind: 'git', + connectionId: TARGET_ID +} as unknown as Repo + +function makeController(): { + controller: RuntimeProjectHostSetupController + addRepo: ReturnType + addRemoteRepo: ReturnType + cloneRepo: ReturnType + projectId: string +} { + const store = { + getProjects: () => projectHostSetupProjectionFromRepos([remoteRepo]).projects, + getProjectHostSetups: () => [], + updateRepo: (_id: string, updates: Record) => ({ ...remoteRepo, ...updates }) + } + const addRepo = vi.fn().mockResolvedValue(remoteRepo) + const addRemoteRepo = vi.fn().mockResolvedValue(remoteRepo) + const cloneRepo = vi.fn().mockResolvedValue(remoteRepo) + const controller = new RuntimeProjectHostSetupController({ + getStore: () => store as never, + listRepos: () => [remoteRepo], + addRepo, + addRemoteRepo, + cloneRepo, + invalidateResolvedWorktrees: vi.fn(), + invalidateWorktreeScan: vi.fn(), + notifyReposChanged: vi.fn() + }) + return { + controller, + addRepo, + addRemoteRepo, + cloneRepo, + projectId: getProjectHostSetupForRepo([], remoteRepo).projectId + } +} + +describe('RuntimeProjectHostSetupController host routing', () => { + it('registers an existing folder on an SSH host instead of refusing it (#11163)', async () => { + const { controller, addRepo, addRemoteRepo, projectId } = makeController() + + const result = await controller.setupExistingFolder({ + projectId, + hostId: `ssh:${TARGET_ID}`, + path: REMOTE_PATH, + kind: 'git' + }) + + expect(addRemoteRepo).toHaveBeenCalledWith({ + connectionId: TARGET_ID, + remotePath: REMOTE_PATH, + kind: 'git' + }) + // The local registration path validates the path against the client filesystem. + expect(addRepo).not.toHaveBeenCalled() + expect(result.repo.id).toBe(remoteRepo.id) + }) + + it('decodes a percent-encoded SSH target back to its connection id', async () => { + const { controller, addRemoteRepo, projectId } = makeController() + + await controller.setupExistingFolder({ + projectId, + hostId: 'ssh:my%20host', + path: REMOTE_PATH, + kind: 'folder' + }) + + expect(addRemoteRepo).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: 'my host', kind: 'folder' }) + ) + }) + + it('still uses the local registration for local and runtime hosts', async () => { + const { controller, addRepo, addRemoteRepo, projectId } = makeController() + + await controller.setupExistingFolder({ + projectId, + hostId: 'local', + path: REMOTE_PATH, + kind: 'git' + }) + + expect(addRepo).toHaveBeenCalledWith(REMOTE_PATH, 'git', 'local') + expect(addRemoteRepo).not.toHaveBeenCalled() + }) + + it('refuses to clone onto an SSH host, because nothing here clones remotely', async () => { + const { controller, cloneRepo, projectId } = makeController() + + await expect( + controller.setupClone({ + projectId, + hostId: `ssh:${TARGET_ID}`, + url: 'https://example.com/app.git', + destination: REMOTE_PATH + }) + ).rejects.toThrow(/Cloning onto an SSH host is not supported/) + expect(cloneRepo).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/runtime/runtime-project-host-setup-controller.ts b/src/main/runtime/runtime-project-host-setup-controller.ts index 28d8b297480..cdc3ab1c60f 100644 --- a/src/main/runtime/runtime-project-host-setup-controller.ts +++ b/src/main/runtime/runtime-project-host-setup-controller.ts @@ -13,7 +13,11 @@ import type { ProjectUpdateArgs } from '../../shared/project-types' import type { Repo } from '../../shared/repo-types' -import { parseExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' +import { + getSshTargetIdForExecutionHost, + parseExecutionHostId, + type ExecutionHostId +} from '../../shared/execution-host' import { getProjectIdForProviderIdentity } from '../../shared/project-host-setup-projection' import { getProjectHostSetupForRepo } from '../../shared/project-host-setup-lookup' import { invalidateAuthorizedRootsCache } from '../ipc/filesystem-auth' @@ -24,18 +28,29 @@ type RuntimeProjectHostSetupDependencies = { getStore: () => RuntimeStore | null listRepos: () => Repo[] addRepo: (path: string, kind: 'folder' | 'git', hostId: ExecutionHostId) => Promise + /** Register an existing path that lives on an SSH host; `addRepo` only reaches local/runtime hosts. */ + addRemoteRepo: (args: { + connectionId: string + remotePath: string + displayName?: string + kind: 'folder' | 'git' + }) => Promise cloneRepo: (url: string, destination: string, hostId: ExecutionHostId) => Promise invalidateResolvedWorktrees: () => void invalidateWorktreeScan: (repoId: string) => void notifyReposChanged: () => void } -function assertHostIsSupported(hostId: ExecutionHostId | null | undefined): void { +// Why clone alone still refuses: nothing in this process clones onto an SSH host. `cloneRepo` runs +// `git clone` on the client, so accepting `ssh:*` here would register the client's copy as the +// host's repo — a local answer to a remote question. Registering an existing remote path, by +// contrast, has a correct implementation this process already uses over IPC. +function assertCloneHostIsSupported(hostId: ExecutionHostId | null | undefined): void { if (parseExecutionHostId(hostId)?.kind !== 'ssh') { return } throw new Error( - 'SSH hosts are not supported by this operation. Set the project up from the Orca desktop app, which owns the SSH connection.' + 'Cloning onto an SSH host is not supported. Clone the repository on the host, then set the project up from that existing folder.' ) } @@ -82,18 +97,25 @@ export class RuntimeProjectHostSetupController { if (!this.deps.getStore()) { throw new Error('runtime_unavailable') } - assertHostIsSupported(args.hostId) + const kind = args.kind === 'folder' ? 'folder' : 'git' const knownRepoIds = new Set(this.deps.listRepos().map((repo) => repo.id)) - const repo = await this.deps.addRepo( - args.path, - args.kind === 'folder' ? 'folder' : 'git', - args.hostId - ) + // Why route rather than refuse: this process owns the SSH connection, and its own IPC handler + // already registers `ssh:*` hosts correctly. Refusing here only made the CLI and runtime RPC + // disagree with the desktop app about what the same process can do. + const sshTargetId = getSshTargetIdForExecutionHost(args.hostId) + const repo = sshTargetId + ? await this.deps.addRemoteRepo({ + connectionId: sshTargetId, + remotePath: args.path, + ...(args.displayName ? { displayName: args.displayName } : {}), + kind + }) + : await this.deps.addRepo(args.path, kind, args.hostId) return this.completeSetup(args, repo, !knownRepoIds.has(repo.id)) } async setupClone(args: ProjectHostSetupCloneArgs): Promise { - assertHostIsSupported(args.hostId) + assertCloneHostIsSupported(args.hostId) const knownRepoIds = new Set(this.deps.listRepos().map((repo) => repo.id)) const repo = await this.deps.cloneRepo(args.url, args.destination, args.hostId) return this.completeSetup( diff --git a/src/main/runtime/runtime-worktree-selection.test.ts b/src/main/runtime/runtime-worktree-selection.test.ts new file mode 100644 index 00000000000..0509d94a2b4 --- /dev/null +++ b/src/main/runtime/runtime-worktree-selection.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from 'vitest' +import { runtimeRepoMatchesExecutionHost } from './runtime-worktree-selection' + +describe('runtimeRepoMatchesExecutionHost', () => { + it('matches an unstamped SSH repo against its own host (#11163)', () => { + // The row spells its ownership as `connectionId`; the request spells it as `ssh:`. + // Rejecting it here makes repo-add/clone dedupe register a second row for the same path. + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'ssh:target-1')).toBe(true) + }) + + it('matches a stamped SSH repo against its own host', () => { + expect( + runtimeRepoMatchesExecutionHost( + { connectionId: 'target-1', executionHostId: 'ssh:target-1' }, + 'ssh:target-1' + ) + ).toBe(true) + }) + + it('rejects an unstamped SSH repo against a different SSH host', () => { + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'ssh:target-2')).toBe( + false + ) + }) + + it('rejects an unstamped SSH repo against local and runtime hosts', () => { + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'local')).toBe(false) + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'runtime:env-1')).toBe( + false + ) + }) + + it('keeps a host-less legacy repo adoptable by any host', () => { + expect(runtimeRepoMatchesExecutionHost({}, 'runtime:env-1')).toBe(true) + expect(runtimeRepoMatchesExecutionHost({}, 'local')).toBe(true) + expect(runtimeRepoMatchesExecutionHost({}, 'ssh:target-1')).toBe(true) + }) + + it('matches any repo when the caller names no host', () => { + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' })).toBe(true) + expect(runtimeRepoMatchesExecutionHost({ executionHostId: 'runtime:env-1' }, null)).toBe(true) + }) + + it('keeps a stamped repo bound to the host it names', () => { + expect(runtimeRepoMatchesExecutionHost({ executionHostId: 'runtime:env-1' }, 'local')).toBe( + false + ) + expect( + runtimeRepoMatchesExecutionHost( + { executionHostId: 'local', connectionId: 'target-1' }, + 'local' + ) + ).toBe(true) + }) +}) diff --git a/src/main/runtime/runtime-worktree-selection.ts b/src/main/runtime/runtime-worktree-selection.ts index 5dd1c4fd9a4..7e3fc5be481 100644 --- a/src/main/runtime/runtime-worktree-selection.ts +++ b/src/main/runtime/runtime-worktree-selection.ts @@ -1,5 +1,5 @@ import type { Repo } from '../../shared/repo-types' -import type { ExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' import { splitWorktreeId } from '../../shared/worktree/id' import type { GitPushTarget } from '../../shared/worktree/types' @@ -38,8 +38,9 @@ export function getRuntimeWorktreeRemovalOptionsKey( } // Null executionHostId means host-unaware: path-only callers match any repo, and the first runtime -// host can adopt a legacy (unstamped) repo. But an unstamped repo with a connectionId is an SSH repo -// (resolves to ssh:), so it must not be adopted/matched by a runtime host at the same path. +// host can adopt a legacy (unstamped) repo. A repo that names a host in *either* spelling matches +// only that host — including its own ssh:, which an executionHostId-only comparison +// used to reject, so an unstamped SSH repo failed to dedupe against itself. export function runtimeRepoMatchesExecutionHost( repo: Pick, executionHostId?: ExecutionHostId | null @@ -47,10 +48,10 @@ export function runtimeRepoMatchesExecutionHost( if (executionHostId == null) { return true } - if (repo.executionHostId != null) { - return repo.executionHostId === executionHostId + if (repo.executionHostId == null && repo.connectionId == null) { + return true } - return repo.connectionId == null + return getRepoExecutionHostId(repo) === executionHostId } export function parseExactWorktreeIdSelector( diff --git a/src/main/runtime/worktree-scan-execution-host-routing.test.ts b/src/main/runtime/worktree-scan-execution-host-routing.test.ts new file mode 100644 index 00000000000..9e40789da82 --- /dev/null +++ b/src/main/runtime/worktree-scan-execution-host-routing.test.ts @@ -0,0 +1,214 @@ +// SSH ownership has two spellings on a repo row: the legacy `connectionId` field and the unified +// `executionHostId: 'ssh:*'`. This suite pins the scan and the terminal launch that follows it for +// the second spelling — the seam #17909 identified but could not test end to end (#11163). +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const electronMocks = vi.hoisted(() => { + const ipcMain = { + on: vi.fn(() => ipcMain), + removeListener: vi.fn(() => ipcMain), + emit: vi.fn(() => true) + } + return { + BrowserWindow: { fromId: vi.fn((): unknown => null) }, + webContents: { fromId: vi.fn((): unknown => null) }, + ipcMain, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } + } +}) +vi.mock('electron', () => electronMocks) + +const getSshGitProviderMock = vi.hoisted(() => vi.fn()) +vi.mock('../providers/ssh-git-dispatch', () => ({ + getSshGitProvider: getSshGitProviderMock, + getSshGitProviderGeneration: vi.fn(() => 0), + SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE: 'unavailable', + requireSshGitProvider: (connectionId: string) => getSshGitProviderMock(connectionId) +})) + +const listWorktreesStrictMock = vi.hoisted(() => vi.fn()) +vi.mock('../git/worktree', async (importOriginal) => ({ + ...(await importOriginal>()), + listWorktreesStrict: listWorktreesStrictMock +})) + +vi.mock('./repo-worktree-admin-fingerprint', () => ({ + readRepoWorktreeAdminFingerprint: vi.fn(async () => null) +})) + +import { OrcaRuntimeService } from './orca-runtime' + +const TARGET_ID = 'remote-1' +const REPO_ID = 'repo-remote' +const REPO_PATH = '/srv/app' +const WORKTREE_PATH = '/srv/app-feature' +const WORKTREE_ID = `${REPO_ID}::${WORKTREE_PATH}` +const MAIN_WORKTREE_ID = `${REPO_ID}::${REPO_PATH}` + +function makeMeta(overrides: Record = {}) { + return { + displayName: 'feature', + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + linkedGitLabMR: null, + linkedGitLabIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 0, + ...overrides + } +} + +/** One repo row owned by an SSH host, stamped with `executionHostId` only — no `connectionId`. */ +function makeStore(repoOverrides: Record) { + const metaById: Record> = { + [WORKTREE_ID]: makeMeta({ + hostId: `ssh:${TARGET_ID}`, + instanceId: '11111111-1111-4111-8111-111111111111' + }), + [MAIN_WORKTREE_ID]: makeMeta({ + displayName: 'main', + hostId: `ssh:${TARGET_ID}`, + instanceId: '22222222-2222-4222-8222-222222222222' + }) + } + const repos = [ + { + id: REPO_ID, + path: REPO_PATH, + displayName: 'app', + badgeColor: 'blue', + addedAt: 1, + ...repoOverrides + } + ] + const store = { + getRepo: (id: string) => repos.find((repo) => repo.id === id), + getRepos: () => repos, + getAllWorktreeMeta: () => metaById, + getWorktreeMeta: (id: string) => metaById[id], + setWorktreeMeta: (id: string, meta: Record) => { + metaById[id] = { ...(metaById[id] ?? makeMeta()), ...meta } as never + return metaById[id] + }, + removeWorktreeMeta: () => {}, + getAllWorktreeLineage: () => ({}), + getAllWorkspaceLineage: () => ({}), + removeWorktreeLineage: vi.fn(), + removeWorkspaceLineage: vi.fn(), + getGitHubCache: () => undefined as never, + getSettings: () => ({ + workspaceDir: '/tmp/workspaces', + nestWorkspaces: false, + refreshLocalBaseRefOnWorktreeCreate: false, + branchPrefix: 'none', + branchPrefixCustom: '' + }), + getProjects: () => [] + } + return store +} + +type RuntimeInternals = { + listResolvedWorktrees: () => Promise<{ id: string; path: string; hostId?: string }[]> +} + +function makeRuntime(repoOverrides: Record): { + runtime: OrcaRuntimeService + list: () => Promise<{ id: string; path: string; hostId?: string }[]> +} { + const runtime = new OrcaRuntimeService(makeStore(repoOverrides) as never) + return { + runtime, + list: () => (runtime as unknown as RuntimeInternals).listResolvedWorktrees() + } +} + +describe('worktree scan execution-host routing', () => { + beforeEach(() => { + getSshGitProviderMock.mockReset() + listWorktreesStrictMock.mockReset() + listWorktreesStrictMock.mockResolvedValue([]) + }) + + it('scans an executionHostId-only SSH repo over its SSH provider, not on the client', async () => { + const listWorktrees = vi.fn(async () => [ + { path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true }, + { path: WORKTREE_PATH, head: 'def', branch: 'feature', isBare: false, isMainWorktree: false } + ]) + getSshGitProviderMock.mockReturnValue({ listWorktrees }) + const { list } = makeRuntime({ executionHostId: `ssh:${TARGET_ID}` }) + + const worktrees = await list() + + expect(getSshGitProviderMock).toHaveBeenCalledWith(TARGET_ID) + expect(listWorktrees).toHaveBeenCalledWith(REPO_PATH) + // A client-side `git worktree list` against a remote path is the silent-substitution failure. + expect(listWorktreesStrictMock).not.toHaveBeenCalled() + expect(worktrees.map((worktree) => worktree.path).sort()).toEqual([REPO_PATH, WORKTREE_PATH]) + expect(worktrees.every((worktree) => worktree.hostId === `ssh:${TARGET_ID}`)).toBe(true) + }) + + it('routes the PTY of an executionHostId-only SSH worktree to its host', async () => { + getSshGitProviderMock.mockReturnValue({ + listWorktrees: async () => [ + { path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true }, + { + path: WORKTREE_PATH, + head: 'def', + branch: 'feature', + isBare: false, + isMainWorktree: false + } + ] + }) + const { runtime } = makeRuntime({ executionHostId: `ssh:${TARGET_ID}` }) + const spawn = vi.fn().mockResolvedValue({ id: 'pty-1' }) + runtime.setPtyController({ + spawn, + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + } as never) + + await runtime.createTerminal(`id:${WORKTREE_ID}`) + + expect(spawn).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: TARGET_ID, cwd: WORKTREE_PATH }) + ) + }) + + it('still routes a legacy connectionId-only SSH repo the same way', async () => { + getSshGitProviderMock.mockReturnValue({ + listWorktrees: async () => [ + { path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true }, + { + path: WORKTREE_PATH, + head: 'def', + branch: 'feature', + isBare: false, + isMainWorktree: false + } + ] + }) + const { runtime } = makeRuntime({ connectionId: TARGET_ID }) + const spawn = vi.fn().mockResolvedValue({ id: 'pty-1' }) + runtime.setPtyController({ + spawn, + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + } as never) + + await runtime.createTerminal(`id:${WORKTREE_ID}`) + + expect(getSshGitProviderMock).toHaveBeenCalledWith(TARGET_ID) + expect(spawn).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: TARGET_ID, cwd: WORKTREE_PATH }) + ) + }) +}) From 36e139ed19e172aba11351657a2ed5c0ed661186 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 17:03:36 -0700 Subject: [PATCH 19/77] docs: update localized Android APK links to 0.0.47 Update localized README links to the latest verified mobile Android release. --- docs/readme/README.es.md | 4 ++-- docs/readme/README.fr.md | 4 ++-- docs/readme/README.ja.md | 4 ++-- docs/readme/README.ko.md | 4 ++-- docs/readme/README.pt.md | 4 ++-- docs/readme/README.zh-CN.md | 4 ++-- 6 files changed, 12 insertions(+), 12 deletions(-) diff --git a/docs/readme/README.es.md b/docs/readme/README.es.md index 638220150b5..f2247e0900d 100644 --- a/docs/readme/README.es.md +++ b/docs/readme/README.es.md @@ -36,7 +36,7 @@ Supervisa y dirige a tus agentes desde el teléfono — recibe una notificación cuando un agente termine y envía instrucciones de seguimiento desde cualquier lugar. -[App Store de iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [APK para Android](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) +[App Store de iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [APK para Android](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) @@ -227,7 +227,7 @@ yay -S stably-orca-bin Vincúlala con tu app de escritorio para supervisar y dirigir a tus agentes desde el teléfono. - **iOS:** [Descargar desde App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) -- **Android:** [Descargar el APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [Descargar el APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.fr.md b/docs/readme/README.fr.md index a64b1af58af..97c78d4e713 100644 --- a/docs/readme/README.fr.md +++ b/docs/readme/README.fr.md @@ -40,7 +40,7 @@ Surveillez et pilotez vos agents depuis votre téléphone — soyez notifié quand un agent termine, et envoyez des instructions de suivi où que vous soyez. -[App Store iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) +[App Store iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) @@ -235,7 +235,7 @@ yay -S stably-orca-bin Associez-la à l'app de bureau pour surveiller et piloter vos agents depuis votre téléphone. - **iOS :** [Télécharger sur l'App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) ou [rejoindre TestFlight](https://testflight.apple.com/join/YjeGMQBA) -- **Android :** [Télécharger l'APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android :** [Télécharger l'APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.ja.md b/docs/readme/README.ja.md index cc589be1cc5..cce2032a67c 100644 --- a/docs/readme/README.ja.md +++ b/docs/readme/README.ja.md @@ -36,7 +36,7 @@ スマートフォンからエージェントを監視・操作 — エージェントの完了を通知で受け取り、どこからでもフォローアップを送信できます。 -[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [ドキュメント →](https://www.onorca.dev/docs/mobile) +[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [ドキュメント →](https://www.onorca.dev/docs/mobile) @@ -227,7 +227,7 @@ yay -S stably-orca-bin デスクトップアプリとペアリングして、スマートフォンからエージェントを監視・操作できます。 - **iOS:** [App Store からダウンロード](https://apps.apple.com/us/app/orca-ide/id6766130217) -- **Android:** [APK をダウンロード](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [APK をダウンロード](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.ko.md b/docs/readme/README.ko.md index 01c8cd71ce1..4a75722ff8c 100644 --- a/docs/readme/README.ko.md +++ b/docs/readme/README.ko.md @@ -36,7 +36,7 @@ 휴대폰에서 에이전트를 모니터링하고 조종하세요 — 에이전트가 완료되면 알림을 받고 어디서든 후속 지시를 보낼 수 있습니다. -[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [문서 →](https://www.onorca.dev/docs/mobile) +[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [문서 →](https://www.onorca.dev/docs/mobile) @@ -230,7 +230,7 @@ yay -S stably-orca-bin 데스크톱 앱과 페어링해 휴대폰에서 에이전트를 모니터링하고 조종하세요. - **iOS:** [App Store에서 다운로드](https://apps.apple.com/us/app/orca-ide/id6766130217) 또는 [TestFlight 참여](https://testflight.apple.com/join/YjeGMQBA) -- **Android:** [APK 0.0.46 다운로드](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [설치 가이드](https://www.onorca.dev/docs/android-apk) +- **Android:** [APK 0.0.47 다운로드](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [설치 가이드](https://www.onorca.dev/docs/android-apk) --- diff --git a/docs/readme/README.pt.md b/docs/readme/README.pt.md index 970fcf6c8a0..86d998a4e5f 100644 --- a/docs/readme/README.pt.md +++ b/docs/readme/README.pt.md @@ -36,7 +36,7 @@ Monitore e conduza seus agentes pelo celular — receba uma notificação quando um agente terminar e envie instruções de acompanhamento de qualquer lugar. -[App Store para iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) +[App Store para iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) @@ -230,7 +230,7 @@ yay -S stably-orca-bin Conecte ao app desktop para monitorar e conduzir seus agentes pelo celular. - **iOS:** [Baixar na App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) ou [entrar no TestFlight](https://testflight.apple.com/join/YjeGMQBA) -- **Android:** [Baixar APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [Baixar APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.zh-CN.md b/docs/readme/README.zh-CN.md index d3ace8a6af4..d7bae3fba9e 100644 --- a/docs/readme/README.zh-CN.md +++ b/docs/readme/README.zh-CN.md @@ -36,7 +36,7 @@ 用手机监控并指挥你的智能体 — 智能体完成时收到通知,随时随地发送后续指令。 -[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [文档 →](https://www.onorca.dev/docs/mobile) +[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [文档 →](https://www.onorca.dev/docs/mobile) @@ -227,7 +227,7 @@ yay -S stably-orca-bin 与桌面应用配对,用手机监控并指挥你的智能体。 - **iOS:** [从 App Store 下载](https://apps.apple.com/us/app/orca-ide/id6766130217) -- **Android:** [下载 APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [下载 APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- From d084a2a36aa6f044f51bfc0227346add1003d23c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 17:46:01 -0700 Subject: [PATCH 20/77] fix(ssh): decide remote-vs-local from the resolved execution host, not a raw field (#18294) `repoIsRemote` read `repo.connectionId` directly. That is one of four spellings of host ownership, so the predicate was wrong in both directions: a row carrying only `executionHostId: 'ssh:'` read as local and got the Linux-only `orca-ide` rename it cannot resolve through the relay shim, while a row that declares itself `local` with a stale `connectionId` read as remote and lost the rename it needs on a Linux desktop. The predicate now resolves the host first and asks "does an SSH target hold this row's files" via `getRepoSshConnectionId`. That keeps a `runtime:` host's nested SSH target remote (that machine reaches the files through its own relay shim) while a runtime with no nested target - a full Orca install - stays local, as do WSL and local. Its call sites did not all want that question: - The four launch-scope sites in main already hold the resolved PTY route on `TerminalWorkspaceLaunchScope.connectionId`. `scope.repo` is documented display metadata and can be a row from a different host than the worktree names, so they now read the route they will actually spawn on. A launch shape that disagrees with its own route is the bug, not a second predicate. - `launchAgentInNewTab` picked its repo row with a host-blind `store.repos.find`, so a worktree that names its own host could be shaped by another host's row. It now resolves through `getConnectionIdFromState`, the same rule the file already used for transcript readability. - `resolveAgentBackgroundLaunchHost` derived the route, the trust write and the launch shape from three reads of the raw field; one resolution now feeds all three. Also converts the raw `repo.connectionId` agent-detection probe eight lines above `buildWorktreeStartupForDraft`'s launch shape, which #17919 deferred precisely because converting it alone would have left that file internally inconsistent. Tests cover two distinct SSH hosts (a single-host fixture passes even when the answer comes off the wrong row, which is how the `ssh:m4air` -> openclaw leak survived review) and a `runtime:` host carrying a nested SSH target. --- ...ca-runtime-agent-session-operation.test.ts | 35 ++++ .../orca-runtime-create-agent-session.ts | 7 +- ...e-get-agent-session-execution-namespace.ts | 5 +- ...resolve-mobile-session-terminal-command.ts | 3 +- ...runtime-resolve-worktree-removal-target.ts | 5 +- .../runtime-worktree-agent-startup.test.ts | 111 +++++++++++- .../runtime/runtime-worktree-agent-startup.ts | 8 +- ...ent-background-session-launch-host.test.ts | 74 ++++++++ .../agent-background-session-launch-host.ts | 11 +- ...h-agent-in-new-tab-host-resolution.test.ts | 167 ++++++++++++++++++ .../src/lib/launch-agent-in-new-tab.ts | 17 +- src/shared/agent-launch-remote.test.ts | 33 ++++ src/shared/agent-launch-remote.ts | 31 +++- 13 files changed, 475 insertions(+), 32 deletions(-) create mode 100644 src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts create mode 100644 src/shared/agent-launch-remote.test.ts diff --git a/src/main/runtime/orca-runtime-agent-session-operation.test.ts b/src/main/runtime/orca-runtime-agent-session-operation.test.ts index ac64dfaa6b7..e99560c7b3a 100644 --- a/src/main/runtime/orca-runtime-agent-session-operation.test.ts +++ b/src/main/runtime/orca-runtime-agent-session-operation.test.ts @@ -149,6 +149,41 @@ describe('agent-session create operation ledger', () => { expect(createTerminal).toHaveBeenCalledOnce() }) + it('shapes the launch for the route it resolved, not a repo row on another host', async () => { + // `scope.repo` is display metadata and can be a row from a different host than the worktree + // names (#11163). Reading it made a locally-routed launch emit the SSH relay shim name. + const runtime = createRuntime({ + supportsAgentSessionClaims: () => true, + supportsAgentSessionCreateOperations: () => true + }) + const internal = runtime as unknown as { + resolveTerminalWorkspaceLaunchScope: ReturnType + } + internal.resolveTerminalWorkspaceLaunchScope.mockResolvedValue({ + id: 'worktree-1', + path: '/repo/worktree-1', + connectionId: null, + // The rival row names openclaw while the worktree resolved to no SSH route at all. + repo: { + id: 'repo-1', + connectionId: 'openclaw', + executionHostId: null, + path: '/srv/openclaw' + }, + folderWorkspace: null + }) + const createTerminal = vi.spyOn(runtime, 'createTerminal').mockResolvedValue(terminal()) + + await runtime.createAgentSession( + request(operationId(), { agent: 'claude-agent-teams', prompt: '' }) + ) + + expect(createTerminal).toHaveBeenCalledWith( + 'id:worktree-1', + expect.objectContaining({ command: expect.stringContaining('orca-ide claude-teams') }) + ) + }) + it('requests exact client legacy fallback before nested SSH side effects', async () => { const runtime = createRuntime() const internal = runtime as unknown as { diff --git a/src/main/runtime/orca-runtime-create-agent-session.ts b/src/main/runtime/orca-runtime-create-agent-session.ts index db2b71a0adc..2ac34f4a920 100644 --- a/src/main/runtime/orca-runtime-create-agent-session.ts +++ b/src/main/runtime/orca-runtime-create-agent-session.ts @@ -16,7 +16,6 @@ import { AGENT_SESSION_OPERATION_PER_CLIENT_LIMIT } from './orca-runtime-core' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { resolveTuiAgentLaunchArgs, @@ -152,9 +151,9 @@ export class OrcaRuntimeWithCreateAgentSession extends OrcaRuntimeWithGetAgentSe throw new Error('Selected agent is disabled. Choose an enabled agent before creating.') } const platform = this.getAgentLaunchPlatformForWorkspace(workspace) - const isRemote = workspace.repo - ? repoIsRemote(workspace.repo) - : Boolean(workspace.connectionId) + // Why: `workspace.repo` is display metadata and may be a row from another host; the launch + // shape must match the PTY route this scope already resolved. + const isRemote = Boolean(workspace.connectionId) const shell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts index 7172331a551..afbb70fc286 100644 --- a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts +++ b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts @@ -11,7 +11,6 @@ import type { } from '../../shared/agent-session-host-authority' import { canonicalizeAgentSessionIdentity } from './agent-session-claim-identity' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { buildAgentResumeStartupPlan } from '../../shared/tui-agent-startup' import { @@ -123,7 +122,9 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim throw new Error('Selected agent is disabled. Choose an enabled agent before resuming.') } const platform = this.getAgentLaunchPlatformForWorkspace(workspace) - const isRemote = workspace.repo ? repoIsRemote(workspace.repo) : Boolean(workspace.connectionId) + // Why: `workspace.repo` is display metadata and may be a row from another host; the launch + // shape must match the PTY route this scope already resolved. + const isRemote = Boolean(workspace.connectionId) const shell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts b/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts index ab63b16ee53..db30b7e2d11 100644 --- a/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts +++ b/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts @@ -5,7 +5,6 @@ import type { WorktreeStartupLaunch } from '../../shared/worktree/launch-types' import type { TuiAgent } from '../../shared/tui-agent' import type { SleepingAgentLaunchConfig } from '../../shared/agent-session-resume' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { buildAgentStartupPlan } from '../../shared/tui-agent-startup' import { @@ -54,7 +53,7 @@ export class OrcaRuntimeWithResolveMobileSessionTerminalCommand extends OrcaRunt // Why: mobile may be iOS while the shell host is Windows/macOS/Linux or SSH Linux; quote for the host shell. const platform = this.getAgentLaunchPlatformForWorkspace(workspace) // Why: SSH runs the CLI through the relay shim (plain `orca`), so the Linux-only `orca-ide` rename must not apply. - const isRemote = workspace.repo ? repoIsRemote(workspace.repo) : repoIsRemote(workspace) + const isRemote = Boolean(workspace.connectionId) const queuedShell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts b/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts index 58fd6231fc7..0d934548332 100644 --- a/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts +++ b/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts @@ -14,7 +14,6 @@ import type { ForceDeleteWorktreeBranchResult } from '../../shared/worktree/crea import type { RuntimeTerminalRename } from '../../shared/runtime-types' import type { TerminalWorkspaceLaunchScope } from './runtime-legacy-worker-terminal-recovery-types' import type { TerminalCreateOptions } from './runtime-terminal-contracts' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' import { resolveBareAgentLaunchCommand } from './runtime-agent-launch-resolution' @@ -175,7 +174,9 @@ export class OrcaRuntimeWithResolveWorktreeRemovalTarget extends OrcaRuntimeWith const settings = store.getSettings() const platform = this.getAgentLaunchPlatformForWorkspace(workspace) - const isRemote = workspace.repo ? repoIsRemote(workspace.repo) : Boolean(workspace.connectionId) + // Why: `workspace.repo` is display metadata and may be a row from another host; the launch + // shape must match the PTY route this scope already resolved. + const isRemote = Boolean(workspace.connectionId) const queuedShell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/runtime-worktree-agent-startup.test.ts b/src/main/runtime/runtime-worktree-agent-startup.test.ts index e276595c4dd..87fcfd9dd58 100644 --- a/src/main/runtime/runtime-worktree-agent-startup.test.ts +++ b/src/main/runtime/runtime-worktree-agent-startup.test.ts @@ -1,14 +1,119 @@ import { describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../shared/repo-types' const mocks = vi.hoisted(() => ({ markCodexProjectTrusted: vi.fn(), markCopilotFolderTrusted: vi.fn(), - markCursorWorkspaceTrusted: vi.fn() + markCursorWorkspaceTrusted: vi.fn(), + detectRemoteAgents: vi.fn(), + detectInstalledAgentsWithShellPathHydration: vi.fn() })) -vi.mock('../agent-trust-presets', () => mocks) +vi.mock('../agent-trust-presets', () => ({ + markCodexProjectTrusted: mocks.markCodexProjectTrusted, + markCopilotFolderTrusted: mocks.markCopilotFolderTrusted, + markCursorWorkspaceTrusted: mocks.markCursorWorkspaceTrusted +})) -import { markLocalWorktreeTrusted } from './runtime-worktree-agent-startup' +vi.mock('../preflight/agent-detection', () => ({ + detectRemoteAgents: mocks.detectRemoteAgents, + detectInstalledAgentsWithShellPathHydration: mocks.detectInstalledAgentsWithShellPathHydration +})) + +import { + buildWorktreeStartupForAgent, + buildWorktreeStartupForDraft, + markLocalWorktreeTrusted +} from './runtime-worktree-agent-startup' + +function makeRepo(fields: Partial): Repo { + return { + id: 'repo-1', + name: 'repo', + path: '/srv/repo', + connectionId: null, + executionHostId: null, + ...fields + } as Repo +} + +const settings = { + agentCmdOverrides: {}, + agentDefaultArgs: {}, + agentDefaultEnv: {}, + disabledTuiAgents: [], + defaultTuiAgent: undefined, + terminalWindowsShell: null +} as never + +/** The launched CLI name is the whole decision: `orca` is the relay shim, `orca-ide` is local. */ +function launchCliNameFor(repo: Repo): string { + return buildWorktreeStartupForAgent({ + repo, + settings, + agent: 'claude-agent-teams', + getLaunchPlatform: () => 'linux', + toSessionOptions: () => undefined + }).startup.command.split(' ')[0]! +} + +describe('buildWorktreeStartupForAgent host resolution', () => { + // Why two hosts: one SSH fixture passes even when the launch shape is resolved off another + // host's row, which is the shape of the `ssh:m4air` -> openclaw leak. + it('drops the Linux-only rename for both spellings of SSH ownership on two hosts', () => { + expect(launchCliNameFor(makeRepo({ connectionId: 'm4air' }))).toBe('orca') + expect(launchCliNameFor(makeRepo({ executionHostId: 'ssh:openclaw' }))).toBe('orca') + }) + + it('keeps the Linux rename for a local row carrying a stale connection', () => { + expect(launchCliNameFor(makeRepo({ connectionId: 'm4air', executionHostId: 'local' }))).toBe( + 'orca-ide' + ) + }) + + it('drops the rename for a runtime host reaching a nested SSH target', () => { + expect( + launchCliNameFor(makeRepo({ connectionId: 'nested', executionHostId: 'runtime:vm-1' })) + ).toBe('orca') + }) + + it('keeps the rename for a runtime host with no nested SSH target', () => { + expect(launchCliNameFor(makeRepo({ executionHostId: 'runtime:vm-1' }))).toBe('orca-ide') + }) +}) + +describe('buildWorktreeStartupForDraft agent detection', () => { + it('probes the SSH host named only by executionHostId instead of this client', async () => { + mocks.detectRemoteAgents.mockResolvedValueOnce(['claude']) + mocks.detectInstalledAgentsWithShellPathHydration.mockResolvedValue([]) + + const result = await buildWorktreeStartupForDraft({ + repo: makeRepo({ executionHostId: 'ssh:openclaw' }), + settings, + draft: 'ship it', + getLaunchPlatform: () => 'linux' + }) + + expect(mocks.detectRemoteAgents).toHaveBeenCalledWith({ connectionId: 'openclaw' }) + expect(mocks.detectInstalledAgentsWithShellPathHydration).not.toHaveBeenCalled() + expect(result?.agent).toBe('claude') + }) + + it('probes this client for a local row carrying a stale connection', async () => { + mocks.detectRemoteAgents.mockClear() + mocks.detectInstalledAgentsWithShellPathHydration.mockResolvedValueOnce(['claude']) + + const result = await buildWorktreeStartupForDraft({ + repo: makeRepo({ connectionId: 'm4air', executionHostId: 'local' }), + settings, + draft: 'ship it', + getLaunchPlatform: () => 'linux' + }) + + expect(mocks.detectRemoteAgents).not.toHaveBeenCalled() + expect(result?.agent).toBe('claude') + }) +}) describe('markLocalWorktreeTrusted', () => { it('waits for the Codex trust write before resolving', async () => { diff --git a/src/main/runtime/runtime-worktree-agent-startup.ts b/src/main/runtime/runtime-worktree-agent-startup.ts index 7c662d633e2..8663771e998 100644 --- a/src/main/runtime/runtime-worktree-agent-startup.ts +++ b/src/main/runtime/runtime-worktree-agent-startup.ts @@ -3,6 +3,7 @@ import type { Repo } from '../../shared/repo-types' import type { TuiAgent } from '../../shared/tui-agent' import type { WorktreeStartupLaunch } from '../../shared/worktree/launch-types' import { repoIsRemote } from '../../shared/agent-launch-remote' +import { getRepoSshConnectionId } from '../../shared/execution-host' import { isTuiAgent, TUI_AGENT_CONFIG } from '../../shared/tui-agent-config' import { isTuiAgentEnabled, pickTuiAgent } from '../../shared/tui-agent-selection' import { @@ -55,10 +56,13 @@ export async function buildWorktreeStartupForDraft( : null if (!agent) { let detected: string[] = [] + // Why: detection has to run on the machine that will run the agent, and SSH ownership has two + // spellings — the raw field probes this client for an `executionHostId: 'ssh:*'`-only repo. + const sshConnectionId = getRepoSshConnectionId(repo) try { // Why: startup-draft fallback can run from sparse runtime launch envs too. - detected = repo.connectionId - ? await detectRemoteAgents({ connectionId: repo.connectionId }) + detected = sshConnectionId + ? await detectRemoteAgents({ connectionId: sshConnectionId }) : await detectInstalledAgentsWithShellPathHydration() } catch { detected = [] diff --git a/src/renderer/src/lib/agent-background-session-launch-host.test.ts b/src/renderer/src/lib/agent-background-session-launch-host.test.ts index 86fd8577fae..ef00efa5145 100644 --- a/src/renderer/src/lib/agent-background-session-launch-host.test.ts +++ b/src/renderer/src/lib/agent-background-session-launch-host.test.ts @@ -71,6 +71,80 @@ describe('resolveAgentBackgroundLaunchHost', () => { ).toThrow('unavailable or ambiguous') }) + // Why two hosts: a single-SSH fixture passes even when the route is read off another host's + // row, which is the shape of the `ssh:m4air` -> openclaw leak. + it('routes both spellings of SSH ownership to their own host', () => { + const legacy = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: 'm4air', + executionHostId: null, + path: '/srv/repo' + } as never + }) + const unified = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: null, + executionHostId: 'ssh:openclaw', + path: '/srv/repo' + } as never + }) + + expect(legacy).toMatchObject({ + connectionId: 'm4air', + isRemote: true, + expectedConnectionId: 'm4air' + }) + expect(unified).toMatchObject({ + connectionId: 'openclaw', + isRemote: true, + expectedConnectionId: 'openclaw' + }) + }) + + it('keeps a local row with a stale connection off the SSH route', () => { + const host = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: 'm4air', + executionHostId: 'local', + path: '/srv/repo' + } as never + }) + + expect(host).toMatchObject({ + connectionId: null, + isRemote: false, + expectedConnectionId: null + }) + }) + + it('keeps a runtime host reaching a nested SSH target remote', () => { + const host = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: 'nested', + executionHostId: 'runtime:vm-1', + path: '/srv/repo' + } as never + }) + + expect(host).toMatchObject({ connectionId: 'nested', isRemote: true }) + }) + it('uses Linux startup quoting for a local WSL folder', () => { const folderPath = '\\\\wsl.localhost\\Ubuntu\\home\\me\\project' const host = resolveAgentBackgroundLaunchHost({ diff --git a/src/renderer/src/lib/agent-background-session-launch-host.ts b/src/renderer/src/lib/agent-background-session-launch-host.ts index 7919f9f3af5..300dec7f6ff 100644 --- a/src/renderer/src/lib/agent-background-session-launch-host.ts +++ b/src/renderer/src/lib/agent-background-session-launch-host.ts @@ -6,6 +6,7 @@ import { getFolderWorkspaceConnectionId } from '@/lib/folder-workspace-connectio import { parseWorkspaceKey } from '../../../shared/workspace-scope' import { isWindowsAbsolutePathLike } from '../../../shared/cross-platform-path' import { repoIsRemote } from '../../../shared/agent-launch-remote' +import { getRepoSshConnectionId } from '../../../shared/execution-host' import { isWslUncPath } from '../../../shared/wsl-paths' type LaunchStore = ReturnType @@ -41,14 +42,18 @@ export function resolveAgentBackgroundLaunchHost(args: { }): AgentBackgroundLaunchHost { const { store, worktreeId, worktreePath, repo } = args if (repo) { + // Why: SSH ownership has two spellings, so the raw field spawns an `executionHostId: 'ssh:*'`-only + // repo on the client with a remote path. One resolution feeds the route, the trust write and the + // launch shape, which must not disagree about the host. + const sshConnectionId = getRepoSshConnectionId(repo) return { - connectionId: repo.connectionId ?? null, + connectionId: sshConnectionId, platform: getAgentLaunchPlatformForRepo( repo, - repo.connectionId ? undefined : getLocalProjectExecutionRuntimeContext(store, worktreeId) + sshConnectionId ? undefined : getLocalProjectExecutionRuntimeContext(store, worktreeId) ), isRemote: repoIsRemote(repo), - expectedConnectionId: repo.connectionId ?? null + expectedConnectionId: sshConnectionId } } const folderWorkspaceConnectionId = resolveFolderWorkspaceConnectionIdForLaunch(store, worktreeId) diff --git a/src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts b/src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts new file mode 100644 index 00000000000..bcb09c1b17c --- /dev/null +++ b/src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts @@ -0,0 +1,167 @@ +// Execution-host coverage for launchAgentInNewTab, split from launch-agent-in-new-tab.test.ts to +// keep both files within the lines budget. + +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const mockCreateTab = vi.fn() +const mockQueueTabStartupCommand = vi.fn() + +type StoreRepo = { + id: string + connectionId: string | null + executionHostId?: string | null + path: string +} + +type StoreWorktree = { + id: string + repoId: string + projectId: string + hostId?: string | null + path: string + displayName: string +} + +const store = { + activeRepoId: 'repo-1', + activeWorktreeId: 'wt-1', + settings: { + agentCmdOverrides: {} as Record, + agentDefaultArgs: {} as Record, + agentDefaultEnv: {} as Record>, + activeRuntimeEnvironmentId: null as string | null + }, + projects: [{ id: 'repo-1', localWindowsRuntimePreference: { kind: 'inherit-global' as const } }], + repos: [] as StoreRepo[], + folderWorkspaces: [] as unknown[], + projectGroups: [] as unknown[], + sshConnectionStates: new Map(), + transientClearedAgentStatusConnectionIds: {} as Record, + worktreesByRepo: {} as Record, + allWorktrees: vi.fn(() => store.worktreesByRepo['repo-1'] ?? []), + tabsByWorktree: { 'wt-1': [{ id: 'tab-1' }] }, + openFiles: [] as { id: string; worktreeId: string }[], + browserTabsByWorktree: {} as Record, + tabBarOrderByWorktree: {} as Record, + terminalLayoutsByTabId: {} as Record< + string, + { activeLeafId: string | null; ptyIdsByLeafId?: Record } + >, + ptyIdsByTabId: {} as Record, + createTab: mockCreateTab, + closeTab: vi.fn(), + queueTabStartupCommand: mockQueueTabStartupCommand, + setActiveTabType: vi.fn(), + setTabBarOrder: vi.fn(), + setAgentStatus: vi.fn(), + seedNativeChatLaunchPrompt: vi.fn(), + seedNativeChatLaunchDraft: vi.fn(), + markNativeChatLaunchPromptFailed: vi.fn() +} + +vi.mock('@/store', () => ({ useAppStore: { getState: () => store } })) + +vi.mock('sonner', () => ({ toast: { message: vi.fn(), error: vi.fn() } })) + +vi.mock('@/components/tab-bar/reconcile-order', () => ({ + reconcileTabOrder: vi.fn( + (_stored, termIds: string[], editorIds: string[], browserIds: string[]) => [ + ...termIds, + ...editorIds, + ...browserIds + ] + ) +})) + +vi.mock('@/lib/agent-paste-draft', () => ({ pasteDraftWhenAgentReady: vi.fn() })) + +vi.mock('@/lib/telemetry', () => ({ + track: vi.fn(), + tuiAgentToAgentKind: (agent: string) => agent +})) + +vi.mock('@/runtime/web-runtime-session', () => ({ + createWebRuntimeSessionTerminal: vi.fn(), + isWebRuntimeSessionActive: vi.fn(() => false), + isWebTerminalSurfaceTabId: vi.fn(() => false) +})) + +function worktreeOn(hostId: string, path: string): StoreWorktree { + return { id: 'wt-1', repoId: 'repo-1', projectId: 'repo-1', hostId, path, displayName: 'main' } +} + +async function launchOnLinux(): Promise { + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + launchAgentInNewTab({ agent: 'claude-agent-teams', worktreeId: 'wt-1', launchPlatform: 'linux' }) +} + +function queuedCommand(): string { + return mockQueueTabStartupCommand.mock.calls[0]?.[1]?.command +} + +describe('launchAgentInNewTab execution host resolution', () => { + beforeEach(() => { + vi.clearAllMocks() + mockCreateTab.mockReturnValue({ id: 'tab-1' }) + store.settings = { + agentCmdOverrides: {}, + agentDefaultArgs: {}, + agentDefaultEnv: {}, + activeRuntimeEnvironmentId: null + } + store.tabsByWorktree = { 'wt-1': [{ id: 'tab-1' }] } + store.openFiles = [] + store.browserTabsByWorktree = {} + store.tabBarOrderByWorktree = {} + store.terminalLayoutsByTabId = {} + store.ptyIdsByTabId = {} + }) + + it('shapes the launch from the worktree host, not a rival repo row on another SSH host', async () => { + // `store.repos.find` is host-blind, so a worktree that names its own host could be shaped by + // an `ssh:openclaw` row it has nothing to do with (#11163). + store.repos = [ + { id: 'repo-1', connectionId: 'openclaw', path: '/srv/openclaw' }, + { id: 'repo-1', connectionId: null, executionHostId: 'local', path: '/repo' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('local', '/repo/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca-ide claude-teams '--dangerously-skip-permissions'") + }) + + it('keeps a worktree on one SSH host remote while a rival row names another', async () => { + store.repos = [ + { id: 'repo-1', connectionId: 'openclaw', path: '/srv/openclaw' }, + { id: 'repo-1', connectionId: null, executionHostId: 'ssh:m4air', path: '/srv/m4air' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('ssh:m4air', '/srv/m4air/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca claude-teams '--dangerously-skip-permissions'") + }) + + it('keeps a runtime host reaching a nested SSH target on the relay shim name', async () => { + store.repos = [ + { id: 'repo-1', connectionId: 'nested', executionHostId: 'runtime:vm-1', path: '/srv/vm' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('runtime:vm-1', '/srv/vm/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca claude-teams '--dangerously-skip-permissions'") + }) + + it('keeps a runtime host with no nested SSH target on the local CLI name', async () => { + store.repos = [ + { id: 'repo-1', connectionId: null, executionHostId: 'runtime:vm-1', path: '/srv/vm' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('runtime:vm-1', '/srv/vm/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca-ide claude-teams '--dangerously-skip-permissions'") + }) +}) diff --git a/src/renderer/src/lib/launch-agent-in-new-tab.ts b/src/renderer/src/lib/launch-agent-in-new-tab.ts index b6cbbb736d8..bf4eda09890 100644 --- a/src/renderer/src/lib/launch-agent-in-new-tab.ts +++ b/src/renderer/src/lib/launch-agent-in-new-tab.ts @@ -22,7 +22,6 @@ import { } from '../../../shared/tui-agent-launch-defaults' import { resolveLocalWindowsAgentStartupShell } from '../../../shared/windows-terminal-shell' import { TUI_AGENT_CONFIG } from '../../../shared/tui-agent-config' -import { repoIsRemote } from '../../../shared/agent-launch-remote' import { seedCommandCodeSubmittedPromptStatus } from '@/lib/command-code-prompt-status-seed' import type { TuiAgent } from '../../../shared/tui-agent' import type { LaunchSource } from '../../../shared/telemetry-events' @@ -96,16 +95,23 @@ export function launchAgentInNewTab(args: LaunchAgentInNewTabArgs): LaunchAgentI const store = useAppStore.getState() const worktree = store.allWorktrees?.().find((entry: { id: string }) => entry.id === worktreeId) const repo = worktree ? store.repos?.find((entry) => entry.id === worktree.repoId) : null + // Why: `store.repos.find` is host-blind and the same repo id can exist on local, SSH and runtime + // hosts, so the row it returns can belong to a different host than the worktree names (#11163). + // The shared resolver answers from the worktree's own host; `undefined` (rival rows disagree) is + // not evidence of a remote, and main rejects that launch anyway. + const worktreeSshConnectionId = getConnectionIdFromState(store, worktreeId) const resolvedLaunchPlatform = launchPlatform ?? (repo ? getAgentLaunchPlatformForRepo( repo, - repo.connectionId ? undefined : getLocalProjectExecutionRuntimeContext(store, worktreeId) + worktreeSshConnectionId + ? undefined + : getLocalProjectExecutionRuntimeContext(store, worktreeId) ) : CLIENT_PLATFORM) // Why: SSH remotes deploy the shim as plain `orca`, so skip the Linux-only `orca-ide` rename for remote launches. - const isRemote = repo ? repoIsRemote(repo) : false + const isRemote = Boolean(worktreeSshConnectionId) const queuedShell = resolveLocalWindowsAgentStartupShell({ platform: resolvedLaunchPlatform, isRemote, @@ -127,9 +133,8 @@ export function launchAgentInNewTab(args: LaunchAgentInNewTabArgs): LaunchAgentI agent, promptDelivery: viewModePromptDelivery, launchDraftText: trimmedPrompt, - nativeChatTranscriptIsLocalReadable: isNativeChatTranscriptLocalReadable( - getConnectionIdFromState(store, worktreeId) - ) + nativeChatTranscriptIsLocalReadable: + isNativeChatTranscriptLocalReadable(worktreeSshConnectionId) } const initialViewModeProps = initialAgentTabViewModeProps(store.settings, initialViewModeOptions) const startupPlanBase = { diff --git a/src/shared/agent-launch-remote.test.ts b/src/shared/agent-launch-remote.test.ts new file mode 100644 index 00000000000..4656dab39fc --- /dev/null +++ b/src/shared/agent-launch-remote.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { repoIsRemote } from './agent-launch-remote' + +describe('repoIsRemote', () => { + it('reads both spellings of SSH ownership on two different hosts', () => { + // Why two hosts: a single-host fixture passes even when the predicate answers from the wrong + // row, which is how the `ssh:m4air` -> openclaw leak survived review. + expect(repoIsRemote({ connectionId: 'm4air', executionHostId: null })).toBe(true) + expect(repoIsRemote({ connectionId: null, executionHostId: 'ssh:openclaw' })).toBe(true) + expect(repoIsRemote({ connectionId: 'm4air', executionHostId: 'ssh:m4air' })).toBe(true) + }) + + it('answers local for a row that declares itself local with a stale connection', () => { + expect(repoIsRemote({ connectionId: 'm4air', executionHostId: 'local' })).toBe(false) + }) + + it('keeps a runtime host with a nested SSH target remote', () => { + expect(repoIsRemote({ connectionId: 'nested-target', executionHostId: 'runtime:vm-1' })).toBe( + true + ) + }) + + it('keeps a runtime host with no nested SSH target local-shaped', () => { + // A runtime with no nested target is a full Orca install, not a relay shim, so it keeps the + // platform CLI name. + expect(repoIsRemote({ connectionId: null, executionHostId: 'runtime:vm-1' })).toBe(false) + }) + + it('keeps plain local and WSL rows local', () => { + expect(repoIsRemote({ connectionId: null, executionHostId: null })).toBe(false) + expect(repoIsRemote({ connectionId: null, executionHostId: 'local' })).toBe(false) + }) +}) diff --git a/src/shared/agent-launch-remote.ts b/src/shared/agent-launch-remote.ts index 08482815859..bec5aae49b9 100644 --- a/src/shared/agent-launch-remote.ts +++ b/src/shared/agent-launch-remote.ts @@ -1,11 +1,26 @@ +import type { Repo } from './repo-types' +import { getRepoSshConnectionId } from './execution-host' + /** - * Why: a repo reached over SSH runs the Orca CLI through the relay shim, which - * is always deployed as plain `orca` (Unix) / `orca.cmd` (Windows). The - * Linux-only `orca-ide` rename — which exists solely to avoid shadowing the - * GNOME Orca screen reader on a local desktop — must not be applied to those - * remotes, or `orca-ide claude-teams` lands on a PATH where it does not exist. - * `connectionId` is the SSH signal; WSL and local stay false. + * Why: a repo reached over SSH runs the Orca CLI through the relay shim, which is always deployed + * as plain `orca` (Unix) / `orca.cmd` (Windows). The Linux-only `orca-ide` rename — which exists + * solely to avoid shadowing the GNOME Orca screen reader on a local desktop — must not be applied + * to those remotes, or `orca-ide claude-teams` lands on a PATH where it does not exist. + * + * The question is "does an SSH target hold this row's files", not "what may this client dial", so + * it resolves the execution host instead of reading the raw `connectionId` field. SSH ownership has + * two spellings and the raw read is wrong in both directions: + * + * - a row carrying only `executionHostId: 'ssh:'` reads as local and gets the `orca-ide` + * rename it cannot resolve on the remote; + * - a row that declares itself `local` while a stale `connectionId` survives reads as remote and + * loses the rename it needs on a Linux desktop. + * + * `runtime:` keeps its nested SSH target (that machine reaches the files through its own relay + * shim), while a runtime host with no nested target is a full Orca install and stays false — as do + * WSL and local. Callers routing a client-local PTY want `getSshTargetIdForExecutionHost` instead; + * callers that already hold a resolved launch connection should read that, not re-derive here. */ -export function repoIsRemote(repo: { connectionId?: string | null }): boolean { - return Boolean(repo.connectionId) +export function repoIsRemote(repo: Pick): boolean { + return getRepoSshConnectionId(repo) !== null } From e827ce2ccb72c23df8a5d6e8ec3c35861b1b1adb Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:01:23 +0000 Subject: [PATCH 21/77] Update README downloads badge --- docs/assets/readme-downloads.svg | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index 0708d09d993..39fbcf45af1 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 36m + + downloads: 37m @@ -15,7 +15,7 @@ downloads downloads - 36m - 36m + 37m + 37m From 953df47fc440db62a21f15df9c3e4fd92e3a3f1d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:04:48 -0700 Subject: [PATCH 22/77] feat(providers): dispatch git and filesystem providers by execution host (#18296) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `const c = repo.connectionId; c ? sshProvider(c) : local()` overloads `null` to mean both "resolved: local" and "could not resolve", so every path that cannot determine the host silently runs remote work on the client (#11163). It also cannot express a `runtime:` host at all. Add a host-keyed dispatch whose input is an `ExecutionHostId` — never null — with `local`, `ssh` and `runtime` as three symmetric entries, and which throws on an id that names no host instead of degrading to this machine. `ssh` carries `provider: null` for "remote, currently unreachable", which is now a different answer from "local" rather than the same one. `runtime:` is a distinct entry rather than a provider because main does not execute runtime hosts at all: they are forwarded over the environment transport, and a runtime row's `connectionId` names a target in the server's namespace. Dialing it from this client's SSH table would trade a silent-local bug for a silent-wrong-host one. First migrations, both to rows resolved via `getRepoExecutionHostId`: - repo-worktrees: an `executionHostId: 'ssh:*'`-only row no longer lists, root-matches, or strict-lists against a same-named local path. - workspace-space-repo-scan: same for the size scan, and `isRemote` no longer contradicts the `executionHostId` emitted beside it. --- .../execution-host-provider-dispatch.test.ts | 105 ++++++++++++ .../execution-host-provider-dispatch.ts | 155 ++++++++++++++++++ src/main/repo-worktrees.test.ts | 61 ++++++- src/main/repo-worktrees.ts | 40 +++-- src/main/workspace-space-repo-scan.ts | 88 +++++----- 5 files changed, 400 insertions(+), 49 deletions(-) create mode 100644 src/main/providers/execution-host-provider-dispatch.test.ts create mode 100644 src/main/providers/execution-host-provider-dispatch.ts diff --git a/src/main/providers/execution-host-provider-dispatch.test.ts b/src/main/providers/execution-host-provider-dispatch.test.ts new file mode 100644 index 00000000000..fd68deea787 --- /dev/null +++ b/src/main/providers/execution-host-provider-dispatch.test.ts @@ -0,0 +1,105 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { + ExecutionHostNotDispatchableError, + requireFilesystemProviderForHost, + requireGitProviderForHost, + resolveFilesystemRouteForHost, + resolveGitRouteForHost, + UnresolvableExecutionHostError +} from './execution-host-provider-dispatch' +import { registerSshGitProvider, unregisterSshGitProvider } from './ssh-git-dispatch' +import { + registerSshFilesystemProvider, + unregisterSshFilesystemProvider +} from './ssh-filesystem-dispatch' + +const connectionId = 'host-dispatch-target' +const gitProvider = { listWorktrees: async () => [] } as never +const filesystemProvider = { readDir: async () => [] } as never + +describe('execution host provider dispatch', () => { + afterEach(() => { + unregisterSshGitProvider(connectionId) + unregisterSshFilesystemProvider(connectionId) + }) + + it('routes `local` to the local entry rather than to a provider', () => { + expect(resolveGitRouteForHost('local')).toEqual({ kind: 'local', hostId: 'local' }) + expect(resolveFilesystemRouteForHost('local')).toEqual({ kind: 'local', hostId: 'local' }) + }) + + it('routes an ssh host to its registered provider', () => { + registerSshGitProvider(connectionId, gitProvider) + registerSshFilesystemProvider(connectionId, filesystemProvider) + + expect(resolveGitRouteForHost(`ssh:${connectionId}`)).toEqual({ + kind: 'ssh', + hostId: `ssh:${connectionId}`, + connectionId, + provider: gitProvider + }) + expect(resolveFilesystemRouteForHost(`ssh:${connectionId}`)).toEqual({ + kind: 'ssh', + hostId: `ssh:${connectionId}`, + connectionId, + provider: filesystemProvider + }) + expect(requireGitProviderForHost(`ssh:${connectionId}`)).toBe(gitProvider) + expect(requireFilesystemProviderForHost(`ssh:${connectionId}`)).toBe(filesystemProvider) + }) + + it('answers `unreachable`, not `local`, for an ssh host with no registered provider', () => { + const route = resolveGitRouteForHost(`ssh:${connectionId}`) + + // The distinction the old `connectionId ? ssh : local` shape could not spell. + expect(route.kind).toBe('ssh') + expect(route.kind === 'ssh' && route.provider).toBeNull() + expect(() => requireGitProviderForHost(`ssh:${connectionId}`)).toThrow( + /Remote connection dropped/ + ) + expect(() => requireFilesystemProviderForHost(`ssh:${connectionId}`)).toThrow( + /Remote connection dropped/ + ) + }) + + it('routes a runtime host to its own entry instead of collapsing it into local', () => { + expect(resolveGitRouteForHost('runtime:env-7')).toEqual({ + kind: 'runtime', + hostId: 'runtime:env-7', + environmentId: 'env-7' + }) + expect(resolveFilesystemRouteForHost('runtime:env-7')).toEqual({ + kind: 'runtime', + hostId: 'runtime:env-7', + environmentId: 'env-7' + }) + }) + + it('refuses to hand a runtime host to this process’s ssh table', () => { + // A runtime repo row carries the *server's* nested target id. Dialling it here would reach a + // same-named target in this client's namespace. + registerSshGitProvider(connectionId, gitProvider) + + expect(() => requireGitProviderForHost('runtime:env-7')).toThrow( + ExecutionHostNotDispatchableError + ) + expect(() => requireFilesystemProviderForHost('runtime:env-7')).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it('refuses to serve a local host from the remote-only accessor', () => { + expect(() => requireGitProviderForHost('local')).toThrow(ExecutionHostNotDispatchableError) + expect(() => requireFilesystemProviderForHost('local')).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it.each([null, undefined, '', 'nonsense', 'ssh:', 'runtime:', 'ssh:a|b'])( + 'throws instead of answering local for the unresolvable host %p', + (hostId) => { + expect(() => resolveGitRouteForHost(hostId)).toThrow(UnresolvableExecutionHostError) + expect(() => resolveFilesystemRouteForHost(hostId)).toThrow(UnresolvableExecutionHostError) + } + ) +}) diff --git a/src/main/providers/execution-host-provider-dispatch.ts b/src/main/providers/execution-host-provider-dispatch.ts new file mode 100644 index 00000000000..1034ceac140 --- /dev/null +++ b/src/main/providers/execution-host-provider-dispatch.ts @@ -0,0 +1,155 @@ +/** + * Host-keyed provider dispatch: one entry per execution host kind, with `local` among them. + * + * The incumbent spelling across main is `const c = repo.connectionId; c ? sshProvider(c) : local()`, + * where `null` means *both* "resolved: this is local" and "could not resolve". Every path that + * cannot determine the host therefore answers "local" and runs remote work on the client — the + * #11163 defect class, which has produced a reproduced cross-host leak (an `ssh:` worktree + * resolving to another target) and near-misses where a transcript that exists only on a remote host + * would have been read locally. The shape also cannot express a `runtime:` host at all. + * + * This module removes that spelling. Its input is an `ExecutionHostId`, which is never null, and an + * id that names no host throws instead of degrading. `getRepoExecutionHostId` / + * `getWorktreeExecutionHostId` / `resolveWorktreeExecutionHost` are the resolution layer that feeds + * it; the last one already answers `unresolved` as a distinct verdict rather than "local". + * + * Why a route union rather than a uniform `getGitProviderForHost(): IGitProvider`, which is the + * VS Code shape (`registerProvider(Schemas.file, …)` symmetric with `Schemas.vscodeRemote`, and + * `ENOPRO` when nothing matches). Two properties of this process, not style preferences: + * + * - `local` git and filesystem work is free functions taking per-worktree execution options + * (`wslDistro`, `sharedLinkPaths`, admission tier), not an `IGitProvider`. There is no local + * provider object to register, and a stateless one would silently drop WSL routing. + * - `runtime:` is not executed in this process *at all*. It is forwarded over the + * environment's transport (`runtimeEnvironments:call`) and the receiving server normalizes it to + * its own `local`. A repo row on a runtime host carries the server's *nested* SSH target in + * `connectionId`; that id is addressable only as the pair (environmentId, targetId). Handing it + * to this client's SSH table would dial a same-named target in the wrong namespace — turning a + * silent-local bug into a silent-wrong-host bug. `host-repo-catalog-snapshot` and + * `host-qualified-worktree-listing` already reject runtime hosts for the same reason. + * + * So the answer is Zed's shape — an enum on the owner (`Local { fs }` vs `Remote { … }`) — and the + * three kinds are symmetric variants of it. Callers switch exhaustively, so `runtime` can no longer + * collapse into `local` by omission. + * + * Note the deliberate second distinction inside the `ssh` variant: `provider: null` means "this host + * is remote and currently unreachable", which is not the same answer as "this host is local" and can + * no longer be spelled the same way. That mirrors the `live` / `unverifiable` / `exited` rule in + * docs/reference/ssh-execution-boundary.md — loss of contact is never evidence of locality. + */ + +import { + parseExecutionHostId, + type ExecutionHostId, + type LOCAL_EXECUTION_HOST_ID, + type ParsedExecutionHost +} from '../../shared/execution-host' +import { getSshGitProvider, SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from './ssh-git-dispatch' +import { + getSshFilesystemProvider, + SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE +} from './ssh-filesystem-dispatch' +import type { IFilesystemProvider, IGitProvider } from './types' + +/** An id that names no execution host. Never degrade to local — that is the whole defect class. */ +export class UnresolvableExecutionHostError extends Error { + constructor(readonly hostId: string | null | undefined) { + super( + `Cannot route work: ${JSON.stringify(hostId ?? null)} names no execution host. ` + + 'Refusing to fall back to this machine.' + ) + this.name = 'UnresolvableExecutionHostError' + } +} + +/** Asking this process for a host it does not execute is a routing mistake, not a fallback. */ +export class ExecutionHostNotDispatchableError extends Error { + constructor(readonly hostId: ExecutionHostId) { + super(`Execution host ${hostId} is not dispatched by this process.`) + this.name = 'ExecutionHostNotDispatchableError' + } +} + +type LocalRoute = { kind: 'local'; hostId: typeof LOCAL_EXECUTION_HOST_ID } +type RuntimeRoute = { kind: 'runtime'; hostId: `runtime:${string}`; environmentId: string } +type SshRoute = { + kind: 'ssh' + hostId: `ssh:${string}` + connectionId: string + /** `null` is "remote, currently unreachable" — never "local". */ + provider: TProvider | null +} + +export type ExecutionHostGitRoute = LocalRoute | RuntimeRoute | SshRoute +export type ExecutionHostFilesystemRoute = LocalRoute | RuntimeRoute | SshRoute + +// Takes an unvalidated string rather than `ExecutionHostId`: validating is the point, and host +// ids also arrive from persistence and IPC where the compiler cannot vouch for them. +function parseRoutableHost(hostId: string | null | undefined): ParsedExecutionHost { + const parsed = parseExecutionHostId(hostId) + if (!parsed) { + throw new UnresolvableExecutionHostError(hostId) + } + return parsed +} + +export function resolveGitRouteForHost(hostId: string | null | undefined): ExecutionHostGitRoute { + const parsed = parseRoutableHost(hostId) + switch (parsed.kind) { + case 'local': + return { kind: 'local', hostId: parsed.id } + case 'ssh': + return { + kind: 'ssh', + hostId: parsed.id, + connectionId: parsed.targetId, + provider: getSshGitProvider(parsed.targetId) ?? null + } + case 'runtime': + return { kind: 'runtime', hostId: parsed.id, environmentId: parsed.environmentId } + } +} + +export function resolveFilesystemRouteForHost( + hostId: string | null | undefined +): ExecutionHostFilesystemRoute { + const parsed = parseRoutableHost(hostId) + switch (parsed.kind) { + case 'local': + return { kind: 'local', hostId: parsed.id } + case 'ssh': + return { + kind: 'ssh', + hostId: parsed.id, + connectionId: parsed.targetId, + provider: getSshFilesystemProvider(parsed.targetId) ?? null + } + case 'runtime': + return { kind: 'runtime', hostId: parsed.id, environmentId: parsed.environmentId } + } +} + +/** For call sites that are structurally remote-only: local and runtime are both routing errors. */ +export function requireGitProviderForHost(hostId: string | null | undefined): IGitProvider { + const route = resolveGitRouteForHost(hostId) + if (route.kind !== 'ssh') { + throw new ExecutionHostNotDispatchableError(route.hostId) + } + if (!route.provider) { + throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) + } + return route.provider +} + +export function requireFilesystemProviderForHost( + hostId: string | null | undefined +): IFilesystemProvider { + const route = resolveFilesystemRouteForHost(hostId) + if (route.kind !== 'ssh') { + throw new ExecutionHostNotDispatchableError(route.hostId) + } + if (!route.provider) { + throw new Error(SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE) + } + return route.provider +} diff --git a/src/main/repo-worktrees.test.ts b/src/main/repo-worktrees.test.ts index d1c3935c1e1..a6b20c5445f 100644 --- a/src/main/repo-worktrees.test.ts +++ b/src/main/repo-worktrees.test.ts @@ -1,4 +1,4 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const { listWorktreeGraphMock, listWorktreesMock, listWorktreesStrictMock } = vi.hoisted(() => ({ listWorktreeGraphMock: vi.fn(), @@ -19,6 +19,8 @@ import { listRepoWorktreeGraph, listRepoWorktrees } from './repo-worktrees' +import { registerSshGitProvider, unregisterSshGitProvider } from './providers/ssh-git-dispatch' +import { WorktreeCatalogUnavailableError } from '../shared/worktree/worktree-catalog-availability' describe('repo-worktrees', () => { beforeEach(() => { @@ -196,6 +198,63 @@ describe('repo-worktrees', () => { expect(listWorktreesStrictMock).not.toHaveBeenCalled() }) + // #11163: a row may spell its owner only as `executionHostId`. Reading `connectionId` answers + // "local" for it and runs the listing against a same-named path on this machine. + describe('rows that spell their owner only as executionHostId', () => { + const sshOnlyRepo = { + id: 'repo-1', + path: '/srv/repo', + displayName: 'repo', + badgeColor: '#000', + addedAt: 0, + kind: 'git' as const, + executionHostId: 'ssh:host-a' as const + } + + afterEach(() => { + unregisterSshGitProvider('host-a') + unregisterSshGitProvider('nested-target') + }) + + it('never lists an ssh-owned row with local git', async () => { + await expect(listRepoWorktrees(sshOnlyRepo)).rejects.toThrow(WorktreeCatalogUnavailableError) + expect(listWorktreesMock).not.toHaveBeenCalled() + }) + + it('lists an ssh-owned row through its registered provider', async () => { + const listWorktrees = vi.fn().mockResolvedValue([{ path: '/srv/repo' }]) + registerSshGitProvider('host-a', { listWorktrees } as never) + + await expect(listRepoWorktrees(sshOnlyRepo)).resolves.toEqual([{ path: '/srv/repo' }]) + expect(listWorktrees).toHaveBeenCalledWith('/srv/repo') + expect(listWorktreesMock).not.toHaveBeenCalled() + }) + + it('keeps an ssh-owned root out of the local repo-root match', () => { + expect(isRepoRoot([sshOnlyRepo], '/srv/repo')).toBe(false) + }) + + it('rejects strict local listing for an ssh-owned row', async () => { + await expect(listLocalRepoWorktreesStrict(sshOnlyRepo)).rejects.toThrow('remote repository') + expect(listWorktreesStrictMock).not.toHaveBeenCalled() + }) + + it('refuses to answer a runtime-owned row from a same-named local target', async () => { + const listWorktrees = vi.fn().mockResolvedValue([{ path: '/wrong/host' }]) + registerSshGitProvider('nested-target', { listWorktrees } as never) + + await expect( + listRepoWorktrees({ + ...sshOnlyRepo, + executionHostId: 'runtime:env-7', + connectionId: 'nested-target' + }) + ).rejects.toThrow(WorktreeCatalogUnavailableError) + expect(listWorktrees).not.toHaveBeenCalled() + expect(listWorktreesMock).not.toHaveBeenCalled() + }) + }) + it('treats Windows repo root casing differences as the same local root', () => { const repos = [ { diff --git a/src/main/repo-worktrees.ts b/src/main/repo-worktrees.ts index b7c6e16f91a..228f303bfa2 100644 --- a/src/main/repo-worktrees.ts +++ b/src/main/repo-worktrees.ts @@ -2,7 +2,8 @@ import type { Repo } from '../shared/repo-types' import type { GitWorktreeInfo } from '../shared/worktree/types' import { listWorktreeGraph, listWorktrees, listWorktreesStrict } from './git/worktree' import { isFolderRepo } from '../shared/repo-kind' -import { getSshGitProvider } from './providers/ssh-git-dispatch' +import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' +import { resolveGitRouteForHost } from './providers/execution-host-provider-dispatch' import { areWorktreePathsEqual } from './ipc/worktree-logic' import { WorktreeCatalogUnavailableError } from '../shared/worktree/worktree-catalog-availability' @@ -16,8 +17,12 @@ function hasLocalRepoWorktreeListOptions(options: LocalRepoWorktreeListOptions | } export function isRepoRoot(repos: Repo[], resolvedTarget: string): boolean { + // Why: `!repo.connectionId` matched a remote path against a local one for a row that spells its + // owner only as `executionHostId: 'ssh:'`. Resolve the host instead of reading one field. return repos.some( - (repo) => !repo.connectionId && areWorktreePathsEqual(repo.path, resolvedTarget) + (repo) => + getRepoExecutionHostId(repo) === LOCAL_EXECUTION_HOST_ID && + areWorktreePathsEqual(repo.path, resolvedTarget) ) } @@ -41,17 +46,24 @@ export async function listRepoWorktrees( if (isFolderRepo(repo)) { return [createFolderWorktree(repo)] } - if (repo.connectionId) { - const provider = getSshGitProvider(repo.connectionId) + const route = resolveGitRouteForHost(getRepoExecutionHostId(repo)) + if (route.kind === 'runtime') { + // A runtime row's `connectionId` names a target in the *server's* namespace, not one this + // client may dial. Reading it here would answer from a same-named local target. + throw new WorktreeCatalogUnavailableError( + `Worktree catalog unavailable for ${repo.path}: host ${route.hostId} is not reachable from this process.` + ) + } + if (route.kind === 'ssh') { // Why: runtime worktree resolution can run before SSH providers have reattached during startup. // Never fall back to local git against a server path, and never report the unreachable host as an // empty catalog (#14004) — callers treat a resolved listing as authoritative. - if (!provider) { + if (!route.provider) { throw new WorktreeCatalogUnavailableError( - `Worktree catalog unavailable for ${repo.path}: SSH connection "${repo.connectionId}" is not connected.` + `Worktree catalog unavailable for ${repo.path}: SSH connection "${route.connectionId}" is not connected.` ) } - return await provider.listWorktrees(repo.path) + return await route.provider.listWorktrees(repo.path) } return hasLocalRepoWorktreeListOptions(options) ? await listWorktrees(repo.path, options) @@ -72,9 +84,15 @@ export async function listRepoWorktreeGraph( if (isFolderRepo(repo)) { return [createFolderWorktree(repo)] } - if (repo.connectionId) { - const provider = getSshGitProvider(repo.connectionId) - return provider ? await provider.listWorktrees(repo.path) : [] + const route = resolveGitRouteForHost(getRepoExecutionHostId(repo)) + // An unreachable remote host answers `[]` here, unlike listRepoWorktrees above, which throws. + // Preserved as-is: this call site's callers treat the graph as best-effort. The inconsistency is + // real but is a separate behavior decision from resolving the host correctly. + if (route.kind === 'runtime') { + return [] + } + if (route.kind === 'ssh') { + return route.provider ? await route.provider.listWorktrees(repo.path) : [] } return hasLocalRepoWorktreeListOptions(options) ? await listWorktreeGraph(repo.path, options) @@ -85,7 +103,7 @@ export async function listLocalRepoWorktreesStrict( repo: Repo, options?: LocalRepoWorktreeListOptions ): Promise { - if (repo.connectionId) { + if (getRepoExecutionHostId(repo) !== LOCAL_EXECUTION_HOST_ID) { throw new Error('Cannot list worktrees for a remote repository') } if (isFolderRepo(repo)) { diff --git a/src/main/workspace-space-repo-scan.ts b/src/main/workspace-space-repo-scan.ts index 9a0bc316403..3bf3c6cf1f4 100644 --- a/src/main/workspace-space-repo-scan.ts +++ b/src/main/workspace-space-repo-scan.ts @@ -9,11 +9,13 @@ import type { WorkspaceSpaceWorktree } from '../shared/workspace-space-types' import { mapWithConcurrency } from '../shared/map-with-concurrency' -import { getRepoExecutionHostId } from '../shared/execution-host' +import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' import { readWorktreeMetaForHost } from './persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from './worktree-metadata-ownership' -import { getSshFilesystemProvider } from './providers/ssh-filesystem-dispatch' -import { getSshGitProvider } from './providers/ssh-git-dispatch' +import { + resolveFilesystemRouteForHost, + resolveGitRouteForHost +} from './providers/execution-host-provider-dispatch' import { createFolderWorktree, listRepoWorktrees } from './repo-worktrees' import { mergeWorktree } from './ipc/worktree-logic' import { getLocalProjectWorktreeGitOptions } from './project-runtime-git-options' @@ -89,16 +91,25 @@ async function listWorktreesForSpaceScan( if (isFolderRepo(repo)) { return { ok: true, worktrees: [createFolderWorktree(repo)] } } - if (repo.connectionId) { - const provider = getSshGitProvider(repo.connectionId) - if (!provider) { + // Why: the raw `connectionId` field answers "local" for a row that spells its owner only as + // `executionHostId: 'ssh:'`, which sizes a same-named path on this machine instead. + const route = resolveGitRouteForHost(getRepoExecutionHostId(repo)) + if (route.kind === 'runtime') { + return { + ok: false, + status: 'unavailable', + error: `Host ${route.hostId} is not reachable from this process.` + } + } + if (route.kind === 'ssh') { + if (!route.provider) { return { ok: false, status: 'unavailable', - error: `SSH connection "${repo.connectionId}" is not connected.` + error: `SSH connection "${route.connectionId}" is not connected.` } } - const worktrees = await provider.listWorktrees(repo.path, { signal }) + const worktrees = await route.provider.listWorktrees(repo.path, { signal }) throwIfWorkspaceSpaceScanAborted(signal) return { ok: true, worktrees } } @@ -175,7 +186,7 @@ export async function scanWorkspaceSpaceRepo(args: { executionHostId: getRepoExecutionHostId(repo), displayName: repo.displayName, path: repo.path, - isRemote: Boolean(repo.connectionId), + isRemote: getRepoExecutionHostId(repo) !== LOCAL_EXECUTION_HOST_ID, worktreeCount: 0, scannedWorktreeCount: 0, unavailableWorktreeCount: 1, @@ -193,7 +204,7 @@ export async function scanWorkspaceSpaceRepo(args: { { totalWorktreeCount: progress.totalWorktreeCount + worktrees.length }, options.onProgress ) - const remoteProvider = repo.connectionId ? getSshFilesystemProvider(repo.connectionId) : undefined + const filesystemRoute = resolveFilesystemRouteForHost(getRepoExecutionHostId(repo)) const rows = await mapWithConcurrency(worktrees, WORKTREE_SCAN_CONCURRENCY, async (worktree) => { throwIfWorkspaceSpaceScanAborted(options.signal) reportProgress( @@ -204,33 +215,36 @@ export async function scanWorkspaceSpaceRepo(args: { }, options.onProgress ) - const row = repo.connectionId - ? remoteProvider - ? await scanRemoteWorkspaceSpaceWorktree( - repo, - worktree, - scannedAt, - remoteProvider, - limiters.remoteFallbackTraversal, - options.signal + const row = + filesystemRoute.kind !== 'local' + ? filesystemRoute.kind === 'ssh' && filesystemRoute.provider + ? await scanRemoteWorkspaceSpaceWorktree( + repo, + worktree, + scannedAt, + filesystemRoute.provider, + limiters.remoteFallbackTraversal, + options.signal + ) + : createUnavailableWorkspaceSpaceRow( + repo, + worktree, + scannedAt, + 'unavailable', + filesystemRoute.kind === 'ssh' + ? `SSH filesystem for "${filesystemRoute.connectionId}" is not connected.` + : `Host ${filesystemRoute.hostId} is not reachable from this process.` + ) + : await limiters.localWorktree(() => + scanLocalWorkspaceSpaceWorktree( + repo, + worktree, + scannedAt, + args.readLocalDuDepthOne, + args.normalizeLocalDuPath, + options.signal + ) ) - : createUnavailableWorkspaceSpaceRow( - repo, - worktree, - scannedAt, - 'unavailable', - `SSH filesystem for "${repo.connectionId}" is not connected.` - ) - : await limiters.localWorktree(() => - scanLocalWorkspaceSpaceWorktree( - repo, - worktree, - scannedAt, - args.readLocalDuDepthOne, - args.normalizeLocalDuPath, - options.signal - ) - ) reportProgress( progress, { @@ -265,7 +279,7 @@ export async function scanWorkspaceSpaceRepo(args: { executionHostId: getRepoExecutionHostId(repo), displayName: repo.displayName, path: repo.path, - isRemote: Boolean(repo.connectionId), + isRemote: getRepoExecutionHostId(repo) !== LOCAL_EXECUTION_HOST_ID, worktreeCount: rows.length, ...summary, error: null From 9cda5a9dc01fedeb5aaa90ea68f572f1a9db3606 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:13:08 -0700 Subject: [PATCH 23/77] fix(worktrees): stop a resolved-worktree snapshot answering for repos it never saw (#18295) * fix(worktrees): stop a resolved-worktree snapshot answering for repos it never saw `listResolvedWorktrees` caches one fleet-wide snapshot for RESOLVED_WORKTREE_CACHE_TTL_MS (1s) and reuses it on time alone. Nothing invalidates it when a repo is registered, so for up to a second after a repo row lands, every caller reads a snapshot computed before that repo existed -- and reads the gap as a verdict. The visible failure is the SSH skill install. `resolveSkillSshTarget` resolves a workspace-scope destination through that snapshot, so installing into a worktree on a host connected moments earlier threw `skill-install-workspace-not-found`: the client asserting a remote workspace is absent on the strength of client-side bookkeeping that had never looked at the host. That is the shape `docs/reference/ssh-execution-boundary.md` rules out -- absence from a client-side set is not evidence about the execution host. It made `tests/e2e/ssh-skill-installation.spec.ts:108` fail 3 runs in 4 locally and deterministically in the Docker SSH lane, where connect-then-install lands inside the one-second window every time. The snapshot now carries the repo-registration revision it was computed under and is only reused while that revision still holds. The counter is the one `bumpLocalWorktreeScanGeneration` already advances on every repo add, removal and update, so the check is O(1) and cannot drift from the mutation sites. * fix(worktrees): key the snapshot on repo mutations only, not on generation reads Two things the headless-reattach lane surfaced. The revision I keyed the snapshot on was `generationSequence`, which `getLocalWorktreeScanGeneration` also advances when it mints a key for a repo id nothing has scanned yet. That is a read, not a mutation, so a read path could discard a snapshot that was still perfectly valid -- the mirror image of the staleness this fixes, and a way to make a lookup fail that would otherwise have succeeded. The counter now advances only where the scan generation is actually bumped: repo add, removal, update, and scan-cache invalidation. Separately, `pty-restore-record-seeding.test.ts` primed the cache by writing its private `resolved` field with a literal spelling out `worktrees`, `platformByRepoId` and `expiresAt`. That literal is a second copy of the cache's freshness contract, so adding a field to the real entry left the fake one failing the check: the primed snapshot was rejected, resolution fell through to a real scan, and the headless fixture -- which has no git -- got `selector_not_found`. It now primes through `getSnapshot` so the cache stamps its own entry and the two cannot drift again. The revision never moved during that test (0 before and after), so nothing was being invalidated; the fake entry simply never satisfied the contract. --- .../ipc/pty-restore-record-seeding.test.ts | 24 ++-- src/main/local-worktree-scan-generation.ts | 18 +++ ...-resolved-worktrees-for-explicit-target.ts | 7 +- .../worktree-scan-cache-ttl.spec.ts | 31 +++++ .../runtime-resolved-worktree-cache.test.ts | 112 ++++++++++++++++++ .../runtime-resolved-worktree-cache.ts | 31 ++++- 6 files changed, 207 insertions(+), 16 deletions(-) create mode 100644 src/main/runtime/runtime-resolved-worktree-cache.test.ts diff --git a/src/main/ipc/pty-restore-record-seeding.test.ts b/src/main/ipc/pty-restore-record-seeding.test.ts index 0434be17e67..0f1646e7812 100644 --- a/src/main/ipc/pty-restore-record-seeding.test.ts +++ b/src/main/ipc/pty-restore-record-seeding.test.ts @@ -12,6 +12,9 @@ import { setupPtyIpcSuite } from './pty-ipc-test-harness' import { getDefaultWorkspaceSession } from '../../shared/constants' import { makePaneKey } from '../../shared/stable-pane-id' import { OrcaRuntimeService } from '../runtime/orca-runtime' +import type { RuntimeResolvedWorktreeCache } from '../runtime/runtime-resolved-worktree-cache' +import type { ResolvedWorktree } from '../runtime/runtime-worktree-path-identity' +import { getWorktreeScanMutationRevision } from '../local-worktree-scan-generation' import { registerPtyHandlers, clearProviderPtyState, @@ -359,15 +362,22 @@ describe('registerPtyHandlers', () => { } as never) // Why: selector resolution shells out to git for real repos; prime the // resolved-worktree cache so this headless fixture resolves offline. + // + // Why through getSnapshot and not a hand-written `resolved` entry: the cache decides freshness + // from fields it stamps itself, so a literal that mirrors them is a second copy of that + // contract and goes stale the moment a field is added. Let the cache stamp its own entry. const worktreeResolutionInternals = runtime as unknown as { - buildResolvedWorktreeFromId(id: string): unknown - resolvedWorktrees: object + buildResolvedWorktreeFromId(id: string): ResolvedWorktree + resolvedWorktrees: RuntimeResolvedWorktreeCache } - Reflect.set(worktreeResolutionInternals.resolvedWorktrees, 'resolved', { - worktrees: [worktreeResolutionInternals.buildResolvedWorktreeFromId(worktreeId)], - platformByRepoId: new Map([[repo.id, process.platform]]), - expiresAt: Date.now() + 60_000 - }) + await worktreeResolutionInternals.resolvedWorktrees.getSnapshot( + async () => ({ + worktrees: [worktreeResolutionInternals.buildResolvedWorktreeFromId(worktreeId)], + platformByRepoId: new Map([[repo.id, process.platform]]) + }), + 60_000, + getWorktreeScanMutationRevision() + ) setLocalPtyProvider({ spawn: vi.fn(async () => ({ id: ptyId, diff --git a/src/main/local-worktree-scan-generation.ts b/src/main/local-worktree-scan-generation.ts index c8cc3bd0ff9..a2c86afcc33 100644 --- a/src/main/local-worktree-scan-generation.ts +++ b/src/main/local-worktree-scan-generation.ts @@ -1,5 +1,6 @@ const generationByRepoId = new Map() let generationSequence = 0 +let mutationRevision = 0 export function getLocalWorktreeScanGeneration(repoId: string): number { const existing = generationByRepoId.get(repoId) @@ -13,6 +14,22 @@ export function getLocalWorktreeScanGeneration(repoId: string): number { export function bumpLocalWorktreeScanGeneration(repoId: string): void { generationByRepoId.set(repoId, ++generationSequence) + mutationRevision += 1 +} + +/** + * Advances on every event above that can change what a worktree scan would find — repo add, + * removal, update, and scan-cache invalidation — and on nothing else. A cache that must not answer + * for repos it never saw compares this in O(1) instead of walking the repo list. + * + * Why not `generationSequence`: that also advances when `getLocalWorktreeScanGeneration` mints a key + * for a repo id nothing has scanned yet, which is a read. Keying a snapshot on it would let a read + * path discard a snapshot that is still perfectly valid. + * + * Ordering-only: the value means nothing outside a same-process comparison. + */ +export function getWorktreeScanMutationRevision(): number { + return mutationRevision } export function isLocalWorktreeScanGenerationCurrent(repoId: string, generation: number): boolean { @@ -21,5 +38,6 @@ export function isLocalWorktreeScanGenerationCurrent(repoId: string, generation: export function resetLocalWorktreeScanGenerationsForTests(): void { generationSequence += 1 + mutationRevision += 1 generationByRepoId.clear() } diff --git a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts index 1f0dbe591a3..152ef547889 100644 --- a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts +++ b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts @@ -5,6 +5,7 @@ import { splitWorktreeIdForFilesystem } from '../../shared/worktree/id' import { isPathInsideOrEqual } from '../../shared/cross-platform-path' import type { ResolvedWorktreeSnapshot } from './runtime-resolved-worktree-cache' import { RESOLVED_WORKTREE_CACHE_TTL_MS } from './orca-runtime-postlude' +import { getWorktreeScanMutationRevision } from '../local-worktree-scan-generation' import { resolveLocalProjectRuntimeForRepo, resolveLocalProjectRuntimesForRepos @@ -65,8 +66,7 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends /** A warm fleet snapshot already answers any selector for free, so scoped scanning must yield to it. */ protected hasFreshResolvedWorktreeCache(): boolean { - const cached = this.resolvedWorktrees.peek() - return Boolean(cached && cached.expiresAt > Date.now()) + return this.resolvedWorktrees.isFresh(getWorktreeScanMutationRevision()) } protected async listResolvedWorktrees(): Promise { @@ -79,7 +79,8 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends } return this.resolvedWorktrees.getSnapshot( () => this.computeResolvedWorktrees(), - RESOLVED_WORKTREE_CACHE_TTL_MS + RESOLVED_WORKTREE_CACHE_TTL_MS, + getWorktreeScanMutationRevision() ) } diff --git a/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts index 8dc8d7db3a2..f8ec37f7c2c 100644 --- a/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts +++ b/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts @@ -5,6 +5,7 @@ import { resolveWorktreeScanCacheTtlMs } from '../orca-runtime-test-mocks.spec' import { store } from '../orca-runtime-test-fixtures.spec' +import { bumpLocalWorktreeScanGeneration } from '../../local-worktree-scan-generation' describe('resolveWorktreeScanCacheTtlMs', () => { const BASE_TTL_MS = 30_000 @@ -84,4 +85,34 @@ describe('resolveWorktreeScanCacheTtlMs', () => { vi.useRealTimers() } }) + + it('scans a repo registered after the last snapshot instead of answering from it', async () => { + // Why: the fleet snapshot only covers the repos that existed when it ran, so for a full TTL it + // reported a just-connected SSH host as having no worktrees at all — and callers that resolve a + // workspace through it turned that gap into `skill-install-workspace-not-found`. + vi.mocked(listWorktrees).mockClear() + const addedPath = '/tmp/repo-registered-later' + const repos = [ + { id: 'repo-1', path: '/tmp/repo', displayName: 'repo', badgeColor: 'blue', addedAt: 1 } + ] + const runtime = new OrcaRuntimeService({ ...store, getRepos: () => repos } as never) + const internals = runtime as unknown as { listResolvedWorktrees: () => Promise } + const scanCallsFor = (path: string): number => + vi.mocked(listWorktrees).mock.calls.filter((call) => call[0] === path).length + + await internals.listResolvedWorktrees() + expect(scanCallsFor(addedPath)).toBe(0) + + repos.push({ + id: 'repo-added', + path: addedPath, + displayName: 'added', + badgeColor: 'blue', + addedAt: 2 + }) + bumpLocalWorktreeScanGeneration('repo-added') + + await internals.listResolvedWorktrees() + expect(scanCallsFor(addedPath)).toBe(1) + }) }) diff --git a/src/main/runtime/runtime-resolved-worktree-cache.test.ts b/src/main/runtime/runtime-resolved-worktree-cache.test.ts new file mode 100644 index 00000000000..31cc4fa2b60 --- /dev/null +++ b/src/main/runtime/runtime-resolved-worktree-cache.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, it } from 'vitest' +import { RuntimeResolvedWorktreeCache } from './runtime-resolved-worktree-cache' +import type { ResolvedWorktreeSnapshot } from './runtime-resolved-worktree-cache' +import { + bumpLocalWorktreeScanGeneration, + getLocalWorktreeScanGeneration, + getWorktreeScanMutationRevision +} from '../local-worktree-scan-generation' + +function snapshotOf(ids: string[]): ResolvedWorktreeSnapshot { + return { + worktrees: ids.map((id) => ({ id }) as ResolvedWorktreeSnapshot['worktrees'][number]), + platformByRepoId: new Map() + } +} + +describe('RuntimeResolvedWorktreeCache', () => { + it('reuses a snapshot inside the TTL while the repo inventory is unchanged', async () => { + const cache = new RuntimeResolvedWorktreeCache() + let computes = 0 + const compute = async (): Promise => { + computes += 1 + return snapshotOf(['repo-1::/a']) + } + + await cache.getSnapshot(compute, 60_000, 7) + const second = await cache.getSnapshot(compute, 60_000, 7) + + expect(computes).toBe(1) + expect(second.worktrees.map((worktree) => worktree.id)).toEqual(['repo-1::/a']) + }) + + it('recomputes when the repo inventory moved, even well inside the TTL', async () => { + // Why: this is the whole point. A snapshot taken before a repo was registered cannot testify + // that the repo's worktrees are absent — callers read the gap as "workspace not found". + const cache = new RuntimeResolvedWorktreeCache() + const results = [snapshotOf(['repo-1::/a']), snapshotOf(['repo-1::/a', 'repo-2::/b'])] + let computes = 0 + const compute = async (): Promise => results[computes++] + + await cache.getSnapshot(compute, 60_000, 7) + const afterRegistration = await cache.getSnapshot(compute, 60_000, 8) + + expect(computes).toBe(2) + expect(afterRegistration.worktrees.map((worktree) => worktree.id)).toEqual([ + 'repo-1::/a', + 'repo-2::/b' + ]) + }) + + it('does not join an in-flight compute that started under a stale inventory', async () => { + const cache = new RuntimeResolvedWorktreeCache() + const computed: number[] = [] + const compute = async (): Promise => { + computed.push(computed.length) + return snapshotOf([]) + } + + const first = cache.getSnapshot(compute, 60_000, 7) + const second = cache.getSnapshot(compute, 60_000, 8) + await Promise.all([first, second]) + + expect(computed).toHaveLength(2) + }) + + it('reports freshness against the inventory the snapshot was computed under', async () => { + const cache = new RuntimeResolvedWorktreeCache() + await cache.getSnapshot(async () => snapshotOf([]), 60_000, 7) + + expect(cache.isFresh(7)).toBe(true) + expect(cache.isFresh(8)).toBe(false) + cache.invalidateResolved() + expect(cache.isFresh(7)).toBe(false) + }) + + it('keeps a primed snapshot servable when nothing mutated', async () => { + // Why: the headless-reattach fixtures prime this cache once and then resolve a selector off it + // without any git available. Losing freshness for a reason other than a mutation strands them + // on a real scan, which is the failure this pairs with — a lookup that finds nothing because + // the snapshot was dropped, not because the worktree is gone. + const cache = new RuntimeResolvedWorktreeCache() + let computes = 0 + const prime = async (): Promise => { + computes += 1 + return snapshotOf(['repo-restore::/tmp/restore-records']) + } + await cache.getSnapshot(prime, 60_000, getWorktreeScanMutationRevision()) + + // A read that mints a scan generation for a repo nothing has scanned yet is not a mutation. + getLocalWorktreeScanGeneration(`repo-never-scanned-${Math.random()}`) + + expect(cache.isFresh(getWorktreeScanMutationRevision())).toBe(true) + const served = await cache.getSnapshot(prime, 60_000, getWorktreeScanMutationRevision()) + expect(computes).toBe(1) + expect(served.worktrees.map((worktree) => worktree.id)).toEqual([ + 'repo-restore::/tmp/restore-records' + ]) + }) +}) + +describe('getWorktreeScanMutationRevision', () => { + it('advances on a repo mutation and not on a first-seen generation read', () => { + const repoId = `repo-${Math.random()}` + const before = getWorktreeScanMutationRevision() + + getLocalWorktreeScanGeneration(repoId) + expect(getWorktreeScanMutationRevision()).toBe(before) + + bumpLocalWorktreeScanGeneration(repoId) + expect(getWorktreeScanMutationRevision()).toBe(before + 1) + }) +}) diff --git a/src/main/runtime/runtime-resolved-worktree-cache.ts b/src/main/runtime/runtime-resolved-worktree-cache.ts index b7ce735eefe..ef7532a6e91 100644 --- a/src/main/runtime/runtime-resolved-worktree-cache.ts +++ b/src/main/runtime/runtime-resolved-worktree-cache.ts @@ -5,9 +5,10 @@ export type ResolvedWorktreeSnapshot = { platformByRepoId: ReadonlyMap } -type ResolvedCache = ResolvedWorktreeSnapshot & { expiresAt: number } +type ResolvedCache = ResolvedWorktreeSnapshot & { expiresAt: number; inventoryRevision: number } type ResolvedInFlight = { generation: number + inventoryRevision: number promise: Promise } export class RuntimeResolvedWorktreeCache { @@ -19,25 +20,43 @@ export class RuntimeResolvedWorktreeCache { return this.resolved } + /** + * Why the revision and not the TTL alone: a snapshot only answers for the repos that were + * registered when it ran. A repo added afterwards — a remote host the user just connected — + * is missing from it for reasons that have nothing to do with what exists on that host, and + * callers read the gap as a verdict that the worktree does not exist. + */ + isFresh(inventoryRevision: number, now = Date.now()): boolean { + return Boolean( + this.resolved && + this.resolved.inventoryRevision === inventoryRevision && + this.resolved.expiresAt > now + ) + } + async getSnapshot( compute: () => Promise, - ttlMs: number + ttlMs: number, + inventoryRevision: number ): Promise { - if (this.resolved && this.resolved.expiresAt > Date.now()) { + if (this.resolved && this.isFresh(inventoryRevision)) { return this.resolved } const generation = this.resolvedGeneration - if (this.resolvedInFlight?.generation === generation) { + if ( + this.resolvedInFlight?.generation === generation && + this.resolvedInFlight.inventoryRevision === inventoryRevision + ) { return this.resolvedInFlight.promise } const promise = compute() - this.resolvedInFlight = { generation, promise } + this.resolvedInFlight = { generation, inventoryRevision, promise } try { const result = await promise if (generation === this.resolvedGeneration) { // Why stamped on completion, not entry: a compute that spent longer than the TTL would // otherwise publish an already-expired entry, so the next poll recomputes the same slow path. - this.resolved = { ...result, expiresAt: Date.now() + ttlMs } + this.resolved = { ...result, inventoryRevision, expiresAt: Date.now() + ttlMs } } return result } finally { From d5750648c289c3bcfc1a7a0c85b995b87b9b4391 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:26:07 -0700 Subject: [PATCH 24/77] fix(runtime): route runtime Git by resolved execution host, not repo connectionId (#18307) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `RuntimeGitTarget` carried `connectionId?: string` and no host id, so `undefined` spelled three different answers at once — "runtime: host", "unresolved", and "genuinely local". Its sole resolver read `store.getRepo(worktree.repoId)?.connectionId` and never looked at `worktree.hostId`, which outranks every repo row, so one arbitrarily chosen row decided the execution host for 36 downstream dispatches. The target now carries `executionHostId: ExecutionHostId` (never null, never optional), resolved through the shared rule that landed with #17909/#17919 and dispatched through the host-keyed routes from #18296. Dispatch sites call `requireRuntimeGitProvider`, where `null` means exactly one thing: the host is `local` and the command runs here as free functions. Four answers that used to collapse into one: - `ssh:x` with a rival row on `ssh:y` — routes to x. Previously the first row won, which is the reproduced cross-host leak. - `local` with a surviving `connectionId` — a row contradicting itself; no SSH connection is handed out. - `runtime:` — throws `ExecutionHostNotDispatchableError`. Its repo row's connection names a target in the *server's* namespace; dialling it here reaches a same-named target on this client. - rival rows disagreeing with no worktree host — `worktree_execution_host_unresolved`, matching the launch path rather than guessing a row. An unreachable SSH host still throws `SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE`; loss of contact is never evidence of locality (docs/reference/ssh-execution-boundary.md). `resolveWorktreeLaunchHost` keeps its exact signature and now delegates to `resolveWorktreeHostRouting`, the same resolution answering "which host is this on" rather than "what may this client dial" — the git target needs the first question because `local` and `runtime:` are two different non-SSH answers. No wire change: `RuntimeGitTarget` is main-process internal, and the SSH and local model-discovery host keys are byte-identical to before. `RuntimeFileTarget` has the same defect in ~30 filesystem dispatches and is deliberately left for a follow-up. --- docs/reference/ssh-execution-boundary.md | 2 +- .../execution-host-provider-dispatch.ts | 5 +- .../orca-runtime-git-branch-diff.test.ts | 2 +- .../orca-runtime-git-diff-budget.test.ts | 2 +- src/main/runtime/orca-runtime-git.test.ts | 42 ++-- ...runtime-persist-headless-terminal-title.ts | 24 +- .../methods/git-diff-transport-budget.test.ts | 2 +- src/main/runtime/runtime-file-command-host.ts | 4 +- ...ntime-git-branch-compare-admission.test.ts | 3 +- .../runtime-git-command-target.test.ts | 87 ++++++++ .../runtime/runtime-git-command-target.ts | 66 +++++- ...ime-git-conflict-operation-routing.test.ts | 1 + src/main/runtime/runtime-git-diff-commands.ts | 54 ++--- ...ntime-git-execution-host-ownership.test.ts | 2 +- .../runtime-git-generation-admission.test.ts | 3 +- .../runtime-git-generation-commands.ts | 106 ++++----- .../runtime/runtime-git-generation-context.ts | 34 +++ .../runtime/runtime-git-staging-commands.ts | 47 ++-- .../runtime-git-status-admission.test.ts | 5 +- .../runtime/runtime-git-status-commands.ts | 59 ++--- .../runtime/runtime-git-sync-commands.test.ts | 3 +- src/main/runtime/runtime-git-sync-commands.ts | 75 ++----- .../runtime-git-target-execution-host.test.ts | 205 ++++++++++++++++++ .../runtime/worktree-launch-host-repo.test.ts | 55 ++++- src/main/runtime/worktree-launch-host-repo.ts | 55 +++-- 25 files changed, 660 insertions(+), 283 deletions(-) create mode 100644 src/main/runtime/runtime-git-command-target.test.ts create mode 100644 src/main/runtime/runtime-git-target-execution-host.test.ts diff --git a/docs/reference/ssh-execution-boundary.md b/docs/reference/ssh-execution-boundary.md index 45b307411f6..87495960b76 100644 --- a/docs/reference/ssh-execution-boundary.md +++ b/docs/reference/ssh-execution-boundary.md @@ -13,7 +13,7 @@ Two consequences, both non-negotiable: The vocabulary is fixed: **`live` / `unverifiable` / `exited`**, taken from the incumbent `UnstoppedPtyVerdict`. Do not introduce synonyms, and never collapse `unverifiable` into either neighbour. `exited` requires positive evidence of absence from the host that owns the process; a transport failure can only ever produce `unverifiable`. -Rule 1 is stated at `src/main/source-control/repo-default-branch.ts:76-78`, `src/main/repo-worktrees.ts:45-48`, `OrcaRuntimeService.probeWorktreeDrift` in `src/main/runtime/orca-runtime.ts`, and `src/renderer/src/lib/connection-context.ts:22-24`. It is enforced throughout `src/main/runtime/orca-runtime-git.ts` by the guard that throws `SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE` whenever `target.connectionId` is set and no provider is registered — grep that constant for the current call sites rather than trusting a count. +Rule 1 is stated at `src/main/source-control/repo-default-branch.ts:76-78`, `src/main/repo-worktrees.ts:45-48`, `OrcaRuntimeService.probeWorktreeDrift` in `src/main/runtime/orca-runtime.ts`, and `src/renderer/src/lib/connection-context.ts:22-24`. It is enforced throughout `src/main/runtime/orca-runtime-git.ts` by `requireRuntimeGitProvider` in `src/main/runtime/runtime-git-command-target.ts`, which routes on the target's resolved `executionHostId` rather than on a repo row's `connectionId`: it throws `SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE` when an SSH host has no registered provider, throws `ExecutionHostNotDispatchableError` for a `runtime:` host this process does not execute, and returns `null` only for `local`. Grep those two names for the current call sites rather than trusting a count. `src/main/runtime/unstopped-pty-verification.ts:12-16` is the reference implementation of rule 2: it keeps `live` / `unverifiable` / `exited` as three distinct verdicts, and treats "we could not ask" as its own answer. diff --git a/src/main/providers/execution-host-provider-dispatch.ts b/src/main/providers/execution-host-provider-dispatch.ts index 1034ceac140..079905b9b37 100644 --- a/src/main/providers/execution-host-provider-dispatch.ts +++ b/src/main/providers/execution-host-provider-dispatch.ts @@ -45,6 +45,7 @@ import { type ParsedExecutionHost } from '../../shared/execution-host' import { getSshGitProvider, SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from './ssh-git-dispatch' +import type { SshGitProvider } from './ssh-git-provider' import { getSshFilesystemProvider, SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE @@ -80,7 +81,9 @@ type SshRoute = { provider: TProvider | null } -export type ExecutionHostGitRoute = LocalRoute | RuntimeRoute | SshRoute +// The SSH table stores `SshGitProvider`; narrowing the route to `IGitProvider` would drop the +// remote-only methods (commit-message plans, push-target materialization) that callers need. +export type ExecutionHostGitRoute = LocalRoute | RuntimeRoute | SshRoute export type ExecutionHostFilesystemRoute = LocalRoute | RuntimeRoute | SshRoute // Takes an unvalidated string rather than `ExecutionHostId`: validating is the point, and host diff --git a/src/main/runtime/orca-runtime-git-branch-diff.test.ts b/src/main/runtime/orca-runtime-git-branch-diff.test.ts index aba155a6895..b27e09fefba 100644 --- a/src/main/runtime/orca-runtime-git-branch-diff.test.ts +++ b/src/main/runtime/orca-runtime-git-branch-diff.test.ts @@ -45,7 +45,7 @@ describe('RuntimeGitCommands branch diff', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/orca-runtime-git-diff-budget.test.ts b/src/main/runtime/orca-runtime-git-diff-budget.test.ts index b2e894d4bc6..3cc401b155c 100644 --- a/src/main/runtime/orca-runtime-git-diff-budget.test.ts +++ b/src/main/runtime/orca-runtime-git-diff-budget.test.ts @@ -57,7 +57,7 @@ function commands( return new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree, - ...(connectionId ? { connectionId } : {}), + executionHostId: connectionId ? (`ssh:${connectionId}` as const) : ('local' as const), ...(localGitOptions ? { localGitOptions } : {}) }), getRuntimeSettings: () => ({}) as GlobalSettings diff --git a/src/main/runtime/orca-runtime-git.test.ts b/src/main/runtime/orca-runtime-git.test.ts index 70fe676f98c..9b27200945e 100644 --- a/src/main/runtime/orca-runtime-git.test.ts +++ b/src/main/runtime/orca-runtime-git.test.ts @@ -86,9 +86,13 @@ function makeWorktree(path: string, linkedIssue: number | null = null): Resolved return worktree as unknown as ResolvedRuntimeGitWorktree } +function localTarget(worktreePath: string, linkedIssue: number | null = null) { + return { worktree: makeWorktree(worktreePath, linkedIssue), executionHostId: 'local' as const } +} + function makeCommands(worktreePath: string): RuntimeGitCommands { return new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({}) as GlobalSettings }) } @@ -125,6 +129,7 @@ describe('RuntimeGitCommands', () => { mocks.getStatus.mockResolvedValue({ entries: [], conflictOperation: 'none' }) const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree('/workspace/feature'), repo: { path: '/workspace/repo', symlinkPaths: ['node_modules'] } as never }), @@ -146,7 +151,7 @@ describe('RuntimeGitCommands', () => { resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), repo: { path: '/remote/repo', symlinkPaths: ['node_modules'] } as never, - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -162,6 +167,7 @@ describe('RuntimeGitCommands', () => { tempDirs.push(worktreePath) const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree(worktreePath), localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -186,7 +192,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -217,6 +223,7 @@ describe('RuntimeGitCommands', () => { it('prioritizes a local single-file discard without losing WSL routing', async () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree('/workspace/repo'), localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -237,7 +244,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -256,7 +263,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -299,7 +306,7 @@ describe('RuntimeGitCommands', () => { message: 'docs: update readme' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({ commitMessageAi: { enabled: true, agentId: 'codex' }, @@ -352,6 +359,7 @@ describe('RuntimeGitCommands', () => { }) const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree(worktreePath), localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -410,7 +418,7 @@ describe('RuntimeGitCommands', () => { message: 'feat: update readme' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({ sourceControlAi: { @@ -471,7 +479,7 @@ describe('RuntimeGitCommands', () => { } }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({ sourceControlAi: { @@ -542,7 +550,7 @@ describe('RuntimeGitCommands', () => { } }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -595,7 +603,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({ @@ -637,7 +645,7 @@ describe('RuntimeGitCommands', () => { mocks.getStagedCommitContext.mockResolvedValue(context) mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -663,7 +671,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 77), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -687,7 +695,7 @@ describe('RuntimeGitCommands', () => { mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const getWorktreeLinkedIssue = vi.fn(() => 321) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, getWorktreeLinkedIssue }) @@ -713,7 +721,7 @@ describe('RuntimeGitCommands', () => { mocks.getStagedCommitContext.mockResolvedValue(context) mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, getWorktreeLinkedIssue: () => null }) @@ -734,7 +742,7 @@ describe('RuntimeGitCommands', () => { mocks.getStagedCommitContext.mockResolvedValue(context) mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, // Why: what the host reports when its store is not initialized yet. getWorktreeLinkedIssue: () => undefined @@ -765,7 +773,7 @@ describe('RuntimeGitCommands', () => { mocks.getPullRequestDraftContext.mockResolvedValue(context) mocks.generatePullRequestFieldsFromContext.mockResolvedValue({ success: true, fields: {} }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, getWorktreeLinkedIssue: () => 321 }) @@ -827,7 +835,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 55), - ...(connectionId ? { connectionId } : {}) + executionHostId: connectionId ? (`ssh:${connectionId}` as const) : ('local' as const) }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts b/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts index 3be605b117a..b9a24685be4 100644 --- a/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts +++ b/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts @@ -8,7 +8,9 @@ import type { } from '../../shared/runtime-types' import type { ResolvedWorktree } from './runtime-worktree-path-identity' import type { Repo } from '../../shared/repo-types' +import { LOCAL_EXECUTION_HOST_ID, type ExecutionHostId } from '../../shared/execution-host' import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-options' +import { resolveWorktreeHostRouting } from './worktree-launch-host-repo' export class OrcaRuntimeWithPersistHeadlessTerminalTitle extends OrcaRuntimeWithMoveHeadlessMobileSessionTab { // Persist a manual terminal rename so a headless rebuild keeps the title @@ -157,19 +159,31 @@ export class OrcaRuntimeWithPersistHeadlessTerminalTitle extends OrcaRuntimeWith return await this.notifier.saveMobileMarkdownTab(worktreeId, tabId, baseVersion, content) } + // Why: `getRepo(id)` is host-blind and never read `worktree.hostId`, which outranks every repo + // row. One id can name rows on local, SSH and runtime hosts at once, so an arbitrary row decided + // the execution host for ~36 downstream Git dispatches: a worktree on one SSH host routed to + // another, and a runtime host's *nested* target got dialled in this client's namespace (#11163). protected async resolveRuntimeGitTarget(worktreeSelector: string): Promise<{ worktree: ResolvedWorktree repo?: Repo - connectionId?: string + executionHostId: ExecutionHostId localGitOptions?: { wslDistro?: string } }> { const store = this.requireStore() const worktree = await this.resolveWorktreeSelector(worktreeSelector) - const repo = store.getRepo(worktree.repoId) - const connectionId = repo?.connectionId ?? undefined + const routing = resolveWorktreeHostRouting(store.getRepos(), worktree) + if (routing.kind === 'ambiguous') { + throw new Error('worktree_execution_host_unresolved') + } + const executionHostId = routing.kind === 'resolved' ? routing.hostId : LOCAL_EXECUTION_HOST_ID + // Metadata only (shared-link paths, source-control AI defaults); routing is `executionHostId`. + const repo = + (routing.kind === 'resolved' ? routing.repo : null) ?? store.getRepo(worktree.repoId) const localGitOptions = - repo && !connectionId ? getLocalProjectWorktreeGitOptions(store, repo) : {} - return { worktree, repo, connectionId, localGitOptions } + repo && executionHostId === LOCAL_EXECUTION_HOST_ID + ? getLocalProjectWorktreeGitOptions(store, repo) + : {} + return { worktree, repo, executionHostId, localGitOptions } } protected async resolveRuntimeFileTarget(worktreeSelector: string): Promise<{ diff --git a/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts b/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts index a4b22aaa891..72ae885c9b7 100644 --- a/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts +++ b/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts @@ -161,7 +161,7 @@ describe('remote git diff transport budget', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: { id: 'wt-1', path: '/remote/repo' } as unknown as ResolvedRuntimeGitWorktree, - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/runtime-file-command-host.ts b/src/main/runtime/runtime-file-command-host.ts index 74f71c9d9ab..b757ab29b14 100644 --- a/src/main/runtime/runtime-file-command-host.ts +++ b/src/main/runtime/runtime-file-command-host.ts @@ -42,9 +42,11 @@ export type RuntimeFileCommandHost = { pathText: string, absolutePath: string ): boolean | Promise + // `executionHostId`, not `connectionId`: a repo row's connection cannot tell `runtime:` from + // `local`, and this contract must not re-introduce that spelling. See runtime-git-command-target. resolveRuntimeGitTarget( selector: string - ): Promise<{ worktree: ResolvedRuntimeFileWorktree; connectionId?: string }> + ): Promise<{ worktree: ResolvedRuntimeFileWorktree; executionHostId: ExecutionHostId }> openFile( worktreeId: string, filePath: string, diff --git a/src/main/runtime/runtime-git-branch-compare-admission.test.ts b/src/main/runtime/runtime-git-branch-compare-admission.test.ts index f2b77c209de..f9d5ffa4cfd 100644 --- a/src/main/runtime/runtime-git-branch-compare-admission.test.ts +++ b/src/main/runtime/runtime-git-branch-compare-admission.test.ts @@ -36,6 +36,7 @@ function makeCommands(overrides: Partial = {}): RuntimeGitDiff head: 'a'.repeat(40) } }, + executionHostId: 'local', localGitOptions: { wslDistro: 'Ubuntu' }, ...overrides } as RuntimeGitTarget @@ -71,7 +72,7 @@ describe('RuntimeGitDiffCommands branch-compare admission', () => { it('forwards background admission to the SSH execution host', async () => { const getBranchCompare = vi.fn().mockResolvedValue({ summary: {}, entries: [] }) mocks.getSshGitProvider.mockReturnValue({ getBranchCompare }) - const commands = makeCommands({ connectionId: 'conn-1' }) + const commands = makeCommands({ executionHostId: 'ssh:conn-1' }) await commands.getRuntimeGitBranchCompare('id:wt-1', 'origin/main', 'background') diff --git a/src/main/runtime/runtime-git-command-target.test.ts b/src/main/runtime/runtime-git-command-target.test.ts new file mode 100644 index 00000000000..0edb9d0518a --- /dev/null +++ b/src/main/runtime/runtime-git-command-target.test.ts @@ -0,0 +1,87 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { ExecutionHostNotDispatchableError } from '../providers/execution-host-provider-dispatch' +import { registerSshGitProvider, unregisterSshGitProvider } from '../providers/ssh-git-dispatch' +import { + localGitOptionsForTarget, + requireRuntimeGitProvider, + runtimeGitRouteForTarget, + type RuntimeGitTarget +} from './runtime-git-command-target' + +const worktree = { + id: 'wt-1', + repoId: 'repo-1', + path: '/srv/app', + git: { path: '/srv/app', branch: 'main', isBare: false, isMainWorktree: false } +} as unknown as RuntimeGitTarget['worktree'] + +function target(overrides: Partial): RuntimeGitTarget { + return { worktree, executionHostId: 'local', ...overrides } +} + +describe('runtime Git target routing', () => { + const registered: string[] = [] + + function register(connectionId: string) { + const provider = { getStatus: async () => ({ entries: [] }) } + registerSshGitProvider(connectionId, provider as never) + registered.push(connectionId) + return provider + } + + afterEach(() => { + for (const connectionId of registered.splice(0)) { + unregisterSshGitProvider(connectionId) + } + }) + + it('routes each ssh host to its own provider', () => { + const m4air = register('m4air') + const openclaw = register('openclaw') + + expect(runtimeGitRouteForTarget(target({ executionHostId: 'ssh:m4air' }))).toEqual({ + kind: 'ssh', + connectionId: 'm4air', + provider: m4air + }) + expect(requireRuntimeGitProvider(target({ executionHostId: 'ssh:openclaw' }))).toBe(openclaw) + }) + + it('answers `local` with no provider, which is the only meaning `null` carries', () => { + expect(runtimeGitRouteForTarget(target({}))).toEqual({ kind: 'local' }) + expect(requireRuntimeGitProvider(target({}))).toBeNull() + }) + + // Loss of contact is never evidence of locality (docs/reference/ssh-execution-boundary.md). + it('keeps an unreachable ssh host remote instead of degrading it to local', () => { + const route = runtimeGitRouteForTarget(target({ executionHostId: 'ssh:gone' })) + + expect(route).toEqual({ kind: 'ssh', connectionId: 'gone', provider: null }) + expect(() => requireRuntimeGitProvider(target({ executionHostId: 'ssh:gone' }))).toThrow( + /Remote connection dropped/ + ) + }) + + it('refuses a runtime host even when a same-named target is registered here', () => { + register('nested-1') + + expect(() => runtimeGitRouteForTarget(target({ executionHostId: 'runtime:env-a' }))).toThrow( + ExecutionHostNotDispatchableError + ) + expect(() => requireRuntimeGitProvider(target({ executionHostId: 'runtime:env-a' }))).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it('keeps WSL routing on the local host and off every other one', () => { + const localGitOptions = { wslDistro: 'Ubuntu' } + + expect(localGitOptionsForTarget(target({ localGitOptions }))).toEqual(localGitOptions) + expect( + localGitOptionsForTarget(target({ executionHostId: 'ssh:m4air', localGitOptions })) + ).toEqual({}) + expect( + localGitOptionsForTarget(target({ executionHostId: 'runtime:env-a', localGitOptions })) + ).toEqual({}) + }) +}) diff --git a/src/main/runtime/runtime-git-command-target.ts b/src/main/runtime/runtime-git-command-target.ts index 47131ce7e63..48c4b5492e2 100644 --- a/src/main/runtime/runtime-git-command-target.ts +++ b/src/main/runtime/runtime-git-command-target.ts @@ -1,7 +1,14 @@ +import { LOCAL_EXECUTION_HOST_ID, type ExecutionHostId } from '../../shared/execution-host' import type { GlobalSettings } from '../../shared/global-settings-types' import type { Repo } from '../../shared/repo-types' import type { GitPushTarget, GitWorktreeInfo, Worktree } from '../../shared/worktree/types' import type { GitRuntimeOptions } from '../git/git-runtime-options' +import { + ExecutionHostNotDispatchableError, + resolveGitRouteForHost +} from '../providers/execution-host-provider-dispatch' +import { SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from '../providers/ssh-git-dispatch' +import type { SshGitProvider } from '../providers/ssh-git-provider' import type { CommitMessageAgentEnvironmentResolvers } from '../text-generation/commit-message-agent-environment' import type { PullRequestLinkedIssueMeta } from '../source-control/pull-request-linked-issue' import { normalizeRuntimeRelativePath } from './runtime-relative-paths' @@ -10,8 +17,21 @@ export type ResolvedRuntimeGitWorktree = Worktree & { git: GitWorktreeInfo } export type RuntimeGitTarget = { worktree: ResolvedRuntimeGitWorktree + /** + * Display and settings metadata only (shared-link paths, source-control AI defaults). It can be + * a same-id row from another host when the worktree's own host carries none, so it must not + * decide routing — `executionHostId` does. + */ repo?: Repo - connectionId?: string + /** + * The host this worktree's Git runs on. Never optional and never null: the field it replaced + * (`connectionId?: string`) spelled "runtime host", "unresolved" and "genuinely local" all as + * `undefined`, so every path that could not resolve answered "local" and ran remote work on the + * client (#11163). Unresolved now fails at resolution time instead of arriving here as a + * silently-local target. + */ + executionHostId: ExecutionHostId + /** Only consulted when `executionHostId` is `local`; see `localGitOptionsForTarget`. */ localGitOptions?: GitRuntimeOptions } @@ -29,8 +49,50 @@ export type RuntimeGitCommandHost = { persistMaterializedPushTarget?(worktreeId: string, pushTarget: GitPushTarget): void } +/** + * The two hosts this process can itself execute a runtime Git command on, narrowed from the shared + * host-keyed route in `src/main/providers/execution-host-provider-dispatch.ts`. + * + * `runtime:` is deliberately not a variant. Its Git is executed by that environment's own + * server, and the SSH target on its repo row is that server's *nested* one — addressable only as + * the pair (environmentId, targetId). Handing that id to this client's SSH table dials a + * same-named target in the wrong namespace, so it throws rather than routing. + */ +export type RuntimeGitRoute = + | { kind: 'local' } + /** `provider: null` is "remote and currently unreachable" — never "run it here". */ + | { kind: 'ssh'; connectionId: string; provider: SshGitProvider | null } + +export function runtimeGitRouteForTarget(target: RuntimeGitTarget): RuntimeGitRoute { + const route = resolveGitRouteForHost(target.executionHostId) + switch (route.kind) { + case 'local': + return { kind: 'local' } + case 'ssh': + return { kind: 'ssh', connectionId: route.connectionId, provider: route.provider } + case 'runtime': + throw new ExecutionHostNotDispatchableError(route.hostId) + } +} + +/** + * `null` means exactly one thing: the host is `local`, and this command runs here as free + * functions. An unreachable SSH host and a `runtime:` host both throw. + */ +export function requireRuntimeGitProvider(target: RuntimeGitTarget): SshGitProvider | null { + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'local') { + return null + } + if (!route.provider) { + throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) + } + return route.provider +} + export function localGitOptionsForTarget(target: RuntimeGitTarget): GitRuntimeOptions { - return target.connectionId ? {} : (target.localGitOptions ?? {}) + // WSL routing describes *this* machine; no remote host may inherit it. + return target.executionHostId === LOCAL_EXECUTION_HOST_ID ? (target.localGitOptions ?? {}) : {} } export function normalizeRuntimeGitRelativePath(filePath: string): string { diff --git a/src/main/runtime/runtime-git-conflict-operation-routing.test.ts b/src/main/runtime/runtime-git-conflict-operation-routing.test.ts index b9d42a0af70..0e6d8d22f5e 100644 --- a/src/main/runtime/runtime-git-conflict-operation-routing.test.ts +++ b/src/main/runtime/runtime-git-conflict-operation-routing.test.ts @@ -22,6 +22,7 @@ describe('getRuntimeGitConflictOperation', () => { const commands = new RuntimeGitStatusCommands({ resolveRuntimeGitTarget: async () => ({ worktree: { path: '/home/me/repo/feature' }, + executionHostId: 'local', localGitOptions: { wslDistro: 'Ubuntu' } }) } as never) diff --git a/src/main/runtime/runtime-git-diff-commands.ts b/src/main/runtime/runtime-git-diff-commands.ts index a6d94775ae5..ef4a4a2290f 100644 --- a/src/main/runtime/runtime-git-diff-commands.ts +++ b/src/main/runtime/runtime-git-diff-commands.ts @@ -14,14 +14,11 @@ import { } from '../git/status' import { awaitWindowsHostGitEnvironmentReady } from '../git/runner' import type { GitAdmissionTier } from '../git/command-runner/git-exec-options' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { normalizeRuntimeRelativePath } from './runtime-relative-paths' import { localGitOptionsForTarget, normalizeRuntimeGitRelativePath, + requireRuntimeGitProvider, type RuntimeGitCommandHost } from './runtime-git-command-target' @@ -38,11 +35,8 @@ export class RuntimeGitDiffCommands { ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return assertGitDiffWithinTransportBudget( await provider.getDiff(target.worktree.path, relativePath, staged, compareAgainstHead), maxContentBytes @@ -63,11 +57,8 @@ export class RuntimeGitDiffCommands { admissionTier: GitAdmissionTier = 'interactive' ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getBranchCompare(target.worktree.path, baseRef, { admissionTier }) } return getBranchCompare(target.worktree.path, baseRef, { @@ -81,11 +72,8 @@ export class RuntimeGitDiffCommands { commitId: string ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getCommitCompare(target.worktree.path, commitId) } return getCommitCompare(target.worktree.path, commitId, { @@ -104,11 +92,8 @@ export class RuntimeGitDiffCommands { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) const oldRelativePath = oldPath ? normalizeRuntimeGitRelativePath(oldPath) : undefined - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const results = await provider.getBranchDiff(target.worktree.path, compare.mergeBase, { includePatch: true, headOid: compare.headOid, @@ -152,11 +137,8 @@ export class RuntimeGitDiffCommands { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeRelativePath(args.filePath) const oldRelativePath = args.oldPath ? normalizeRuntimeRelativePath(args.oldPath) : undefined - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return assertGitDiffWithinTransportBudget( await provider.getCommitDiff(target.worktree.path, { commitOid: args.commitOid, @@ -192,11 +174,8 @@ export class RuntimeGitDiffCommands { ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const normalizedRelativePath = normalizeRuntimeGitRelativePath(relativePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getRemoteFileUrl(target.worktree.path, normalizedRelativePath, line) } await awaitWindowsHostGitEnvironmentReady({ cwd: target.worktree.path }) @@ -208,11 +187,8 @@ export class RuntimeGitDiffCommands { sha: string ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getRemoteCommitUrl(target.worktree.path, sha) } await awaitWindowsHostGitEnvironmentReady({ cwd: target.worktree.path }) diff --git a/src/main/runtime/runtime-git-execution-host-ownership.test.ts b/src/main/runtime/runtime-git-execution-host-ownership.test.ts index 8de68b2e9ca..9d42d003eb6 100644 --- a/src/main/runtime/runtime-git-execution-host-ownership.test.ts +++ b/src/main/runtime/runtime-git-execution-host-ownership.test.ts @@ -38,7 +38,7 @@ function remoteCommands(): RuntimeGitCommands { git: { path: '/remote/repo', branch: 'main', isBare: false, isMainWorktree: false } } as unknown as ResolvedRuntimeGitWorktree return new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree, connectionId: 'ssh-1' }), + resolveRuntimeGitTarget: async () => ({ worktree, executionHostId: 'ssh:ssh-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) } diff --git a/src/main/runtime/runtime-git-generation-admission.test.ts b/src/main/runtime/runtime-git-generation-admission.test.ts index 4d93779c7a8..6631930082d 100644 --- a/src/main/runtime/runtime-git-generation-admission.test.ts +++ b/src/main/runtime/runtime-git-generation-admission.test.ts @@ -60,6 +60,7 @@ function makeTarget(path: string, overrides: Partial = {}): Ru path, git: { path, branch: 'main', isBare: false, isMainWorktree: false, head: 'a'.repeat(40) } } as RuntimeGitTarget['worktree'], + executionHostId: 'local', ...overrides } } @@ -148,7 +149,7 @@ describe('RuntimeGitGenerationCommands admission', () => { }) const commands = makeCommands( makeTarget('/remote/repo', { - connectionId: 'conn-1', + executionHostId: 'ssh:conn-1', localGitOptions: { wslDistro: 'Ubuntu' } }) ) diff --git a/src/main/runtime/runtime-git-generation-commands.ts b/src/main/runtime/runtime-git-generation-commands.ts index c741525e0a0..bea10b2ef7d 100644 --- a/src/main/runtime/runtime-git-generation-commands.ts +++ b/src/main/runtime/runtime-git-generation-commands.ts @@ -3,12 +3,8 @@ import { getCommitMessageModelDiscoveryHostKey } from '../../shared/commit-messa import type { HostedReviewProvider } from '../../shared/hosted-review' import { withLinkedIssueDraftContext } from '../../shared/source-control-ai-action-variables' import type { TuiAgent } from '../../shared/tui-agent' -import { gitExecFileAsync } from '../git/runner' import { getStagedCommitContext } from '../git/status' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' +import { SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from '../providers/ssh-git-dispatch' import { loadPullRequestLinkedIssue } from '../source-control/pull-request-linked-issue' import { resolveHostedReviewBodyForGeneration } from '../source-control/pull-request-template' import { prepareLocalCommitMessageAgentEnv } from '../text-generation/commit-message-agent-environment' @@ -25,13 +21,18 @@ import { type GeneratePullRequestFieldsResult } from '../text-generation/commit-message-text-generation' import { getPullRequestDraftContext } from '../text-generation/pull-request-context' -import { localGitOptionsForTarget, type RuntimeGitCommandHost } from './runtime-git-command-target' +import { + localGitOptionsForTarget, + runtimeGitRouteForTarget, + type RuntimeGitCommandHost +} from './runtime-git-command-target' import { getRuntimeGitGenerationSettings, linkedIssueForTarget, linkedIssueMetaForTarget, localAgentRuntimeTargetForTarget, localTextGenerationTargetForTarget, + pullRequestDraftGitExec, type RuntimeCommitMessageSettingsOverride } from './runtime-git-generation-context' @@ -43,9 +44,10 @@ export class RuntimeGitGenerationCommands { settingsOverride?: RuntimeCommitMessageSettingsOverride ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) + const route = runtimeGitRouteForTarget(target) const discoveryHostKey = settingsOverride?.commitMessageDiscoveryHostKey ?? - getCommitMessageModelDiscoveryHostKey(target.connectionId ?? null) + getCommitMessageModelDiscoveryHostKey(route.kind === 'ssh' ? route.connectionId : null) const resolvedSettings = settingsOverride?.sourceControlAiResolvedParams ? { ok: true as const, params: settingsOverride.sourceControlAiResolvedParams } : resolveCommitMessageSettings( @@ -62,8 +64,8 @@ export class RuntimeGitGenerationCommands { return { success: false, error: resolvedSettings.error } } - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { + if (route.kind === 'ssh') { + const provider = route.provider if (!provider) { return { success: false, error: SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } } @@ -118,9 +120,11 @@ export class RuntimeGitGenerationCommands { async cancelRuntimeGenerateCommitMessage(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - await provider?.cancelGenerateCommitMessage(target.worktree.path, 'commit-message') + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'ssh') { + // Cancelling an unreachable host is a no-op, not a local cancel: the local registry is keyed + // by path and would abort an unrelated generation running here for the same path. + await route.provider?.cancelGenerateCommitMessage(target.worktree.path, 'commit-message') return { ok: true } } cancelGenerateCommitMessageLocal(target.worktree.path) @@ -140,9 +144,10 @@ export class RuntimeGitGenerationCommands { settingsOverride?: RuntimeCommitMessageSettingsOverride ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) + const route = runtimeGitRouteForTarget(target) const discoveryHostKey = settingsOverride?.commitMessageDiscoveryHostKey ?? - getCommitMessageModelDiscoveryHostKey(target.connectionId ?? null) + getCommitMessageModelDiscoveryHostKey(route.kind === 'ssh' ? route.connectionId : null) const resolvedSettings = settingsOverride?.sourceControlAiResolvedParams ? { ok: true as const, params: settingsOverride.sourceControlAiResolvedParams } : resolveCommitMessageSettings( @@ -159,8 +164,8 @@ export class RuntimeGitGenerationCommands { return { success: false, error: resolvedSettings.error } } - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId && !provider) { + const provider = route.kind === 'ssh' ? route.provider : null + if (route.kind === 'ssh' && !provider) { return { success: false, error: SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } } const issueMeta = linkedIssueMetaForTarget(this.host, target) @@ -168,56 +173,30 @@ export class RuntimeGitGenerationCommands { meta: issueMeta, provider: input.provider, repoPath: target.worktree.path, - connectionId: target.connectionId, - localGitOptions: target.connectionId - ? {} - : { - ...localGitOptionsForTarget(target), - admissionTier: 'interactive' - } + connectionId: route.kind === 'ssh' ? route.connectionId : undefined, + localGitOptions: + route.kind === 'ssh' + ? {} + : { + ...localGitOptionsForTarget(target), + admissionTier: 'interactive' + } }) let context: Awaited> try { const currentBody = await resolveHostedReviewBodyForGeneration({ body: input.body, repoPath: target.worktree.path, - connectionId: target.connectionId, + connectionId: route.kind === 'ssh' ? route.connectionId : undefined, provider: input.provider, useTemplate: input.useTemplate }) - context = target.connectionId - ? await getPullRequestDraftContext( - (argv, commandOptions) => { - const timeoutMs = commandOptions?.timeoutMs ?? commandOptions?.timeout - return timeoutMs === undefined - ? provider!.exec(argv, target.worktree.path) - : provider!.exec(argv, target.worktree.path, { timeoutMs }) - }, - { - base: input.base, - currentTitle: input.title, - currentBody, - currentDraft: input.draft - } - ) - : await getPullRequestDraftContext( - (argv, options) => - gitExecFileAsync(argv, { - cwd: target.worktree.path, - ...localGitOptionsForTarget(target), - ...(options?.maxBuffer === undefined ? {} : { maxBuffer: options.maxBuffer }), - ...(options?.timeoutMs === undefined && options?.timeout === undefined - ? {} - : { timeout: options?.timeoutMs ?? options?.timeout }), - admissionTier: 'interactive' - }), - { - base: input.base, - currentTitle: input.title, - currentBody, - currentDraft: input.draft - } - ) + context = await getPullRequestDraftContext(pullRequestDraftGitExec(target, route), { + base: input.base, + currentTitle: input.title, + currentBody, + currentDraft: input.draft + }) } catch (error) { return { success: false, @@ -234,7 +213,7 @@ export class RuntimeGitGenerationCommands { ...(linkedIssueDetails ? { linkedIssueDetails } : {}) } - if (target.connectionId) { + if (route.kind === 'ssh') { return generatePullRequestFieldsFromContext(context, resolvedSettings.params, { kind: 'remote', cwd: target.worktree.path, @@ -260,9 +239,9 @@ export class RuntimeGitGenerationCommands { async cancelRuntimeGeneratePullRequestFields(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - await provider?.cancelGenerateCommitMessage(target.worktree.path, 'pull-request-fields') + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'ssh') { + await route.provider?.cancelGenerateCommitMessage(target.worktree.path, 'pull-request-fields') return { ok: true } } cancelGeneratePullRequestFieldsLocal(target.worktree.path) @@ -279,10 +258,11 @@ export class RuntimeGitGenerationCommands { const agentCommandOverride = settingsOverride?.agentCmdOverrides?.[typedAgentId] ?? this.host.getRuntimeSettings().agentCmdOverrides?.[typedAgentId] - if (target.connectionId) { - const provider = getSshGitProvider(target.connectionId) + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'ssh') { + const provider = route.provider if (!provider) { - return { success: false, error: `No git provider for connection "${target.connectionId}"` } + return { success: false, error: `No git provider for connection "${route.connectionId}"` } } return discoverCommitMessageModelsRemote( typedAgentId, diff --git a/src/main/runtime/runtime-git-generation-context.ts b/src/main/runtime/runtime-git-generation-context.ts index 70873155b56..141b5f18220 100644 --- a/src/main/runtime/runtime-git-generation-context.ts +++ b/src/main/runtime/runtime-git-generation-context.ts @@ -1,4 +1,6 @@ import type { GlobalSettings } from '../../shared/global-settings-types' +import { gitExecFileAsync } from '../git/runner' +import type { getPullRequestDraftContext } from '../text-generation/pull-request-context' import { mergeLegacyCommitMessageAiIntoSourceControlAi, type ResolvedSourceControlAiGenerationParams @@ -10,9 +12,41 @@ import type { PullRequestLinkedIssueMeta } from '../source-control/pull-request- import { localGitOptionsForTarget, type RuntimeGitCommandHost, + type RuntimeGitRoute, type RuntimeGitTarget } from './runtime-git-command-target' +type PullRequestDraftGitExec = Parameters[0] + +/** Runs the PR draft-context probes on whichever host `route` resolved to. */ +export function pullRequestDraftGitExec( + target: RuntimeGitTarget, + route: RuntimeGitRoute +): PullRequestDraftGitExec { + if (route.kind === 'ssh') { + const provider = route.provider + if (!provider) { + throw new Error('ssh_git_provider_unavailable') + } + return (argv, options) => { + const timeoutMs = options?.timeoutMs ?? options?.timeout + return timeoutMs === undefined + ? provider.exec(argv, target.worktree.path) + : provider.exec(argv, target.worktree.path, { timeoutMs }) + } + } + return (argv, options) => + gitExecFileAsync(argv, { + cwd: target.worktree.path, + ...localGitOptionsForTarget(target), + ...(options?.maxBuffer === undefined ? {} : { maxBuffer: options.maxBuffer }), + ...(options?.timeoutMs === undefined && options?.timeout === undefined + ? {} + : { timeout: options?.timeoutMs ?? options?.timeout }), + admissionTier: 'interactive' + }) +} + export type RuntimeCommitMessageSettingsOverride = Partial< Pick > & { diff --git a/src/main/runtime/runtime-git-staging-commands.ts b/src/main/runtime/runtime-git-staging-commands.ts index a856be7bf7b..97f791ac575 100644 --- a/src/main/runtime/runtime-git-staging-commands.ts +++ b/src/main/runtime/runtime-git-staging-commands.ts @@ -6,13 +6,10 @@ import { stageFile, unstageFile } from '../git/status' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { localGitOptionsForTarget, normalizeRuntimeGitRelativePath, + requireRuntimeGitProvider, type RuntimeGitCommandHost } from './runtime-git-command-target' @@ -22,11 +19,8 @@ export class RuntimeGitStagingCommands { async stageRuntimeGitPath(worktreeSelector: string, filePath: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.stageFile(target.worktree.path, relativePath) return { ok: true } } @@ -40,11 +34,8 @@ export class RuntimeGitStagingCommands { async unstageRuntimeGitPath(worktreeSelector: string, filePath: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.unstageFile(target.worktree.path, relativePath) return { ok: true } } @@ -61,11 +52,8 @@ export class RuntimeGitStagingCommands { ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePaths = filePaths.map((path) => normalizeRuntimeGitRelativePath(path)) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.bulkStageFiles(target.worktree.path, relativePaths) return { ok: true } } @@ -82,11 +70,8 @@ export class RuntimeGitStagingCommands { ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePaths = filePaths.map((path) => normalizeRuntimeGitRelativePath(path)) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.bulkUnstageFiles(target.worktree.path, relativePaths) return { ok: true } } @@ -103,11 +88,8 @@ export class RuntimeGitStagingCommands { ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePaths = filePaths.map((path) => normalizeRuntimeGitRelativePath(path)) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.bulkDiscardChanges(target.worktree.path, relativePaths) return { ok: true } } @@ -121,11 +103,8 @@ export class RuntimeGitStagingCommands { async discardRuntimeGitPath(worktreeSelector: string, filePath: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.discardChanges(target.worktree.path, relativePath) return { ok: true } } diff --git a/src/main/runtime/runtime-git-status-admission.test.ts b/src/main/runtime/runtime-git-status-admission.test.ts index 1a0c5a0a47f..8f7525b74f6 100644 --- a/src/main/runtime/runtime-git-status-admission.test.ts +++ b/src/main/runtime/runtime-git-status-admission.test.ts @@ -30,7 +30,10 @@ describe('runtime git status admission', () => { return { entries: [], conflictOperation: 'none' } }) const commands = new RuntimeGitStatusCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: { path: '/workspace/feature' } }) + resolveRuntimeGitTarget: async () => ({ + worktree: { path: '/workspace/feature' }, + executionHostId: 'local' + }) } as never) const runtime = { getRuntimeId: () => 'test-runtime', diff --git a/src/main/runtime/runtime-git-status-commands.ts b/src/main/runtime/runtime-git-status-commands.ts index 28390cee7c9..bc6213a959f 100644 --- a/src/main/runtime/runtime-git-status-commands.ts +++ b/src/main/runtime/runtime-git-status-commands.ts @@ -14,12 +14,12 @@ import { getSubmoduleStatus as getGitSubmoduleStatus } from '../git/status' import type { GitProviderStatusOptions } from '../providers/types' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { getWorktreeSharedLinkPaths } from '../git/worktree-shared-directories' -import { localGitOptionsForTarget, type RuntimeGitCommandHost } from './runtime-git-command-target' +import { + localGitOptionsForTarget, + requireRuntimeGitProvider, + type RuntimeGitCommandHost +} from './runtime-git-command-target' export class RuntimeGitStatusCommands { constructor(private readonly host: RuntimeGitCommandHost) {} @@ -29,11 +29,8 @@ export class RuntimeGitStatusCommands { options?: GitProviderStatusOptions ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return options ? provider.getStatus(target.worktree.path, options) : provider.getStatus(target.worktree.path) @@ -56,11 +53,8 @@ export class RuntimeGitStatusCommands { area: GitStagingArea = 'unstaged' ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getSubmoduleStatus(target.worktree.path, submodulePath, area) } return getGitSubmoduleStatus(target.worktree.path, submodulePath, { @@ -75,11 +69,8 @@ export class RuntimeGitStatusCommands { relativePaths: string[] ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.checkIgnoredPaths(target.worktree.path, relativePaths) } return checkIgnoredPaths(target.worktree.path, relativePaths, { @@ -93,11 +84,8 @@ export class RuntimeGitStatusCommands { options: GitHistoryOptions = {} ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getHistory(target.worktree.path, options) } return getGitHistory(target.worktree.path, { @@ -109,11 +97,8 @@ export class RuntimeGitStatusCommands { async getRuntimeGitConflictOperation(worktreeSelector: string): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.detectConflictOperation(target.worktree.path) } return detectConflictOperation(target.worktree.path, localGitOptionsForTarget(target)) @@ -124,11 +109,8 @@ export class RuntimeGitStatusCommands { branch: string ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.checkoutBranch(target.worktree.path, branch) return { ok: true, branch } } @@ -141,11 +123,8 @@ export class RuntimeGitStatusCommands { async listRuntimeGitLocalBranches(worktreeSelector: string): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.listLocalBranches(target.worktree.path) } return listLocalBranches(target.worktree.path, localGitOptionsForTarget(target)) diff --git a/src/main/runtime/runtime-git-sync-commands.test.ts b/src/main/runtime/runtime-git-sync-commands.test.ts index f662c909cb6..b34aa171788 100644 --- a/src/main/runtime/runtime-git-sync-commands.test.ts +++ b/src/main/runtime/runtime-git-sync-commands.test.ts @@ -59,6 +59,7 @@ describe('RuntimeGitSyncCommands admission', () => { it('prioritizes local runtime git actions and preserves host routing', async () => { const commands = new RuntimeGitSyncCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree, localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -107,7 +108,7 @@ describe('RuntimeGitSyncCommands admission', () => { const commands = new RuntimeGitSyncCommands({ resolveRuntimeGitTarget: async () => ({ worktree, - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/runtime-git-sync-commands.ts b/src/main/runtime/runtime-git-sync-commands.ts index f68b82aac79..542d459f743 100644 --- a/src/main/runtime/runtime-git-sync-commands.ts +++ b/src/main/runtime/runtime-git-sync-commands.ts @@ -5,16 +5,13 @@ import { gitSyncForkDefaultBranch } from '../git/fork-sync' import { gitFastForward, gitFetch, gitPull, gitPullRebaseFromBase, gitPush } from '../git/remote' import { abortMerge, abortRebase, commitChanges } from '../git/status' import { getUpstreamStatus } from '../git/upstream' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { materializeWorktreePushTargetRemote, materializeWorktreePushTargetRemoteSsh } from '../ipc/worktree-remote' import { localGitOptionsForTarget, + requireRuntimeGitProvider, type RuntimeGitCommandHost, type RuntimeGitTarget } from './runtime-git-command-target' @@ -37,11 +34,8 @@ export class RuntimeGitSyncCommands { async abortRuntimeGitMerge(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.abortMerge(target.worktree.path) return { ok: true } } @@ -54,11 +48,8 @@ export class RuntimeGitSyncCommands { async abortRuntimeGitRebase(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.abortRebase(target.worktree.path) return { ok: true } } @@ -74,11 +65,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getUpstreamStatus(target.worktree.path, pushTarget) } return getUpstreamStatus(target.worktree.path, pushTarget, localGitOptionsForTarget(target)) @@ -89,11 +77,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -123,11 +108,8 @@ export class RuntimeGitSyncCommands { expectedUpstream: GitForkSyncExpectedUpstream ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.syncForkDefaultBranch(target.worktree.path, expectedUpstream) } return gitSyncForkDefaultBranch(target.worktree.path, expectedUpstream, { @@ -141,11 +123,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -175,11 +154,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -206,11 +182,8 @@ export class RuntimeGitSyncCommands { async rebaseRuntimeGitFromBase(worktreeSelector: string, baseRef: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.rebaseFromBase(target.worktree.path, baseRef) return { ok: true } } @@ -228,11 +201,8 @@ export class RuntimeGitSyncCommands { forceWithLease?: boolean ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -268,11 +238,8 @@ export class RuntimeGitSyncCommands { throw new Error('Commit message is required') } const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.commit(target.worktree.path, message) } return commitChanges(target.worktree.path, message, { diff --git a/src/main/runtime/runtime-git-target-execution-host.test.ts b/src/main/runtime/runtime-git-target-execution-host.test.ts new file mode 100644 index 00000000000..e4d21c8b477 --- /dev/null +++ b/src/main/runtime/runtime-git-target-execution-host.test.ts @@ -0,0 +1,205 @@ +// `resolveRuntimeGitTarget` read `store.getRepo(worktree.repoId)?.connectionId` and never looked at +// `worktree.hostId`, so one arbitrarily chosen row decided the execution host for ~36 downstream +// Git dispatches. `undefined` there meant "runtime host", "unresolved" and "genuinely local" at +// once (#11163). These cases pin all four answers end to end, through the real SSH provider table. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } +})) + +import type * as GitStatusModule from '../git/status' + +const mocks = vi.hoisted(() => ({ getStatus: vi.fn() })) + +vi.mock('../git/status', async () => ({ + ...(await vi.importActual('../git/status')), + getStatus: mocks.getStatus +})) + +import { ExecutionHostNotDispatchableError } from '../providers/execution-host-provider-dispatch' +import { registerSshGitProvider, unregisterSshGitProvider } from '../providers/ssh-git-dispatch' +import { OrcaRuntimeService } from './orca-runtime' + +const REMOTE_PATH = '/srv/app-feature' +const WORKTREE_ID = 'repo-shared::/srv/app-feature' + +type RuntimeInternals = { + resolveWorktreeSelector: (selector: string) => Promise +} + +function makeRuntime(repos: readonly Record[], hostId?: string) { + const store = { + getSettings: () => ({ disabledTuiAgents: [], workspaceDir: '/tmp/workspaces' }), + getProjectHostSetups: () => [], + getProjects: () => [], + getRepos: () => repos, + getRepo: (id: string) => repos.find((repo) => repo.id === id) + } + const runtime = new OrcaRuntimeService(store as never) + vi.spyOn(runtime as unknown as RuntimeInternals, 'resolveWorktreeSelector').mockResolvedValue({ + id: WORKTREE_ID, + repoId: 'repo-shared', + path: REMOTE_PATH, + git: { path: REMOTE_PATH, branch: 'main', isBare: false, isMainWorktree: false }, + ...(hostId ? { hostId } : {}) + }) + return runtime +} + +function stubProvider() { + return { getStatus: vi.fn().mockResolvedValue({ entries: [] }) } +} + +describe('runtime Git target execution host', () => { + const registered: string[] = [] + + function register(connectionId: string) { + const provider = stubProvider() + registerSshGitProvider(connectionId, provider as never) + registered.push(connectionId) + return provider + } + + beforeEach(() => { + vi.restoreAllMocks() + mocks.getStatus.mockReset().mockResolvedValue({ entries: [] }) + }) + + afterEach(() => { + for (const connectionId of registered.splice(0)) { + unregisterSshGitProvider(connectionId) + } + }) + + // The case whose absence let the original cross-host leak through review: two SSH hosts, and the + // rival row is the one `getRepo` returns first. + it('serves an ssh worktree from the host it names, not from a rival row on another ssh host', async () => { + const openclaw = register('openclaw') + const m4air = register('m4air') + const runtime = makeRuntime( + [ + { id: 'repo-shared', path: '/home/me/app', connectionId: 'openclaw' }, + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' } + ], + 'ssh:m4air' + ) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).toHaveBeenCalledWith(REMOTE_PATH) + expect(openclaw.getStatus).not.toHaveBeenCalled() + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it("routes to the worktree's host even when the only repo row names a different ssh host", async () => { + const openclaw = register('openclaw') + const m4air = register('m4air') + const runtime = makeRuntime( + [{ id: 'repo-shared', path: '/home/me/app', connectionId: 'openclaw' }], + 'ssh:m4air' + ) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).toHaveBeenCalledWith(REMOTE_PATH) + expect(openclaw.getStatus).not.toHaveBeenCalled() + }) + + // `local` has no SSH namespace to nest in, so a surviving `connectionId` is a row contradicting + // itself. The old shape handed it out and dialled a remote host for a local workspace. + it('ignores a stale connection on a row that declares itself local', async () => { + const m4air = register('m4air') + const runtime = makeRuntime( + [ + { + id: 'repo-shared', + path: '/home/me/app', + executionHostId: 'local', + connectionId: 'm4air' + } + ], + 'local' + ) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).not.toHaveBeenCalled() + expect(mocks.getStatus).toHaveBeenCalled() + }) + + // A `runtime:` row's `connectionId` names a target in the *server's* namespace. Dialling it here + // reaches a same-named target on this client — a silent-wrong-host answer, worse than the + // silent-local one it replaced. + it('refuses a runtime host whose nested ssh target is also registered on this client', async () => { + const impostor = register('nested-1') + const runtime = makeRuntime( + [ + { + id: 'repo-shared', + path: '/srv/app', + executionHostId: 'runtime:env-a', + connectionId: 'nested-1' + } + ], + 'runtime:env-a' + ) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + ExecutionHostNotDispatchableError + ) + expect(impostor.getStatus).not.toHaveBeenCalled() + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it('refuses a runtime host with no nested ssh target rather than answering locally', async () => { + const runtime = makeRuntime( + [{ id: 'repo-shared', path: '/srv/app', executionHostId: 'runtime:env-a' }], + 'runtime:env-a' + ) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + ExecutionHostNotDispatchableError + ) + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it('refuses rather than guessing when rival rows disagree and the worktree names no host', async () => { + register('m4air') + const runtime = makeRuntime([ + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }, + { id: 'repo-shared', path: '/home/me/app' } + ]) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + 'worktree_execution_host_unresolved' + ) + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it('still answers from the single row when the worktree names no host', async () => { + const m4air = register('m4air') + const runtime = makeRuntime([{ id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }]) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).toHaveBeenCalledWith(REMOTE_PATH) + }) + + // Losing contact with a remote host is never evidence that its files are here + // (docs/reference/ssh-execution-boundary.md). + it('reports the dropped connection instead of reading a remote path locally', async () => { + const runtime = makeRuntime( + [{ id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }], + 'ssh:m4air' + ) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + /Remote connection dropped/ + ) + expect(mocks.getStatus).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/runtime/worktree-launch-host-repo.test.ts b/src/main/runtime/worktree-launch-host-repo.test.ts index 8a81f2424d1..9525e74b0be 100644 --- a/src/main/runtime/worktree-launch-host-repo.test.ts +++ b/src/main/runtime/worktree-launch-host-repo.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' +import { resolveWorktreeHostRouting, resolveWorktreeLaunchHost } from './worktree-launch-host-repo' // Why (#11163): the terminal launch scope read // `store.getRepo(worktree.repoId)?.connectionId ?? null` — one spelling of one arbitrarily chosen @@ -92,3 +92,56 @@ describe('resolveWorktreeLaunchHost', () => { }) }) }) + +// The same resolution answering "which host is this on" rather than "what may this client dial". +// The runtime Git target needs the first question, because `local` and `runtime:` are two different +// non-SSH answers and only one of them may run here. +describe('resolveWorktreeHostRouting', () => { + const runtimeRow = { + id: 'r', + path: '/p', + connectionId: 'ssh-nested', + executionHostId: 'runtime:env-a' as const + } + + it('keeps `runtime:` distinct from `local` where the launch answer collapses them', () => { + expect( + resolveWorktreeHostRouting([runtimeRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', hostId: 'runtime:env-a', repo: runtimeRow }) + // Both answer "no connection this client may dial"; only the routing view says which host. + expect( + resolveWorktreeLaunchHost([runtimeRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', repo: runtimeRow, connectionId: null }) + }) + + it('answers the host the worktree names over a rival row on another ssh host', () => { + const rows = [ + { id: 'r', path: '/p', connectionId: 'openclaw' }, + { id: 'r', path: '/q', connectionId: 'm4air' } + ] + expect(resolveWorktreeHostRouting(rows, { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + repo: rows[1] + }) + // A row on some other host is not evidence about this one, so it contributes no metadata either. + expect(resolveWorktreeHostRouting([rows[0]], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + repo: null + }) + }) + + it('separates "nobody carries this id" from "rival rows disagree"', () => { + expect(resolveWorktreeHostRouting([], { repoId: 'r' })).toEqual({ kind: 'unowned' }) + expect( + resolveWorktreeHostRouting( + [ + { id: 'r', path: '/p' }, + { id: 'r', path: '/q', connectionId: 'm4air' } + ], + { repoId: 'r' } + ) + ).toEqual({ kind: 'ambiguous' }) + }) +}) diff --git a/src/main/runtime/worktree-launch-host-repo.ts b/src/main/runtime/worktree-launch-host-repo.ts index fa2641655b3..4decb7acb56 100644 --- a/src/main/runtime/worktree-launch-host-repo.ts +++ b/src/main/runtime/worktree-launch-host-repo.ts @@ -3,7 +3,7 @@ import { resolveWorktreeExecutionHost, type ExecutionHostOwnerRow } from '../../shared/worktree-execution-host-resolution' -import { getSshTargetIdForExecutionHost } from '../../shared/execution-host' +import { getSshTargetIdForExecutionHost, type ExecutionHostId } from '../../shared/execution-host' import type { Repo } from '../../shared/repo-types' export type LaunchHostRepo = Pick @@ -12,32 +12,53 @@ export type WorktreeLaunchHostResolution = | { kind: 'resolved'; repo: T | null; connectionId: string | null } | { kind: 'ambiguous' } +export type WorktreeHostRouting = + /** `repo` is metadata; the host is the routing answer. */ + | { kind: 'resolved'; hostId: ExecutionHostId; repo: T | null } + /** No row carries this repo id and the worktree names no host — nothing ever named a host. */ + | { kind: 'unowned' } + /** Rival rows disagree about the host; guessing one is the cross-host leak. */ + | { kind: 'ambiguous' } + /** * Main-side adapter over the shared execution-host rule * (`src/shared/worktree-execution-host-resolution.ts`), which the renderer's owner index answers - * with too. Two things are local to this side: - * - * - rival rows that disagree about the host are `ambiguous` and the launch scope throws, while an - * id nobody carries stays "no repo, no connection" — the launch path's long-standing behaviour - * for a worktree whose repo row has gone; - * - the connection comes off the *host*, not the resolved row. This is a client-dialable PTY - * route, so a `runtime:` host contributes nothing: its nested SSH target belongs to that - * machine's namespace and spawning against it here would dial the wrong box. The renderer wants - * the opposite answer from the same resolution, which is why the shared type carries both. + * with too. What is local to this side is the disposal of the two `unresolved` reasons: rival rows + * that disagree about the host are `ambiguous` and callers throw, while an id nobody carries is + * `unowned` — the launch path's long-standing behaviour for a worktree whose repo row has gone. + */ +export function resolveWorktreeHostRouting( + repos: readonly T[], + worktree: { repoId: string; hostId?: string | null } +): WorktreeHostRouting { + const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) + if (resolution.kind === 'unresolved') { + return resolution.reason === 'ambiguous' ? { kind: 'ambiguous' } : { kind: 'unowned' } + } + return { kind: 'resolved', hostId: resolution.hostId, repo: resolution.owner } +} + +/** + * The same resolution, answering "what may this client dial" rather than "which host is this on". + * The connection comes off the *host*, not the resolved row: this is a client-dialable PTY route, + * so a `runtime:` host contributes nothing — its nested SSH target belongs to that machine's + * namespace and spawning against it here would dial the wrong box. The renderer wants the opposite + * answer from the same resolution, which is why the shared type carries both. */ export function resolveWorktreeLaunchHost( repos: readonly T[], worktree: { repoId: string; hostId?: string | null } ): WorktreeLaunchHostResolution { - const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) - if (resolution.kind === 'unresolved') { - return resolution.reason === 'ambiguous' - ? { kind: 'ambiguous' } - : { kind: 'resolved', repo: null, connectionId: null } + const routing = resolveWorktreeHostRouting(repos, worktree) + if (routing.kind === 'ambiguous') { + return { kind: 'ambiguous' } + } + if (routing.kind === 'unowned') { + return { kind: 'resolved', repo: null, connectionId: null } } return { kind: 'resolved', - repo: resolution.owner, - connectionId: getSshTargetIdForExecutionHost(resolution.hostId) + repo: routing.repo, + connectionId: getSshTargetIdForExecutionHost(routing.hostId) } } From 40d9927f014199ac7a96b669df14c2246a445814 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:52:19 -0700 Subject: [PATCH 25/77] fix(native-chat): show images in Codex structured chat (#18266) * fix(native-chat): render structured image refs from their runtime owner * fix(native-chat): keep transcript image keys stable * fix(native-chat): memoize the image runtime owner * fix(native-chat): keep image preview observation scoped * fix(native-chat): resolve runtime-only image owners * fix(native-chat): retain image preview cache leases --------- Co-authored-by: Merge Sim --- .../editor/local-image-src-cache.ts | 231 ++++++++++++++ .../editor/local-image-src-reader.ts | 23 ++ .../editor/rich-markdown-extensions.ts | 11 +- .../editor/rich-markdown-local-image.test.ts | 26 +- .../editor/useLocalImageSrc.test.ts | 116 +++++++ .../src/components/editor/useLocalImageSrc.ts | 291 ++++++------------ .../native-chat/NativeChatMessageList.tsx | 52 ++-- .../NativeChatStructuredSession.tsx | 3 + .../NativeChatTranscriptChrome.test.tsx | 209 +++++++++++++ .../NativeChatTranscriptChrome.tsx | 235 +++++++++++++- .../NativeChatTypingIndicatorRow.tsx | 21 ++ .../native-chat/native-chat-file-link.test.ts | 22 ++ .../native-chat/native-chat-file-link.ts | 13 +- .../native-chat-image-runtime-context.test.ts | 96 ++++++ .../native-chat-image-runtime-context.ts | 184 +++++++++++ 15 files changed, 1290 insertions(+), 243 deletions(-) create mode 100644 src/renderer/src/components/editor/local-image-src-cache.ts create mode 100644 src/renderer/src/components/editor/local-image-src-reader.ts create mode 100644 src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx create mode 100644 src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx create mode 100644 src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts create mode 100644 src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts diff --git a/src/renderer/src/components/editor/local-image-src-cache.ts b/src/renderer/src/components/editor/local-image-src-cache.ts new file mode 100644 index 00000000000..410ce1c4502 --- /dev/null +++ b/src/renderer/src/components/editor/local-image-src-cache.ts @@ -0,0 +1,231 @@ +import { + clearLocalImageCachePins, + isLocalImageCacheKeyPinned, + pinLocalImageCacheKey, + prunePinnedLocalImageCache, + unpinLocalImageCacheKey +} from './local-image-cache-pinning' + +const BLOB_URL_CACHE_MAX_SIZE = 100 +// Keep retained decoded image data bounded as well as entry count. +const BLOB_URL_CACHE_MAX_BYTES = 128 * 1024 * 1024 + +export const blobUrlCache = new Map() +const blobUrlCacheBytes = new Map() +export const inFlightBlobUrlLoads = new Map>() +// Incremented on release so a read resolving after its last consumer left +// cannot repopulate the cache. +const cacheKeyVersions = new Map() + +export function getLocalImageCacheKeyVersion(key: string): number { + return cacheKeyVersions.get(key) ?? 0 +} + +export function cleanupLocalImageCacheKeyVersion(key: string): void { + if ( + !blobUrlCache.has(key) && + !inFlightBlobUrlLoads.has(key) && + !isLocalImageCacheKeyPinned(key) + ) { + cacheKeyVersions.delete(key) + } +} + +let cacheGeneration = 0 +const cacheListeners = new Set<() => void>() +const pendingBlobUrlRevocations = new Set() +let pendingBlobUrlRevocationTimer: ReturnType | null = null + +function pruneImageCache(): void { + prunePinnedLocalImageCache(blobUrlCache, BLOB_URL_CACHE_MAX_SIZE, (url) => { + URL.revokeObjectURL(url) + }) + for (const key of blobUrlCacheBytes.keys()) { + if (!blobUrlCache.has(key)) { + blobUrlCacheBytes.delete(key) + cleanupLocalImageCacheKeyVersion(key) + } + } + let retainedBytes = 0 + for (const byteLength of blobUrlCacheBytes.values()) { + retainedBytes += byteLength + } + while (retainedBytes > BLOB_URL_CACHE_MAX_BYTES) { + const key = Array.from(blobUrlCache.keys()).find( + (candidate) => !isLocalImageCacheKeyPinned(candidate) + ) + if (key === undefined) { + return + } + const url = blobUrlCache.get(key) + blobUrlCache.delete(key) + retainedBytes -= blobUrlCacheBytes.get(key) ?? 0 + blobUrlCacheBytes.delete(key) + if (url) { + URL.revokeObjectURL(url) + } + cleanupLocalImageCacheKeyVersion(key) + } +} + +export function cacheLocalImageBlob( + key: string, + url: string, + byteLength: number, + expectedVersion?: number +): boolean { + if ( + (expectedVersion !== undefined && getLocalImageCacheKeyVersion(key) !== expectedVersion) || + byteLength > BLOB_URL_CACHE_MAX_BYTES + ) { + URL.revokeObjectURL(url) + cleanupLocalImageCacheKeyVersion(key) + return false + } + const previousUrl = blobUrlCache.get(key) + const previousBytes = blobUrlCacheBytes.get(key) ?? 0 + let retainedBytes = 0 + for (const bytes of blobUrlCacheBytes.values()) { + retainedBytes += bytes + } + let projectedEntries = blobUrlCache.size + (previousUrl === undefined ? 1 : 0) + let projectedBytes = retainedBytes - previousBytes + byteLength + + // Evict only unpinned entries. If visible leases consume the budget, fail + // closed so a late/non-visible decode cannot make retention unbounded. + while (projectedEntries > BLOB_URL_CACHE_MAX_SIZE || projectedBytes > BLOB_URL_CACHE_MAX_BYTES) { + const candidate = Array.from(blobUrlCache.keys()).find( + (candidateKey) => candidateKey !== key && !isLocalImageCacheKeyPinned(candidateKey) + ) + if (candidate === undefined) { + URL.revokeObjectURL(url) + cleanupLocalImageCacheKeyVersion(key) + return false + } + const candidateUrl = blobUrlCache.get(candidate) + const candidateBytes = blobUrlCacheBytes.get(candidate) ?? 0 + blobUrlCache.delete(candidate) + blobUrlCacheBytes.delete(candidate) + projectedEntries -= 1 + projectedBytes -= candidateBytes + if (candidateUrl) { + URL.revokeObjectURL(candidateUrl) + } + cleanupLocalImageCacheKeyVersion(candidate) + } + if (previousUrl !== undefined && previousUrl !== url) { + URL.revokeObjectURL(previousUrl) + } + blobUrlCacheBytes.delete(key) + blobUrlCacheBytes.set(key, byteLength) + blobUrlCache.set(key, url) + return true +} + +export function getLocalImageCacheGeneration(): number { + return cacheGeneration +} + +export function pinLocalImageCache(key: string): void { + pinLocalImageCacheKey(key) +} + +export function unpinLocalImageCache(key: string): void { + unpinLocalImageCacheKey(key) + pruneImageCache() + cleanupLocalImageCacheKeyVersion(key) +} + +export function subscribeToLocalImageCacheInvalidation(listener: () => void): () => void { + cacheListeners.add(listener) + return () => cacheListeners.delete(listener) +} + +function revokePendingBlobUrls(): void { + pendingBlobUrlRevocationTimer = null + for (const url of pendingBlobUrlRevocations) { + URL.revokeObjectURL(url) + } + pendingBlobUrlRevocations.clear() +} + +function scheduleBlobUrlRevocation(urls: string[]): void { + for (const url of urls) { + pendingBlobUrlRevocations.add(url) + } + if (pendingBlobUrlRevocationTimer !== null || pendingBlobUrlRevocations.size === 0) { + return + } + pendingBlobUrlRevocationTimer = setTimeout(revokePendingBlobUrls, 30_000) +} + +export function invalidateLocalImageCache(): void { + const staleUrls = Array.from(blobUrlCache.values()) + blobUrlCache.clear() + blobUrlCacheBytes.clear() + inFlightBlobUrlLoads.clear() + cacheKeyVersions.clear() + cacheGeneration += 1 + for (const listener of cacheListeners) { + listener() + } + if (staleUrls.length > 0) { + scheduleBlobUrlRevocation(staleUrls) + } +} + +export function releaseLocalImageBlob(key: string): void { + if (isLocalImageCacheKeyPinned(key)) { + return + } + cacheKeyVersions.set(key, getLocalImageCacheKeyVersion(key) + 1) + const inFlight = inFlightBlobUrlLoads.get(key) + if (inFlight) { + // A released lease must not be reused by a later visible lease: the + // released read may resolve null or be stale for the next owner. + inFlightBlobUrlLoads.delete(key) + } + const url = blobUrlCache.get(key) + if (url) { + blobUrlCache.delete(key) + blobUrlCacheBytes.delete(key) + URL.revokeObjectURL(url) + } + if (!inFlight) { + cleanupLocalImageCacheKeyVersion(key) + } +} + +export function resetLocalImageCacheState(): void { + if (pendingBlobUrlRevocationTimer !== null) { + clearTimeout(pendingBlobUrlRevocationTimer) + pendingBlobUrlRevocationTimer = null + } + revokePendingBlobUrls() + for (const url of blobUrlCache.values()) { + URL.revokeObjectURL(url) + } + blobUrlCache.clear() + blobUrlCacheBytes.clear() + clearLocalImageCachePins() + inFlightBlobUrlLoads.clear() + cacheKeyVersions.clear() + cacheGeneration = 0 + pendingBlobUrlRevocations.clear() + cacheListeners.clear() +} + +export function disposeLocalImageCacheState(): void { + if (typeof window !== 'undefined') { + window.removeEventListener('focus', invalidateLocalImageCache) + } + resetLocalImageCacheState() +} + +if (typeof window !== 'undefined') { + window.addEventListener('focus', invalidateLocalImageCache) +} + +if (import.meta !== undefined && import.meta.hot) { + import.meta.hot.dispose(disposeLocalImageCacheState) +} diff --git a/src/renderer/src/components/editor/local-image-src-reader.ts b/src/renderer/src/components/editor/local-image-src-reader.ts new file mode 100644 index 00000000000..5e845031bd0 --- /dev/null +++ b/src/renderer/src/components/editor/local-image-src-reader.ts @@ -0,0 +1,23 @@ +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { readRuntimeFilePreview } from '@/runtime/runtime-file-client' + +export function readLocalImagePreview( + absolutePath: string, + connectionId?: string | null, + runtimeContext?: Omit & { connectionId?: string | null } +) { + try { + if (!runtimeContext) { + return window.api.fs.readFile({ + filePath: absolutePath, + connectionId: connectionId ?? undefined + }) + } + return readRuntimeFilePreview( + { ...runtimeContext, connectionId: runtimeContext.connectionId ?? connectionId ?? undefined }, + absolutePath + ) + } catch (error) { + return Promise.reject(error) + } +} diff --git a/src/renderer/src/components/editor/rich-markdown-extensions.ts b/src/renderer/src/components/editor/rich-markdown-extensions.ts index 9876f2871cc..42904291a23 100644 --- a/src/renderer/src/components/editor/rich-markdown-extensions.ts +++ b/src/renderer/src/components/editor/rich-markdown-extensions.ts @@ -12,7 +12,11 @@ import { TableRow } from '@tiptap/extension-table-row' import { BlockMath, InlineMath } from '@tiptap/extension-mathematics' import { Markdown } from '@tiptap/markdown' import { createLowlight, common } from 'lowlight' -import { loadLocalImageSrc, onImageCacheInvalidated } from './useLocalImageSrc' +import { + acquireLocalImageSrcLease, + loadLocalImageSrc, + onImageCacheInvalidated +} from './useLocalImageSrc' import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' import { createRawMarkdownHtmlBlock, @@ -126,14 +130,18 @@ export function createRichMarkdownExtensions({ let currentSrc = node.attrs.src as string | undefined let currentContextVersion = getImageContextVersion(this.storage) + let releaseImageLease: (() => void) | undefined const loadImage = (src: string | undefined): void => { + releaseImageLease?.() + releaseImageLease = undefined const fp = this.storage.filePath as string const runtimeContext = this.storage.runtimeContext as | RuntimeFileOperationArgs | undefined const contextVersionAtLoad = getImageContextVersion(this.storage) if (src && fp) { + releaseImageLease = acquireLocalImageSrcLease(src, fp, undefined, runtimeContext) void loadLocalImageSrc(src, fp, undefined, runtimeContext).then((resolved) => { if (currentSrc !== src || currentContextVersion !== contextVersionAtLoad) { return @@ -187,6 +195,7 @@ export function createRichMarkdownExtensions({ return true }, destroy: () => { + releaseImageLease?.() if (reloadListeners instanceof Set) { reloadListeners.delete(reloadForContextChange) } diff --git a/src/renderer/src/components/editor/rich-markdown-local-image.test.ts b/src/renderer/src/components/editor/rich-markdown-local-image.test.ts index e343e918619..bcd4e64c6f1 100644 --- a/src/renderer/src/components/editor/rich-markdown-local-image.test.ts +++ b/src/renderer/src/components/editor/rich-markdown-local-image.test.ts @@ -4,7 +4,7 @@ import { Editor } from '@tiptap/core' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { createRichMarkdownExtensions } from './rich-markdown-extensions' import { createRichMarkdownEditorCodec } from './rich-markdown-source-transport' -import { resetLocalImageSrcStateForTests } from './useLocalImageSrc' +import { releaseLocalImageSrc, resetLocalImageSrcStateForTests } from './useLocalImageSrc' import { setRichMarkdownImageResolverContext } from './rich-markdown-image-context' async function flushPromises(): Promise { @@ -18,6 +18,7 @@ describe('rich markdown local images', () => { beforeEach(() => { resetLocalImageSrcStateForTests() vi.spyOn(URL, 'createObjectURL').mockReturnValue('blob:rich-local-image') + vi.spyOn(URL, 'revokeObjectURL').mockImplementation(() => undefined) globalThis.window.api = { ...globalThis.window.api, fs: { @@ -64,4 +65,27 @@ describe('rich markdown local images', () => { editor.destroy() } }) + + it('keeps a displayed image leased when another surface releases the same cache entry', async () => { + const host = document.createElement('div') + document.body.appendChild(host) + const editor = new Editor({ + element: host, + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content: '![](diagram.png)', + contentType: 'markdown' + }) + + try { + setRichMarkdownImageResolverContext(editor, { filePath: '/repo/docs/readme.md' }) + await flushPromises() + + releaseLocalImageSrc('diagram.png', '/repo/docs/readme.md') + + expect(URL.revokeObjectURL).not.toHaveBeenCalledWith('blob:rich-local-image') + expect(host.querySelector('img')?.src).toBe('blob:rich-local-image') + } finally { + editor.destroy() + } + }) }) diff --git a/src/renderer/src/components/editor/useLocalImageSrc.test.ts b/src/renderer/src/components/editor/useLocalImageSrc.test.ts index eaf5b031928..1c93693ff1c 100644 --- a/src/renderer/src/components/editor/useLocalImageSrc.test.ts +++ b/src/renderer/src/components/editor/useLocalImageSrc.test.ts @@ -7,9 +7,16 @@ import { getLocalImageCacheKey, invalidateLocalImageSrcCacheForTests, loadLocalImageSrc, + releaseLocalImageSrc, resetLocalImageSrcStateForTests, useLocalImageSrc } from './useLocalImageSrc' +import { + blobUrlCache, + cacheLocalImageBlob, + getLocalImageCacheKeyVersion, + pinLocalImageCache +} from './local-image-src-cache' type PreviewResult = { content: string @@ -128,6 +135,37 @@ describe('loadLocalImageSrc', () => { expect(URL.createObjectURL).toHaveBeenCalledTimes(1) }) + it('lets a mounted preview adopt an in-flight prewarm read', async () => { + const read = deferred() + const readFile = vi.fn().mockReturnValue(read.promise) + const renders: (string | undefined)[] = [] + vi.spyOn(URL, 'createObjectURL').mockReturnValue('blob:prewarmed') + setReadFile(readFile) + + const prewarm = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + const container = document.createElement('div') + const root: Root = createRoot(container) + await act(async () => { + root.render( + createElement(HookProbe, { + filePath: '/repo/docs/readme.md', + onRender: (displaySrc) => renders.push(displaySrc), + src: 'diagram.png' + }) + ) + }) + expect(readFile).toHaveBeenCalledTimes(1) + + await act(async () => { + read.resolve(binaryPreview()) + await flushPromises() + }) + + await expect(prewarm).resolves.toBe('blob:prewarmed') + expect(renders.at(-1)).toBe('blob:prewarmed') + root.unmount() + }) + it('does not revoke blob URLs still used by mounted previews during eviction', async () => { const readFile = vi.fn().mockResolvedValue(binaryPreview()) let nextUrl = 0 @@ -230,6 +268,84 @@ describe('loadLocalImageSrc', () => { expect(URL.revokeObjectURL).not.toHaveBeenCalledWith('blob:newer') }) + it('does not retain a read that resolves after its preview lease is released', async () => { + const read = deferred() + const readFile = vi.fn().mockReturnValue(read.promise) + vi.spyOn(URL, 'createObjectURL').mockReturnValue('blob:released') + setReadFile(readFile) + + const pending = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + releaseLocalImageSrc('diagram.png', '/repo/docs/readme.md') + read.resolve(binaryPreview()) + + await expect(pending).resolves.toBeNull() + expect(blobUrlCache.size).toBe(0) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:released') + }) + + it('starts a fresh read when a released lease becomes visible again', async () => { + const firstRead = deferred() + const secondRead = deferred() + const readFile = vi + .fn() + .mockReturnValueOnce(firstRead.promise) + .mockReturnValueOnce(secondRead.promise) + vi.spyOn(URL, 'createObjectURL') + .mockReturnValueOnce('blob:fresh') + .mockReturnValueOnce('blob:stale') + setReadFile(readFile) + + const stale = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + releaseLocalImageSrc('diagram.png', '/repo/docs/readme.md') + const fresh = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + expect(readFile).toHaveBeenCalledTimes(2) + + secondRead.resolve(binaryPreview('AQ==')) + await expect(fresh).resolves.toBe('blob:fresh') + firstRead.resolve(binaryPreview()) + await expect(stale).resolves.toBeNull() + expect(blobUrlCache.size).toBe(1) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:stale') + }) + + it('cleans version metadata for released unique paths', () => { + for (let index = 0; index < 500; index += 1) { + const path = `/repo/docs/image-${index}.png` + releaseLocalImageSrc(path, '/repo/docs/readme.md') + expect(getLocalImageCacheKeyVersion(getLocalImageCacheKey(path, undefined, undefined))).toBe( + 0 + ) + } + }) + + it('fails closed when pinned previews already consume the entry or byte budget', () => { + for (let index = 0; index < 100; index += 1) { + const key = `pinned-${index}` + pinLocalImageCache(key) + expect(cacheLocalImageBlob(key, `blob:${index}`, 1)).toBe(true) + } + + expect(cacheLocalImageBlob('pinned-overflow', 'blob:overflow', 1)).toBe(false) + expect(blobUrlCache.size).toBe(100) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:overflow') + + expect( + cacheLocalImageBlob('large-overflow', 'blob:large-overflow', 128 * 1024 * 1024 + 1) + ).toBe(false) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:large-overflow') + }) + + it('does not exceed the decoded-byte budget when all retained entries are pinned', () => { + const retainedBytes = 80 * 1024 * 1024 + pinLocalImageCache('large-pinned-1') + pinLocalImageCache('large-pinned-2') + expect(cacheLocalImageBlob('large-pinned-1', 'blob:large-1', retainedBytes)).toBe(true) + expect(cacheLocalImageBlob('large-pinned-2', 'blob:large-2', retainedBytes)).toBe(false) + + expect(blobUrlCache.size).toBe(1) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:large-2') + }) + it('keeps runtime owners in separate image cache entries', async () => { const readFile = vi.fn().mockResolvedValue(binaryPreview()) vi.spyOn(URL, 'createObjectURL') diff --git a/src/renderer/src/components/editor/useLocalImageSrc.ts b/src/renderer/src/components/editor/useLocalImageSrc.ts index bd4cccd9e16..38339ee36ea 100644 --- a/src/renderer/src/components/editor/useLocalImageSrc.ts +++ b/src/renderer/src/components/editor/useLocalImageSrc.ts @@ -1,22 +1,21 @@ import { useEffect, useState } from 'react' import { resolveImageAbsolutePath } from './markdown-preview-links' import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' -import { readRuntimeFilePreview } from '@/runtime/runtime-file-client' +import { readLocalImagePreview } from './local-image-src-reader' import { - clearLocalImageCachePins, - pinLocalImageCacheKey, - prunePinnedLocalImageCache, - unpinLocalImageCacheKey -} from './local-image-cache-pinning' - -// Why: the renderer is served from http://localhost in dev mode, so file:// -// URLs in tags are blocked by cross-origin restrictions. Loading images -// via the existing fs.readFile IPC and converting to blob URLs bypasses this -// limitation and works identically in both dev and production modes. - -const BLOB_URL_CACHE_MAX_SIZE = 100 -const blobUrlCache = new Map() -const inFlightBlobUrlLoads = new Map>() + blobUrlCache, + cacheLocalImageBlob, + cleanupLocalImageCacheKeyVersion, + getLocalImageCacheGeneration, + getLocalImageCacheKeyVersion, + inFlightBlobUrlLoads, + invalidateLocalImageCache, + pinLocalImageCache, + releaseLocalImageBlob, + resetLocalImageCacheState, + subscribeToLocalImageCacheInvalidation, + unpinLocalImageCache +} from './local-image-src-cache' export function getLocalImageCacheKey( absolutePath: string, @@ -28,131 +27,32 @@ export function getLocalImageCacheKey( return [ runtimeEnvironmentId, runtimeContext?.connectionId ?? connectionId ?? 'local', + runtimeContext?.expectedExecutionHostId ?? 'unknown-host', + runtimeContext?.expectedSshTargetId ?? '', + runtimeContext?.expectedSshConnectionGeneration?.toString() ?? '', runtimeContext?.expectedExternalSshTargetId ?? '', runtimeContext?.worktreeId ?? 'unknown-worktree', + runtimeContext?.worktreePath ?? '', absolutePath ].join('\0') } -// Why: blob URLs hold references to in-memory Blob objects; without eviction -// the cache grows without bound and leaks memory. We evict the oldest entry -// (Map iteration order is insertion order) and revoke its blob URL so the -// browser can free the underlying data. -function cacheBlobUrl(key: string, url: string): void { - const previousUrl = blobUrlCache.get(key) - if (previousUrl !== undefined) { - blobUrlCache.delete(key) - if (previousUrl !== url) { - // Why: cache replacements must release the superseded Blob even when - // they come from rare stale state or future loader changes. - URL.revokeObjectURL(previousUrl) - } - } - blobUrlCache.set(key, url) - prunePinnedLocalImageCache(blobUrlCache, BLOB_URL_CACHE_MAX_SIZE, URL.revokeObjectURL) -} - -const cacheListeners = new Set<() => void>() -let cacheGeneration = 0 -const pendingBlobUrlRevocations = new Set() -let pendingBlobUrlRevocationTimer: ReturnType | null = null - -function base64ToBlobUrl(base64: string, mimeType: string): string { +function base64ToBlobUrl(base64: string, mimeType: string): { url: string; byteLength: number } { const binary = atob(base64.replace(/\s/g, '')) const bytes = new Uint8Array(binary.length) for (let i = 0; i < binary.length; i += 1) { bytes[i] = binary.charCodeAt(i) } - return URL.createObjectURL(new Blob([bytes], { type: mimeType })) -} - -function revokePendingBlobUrls(): void { - pendingBlobUrlRevocationTimer = null - for (const url of pendingBlobUrlRevocations) { - URL.revokeObjectURL(url) - } - pendingBlobUrlRevocations.clear() -} - -function scheduleBlobUrlRevocation(urls: string[]): void { - for (const url of urls) { - pendingBlobUrlRevocations.add(url) - } - if (pendingBlobUrlRevocationTimer !== null || pendingBlobUrlRevocations.size === 0) { - return - } - pendingBlobUrlRevocationTimer = setTimeout(revokePendingBlobUrls, 30_000) -} - -// Why: when the user switches back to the app after deleting or replacing -// image files externally, clearing the cache forces the preview to pick up -// the current filesystem state instead of showing stale in-memory blob URLs. -// Old blob URLs are revoked after a short delay so that elements still -// display the old data while the fresh IPC load completes, avoiding a visible -// flash. The 30-second window is generous enough for even slow IPC reads. -function invalidateImageCache(): void { - const staleUrls = Array.from(blobUrlCache.values()) - blobUrlCache.clear() - inFlightBlobUrlLoads.clear() - cacheGeneration += 1 - for (const listener of cacheListeners) { - listener() - } - // Why: defer revocation so the browser keeps the old blob data readable - // until replacement IPC loads complete, then free the underlying memory. - // 30 seconds is generous enough to cover slow machines or large images - // without risking a visible broken-image flash. - if (staleUrls.length > 0) { - scheduleBlobUrlRevocation(staleUrls) + return { + url: URL.createObjectURL(new Blob([bytes], { type: mimeType })), + byteLength: bytes.byteLength } } -function disposeImageCacheModuleState(): void { - if (typeof window !== 'undefined') { - window.removeEventListener('focus', invalidateImageCache) - } - if (pendingBlobUrlRevocationTimer !== null) { - clearTimeout(pendingBlobUrlRevocationTimer) - pendingBlobUrlRevocationTimer = null - } - revokePendingBlobUrls() - for (const url of blobUrlCache.values()) { - URL.revokeObjectURL(url) - } - blobUrlCache.clear() - clearLocalImageCachePins() - inFlightBlobUrlLoads.clear() - cacheListeners.clear() -} - -if (typeof window !== 'undefined') { - window.addEventListener('focus', invalidateImageCache) -} - -if (import.meta !== undefined && import.meta.hot) { - // Why: Vite can re-evaluate this module without a full renderer reload. - // Disposing the module-level listener and blob URLs prevents dev-session leaks. - import.meta.hot.dispose(disposeImageCacheModuleState) -} - -/** - * Subscribe to cache invalidation events (fired on window re-focus). - * Returns an unsubscribe function. - */ -export function onImageCacheInvalidated(listener: () => void): () => void { - cacheListeners.add(listener) - return () => { - cacheListeners.delete(listener) - } -} +export const onImageCacheInvalidated = subscribeToLocalImageCacheInvalidation function isExternalUrl(src: string): boolean { - return ( - src.startsWith('http://') || - src.startsWith('https://') || - src.startsWith('data:') || - src.startsWith('blob:') - ) + return /^(?:https?|data|blob):/i.test(src) } /** @@ -165,32 +65,22 @@ export function useLocalImageSrc( rawSrc: string | undefined, filePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null ): string | undefined { - const [generation, setGeneration] = useState(cacheGeneration) + const [generation, setGeneration] = useState(getLocalImageCacheGeneration()) useEffect(() => { - if (!rawSrc || isExternalUrl(rawSrc)) { - return - } - const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) - if (!absolutePath) { - return - } - const cacheKey = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) - pinLocalImageCacheKey(cacheKey) - return () => { - unpinLocalImageCacheKey(cacheKey) - prunePinnedLocalImageCache(blobUrlCache, BLOB_URL_CACHE_MAX_SIZE, URL.revokeObjectURL) - } + return acquireLocalImageSrcLease(rawSrc, filePath, connectionId, runtimeContext) }, [rawSrc, filePath, connectionId, runtimeContext]) useEffect(() => { - return onImageCacheInvalidated(() => setGeneration(cacheGeneration)) + return onImageCacheInvalidated(() => setGeneration(getLocalImageCacheGeneration())) }, []) const [displaySrc, setDisplaySrc] = useState(() => { - if (!rawSrc) { + if (!rawSrc || runtimeContext === null) { return undefined } if (isExternalUrl(rawSrc)) { @@ -207,7 +97,7 @@ export function useLocalImageSrc( }) useEffect(() => { - if (!rawSrc) { + if (!rawSrc || runtimeContext === null) { setDisplaySrc(undefined) return } @@ -236,7 +126,7 @@ export function useLocalImageSrc( if (cancelled) { return } - setDisplaySrc(cacheGeneration === effectGeneration && url ? url : undefined) + setDisplaySrc(getLocalImageCacheGeneration() === effectGeneration && url ? url : undefined) }) .catch(() => { if (!cancelled) { @@ -261,16 +151,16 @@ export async function loadLocalImageSrc( rawSrc: string, filePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null ): Promise { - if ( - rawSrc.startsWith('http://') || - rawSrc.startsWith('https://') || - rawSrc.startsWith('data:') || - rawSrc.startsWith('blob:') - ) { + if (isExternalUrl(rawSrc)) { return rawSrc } + if (runtimeContext === null) { + return null + } const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) if (!absolutePath) { @@ -289,8 +179,13 @@ export async function loadLocalImageSrc( export function loadLocalImageAbsolutePath( absolutePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null ): Promise { + if (runtimeContext === null) { + return Promise.resolve(null) + } const cacheKey = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) const cached = blobUrlCache.get(cacheKey) if (cached) { @@ -302,73 +197,79 @@ export function loadLocalImageAbsolutePath( return inFlight } - const readGeneration = cacheGeneration - const loadPromise = readImagePreview(absolutePath, connectionId, runtimeContext) + const readGeneration = getLocalImageCacheGeneration() + const readLeaseVersion = getLocalImageCacheKeyVersion(cacheKey) + const loadPromise = readLocalImagePreview(absolutePath, connectionId, runtimeContext) .then((result) => { - if (!result.isBinary || !result.content || cacheGeneration !== readGeneration) { - // Why: local image paths must stay behind IPC/runtime authorization; - // handing raw file: or relative paths back to Chromium can escape it. + if ( + !result.isBinary || + !result.content || + getLocalImageCacheGeneration() !== readGeneration + ) { return null } - const url = base64ToBlobUrl(result.content, result.mimeType ?? 'image/png') - if (cacheGeneration !== readGeneration) { + const { url, byteLength } = base64ToBlobUrl(result.content, result.mimeType ?? 'image/png') + if (getLocalImageCacheGeneration() !== readGeneration) { URL.revokeObjectURL(url) return null } - cacheBlobUrl(cacheKey, url) - return url + return cacheLocalImageBlob(cacheKey, url, byteLength, readLeaseVersion) ? url : null }) .catch(() => null) .finally(() => { if (inFlightBlobUrlLoads.get(cacheKey) === loadPromise) { inFlightBlobUrlLoads.delete(cacheKey) } + cleanupLocalImageCacheKeyVersion(cacheKey) }) inFlightBlobUrlLoads.set(cacheKey, loadPromise) return loadPromise } export function resetLocalImageSrcStateForTests(): void { - if (pendingBlobUrlRevocationTimer !== null) { - clearTimeout(pendingBlobUrlRevocationTimer) - pendingBlobUrlRevocationTimer = null - } - revokePendingBlobUrls() - for (const url of blobUrlCache.values()) { - URL.revokeObjectURL(url) - } - blobUrlCache.clear() - clearLocalImageCachePins() - inFlightBlobUrlLoads.clear() - cacheGeneration = 0 - pendingBlobUrlRevocations.clear() - cacheListeners.clear() + resetLocalImageCacheState() } export function invalidateLocalImageSrcCacheForTests(): void { - invalidateImageCache() + invalidateLocalImageCache() } -function readImagePreview( - absolutePath: string, +export function acquireLocalImageSrcLease( + rawSrc: string | undefined, + filePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } -) { - try { - if (!runtimeContext) { - return window.api.fs.readFile({ - filePath: absolutePath, - connectionId: connectionId ?? undefined - }) - } - return readRuntimeFilePreview( - { - ...runtimeContext, - connectionId: runtimeContext.connectionId ?? connectionId ?? undefined - }, - absolutePath - ) - } catch (error) { - return Promise.reject(error) + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null +): (() => void) | undefined { + if (!rawSrc || isExternalUrl(rawSrc) || runtimeContext === null) { + return undefined } + const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) + if (!absolutePath) { + return undefined + } + const key = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) + pinLocalImageCache(key) + return () => unpinLocalImageCache(key) +} + +/** Evict one no-longer-visible transcript preview immediately. */ +export function releaseLocalImageSrc( + rawSrc: string, + filePath: string, + connectionId?: string | null, + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null +): void { + if (!rawSrc || isExternalUrl(rawSrc) || runtimeContext === null) { + return + } + const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) + if (!absolutePath) { + return + } + const key = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) + releaseLocalImageBlob(key) } diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx index 988db65f730..12b10e0f714 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx @@ -16,34 +16,16 @@ import { shouldShowNativeChatTypingIndicator } from './native-chat-typing-indica import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' import { useNativeChatTurnStatus } from './use-native-chat-turn-status' import { nativeChatProseToMarkdown } from './native-chat-prose' +import { NativeChatTypingIndicatorRow } from './NativeChatTypingIndicatorRow' import { NativeChatAgentControls, NativeChatImageAttachments, ProviderFrameRow } from './NativeChatTranscriptChrome' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' export { ProviderFrameRow } from './NativeChatTranscriptChrome' -function TypingIndicatorRow(): React.JSX.Element { - return ( -
-
- {[0, 1, 2].map((i) => ( - - ))} -
-
- ) -} - function geometryOf(el: HTMLElement): ScrollGeometry { return { scrollTop: el.scrollTop, scrollHeight: el.scrollHeight, clientHeight: el.clientHeight } } @@ -62,7 +44,8 @@ function MessageRow({ allowFileUriLinks = false, deliveryFailed = false, activityExpandOverride, - structuredActivityUi = true + structuredActivityUi = true, + runtimeContext }: { message: NativeChatMessage expandSignal: boolean @@ -74,6 +57,7 @@ function MessageRow({ deliveryFailed?: boolean activityExpandOverride?: boolean structuredActivityUi?: boolean + runtimeContext?: RuntimeFileOperationArgs | null }): React.JSX.Element | null { const rowRef = useRef(null) const { prose, tools } = useMemo(() => splitNativeChatBlocks(message.blocks), [message.blocks]) @@ -113,7 +97,11 @@ function MessageRow({
{markdown ? ( <> - + ) : ( - + )}
{deliveryFailed ? ( @@ -152,7 +144,11 @@ function MessageRow({ isSystem && 'text-xs text-muted-foreground' )} > - + {markdown ? ( /** Turn timing/disclosure is available only on the structured Codex lane. */ showTurnStatus?: boolean + runtimeContext?: RuntimeFileOperationArgs | null }): React.JSX.Element { const scrollRef = useRef(null) const contentRef = useRef(null) @@ -231,7 +229,6 @@ export function NativeChatMessageList({ const stuckToBottomRef = useRef(stuckToBottom) stuckToBottomRef.current = stuckToBottom - const { hasMore, loadingEarlier, loadEarlier } = session // Keep hidden harness turns as fold boundaries, then strip them before render. @@ -401,6 +398,7 @@ export function NativeChatMessageList({ deliveryFailed={failedDeliveryMessageIds?.has(message.id) === true} structuredActivityUi={showTurnStatus} activityExpandOverride={turnKey ? expandedTurnIds.has(turnKey) : undefined} + runtimeContext={runtimeContext} /> {showTurnStatus && status && @@ -430,7 +428,7 @@ export function NativeChatMessageList({ workedSeconds={turnStatuses.active.workedSeconds} /> ) : null} - {!showTurnStatus && showTypingIndicator ? : null} + {!showTurnStatus && showTypingIndicator ? : null} {showJump ? ( diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx index 7a02829a05a..f7d2fd62663 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx @@ -22,6 +22,7 @@ import { useStructuredAgentSession } from './use-structured-agent-session' import { translate } from '@/i18n/i18n' import { NativeChatOrchestrationPausedNotice } from './NativeChatOrchestrationPausedNotice' import { useNativeChatPasteBridge } from './use-native-chat-paste-bridge' +import { useNativeChatImageRuntimeContext } from './native-chat-image-runtime-context' function encodeQuestionAnswer(questionId: string, answer: string): string { return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` @@ -80,6 +81,7 @@ export function NativeChatStructuredSession(props: { const viewState = selectNativeChatViewState(session) const fontScale = useNativeChatFontScale(viewState.kind === 'ready') const fileLinkContext = useNativeChatFileLinkContext(props.tabId) + const imageRuntimeContext = useNativeChatImageRuntimeContext(props.tabId) const fileLinkClick = useNativeChatFileLinkClick(props.allowFileUriLinks ? fileLinkContext : null) const prompt = controller.prompts[0] ?? null const questionBody = prompt?.body.kind === 'question' ? prompt.body : null @@ -145,6 +147,7 @@ export function NativeChatStructuredSession(props: { showTurnStatus={props.agent === 'codex'} onLinkClick={fileLinkClick} allowFileUriLinks={fileLinkClick !== undefined} + runtimeContext={props.agent === 'codex' ? imageRuntimeContext : undefined} /> )} diff --git a/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx new file mode 100644 index 00000000000..3992bbce1f5 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx @@ -0,0 +1,209 @@ +// @vitest-environment happy-dom + +import { act, createElement } from 'react' +import { createRoot } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { + invalidateLocalImageSrcCacheForTests, + resetLocalImageSrcStateForTests +} from '@/components/editor/useLocalImageSrc' +import { NativeChatImageAttachments } from './NativeChatTranscriptChrome' + +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +function runtimeContext(worktreeId: string): RuntimeFileOperationArgs { + return { + settings: { activeRuntimeEnvironmentId: null }, + worktreeId, + worktreePath: `/repo/${worktreeId}`, + expectedExecutionHostId: 'local' + } +} + +async function flushPromises(): Promise { + await Promise.resolve() + await Promise.resolve() +} + +beforeEach(() => { + resetLocalImageSrcStateForTests() + vi.stubGlobal('IntersectionObserver', undefined) + let urlSequence = 0 + vi.spyOn(URL, 'createObjectURL').mockImplementation(() => `blob:owner-${++urlSequence}`) + vi.spyOn(URL, 'revokeObjectURL').mockImplementation(() => undefined) + window.api = { + fs: { + readFile: vi.fn().mockResolvedValue({ + content: 'AA==', + isBinary: true, + mimeType: 'image/png' + }) + } + } as unknown as Window['api'] +}) + +afterEach(() => { + resetLocalImageSrcStateForTests() + vi.unstubAllGlobals() + vi.restoreAllMocks() +}) + +describe('NativeChatImageAttachments', () => { + it('pools visibility observation across image refs', async () => { + class FakeIntersectionObserver { + static instances: FakeIntersectionObserver[] = [] + readonly observe = vi.fn() + readonly unobserve = vi.fn() + readonly disconnect = vi.fn() + + constructor(_callback: IntersectionObserverCallback) { + FakeIntersectionObserver.instances.push(this) + } + } + vi.stubGlobal('IntersectionObserver', FakeIntersectionObserver) + + const container = document.createElement('div') + const root = createRoot(container) + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks: [ + { type: 'image-ref' as const, path: '/repo/one.png' }, + { type: 'image-ref' as const, path: '/repo/two.png' }, + { type: 'image-ref' as const, path: '/repo/three.png' } + ], + runtimeContext: runtimeContext('wt-1') + }) + ) + await flushPromises() + }) + + expect(FakeIntersectionObserver.instances).toHaveLength(1) + expect(FakeIntersectionObserver.instances[0]?.observe).toHaveBeenCalledTimes(3) + + root.unmount() + expect(FakeIntersectionObserver.instances[0]?.unobserve).toHaveBeenCalledTimes(3) + expect(FakeIntersectionObserver.instances[0]?.disconnect).toHaveBeenCalledOnce() + }) + + it('preserves same-image errors but retries when the runtime owner changes', async () => { + const container = document.createElement('div') + const root = createRoot(container) + const blocks = [{ type: 'image-ref' as const, path: '/repo/image.png' }] + const ownerOne = runtimeContext('wt-1') + + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks, + runtimeContext: ownerOne + }) + ) + await flushPromises() + }) + const firstOwnerSrc = container.querySelector('img')?.getAttribute('src') + expect(firstOwnerSrc).toBe('blob:owner-1') + + await act(async () => { + container.querySelector('img')?.dispatchEvent(new Event('error')) + }) + expect(container.querySelector('img')).toBeNull() + + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks, + runtimeContext: ownerOne + }) + ) + await flushPromises() + }) + expect(container.querySelector('img')).toBeNull() + + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks, + runtimeContext: runtimeContext('wt-2') + }) + ) + await flushPromises() + }) + expect(container.querySelector('img')?.getAttribute('src')).not.toBe(firstOwnerSrc) + expect(window.api.fs.readFile).toHaveBeenCalledTimes(2) + + root.unmount() + }) + + it('retries a failed thumbnail after the image cache refreshes', async () => { + const container = document.createElement('div') + const root = createRoot(container) + const props = { + blocks: [{ type: 'image-ref' as const, path: '/repo/image.png' }], + runtimeContext: runtimeContext('wt-1') + } + + await act(async () => { + root.render(createElement(NativeChatImageAttachments, props)) + await flushPromises() + }) + expect(container.querySelector('img')?.getAttribute('src')).toBe('blob:owner-1') + + await act(async () => { + container.querySelector('img')?.dispatchEvent(new Event('error')) + }) + expect(container.querySelector('img')).toBeNull() + + await act(async () => { + invalidateLocalImageSrcCacheForTests() + await flushPromises() + }) + + expect(container.querySelector('img')?.getAttribute('src')).toBe('blob:owner-2') + root.unmount() + }) + + it('keeps the observed element stable while a preview is materialized', async () => { + let callback: IntersectionObserverCallback | undefined + class FakeIntersectionObserver { + readonly observe = vi.fn() + readonly unobserve = vi.fn() + readonly disconnect = vi.fn() + + constructor(nextCallback: IntersectionObserverCallback) { + callback = nextCallback + } + } + vi.stubGlobal('IntersectionObserver', FakeIntersectionObserver) + + const container = document.createElement('div') + const root = createRoot(container) + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks: [{ type: 'image-ref' as const, path: '/repo/image.png' }], + runtimeContext: runtimeContext('wt-1') + }) + ) + await flushPromises() + }) + + const observedElement = container.firstElementChild + expect(observedElement).not.toBeNull() + if (!observedElement || !callback) { + throw new Error('image preview did not register visibility observation') + } + const notifyVisibility = callback + await act(async () => { + notifyVisibility( + [{ target: observedElement, isIntersecting: true } as IntersectionObserverEntry], + {} as IntersectionObserver + ) + await flushPromises() + }) + + expect(container.firstElementChild).toBe(observedElement) + root.unmount() + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx index 206f0f8df5e..cc951002431 100644 --- a/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx +++ b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx @@ -1,40 +1,241 @@ +import { useEffect, useRef, useState } from 'react' import { ArrowUp, Image as ImageIcon } from 'lucide-react' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' import { basename } from '@/lib/path' import type { NativeChatBlock } from '../../../../shared/native-chat-types' -import { isNativeChatPastedImagePath } from './native-chat-image-paste' import { NativeChatCopyButton } from './NativeChatCopyButton' import { nativeChatProviderFrameSummary } from '../../../../shared/native-chat-provider-frame-summary' +import { Dialog, DialogContent, DialogDescription, DialogTitle } from '@/components/ui/dialog' +import { + getLocalImageCacheKey, + useLocalImageSrc, + releaseLocalImageSrc +} from '@/components/editor/useLocalImageSrc' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { isNativeChatPastedImagePath } from './native-chat-image-paste' + +type VisibilityListener = (isVisible: boolean) => void + +const visibilityListeners = new Map() +let visibilityObserver: IntersectionObserver | null = null + +function observeTranscriptVisibility(element: Element, listener: VisibilityListener): () => void { + if (typeof IntersectionObserver === 'undefined') { + listener(true) + return () => {} + } + + visibilityObserver ??= new IntersectionObserver( + (entries) => { + for (const entry of entries) { + visibilityListeners.get(entry.target)?.(entry.isIntersecting) + } + }, + { rootMargin: '128px' } + ) + visibilityListeners.set(element, listener) + visibilityObserver.observe(element) + + return () => { + visibilityListeners.delete(element) + visibilityObserver?.unobserve(element) + if (visibilityListeners.size === 0) { + visibilityObserver?.disconnect() + visibilityObserver = null + } + } +} + +function renderableImageSource(source: string | undefined): boolean { + return Boolean(source && /^(?:https?|data|blob):/i.test(source)) +} + +function transcriptImageIdentity( + block: Extract, + runtimeContext: RuntimeFileOperationArgs | null | undefined +): string { + const source = block.url?.trim() || block.path + const filePath = block.path ?? source ?? '' + if (renderableImageSource(source)) { + return `external\0${source ?? ''}` + } + return `${source ?? ''}\0${filePath}\0${ + runtimeContext === null + ? 'unresolved' + : runtimeContext === undefined + ? 'pending' + : getLocalImageCacheKey(source ?? '', runtimeContext.connectionId, runtimeContext) + }` +} + +function TranscriptImagePreview({ + block, + runtimeContext +}: { + block: Extract + runtimeContext: RuntimeFileOperationArgs | null | undefined +}): React.JSX.Element { + const [open, setOpen] = useState(false) + const [near, setNear] = useState(false) + const [thumbnailErrorSrc, setThumbnailErrorSrc] = useState(null) + const [dialogErrorSrc, setDialogErrorSrc] = useState(null) + const ref = useRef(null) + const source = block.url?.trim() || block.path + const filePath = block.path ?? source ?? '' + const external = renderableImageSource(source) + const leaseActive = near || open + const localSrc = useLocalImageSrc( + leaseActive && !external && runtimeContext !== undefined ? source : undefined, + filePath, + runtimeContext?.connectionId, + runtimeContext + ) + const displaySrc = external && leaseActive ? source : localSrc + const label = + block.alt?.trim() || + (block.path && isNativeChatPastedImagePath(block.path) + ? translate('components.native-chat.composer.pastedImageLabel', 'Pasted image') + : block.path + ? basename(block.path) + : 'Image') + const viewImageLabel = translate('components.native-chat.composer.viewAttachment', 'View image') + const fallback = ( +
+ + {label} +
+ ) + + useEffect(() => { + const element = ref.current + if (!element) { + return + } + return observeTranscriptVisibility(element, setNear) + }, []) + useEffect(() => { + const context = runtimeContext + if (!source || external || context === undefined || context === null) { + return + } + if (!leaseActive) { + releaseLocalImageSrc(source, filePath, context.connectionId, context) + } + return () => releaseLocalImageSrc(source, filePath, context.connectionId, context) + }, [external, filePath, leaseActive, runtimeContext, source]) + + const showPreview = + leaseActive && + Boolean(displaySrc) && + displaySrc !== thumbnailErrorSrc && + Boolean(source) && + (external || runtimeContext !== null) + + if (!showPreview) { + return
{fallback}
+ } + return ( +
+ + + + {label} + + {translate('components.native-chat.composer.imagePreview', 'Full-size image preview')} + +
+ {displaySrc && displaySrc !== dialogErrorSrc ? ( + {label} setDialogErrorSrc(displaySrc)} + className="max-h-[75vh] max-w-full object-contain" + /> + ) : ( + fallback + )} +
+
+
+
+ ) +} export function NativeChatImageAttachments({ - blocks + blocks, + runtimeContext, + enablePreview = runtimeContext !== undefined }: { blocks: NativeChatBlock[] + runtimeContext?: RuntimeFileOperationArgs | null + /** Keep legacy terminal chips unchanged until that lane opts into previews. */ + enablePreview?: boolean }): React.JSX.Element | null { const images = blocks.filter((block) => block.type === 'image-ref') if (images.length === 0) { return null } + const imageKeyCounts = new Map() + if (!enablePreview) { + return ( +
+ {images.map((image) => { + const label = image.alt ?? image.path ?? image.url ?? 'Image' + const imageKeyBase = `${label}-${image.url ?? ''}-${image.path ?? ''}` + const occurrence = imageKeyCounts.get(imageKeyBase) ?? 0 + imageKeyCounts.set(imageKeyBase, occurrence + 1) + const name = + image.path && isNativeChatPastedImagePath(image.path) + ? translate('components.native-chat.composer.pastedImageLabel', 'Pasted image') + : image.path + ? basename(image.path) + : label + return ( +
+ + {name} +
+ ) + })} +
+ ) + } return (
- {images.map((image, index) => { + {images.map((image) => { const label = image.alt ?? image.path ?? image.url ?? 'Image' - const name = - image.path && isNativeChatPastedImagePath(image.path) - ? translate('components.native-chat.composer.pastedImageLabel', 'Pasted image') - : image.path - ? basename(image.path) - : label + const imageKeyBase = `${label}-${image.url ?? ''}-${image.path ?? ''}` + const occurrence = imageKeyCounts.get(imageKeyBase) ?? 0 + imageKeyCounts.set(imageKeyBase, occurrence + 1) + const identity = transcriptImageIdentity(image, runtimeContext) return ( -
- - {name} -
+ ) })}
diff --git a/src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx b/src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx new file mode 100644 index 00000000000..59969c1dece --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx @@ -0,0 +1,21 @@ +import { translate } from '@/i18n/i18n' + +export function NativeChatTypingIndicatorRow(): React.JSX.Element { + return ( +
+
+ {[0, 1, 2].map((i) => ( + + ))} +
+
+ ) +} diff --git a/src/renderer/src/components/native-chat/native-chat-file-link.test.ts b/src/renderer/src/components/native-chat/native-chat-file-link.test.ts index ab5d273ce54..eaf63279fe4 100644 --- a/src/renderer/src/components/native-chat/native-chat-file-link.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-file-link.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import type { Tab } from '../../../../shared/tab-types' import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { AppState } from '@/store/types' +import { folderWorkspaceKey } from '../../../../shared/workspace-scope' import { resolveNativeChatFileLink, resolveNativeChatFileLinkContext, @@ -111,6 +112,27 @@ describe('resolveNativeChatFileLinkContext', () => { runtimeEnvironmentId: null }) }) + + it('resolves a folder workspace tab from its folder path when no projected worktree path exists', () => { + const folderId = 'folder-1' + const folderKey = folderWorkspaceKey(folderId) + const folderTab = terminalTab({ worktreeId: folderKey }) + expect( + resolveNativeChatFileLinkContext( + state({ + tabsByWorktree: { [folderKey]: [folderTab] }, + getKnownWorktreeById: () => undefined, + folderWorkspaces: [{ id: folderId, folderPath: '/workspace/platform' } as never], + worktreesByRepo: {} + }), + folderTab.id + ) + ).toEqual({ + worktreeId: folderKey, + worktreePath: '/workspace/platform', + runtimeEnvironmentId: null + }) + }) }) describe('resolveNativeChatFileLink', () => { diff --git a/src/renderer/src/components/native-chat/native-chat-file-link.ts b/src/renderer/src/components/native-chat/native-chat-file-link.ts index c2d77d066c5..431dc9a556e 100644 --- a/src/renderer/src/components/native-chat/native-chat-file-link.ts +++ b/src/renderer/src/components/native-chat/native-chat-file-link.ts @@ -6,6 +6,7 @@ import { } from '@/lib/explicit-file-link-target' import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' import type { AppState } from '@/store/types' +import { parseWorkspaceKey } from '../../../../shared/workspace-scope' export type NativeChatFileLinkContext = { worktreeId: string @@ -86,13 +87,21 @@ export function resolveNativeChatFileLinkContext( const worktree = knownWorktree?.path ? knownWorktree : findWorktreeFallback(state.worktreesByRepo, worktreeId) - if (!worktree?.path) { + const workspaceScope = parseWorkspaceKey(worktreeId) + const worktreePath = + worktree?.path ?? + (workspaceScope?.type === 'folder' + ? (state.folderWorkspaces.find( + (workspace) => workspace.id === workspaceScope.folderWorkspaceId + )?.folderPath ?? null) + : null) + if (!worktreePath) { return null } return { worktreeId, - worktreePath: worktree.path, + worktreePath, runtimeEnvironmentId: getRuntimeEnvironmentIdForWorktree(state, worktreeId) } } diff --git a/src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts new file mode 100644 index 00000000000..cfda20998fa --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts @@ -0,0 +1,96 @@ +import { describe, expect, it } from 'vitest' +import { shallow } from 'zustand/shallow' +import type { AppState } from '@/store/types' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import { + resolveNativeChatImageRuntimeContext, + selectNativeChatImageOwnerState +} from './native-chat-image-runtime-context' + +function state(): AppState { + const tab: TerminalTab = { + id: 'tab-1', + ptyId: null, + worktreeId: 'wt-1', + title: 'Terminal 1', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 0 + } + const worktree = { + id: 'wt-1', + repoId: 'repo', + path: '/repo/worktree', + hostId: 'local' + } + return { + activeWorkspaceExecutionHostId: 'local', + activeWorktreeId: 'wt-1', + detectedWorktreesByRepo: {}, + folderWorkspaces: [], + getKnownWorktreeById: () => worktree, + projectGroups: [], + removedRuntimeEnvironmentIds: new Set(), + repos: [{ id: 'repo', path: '/repo' }], + restoredRuntimeHostIdByWorkspaceSessionKey: {}, + runtimeEnvironmentCatalogHydrated: true, + runtimeEnvironments: [], + settings: { activeRuntimeEnvironmentId: null }, + sshConnectionStates: {}, + sshStateByEnvironment: {}, + tabsByWorktree: { 'wt-1': [tab] }, + unifiedTabsByWorktree: {}, + worktreesByRepo: { repo: [worktree] } + } as unknown as AppState +} + +describe('resolveNativeChatImageRuntimeContext', () => { + it('keeps unrelated store writes out of the image-owner selector', () => { + const storeState = state() + const first = selectNativeChatImageOwnerState(storeState) + const second = selectNativeChatImageOwnerState({ + ...storeState, + agentStatusByPaneKey: {} as AppState['agentStatusByPaneKey'] + }) + + expect(shallow(second, first)).toBe(true) + }) + + it('reuses derived settings when owner inputs are unchanged', () => { + const storeState = state() + const first = resolveNativeChatImageRuntimeContext(storeState, 'tab-1') + const second = resolveNativeChatImageRuntimeContext(storeState, 'tab-1') + + expect(first).not.toBeNull() + expect(second?.settings).toBe(first?.settings) + expect(shallow(second, first)).toBe(true) + }) + + it('derives a runtime host from an owner-only route during paired hydration', () => { + const storeState = state() + const ownerOnlyWorktree = { + id: 'wt-1', + repoId: 'repo', + path: '/repo/worktree', + runtimeOwnerEnvironmentId: 'owner-a' + } + const ownerState = { + ...storeState, + activeWorktreeId: null, + activeWorkspaceExecutionHostId: null, + getKnownWorktreeById: () => ownerOnlyWorktree, + worktreesByRepo: { repo: [ownerOnlyWorktree] }, + runtimeEnvironments: [{ id: 'owner-a' }] + } as unknown as AppState + + const context = resolveNativeChatImageRuntimeContext(ownerState, 'tab-1') + + expect(context).toMatchObject({ + worktreeId: 'wt-1', + worktreePath: '/repo/worktree', + expectedExecutionHostId: 'local', + settings: { activeRuntimeEnvironmentId: 'owner-a' } + }) + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts new file mode 100644 index 00000000000..600c5b63c1d --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts @@ -0,0 +1,184 @@ +import { useAppStore } from '@/store' +import type { AppState } from '@/store/types' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { + settingsForWorktreeOperationRoute, + resolveWorktreeOperationRouteResult +} from '@/lib/worktree-operation-route' +import { resolveNativeChatFileLinkContext } from './native-chat-file-link' +import { captureDirectSshMutationExpectation } from '@/lib/ssh-mutation-expectation' +import { + parseExecutionHostId, + toRuntimeExecutionHostId, + type ExecutionHostId +} from '../../../../shared/execution-host' +import { parseWorkspaceKey } from '../../../../shared/workspace-scope' +import { useMemo } from 'react' +import { useShallow } from 'zustand/react/shallow' + +/** The transcript must not read until ownership and its path are both known. */ +export type NativeChatImageRuntimeContext = RuntimeFileOperationArgs | null + +type OwnerState = Pick< + AppState, + | 'settings' + | 'repos' + | 'worktreesByRepo' + | 'detectedWorktreesByRepo' + | 'folderWorkspaces' + | 'projectGroups' + | 'runtimeEnvironments' + | 'runtimeEnvironmentCatalogHydrated' + | 'removedRuntimeEnvironmentIds' + | 'sshConnectionStates' + | 'sshStateByEnvironment' + | 'activeWorktreeId' + | 'activeWorkspaceExecutionHostId' + | 'restoredRuntimeHostIdByWorkspaceSessionKey' + | 'getKnownWorktreeById' + | 'tabsByWorktree' + | 'unifiedTabsByWorktree' +> + +// Keep the subscription limited to fields that can change image ownership. The +// derived context is computed during render, after Zustand has filtered updates. +export function selectNativeChatImageOwnerState(state: AppState): OwnerState { + return { + settings: state.settings, + repos: state.repos, + worktreesByRepo: state.worktreesByRepo, + detectedWorktreesByRepo: state.detectedWorktreesByRepo, + folderWorkspaces: state.folderWorkspaces, + projectGroups: state.projectGroups, + runtimeEnvironments: state.runtimeEnvironments, + runtimeEnvironmentCatalogHydrated: state.runtimeEnvironmentCatalogHydrated, + removedRuntimeEnvironmentIds: state.removedRuntimeEnvironmentIds, + sshConnectionStates: state.sshConnectionStates, + sshStateByEnvironment: state.sshStateByEnvironment, + activeWorktreeId: state.activeWorktreeId, + activeWorkspaceExecutionHostId: state.activeWorkspaceExecutionHostId, + restoredRuntimeHostIdByWorkspaceSessionKey: state.restoredRuntimeHostIdByWorkspaceSessionKey, + getKnownWorktreeById: state.getKnownWorktreeById, + tabsByWorktree: state.tabsByWorktree, + unifiedTabsByWorktree: state.unifiedTabsByWorktree + } +} + +// Route settings are cloned for the runtime operation contract. Reuse that +// clone while the store's source settings and selected runtime are unchanged so +// consumers do not treat an unrelated store update as a new image owner. +const settingsBySource = new WeakMap>() + +function stableSettingsForRoute( + settings: AppState['settings'], + runtimeEnvironmentId: string | null +): AppState['settings'] { + if (!settings) { + return settingsForWorktreeOperationRoute(settings, { + executionHostId: null, + runtimeEnvironmentId + }) + } + const source = settings as object + let byRuntime = settingsBySource.get(source) + if (!byRuntime) { + byRuntime = new Map() + settingsBySource.set(source, byRuntime) + } + const cacheKey = runtimeEnvironmentId ?? '' + const cached = byRuntime.get(cacheKey) + if (cached) { + return cached + } + const resolved = settingsForWorktreeOperationRoute(settings, { + executionHostId: null, + runtimeEnvironmentId + }) + byRuntime.set(cacheKey, resolved) + return resolved +} + +function resolvePath( + state: OwnerState, + worktreeId: string, + hostId: ExecutionHostId | null +): string | null { + const known = state.getKnownWorktreeById(worktreeId, hostId ?? undefined) + if (known?.path) { + return known.path + } + const workspace = parseWorkspaceKey(worktreeId) + if (workspace?.type === 'folder') { + return ( + state.folderWorkspaces.find((entry) => entry.id === workspace.folderWorkspaceId) + ?.folderPath ?? null + ) + } + for (const worktrees of Object.values(state.worktreesByRepo ?? {})) { + const match = worktrees.find( + (entry) => entry.id === worktreeId && (!hostId || entry.hostId === hostId) + ) + if (match?.path) { + return match.path + } + } + return null +} + +export function resolveNativeChatImageRuntimeContext( + state: OwnerState, + tabId: string +): NativeChatImageRuntimeContext { + const linkContext = resolveNativeChatFileLinkContext(state, tabId) + if (!linkContext) { + return null + } + const routeResolution = resolveWorktreeOperationRouteResult(state, linkContext.worktreeId) + if (routeResolution.kind !== 'resolved') { + return null + } + const route = routeResolution.route + const executionHostId = + route.executionHostId ?? + (route.runtimeEnvironmentId ? toRuntimeExecutionHostId(route.runtimeEnvironmentId) : null) + if (!executionHostId) { + return null + } + const worktreePath = resolvePath(state, linkContext.worktreeId, executionHostId) + if (!worktreePath) { + return null + } + const host = parseExecutionHostId(executionHostId) + if (!host) { + return null + } + const context: RuntimeFileOperationArgs = { + settings: stableSettingsForRoute(state.settings, route.runtimeEnvironmentId), + worktreeId: linkContext.worktreeId, + worktreePath, + expectedExecutionHostId: host.kind === 'ssh' ? host.id : 'local' + } + if (host.kind === 'ssh') { + try { + const expectation = captureDirectSshMutationExpectation( + state, + host.targetId, + route.runtimeEnvironmentId + ) + context.expectedSshTargetId = expectation.expectedSshTargetId + context.expectedSshConnectionGeneration = expectation.expectedSshConnectionGeneration + if (!route.runtimeEnvironmentId) { + context.connectionId = host.targetId + context.expectedExternalSshTargetId = host.targetId + } + } catch { + return null + } + } + return context +} + +export function useNativeChatImageRuntimeContext(tabId: string): NativeChatImageRuntimeContext { + const ownerState = useAppStore(useShallow(selectNativeChatImageOwnerState)) + return useMemo(() => resolveNativeChatImageRuntimeContext(ownerState, tabId), [ownerState, tabId]) +} From 21210aad3428a2ab52aba0e49e2c46a2922dfdc3 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 20:23:11 -0700 Subject: [PATCH 26/77] fix(native-chat): make structured Codex launches race-resistant (#18251) * fix(native-chat): cancel close-racing structured launches * fix(native-chat): make structured launches observable and recoverable * fix(native-chat): reconcile merged session tab publications * refactor(native-chat): unify host snapshot versioning * refactor(native-chat): complete launches from host snapshots * fix(native-chat): replay unknown launches by intent * fix(native-chat): guard duplicate launches and bound sync recovery * test(native-chat): type owner fixture * test(native-chat): type owner fixture * fix(native-chat): back off structured session resubscribe * fix(native-chat): fence delayed local session snapshots * fix(native-chat): retry initial session sync safely * fix(native-chat): refresh before sync retry * test(native-chat): cover folder sync cursor cleanup * fix(native-chat): retry failed structured session subscriptions --------- Co-authored-by: Merge Sim --- ...ime-apply-mobile-session-tab-navigation.ts | 2 +- ...time-close-headless-mobile-terminal-tab.ts | 2 +- ...time-close-structured-agent-session-tab.ts | 8 +- ...e-runtime-owned-mobile-session-terminal.ts | 2 +- ...act-persisted-terminal-surface-identity.ts | 2 +- ...ile-session-tabs-from-workspace-session.ts | 2 +- ...untime-move-headless-mobile-session-tab.ts | 6 +- ...time-persist-headless-session-tab-props.ts | 4 +- ...me-persist-terminal-surface-retirements.ts | 2 +- ...lish-pty-backed-mobile-session-terminal.ts | 2 +- ...le-headless-mobile-session-browser-tabs.ts | 2 +- ...renderer-session-owned-mobile-terminals.ts | 2 +- ...tore-structured-agent-session-tabs-once.ts | 4 +- src/main/runtime/orca-runtime-runtime-id.ts | 15 ++ .../orca-runtime-stop-requested-pty-ids.ts | 2 +- ...mobile-snapshot-has-stale-preserved-tab.ts | 20 +- .../orca-runtime-sync-mobile-session-tabs.ts | 4 +- .../runtime/orca-runtime-sync-window-graph.ts | 2 +- .../mobile-session-tabs-part-06.spec.ts | 2 +- ...-touch-mobile-session-tabs-for-worktree.ts | 2 +- .../components/tab-bar/QuickLaunchButton.tsx | 22 +- .../TabBarCreateEntry.keyboard.test.tsx | 30 +++ .../components/tab-bar/TabBarCreateEntry.tsx | 27 +- .../tab-bar/TabBarCreateEntryRow.tsx | 13 +- ...abCloseCommands.structured-session.test.ts | 6 + .../tab-group/useTabGroupTabCloseCommands.ts | 5 + ...launch-agent-structured-chat-guard.test.ts | 29 ++- .../structured-agent-session-launch.test.ts | 162 ++++++++++-- .../lib/structured-agent-session-launch.ts | 239 +++++++++++++++--- ...local-structured-session-tabs-sync.test.ts | 211 +++++++++++++++- .../local-structured-session-tabs-sync.ts | 166 ++++++++++-- .../src/runtime/web-session-tabs-sync.test.ts | 34 +++ .../publisher-identity-fences.ts | 16 ++ .../tracking-decisions.ts | 22 +- 34 files changed, 931 insertions(+), 138 deletions(-) diff --git a/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts b/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts index 912a38b5abc..a2552ef12c3 100644 --- a/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts +++ b/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts @@ -142,7 +142,7 @@ export class OrcaRuntimeWithApplyMobileSessionTabNavigation extends OrcaRuntimeW tabs } this.persistHeadlessTerminalActiveLeaf(worktreeId, activeTab) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } diff --git a/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts b/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts index 2786f8ac8a4..e0ac1bed9d0 100644 --- a/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts +++ b/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts @@ -107,7 +107,7 @@ export class OrcaRuntimeWithCloseHeadlessMobileTerminalTab extends OrcaRuntimeWi : {}), tabs: nextTabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } diff --git a/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts b/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts index 57703367a52..9c6e5cda532 100644 --- a/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts +++ b/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts @@ -39,7 +39,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi })), tabs: nextTabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } @@ -49,7 +49,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi protected republishMobileSessionTabsSnapshot(worktreeId: string): void { const snapshot = this.mobileSessionTabsByWorktree.get(worktreeId) if (snapshot) { - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...snapshot, snapshotVersion: snapshot.snapshotVersion + 1 }) @@ -161,7 +161,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi })), tabs: nextTabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return true } @@ -203,7 +203,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi focusesHost, publicationEpoch: `headless:${Date.now().toString(36)}` }) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) // Why: browser group membership is otherwise live-only; persist it so a // later rebuild keeps the browser in its group instead of coalescing left. if (placedInTargetGroup && nextSnapshot.tabGroupLayout) { diff --git a/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts b/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts index ed9c9cec036..ee0b87693a8 100644 --- a/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts +++ b/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts @@ -139,7 +139,7 @@ export class OrcaRuntimeWithCreateRuntimeOwnedMobileSessionTerminal extends Orca ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, next) + this.storeMobileSessionSnapshot(worktreeId, next) const result = this.toMobileSessionTabsResult(next) const changeSequence = ++this.mobileSessionTabsChangeSequence for (const subscription of this.mobileSessionTabListeners) { diff --git a/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts b/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts index e71b0cb8010..6956644c748 100644 --- a/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts +++ b/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts @@ -65,7 +65,7 @@ export class OrcaRuntimeWithHasExactPersistedTerminalSurfaceIdentity extends Orc exactOnly: true }) if (retired) { - this.mobileSessionTabsByWorktree.set(candidate.worktreeId, retired.snapshot) + this.storeMobileSessionSnapshot(candidate.worktreeId, retired.snapshot) this.notifyMobileSessionTabsChanged(candidate.worktreeId) } } diff --git a/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts b/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts index 1a356c0cdb8..140093fc204 100644 --- a/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts +++ b/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts @@ -236,7 +236,7 @@ export class OrcaRuntimeWithHydrateHeadlessMobileSessionTabsFromWorkspaceSession if (existing && headlessMobileSnapshotContentUnchanged(existing, nextSnapshot)) { continue } - this.mobileSessionTabsByWorktree.set(entryWorktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(entryWorktreeId, nextSnapshot) } return reconciledWorktreeIds } diff --git a/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts b/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts index 19a584fbe17..caaf892e65b 100644 --- a/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts +++ b/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts @@ -72,7 +72,7 @@ export class OrcaRuntimeWithMoveHeadlessMobileSessionTab extends OrcaRuntimeWith if (nextGroups.length > 1 && snapshot.tabGroupLayout) { this.persistHeadlessTabGroups(worktreeId, nextGroups, snapshot.tabGroupLayout) } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return { moved: true } } @@ -111,7 +111,7 @@ export class OrcaRuntimeWithMoveHeadlessMobileSessionTab extends OrcaRuntimeWith tabGroupLayout: split.layout } this.persistHeadlessTabGroups(worktreeId, split.groups, split.layout) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return { moved: true } } @@ -147,7 +147,7 @@ export class OrcaRuntimeWithMoveHeadlessMobileSessionTab extends OrcaRuntimeWith tabGroupLayout: layout } this.persistHeadlessTabGroups(worktreeId, moved.groups, layout) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return { moved: true } } diff --git a/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts b/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts index e8eee89a1db..10489c6c701 100644 --- a/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts +++ b/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts @@ -95,7 +95,7 @@ export class OrcaRuntimeWithPersistHeadlessSessionTabProps extends OrcaRuntimeWi snapshotVersion: snapshot.snapshotVersion + 1, tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } @@ -181,7 +181,7 @@ export class OrcaRuntimeWithPersistHeadlessSessionTabProps extends OrcaRuntimeWi snapshotVersion: snapshot.snapshotVersion + 1, tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } } diff --git a/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts b/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts index 1d275a91865..5c70a9e518a 100644 --- a/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts +++ b/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts @@ -173,7 +173,7 @@ export class OrcaRuntimeWithPersistTerminalSurfaceRetirements extends OrcaRuntim : {}) }) if (retired) { - this.mobileSessionTabsByWorktree.set(worktreeId, retired.snapshot) + this.storeMobileSessionSnapshot(worktreeId, retired.snapshot) this.notifyMobileSessionTabsChanged(worktreeId) } } diff --git a/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts b/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts index b1ee888a65f..baad15289d1 100644 --- a/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts +++ b/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts @@ -142,7 +142,7 @@ export class OrcaRuntimeWithPublishPtyBackedMobileSessionTerminal extends OrcaRu ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, next) + this.storeMobileSessionSnapshot(worktreeId, next) if (args.notify !== false) { this.notifyMobileSessionTabsChanged(worktreeId) } diff --git a/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts b/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts index a97746eaa30..6c4606ec5aa 100644 --- a/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts +++ b/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts @@ -47,7 +47,7 @@ export class OrcaRuntimeWithReconcileHeadlessMobileSessionBrowserTabs extends Or const active = activeStillPresent ? null : (nextTabs.find((tab) => tab.isActive) ?? nextTabs[0] ?? null) - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...existing, publicationEpoch: `headless-hydrated:${Date.now().toString(36)}`, snapshotVersion: existing.snapshotVersion + 1, diff --git a/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts b/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts index b5efd465589..bfab8f0ba9a 100644 --- a/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts +++ b/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts @@ -39,7 +39,7 @@ export class OrcaRuntimeWithRestoreLivePairedRendererSessionOwnedMobileTerminals continue } if (!existing) { - this.mobileSessionTabsByWorktree.set(targetWorktreeId, { + this.storeMobileSessionSnapshot(targetWorktreeId, { worktree: targetWorktreeId, publicationEpoch: `renderer-rescue:${Date.now().toString(36)}`, snapshotVersion: 0, diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index be255d723c1..36b5f65b12e 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -94,7 +94,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu ), tabs: existing.tabs.map((tab) => ({ ...tab, isActive: tab.id === id })) } - this.mobileSessionTabsByWorktree.set(input.workspaceId, snapshot) + this.storeMobileSessionSnapshot(input.workspaceId, snapshot) if (input.notify !== false) { this.emitMobileSessionTabsSnapshot(snapshot) } @@ -143,7 +143,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.mobileSessionTabsByWorktree.set(input.workspaceId, snapshot) + this.storeMobileSessionSnapshot(input.workspaceId, snapshot) if (input.notify !== false) { this.emitMobileSessionTabsSnapshot(snapshot) } diff --git a/src/main/runtime/orca-runtime-runtime-id.ts b/src/main/runtime/orca-runtime-runtime-id.ts index da55f2229b6..298cc2b7cb7 100644 --- a/src/main/runtime/orca-runtime-runtime-id.ts +++ b/src/main/runtime/orca-runtime-runtime-id.ts @@ -96,6 +96,21 @@ export class OrcaRuntimeWithRuntimeId { protected mobileSessionTabsByWorktree = new Map() + /** Single host writer for mobile session snapshots; versions are total-order stamps. */ + protected storeMobileSessionSnapshot( + worktreeId: string, + snapshot: RuntimeMobileSessionTabsSnapshot + ): RuntimeMobileSessionTabsSnapshot { + const existing = this.mobileSessionTabsByWorktree.get(worktreeId) + const snapshotVersion = existing + ? Math.max(snapshot.snapshotVersion, existing.snapshotVersion + 1) + : snapshot.snapshotVersion + const stamped = + snapshotVersion === snapshot.snapshotVersion ? snapshot : { ...snapshot, snapshotVersion } + this.mobileSessionTabsByWorktree.set(worktreeId, stamped) + return stamped + } + protected structuredAgentSessionTabRestorePromise: Promise | null = null protected structuredAgentSessionStartupRestorePromise: Promise | null = null diff --git a/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts b/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts index 233e739911b..278ccd28a74 100644 --- a/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts +++ b/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts @@ -57,7 +57,7 @@ export class OrcaRuntimeWithStopRequestedPtyIds extends OrcaRuntimeWithRuntimeId resolveOwner: (handle) => this.resolveNativeChatLaunchDraftOwner(handle), listMobileSnapshots: () => this.mobileSessionTabsByWorktree, setMobileSnapshot: (worktreeId, snapshot) => - this.mobileSessionTabsByWorktree.set(worktreeId, snapshot), + this.storeMobileSessionSnapshot(worktreeId, snapshot), scheduleMobileSnapshot: (worktreeId) => this.scheduleMobileSessionTabsChanged(worktreeId), notifyResolved: (tabId, resolution, event) => { this.notifier?.nativeChatLaunchDraftResolved?.(tabId, resolution) diff --git a/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts b/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts index 524018d889f..c32349352a1 100644 --- a/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts +++ b/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts @@ -8,7 +8,6 @@ import type { import { getMobileSessionSnapshotTabIdentityKeys } from './mobile-session-tab-merge' import { getRuntimeBrowserPageRegistry } from './runtime-browser-page-registry' import { sameRuntimeBrowserPlacement } from '../../shared/runtime-browser-placement' -import { createHash } from 'node:crypto' import type { ClientHostedBrowserRowsEvent } from '../../shared/client-hosted-browser-rows' export class OrcaRuntimeWithStoredMobileSnapshotHasStalePreservedTab extends OrcaRuntimeWithMergePreservedHeadlessMobileSessionTabs { @@ -115,23 +114,14 @@ export class OrcaRuntimeWithStoredMobileSnapshotHasStalePreservedTab extends Orc protected getMergedMobileSessionPublicationEpoch( snapshot: RuntimeMobileSessionTabsSnapshot, - preservedTabs: readonly RuntimeMobileSessionSnapshotTab[] + _preservedTabs: readonly RuntimeMobileSessionSnapshotTab[] ): string { // Why: preserved snapshots can merge repeatedly; strip the prior merge suffix first so the publication epoch stays idempotent. const normalizedPublicationEpoch = snapshot.publicationEpoch.split(':headless-merge:')[0] - const signature = createHash('sha1') - .update( - preservedTabs - .map((tab) => - tab.type === 'terminal' - ? `${tab.id}:${tab.parentTabId}:${tab.ptyId ?? ''}:${tab.leafId}` - : tab.id - ) - .join('|') - ) - .digest('hex') - .slice(0, 12) - return `${normalizedPublicationEpoch}:headless-merge:${signature}` + // The epoch identifies the publisher generation, not the merged content. + // Content changes are ordered by snapshotVersion, so encoding a merge hash + // here would make the identity oscillate and permanently fence later rows. + return normalizedPublicationEpoch } /** Serves a hydrating host renderer; the publisher counts this as a delivery, not a read. */ diff --git a/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts b/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts index 4d54a322f3e..51997577179 100644 --- a/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts +++ b/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts @@ -167,7 +167,7 @@ export class OrcaRuntimeWithSyncMobileSessionTabs extends OrcaRuntimeWithWriteOr const storedVersion = existing ? Math.max(nextSnapshot.snapshotVersion, existing.snapshotVersion + 1) : nextSnapshot.snapshotVersion - this.mobileSessionTabsByWorktree.set( + this.storeMobileSessionSnapshot( snapshot.worktree, storedVersion === nextSnapshot.snapshotVersion ? nextSnapshot @@ -195,7 +195,7 @@ export class OrcaRuntimeWithSyncMobileSessionTabs extends OrcaRuntimeWithWriteOr preserved.tabs.length === existing.tabs.length && preserved.tabs.every((tab, index) => tab === existing.tabs[index]) if (!preservedIsNoOp) { - this.mobileSessionTabsByWorktree.set(worktreeId, preserved) + this.storeMobileSessionSnapshot(worktreeId, preserved) } // Why: the stored entry is no longer the renderer's publication, so a // future renderer frame must be re-merged even if it reuses the pair. diff --git a/src/main/runtime/orca-runtime-sync-window-graph.ts b/src/main/runtime/orca-runtime-sync-window-graph.ts index f9c6ab358e3..7d44d4af2c1 100644 --- a/src/main/runtime/orca-runtime-sync-window-graph.ts +++ b/src/main/runtime/orca-runtime-sync-window-graph.ts @@ -238,7 +238,7 @@ export class OrcaRuntimeWithSyncWindowGraph extends OrcaRuntimeWithAttachWindow // the PTY touch path does) or the re-emitted payload — e.g. the // pending-handle → ready flip — is discarded and the client stays stale. // The accepted-renderer tracking is untouched: this is a main-local bump. - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...stored, snapshotVersion: stored.snapshotVersion + 1 }) diff --git a/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts b/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts index 27e052e1594..be1d4c8175c 100644 --- a/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts +++ b/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts @@ -595,7 +595,7 @@ describe('OrcaRuntimeService', () => { const secondMerge = await runtime.listMobileSessionTabs(`id:${TEST_WORKTREE_ID}`) expect(secondMerge.publicationEpoch).toBe(firstMerge.publicationEpoch) - expect(secondMerge.publicationEpoch.match(/:headless-merge:/g) ?? []).toHaveLength(1) + expect(secondMerge.publicationEpoch).toBe('headless:stable-epoch') }) it('keeps the graph ready when a mobile snapshot references a removed folder workspace', () => { diff --git a/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts b/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts index 1481709ee66..b7ca09af705 100644 --- a/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts +++ b/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts @@ -21,7 +21,7 @@ export class OrcaRuntimeWithTouchMobileSessionTabsForWorktree extends OrcaRuntim if (!snapshot) { return } - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...snapshot, snapshotVersion: snapshot.snapshotVersion + 1 }) diff --git a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx index 6af2508048c..85d978e3bf6 100644 --- a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx +++ b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx @@ -1,5 +1,5 @@ import React, { useCallback } from 'react' -import { Settings as SettingsIcon } from 'lucide-react' +import { Loader2, Settings as SettingsIcon } from 'lucide-react' import { toast } from 'sonner' import { DropdownMenuItem, DropdownMenuShortcut } from '@/components/ui/dropdown-menu' import { getAgentCatalog, AgentIcon } from '@/lib/agent-catalog' @@ -15,6 +15,7 @@ import { filterEnabledTuiAgents } from '../../../../shared/tui-agent-selection' import { translate } from '@/i18n/i18n' +import { useStructuredCodexLaunchStatus } from '@/lib/structured-agent-session-launch' export type QuickLaunchAgentMenuItemsProps = { worktreeId: string @@ -116,6 +117,7 @@ function QuickLaunchAgentMenuItemsInner({ const openSettingsPage = useAppStore((s) => s.openSettingsPage) const openSettingsTarget = useAppStore((s) => s.openSettingsTarget) const newAgentShortcut = useOptionalShortcutLabel('tab.newAgent') + const structuredCodexLaunchStatus = useStructuredCodexLaunchStatus(worktreeId) const openAgentSettings = useCallback(() => { openSettingsTarget({ pane: 'agents', repoId: null }) @@ -197,21 +199,31 @@ function QuickLaunchAgentMenuItemsInner({ {agents.map((agent) => { const entry = getCatalogEntry(agent) const label = entry?.label ?? agent + const isStructuredCodexPending = + agent === 'codex' && structuredCodexLaunchStatus === 'pending' + const menuLabel = isStructuredCodexPending ? 'Starting Codex chat…' : label const showsDefaultAgentShortcut = newAgentShortcut !== null && defaultAgent !== 'blank' && agent === defaultAgent return ( runLaunch(agent)} className="gap-2 rounded-[7px] px-2 py-1.5 text-[12px] leading-5 font-medium" title={translate( 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', - 'Launch {{value0}} in a new terminal', - { value0: label } + isStructuredCodexPending + ? 'Starting Codex chat…' + : 'Launch {{value0}} in a new terminal', + isStructuredCodexPending ? undefined : { value0: label } )} > - - {label} + {isStructuredCodexPending ? ( +