Merge remote-tracking branch 'origin/main' into integrate-fixes

This commit is contained in:
Jinwoo-H
2026-09-04 04:34:13 -04:00
93 changed files with 6159 additions and 840 deletions
@@ -111,14 +111,19 @@ describe('incident monitor evaluator', () => {
})
})
it('freezes when postgres retries exceed the recalibrated ceiling', () => {
const sample = healthySample()
sample.sources['relay-logs']!.signals['relay.postgres_retries'] =
signal(INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries + 1)
expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({
// Why: the global relay_cells lock made retries a steady-state rate (24 h p99
// 1320/5min on 2026-09-04); the bar fences only unbounded growth beyond that.
it('tolerates the measured healthy retry rate and freezes above the bar', () => {
const healthy = healthySample()
healthy.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(1504)
expect(evaluateIncidentSample(healthy, startedAt).status).toBe('green')
const incident = healthySample()
incident.sources['relay-logs']!.signals['relay.postgres_retries'] = signal(2001)
expect(evaluateIncidentSample(incident, startedAt)).toMatchObject({
status: 'freeze',
failures: [
expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 300 })
expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 2000 })
]
})
})
+15 -6
View File
@@ -32,11 +32,20 @@ export const INCIDENT_MONITOR_THRESHOLDS = {
relayPoolWaiting: 800,
relayPoolWaitMs: 2_500,
// Why: successful lock retries are the contention machinery working, not harm.
// Healthy 2026-08-26 baseline bursts to 234/5min (26% of windows crossed the old
// bar of 20, set unmeasured at the monitor's 2026-07-28 birth); the 2026-08-23
// incident ran ~2,200-3,000/5min. 300 clears healthy bursts with ~10x incident
// margin; relayPostgresRetryExhausted below bounds the terminally failed share.
relayPostgresRetries: 300,
// Recalibrated 2026-09-04 from 300, which was set 2026-08-26 when healthy bursts
// reached 234/5min. The global relay_cells FOR UPDATE lock has since become the
// fleet's steady state: measured fleet-wide (director + cells, summed per five
// minutes) 2026-09-03T05Z..2026-09-04T05Z p50 430 / p90 924 / p99 1320 / max
// 1504, with 55% of windows over 300 and only 22% of 15-minute gates clean, so
// the bar blocked the very cell roll that carries the 500 ms lock wait (#18521)
// and the beginProof crash guard to the cells. The 2026-08-23 lock incident on
// this same metric peaked at 1510 in one window and 646 in the next, so it is
// not separable from today's contention by retries alone; it is caught by
// relayPostgresRetryExhausted (467 at the peak vs a 300 bar), director
// concurrency, and the pool bars. 2000 passes every healthy 15-minute window
// measured in the last 24 h and still fences unbounded growth. Re-tighten once
// the fleet is on the 500 ms lock wait and the baseline is re-measured.
relayPostgresRetries: 2000,
// Why: 300 per five minutes, recalibrated 2026-09-04 from a bar of zero that no
// production window has cleared since #18521 shipped to the director. That
// change cut the request-path cell-inventory wait from the 1 s pool lock_timeout
@@ -48,7 +57,7 @@ export const INCIDENT_MONITOR_THRESHOLDS = {
// quiet hours p50 2 / max 36; pre-#18521 daytime p50 10 / p90 25 / max 87;
// post-#18521 p50 42 / p90 147 / max 220. The 2026-08-23 lock incident peaked
// at 467. 300 clears every measured healthy window and still sits below the
// incident shape; relayPostgresRetries above stays the ~10x discriminator.
// incident shape; retries above fence only unbounded growth.
// User-facing /v1/assign 503 share did not move with #18521 (13.9% old image
// vs 12.3% new, same evening), so exhaustion is not a proxy for user harm.
relayPostgresRetryExhausted: 300,
+18 -2
View File
@@ -99,7 +99,7 @@ durably marked consumed before mutation and cannot authorize another run.
| Cloud SQL deadlocks | over 0 |
| Relay pool waiters | over 800 |
| Relay pool wait | over 2,500 ms |
| PostgreSQL retries in five minutes | over 300 |
| PostgreSQL retries in five minutes | over 2,000 |
| Exhausted PostgreSQL retries in five minutes | over 300 |
| Director instances | outside 5–6 |
| Director CPU or memory | over 80% |
@@ -138,7 +138,23 @@ heartbeats, and matching live admission.
`jsonPayload.event="orca_relay_postgres_transaction_retry"` in production
logs: healthy-day bursts reach 234/5min with zero exhausted retries and 26%
of five-minute windows over 20, while the 2026-08-23 lock-contention
incident ran roughly 2,200–3,000/5min.
incident ran roughly 2,200–3,000/5min by raw log-line count (the gate's
own `orca_relay_postgres_retries` metric read 1,510 for that window; see the
2026-09-04 entry).
- Recalibrated the PostgreSQL-retry freeze from 300 to 2,000 per five minutes
(2026-09-04). Basis: the global `relay_cells FOR UPDATE` lock made
successful retries a steady-state rate. Measured fleet-wide (director +
cells, summed per five minutes from the `orca_relay_postgres_retries`
log metric) over 2026-09-03T05Z..2026-09-04T05Z: p50 430 / p90 924 /
p99 1,320 / max 1,504; 55% of windows over 300; only 22% of 15-minute gates
clean at 300 versus 100% at 2,000. Three read-only dry-runs on 2026-09-04
froze on this bar (runs 33836470590, 33838698725) or on a genuine six-cell
crash storm (33837160275), blocking the same-cap roll that carries #18521
and the `beginProof` crash guard to the 23 cells. The 2026-08-23 incident
on this metric peaked at 1,510 then 646, so retries alone no longer
separate it from today's baseline; the exhausted-retry bar (incident peak
467 vs bar 300), director concurrency, and the pool bars carry that role.
Re-tighten after the fleet is on the 500 ms lock wait.
- Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five
minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory
lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters
+1 -1
View File
@@ -492,7 +492,7 @@
"ko": "agent CLI를 찾지 못했습니다. 하나를 설치하거나 설정에서 기본 agent를 선택하세요."
},
"auto.components.Terminal.7958465754": {
"ko": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?"
"ko": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?"
},
"auto.components.Terminal.cdc9ac4b2d": {
"ko": "편집기"
@@ -1,7 +1,14 @@
import type { ChildProcess } from 'node:child_process'
import { describe, expect, it, vi } from 'vitest'
import { beforeEach, describe, expect, it, vi } from 'vitest'
import {
findSelfInitiatedTreeKills,
resetSelfInitiatedTreeKillLogForTest
} from '../crash-reporting/self-initiated-tree-kill-log'
import { terminateCodexAppServerProcessTree } from './codex-app-server-process-teardown'
/** Above pid_max on every supported POSIX host, so the group signal is a real ESRCH. */
const UNREACHABLE_PGID = 2_147_483_647
function child() {
return {
pid: 1234,
@@ -10,6 +17,10 @@ function child() {
}
describe('terminateCodexAppServerProcessTree', () => {
beforeEach(() => {
resetSelfInitiatedTreeKillLogForTest()
})
it('waits for the Windows tree kill before releasing the wrapper', async () => {
const target = child()
const release = Promise.withResolvers<void>()
@@ -120,6 +131,55 @@ describe('terminateCodexAppServerProcessTree', () => {
expect(target.kill).not.toHaveBeenCalled()
})
/**
* `selfInitiatedTreeKillCount` decides whether a `render-process-gone` was
* ours. A group that had already exited was killed by nobody, so crediting it
* puts a suspect in the five-second window that Orca never issued. Exercised
* through the real `process.kill(-pgid)` because the swallow being tested
* lives in the production default, not in an injectable seam.
*/
it('does not claim a snapshot group that was already gone', async () => {
const target = { pid: UNREACHABLE_PGID, kill: vi.fn(() => true) as ChildProcess['kill'] }
await expect(
terminateCodexAppServerProcessTree(target, undefined, {
platform: 'darwin',
captureDescendants: async () => ({
rootPgid: UNREACHABLE_PGID,
descendants: [],
capturedAtMs: 1
}),
terminateDescendants: async () => true
})
).resolves.toBe(true)
expect(target.kill).toHaveBeenLastCalledWith('SIGKILL')
expect(findSelfInitiatedTreeKills(Date.now())).toEqual([])
})
it('claims a snapshot group the signal actually reached', async () => {
const target = child()
const signalProcessGroup = vi.fn()
await expect(
terminateCodexAppServerProcessTree(target, undefined, {
platform: 'darwin',
captureDescendants: async () => ({ rootPgid: 1234, descendants: [], capturedAtMs: 1 }),
terminateDescendants: async () => true,
signalProcessGroup
})
).resolves.toBe(true)
expect(signalProcessGroup).toHaveBeenCalledWith(1234, 'SIGKILL')
expect(findSelfInitiatedTreeKills(Date.now())).toEqual([
expect.objectContaining({
pid: 1234,
site: 'codex-app-server-teardown',
scope: 'posix-process-group'
})
])
})
it('tears down 40 dedicated groups without process-table scans or cross-group fanout', async () => {
const killMocks = Array.from({ length: 40 }, () => vi.fn(() => true))
const targets = killMocks.map((kill, index) => ({
@@ -128,19 +128,24 @@ async function terminatePosixTree(
if (descendantsExited && snapshot.rootPgid === rootPid) {
const signalGroup =
deps.signalProcessGroup ??
((pgid: number, signal: NodeJS.Signals) => {
try {
process.kill(-pgid, signal)
} catch {
// Group already exited.
}
((pgid: number, signal: NodeJS.Signals) => process.kill(-pgid, signal))
let groupSignalled = false
try {
signalGroup(snapshot.rootPgid, 'SIGKILL')
groupSignalled = true
} catch {
// Already-gone is still the desired outcome, but nothing here killed it,
// and a crumb for a kill we never landed is a false render-process-gone suspect.
}
if (groupSignalled) {
// Outside the try, as in terminateDedicatedPosixGroup: that catch is the
// already-gone contract, not a breadcrumb handler.
recordSelfInitiatedTreeKill({
pid: snapshot.rootPgid,
site: 'codex-app-server-teardown',
scope: 'posix-process-group'
})
signalGroup(snapshot.rootPgid, 'SIGKILL')
recordSelfInitiatedTreeKill({
pid: snapshot.rootPgid,
site: 'codex-app-server-teardown',
scope: 'posix-process-group'
})
}
}
if (!descendantsExited) {
child.kill('SIGCONT')
@@ -1,77 +0,0 @@
import type { CrashReportDetailValue } from '../../shared/crash-reporting'
// ─── System memory at gone time ─────────────────────────────────────
// Why: the system outlives the crashed process, so this IS sampleable at
// process-gone — it separates "renderer grew huge" from "machine out of
// memory/commit", which the per-process buckets alone cannot.
// Timing honesty: this reads AFTER the crashed process's memory returned to
// the OS, so free/swapFree can look healthier than they were at kill time.
// Platform honesty: swap* exist on Windows/Linux only. On Linux `free` is
// /proc/meminfo MemFree and is NOT the pressure signal — it excludes page cache
// and other reclaimable memory; `available` (MemAvailable, Linux-only) is. On
// macOS `free` is near-meaningless (file cache and compression keep it low on
// healthy machines); fileBacked/purgeable are the only reclaimability proxy this
// API gives there, and none of these fields answers "was the machine under
// pressure" on macOS — that needs a signal Electron does not expose.
type CrashReportDetails = Record<string, CrashReportDetailValue>
export function memoryKBFieldMB(value: unknown): number | undefined {
const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined
return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024)
}
type SystemMemoryInfoLike = {
total?: unknown
free?: unknown
available?: unknown
swapTotal?: unknown
swapFree?: unknown
fileBacked?: unknown
purgeable?: unknown
}
type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null
function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null {
const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike })
.getSystemMemoryInfo
if (typeof read !== 'function') {
return null
}
try {
return read.call(process)
} catch {
return null
}
}
let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo
export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void {
systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo
}
export function getSystemMemoryAtGoneDetails(): CrashReportDetails {
const info = systemMemoryInfoReader()
if (!info) {
return {}
}
const details: CrashReportDetails = {}
const fields: readonly [keyof SystemMemoryInfoLike, string][] = [
['total', 'systemMemoryTotalMB'],
['free', 'systemMemoryFreeMB'],
['available', 'systemMemoryAvailableMB'],
['swapTotal', 'systemMemorySwapTotalMB'],
['swapFree', 'systemMemorySwapFreeMB'],
['fileBacked', 'systemMemoryFileBackedMB'],
['purgeable', 'systemMemoryPurgeableMB']
]
for (const [field, key] of fields) {
const mb = memoryKBFieldMB(info[field])
if (mb !== undefined) {
details[key] = mb
}
}
return details
}
@@ -0,0 +1,379 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
import {
getSystemMemoryDetails,
setSystemMemoryInfoReaderForTest,
withSwapVolumeFreeSpace
} from './system-memory-details'
import {
readSwapVolumeFreeSpace,
setSwapVolumeFreeSpaceReaderForTest,
type SwapVolumeFreeSpace
} from './swap-volume-free-space'
import { samplePreGoneSystemMemory } from './pre-gone-host-memory'
import {
buildProcessGoneCrashDetails,
resetPreGoneCrashSamplingForTest,
samplePreGoneProcessMetrics,
startPreGoneCrashSampling
} from './process-gone-diagnostics'
type MetricFixture = {
pid: number
creationTime: number
type: string
memory: { workingSetSize: number; peakWorkingSetSize?: number; privateBytes?: number }
}
const { appMetricsMock } = vi.hoisted(() => ({
appMetricsMock: vi.fn<() => MetricFixture[]>(() => [])
}))
vi.mock('electron', () => ({ app: { getAppMetrics: appMetricsMock } }))
const BROWSER_AND_RENDERER: MetricFixture[] = [
{ pid: 10, creationTime: 1, type: 'Browser', memory: { workingSetSize: 1024 * 250 } },
{
pid: 11,
creationTime: 2,
type: 'Tab',
memory: { workingSetSize: 1024 * 400, peakWorkingSetSize: 1024 * 420, privateBytes: 1024 * 260 }
}
]
const BROWSER_ONLY: MetricFixture[] = [BROWSER_AND_RENDERER[0]]
const UNDER_COMMIT_PRESSURE = {
total: 16_000 * 1024,
free: 400 * 1024,
swapTotal: 48_000 * 1024,
swapFree: 200 * 1024
}
const AFTER_THE_CORPSE_RELEASED = {
total: 16_000 * 1024,
free: 3_000 * 1024,
swapTotal: 48_000 * 1024,
swapFree: 2_900 * 1024
}
// Commit limit ~= RAM: a disabled or fixed pagefile, which no amount of empty
// disk can grow into. `swapTotal > total` is all this API can say about that.
const FIXED_PAGEFILE_UNDER_PRESSURE = {
total: 16_000 * 1024,
free: 300 * 1024,
swapTotal: 16_100 * 1024,
swapFree: 180 * 1024
}
const NO_PAGEFILE_UNDER_PRESSURE = {
...FIXED_PAGEFILE_UNDER_PRESSURE,
swapTotal: 15_900 * 1024
}
const BEFORE_THE_STORM = {
total: 16_000 * 1024,
free: 9_000 * 1024,
swapTotal: 48_000 * 1024,
swapFree: 30_000 * 1024
}
describe('pre-gone host memory', () => {
beforeEach(() => {
resetPreGoneCrashSamplingForTest()
setSystemMemoryInfoReaderForTest(null)
setSwapVolumeFreeSpaceReaderForTest(null)
appMetricsMock.mockClear()
appMetricsMock.mockReturnValue(BROWSER_AND_RENDERER)
})
it('carries a pre-gone host reading, not only the post-mortem one', async () => {
setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE)
setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' }))
await samplePreGoneSystemMemory(Date.now() - 5_000)
// The renderer dies; its ~400 MB returns to the OS, so the gone-time read
// now shows a much healthier machine than the one that refused the alloc.
setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED)
appMetricsMock.mockReturnValue(BROWSER_ONLY)
const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer')
expect(details.systemMemorySwapFreeMB).toBe(2_900)
expect(details.systemMemoryPreGoneSwapFreeMB).toBe(200)
expect(details.systemMemoryPreGoneFreeMB).toBe(400)
expect(details.systemMemoryPreGoneTotalMB).toBe(16_000)
// Why: host memory keeps its own key family, so a `systemMemory` prefix scan sees both reads.
expect(
Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem'))
).toEqual([])
})
// Why this decides the cluster: 200 MB available commit is only a REFUSAL when
// the pagefile cannot grow, which is what the volume's free space says.
it('reports swap-volume free space so low commit can be told from refused commit', async () => {
setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE)
setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' }))
await samplePreGoneSystemMemory(Date.now() - 5_000)
const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer')
expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(120)
// Which volume was measured: Windows only names the DEFAULT pagefile drive.
expect(details.systemMemoryPreGoneSwapVolume).toBe('C:')
})
it('omits swap-volume free space on Linux, where swap cannot grow into free disk', async () => {
// Linux swap is a fixed partition, a fixed-size swapfile, or zram; reporting
// root-fs free space next to SwapFreeMB 0 would read as headroom that is not there.
setSwapVolumeFreeSpaceReaderForTest(null)
await expect(readSwapVolumeFreeSpace('linux')).resolves.toBeUndefined()
})
it('labels the reading with the pressure verdict the platform can actually give', () => {
// Windows available commit is only a REFUSAL when the pagefile cannot grow,
// which nothing here proves, so no label may read as that verdict.
setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE)
const windowsCommit = getSystemMemoryDetails('win32')
expect(windowsCommit.systemMemoryPressureSignal).toBe('available-commit-unqualified')
expect(
withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32')
.systemMemoryPressureSignal
).toBe('available-commit-volume-cotimed')
// A volume number from a different moment describes a different machine.
expect(
withSwapVolumeFreeSpace(windowsCommit, { freeMB: 120, volume: 'C:' }, 'win32', false)
.systemMemoryPressureSignal
).toBe('available-commit-unqualified')
setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, free: 400 * 1024 }))
expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('none')
setSystemMemoryInfoReaderForTest(() => ({ total: 16_000 * 1024, available: 900 * 1024 }))
expect(getSystemMemoryDetails('linux').systemMemoryPressureSignal).toBe('mem-available')
// darwin free/fileBacked/purgeable answer reclaimability, never pressure.
setSystemMemoryInfoReaderForTest(() => ({
total: 16_000 * 1024,
free: 272 * 1024,
fileBacked: 2_694 * 1024,
purgeable: 0
}))
expect(getSystemMemoryDetails('darwin').systemMemoryPressureSignal).toBe('none')
})
// Why this and not the volume number: the branch's own repro needed a pagefile
// that CANNOT grow to kill anything, and neither the pagefile maximum nor its
// drive is readable here — `swapVolumeAnchor` measures SystemRoot's volume,
// which a relocated pagefile does not live on.
it('never reads free disk as proof the pagefile could have grown', () => {
setSystemMemoryInfoReaderForTest(() => FIXED_PAGEFILE_UNDER_PRESSURE)
const fixedPagefile = withSwapVolumeFreeSpace(
getSystemMemoryDetails('win32'),
{ freeMB: 812_000, volume: 'C:' },
'win32'
)
// 180 MB of commit beside 812 GB of free disk: co-timed, and still not a
// verdict — reading it as "the pagefile had room" is the opposite conclusion.
expect(fixedPagefile.systemMemoryPressureSignal).toBe('available-commit-volume-cotimed')
// The one decisive win32 case: commit limit at or below RAM means there is
// no pagefile behind it, so the floor cannot heal however empty the disk is.
setSystemMemoryInfoReaderForTest(() => NO_PAGEFILE_UNDER_PRESSURE)
expect(
withSwapVolumeFreeSpace(
getSystemMemoryDetails('win32'),
{ freeMB: 812_000, volume: 'C:' },
'win32'
).systemMemoryPressureSignal
).toBe('available-commit-hard-capped')
})
// Why the verdict and not just the field: a statfs issued on a healthy host at
// t=0 that resolves 20 s into a commit storm prints "200 MB commit, 40 GB of
// pagefile headroom" — which reads as NOT a commit refusal, the opposite
// conclusion, under the branch's most confident label.
it('will not let a statfs that outlived its tick qualify the win32 commit verdict', async () => {
const platform = Object.getOwnPropertyDescriptor(process, 'platform')!
Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' })
vi.useFakeTimers()
let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {}
try {
setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM)
setSwapVolumeFreeSpaceReaderForTest(
() =>
new Promise<SwapVolumeFreeSpace>((resolve) => {
resolveVolume = resolve
})
)
void samplePreGoneSystemMemory(0)
// The storm arrives; the in-flight latch makes every tick skip the merge,
// so the pending statfs is as old as the tick that STARTED it.
setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE)
await samplePreGoneSystemMemory(10_000)
await samplePreGoneSystemMemory(20_000)
resolveVolume({ freeMB: 40_000, volume: 'C:' })
await vi.advanceTimersByTimeAsync(0)
vi.setSystemTime(20_000)
const stale = buildProcessGoneCrashDetails({}, 'renderer')
expect(stale.systemMemoryPreGoneSwapFreeMB).toBe(200)
// The pre-storm volume number still ships — but carrying its own age, and
// without promoting the verdict the analyst reads.
expect(stale.systemMemoryPreGoneSwapVolumeFreeMB).toBe(40_000)
expect(stale.systemMemoryPreGoneSampleAgeMs).toBe(0)
expect(stale.systemMemoryPreGoneSwapVolumeAgeMs).toBe(20_000)
expect(stale.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified')
// The next tick's statfs answers on its own tick, so it qualifies again.
setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 900, volume: 'C:' }))
await samplePreGoneSystemMemory(30_000)
vi.setSystemTime(30_000)
const fresh = buildProcessGoneCrashDetails({}, 'renderer')
expect(fresh.systemMemoryPreGoneSwapVolumeFreeMB).toBe(900)
expect(fresh.systemMemoryPreGoneSwapVolumeAgeMs).toBe(0)
expect(fresh.systemMemoryPreGonePressureSignal).toBe('available-commit-volume-cotimed')
} finally {
vi.useRealTimers()
Object.defineProperty(process, 'platform', platform)
}
})
// Round 5: sample identity alone could not see these ticks. A host read that
// returns nothing leaves the sample object in place, so `sample === issuedFor`
// still held 25 s and two ticks later and the statfs re-qualified the verdict.
it('will not let ticks with a failed host read pass a stale statfs off as co-timed', async () => {
const platform = Object.getOwnPropertyDescriptor(process, 'platform')!
Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' })
vi.useFakeTimers()
let resolveVolume: (value: SwapVolumeFreeSpace) => void = () => {}
try {
setSystemMemoryInfoReaderForTest(() => BEFORE_THE_STORM)
setSwapVolumeFreeSpaceReaderForTest(
() =>
new Promise<SwapVolumeFreeSpace>((resolve) => {
resolveVolume = resolve
})
)
void samplePreGoneSystemMemory(0)
// GlobalMemoryStatusEx starts failing: the sample is neither replaced nor erased.
setSystemMemoryInfoReaderForTest(() => null)
await samplePreGoneSystemMemory(10_000)
await samplePreGoneSystemMemory(20_000)
resolveVolume({ freeMB: 40_000, volume: 'C:' })
await vi.advanceTimersByTimeAsync(0)
vi.setSystemTime(25_000)
const details = buildProcessGoneCrashDetails({}, 'renderer')
// 25 s of lag: the label must not say co-timed beside that age.
expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(25_000)
expect(details.systemMemoryPreGonePressureSignal).toBe('available-commit-unqualified')
} finally {
vi.useRealTimers()
Object.defineProperty(process, 'platform', platform)
}
})
it("arms the host sampler on its own unref'd 10 s timer, not the metric sweep's", async () => {
vi.useFakeTimers()
vi.setSystemTime(0)
const readHostMemory = vi.fn(() => UNDER_COMMIT_PRESSURE)
setSystemMemoryInfoReaderForTest(readHostMemory)
setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 120, volume: 'C:' }))
const setIntervalSpy = vi.spyOn(globalThis, 'setInterval')
try {
startPreGoneCrashSampling()
// Literal millisecond values: asserting the constants against themselves
// would let a cadence regression through, and 37 s of staleness is the bug.
expect(setIntervalSpy.mock.calls.map(([, ms]) => ms)).toEqual([60_000, 10_000])
for (const { value } of setIntervalSpy.mock.results) {
expect((value as NodeJS.Timeout).hasRef()).toBe(false)
}
expect(readHostMemory).toHaveBeenCalledTimes(1)
readHostMemory.mockReturnValue(AFTER_THE_CORPSE_RELEASED)
await vi.advanceTimersByTimeAsync(10_000)
// One host tick, no extra metric sweep: the two samplers run independently.
expect(readHostMemory).toHaveBeenCalledTimes(2)
expect(appMetricsMock).toHaveBeenCalledTimes(1)
const details = buildProcessGoneCrashDetails({}, 'renderer')
expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0)
expect(details.systemMemoryPreGoneSwapFreeMB).toBe(2_900)
} finally {
setIntervalSpy.mockRestore()
vi.useRealTimers()
}
})
it('commits the host reading without waiting on the swap-volume statfs', async () => {
// Why: statfs is slowest during the paging storm this sampler targets, and
// a hung volume must not stall or silently skip host sampling.
setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE)
setSwapVolumeFreeSpaceReaderForTest(() => new Promise(() => {}))
void samplePreGoneSystemMemory(Date.now() - 5_000)
expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(200)
// A second tick still refreshes the reading while that statfs hangs.
setSystemMemoryInfoReaderForTest(() => AFTER_THE_CORPSE_RELEASED)
void samplePreGoneSystemMemory(Date.now())
expect(buildProcessGoneCrashDetails({}, 'renderer').systemMemoryPreGoneSwapFreeMB).toBe(2_900)
})
it('publishes no pre-gone host keys when every memory field failed to read', async () => {
// Why not "no keys at all": the reading always carries its signal label, so a
// committed empty one would ship an age and a volume number with no memory
// numbers beside them — a disk-free figure standing in for a host reading.
setSystemMemoryInfoReaderForTest(() => ({ total: Number.NaN, free: undefined }))
await samplePreGoneSystemMemory(Date.now())
const details = buildProcessGoneCrashDetails({}, 'renderer')
expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([])
})
it('carries the last volume reading forward, aged, instead of dropping it', async () => {
vi.useFakeTimers()
try {
setSystemMemoryInfoReaderForTest(() => UNDER_COMMIT_PRESSURE)
setSwapVolumeFreeSpaceReaderForTest(() => Promise.resolve({ freeMB: 42, volume: 'C:' }))
await samplePreGoneSystemMemory(0)
// The next tick's statfs hangs — during the paging storm this targets, that
// is the normal case — so the tick has no volume reading of its own, and
// the sample that replaces the last one would otherwise drop the field.
setSwapVolumeFreeSpaceReaderForTest(() => new Promise<never>(() => {}))
void samplePreGoneSystemMemory(10_000)
vi.setSystemTime(10_000)
const details = buildProcessGoneCrashDetails({}, 'renderer')
expect(details.systemMemoryPreGoneSwapVolumeFreeMB).toBe(42)
expect(details.systemMemoryPreGoneSwapVolume).toBe('C:')
expect(details.systemMemoryPreGoneSampleAgeMs).toBe(0)
// Carried, not re-read: it ships at its real age, never as a fresh number.
expect(details.systemMemoryPreGoneSwapVolumeAgeMs).toBe(10_000)
} finally {
vi.useRealTimers()
}
})
it('keeps a failed host read from erasing the process-metric sample', async () => {
samplePreGoneProcessMetrics(Date.now() - 5_000)
setSystemMemoryInfoReaderForTest(() => {
throw new Error('getSystemMemoryInfo unavailable')
})
await samplePreGoneSystemMemory(Date.now() - 5_000)
setSystemMemoryInfoReaderForTest(null)
const details = buildProcessGoneCrashDetails({ processType: 'renderer' }, 'renderer')
expect(details.processMetricsPreGoneRendererWorkingSetMB).toBe(400)
expect(Object.keys(details).filter((key) => key.startsWith('systemMemoryPreGone'))).toEqual([])
})
})
@@ -0,0 +1,164 @@
import type { CrashReportDetailValue } from '../../shared/crash-reporting'
import { readSwapVolumeFreeSpace } from './swap-volume-free-space'
import {
getSystemMemoryDetails,
SYSTEM_MEMORY_KEY_PREFIX,
withSwapVolumeFreeSpace
} from './system-memory-details'
// ─── Pre-gone host memory sampling ──────────────────────────────────
// Why sample at all: the gone-time host read lands after the corpse released
// its pages, so it reports a healthier machine than the one that refused the
// allocation.
// Why 10 s and not the 60 s process-metrics cadence: at 60 s, four of five
// field OOMs carried a ~37 s old host reading — far too stale to see a
// transient commit refusal. A refusal shorter than the interval stays
// invisible; no cadence fixes that.
export const PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS = 10_000
type CrashReportDetails = Record<string, CrashReportDetailValue>
type PreGoneSystemMemorySample = {
details: CrashReportDetails
sampledAtMs: number
/** Tick that ISSUED the statfs now merged in — never the tick it resolved on. */
swapVolumeSampledAtMs?: number
}
let preGoneSample: PreGoneSystemMemorySample | null = null
let preGoneTimer: ReturnType<typeof setInterval> | null = null
let swapVolumeReadInFlight = false
let samplingGeneration = 0
let sampleTick = 0
const PRESSURE_SIGNAL_KEY = `${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`
/**
* Carries the last volume reading onto the sample that replaces its own.
*
* Why: a statfs slower than one tick would otherwise make the field vanish from
* the reports it exists for — the next tick replaces the sample wholesale, and
* the in-flight latch keeps intervening ticks from merging anything. It ships
* with its own (now larger) age and, not being co-timed, never names the label.
*/
function withCarriedSwapVolume(sample: PreGoneSystemMemorySample): PreGoneSystemMemorySample {
const previous = preGoneSample
if (!previous || previous.swapVolumeSampledAtMs === undefined) {
return sample
}
const freeMB = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`]
const volume = previous.details[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`]
if (typeof freeMB !== 'number' || typeof volume !== 'string') {
return sample
}
return {
...sample,
details: withSwapVolumeFreeSpace(sample.details, { freeMB, volume }, process.platform, false),
swapVolumeSampledAtMs: previous.swapVolumeSampledAtMs
}
}
function commitHostMemorySample(nowMs: number): boolean {
try {
const details = getSystemMemoryDetails()
// Why not `length === 0`: the signal label is appended unconditionally, so a
// reading that resolved no memory field at all still arrives with one key.
if (!Object.keys(details).some((key) => key !== PRESSURE_SIGNAL_KEY)) {
return false
}
preGoneSample = withCarriedSwapVolume({ details, sampledAtMs: nowMs })
return true
} catch {
// Why: a failed read must not erase the previous good sample.
return false
}
}
async function mergeSwapVolumeFreeSpace(issuedOnTick: number): Promise<void> {
if (swapVolumeReadInFlight) {
return
}
swapVolumeReadInFlight = true
const generation = samplingGeneration
const issuedFor = preGoneSample
try {
const volume = await readSwapVolumeFreeSpace()
if (volume && preGoneSample && generation === samplingGeneration) {
// Why only its own tick qualifies: a statfs that outlived its tick carries a
// pre-storm volume number, and the latch makes that lag unbounded. It still
// ships beside its age, but it may not decide the verdict.
// Why the tick counter and not sample identity: a tick whose host read fails
// leaves the sample object in place, so identity alone reads as co-timed.
const coTimed = issuedOnTick === sampleTick
preGoneSample = {
...preGoneSample,
details: withSwapVolumeFreeSpace(preGoneSample.details, volume, process.platform, coTimed),
swapVolumeSampledAtMs: issuedFor?.sampledAtMs
}
}
} catch {
// Why: the memory reading is already committed and stands on its own.
} finally {
swapVolumeReadInFlight = false
}
}
export async function samplePreGoneSystemMemory(nowMs: number = Date.now()): Promise<void> {
// Why commit before awaiting: the volume read is a statfs, and under the very
// paging storm this targets it is slowest — it must never delay, or (via an
// in-flight latch) skip, the cheap synchronous host reading.
const tick = ++sampleTick
if (!commitHostMemorySample(nowMs)) {
return
}
await mergeSwapVolumeFreeSpace(tick)
}
export function startPreGoneSystemMemorySampling(
intervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS
): void {
if (preGoneTimer) {
return
}
void samplePreGoneSystemMemory()
preGoneTimer = setInterval(() => void samplePreGoneSystemMemory(), intervalMs)
preGoneTimer.unref?.()
}
export function resetPreGoneSystemMemorySamplingForTest(): void {
if (preGoneTimer) {
clearInterval(preGoneTimer)
}
preGoneTimer = null
preGoneSample = null
swapVolumeReadInFlight = false
// Why bump: an already-awaited volume read must not repopulate a reset sample.
samplingGeneration += 1
}
/** Keyed as `systemMemoryPreGone*` so a scan over the `systemMemory` family sees both reads. */
export function preGoneSystemMemoryDetails(nowMs: number): CrashReportDetails {
if (!preGoneSample) {
return {}
}
const details: CrashReportDetails = {
[`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSampleAgeMs`]: Math.max(
0,
nowMs - preGoneSample.sampledAtMs
)
}
// Why its own age: the volume read resolves out of band, so it can be older
// than the memory reading printed beside it, and that gap must be readable.
if (preGoneSample.swapVolumeSampledAtMs !== undefined) {
details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGoneSwapVolumeAgeMs`] = Math.max(
0,
nowMs - preGoneSample.swapVolumeSampledAtMs
)
}
for (const [key, value] of Object.entries(preGoneSample.details)) {
details[`${SYSTEM_MEMORY_KEY_PREFIX}PreGone${key.slice(SYSTEM_MEMORY_KEY_PREFIX.length)}`] =
value
}
return details
}
@@ -2,11 +2,11 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import {
buildProcessGoneCrashDetails,
collectProcessGoneMetricDetails,
resetPreGoneProcessMetricsSamplingForTest,
resetPreGoneCrashSamplingForTest,
samplePreGoneProcessMetrics,
startPreGoneProcessMetricsSampling
startPreGoneCrashSampling
} from './process-gone-diagnostics'
import { setSystemMemoryInfoReaderForTest } from './gone-time-system-memory'
import { setSystemMemoryInfoReaderForTest } from './system-memory-details'
type MetricFixture = {
pid?: number
@@ -27,7 +27,7 @@ vi.mock('electron', () => ({
describe('process gone diagnostics', () => {
beforeEach(() => {
resetPreGoneProcessMetricsSamplingForTest()
resetPreGoneCrashSamplingForTest()
setSystemMemoryInfoReaderForTest(null)
})
@@ -141,8 +141,8 @@ describe('process gone diagnostics', () => {
appMetricsMock.mockReturnValue([
{ pid: 30, type: 'Tab', memory: { workingSetSize: 1024 * 100 } }
])
startPreGoneProcessMetricsSampling(1_000)
startPreGoneProcessMetricsSampling(1_000)
startPreGoneCrashSampling(1_000)
startPreGoneCrashSampling(1_000)
// A crash inside the first interval already has a sample to draw from.
expect(buildProcessGoneCrashDetails({}, 'renderer')).toMatchObject({
@@ -582,12 +582,12 @@ describe('process gone diagnostics', () => {
it("arms an unref'd interval so sampling never holds the event loop open", () => {
const setIntervalSpy = vi.spyOn(globalThis, 'setInterval')
try {
startPreGoneProcessMetricsSampling(60_000)
startPreGoneCrashSampling(60_000)
const timer = setIntervalSpy.mock.results[0]?.value as NodeJS.Timeout
expect(timer.hasRef()).toBe(false)
} finally {
setIntervalSpy.mockRestore()
resetPreGoneProcessMetricsSamplingForTest()
resetPreGoneCrashSamplingForTest()
}
})
@@ -641,7 +641,7 @@ describe('process gone diagnostics', () => {
expect(details.systemMemoryTotalMB).toBe(16_384)
})
it('samples system memory at gone time but never into the pre-gone snapshot', () => {
it('samples system memory at gone time but never into the processMetrics family', () => {
appMetricsMock.mockReturnValue([{ pid: 1, type: 'Browser', memory: { workingSetSize: 0 } }])
samplePreGoneProcessMetrics()
setSystemMemoryInfoReaderForTest(() => ({
@@ -658,7 +658,9 @@ describe('process gone diagnostics', () => {
systemMemorySwapTotalMB: 8_192,
systemMemorySwapFreeMB: 40
})
expect(details.processMetricsPreGoneSystemMemoryTotalMB).toBeUndefined()
expect(
Object.keys(details).filter((key) => key.startsWith('processMetricsPreGoneSystem'))
).toEqual([])
})
it('leaves records unflagged when the crashed bucket is still populated', () => {
@@ -3,7 +3,13 @@ import {
sanitizeCrashReportDetails,
type CrashReportDetailValue
} from '../../shared/crash-reporting'
import { getSystemMemoryAtGoneDetails, memoryKBFieldMB } from './gone-time-system-memory'
import { getSystemMemoryDetails, memoryKBFieldMB } from './system-memory-details'
import {
PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS,
preGoneSystemMemoryDetails,
resetPreGoneSystemMemorySamplingForTest,
startPreGoneSystemMemorySampling
} from './pre-gone-host-memory'
type ProcessMetricLike = {
pid?: unknown
@@ -204,8 +210,9 @@ export function samplePreGoneProcessMetrics(nowMs: number = Date.now()): void {
}
}
export function startPreGoneProcessMetricsSampling(
intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS
export function startPreGoneCrashSampling(
intervalMs: number = PROCESS_METRICS_PRE_GONE_SAMPLE_INTERVAL_MS,
systemMemoryIntervalMs: number = PRE_GONE_SYSTEM_MEMORY_SAMPLE_INTERVAL_MS
): void {
if (preGoneSampleTimer) {
return
@@ -213,14 +220,16 @@ export function startPreGoneProcessMetricsSampling(
samplePreGoneProcessMetrics()
preGoneSampleTimer = setInterval(() => samplePreGoneProcessMetrics(), intervalMs)
preGoneSampleTimer.unref?.()
startPreGoneSystemMemorySampling(systemMemoryIntervalMs)
}
export function resetPreGoneProcessMetricsSamplingForTest(): void {
export function resetPreGoneCrashSamplingForTest(): void {
if (preGoneSampleTimer) {
clearInterval(preGoneSampleTimer)
}
preGoneSampleTimer = null
preGoneSample = null
resetPreGoneSystemMemorySamplingForTest()
}
const PROCESS_METRICS_KEY_PREFIX = 'processMetrics'
@@ -271,7 +280,7 @@ export function buildProcessGoneCrashDetails(
const crashDetails: CrashReportDetails = {
...sanitizedDetails,
...liveMetricDetails,
...getSystemMemoryAtGoneDetails()
...getSystemMemoryDetails()
}
// Why: with the crasher gone, Largest names a survivor — flag that so the
// live buckets are read as "everyone else", not as the crashed process.
@@ -290,8 +299,10 @@ export function buildProcessGoneCrashDetails(
if (liveMetricDetails[crashedBucketCountKey] === 0 || sampledSameBucketProcessVanished) {
crashDetails.processMetricsCrashedProcessAbsent = true
}
const nowMs = Date.now()
if (preGoneSample) {
Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, Date.now()))
Object.assign(crashDetails, preGoneSampleDetails(preGoneSample, nowMs))
}
Object.assign(crashDetails, preGoneSystemMemoryDetails(nowMs))
return crashDetails
}
@@ -0,0 +1,67 @@
import { statfs } from 'node:fs/promises'
import path from 'node:path'
// Why: a system-managed Windows pagefile — and a macOS swapfile — only grows
// into free space on its own volume, so low available commit is a REFUSED
// allocation only when that volume is full too. Linux is excluded on purpose:
// its swap is a fixed partition, a fixed-size swapfile, or zram, none of which
// grow into root-fs free space, so the number would read as headroom that
// cannot exist. The measured volume ships alongside because Windows only names
// the DEFAULT pagefile drive; a relocated pagefile lives elsewhere.
const BYTES_PER_MB = 1024 * 1024
export type SwapVolumeFreeSpace = {
freeMB: number
/** Which volume was measured, separator-trimmed so redaction sees no path. */
volume: string
}
type SwapVolumeFreeSpaceReader = (
platform: NodeJS.Platform
) => Promise<SwapVolumeFreeSpace | undefined>
function swapVolumeAnchor(platform: NodeJS.Platform): string | undefined {
if (platform === 'win32') {
const anchor = process.env.SystemRoot || process.env.SystemDrive
return anchor ? path.parse(anchor).root || anchor : undefined
}
return platform === 'darwin' ? path.sep : undefined
}
function volumeLabel(root: string): string {
const trimmed = root.replace(/[\\/]+$/, '')
return trimmed.length > 0 ? trimmed : root
}
async function statfsSwapVolumeFreeSpace(
platform: NodeJS.Platform
): Promise<SwapVolumeFreeSpace | undefined> {
const root = swapVolumeAnchor(platform)
if (!root) {
return undefined
}
try {
const stats = await statfs(root)
const bytes = Number(stats.bsize) * Number(stats.bavail)
return Number.isFinite(bytes)
? { freeMB: Math.round(Math.max(0, bytes) / BYTES_PER_MB), volume: volumeLabel(root) }
: undefined
} catch {
return undefined
}
}
let swapVolumeFreeSpaceReader: SwapVolumeFreeSpaceReader = statfsSwapVolumeFreeSpace
export function setSwapVolumeFreeSpaceReaderForTest(
reader: SwapVolumeFreeSpaceReader | null
): void {
swapVolumeFreeSpaceReader = reader ?? statfsSwapVolumeFreeSpace
}
export function readSwapVolumeFreeSpace(
platform: NodeJS.Platform = process.platform
): Promise<SwapVolumeFreeSpace | undefined> {
return swapVolumeFreeSpaceReader(platform)
}
@@ -0,0 +1,161 @@
import type { CrashReportDetailValue } from '../../shared/crash-reporting'
import type { SwapVolumeFreeSpace } from './swap-volume-free-space'
// ─── Host system memory for crash reports ───────────────────────────
// Why: the system outlives the crashed process, so this IS sampleable at
// process-gone — it separates "renderer grew huge" from "machine out of
// memory/commit", which the per-process buckets alone cannot. The gone-time
// caller reads AFTER the corpse returned its pages, so free/swapFree read
// healthier than at kill time; the pre-gone sampler carries a live reading past
// that.
// Every reading is labelled `systemMemoryPressureSignal` so no report can be
// read as a pressure verdict the platform never gave:
// win32 — swapFree is MEMORYSTATUSEX.ullAvailPageFile, i.e. available
// COMMIT, which pagefile growth can heal (a 127 MB commit floor healed to
// 2029 MB mid-hold on the win-lowspec repro, killing nothing). Free space on
// the swap volume does NOT establish that it could: a fixed-size or disabled
// pagefile grows into no amount of empty disk, its maximum is unreadable
// here (needs a registry read), and the measured volume is only the DEFAULT
// pagefile drive. So a co-timed volume reading is context beside the commit
// number — `available-commit-volume-cotimed` — never a verdict. The one
// decisive win32 case is a commit limit at or below RAM: no pagefile exists
// to grow, so the floor cannot heal (`available-commit-hard-capped`).
// linux — MemAvailable is the real signal; MemFree is not (it excludes page
// cache and other reclaimable memory).
// darwin — none. `free` stays low on healthy machines and
// fileBacked/purgeable are only a reclaimability proxy. The real signal
// needs `memory_pressure -Q`; Orca's reader for it
// (src/main/memory/host-memory.ts) is on-demand, and spawning a subprocess
// on a 10 s app-lifetime timer costs more than the gap it closes.
type CrashReportDetails = Record<string, CrashReportDetailValue>
export const SYSTEM_MEMORY_KEY_PREFIX = 'systemMemory'
export function memoryKBFieldMB(value: unknown): number | undefined {
const kb = typeof value === 'number' && Number.isFinite(value) ? value : undefined
return kb === undefined ? undefined : Math.round(Math.max(0, kb) / 1024)
}
type SystemMemoryInfoLike = {
total?: unknown
free?: unknown
available?: unknown
swapTotal?: unknown
swapFree?: unknown
fileBacked?: unknown
purgeable?: unknown
}
type SystemMemoryInfoReader = () => SystemMemoryInfoLike | null
/** How far this reading may be read as a "was the host under pressure" verdict. */
export type SystemMemoryPressureSignal =
| 'available-commit-hard-capped'
| 'available-commit-volume-cotimed'
| 'available-commit-unqualified'
| 'mem-available'
| 'none'
function readElectronSystemMemoryInfo(): SystemMemoryInfoLike | null {
const read = (process as NodeJS.Process & { getSystemMemoryInfo?: () => SystemMemoryInfoLike })
.getSystemMemoryInfo
if (typeof read !== 'function') {
return null
}
try {
return read.call(process)
} catch {
return null
}
}
let systemMemoryInfoReader: SystemMemoryInfoReader = readElectronSystemMemoryInfo
export function setSystemMemoryInfoReaderForTest(reader: SystemMemoryInfoReader | null): void {
systemMemoryInfoReader = reader ?? readElectronSystemMemoryInfo
}
function numericDetail(details: CrashReportDetails, suffix: string): number | undefined {
const value = details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`]
return typeof value === 'number' ? value : undefined
}
/** Windows commit limit = RAM + pagefile, so a limit at or below RAM has no pagefile behind it. */
function pagefileBacksCommit(details: CrashReportDetails): boolean | undefined {
const total = numericDetail(details, 'TotalMB')
const swapTotal = numericDetail(details, 'SwapTotalMB')
return total === undefined || swapTotal === undefined ? undefined : swapTotal > total
}
function pressureSignal(
platform: NodeJS.Platform,
details: CrashReportDetails,
volumeCoTimed = true
): SystemMemoryPressureSignal {
if (platform === 'win32' && `${SYSTEM_MEMORY_KEY_PREFIX}SwapFreeMB` in details) {
if (pagefileBacksCommit(details) === false) {
return 'available-commit-hard-capped'
}
return volumeCoTimed && `${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB` in details
? 'available-commit-volume-cotimed'
: 'available-commit-unqualified'
}
if (platform === 'linux' && `${SYSTEM_MEMORY_KEY_PREFIX}AvailableMB` in details) {
return 'mem-available'
}
return 'none'
}
export function getSystemMemoryDetails(
platform: NodeJS.Platform = process.platform
): CrashReportDetails {
const info = systemMemoryInfoReader()
if (!info) {
return {}
}
const details: CrashReportDetails = {}
const fields: readonly [keyof SystemMemoryInfoLike, string][] = [
['total', 'TotalMB'],
['free', 'FreeMB'],
['available', 'AvailableMB'],
['swapTotal', 'SwapTotalMB'],
['swapFree', 'SwapFreeMB'],
['fileBacked', 'FileBackedMB'],
['purgeable', 'PurgeableMB']
]
for (const [field, suffix] of fields) {
const mb = memoryKBFieldMB(info[field])
if (mb !== undefined) {
details[`${SYSTEM_MEMORY_KEY_PREFIX}${suffix}`] = mb
}
}
details[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, details)
return details
}
/**
* Merges the statfs-derived volume datum, which needs an await and so is only
* reachable from the periodic sampler, and relabels the reading it sits beside.
*
* `coTimed` false means the statfs outlived the tick that issued it, so this
* volume number and the commit number beside it describe different moments —
* during a pagefile-growth storm that is exactly when they diverge, and a
* pre-storm 40 GB printed next to 200 MB of commit reads as "the pagefile had
* room", the opposite conclusion. The datum still ships (with its own age), but
* only a co-timed one is named in the label.
*/
export function withSwapVolumeFreeSpace(
details: CrashReportDetails,
volume: SwapVolumeFreeSpace,
platform: NodeJS.Platform = process.platform,
coTimed = true
): CrashReportDetails {
const merged: CrashReportDetails = {
...details,
[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolumeFreeMB`]: volume.freeMB,
[`${SYSTEM_MEMORY_KEY_PREFIX}SwapVolume`]: volume.volume
}
merged[`${SYSTEM_MEMORY_KEY_PREFIX}PressureSignal`] = pressureSignal(platform, merged, coTimed)
return merged
}
@@ -49,6 +49,7 @@ function recordRendererBreadcrumbTrace(
const DUPLICATE_TAB_OWNER_BREADCRUMB = 'terminal_tab_id_owned_by_multiple_worktrees'
const PARK_VERDICT_CHURN_BREADCRUMB = 'terminal_park_verdict_churn'
const REACT_COMMIT_CASCADE_BREADCRUMB = 'react_commit_cascade'
const REPLAY_GUARD_WEDGED_BREADCRUMB = 'terminal_replay_guard_wedged_release'
const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([
'renderer_error',
'renderer_unhandled_rejection',
@@ -56,6 +57,7 @@ const COALESCED_RENDERER_BREADCRUMB_NAMES = new Set([
DUPLICATE_TAB_OWNER_BREADCRUMB,
PARK_VERDICT_CHURN_BREADCRUMB,
REACT_COMMIT_CASCADE_BREADCRUMB,
REPLAY_GUARD_WEDGED_BREADCRUMB,
TERMINAL_WEBGL_DIAGNOSTIC_BREADCRUMB
])
const RENDERER_BREADCRUMB_COALESCE_MS = 30_000
@@ -69,6 +71,11 @@ const RENDERER_BREADCRUMB_COALESCE_MS = 30_000
// 30-entry ring to two such bursts. `suppressedSinceLast` keeps the pane count
// — the only signal these carry — in one slot.
const NAME_ONLY_COALESCED_BREADCRUMB_NAMES = new Set(['terminal_safe_fit_retry_exhausted'])
// Why: the 30-slot ring is the scarce sink; the durable span stream is not. For
// bounded-rate pane telemetry whose multiplicity is the whole signal, spans are the
// only place a burst survives the restart that clears the ring, so coalesce the ring
// but keep every event's span.
const PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES = new Set([REPLAY_GUARD_WEDGED_BREADCRUMB])
function rendererBreadcrumbCoalesceKey(
name: string,
@@ -77,6 +84,13 @@ function rendererBreadcrumbCoalesceKey(
if (NAME_ONLY_COALESCED_BREADCRUMB_NAMES.has(name)) {
return name
}
// Why presence and not value: `ptyId`/`tabIdHash` are absent on the restore call
// site (layout-serialization restoreScrollbackBuffers) and present on reattach, so
// their presence is the call-site identity a mixed burst would otherwise lose. Four
// slots per storm at most, regardless of pane count.
if (name === REPLAY_GUARD_WEDGED_BREADCRUMB) {
return `${name}:${data?.ptyId ? 'pty' : ''}:${data?.tabIdHash ? 'tab' : ''}`
}
// Why trigger and not name alone: `burst` means damping engaged a commit
// short of React #185, `window` means slow benign churn. Collapsing them
// would drop the near-crash signal into a slow-churn slot. Still bounded —
@@ -191,9 +205,13 @@ export function recordRendererBreadcrumbFromRenderer(
minIntervalMs: RENDERER_BREADCRUMB_COALESCE_MS,
...(origin ? { origin } : {})
})
// Why: tracing every suppressed duplicate would preserve the same
// serialization and disk churn that breadcrumb coalescing removes.
if (coalesceResult) {
if (PER_EVENT_TRACED_COALESCED_BREADCRUMB_NAMES.has(args.name)) {
// Why the raw data: every event already gets its own span, so folding the ring's
// running count in here would double-count in any span-stream total.
recordRendererBreadcrumbTrace(args.name, data)
} else if (coalesceResult) {
// Why gated: tracing every suppressed duplicate would preserve the same
// serialization and disk churn that breadcrumb coalescing removes.
recordRendererBreadcrumbTrace(
args.name,
coalesceResult.suppressedSinceLast > 0
@@ -0,0 +1,128 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import {
clearCrashBreadcrumbsForTest,
getCrashBreadcrumbSnapshot,
recordCrashBreadcrumb
} from '../crash-reporting/crash-breadcrumb-store'
import { recordRendererBreadcrumbFromRenderer } from './crash-reporting-renderer-breadcrumbs'
type SpanOptions = { attributes: Record<string, unknown> }
const startSpanMock = vi.fn((_name: string, _options: SpanOptions) => ({ end: () => {} }))
vi.mock('../observability/tracer', () => ({
startSpan: (name: string, options: SpanOptions) => startSpanMock(name, options)
}))
const WEDGE_BREADCRUMB = 'terminal_replay_guard_wedged_release'
/** Reattach-path shape: identity-bearing (`tabIdHash`, optionally `ptyId`). */
function emitReattachWedge(pane: number, withPtyId = false): void {
recordRendererBreadcrumbFromRenderer({
name: WEDGE_BREADCRUMB,
data: {
paneId: pane,
leafIdHash: `leaf${String(pane).padStart(5, '0')}`,
tabIdHash: `tab${String(pane).padStart(6, '0')}`,
worktreeIdHash: 'caa15fa9',
...(withPtyId ? { ptyId: `…@@pty-${pane}` } : {})
}
})
}
/** Restore-path shape (restoreScrollbackBuffers): no tabIdHash, no ptyId. */
function emitRestoreWedge(pane: number): void {
recordRendererBreadcrumbFromRenderer({
name: WEDGE_BREADCRUMB,
data: { paneId: pane, leafIdHash: `leaf${String(pane).padStart(5, '0')}` }
})
}
function wedgeCrumbs(): ReturnType<typeof getCrashBreadcrumbSnapshot> {
return getCrashBreadcrumbSnapshot().filter((entry) => entry.name === WEDGE_BREADCRUMB)
}
function wedgeSpanCount(): number {
return startSpanMock.mock.calls.filter(
(call) => call[1].attributes['breadcrumb.name'] === WEDGE_BREADCRUMB
).length
}
beforeEach(() => {
startSpanMock.mockClear()
})
afterEach(() => {
clearCrashBreadcrumbsForTest()
})
// One mount/reveal/wake transition expires every in-flight replay write at once, so
// the burst reaches the 30-slot ring as N distinct entries. Field span streams measure
// bursts of 26 in 0.96s and 62 over 85s. No captured report in the 09-02 corpus shows
// a ring that actually drained — all nine bursts predate their report's ring window —
// so this bounds a demonstrated hazard, not an observed loss, and must not cost the
// durable span evidence that did carry those bursts.
describe('replay-guard wedge burst against the fixed-size breadcrumb ring', () => {
it('costs one ring slot per call site and preserves the pre-crash trail', () => {
for (let index = 0; index < 10; index += 1) {
recordCrashBreadcrumb(`pre_crash_evidence_${index}`, { index })
}
for (let pane = 0; pane < 26; pane += 1) {
emitReattachWedge(pane)
}
const snapshot = getCrashBreadcrumbSnapshot()
expect(snapshot.filter((entry) => entry.name.startsWith('pre_crash_evidence_'))).toHaveLength(
10
)
expect(wedgeCrumbs()).toHaveLength(1)
})
it('carries the burst multiplicity into the ring as suppressedSinceLast', () => {
for (let pane = 0; pane < 26; pane += 1) {
emitReattachWedge(pane)
}
// 26 emissions: one owns the slot, 25 fold into it.
expect(wedgeCrumbs()[0]?.data?.suppressedSinceLast).toBe(25)
})
// The 121-event field corpus lives entirely in the renderer.breadcrumb span stream,
// and the ring is cleared by the restart that usually precedes the crash report, so
// ring coalescing must not suppress the per-event spans.
it('still emits one durable span per wedge event', () => {
for (let pane = 0; pane < 26; pane += 1) {
emitReattachWedge(pane)
}
expect(wedgeSpanCount()).toBe(26)
// Why no count on the span: one span per event already carries the multiplicity.
expect(
startSpanMock.mock.calls.some((call) =>
JSON.stringify(call[1]).includes('suppressedSinceLast')
)
).toBe(false)
})
// Bundle 8907a508 mixes restore-path (identity-less) and reattach-path crumbs in one
// window; name-only keying would report only the last one's shape.
it('keeps restore-path and reattach-path call sites in separate slots', () => {
emitRestoreWedge(1)
emitRestoreWedge(2)
emitReattachWedge(3)
emitReattachWedge(4, true)
const crumbs = wedgeCrumbs()
expect(crumbs).toHaveLength(3)
expect(crumbs.map((crumb) => Boolean(crumb.data?.tabIdHash))).toEqual([false, true, true])
expect(crumbs.map((crumb) => Boolean(crumb.data?.ptyId))).toEqual([false, false, true])
})
it('bounds a many-pane burst to one slot within a call site', () => {
for (let pane = 0; pane < 40; pane += 1) {
emitReattachWedge(pane, pane % 2 === 0)
}
expect(wedgeCrumbs()).toHaveLength(2)
})
})
@@ -0,0 +1,372 @@
import { mkdir, mkdtemp, realpath, rm, symlink, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join, resolve } from 'node:path'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type { Store } from '../persistence'
import type * as RepoWorktrees from '../repo-worktrees'
import { listRepoWorktreeGraph } from '../repo-worktrees'
import type * as ProjectGroupsModule from '../../shared/project-groups'
import { buildProjectGroupChildIndex, getProjectGroupSubtreeIds } from '../../shared/project-groups'
import { isPathInsideOrEqual } from '../../shared/cross-platform-path'
import { getWorktreeMirrorDistro } from '../project-runtime-git-options'
import type { FolderWorkspace } from '../../shared/folder-workspace-types'
import type { ProjectGroup } from '../../shared/project-group-types'
import type { Project } from '../../shared/project-types'
import type { Repo } from '../../shared/repo-types'
import { getAllowedRoots } from './filesystem-allowed-roots'
import { authorizeExternalPath, resolveAuthorizedPath } from './filesystem-auth'
import { invalidateAuthorizedRootsCache } from './registered-worktree-roots-cache'
import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic'
vi.mock('../repo-worktrees', async () => {
const actual = await vi.importActual<typeof RepoWorktrees>('../repo-worktrees')
return { ...actual, listRepoWorktreeGraph: vi.fn(async () => []) }
})
vi.mock('../../shared/project-groups', async () => {
const actual = await vi.importActual<typeof ProjectGroupsModule>('../../shared/project-groups')
return {
...actual,
buildProjectGroupChildIndex: vi.fn(actual.buildProjectGroupChildIndex),
getProjectGroupSubtreeIds: vi.fn(actual.getProjectGroupSubtreeIds)
}
})
type StoreFixture = {
repos: Repo[]
projects: Project[]
projectGroups: ProjectGroup[]
folderWorkspaces: FolderWorkspace[]
workspaceDir?: string
}
type StoreCallCounts = {
getRepos: number
getProjects: number
getProjectGroups: number
getFolderWorkspaces: number
}
function makeCountingStore(fixture: StoreFixture): { store: Store; counts: StoreCallCounts } {
const counts: StoreCallCounts = {
getRepos: 0,
getProjects: 0,
getProjectGroups: 0,
getFolderWorkspaces: 0
}
const store = {
getRepos: () => {
counts.getRepos += 1
// Match the real store, which rehydrates fresh repo objects on every read.
return fixture.repos.map((repo) => ({ ...repo }))
},
getProjects: () => {
counts.getProjects += 1
return fixture.projects.map((project) => ({ ...project }))
},
getProjectGroups: () => {
counts.getProjectGroups += 1
return fixture.projectGroups.map((group) => ({ ...group }))
},
getFolderWorkspaces: () => {
counts.getFolderWorkspaces += 1
return fixture.folderWorkspaces.map((workspace) => ({ ...workspace }))
},
getSettings: () => ({ nestWorkspaces: false, workspaceDir: fixture.workspaceDir ?? '' })
} as unknown as Store
return { store, counts }
}
/**
* The pre-change `getAllowedRoots` algorithm, kept verbatim so the equivalence test compares the
* new root list against the old one rather than against a hand-written expectation.
*/
function referenceAllowedRoots(store: Store): string[] {
const scopeStore = store as unknown as {
getRepos: () => Repo[]
getProjectGroups?: () => ProjectGroup[]
getFolderWorkspaces?: () => FolderWorkspace[]
getSettings: () => { workspaceDir?: string; nestWorkspaces?: boolean }
}
const localRepos = scopeStore.getRepos().filter((repo) => !repo.connectionId)
const settings = scopeStore.getSettings()
const scopeRepos = scopeStore.getRepos()
const projectGroups = scopeStore.getProjectGroups?.() ?? []
const isRemoteOnly = (
folderPath: string,
projectGroupId: string,
connectionId: string | null | undefined
): boolean => {
if (connectionId) {
return true
}
const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId)
const candidates = scopeRepos.filter(
(repo) =>
(typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) ||
isPathInsideOrEqual(folderPath, repo.path)
)
return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId))
}
const folderScopeRoots: string[] = []
for (const group of projectGroups) {
if (group.parentPath && !isRemoteOnly(group.parentPath, group.id, group.connectionId)) {
folderScopeRoots.push(resolve(group.parentPath))
}
}
for (const workspace of scopeStore.getFolderWorkspaces?.() ?? []) {
const connectionId =
workspace.connectionId ??
projectGroups.find((group) => group.id === workspace.projectGroupId)?.connectionId ??
null
if (!isRemoteOnly(workspace.folderPath, workspace.projectGroupId, connectionId)) {
folderScopeRoots.push(resolve(workspace.folderPath))
}
}
const roots = [...localRepos.map((repo) => resolve(repo.path)), ...folderScopeRoots]
if (settings.workspaceDir) {
if (localRepos.length === 0) {
roots.push(resolve(settings.workspaceDir))
} else {
for (const repo of localRepos) {
roots.push(
resolve(
computeWorkspaceRoot(
repo.path,
getWorktreePathSettings(repo, settings as never, getWorktreeMirrorDistro(store, repo))
)
)
)
}
}
}
return roots
}
function makeRepo(overrides: Partial<Repo> & Pick<Repo, 'id' | 'path'>): Repo {
return {
displayName: overrides.id,
badgeColor: '#000000',
addedAt: 1,
kind: 'git',
...overrides
}
}
function makeGroup(overrides: Partial<ProjectGroup> & Pick<ProjectGroup, 'id'>): ProjectGroup {
return {
name: overrides.id,
parentPath: null,
parentGroupId: null,
createdFrom: 'folder-scan',
tabOrder: 0,
isCollapsed: false,
color: null,
createdAt: 1,
updatedAt: 1,
...overrides
}
}
function makeWorkspace(
overrides: Partial<FolderWorkspace> & Pick<FolderWorkspace, 'id' | 'folderPath'>
): FolderWorkspace {
return {
projectGroupId: 'group-root',
name: overrides.id,
comment: '',
linkedTask: null,
isArchived: false,
isUnread: false,
isPinned: false,
sortOrder: 1,
lastActivityAt: 1,
createdAt: 1,
updatedAt: 1,
...overrides
}
}
/** Repos, nested groups, folder workspaces (one not a git worktree), and an SSH repo. */
function makeMixedFixture(): StoreFixture {
const repos = [
makeRepo({ id: 'repo-local', path: '/repos/app', projectGroupId: 'group-root' }),
makeRepo({ id: 'repo-nested', path: '/repos/nested', projectGroupId: 'group-child' }),
makeRepo({ id: 'repo-folder', path: '/folders/plain', kind: 'folder' }),
makeRepo({
id: 'repo-ssh',
path: '/remote/app',
connectionId: 'ssh-1',
projectGroupId: 'group-remote'
})
]
const projectGroups = [
makeGroup({ id: 'group-root', parentPath: '/folders/root' }),
makeGroup({ id: 'group-child', parentGroupId: 'group-root', parentPath: '/folders/child' }),
makeGroup({ id: 'group-grandchild', parentGroupId: 'group-child' }),
makeGroup({ id: 'group-remote', parentPath: '/remote/scope' }),
makeGroup({ id: 'group-connection', parentPath: '/remote/via-group', connectionId: 'ssh-1' })
]
const folderWorkspaces = [
makeWorkspace({ id: 'ws-git', folderPath: '/folders/root/feature' }),
// Not a git worktree: a plain folder workspace under a folder-kind repo.
makeWorkspace({
id: 'ws-plain',
folderPath: '/folders/plain/scratch',
projectGroupId: 'group-child'
}),
makeWorkspace({ id: 'ws-remote', folderPath: '/remote/ws', projectGroupId: 'group-remote' }),
makeWorkspace({
id: 'ws-connection',
folderPath: '/remote/direct',
projectGroupId: 'group-connection'
}),
makeWorkspace({
id: 'ws-unlinked',
folderPath: '/folders/unlinked',
projectGroupId: 'group-orphan'
})
]
const projects: Project[] = [
{
id: 'project-1',
displayName: 'App',
badgeColor: '#000000',
sourceRepoIds: ['repo-local', 'repo-nested'],
createdAt: 1,
updatedAt: 1
},
{
id: 'project-2',
displayName: 'Folder',
badgeColor: '#000000',
sourceRepoIds: ['repo-folder'],
createdAt: 1,
updatedAt: 1
}
]
return { repos, projects, projectGroups, folderWorkspaces, workspaceDir: '/workspaces' }
}
beforeEach(() => {
invalidateAuthorizedRootsCache()
vi.mocked(buildProjectGroupChildIndex).mockClear()
vi.mocked(getProjectGroupSubtreeIds).mockClear()
})
describe('getAllowedRoots', () => {
it('produces the same roots as the pre-change implementation', () => {
const { store } = makeCountingStore(makeMixedFixture())
expect(getAllowedRoots(store)).toEqual(referenceAllowedRoots(store))
})
it('reads the store once and indexes project groups once per build', () => {
const fixture = makeMixedFixture()
const { store, counts } = makeCountingStore(fixture)
getAllowedRoots(store)
expect.soft(counts.getRepos).toBe(1)
expect.soft(counts.getProjectGroups).toBe(1)
expect.soft(counts.getFolderWorkspaces).toBe(1)
// Batched runtime resolution scans the project list once, not once per local repo.
expect.soft(counts.getProjects).toBe(1)
// The per-scope subtree walk no longer rebuilds the parent->children index.
expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(1)
expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled()
})
})
describe('resolveAuthorizedPath allowed-root reuse', () => {
let repoRoot: string
let outsideRoot: string
let store: Store
let counts: StoreCallCounts
beforeEach(async () => {
repoRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-allowed-roots-'))
outsideRoot = await mkdtemp(join(await realpath(tmpdir()), 'orca-outside-'))
const fixture = makeMixedFixture()
fixture.repos = [makeRepo({ id: 'repo-local', path: repoRoot }), ...fixture.repos]
fixture.projects[0]!.sourceRepoIds = ['repo-local']
;({ store, counts } = makeCountingStore(fixture))
})
afterEach(async () => {
await rm(repoRoot, { recursive: true, force: true })
await rm(outsideRoot, { recursive: true, force: true })
})
it('builds the allowed-root list once per call across repeated reads', async () => {
const dirPath = join(repoRoot, 'src')
await mkdir(dirPath)
await writeFile(join(dirPath, 'index.ts'), 'export {}\n')
const callCount = 5
for (let index = 0; index < callCount; index += 1) {
await resolveAuthorizedPath(dirPath, store)
await resolveAuthorizedPath(join(dirPath, 'index.ts'), store)
}
const buildCount = callCount * 2
// One build per authorization, not one per raw-path check plus one per realpath check.
expect.soft(counts.getFolderWorkspaces).toBe(buildCount)
expect.soft(counts.getRepos).toBe(buildCount)
expect.soft(counts.getProjects).toBe(buildCount)
expect.soft(vi.mocked(buildProjectGroupChildIndex)).toHaveBeenCalledTimes(buildCount)
expect.soft(vi.mocked(getProjectGroupSubtreeIds)).not.toHaveBeenCalled()
})
// Why (both symlink cases): creating a symlink on Windows needs elevation or
// Developer Mode, so these would fail EPERM in setup rather than exercise the
// escape check. Every non-symlink case still runs there.
it.skipIf(process.platform === 'win32')(
'still refuses a symlink that escapes every allowed root',
async () => {
const secret = join(outsideRoot, 'secret.txt')
await writeFile(secret, 'secret\n')
const escape = join(repoRoot, 'escape.txt')
await symlink(secret, escape)
await expect(resolveAuthorizedPath(escape, store)).rejects.toThrow('Access denied')
expect(vi.mocked(listRepoWorktreeGraph)).toHaveBeenCalled()
}
)
it('builds no allowed-root list at all for a granted external path', async () => {
const external = join(outsideRoot, 'external.md')
await writeFile(external, 'notes\n')
authorizeExternalPath(external)
counts.getRepos = 0
counts.getProjects = 0
counts.getFolderWorkspaces = 0
for (let index = 0; index < 5; index += 1) {
await expect(resolveAuthorizedPath(external, store)).resolves.toBe(external)
}
// The grant answers on its own; hoisting the snapshot must not turn zero builds into one per read.
expect.soft(counts.getRepos).toBe(0)
expect.soft(counts.getProjects).toBe(0)
expect.soft(counts.getFolderWorkspaces).toBe(0)
expect.soft(vi.mocked(buildProjectGroupChildIndex)).not.toHaveBeenCalled()
})
it.skipIf(process.platform === 'win32')(
'still refuses a directory symlink that escapes every allowed root',
async () => {
const outsideDir = join(outsideRoot, 'nested')
await mkdir(outsideDir)
await writeFile(join(outsideDir, 'file.txt'), 'secret\n')
const escape = join(repoRoot, 'escape-dir')
await symlink(outsideDir, escape)
await expect(resolveAuthorizedPath(join(escape, 'file.txt'), store)).rejects.toThrow(
'Access denied'
)
}
)
})
+44 -15
View File
@@ -1,9 +1,16 @@
import { resolve } from 'node:path'
import type { Store } from '../persistence'
import { computeWorkspaceRoot, getWorktreePathSettings } from './worktree-logic'
import { getWorktreeMirrorDistro } from '../project-runtime-git-options'
import {
getWorktreeMirrorDistroForRuntime,
resolveLocalProjectRuntimesForRepos
} from '../project-runtime-git-options'
import { isPathInsideOrEqual } from '../../shared/cross-platform-path'
import { getProjectGroupSubtreeIds } from '../../shared/project-groups'
import {
buildProjectGroupChildIndex,
collectProjectGroupSubtreeIds,
type ProjectGroupChildIndex
} from '../../shared/project-groups'
import type { FolderWorkspace } from '../../shared/folder-workspace-types'
import type { ProjectGroup } from '../../shared/project-group-types'
import type { Repo } from '../../shared/repo-types'
@@ -11,18 +18,22 @@ import type { Repo } from '../../shared/repo-types'
type FolderScopeStore = Pick<Store, 'getRepos'> &
Partial<Pick<Store, 'getProjectGroups' | 'getFolderWorkspaces'>>
// Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths.
function filterLocalRepos(repos: readonly Repo[]): Repo[] {
return repos.filter((repo) => !repo.connectionId)
}
export function getLocalRepos(store: Store) {
// Why: SSH repo paths are remote-host paths; treating them as local roots could authorize unrelated local folders or probe SSH-only paths.
return store.getRepos().filter((repo) => !repo.connectionId)
return filterLocalRepos(store.getRepos())
}
function getFolderScopeCandidateRepos(
folderPath: string,
projectGroupId: string,
projectGroups: readonly ProjectGroup[],
childGroupIndex: ProjectGroupChildIndex,
repos: readonly Repo[]
): Repo[] {
const groupIds = getProjectGroupSubtreeIds(projectGroups, projectGroupId)
const groupIds = collectProjectGroupSubtreeIds(childGroupIndex, projectGroupId)
return repos.filter(
(repo) =>
(typeof repo.projectGroupId === 'string' && groupIds.has(repo.projectGroupId)) ||
@@ -34,13 +45,18 @@ function isRemoteOnlyFolderScope(
folderPath: string,
projectGroupId: string,
connectionId: string | null | undefined,
projectGroups: readonly ProjectGroup[],
childGroupIndex: ProjectGroupChildIndex,
repos: readonly Repo[]
): boolean {
if (connectionId) {
return true
}
const candidates = getFolderScopeCandidateRepos(folderPath, projectGroupId, projectGroups, repos)
const candidates = getFolderScopeCandidateRepos(
folderPath,
projectGroupId,
childGroupIndex,
repos
)
return candidates.length > 0 && candidates.every((repo) => Boolean(repo.connectionId))
}
@@ -55,16 +71,22 @@ function getFolderWorkspaceConnectionId(
)
}
function getLocalFolderScopeRoots(store: Store): string[] {
function getLocalFolderScopeRoots(store: Store, repos: readonly Repo[]): string[] {
const scopeStore = store as FolderScopeStore
const repos = scopeStore.getRepos()
// Why: many filesystem tests use narrow Store doubles; folder scopes are additive.
const projectGroups = scopeStore.getProjectGroups?.() ?? []
const childGroupIndex = buildProjectGroupChildIndex(projectGroups)
const roots: string[] = []
for (const group of projectGroups) {
if (
group.parentPath &&
!isRemoteOnlyFolderScope(group.parentPath, group.id, group.connectionId, projectGroups, repos)
!isRemoteOnlyFolderScope(
group.parentPath,
group.id,
group.connectionId,
childGroupIndex,
repos
)
) {
roots.push(resolve(group.parentPath))
}
@@ -75,7 +97,7 @@ function getLocalFolderScopeRoots(store: Store): string[] {
workspace.folderPath,
workspace.projectGroupId,
getFolderWorkspaceConnectionId(workspace, projectGroups),
projectGroups,
childGroupIndex,
repos
)
) {
@@ -86,16 +108,19 @@ function getLocalFolderScopeRoots(store: Store): string[] {
}
export function getAllowedRoots(store: Store): string[] {
const localRepos = getLocalRepos(store)
// Why one read: `getRepos` rehydrates every repo, and this runs twice per filesystem IPC.
const repos = store.getRepos()
const localRepos = filterLocalRepos(repos)
const settings = store.getSettings()
const roots = [
...localRepos.map((repo) => resolve(repo.path)),
...getLocalFolderScopeRoots(store)
...getLocalFolderScopeRoots(store, repos)
]
if (settings.workspaceDir) {
if (localRepos.length === 0) {
roots.push(resolve(settings.workspaceDir))
} else {
const projectRuntimeByRepoId = resolveLocalProjectRuntimesForRepos(store, localRepos)
for (const repo of localRepos) {
roots.push(
resolve(
@@ -104,7 +129,11 @@ export function getAllowedRoots(store: Store): string[] {
// Why enriched here too: placement has to agree with the create
// flow, or renderer file access is denied for a worktree Orca
// just put on the WSL side.
getWorktreePathSettings(repo, settings, getWorktreeMirrorDistro(store, repo))
getWorktreePathSettings(
repo,
settings,
getWorktreeMirrorDistroForRuntime(projectRuntimeByRepoId.get(repo.id))
)
)
)
)
+51 -14
View File
@@ -43,7 +43,24 @@ export function authorizeExternalPath(targetPath: string): void {
} catch {}
}
export function isPathAllowed(targetPath: string, store: Store): boolean {
/**
* One allowed-root list shared by every check in a single authorization.
*
* Lazy so a path already covered by an external grant still builds nothing at all, the way it did
* before the list was hoisted out of the individual checks.
*/
type AllowedRootsSnapshot = { get: () => readonly string[] }
function createAllowedRootsSnapshot(store: Store): AllowedRootsSnapshot {
let roots: readonly string[] | undefined
return { get: () => (roots ??= getAllowedRoots(store)) }
}
export function isPathAllowed(
targetPath: string,
store: Store,
allowedRoots?: AllowedRootsSnapshot
): boolean {
const resolvedTarget = resolve(targetPath)
if (authorizedExternalPaths.has(resolvedTarget)) {
return true
@@ -53,7 +70,9 @@ export function isPathAllowed(targetPath: string, store: Store): boolean {
return true
}
}
return getAllowedRoots(store).some((root) => isDescendantOrEqual(resolvedTarget, root))
return (allowedRoots?.get() ?? getAllowedRoots(store)).some((root) =>
isDescendantOrEqual(resolvedTarget, root)
)
}
export type ResolveAuthorizedPathOptions = {
@@ -69,7 +88,10 @@ export async function resolveAuthorizedPath(
options: ResolveAuthorizedPathOptions = {}
): Promise<string> {
const resolvedTarget = resolve(targetPath)
if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store))) {
// Why: the roots depend only on store state, not on the candidate path, so one snapshot serves
// every authorization below; each candidate is still checked against it in full.
const allowedRoots = createAllowedRootsSnapshot(store)
if (!(await isPathAllowedIncludingRegisteredWorktrees(resolvedTarget, store, { allowedRoots }))) {
throw new Error(PATH_ACCESS_DENIED_MESSAGE)
}
@@ -80,14 +102,15 @@ export async function resolveAuthorizedPath(
realParent = await realpath(dirname(resolvedTarget))
} catch (error) {
if (isENOENT(error)) {
return resolveAuthorizedMissingPath(resolvedTarget, store)
return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots)
}
throw error
}
const candidateTarget = resolve(realParent, basename(resolvedTarget))
if (
!(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, {
canonicalSourcePath: resolvedTarget
canonicalSourcePath: resolvedTarget,
allowedRoots
}))
) {
throw new Error(PATH_ACCESS_DENIED_MESSAGE)
@@ -100,7 +123,8 @@ export async function resolveAuthorizedPath(
const realTarget = resolve(await realpath(resolvedTarget))
if (
!(await isPathAllowedIncludingRegisteredWorktrees(realTarget, store, {
canonicalSourcePath: resolvedTarget
canonicalSourcePath: resolvedTarget,
allowedRoots
}))
) {
throw new Error(PATH_ACCESS_DENIED_MESSAGE)
@@ -110,11 +134,15 @@ export async function resolveAuthorizedPath(
if (!isENOENT(error)) {
throw error
}
return resolveAuthorizedMissingPath(resolvedTarget, store)
return resolveAuthorizedMissingPath(resolvedTarget, store, allowedRoots)
}
}
async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store): Promise<string> {
async function resolveAuthorizedMissingPath(
resolvedTarget: string,
store: Store,
allowedRoots: AllowedRootsSnapshot
): Promise<string> {
let existingAncestor = resolvedTarget
const missingSegments: string[] = []
@@ -124,7 +152,8 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store
const candidateTarget = resolve(realAncestor, ...missingSegments)
if (
!(await isPathAllowedIncludingRegisteredWorktrees(candidateTarget, store, {
canonicalSourcePath: resolvedTarget
canonicalSourcePath: resolvedTarget,
allowedRoots
}))
) {
throw new Error(PATH_ACCESS_DENIED_MESSAGE)
@@ -148,9 +177,9 @@ async function resolveAuthorizedMissingPath(resolvedTarget: string, store: Store
async function isPathAllowedIncludingRegisteredWorktrees(
targetPath: string,
store: Store,
options: { canonicalSourcePath?: string } = {}
options: { canonicalSourcePath?: string; allowedRoots?: AllowedRootsSnapshot } = {}
): Promise<boolean> {
if (isPathAllowed(targetPath, store)) {
if (isPathAllowed(targetPath, store, options.allowedRoots)) {
return true
}
@@ -158,7 +187,14 @@ async function isPathAllowedIncludingRegisteredWorktrees(
return true
}
if (await isPathAllowedByCanonicalAllowedRoot(targetPath, options.canonicalSourcePath, store)) {
if (
await isPathAllowedByCanonicalAllowedRoot(
targetPath,
options.canonicalSourcePath,
store,
options.allowedRoots
)
) {
return true
}
@@ -178,12 +214,13 @@ async function isPathAllowedIncludingRegisteredWorktrees(
async function isPathAllowedByCanonicalAllowedRoot(
targetPath: string,
sourcePath: string | undefined,
store: Store
store: Store,
allowedRoots?: AllowedRootsSnapshot
): Promise<boolean> {
if (!sourcePath) {
return false
}
for (const root of getAllowedRoots(store)) {
for (const root of allowedRoots?.get() ?? getAllowedRoots(store)) {
const resolvedRoot = resolve(root)
if (!isDescendantOrEqual(sourcePath, resolvedRoot)) {
continue
+29 -1
View File
@@ -1,4 +1,5 @@
import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment'
import { recordCoalescedDurableCrashBreadcrumb } from './crash-reporting/durable-crash-breadcrumb'
/**
* PIDs of Orca's own Chromium processes — browser, renderers, GPU, utilities.
@@ -11,6 +12,14 @@ import { getAppEnvironment, hasAppEnvironment } from '../shared/app-environment'
* Empty on a Node host and empty on failure: that is "no refusal proven", never
* "safe to kill" — callers must keep every other guard they already have.
*
* Why failure stays open rather than refusing everything: a refusal is not free.
* `terminateWindowsProcessTree` resolves without killing, and
* `killSourceControlAgentProcess` returns that straight to a caller that then
* releases the managed-home lock, so failing closed would trade one unreadable
* metrics table for every PTY, git, codex and notebook tree in main leaking at
* once. The `own_chromium_pids_unreadable` crumb is the price of that choice:
* without it a throw is byte-identical to "no Chromium on this host".
*
* Host coverage: only Electron main installs a Chromium-backed AppEnvironment
* (main-process-preflight). The standalone daemon installs none and `orcad`
* installs a Node one whose `getAppMetrics()` is `[]`, so this set is empty in
@@ -30,7 +39,26 @@ export function readOrcaChromiumProcessPids(): ReadonlySet<number> {
.map((metric) => metric.pid)
.filter((pid) => Number.isInteger(pid) && pid > 0)
return new Set(pids)
} catch {
} catch (error) {
recordUnreadableOwnChromiumMetrics(error)
return new Set()
}
}
// Why coalesced: the gate reads this set on every tree kill, so a persistently
// broken metrics table would otherwise flood the 30-slot ring it shares.
const UNREADABLE_METRICS_COALESCE_MS = 60_000
function recordUnreadableOwnChromiumMetrics(error: unknown): void {
try {
recordCoalescedDurableCrashBreadcrumb({
name: 'own_chromium_pids_unreadable',
data: { cause: error instanceof Error ? error.message : String(error) },
coalesceKey: 'own-chromium-pids-unreadable',
minIntervalMs: UNREADABLE_METRICS_COALESCE_MS
})
} catch {
// Diagnostics must never turn an admitted kill into a thrown one: callers
// read this set outside their own try.
}
}
@@ -143,6 +143,40 @@ describe('refusing to tree-kill our own Chromium processes', () => {
)
})
/**
* Fail-open is the deliberate choice — see `orca-chromium-process-pids.ts` for
* why refusing everything is worse — so the crumb is the only thing that keeps
* an unreadable metrics table distinguishable from a host that has no Chromium.
*/
it('leaves proof, and still admits the kill, when the Chromium metrics cannot be read', () => {
appMetricsMock.mockImplementation(() => {
throw new Error('getAppMetrics unavailable')
})
expect([...readOrcaChromiumProcessPids()]).toEqual([])
// Coalesced: the gate reads this set on every kill, so a broken table must
// not evict the ring it shares with the refusal crumb.
expect([...readOrcaChromiumProcessPids()]).toEqual([])
expect(
admitSelfInitiatedTreeKill({
pid: RENDERER_PID,
site: 'pty-descendant-sweep',
scope: 'win-taskkill-tree'
})
).toBe(true)
expect(
getCrashBreadcrumbSnapshot().filter(
(breadcrumb) => breadcrumb.name === 'own_chromium_pids_unreadable'
)
).toEqual([
expect.objectContaining({
name: 'own_chromium_pids_unreadable',
data: expect.objectContaining({ cause: 'getAppMetrics unavailable' })
})
])
})
it('refuses an own-Chromium pid at the gate the account teardowns share', () => {
expect(
admitSelfInitiatedTreeKill({
+6 -1
View File
@@ -102,7 +102,12 @@ export function getWorktreeMirrorDistro(
store: ProjectRuntimeResolutionStore,
repo: Repo
): string | undefined {
const projectRuntime = resolveLocalProjectRuntimeForRepo(store, repo)
return getWorktreeMirrorDistroForRuntime(resolveLocalProjectRuntimeForRepo(store, repo))
}
export function getWorktreeMirrorDistroForRuntime(
projectRuntime: ProjectExecutionRuntimeResolution | undefined
): string | undefined {
if (!projectRuntime || projectRuntime.status !== 'resolved') {
return undefined
}
+2 -1
View File
@@ -44,7 +44,8 @@ export class SshGitReadProvider {
}
}
private invalidateGitReads(): void {
/** Overridden by subclasses that own additional read caches (worktree listings). */
protected invalidateGitReads(): void {
this.gitDiffReadDedupe.clear()
this.statusReadLeaseOwner.invalidate()
this.upstreamStatusReadOwner.invalidate()
@@ -0,0 +1,156 @@
/**
* Local repos coalesce concurrent `git worktree list` scans (`shareWorktreeScan`); the SSH path
* branched away from that and paid one relay round trip per independent caller (`worktrees:list`,
* `worktrees:listAll`, the space repo scan, provisioned-root adoption). These are call counters.
*/
import { describe, expect, it } from 'vitest'
import { SshGitProvider } from './ssh-git-provider'
import { createMockMux, type MockMultiplexer } from './ssh-git-provider-test-harness'
const REPO_PATH = '/home/user/repo'
const WORKTREES = [
{ path: REPO_PATH, head: 'abc123', branch: 'main', isBare: false, isMainWorktree: true }
]
type Deferred = { resolve: (value: unknown) => void; reject: (error: unknown) => void }
/** Holds `git.listWorktrees` open so overlap is deterministic; answers everything else at once. */
function createPendingListMux(): { mux: MockMultiplexer; listDeferreds: Deferred[] } {
const mux = createMockMux()
const listDeferreds: Deferred[] = []
mux.request.mockImplementation((method: string) => {
if (method !== 'git.listWorktrees') {
return Promise.resolve(undefined)
}
return new Promise((resolve, reject) => {
listDeferreds.push({ resolve, reject })
})
})
return { mux, listDeferreds }
}
const flush = (): Promise<void> => new Promise((resolve) => setTimeout(resolve, 0))
function countListRequests(mux: MockMultiplexer): number {
return mux.request.mock.calls.filter((call) => call[0] === 'git.listWorktrees').length
}
describe('SSH git.listWorktrees in-flight dedupe', () => {
it('collapses concurrent listings of one repo into a single relay request', async () => {
const { mux, listDeferreds } = createPendingListMux()
const provider = new SshGitProvider('conn-1', mux as never)
const listings = Array.from({ length: 6 }, () => provider.listWorktrees(REPO_PATH))
await flush()
expect(countListRequests(mux)).toBe(1)
expect(mux.request).toHaveBeenCalledWith(
'git.listWorktrees',
{ repoPath: REPO_PATH },
{ signal: undefined }
)
listDeferreds[0].resolve(WORKTREES)
expect(await Promise.all(listings)).toEqual(Array.from({ length: 6 }, () => WORKTREES))
})
it('does not share across repos or connections', async () => {
const { mux } = createPendingListMux()
const provider = new SshGitProvider('conn-1', mux as never)
void provider.listWorktrees(REPO_PATH)
void provider.listWorktrees('/home/user/other')
await flush()
expect(countListRequests(mux)).toBe(2)
const second = createPendingListMux()
void new SshGitProvider('conn-2', second.mux as never).listWorktrees(REPO_PATH)
await flush()
expect(countListRequests(second.mux)).toBe(1)
expect(countListRequests(mux)).toBe(2)
})
it('keeps a signalled listing on its own request', async () => {
const { mux } = createPendingListMux()
const provider = new SshGitProvider('conn-1', mux as never)
void provider.listWorktrees(REPO_PATH)
await flush()
const controller = new AbortController()
void provider.listWorktrees(REPO_PATH, { signal: controller.signal })
await flush()
expect(countListRequests(mux)).toBe(2)
expect(mux.request).toHaveBeenCalledWith(
'git.listWorktrees',
{ repoPath: REPO_PATH },
{ signal: controller.signal }
)
})
it('re-requests after the shared listing settles instead of caching it', async () => {
const { mux, listDeferreds } = createPendingListMux()
const provider = new SshGitProvider('conn-1', mux as never)
const first = provider.listWorktrees(REPO_PATH)
await flush()
listDeferreds[0].resolve(WORKTREES)
await first
void provider.listWorktrees(REPO_PATH)
await flush()
expect(countListRequests(mux)).toBe(2)
})
it.each([
['addWorktree', (p: SshGitProvider) => p.addWorktree(REPO_PATH, 'feature', '/home/user/feat')],
['removeWorktree', (p: SshGitProvider) => p.removeWorktree('/home/user/feat')]
])('invalidates the shared listing after %s', async (_name, mutate) => {
const { mux } = createPendingListMux()
const provider = new SshGitProvider('conn-1', mux as never)
void provider.listWorktrees(REPO_PATH)
await flush()
expect(countListRequests(mux)).toBe(1)
await mutate(provider)
// The catalog moved, so a joiner must not inherit the pre-mutation scan.
void provider.listWorktrees(REPO_PATH)
await flush()
expect(countListRequests(mux)).toBe(2)
})
it('shares a failed listing with its joiners and re-requests afterwards', async () => {
const { mux, listDeferreds } = createPendingListMux()
const provider = new SshGitProvider('conn-1', mux as never)
const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)]
await flush()
expect(countListRequests(mux)).toBe(1)
const failure = new Error('relay request failed')
listDeferreds[0].reject(failure)
await expect(listings[0]).rejects.toBe(failure)
await expect(listings[1]).rejects.toBe(failure)
void provider.listWorktrees(REPO_PATH)
await flush()
expect(countListRequests(mux)).toBe(2)
})
it('refuses an unauthoritative relay answer for every joiner (#14004)', async () => {
const { mux, listDeferreds } = createPendingListMux()
const provider = new SshGitProvider('conn-1', mux as never)
const listings = [provider.listWorktrees(REPO_PATH), provider.listWorktrees(REPO_PATH)]
await flush()
listDeferreds[0].resolve([])
await expect(listings[0]).rejects.toThrow()
await expect(listings[1]).rejects.toThrow()
})
})
@@ -2,6 +2,7 @@ import type { GitStatusResult } from '../../shared/git-status-types'
import type { RemoveWorktreeResult } from '../../shared/worktree/create-types'
import type { GitWorktreeInfo } from '../../shared/worktree/types'
import { CapabilityProbeCache } from '../../shared/capability-probe-cache'
import { InFlightPromiseDedupe, stableInFlightKey } from '../../shared/in-flight-promise-dedupe'
import { assertAuthoritativeWorktreeCatalog } from '../../shared/worktree/worktree-catalog-availability'
import { isJsonRpcMethodNotFoundError } from './ssh-git-relay-errors'
import { SshGitReviewHeadProvider } from './ssh-git-review-head-provider'
@@ -29,16 +30,35 @@ export class SshGitWorktreeProvider extends SshGitReviewHeadProvider {
private readonly worktreeIsCleanCapabilityCache = new CapabilityProbeCache<
typeof WORKTREE_IS_CLEAN_CAPABILITY
>(Number.POSITIVE_INFINITY)
// Scoped to this provider instance, so two SSH hosts never share an entry.
private readonly worktreeListDedupe = new InFlightPromiseDedupe<GitWorktreeInfo[]>()
protected override invalidateGitReads(): void {
super.invalidateGitReads()
this.worktreeListDedupe.clear()
}
/** Un-signalled reads of one repo coalesce onto the request already in flight; nothing is cached. */
async listWorktrees(
repoPath: string,
options?: { signal?: AbortSignal }
): Promise<GitWorktreeInfo[]> {
const response = await this.mux.request(
'git.listWorktrees',
{ repoPath },
{ signal: options?.signal }
// Why: same rule as shareWorktreeScan — one caller's abort must not cancel the scan its
// joiners are still waiting on, so a signalled read keeps its own request.
if (options?.signal) {
return this.requestWorktreeList(repoPath, options.signal)
}
return this.worktreeListDedupe.run(stableInFlightKey(['listWorktrees', repoPath]), () =>
this.requestWorktreeList(repoPath)
)
}
/** The one real relay round trip a coalesced read's joiners all wait on. */
private async requestWorktreeList(
repoPath: string,
signal?: AbortSignal
): Promise<GitWorktreeInfo[]> {
const response = await this.mux.request('git.listWorktrees', { repoPath }, { signal })
// Why (#14004): relays before this fix answered a failed worktree scan with `[]`. Mixed versions are
// normal, so refuse the shape here too — a Git repo always lists its own checkout.
return assertAuthoritativeWorktreeCatalog<GitWorktreeInfo>(response, repoPath)
@@ -0,0 +1,68 @@
/**
* Ratchet (#18419): `pty.inspectProcess` must NOT be in-flight coalesced the way the sibling git
* reads in `SshGitReadProvider` are. The host mints one `observationEpoch` per request and the
* pane foreground reader commits that epoch per read, so a shared reply reads as a stale replay to
* the second reader to settle — see the companion renderer proof in
* `src/renderer/src/components/terminal-pane/pane-foreground-inspect-observation-identity.test.ts`.
* These are request counters, not timings.
*/
import { describe, expect, it, vi } from 'vitest'
import { createSshPtyProviderRpcOperations } from './ssh-pty-provider-rpc-operations'
const RELAY_PTY_ID = 'pty-1'
const APP_PTY_ID = `ssh:conn-1@@${RELAY_PTY_ID}`
const INCARNATION_ID = 'inc-1'
/** Answers `pty.inspectProcess` with a fresh host observation per request, held open on demand. */
function createInspectingOperations(): {
operations: ReturnType<typeof createSshPtyProviderRpcOperations>
request: ReturnType<typeof vi.fn>
resolvers: ((value: unknown) => void)[]
} {
const resolvers: ((value: unknown) => void)[] = []
const request = vi.fn(() => new Promise((resolve) => resolvers.push(resolve)))
return {
operations: createSshPtyProviderRpcOperations({
mux: { request } as never,
toRelayPtyId: () => RELAY_PTY_ID
}),
request,
resolvers
}
}
const flush = (): Promise<void> => new Promise((resolve) => setTimeout(resolve, 0))
describe('SSH pty.inspectProcess observation identity', () => {
it('gives each overlapping probe of one pane+incarnation its own host observation', async () => {
const { operations, request, resolvers } = createInspectingOperations()
const first = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID })
const second = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID })
await flush()
expect(request).toHaveBeenCalledTimes(2)
resolvers[0]({ foregroundProcess: 'claude', observationEpoch: 1 })
resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 2 })
// Each read settles on the observation minted for it, never a neighbour's.
expect(await first).toMatchObject({ observationEpoch: 1 })
expect(await second).toMatchObject({ observationEpoch: 2 })
})
it('does not share a failed probe with an overlapping one', async () => {
const { operations, request, resolvers } = createInspectingOperations()
const failing = operations.inspectProcess(APP_PTY_ID, { expectedIncarnationId: INCARNATION_ID })
const overlapping = operations.inspectProcess(APP_PTY_ID, {
expectedIncarnationId: INCARNATION_ID
})
await flush()
expect(request).toHaveBeenCalledTimes(2)
resolvers[0](Promise.reject(new Error('relay dropped the probe')))
resolvers[1]({ foregroundProcess: 'claude', observationEpoch: 1 })
await expect(failing).rejects.toThrow('relay dropped the probe')
expect(await overlapping).toMatchObject({ observationEpoch: 1 })
})
})
@@ -50,6 +50,10 @@ export function createSshPtyProviderRpcOperations({ mux, toRelayPtyId }: SshPtyP
const result = await mux.request('pty.getForegroundProcess', { id: toRelayPtyId(id) })
return result as string | null
},
// Do NOT in-flight coalesce this the way the sibling git reads are: the host mints one
// `observationEpoch` per request and the pane foreground reader commits it per read, so a
// shared reply reads as a stale replay and degrades a `live` identity read to `unverifiable`.
// Guarded by ssh-pty-inspect-observation-identity.test.ts; #17525 removes the poll.
inspectProcess: async (
id: string,
options?: { expectedIncarnationId?: string }
@@ -0,0 +1,62 @@
/**
* Telling a relay daemon's own service processes apart from the work it holds.
*
* The reap gate used to ask `pgrep -P <relay> | grep -c .` and demand zero. But the daemon
* forks service children of its own — `relay-ai-vault-service.js` is spawned lazily and then
* never exits — so that count is permanently non-zero on any relay that has touched the AI
* Vault, whether or not it holds a single PTY. A superseded, disconnected relay holding
* nothing therefore reported `retained-live-work` forever, its version directory stayed
* pinned against GC by its own live socket, and the population grew without bound (#13614).
*
* The asymmetry below is the whole safety argument, and it follows
* docs/reference/ssh-execution-boundary.md: *subtracting a child we can positively identify
* as relay infrastructure is sound; assuming anything about a child we cannot identify is
* not.* An argv that does not match, an argv `ps` would not print, and a host without
* `pgrep` all count against the relay and keep it unreapable. Losing sight of a child is
* never evidence that it holds nothing.
*/
import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts'
import { shellEscape } from './ssh-connection-utils'
/** Shell variable set to the daemon's direct-child count, or `unknown`. */
export const RELAY_CHILD_COUNT_VAR = 'kids'
/** Shell variable set to the count of children not identified as relay services, or `unknown`. */
export const RELAY_UNRECOGNIZED_CHILD_COUNT_VAR = 'unrecognized_kids'
/**
* `case` patterns matching a service child's argv. Suffix-anchored on purpose: both entries
* are forked with no script arguments, so the argv ends at the filename, and the leading `/`
* requires the absolute path the daemon forks rather than a bare mention of the name. A
* future arg would stop matching and the relay would go back to being retained — the safe
* direction to fail in.
*/
function serviceChildArgvPatterns(): string {
return RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.map(
(filename) => `*${shellEscape(`/${filename}`)}`
).join('|')
}
/**
* POSIX shell that censuses the direct children of `$pid`, setting `kids` and
* `unrecognized_kids`. Both stay `unknown` when the host cannot enumerate children at all.
*/
export function relayDaemonChildCensusShell(): string[] {
return [
`${RELAY_CHILD_COUNT_VAR}=unknown`,
`${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=unknown`,
'if command -v pgrep >/dev/null 2>&1; then',
` ${RELAY_CHILD_COUNT_VAR}=0`,
` ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=0`,
' for kid in $(pgrep -P "$pid" 2>/dev/null); do',
` ${RELAY_CHILD_COUNT_VAR}=$((${RELAY_CHILD_COUNT_VAR}+1))`,
' kid_args=$(ps -o args= -p "$kid" 2>/dev/null | tr -d "\\n")',
' case "$kid_args" in',
` ${serviceChildArgvPatterns()}) ;;`,
// An unreadable or unrecognised argv lands here, which is what keeps the relay retained.
` *) ${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}=$((${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}+1)) ;;`,
' esac',
' done',
'fi'
]
}
@@ -4,7 +4,7 @@
* generated scripts through /bin/sh against real unix sockets and real processes.
*/
import { execFile, spawn, type ChildProcess } from 'node:child_process'
import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'
import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { afterAll, afterEach, beforeAll, describe, expect, it } from 'vitest'
@@ -15,21 +15,32 @@ import {
type RelayEndpointIncumbent
} from './ssh-relay-endpoint-incumbent'
import { reapEmptyRelayHuskCommand } from './ssh-relay-endpoint-takeover'
import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts'
const posixOnly = process.platform === 'win32' ? describe.skip : describe
const FAKE_RELAY_SOURCE = `
const net = require('net')
const path = require('path')
const sock = process.argv[process.argv.indexOf('--sock-path') + 1]
function spawnChild(args) {
require('child_process').spawn(process.execPath, args, { stdio: 'ignore' })
}
if (process.argv.includes('--with-child')) {
require('child_process').spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000)'], {
stdio: 'ignore'
})
spawnChild(['-e', 'setTimeout(() => {}, 60000)'])
}
// Why forked the same way production does: the exclusion is argv-shaped, so a hand-written
// stand-in would test the test rather than the shell that runs on someone's host.
for (const name of process.argv.filter((arg) => arg.startsWith('--service-child='))) {
spawnChild([path.join(__dirname, name.slice('--service-child='.length))])
}
net.createServer(() => {}).listen(sock, () => process.stdout.write('READY\\n'))
process.on('SIGTERM', () => process.exit(0))
`
// Self-limiting: these are orphaned when the relay under test is reaped.
const IDLE_SERVICE_SOURCE = 'setTimeout(() => {}, 60000)\n'
function sh(script: string): Promise<string> {
return new Promise((resolve, reject) => {
execFile('/bin/sh', ['-c', script], { timeout: 20_000 }, (error, stdout) => {
@@ -43,14 +54,21 @@ function sh(script: string): Promise<string> {
}
let workDir: string
let pgreplessBinDir: string
let hasLsof = false
const running: ChildProcess[] = []
function startFakeRelay(sockPath: string, withChild = false): Promise<ChildProcess> {
function startFakeRelay(
sockPath: string,
options: { withChild?: boolean; serviceChildren?: readonly string[] } = {}
): Promise<ChildProcess> {
const args = [join(workDir, 'relay.js'), '--sock-path', sockPath]
if (withChild) {
if (options.withChild) {
args.push('--with-child')
}
for (const name of options.serviceChildren ?? []) {
args.push(`--service-child=${name}`)
}
const child = spawn(process.execPath, args, { stdio: ['ignore', 'pipe', 'ignore'] })
running.push(child)
return new Promise((resolve, reject) => {
@@ -68,9 +86,31 @@ async function probe(sockPath: string): Promise<RelayEndpointIncumbent> {
return parseRelayEndpointIncumbentProbe(sockPath, output)
}
/** The relay forks its children after it starts listening, so the probe can race them. */
async function waitForChildCount(
sockPath: string,
expected: number
): Promise<RelayEndpointIncumbent> {
let incumbent = await probe(sockPath)
for (let attempt = 0; attempt < 50 && incumbent.holders[0]?.childCount !== expected; attempt++) {
await new Promise((resolve) => setTimeout(resolve, 100))
incumbent = await probe(sockPath)
}
return incumbent
}
beforeAll(async () => {
workDir = mkdtempSync(join(tmpdir(), 'orca-relay-incumbent-'))
writeFileSync(join(workDir, 'relay.js'), FAKE_RELAY_SOURCE)
for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) {
writeFileSync(join(workDir, filename), IDLE_SERVICE_SOURCE)
}
writeFileSync(join(workDir, 'looks-like-relay-watcher.js'), IDLE_SERVICE_SOURCE)
pgreplessBinDir = join(workDir, 'pgrepless-bin')
mkdirSync(pgreplessBinDir)
for (const tool of ['ps', 'tr']) {
symlinkSync((await sh(`command -v ${tool}`)).trim(), join(pgreplessBinDir, tool))
}
hasLsof = await sh('command -v lsof >/dev/null 2>&1 && echo yes || echo no').then(
(out) => out.trim() === 'yes'
)
@@ -105,13 +145,51 @@ posixOnly('relay endpoint probe against a real socket', () => {
return
}
expect(incumbent.holders.map((holder) => holder.pid)).toEqual([relay.pid])
expect(incumbent.holders[0]).toMatchObject({ matchesRelayArgv: true, childCount: 0 })
expect(incumbent.holders[0]).toMatchObject({
matchesRelayArgv: true,
childCount: 0,
unrecognizedChildCount: 0
})
expect(isReapableRelayHusk(incumbent)).toBe(true)
})
it("counts the daemon's own service children but does not hold them against it", async () => {
const sockPath = join(workDir, 'services.sock')
await startFakeRelay(sockPath, { serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES })
const incumbent = await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length)
expect(incumbent.holders[0].childCount).toBe(RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length)
expect(incumbent.holders[0].unrecognizedChildCount).toBe(0)
expect(isReapableRelayHusk(incumbent)).toBe(true)
})
it('still retains a relay holding work alongside its service children', async () => {
const sockPath = join(workDir, 'services-and-work.sock')
await startFakeRelay(sockPath, {
withChild: true,
serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES
})
const incumbent = await waitForChildCount(
sockPath,
RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length + 1
)
expect(incumbent.holders[0].unrecognizedChildCount).toBe(1)
expect(isReapableRelayHusk(incumbent)).toBe(false)
})
it('does not excuse a child that merely mentions a service entry name', async () => {
const sockPath = join(workDir, 'lookalike.sock')
await startFakeRelay(sockPath, { serviceChildren: ['looks-like-relay-watcher.js'] })
const incumbent = await waitForChildCount(sockPath, 1)
expect(incumbent.holders[0].unrecognizedChildCount).toBe(1)
expect(isReapableRelayHusk(incumbent)).toBe(false)
})
it('refuses to call a relay with a live child an empty husk', async () => {
const sockPath = join(workDir, 'busy.sock')
await startFakeRelay(sockPath, true)
await startFakeRelay(sockPath, { withChild: true })
const incumbent = await probe(sockPath)
expect(incumbent.verdict).toBe('live')
@@ -150,12 +228,34 @@ posixOnly('empty relay husk reap against a real process', () => {
it('refuses to signal a relay that acquired a child after it was probed', async () => {
const sockPath = join(workDir, 'raced.sock')
const relay = await startFakeRelay(sockPath, true)
const relay = await startFakeRelay(sockPath, { withChild: true })
const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath))
expect(output.trim()).toBe('BUSY')
expect(relay.killed).toBe(false)
})
it('terminates a relay whose only children are its own service processes (#13614)', async () => {
const sockPath = join(workDir, 'service-husk.sock')
const relay = await startFakeRelay(sockPath, {
serviceChildren: RELAY_DAEMON_SERVICE_ENTRY_FILENAMES
})
await waitForChildCount(sockPath, RELAY_DAEMON_SERVICE_ENTRY_FILENAMES.length)
const output = await sh(reapEmptyRelayHuskCommand(relay.pid!, sockPath))
expect(output.trim()).toBe('GONE')
})
it('refuses to signal when the host cannot enumerate children at all', async () => {
const sockPath = join(workDir, 'no-pgrep.sock')
const relay = await startFakeRelay(sockPath)
// A PATH carrying every tool the script needs except `pgrep`: the census answers
// `unknown`, which must reach BUSY rather than the zero a missing tool would imply.
const output = await sh(
`PATH=${pgreplessBinDir}\n${reapEmptyRelayHuskCommand(relay.pid!, sockPath)}`
)
expect(output.trim()).toBe('BUSY')
expect(relay.killed).toBe(false)
})
it('refuses to signal a pid whose argv is not this relay at this socket', async () => {
const sockPath = join(workDir, 'mismatch.sock')
await startFakeRelay(sockPath)
@@ -32,17 +32,24 @@ describe('parseRelayEndpointIncumbentProbe', () => {
it('reports live when the socket accepted a connection', () => {
const incumbent = parseRelayEndpointIncumbentProbe(
SOCK,
probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=4242 yes 13'])
probeOutput([
'PRESENT=yes',
'LISTEN=accepted',
'HOLDERS_SOURCE=lsof',
'HOLDER=4242 yes 13 11'
])
)
expect(incumbent.verdict).toBe('live')
expect(incumbent.evidence).toBe('accepted-connection')
expect(incumbent.holders).toEqual([{ pid: 4242, matchesRelayArgv: true, childCount: 13 }])
expect(incumbent.holders).toEqual([
{ pid: 4242, matchesRelayArgv: true, childCount: 13, unrecognizedChildCount: 11 }
])
})
it('reports live when a process still holds an inode that refuses connections', () => {
const incumbent = parseRelayEndpointIncumbentProbe(
SOCK,
probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2'])
probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=91 yes 2 2'])
)
expect(incumbent.verdict).toBe('live')
expect(incumbent.evidence).toBe('holder-process')
@@ -85,7 +92,12 @@ describe('parseRelayEndpointIncumbentProbe', () => {
it('drops holder lines that do not carry a usable pid', () => {
const incumbent = parseRelayEndpointIncumbentProbe(
SOCK,
probeOutput(['PRESENT=yes', 'LISTEN=refused', 'HOLDERS_SOURCE=lsof', 'HOLDER=- no unknown'])
probeOutput([
'PRESENT=yes',
'LISTEN=refused',
'HOLDERS_SOURCE=lsof',
'HOLDER=- no unknown unknown'
])
)
expect(incumbent.holders).toEqual([])
expect(incumbent.verdict).toBe('exited')
@@ -94,9 +106,24 @@ describe('parseRelayEndpointIncumbentProbe', () => {
it('keeps an unreadable child count as null rather than zero', () => {
const [holder] = parseRelayEndpointIncumbentProbe(
SOCK,
probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes unknown'])
probeOutput([
'PRESENT=yes',
'LISTEN=accepted',
'HOLDERS_SOURCE=lsof',
'HOLDER=7 yes unknown unknown'
])
).holders
expect(holder.childCount).toBeNull()
expect(holder.unrecognizedChildCount).toBeNull()
})
it('keeps a holder line with no unrecognized-child field unreapable', () => {
const incumbent = parseRelayEndpointIncumbentProbe(
SOCK,
probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=7 yes 0'])
)
expect(incumbent.holders[0].unrecognizedChildCount).toBeNull()
expect(isReapableRelayHusk(incumbent)).toBe(false)
})
})
@@ -172,27 +199,36 @@ describe('mayLaunchOverRelayEndpoint', () => {
describe('isReapableRelayHusk', () => {
const husk = parseRelayEndpointIncumbentProbe(
SOCK,
probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0'])
probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 0 0'])
)
it('accepts a single proven relay holder with zero children', () => {
it('accepts a single proven relay holder with no unaccounted-for children', () => {
expect(isReapableRelayHusk(husk)).toBe(true)
})
it('refuses a relay that still holds children', () => {
it('accepts a relay whose only children are its own service processes (#13614)', () => {
const withServices = parseRelayEndpointIncumbentProbe(
SOCK,
probeOutput(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=500 yes 2 0'])
)
expect(withServices.holders[0].childCount).toBe(2)
expect(isReapableRelayHusk(withServices)).toBe(true)
})
it('refuses a relay that still holds children it could not account for', () => {
expect(
isReapableRelayHusk({
...husk,
holders: [{ pid: 500, matchesRelayArgv: true, childCount: 1 }]
holders: [{ pid: 500, matchesRelayArgv: true, childCount: 3, unrecognizedChildCount: 1 }]
})
).toBe(false)
})
it('refuses a holder whose child count could not be read', () => {
it('refuses a holder whose unrecognized-child count could not be read', () => {
expect(
isReapableRelayHusk({
...husk,
holders: [{ pid: 500, matchesRelayArgv: true, childCount: null }]
holders: [{ pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: null }]
})
).toBe(false)
})
@@ -201,7 +237,7 @@ describe('isReapableRelayHusk', () => {
expect(
isReapableRelayHusk({
...husk,
holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0 }]
holders: [{ pid: 500, matchesRelayArgv: false, childCount: 0, unrecognizedChildCount: 0 }]
})
).toBe(false)
})
@@ -211,8 +247,8 @@ describe('isReapableRelayHusk', () => {
isReapableRelayHusk({
...husk,
holders: [
{ pid: 500, matchesRelayArgv: true, childCount: 0 },
{ pid: 501, matchesRelayArgv: true, childCount: 0 }
{ pid: 500, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 },
{ pid: 501, matchesRelayArgv: true, childCount: 0, unrecognizedChildCount: 0 }
]
})
).toBe(false)
+35 -12
View File
@@ -20,6 +20,11 @@
*/
import type { SshConnection } from './ssh-connection'
import { shellEscape } from './ssh-connection-utils'
import {
RELAY_CHILD_COUNT_VAR,
RELAY_UNRECOGNIZED_CHILD_COUNT_VAR,
relayDaemonChildCensusShell
} from './relay-daemon-service-children'
import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers'
import { isWindowsRemoteHost, type RemoteHostPlatform } from './ssh-remote-platform'
@@ -38,6 +43,12 @@ export type RelayEndpointHolder = {
matchesRelayArgv: boolean
/** Direct children, or null when `pgrep` could not answer. Never guessed. */
childCount: number | null
/**
* Direct children *not* positively identified as the daemon's own service processes, or
* null when the host could not enumerate them. This — not `childCount` — is what says
* whether the relay holds anything; see relay-daemon-service-children.ts.
*/
unrecognizedChildCount: number | null
}
export type RelayEndpointIncumbent = {
@@ -95,11 +106,9 @@ export function relayEndpointIncumbentProbeCommand(nodePath: string, sockPath: s
' args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")',
' match=no',
' case "$args" in *relay.js*"$sock"*) match=yes ;; esac',
' kids=unknown',
' if command -v pgrep >/dev/null 2>&1; then',
' kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)',
' fi',
' printf \'HOLDER=%s %s %s\\n\' "$pid" "$match" "$kids"',
...relayDaemonChildCensusShell().map((line) => ` ${line}`),
' printf \'HOLDER=%s %s %s %s\\n\' "$pid" "$match" ' +
`"$${RELAY_CHILD_COUNT_VAR}" "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}"`,
' done',
'else',
" printf 'HOLDERS_SOURCE=unavailable\\n'",
@@ -159,19 +168,25 @@ export function parseRelayEndpointIncumbentProbe(
}
function parseHolder(value: string): RelayEndpointHolder | null {
const [rawPid, rawMatch, rawKids] = value.split(/\s+/)
const [rawPid, rawMatch, rawKids, rawUnrecognized] = value.split(/\s+/)
const pid = Number.parseInt(rawPid ?? '', 10)
if (!Number.isInteger(pid) || pid <= 0) {
return null
}
const childCount = Number.parseInt(rawKids ?? '', 10)
return {
pid,
matchesRelayArgv: rawMatch === 'yes',
childCount: Number.isInteger(childCount) && childCount >= 0 ? childCount : null
childCount: parseChildCount(rawKids),
unrecognizedChildCount: parseChildCount(rawUnrecognized)
}
}
/** `unknown`, a missing field, and anything unparseable are all "could not tell" — never 0. */
function parseChildCount(raw: string | undefined): number | null {
const count = Number.parseInt(raw ?? '', 10)
return Number.isInteger(count) && count >= 0 ? count : null
}
function unverifiableEndpoint(sockPath: string): RelayEndpointIncumbent {
return {
sockPath,
@@ -239,8 +254,12 @@ export function mayLaunchOverRelayEndpoint(incumbent: RelayEndpointIncumbent): b
/**
* A live relay that provably holds nothing: identity confirmed against its argv, exactly one
* holder, and zero children. Reaping it destroys no user work. Anything less is retained —
* killing the wrong pid on someone's remote host is the worst outcome available here.
* holder, and no child the host could not account for as one of the daemon's own service
* processes. Reaping it destroys no user work. Anything less is retained — killing the wrong
* pid on someone's remote host is the worst outcome available here.
*
* Why not `childCount === 0`: the daemon's AI Vault sidecar never exits once spawned, so that
* gate was unreachable for any relay that had ever served a vault request (#13614).
*/
export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean {
if (incumbent.verdict !== 'live' || !incumbent.holdersEnumerable) {
@@ -250,12 +269,16 @@ export function isReapableRelayHusk(incumbent: RelayEndpointIncumbent): boolean
return false
}
const [holder] = incumbent.holders
return holder.matchesRelayArgv && holder.childCount === 0
return holder.matchesRelayArgv && holder.unrecognizedChildCount === 0
}
export function describeRelayEndpointIncumbent(incumbent: RelayEndpointIncumbent): string {
const holders = incumbent.holders
.map((holder) => `${holder.pid}(children=${holder.childCount ?? 'unknown'})`)
.map(
(holder) =>
`${holder.pid}(children=${holder.childCount ?? 'unknown'},` +
`unrecognized=${holder.unrecognizedChildCount ?? 'unknown'})`
)
.join(',')
return (
`${incumbent.sockPath} verdict=${incumbent.verdict} evidence=${incumbent.evidence} ` +
@@ -14,6 +14,7 @@ import {
resolveRelayEndpointBeforeRelaunch
} from './ssh-relay-endpoint-takeover'
import { RelayVersionMismatchError } from './ssh-relay-version-mismatch-error'
import { RELAY_DAEMON_SERVICE_ENTRY_FILENAMES } from '../../shared/relay-artifacts'
import type { SshConnection } from './ssh-connection'
import { getRemoteHostPlatform } from './ssh-remote-platform'
@@ -42,7 +43,7 @@ beforeEach(() => {
describe('incumbent alive and refusing', () => {
it('refuses to rebind a live relay holding PTYs, and signals nothing', async () => {
execCommand.mockResolvedValueOnce(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11'])
)
await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError)
// The whole point of #8585: the incumbent's socket must survive so it is not orphaned.
@@ -52,9 +53,9 @@ describe('incumbent alive and refusing', () => {
it('names the incumbent pid and the Reset Relay escape hatch in the error', async () => {
execCommand.mockResolvedValue(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11'])
)
await expect(resolve()).rejects.toThrow(/3669803\(children=13\)/)
await expect(resolve()).rejects.toThrow(/3669803\(children=13,unrecognized=11\)/)
await expect(resolve()).rejects.toThrow(/Reset Relay/)
})
@@ -70,7 +71,7 @@ describe('incumbent alive and refusing', () => {
it('reaps a live relay only when it provably holds nothing, and confirms it is gone', async () => {
execCommand
.mockResolvedValueOnce(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0'])
)
.mockResolvedValueOnce('GONE\n')
await expect(resolve()).resolves.toMatchObject({ verdict: 'live' })
@@ -80,7 +81,7 @@ describe('incumbent alive and refusing', () => {
it('does not launch over an empty relay whose death could not be confirmed', async () => {
execCommand
.mockResolvedValueOnce(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0'])
)
.mockResolvedValueOnce('LIVE\n')
await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError)
@@ -89,7 +90,7 @@ describe('incumbent alive and refusing', () => {
it('does not launch over a relay the host refused to signal on its own re-check', async () => {
execCommand
.mockResolvedValueOnce(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0'])
)
.mockResolvedValueOnce('BUSY\n')
await expect(resolve()).rejects.toSatisfy(isRelayEndpointHeldError)
@@ -138,9 +139,19 @@ describe('reapEmptyRelayHuskCommand', () => {
})
it('aborts without signalling when the host cannot count children', () => {
expect(reapEmptyRelayHuskCommand(4242, SOCK)).toContain(
"command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }"
)
const command = reapEmptyRelayHuskCommand(4242, SOCK)
// The census leaves both counters at `unknown` without pgrep, and the gate demands "0".
expect(command).toContain('unrecognized_kids=unknown')
expect(command).toContain('command -v pgrep >/dev/null 2>&1')
expect(command).toContain('[ "$unrecognized_kids" = "0" ] ||')
})
it('subtracts only the daemon service children it can name from the reap gate', () => {
const command = reapEmptyRelayHuskCommand(4242, SOCK)
for (const filename of RELAY_DAEMON_SERVICE_ENTRY_FILENAMES) {
expect(command).toContain(`*'/${filename}'`)
}
expect(command).toContain('unrecognized_kids=$((unrecognized_kids+1))')
})
})
+10 -5
View File
@@ -2,13 +2,17 @@
* Deciding whether a relay socket path is ours to take, and acting on the answer.
*
* The only destructive action available here is a SIGTERM to a relay that has been proven —
* by argv, by socket-holder enumeration, and by a zero child count re-checked on the host
* immediately before the signal — to hold nothing at all. Everything else is left running.
* by argv, by socket-holder enumeration, and by a child census re-run on the host immediately
* before the signal — to hold nothing at all. Everything else is left running.
* Per docs/reference/ssh-execution-boundary.md, a relay we merely failed to reach is
* `unverifiable`, and `unverifiable` never authorizes a kill or a rebind.
*/
import type { SshConnection } from './ssh-connection'
import { shellEscape } from './ssh-connection-utils'
import {
RELAY_UNRECOGNIZED_CHILD_COUNT_VAR,
relayDaemonChildCensusShell
} from './relay-daemon-service-children'
import { execCommand, isUnconfirmedSshCommandTermination } from './ssh-relay-deploy-helpers'
import {
describeRelayEndpointIncumbent,
@@ -39,9 +43,10 @@ export function reapEmptyRelayHuskCommand(pid: number, sockPath: string): string
`sock=${shellEscape(sockPath)}`,
'args=$(ps -o args= -p "$pid" 2>/dev/null | tr "\\n" " ")',
'case "$args" in *relay.js*"$sock"*) ;; *) printf \'MISMATCH\\n\'; exit 0 ;; esac',
"command -v pgrep >/dev/null 2>&1 || { printf 'BUSY\\n'; exit 0; }",
'kids=$(pgrep -P "$pid" 2>/dev/null | grep -c .)',
'[ "$kids" = "0" ] || { printf \'BUSY\\n\'; exit 0; }',
// Why the same census as the probe: `unknown` (no pgrep) and any child this host could
// not account for as a relay service both land on BUSY, so nothing is signalled.
...relayDaemonChildCensusShell(),
`[ "$${RELAY_UNRECOGNIZED_CHILD_COUNT_VAR}" = "0" ] || { printf 'BUSY\\n'; exit 0; }`,
// SIGTERM only: the relay's own handler disposes and unlinks. SIGKILL would leave the
// socket inode behind and skip that shutdown path for no gain on an empty daemon.
'kill -TERM "$pid" 2>/dev/null || true',
@@ -67,7 +67,7 @@ describe('classifySupersededRelay', () => {
'PRESENT=yes',
'LISTEN=accepted',
'HOLDERS_SOURCE=lsof',
'HOLDER=3669803 yes 13'
'HOLDER=3669803 yes 13 11'
])
)
).toBe('retained-live-work')
@@ -76,7 +76,7 @@ describe('classifySupersededRelay', () => {
it('nominates only a proven empty relay for reaping', () => {
expect(
classifySupersededRelay(
incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0'])
incumbent(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0'])
)
).toBe('reap-candidate')
})
@@ -101,7 +101,7 @@ describe('sweepSupersededRelayEndpoints', () => {
execCommand
.mockResolvedValueOnce(`${OLD_SOCK}\n`)
.mockResolvedValueOnce(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=3669803 yes 13 11'])
)
const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP)
expect(findings).toHaveLength(1)
@@ -114,7 +114,7 @@ describe('sweepSupersededRelayEndpoints', () => {
execCommand
.mockResolvedValueOnce(`${OLD_SOCK}\n`)
.mockResolvedValueOnce(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0'])
)
.mockResolvedValueOnce('GONE\n')
const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP)
@@ -126,7 +126,7 @@ describe('sweepSupersededRelayEndpoints', () => {
execCommand
.mockResolvedValueOnce(`${OLD_SOCK}\n`)
.mockResolvedValueOnce(
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 0'])
probe(['PRESENT=yes', 'LISTEN=accepted', 'HOLDERS_SOURCE=lsof', 'HOLDER=80583 yes 2 0'])
)
.mockResolvedValueOnce('LIVE\n')
const findings = await sweepSupersededRelayEndpoints(CONN, HOST, SWEEP)
+15 -5
View File
@@ -11,14 +11,24 @@ export {
// to leave room for the `/c` wrapper sshd adds before cmd.exe counts the line.
const WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS = 8_000
export function powerShellCommand(script: string): string {
const inline = encodedPowerShellCommand(script)
/**
* `pwsh.exe` is PowerShell 7. It is not present on a stock Windows install, so it is only ever
* chosen after a probe — but where it exists it reads a redirected stdin correctly, which Windows
* PowerShell 5.1 does not (see `system-ssh-file-binary-transfer.ts`).
*/
export type WindowsPowerShellExecutable = 'powershell.exe' | 'pwsh.exe'
export function powerShellCommand(
script: string,
executable: WindowsPowerShellExecutable = 'powershell.exe'
): string {
const inline = encodedPowerShellCommand(script, executable)
if (inline.length <= WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) {
return inline
}
// Why: these scripts are repetitive enough that gzip beats the UTF-16LE tax by
// ~4x, which is the difference between a line cmd.exe runs and one it refuses.
const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script))
const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script), executable)
if (compressed.length > WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) {
throw new Error(
`Remote Windows command needs ${compressed.length} characters; Orca budgets ${WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS} for a line sshd hands to cmd.exe, which itself refuses more than ${CMD_EXE_COMMAND_LINE_MAX_CHARS}.`
@@ -27,8 +37,8 @@ export function powerShellCommand(script: string): string {
return compressed
}
function encodedPowerShellCommand(script: string): string {
return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}`
function encodedPowerShellCommand(script: string, executable: WindowsPowerShellExecutable): string {
return `${executable} -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}`
}
/** Orca-prefixed names so the payload can never shadow the bootstrap's own state. */
@@ -4,6 +4,10 @@ import { CMD_EXE_COMMAND_LINE_MAX_CHARS } from '../providers/windows-shell-args'
import { getRemoteHostPlatform } from './ssh-remote-platform'
import { tryStealInstallLockCommand } from './ssh-relay-install-lock-commands'
import { decodeRemotePowerShellScript, powerShellCommand } from './ssh-remote-powershell'
import {
makeWindowsPublishStagedFileCommand,
makeWindowsWriteFileCommand
} from './system-ssh-windows-file-write'
import {
cleanupOwnedRelayUploadStageCommand,
promoteOwnedRelayUploadStageCommand,
@@ -38,6 +42,17 @@ describe('Windows remote command line limit', () => {
[
'steal stale install lock',
tryStealInstallLockCommand(windows, 'C:\\Users\\orca\\.orca-remote\\relay', 1_200)
],
// F11 flagged these two as uncovered. They carry one path literal each, so they are the file
// commands whose length a caller can actually move.
['write file', makeWindowsWriteFileCommand('C:\\Users\\orca\\.orca-remote\\relay.js')],
[
'publish staged file',
makeWindowsPublishStagedFileCommand(
'C:\\Users\\orca\\.orca-remote\\relay.js.orca-partial-0123456789ab',
'C:\\Users\\orca\\.orca-remote\\relay.js',
'create'
)
]
])('keeps the %s command inside what sshd\u2019s cmd.exe accepts', (_name, command) => {
expect(command.length).toBeLessThanOrEqual(CMD_EXE_COMMAND_LINE_MAX_CHARS)
@@ -76,3 +91,32 @@ describe('Windows remote command line limit', () => {
)
})
})
/**
* F11 asked whether a pathological path could reach the budget, and what happens if it does.
* Measured: the inline encoding crosses 8000 at roughly 2500 high-entropy path characters — an
* order of magnitude past what Windows itself accepts — and the failure is a throw before any ssh
* is spawned, never a hang.
*/
describe('Windows file command budget headroom', () => {
it('absorbs a path far longer than Windows will accept', () => {
const deep = `C:\\Users\\orca\\${'segment\\'.repeat(30)}relay.js`
expect(deep.length).toBeGreaterThan(260)
expect(makeWindowsWriteFileCommand(deep).length).toBeLessThanOrEqual(
CMD_EXE_COMMAND_LINE_MAX_CHARS
)
})
it('throws rather than spawning a line cmd.exe would refuse', () => {
// Random segments so gzip cannot rescue it, which is the only way to reach the ceiling at all.
const incompressible = Array.from(
{ length: 400 },
(_unused, index) => `${index}-${Math.random().toString(36).slice(2)}`
).join('\\')
expect(() => makeWindowsWriteFileCommand(`C:\\${incompressible}\\f.bin`)).toThrow(
/Orca budgets 8000/
)
})
})
+50 -35
View File
@@ -709,9 +709,13 @@ describe('spawnSystemSsh', () => {
expect(args[standaloneControlIdx + 1]).toBe('none')
})
it('writes files to Windows system SSH targets with PowerShell stdin bytes', async () => {
const proc = createEventedProcess()
spawnMock.mockImplementation(() => closeOnceSpawned(proc))
it('sends Windows file writes over sftp, not through a remote PowerShell stdin', async () => {
const spawned: EventedProcess[] = []
spawnMock.mockImplementation(() => {
const proc = createEventedProcess()
spawned.push(proc)
return closeOnceSpawned(proc)
})
const hostPlatform = getRemoteHostPlatform('win32-x64')
const promise = writeFileViaSystemSsh(
@@ -722,16 +726,24 @@ describe('spawnSystemSsh', () => {
)
await expect(promise).resolves.toBeUndefined()
const args = spawnMock.mock.calls[0][1] as string[]
const remoteCommand = args.at(-1) ?? ''
expect(remoteCommand).toContain('powershell.exe')
expect(remoteCommand).not.toContain('/bin/sh')
expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('0.1.0', 'utf-8'))
// #16432, re-measured: Windows PowerShell 5.1 can lose a redirected stdin for good when a read
// finds it momentarily empty, so the bytes must not travel that way at all.
const batch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '')
expect(batch).toContain('put ')
expect(batch).toContain('/C:/Users/me/.orca-remote/relay/.version.orca-partial-')
const sftpArgs = spawnMock.mock.calls[0][1] as string[]
expect(sftpArgs).toContain('-b')
// The rename that publishes it reads the staged file, never a pipe.
const publish = (spawnMock.mock.calls[1][1] as string[]).at(-1) ?? ''
expect(publish).toContain('powershell.exe')
expect(decodePowerShellCommand(publish)).toContain(
'[System.IO.File]::Replace($staging, $path, [NullString]::Value)'
)
expect(publish).not.toContain('/bin/sh')
})
it('writes binary buffers to Windows system SSH targets with CreateNew mode', async () => {
const proc = createEventedProcess()
spawnMock.mockImplementation(() => closeOnceSpawned(proc))
it('enforces an exclusive Windows buffer write at the rename, where it is atomic', async () => {
spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess()))
const hostPlatform = getRemoteHostPlatform('win32-x64')
const promise = writeBufferViaSystemSsh(
@@ -742,12 +754,11 @@ describe('spawnSystemSsh', () => {
)
await expect(promise).resolves.toBeUndefined()
const args = spawnMock.mock.calls[0][1] as string[]
const remoteCommand = args.at(-1) ?? ''
expect(remoteCommand).toContain('powershell.exe')
expect(decodePowerShellCommand(remoteCommand)).toContain('CreateNew')
expect(remoteCommand).not.toContain('/bin/sh')
expect(proc.stdin.end).toHaveBeenCalledWith(Buffer.from('png'))
const publish = decodePowerShellCommand((spawnMock.mock.calls[1][1] as string[]).at(-1) ?? '')
// `File::Move` raising on an existing destination is what carries the exclusive contract now;
// a `CreateNew` on the staged file would only refuse a leftover of our own.
expect(publish).toContain('[System.IO.File]::Move($staging, $path)')
expect(publish).not.toContain('[System.IO.File]::Delete($path)')
})
it('downloads files from Windows system SSH targets with PowerShell stdout bytes', async () => {
@@ -779,8 +790,7 @@ describe('spawnSystemSsh', () => {
})
it('forces standalone SSH for Windows file writes when requested', async () => {
const proc = createEventedProcess()
spawnMock.mockImplementation(() => closeOnceSpawned(proc))
spawnMock.mockImplementation(() => closeOnceSpawned(createEventedProcess()))
const hostPlatform = getRemoteHostPlatform('win32-x64')
const promise = writeFileViaSystemSsh(
@@ -791,10 +801,14 @@ describe('spawnSystemSsh', () => {
)
await expect(promise).resolves.toBeUndefined()
const args = spawnMock.mock.calls[0][1] as string[]
const standaloneControlIdx = args.indexOf('-S')
const sftpArgs = spawnMock.mock.calls[0][1] as string[]
// sftp's own `-S` names a program to run, so the same request has to be spelled as an option.
expect(sftpArgs).not.toContain('-S')
expect(sftpArgs).toContain('ControlPath=none')
const publishArgs = spawnMock.mock.calls[1][1] as string[]
const standaloneControlIdx = publishArgs.indexOf('-S')
expect(standaloneControlIdx).toBeGreaterThan(-1)
expect(args[standaloneControlIdx + 1]).toBe('none')
expect(publishArgs[standaloneControlIdx + 1]).toBe('none')
})
it('uploads a Windows directory as a mkdir batch plus per-file writes, never one blob', async () => {
@@ -819,19 +833,20 @@ describe('spawnSystemSsh', () => {
rmSync(localDir, { recursive: true, force: true })
}
// #16432: directories first, then the file — but both over sftp now, so the only PowerShell
// left is the rename that publishes the staged file, which reads a file rather than a pipe.
const mkdirBatch = String(spawned[0]!.stdin.end.mock.calls[0]?.[0] ?? '')
expect(mkdirBatch).toBe('-mkdir "/C:/Users/me/.orca-remote/relay"\n')
const putBatch = String(spawned[1]!.stdin.end.mock.calls[0]?.[0] ?? '')
expect(putBatch).toContain('put ')
expect(putBatch).toContain('/C:/Users/me/.orca-remote/relay/relay.js.orca-partial-')
const commands = spawnMock.mock.calls.map((call) => (call[1] as string[]).at(-1) ?? '')
// #16432: directories first (metadata only), then the file bytes on their own stdin. One batch
// meant base64-ing the whole bundle into a single PowerShell string, which the remote never read.
expect(commands).toHaveLength(2)
expect(commands.every((command) => command.includes('powershell.exe'))).toBe(true)
expect(commands.every((command) => !command.includes('/bin/sh'))).toBe(true)
expect(commands.join('\n')).not.toContain('tar -xzf')
expect(JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string)).toEqual([
'C:/Users/me/.orca-remote/relay'
])
expect(Buffer.from(spawned[1].stdin.end.mock.calls[0]?.[0] as Buffer).toString('utf-8')).toBe(
'console.log("relay")'
)
// Nothing base64s the bundle into one PowerShell string any more, and nothing reads one.
expect(
commands.some((command) => decodePowerShellCommand(command).includes('OpenStandardInput'))
).toBe(false)
})
it('forces standalone SSH for Windows upload packages when requested', async () => {
@@ -855,9 +870,9 @@ describe('spawnSystemSsh', () => {
}
const args = spawnMock.mock.calls[0][1] as string[]
const standaloneControlIdx = args.indexOf('-S')
expect(standaloneControlIdx).toBeGreaterThan(-1)
expect(args[standaloneControlIdx + 1]).toBe('none')
// The first spawn is the sftp client, whose own `-S` names a program to run.
expect(args).not.toContain('-S')
expect(args).toContain('ControlPath=none')
})
it('throws when no system ssh is found', () => {
+74 -162
View File
@@ -1,5 +1,7 @@
import { constants, createWriteStream } from 'node:fs'
import { lstat, open } from 'node:fs/promises'
import { lstat, mkdtemp, open, rm, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import type { Writable } from 'node:stream'
import { pipeline } from 'node:stream/promises'
import type { SshTarget } from '../../shared/ssh-types'
@@ -16,6 +18,16 @@ import {
throwIfAborted,
waitForChannelClose
} from './system-ssh-operation-lifecycle'
import {
writeWindowsRemoteFile,
type WindowsWriteSource
} from './system-ssh-windows-write-strategy'
export {
WINDOWS_STDIN_WRITE_CHUNK_BYTES,
WINDOWS_STDIN_WRITE_TIMEOUT_MS
} from './system-ssh-windows-write-strategy'
export { WINDOWS_STAGED_WRITE_SUFFIX } from './system-ssh-windows-file-write'
type SystemSshOperationOptions = SystemSshBuildArgsOptions & {
signal?: AbortSignal
@@ -74,13 +86,16 @@ export async function writeBufferViaSystemSsh(
): Promise<void> {
throwIfAborted(options?.signal)
if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) {
await writeWindowsBytesViaSystemSsh(
await writeWindowsRemoteFile(
target,
remotePath,
contents.length,
(offset, maxBytes) =>
Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))),
options
{
totalBytes: contents.length,
readChunk: (offset, maxBytes) =>
Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))),
withLocalFile: (send) => withTemporaryLocalFile(contents, send)
},
options ?? {}
)
return
}
@@ -127,20 +142,19 @@ export async function uploadFileViaSystemSsh(
throwIfAborted(options?.signal)
if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) {
// #16432: a Windows host cannot take a whole file through one stdin, however the local side
// paces it — see WINDOWS_STDIN_WRITE_CHUNK_BYTES. This is the path that carries the large
// files, so it is the one that has to be chunked and bounded.
await writeWindowsBytesViaSystemSsh(
target,
remotePath,
openedStat.size,
async (offset, maxBytes) => {
// This is the path that carries the large files, so it is the one the transport choice is
// made for; see the #16432 note below.
const source: WindowsWriteSource = {
totalBytes: openedStat.size,
readChunk: async (offset, maxBytes) => {
const buffer = Buffer.allocUnsafe(Math.min(maxBytes, openedStat.size - offset))
const { bytesRead } = await handle.read(buffer, 0, buffer.length, offset)
return buffer.subarray(0, bytesRead)
},
options
)
// The verified local file is already exactly the payload, so sftp sends it as is.
withLocalFile: (send) => send(localPath)
}
await writeWindowsRemoteFile(target, remotePath, source, options ?? {})
return
}
@@ -173,158 +187,56 @@ export async function uploadFileViaSystemSsh(
}
/**
* #16432: Windows PowerShell 5.1 stops draining a redirected stdin over a non-pty ssh exec
* somewhere between 50KB and 1MB, depending on the host's `DefaultShell`, and it hangs rather than
* failing. The reporter measured that on both constructs he tried — `[Console]::In.ReadToEnd()` and
* `new IO.StreamReader([Console]::OpenStandardInput())`, the latter reading incrementally, which is
* why the limit cannot be attributed to materializing the payload. `Stream.CopyTo` reads the same
* `[Console]::OpenStandardInput()` object with the same incremental `Read` loop, so nothing in it
* escapes that limit either: no single write may exceed what one stdin is known to carry.
* #16432, re-measured: the constraint is not a size limit, and it is not cmd.exe's.
*
* 32KB is an order of magnitude under the low end of the measured range, and under 50KB, which the
* reporter measured succeeding against a stream reader on the worse of the two `DefaultShell`
* settings.
*/
export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024
/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */
export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000
/** Suffix for the path a multi-exec Windows write lands on before it is published by rename. */
export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial'
/**
* Splits one logical Windows write into stdin-sized execs.
* A read on Windows PowerShell 5.1's redirected-stdin handle over a non-pty ssh exec can die
* permanently when it finds the stream momentarily empty: no further bytes arrive, and no EOF ever
* does. It is probabilistic per such read — not a size threshold, and not certain on the first one.
* Measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2 with `DefaultShell = cmd.exe`, by
* replacing the copy loop with a counting reader:
*
* A write that needs more than one exec cannot land on the destination directly: a chunk failing
* mid-file would leave a truncated artifact under the real name with nothing marking it incomplete,
* and the retry would then meet its own leftovers — under `exclusive` the retry's `CreateNew` fails
* on them. Multi-exec creates therefore land on a staging path and are published by a rename, which
* is also where `exclusive` is enforced: once, at the destination, instead of smeared across the
* first chunk. A caller-requested append cannot be staged without reading the remote file back, so
* it keeps writing straight through, as its own protocol already implies.
* - a 1.5s gap before any byte, which forces the first read to find nothing -> 0 bytes, 6 of 6
* - one byte, a 1.5s gap, then 32767 more -> exactly 1 byte, then nothing
* - 32768, a 1.5s gap, then 32768 more -> exactly 32768, then nothing
* - a continuous 2MB -> 167936 / 270336 / 372736, then nothing
*
* Those three 2MB death points are one payload run three times under the same conditions, which is
* what rules out a threshold: a stream that died at a fixed point would not vary by 2x. Independently reproduced by
* a second harness, where one 1.9MB counted read survived 39 reads to completion and another died
* after 11 — same construct, same payload.
*
* A payload small enough to arrive in one burst usually presents only one read that can find the
* stream empty (the one waiting for EOF), which is why 32KB mostly works: it still failed 15 times
* in 120 with the host under load, and 1 in 40 on a quiet one. Neither rate is survivable across
* the 62 execs a 1.9MB file needs — even 2.5% compounds to roughly four uploads in five failing —
* and no chunk size helps, because the client does not control whether its bytes arrive together.
*
* The same host, same `DefaultShell`, same connection pattern contradicts every size-limit reading:
* `findstr` took 2,016,000 bytes through one exec's stdin, and PowerShell 7 took 2MB. So cmd.exe is
* not the ceiling and neither is ~50KB. Writes now go over sftp, which moves the whole payload
* without any remote process reading a pipe; see `system-ssh-windows-write-strategy.ts` for the
* fallback order.
*
* Successes are never partial. Across every run in both harnesses a failed write hung; not one
* produced a short file, so this defect cannot silently truncate an upload.
*/
async function writeWindowsBytesViaSystemSsh(
target: SshTarget,
remotePath: string,
totalBytes: number,
readChunk: (offset: number, maxBytes: number) => Promise<Buffer>,
options: SystemSshWriteBufferOptions
): Promise<void> {
throwIfAborted(options.signal)
const staged = !options.append && totalBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES
const writePath = staged ? `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` : remotePath
let offset = 0
// An empty write still has to run: it is what creates (or truncates) the file.
do {
const chunk = await readChunk(offset, WINDOWS_STDIN_WRITE_CHUNK_BYTES)
if (chunk.length === 0 && offset < totalBytes) {
throw new Error(`Source ran short during upload of ${remotePath}`)
}
await writeWindowsChunkViaSystemSsh(
target,
writePath,
chunk,
{
...options,
append: staged ? offset > 0 : options.append === true || offset > 0,
exclusive: staged ? false : options.exclusive === true && offset === 0
},
offset
)
offset += chunk.length
} while (offset < totalBytes)
if (staged) {
await publishWindowsStagedWrite(target, writePath, remotePath, options)
/** A staged write is materialized locally first when the source is a buffer rather than a file. */
async function withTemporaryLocalFile<T>(
contents: Buffer,
send: (localPath: string) => Promise<T>
): Promise<T> {
const directory = await mkdtemp(join(tmpdir(), 'orca-win-upload-'))
const localPath = join(directory, 'payload.bin')
try {
// 0600: the payload can be repository content, and tmpdir is shared on every platform.
await writeFile(localPath, contents, { mode: 0o600 })
return await send(localPath)
} finally {
await rm(directory, { recursive: true, force: true }).catch(() => {})
}
}
async function writeWindowsChunkViaSystemSsh(
target: SshTarget,
remotePath: string,
chunk: Buffer,
options: SystemSshWriteBufferOptions,
offset: number
): Promise<void> {
throwIfAborted(options.signal)
const channel = spawnSystemSshCommand(target, makeWindowsWriteFileCommand(remotePath, options), {
wrapCommand: false,
...getSystemSshBuildArgsFromOperationOptions(options)
})
const closePromise = awaitWithSystemSshAbort(
options.signal,
() => channel.close(),
waitForChannelClose(
channel,
`write ${remotePath} at offset ${offset}`,
WINDOWS_STDIN_WRITE_TIMEOUT_MS
)
)
if (!options.signal?.aborted) {
channel.stdin.end(chunk)
}
await closePromise
}
async function publishWindowsStagedWrite(
target: SshTarget,
stagingPath: string,
remotePath: string,
options: SystemSshWriteBufferOptions
): Promise<void> {
throwIfAborted(options.signal)
const channel = spawnSystemSshCommand(
target,
makeWindowsPublishStagedFileCommand(stagingPath, remotePath, options.exclusive === true),
{ wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) }
)
const closePromise = awaitWithSystemSshAbort(
options.signal,
() => channel.close(),
waitForChannelClose(channel, `publish ${remotePath}`, WINDOWS_STDIN_WRITE_TIMEOUT_MS)
)
if (!options.signal?.aborted) {
channel.stdin.end()
}
await closePromise
}
function makeWindowsWriteFileCommand(
remotePath: string,
options?: { append?: boolean; exclusive?: boolean }
): string {
const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create'
return powerShellCommand(
[
'$ErrorActionPreference = "Stop"',
`$path = ${powerShellLiteral(remotePath)}`,
'$parent = [System.IO.Path]::GetDirectoryName($path)',
'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }',
'$inputStream = [Console]::OpenStandardInput()',
`$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`,
'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }'
].join('; ')
)
}
// `File::Move` throws when the destination exists, which is exactly the exclusive contract; the
// non-exclusive caller asked to replace, so it deletes first (a no-op on an absent path).
function makeWindowsPublishStagedFileCommand(
stagingPath: string,
remotePath: string,
exclusive: boolean
): string {
return powerShellCommand(
[
'$ErrorActionPreference = "Stop"',
`$staging = ${powerShellLiteral(stagingPath)}`,
`$path = ${powerShellLiteral(remotePath)}`,
...(exclusive ? [] : ['[System.IO.File]::Delete($path)']),
'[System.IO.File]::Move($staging, $path)'
].join('; ')
)
}
function makePosixWriteFileCommand(
remotePath: string,
options?: { append?: boolean; exclusive?: boolean }
+61 -19
View File
@@ -27,6 +27,12 @@ import {
WINDOWS_STDIN_WRITE_TIMEOUT_MS,
writeBufferViaSystemSsh
} from './system-ssh-file-binary-transfer'
import {
isSftpPathUnsupportedError,
isSftpUnavailableError,
makeDirectoriesViaSftp
} from './system-ssh-sftp-transfer'
import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities'
type SystemSshOperationOptions = SystemSshBuildArgsOptions & {
signal?: AbortSignal
@@ -161,9 +167,14 @@ async function collectWindowsUploadPlan(
return plan
}
// Why the JSON envelope survives here: a path list is metadata, so this payload stays in the
// hundreds of bytes even for a deep tree. Batched anyway, so a pathological tree cannot walk back
// into the same stdin size that wedges PowerShell.
/**
* Creates the upload's directories, preferring sftp's own `mkdir`.
*
* The PowerShell fallback keeps the JSON envelope, batched under one stdin's worth: a path list is
* metadata, so it stays in the hundreds of bytes even for a deep tree. It is still a redirected
* stdin read though, so on Windows PowerShell 5.1 it carries the same defect as any other — which
* is why sftp is tried first even for a payload this small.
*/
async function createWindowsUploadDirectories(
target: SshTarget,
directories: readonly string[],
@@ -175,23 +186,27 @@ async function createWindowsUploadDirectories(
if (batch.length === 0) {
return
}
const pending = batch
const payload = JSON.stringify(batch)
batch = []
batchBytes = 0
throwIfAborted(options.signal)
const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), {
wrapCommand: false,
...getSystemSshBuildArgsFromOperationOptions(options)
})
const closePromise = awaitWithSystemSshAbort(
options.signal,
() => channel.close(),
waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS)
await getWindowsRemoteWriteCapabilities(target).runWithFallback(
'sftp-subsystem',
async () => {
try {
await makeDirectoriesViaSftp(target, pending, options)
} catch (error) {
// A directory sftp cannot address is this batch's problem, not the host's verdict.
if (!isSftpPathUnsupportedError(error)) {
throw error
}
await createWindowsUploadDirectoriesViaPowerShell(target, payload, options)
}
},
() => createWindowsUploadDirectoriesViaPowerShell(target, payload, options),
isSftpUnavailableError
)
if (!options.signal?.aborted) {
channel.stdin.end(payload)
}
await closePromise
}
for (const directory of directories) {
const entryBytes = Buffer.byteLength(directory) + 4
@@ -204,17 +219,44 @@ async function createWindowsUploadDirectories(
await flush()
}
async function createWindowsUploadDirectoriesViaPowerShell(
target: SshTarget,
payload: string,
options: SystemSshOperationOptions
): Promise<void> {
const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), {
wrapCommand: false,
...getSystemSshBuildArgsFromOperationOptions(options)
})
const closePromise = awaitWithSystemSshAbort(
options.signal,
() => channel.close(),
waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS)
)
if (!options.signal?.aborted) {
channel.stdin.end(payload)
}
await closePromise
}
function makeWindowsCreateDirectoriesCommand(): string {
return powerShellCommand(
[
'$ErrorActionPreference = "Stop"',
// The reporter measured this reader surviving 50KB where `[Console]::In` wedged at the same
// size (#16432); the batch above stays under that.
// Reached only where the host has no sftp subsystem. Windows PowerShell 5.1 can lose a
// redirected stdin for good when a read finds it empty (#16432); a batch this small usually
// arrives in one piece, and "usually" is exactly why sftp is preferred.
'$reader = New-Object System.IO.StreamReader([Console]::OpenStandardInput())',
'try { $json = $reader.ReadToEnd() } finally { $reader.Dispose() }',
'if ([string]::IsNullOrWhiteSpace($json)) { return }',
'foreach ($path in @($json | ConvertFrom-Json)) {',
' $null = [System.IO.Directory]::CreateDirectory([string]$path)',
// `[string[]]`, not `@(...)`: ConvertFrom-Json emits the parsed array as a single pipeline
// object, so `@(...)` wraps it in *another* array and the loop variable binds to the whole
// thing. `[string]` of that is the paths joined by spaces, which CreateDirectory rejects with
// "The given path's format is not supported". It only ever worked for a one-element batch,
// where stringifying a single-element array happens to yield the element. Measured on
// WindowsPowerShell 5.1.26100 against a three-directory tree.
'foreach ($path in [string[]]($json | ConvertFrom-Json)) {',
' $null = [System.IO.Directory]::CreateDirectory($path)',
'}'
].join('; ')
)
+141
View File
@@ -0,0 +1,141 @@
/**
* `buildSshArgs` is shared with the sftp client, and three of its flags mean something else there.
* Every case below is a silent wrong-target rather than an error if the translation is skipped,
* which is why the fallback is "refuse and use another transport", never "pass it through".
*/
import { describe, expect, it } from 'vitest'
import {
SftpArgTranslationError,
translateSshArgsToSftpArgs,
withSftpKeepalive
} from './system-ssh-sftp-args'
describe('translateSshArgsToSftpArgs', () => {
it('sends the port as an option, since sftp -p preserves mtimes instead', () => {
const args = translateSshArgsToSftpArgs(['-p', '2222', '--', 'dev@win.example'])
expect(args).toEqual(['-o', 'Port=2222', '--', 'dev@win.example'])
})
it('sends the login name as an option, since sftp has no -l', () => {
// `buildSshArgs` emits `-l` for a config alias no Host block claims. Throwing here would send
// exactly those hosts to the transport this PR exists to stop using, silently.
const args = translateSshArgsToSftpArgs(['-l', 'neil', '--', 'awin'])
expect(args).toEqual(['-o', 'User=neil', '--', 'awin'])
})
it('translates the whole unclaimed-alias shape buildSshArgs emits', () => {
const args = translateSshArgsToSftpArgs([
'-o',
'BatchMode=no',
'-T',
'-S',
'none',
'-o',
'Hostname=192.168.0.186',
'-p',
'2222',
'-l',
'neil',
'--',
'awin'
])
expect(args).toEqual([
'-o',
'BatchMode=no',
'-o',
'ControlPath=none',
'-o',
'Hostname=192.168.0.186',
'-o',
'Port=2222',
'-o',
'User=neil',
'--',
'awin'
])
})
it('spells ControlPath=none out, since sftp -S names a program to run', () => {
// `sftp -S none` would try to exec a binary called `none`.
const args = translateSshArgsToSftpArgs(['-S', 'none', '--', 'dev@win.example'])
expect(args).toEqual(['-o', 'ControlPath=none', '--', 'dev@win.example'])
})
it('refuses any other -S, which would hand sftp an ssh binary Orca did not choose', () => {
expect(() => translateSshArgsToSftpArgs(['-S', '/tmp/ctl.sock'])).toThrow(
SftpArgTranslationError
)
})
it('drops -T, which sftp does not have', () => {
expect(translateSshArgsToSftpArgs(['-T', '--', 'host'])).toEqual(['--', 'host'])
})
it('passes through the flags both clients spell the same way', () => {
const args = translateSshArgsToSftpArgs([
'-F',
'/tmp/config',
'-o',
'BatchMode=yes',
'-i',
'/tmp/key',
'-J',
'jump.example',
'--',
'dev@win.example'
])
expect(args).toEqual([
'-F',
'/tmp/config',
'-o',
'BatchMode=yes',
'-i',
'/tmp/key',
'-J',
'jump.example',
'--',
'dev@win.example'
])
})
it('takes everything after -- as the destination without reinterpreting it', () => {
// A host literally named `-p` is not a flag once `--` has been seen.
expect(translateSshArgsToSftpArgs(['--', '-p'])).toEqual(['--', '-p'])
})
it('refuses an unknown flag rather than guessing what sftp would do with it', () => {
// The point of the throw: a flag added to buildSshArgs later must degrade to another
// transport, not reach sftp carrying a different meaning.
expect(() => translateSshArgsToSftpArgs(['-A', '--', 'host'])).toThrow(SftpArgTranslationError)
})
it('refuses a value flag with no value', () => {
expect(() => translateSshArgsToSftpArgs(['-o'])).toThrow(SftpArgTranslationError)
})
})
describe('withSftpKeepalive', () => {
it('asks OpenSSH to notice a dead peer, since the transfer itself has no wall-clock bound', () => {
expect(withSftpKeepalive(['--', 'host'])).toEqual([
'-o',
'ServerAliveInterval=15',
'-o',
'ServerAliveCountMax=3',
'--',
'host'
])
})
it('leaves a caller-stated keepalive policy alone', () => {
const args = withSftpKeepalive(['-o', 'ServerAliveInterval=60', '--', 'host'])
expect(args.filter((arg) => arg.startsWith('ServerAliveInterval'))).toEqual([
'ServerAliveInterval=60'
])
})
})
+95
View File
@@ -0,0 +1,95 @@
/**
* Rewrites `buildSshArgs` output for the sftp(1) client.
*
* Three flags ssh and sftp share spell different things: sftp's `-p` is "preserve mtime", its `-S`
* names the ssh binary to run, and it has no `-T` at all. Passing ssh's list through unchanged
* would silently connect to the wrong port and try to exec a program called `none`.
*
* Anything this table does not recognize throws. A flag added to `buildSshArgs` later must degrade
* to the non-sftp transfer path, never reach sftp carrying a different meaning.
*/
/** `buildSshArgs` emitted a flag with no sftp equivalent; the caller should use another transport. */
export class SftpArgTranslationError extends Error {
constructor(flag: string) {
super(`No sftp equivalent for system ssh argument ${JSON.stringify(flag)}`)
this.name = 'SftpArgTranslationError'
}
}
/** Flags whose spelling and meaning are identical in both clients. */
const PASSTHROUGH_VALUE_FLAGS = new Set(['-F', '-o', '-i', '-J'])
export function translateSshArgsToSftpArgs(sshArgs: readonly string[]): string[] {
const sftpArgs: string[] = []
let index = 0
while (index < sshArgs.length) {
const flag = sshArgs[index]!
if (flag === '--') {
// Everything after `--` is the destination, which both clients spell the same way.
sftpArgs.push(...sshArgs.slice(index))
return sftpArgs
}
const value = sshArgs[index + 1]
if (PASSTHROUGH_VALUE_FLAGS.has(flag)) {
if (value === undefined) {
throw new SftpArgTranslationError(flag)
}
sftpArgs.push(flag, value)
index += 2
continue
}
if (flag === '-T') {
// sftp never allocates a tty, so ssh's "no tty" request has nothing to translate to.
index += 1
continue
}
if (flag === '-p') {
if (value === undefined) {
throw new SftpArgTranslationError(flag)
}
sftpArgs.push('-o', `Port=${value}`)
index += 2
continue
}
if (flag === '-l') {
// sftp has no `-l`; the login name is an option there. `buildSshArgs` emits this for an
// unclaimed config alias, so throwing would route those hosts down the defective path and
// then cache the refusal against them for half an hour.
if (value === undefined) {
throw new SftpArgTranslationError(flag)
}
sftpArgs.push('-o', `User=${value}`)
index += 2
continue
}
if (flag === '-S') {
// ssh's `-S none` is ControlPath=none; sftp's `-S` would run a binary called `none`.
if (value !== 'none') {
throw new SftpArgTranslationError(flag)
}
sftpArgs.push('-o', 'ControlPath=none')
index += 2
continue
}
throw new SftpArgTranslationError(flag)
}
return sftpArgs
}
/**
* A transfer that stalls mid-stream has no per-write bound to catch it, so ask OpenSSH to notice a
* dead peer itself. Only added when the caller has not already stated a keepalive policy.
*/
export function withSftpKeepalive(sftpArgs: readonly string[]): string[] {
const hasOption = (name: string): boolean =>
sftpArgs.some((arg, position) => sftpArgs[position - 1] === '-o' && arg.startsWith(`${name}=`))
const keepalive: string[] = []
if (!hasOption('ServerAliveInterval')) {
keepalive.push('-o', 'ServerAliveInterval=15')
}
if (!hasOption('ServerAliveCountMax')) {
keepalive.push('-o', 'ServerAliveCountMax=3')
}
return [...keepalive, ...sftpArgs]
}
+59
View File
@@ -0,0 +1,59 @@
/**
* Both functions here guard against the same measured failure: sftp's batch lexer treats `\` as an
* escape, so a Windows path handed over raw is silently mis-targeted *and the client still exits
* 0*. On Windows 11 / OpenSSH 10.0p2, `put src C:\Users\neil\qt\a.bin` created a file literally
* named `C` in the start directory and reported success.
*/
import { describe, expect, it } from 'vitest'
import {
quoteSftpBatchArgument,
toSftpRemotePath,
UnsupportedSftpPathError
} from './system-ssh-sftp-path'
describe('toSftpRemotePath', () => {
it('roots a drive path under /, which is the namespace the Windows sftp-server exposes', () => {
// `pwd` in that session reports `/C:/Users/dev`.
expect(toSftpRemotePath('C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin')
})
it('accepts a path already in that namespace unchanged', () => {
expect(toSftpRemotePath('/C:/Users/dev/f.bin')).toBe('/C:/Users/dev/f.bin')
})
it('converts the separators Orca stores paths with', () => {
expect(toSftpRemotePath('C:\\Users\\dev\\f.bin')).toBe('/C:/Users/dev/f.bin')
})
it('declines a UNC path rather than guessing where it lands', () => {
// A guess here writes real bytes to the wrong place; declining falls back to another transport.
expect(() => toSftpRemotePath('//server/share/f.bin')).toThrow(UnsupportedSftpPathError)
})
it('declines a relative path, which would resolve against the session start directory', () => {
expect(() => toSftpRemotePath('Users/dev/f.bin')).toThrow(UnsupportedSftpPathError)
})
})
describe('quoteSftpBatchArgument', () => {
it('escapes the backslashes in a Windows client local path', () => {
// Unescaped, sftp reads this as C:srcf.bin and fails to find the source.
expect(quoteSftpBatchArgument('C:\\src\\f.bin')).toBe('"C:\\\\src\\\\f.bin"')
})
it('keeps a path with spaces as one argument', () => {
expect(quoteSftpBatchArgument('/tmp/two words.bin')).toBe('"/tmp/two words.bin"')
})
it('escapes an embedded quote, which would otherwise end the argument early', () => {
expect(quoteSftpBatchArgument('/tmp/dq".bin')).toBe('"/tmp/dq\\".bin"')
})
it('refuses a line break, which would split one batch command into two', () => {
expect(() => quoteSftpBatchArgument('/tmp/a\nrm -rf b')).toThrow(UnsupportedSftpPathError)
})
it('refuses a NUL, which truncates the argument', () => {
expect(() => quoteSftpBatchArgument('/tmp/a\0b')).toThrow(UnsupportedSftpPathError)
})
})
+46
View File
@@ -0,0 +1,46 @@
import { normalizeWindowsRemotePath } from './ssh-remote-platform'
/**
* A path this transfer cannot express to sftp. Callers treat it as "use another transport", never
* as a transfer failure.
*/
export class UnsupportedSftpPathError extends Error {
constructor(path: string) {
super(`Path cannot be addressed over sftp: ${JSON.stringify(path)}`)
this.name = 'UnsupportedSftpPathError'
}
}
/**
* Converts a Windows remote path to the namespace OpenSSH's Windows sftp-server exposes, which
* roots every drive under `/`: `C:/Users/dev/f` is `/C:/Users/dev/f`, and `pwd` there reports
* `/C:/Users/dev`.
*/
export function toSftpRemotePath(remotePath: string): string {
const normalized = normalizeWindowsRemotePath(remotePath)
if (/^\/[a-zA-Z]:\//.test(normalized)) {
return normalized
}
if (/^[a-zA-Z]:\//.test(normalized)) {
return `/${normalized}`
}
// UNC (`//server/share`) and relative paths have no settled mapping in this namespace, and a
// guess here writes real bytes to the wrong place. Decline instead.
throw new UnsupportedSftpPathError(remotePath)
}
/**
* Quotes one argument of an sftp batch line.
*
* Escaping is load-bearing, not cosmetic: sftp's batch lexer treats `\` as an escape even inside
* double quotes, so an unescaped Windows local path `C:\src\f.bin` is read as `C:srcf.bin`, and an
* unescaped destination `C:\Users\dev\f.bin` writes a file literally named `C` in the start
* directory — while sftp still exits 0. Both measured on Windows 11 / OpenSSH 10.0p2.
*/
export function quoteSftpBatchArgument(value: string): string {
if (/[\n\r\0]/.test(value)) {
// A line break would split one batch command into two; NUL truncates the argument.
throw new UnsupportedSftpPathError(value)
}
return `"${value.replace(/([\\"])/g, '\\$1')}"`
}
+191
View File
@@ -0,0 +1,191 @@
import { accessSync, constants, existsSync, statSync } from 'node:fs'
import { posix, win32 } from 'node:path'
import type { SshTarget } from '../../shared/ssh-types'
import { buildSshArgs, type SystemSshBuildArgsOptions } from './system-ssh-args'
import { findSystemSsh } from './system-ssh-binary'
import {
SftpArgTranslationError,
translateSshArgsToSftpArgs,
withSftpKeepalive
} from './system-ssh-sftp-args'
import {
quoteSftpBatchArgument,
toSftpRemotePath,
UnsupportedSftpPathError
} from './system-ssh-sftp-path'
import { throwIfAborted } from './system-ssh-operation-lifecycle'
import { runProcess } from '../../shared/child-process/run-process'
/** The host answered, but not with an sftp subsystem. The caller must fall back, not fail. */
export class SftpSubsystemUnavailableError extends Error {
constructor(detail: string) {
super(`Remote host has no usable sftp subsystem: ${detail}`)
this.name = 'SftpSubsystemUnavailableError'
}
}
/**
* True for the errors that mean "this host cannot serve sftp at all".
*
* Host-scoped, and therefore the only errors safe to remember: a capability cache keyed by host
* turns anything it accepts into a verdict about every later write to that host. Deliberately
* narrow — a permission denial or a missing directory is a real failure that must surface, not a
* reason to retry the whole upload down a slower path.
*/
export function isSftpUnavailableError(error: unknown): boolean {
return error instanceof SftpSubsystemUnavailableError || error instanceof SftpArgTranslationError
}
/**
* True when *this path* cannot be spelled for sftp, which says nothing about the host.
*
* Kept apart from the host verdict on purpose. A UNC destination, or a local file whose name
* contains a newline, is a property of one operation; caching it would degrade every subsequent
* write to that host for the cache's whole retry window on the strength of one odd filename.
*/
export function isSftpPathUnsupportedError(error: unknown): boolean {
return error instanceof UnsupportedSftpPathError
}
/** Neither kind of refusal moves a byte, so a staged file cannot exist to sweep. */
export function isSftpRefusalBeforeStaging(error: unknown): boolean {
return isSftpUnavailableError(error) || isSftpPathUnsupportedError(error)
}
function systemSftpCandidates(sshPath: string | null, platform: NodeJS.Platform): string[] {
const pathApi = platform === 'win32' ? win32 : posix
const executable = platform === 'win32' ? 'sftp.exe' : 'sftp'
const candidates: string[] = []
// Why the ssh binary's own directory first: a host with two OpenSSH installs must pair the sftp
// client with the ssh that `buildSshArgs` was built for, not whichever one PATH happens to reach.
if (sshPath) {
candidates.push(pathApi.join(pathApi.dirname(sshPath), executable))
}
if (platform === 'win32') {
const systemRoot = process.env.SystemRoot || process.env.WINDIR
if (systemRoot) {
candidates.push(win32.join(systemRoot, 'System32', 'OpenSSH', executable))
}
} else {
candidates.push('/usr/bin/sftp', '/usr/local/bin/sftp', '/opt/homebrew/bin/sftp')
}
return candidates
}
/** Locate the sftp client paired with the system ssh binary. Returns null when there is none. */
export function findSystemSftp(): string | null {
if (process.env.ORCA_SYSTEM_SFTP_PATH) {
return process.env.ORCA_SYSTEM_SFTP_PATH
}
const sshPath = findSystemSsh()
for (const candidate of systemSftpCandidates(sshPath, process.platform)) {
try {
if (!statSync(candidate).isFile()) {
continue
}
if (process.platform !== 'win32') {
accessSync(candidate, constants.X_OK)
}
return candidate
} catch {
continue
}
}
return findSftpOnPath()
}
function findSftpOnPath(): string | null {
const pathValue = process.env.PATH
if (!pathValue) {
return null
}
const pathApi = process.platform === 'win32' ? win32 : posix
const executable = process.platform === 'win32' ? 'sftp.exe' : 'sftp'
for (const entry of pathValue.split(pathApi.delimiter)) {
const directory = entry.trim().replace(/^"|"$/g, '')
if (!directory) {
continue
}
const candidate = pathApi.join(directory, executable)
if (existsSync(candidate)) {
return candidate
}
}
return null
}
/**
* OpenSSH prints this when the server refuses the subsystem — a host with `Subsystem sftp`
* commented out, or an internal-sftp block that does not apply to this user.
*/
const SUBSYSTEM_REFUSED_PATTERN = /subsystem request failed|no such file or directory.*sftp-server/i
export type SftpBatchOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal }
/**
* Runs one sftp batch script.
*
* The script goes to the *local* sftp client's stdin, which is the point: no remote process ever
* reads a redirected stdin, so none of this rides the Windows PowerShell stdin defect.
*/
export async function runSftpBatch(
target: SshTarget,
commands: readonly string[],
options?: SftpBatchOptions
): Promise<void> {
throwIfAborted(options?.signal)
const sftpPath = findSystemSftp()
if (!sftpPath) {
throw new SftpSubsystemUnavailableError('no sftp client binary found alongside ssh')
}
const args = withSftpKeepalive(translateSshArgsToSftpArgs(buildSshArgs(target, options)))
let result
try {
result = await runProcess({
program: sftpPath,
args: ['-b', '-', ...args],
// `-b -` takes the script on stdin, and that stdin is the *local* client's — no remote
// process reads a pipe anywhere in this transfer, which is the whole point of preferring it.
input: `${commands.join('\n')}\n`,
// Why no timeout: a large upload is legitimately slow, and a wall-clock cap would fail a
// healthy transfer on a slow link. A dead peer is caught by the ServerAlive options instead.
timeoutMs: null,
signal: options?.signal
})
} catch (error) {
// A client that will not start is "this host cannot do sftp" from the caller's side, not a
// transfer failure: the payload never left. Falling back is the only useful answer.
throw new SftpSubsystemUnavailableError(
`sftp client at ${sftpPath} could not be started: ${error instanceof Error ? error.message : String(error)}`
)
}
if (result.code === 0) {
return
}
throwIfAborted(options?.signal)
const detail = result.stderr.trim()
if (SUBSYSTEM_REFUSED_PATTERN.test(detail)) {
throw new SftpSubsystemUnavailableError(detail)
}
throw new Error(`sftp batch failed (exit ${result.code}): ${detail}`)
}
/**
* Creates remote directories, parents first.
*
* `-mkdir` keeps sftp going when a directory is already there; batch mode otherwise aborts the
* whole script on the first non-zero status, which for an idempotent tree walk is not a failure.
*/
export function makeDirectoriesViaSftp(
target: SshTarget,
remoteDirectories: readonly string[],
options?: SftpBatchOptions
): Promise<void> {
const commands = remoteDirectories.map(
(directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}`
)
if (commands.length === 0) {
return Promise.resolve()
}
return runSftpBatch(target, commands, options)
}
@@ -0,0 +1,138 @@
import { randomBytes } from 'node:crypto'
import { powerShellCommand, powerShellLiteral } from './ssh-remote-powershell'
import { normalizeWindowsRemotePath } from './ssh-remote-platform'
/**
* Suffix marking the path a Windows write lands on before it is published by rename.
*
* The random tail is the fix for a measured harm, not decoration. A write that loses contact with
* the host leaves a remote process that may still hold the staging file open exclusively, and
* `docs/reference/ssh-execution-boundary.md` is explicit that losing contact is not evidence that
* process died — so the retry must not reuse the name it may still own. A fresh name per attempt
* means a retry never meets its predecessor's lock; the abandoned file is cleaned up best-effort
* and never treated as proof of anything.
*/
export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial'
export function makeWindowsStagingPath(remotePath: string): string {
return `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}-${randomBytes(6).toString('hex')}`
}
export type WindowsPublishMode = 'create' | 'exclusive' | 'append'
/**
* Publishes a staged upload onto its real name.
*
* Every branch reads the staged *file*, never a redirected stdin, which is what makes this safe on
* a host whose Windows PowerShell 5.1 cannot drain a piped stdin.
*
* The replacing branch must never delete the destination first. Deleting and then moving loses the
* user's existing file outright if the move fails, and exposes a window where a reader sees no file
* at all — a worse outcome than the truncated-partial this staging discipline exists to prevent.
* `File.Replace` is the atomic swap (Win32 `ReplaceFile`), and it requires the destination to
* exist, so an absent one falls back to a plain `Move`. That fallback is raced deliberately: if the
* destination appears in between, `Move` throws, the staged file survives, and the destination is
* left exactly as whoever created it left it.
*
* `File::Move` throwing on an existing destination is also precisely the exclusive contract, which
* is why that branch needs nothing else.
*/
export function makeWindowsPublishStagedFileCommand(
stagingPath: string,
remotePath: string,
mode: WindowsPublishMode
): string {
const preamble = [
'$ErrorActionPreference = "Stop"',
`$staging = ${powerShellLiteral(stagingPath)}`,
`$path = ${powerShellLiteral(remotePath)}`,
'$parent = [System.IO.Path]::GetDirectoryName($path)',
'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }'
]
if (mode === 'append') {
return powerShellCommand(
[
...preamble,
// Not atomic, and cannot cheaply be: appending is defined as extending the destination, so
// a failure part-way leaves it longer than it was rather than destroyed. The caller's
// chunked-append protocol already restarts from its own offset.
'$in = [System.IO.File]::OpenRead($staging)',
'$out = [System.IO.File]::Open($path, [System.IO.FileMode]::Append, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)',
'try { $in.CopyTo($out) } finally { $out.Dispose(); $in.Dispose() }',
'[System.IO.File]::Delete($staging)'
].join('; ')
)
}
if (mode === 'exclusive') {
return powerShellCommand([...preamble, '[System.IO.File]::Move($staging, $path)'].join('; '))
}
return powerShellCommand(
[
...preamble,
// `[NullString]::Value`, not `$null`: PowerShell coerces a bare `$null` to an empty string
// when binding a .NET `string` parameter, and `Replace` rejects that with "The path is not
// of a legal form" — so every publish would fail. Measured on WindowsPowerShell 5.1.26100.
'try { [System.IO.File]::Replace($staging, $path, [NullString]::Value) } catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }'
].join('; ')
)
}
/** Best-effort removal of a staged file whose write was abandoned. Never asserts the writer died. */
export function makeWindowsDiscardStagedFileCommand(stagingPath: string): string {
return powerShellCommand(
[
// Deliberately not `Stop`: the previous writer may still hold this file, and that is a
// possibility to tolerate, not an error to report. The unique staging name means a leftover
// blocks nothing; sweeping it is housekeeping.
'$ErrorActionPreference = "SilentlyContinue"',
`$staging = ${powerShellLiteral(stagingPath)}`,
'[System.IO.File]::Delete($staging)'
].join('; ')
)
}
/**
* The ancestor directories of a Windows remote path, drive root first.
*
* sftp's `mkdir` creates one level, so a batch has to name each level itself. The drive root is
* excluded: `-mkdir "/C:/"` is not a directory anyone creates.
*/
export function windowsRemoteAncestorDirectories(remotePath: string): string[] {
const normalized = normalizeWindowsRemotePath(remotePath)
const segments = normalized.split('/')
segments.pop()
const ancestors: string[] = []
// Start past the drive (`C:`) or the UNC host, which are never created.
for (let depth = 2; depth <= segments.length; depth += 1) {
const directory = segments.slice(0, depth).join('/')
if (directory) {
ancestors.push(directory)
}
}
return ancestors
}
/**
* `[Console]::OpenStandardInput()` into a `FileStream`, used only by the two stdin fallbacks.
*
* On Windows PowerShell 5.1 this is the defective read; see the strategy comment in
* `system-ssh-file-binary-transfer.ts`. It is correct under PowerShell 7.
*/
export function makeWindowsWriteFileCommand(
remotePath: string,
options?: { append?: boolean; exclusive?: boolean; executable?: 'powershell.exe' | 'pwsh.exe' }
): string {
const fileMode = options?.append ? 'Append' : options?.exclusive ? 'CreateNew' : 'Create'
return powerShellCommand(
[
'$ErrorActionPreference = "Stop"',
`$path = ${powerShellLiteral(remotePath)}`,
'$parent = [System.IO.Path]::GetDirectoryName($path)',
'if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }',
'$inputStream = [Console]::OpenStandardInput()',
`$outputStream = [System.IO.File]::Open($path, [System.IO.FileMode]::${fileMode}, [System.IO.FileAccess]::Write, [System.IO.FileShare]::None)`,
'try { $inputStream.CopyTo($outputStream) } finally { $outputStream.Dispose() }'
].join('; '),
options?.executable ?? 'powershell.exe'
)
}
+526 -133
View File
@@ -1,29 +1,40 @@
/**
* #16432: the Windows relay upload pushed the whole bundle into one PowerShell stdin, which
* Windows PowerShell 5.1 cannot drain over a non-pty ssh exec — the remote blocks forever, and
* `waitForChannelClose()` had no timeout, so the UI sat at "Connecting…" with no error. Covered
* here: no write exceeds one stdin's worth on any Windows path (bundle upload *and* single-file
* upload, which is the one that carries large files), a partial write never lands under the real
* name, and a remote that never closes fails instead of hanging.
* #16432. The original fix chunked the payload because the constraint was believed to be a ~50KB
* cmd.exe stdin ceiling. Re-measured on Windows 11 26200.9168 / OpenSSH_for_Windows_10.0p2, it is
* not a size limit and not cmd.exe's: a read on Windows PowerShell 5.1's redirected-stdin handle
* over a non-pty ssh exec can die permanently when it finds the stream momentarily empty, taking
* both the remaining data and the EOF with it. It is probabilistic per such read — identical 2MB
* payloads died at 167936, 270336 and 372736 — so a 32KB chunk still failed 15 times in 120 under
* load, while `findstr` took 2,016,000 bytes through one exec on the same host.
*
* So the covering property is no longer "every write is small". It is "the bytes do not cross a
* remote process's stdin at all": sftp first, PowerShell 7 next, and Windows PowerShell 5.1 last,
* bounded and loud. The staging-and-rename discipline is kept on every path, with a unique staging
* name per attempt so a retry never meets a predecessor's lock.
*/
import { EventEmitter } from 'node:events'
import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs'
import { rm } from 'node:fs/promises'
import { readFile, rm, stat } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { PassThrough, Writable } from 'node:stream'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type * as SystemSshOperationLifecycle from './system-ssh-operation-lifecycle'
const { spawnSystemSshCommandMock, waitForChannelCloseSpy } = vi.hoisted(() => ({
const { spawnSystemSshCommandMock, waitForChannelCloseSpy, runProcessMock } = vi.hoisted(() => ({
spawnSystemSshCommandMock: vi.fn(),
waitForChannelCloseSpy: vi.fn()
waitForChannelCloseSpy: vi.fn(),
runProcessMock: vi.fn()
}))
vi.mock('./system-ssh-command', () => ({
spawnSystemSshCommand: spawnSystemSshCommandMock
}))
vi.mock('../../shared/child-process/run-process', () => ({
runProcess: runProcessMock
}))
// Delegates to the real implementation; the spy only records whether each wait was given a bound.
vi.mock('./system-ssh-operation-lifecycle', async (importActual) => {
const actual = (await importActual()) as typeof SystemSshOperationLifecycle
@@ -41,6 +52,11 @@ import {
} from './system-ssh-file-binary-transfer'
import { waitForChannelClose } from './system-ssh-operation-lifecycle'
import { getRemoteHostPlatform } from './ssh-remote-platform'
import {
clearWindowsRemoteWriteCapabilitiesForTests,
getWindowsRemoteWriteCapabilities
} from './system-ssh-windows-write-capabilities'
import { explainWindowsPowerShellStdinFailure } from './system-ssh-windows-write-strategy'
import type { SshTarget } from '../../shared/ssh-types'
type FakeChannel = EventEmitter & {
@@ -50,7 +66,12 @@ type FakeChannel = EventEmitter & {
written: Buffer
}
const target = { id: 'win-1', host: 'win.example', username: 'dev' } as unknown as SshTarget
const target = {
id: 'win-1',
host: 'win.example',
username: 'dev',
port: 22
} as unknown as SshTarget
const hostPlatform = getRemoteHostPlatform('win32-x64')
const remoteRoot = 'C:/Users/dev/.orca-remote'
@@ -78,96 +99,238 @@ function createFakeChannel(onEnd: (channel: FakeChannel) => void): FakeChannel {
return channel
}
type RecordedCommand = { script: string; stdin: Buffer }
type RecordedCommand = { script: string; executable: string; stdin: Buffer }
type RecordedSftpBatch = { args: string[]; script: string }
describe('Windows upload stdin framing', () => {
let localDir: string
const commands: RecordedCommand[] = []
/** Index of the spawn that should report a non-zero exit, to model a chunk failing mid-file. */
let failAtSpawn = -1
const sftpBatches: RecordedSftpBatch[] = []
const commands: RecordedCommand[] = []
/** Index of the exec that should report a non-zero exit, to model a chunk failing mid-file. */
let failAtSpawn = -1
let localDir: string
const fileWrites = (): RecordedCommand[] =>
commands.filter((command) => command.script.includes('FileMode]::'))
const writtenPath = (command: RecordedCommand): string =>
/\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1].replace(/''/g, "'") ?? ''
const fileMode = (command: RecordedCommand): string | undefined =>
/FileMode\]::(\w+)/.exec(command.script)?.[1]
const fileWrites = (): RecordedCommand[] =>
commands.filter((command) => command.script.includes('OpenStandardInput'))
const writtenPath = (command: RecordedCommand): string =>
/\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1]?.replace(/''/g, "'") ?? ''
const fileMode = (command: RecordedCommand): string | undefined =>
/FileMode\]::(\w+)/.exec(command.script)?.[1]
const putLines = (): string[] =>
sftpBatches.flatMap((batch) => batch.script.split('\n').filter((line) => line.startsWith('put ')))
const putDestination = (line: string): string => /put "(?:[^"]*)" "([^"]*)"/.exec(line)?.[1] ?? ''
const putSource = (line: string): string => /put "([^"]*)"/.exec(line)?.[1] ?? ''
beforeEach(() => {
commands.length = 0
failAtSpawn = -1
waitForChannelCloseSpy.mockClear()
localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-'))
spawnSystemSshCommandMock.mockReset()
spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => {
const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1
return createFakeChannel((channel) => {
commands.push({ script: decodePowerShellCommand(command), stdin: channel.written })
setImmediate(() =>
spawnIndex === failAtSpawn
? channel.emit('close', 1, null)
: channel.emit('close', 0, null)
)
/** Makes every sftp batch succeed, recording what it was asked to do. */
function acceptSftp(): void {
runProcessMock.mockImplementation(
async (spec: { args: string[]; input: string; program: string }) => {
const script = spec.input
sftpBatches.push({ args: spec.args, script })
// Model the real client: `put` copies the local file, so read it while it still exists.
for (const line of script.split('\n').filter((entry) => entry.startsWith('put '))) {
await readFile(putSource(line))
}
return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false }
}
)
}
/** Models a host whose sshd has no `Subsystem sftp` line. */
function refuseSftp(): void {
runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => {
sftpBatches.push({ args: spec.args, script: spec.input })
return {
code: 255,
signal: null,
stdout: '',
stderr: 'subsystem request failed on channel 0\nConnection closed',
timedOut: false
}
})
}
/** Models a host with no PowerShell 7, which cmd.exe reports as an unrecognized command. */
function refusePwsh(): void {
spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => {
const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1
const executable = command.split(' ')[0] ?? ''
return createFakeChannel((channel) => {
commands.push({
script: decodePowerShellCommand(command),
executable,
stdin: channel.written
})
setImmediate(() => {
if (executable === 'pwsh.exe') {
channel.stderr.write(
"'pwsh.exe' is not recognized as an internal or external command,\noperable program or batch file."
)
channel.emit('close', 9009, null)
return
}
channel.emit('close', spawnIndex === failAtSpawn ? 1 : 0, null)
})
})
})
}
afterEach(async () => {
await rm(localDir, { recursive: true, force: true })
beforeEach(() => {
commands.length = 0
sftpBatches.length = 0
failAtSpawn = -1
clearWindowsRemoteWriteCapabilitiesForTests()
waitForChannelCloseSpy.mockClear()
localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-'))
process.env.ORCA_SYSTEM_SFTP_PATH = '/usr/bin/sftp'
runProcessMock.mockReset()
acceptSftp()
spawnSystemSshCommandMock.mockReset()
spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => {
const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1
return createFakeChannel((channel) => {
commands.push({
script: decodePowerShellCommand(command),
executable: command.split(' ')[0] ?? '',
stdin: channel.written
})
setImmediate(() =>
spawnIndex === failAtSpawn ? channel.emit('close', 1, null) : channel.emit('close', 0, null)
)
})
})
})
it('never pushes a whole artifact bundle into one PowerShell stdin', async () => {
mkdirSync(join(localDir, 'node'), { recursive: true })
// Comfortably past the ~50KB point at which the reporter measured PowerShell 5.1 wedging.
writeFileSync(join(localDir, 'node', 'relay.js'), Buffer.alloc(600 * 1024, 0x61))
writeFileSync(join(localDir, 'index.js'), Buffer.alloc(300 * 1024, 0x62))
afterEach(async () => {
delete process.env.ORCA_SYSTEM_SFTP_PATH
await rm(localDir, { recursive: true, force: true })
})
await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform })
const largest = Math.max(...commands.map((command) => command.stdin.length))
expect(largest).toBeLessThanOrEqual(WINDOWS_STDIN_WRITE_CHUNK_BYTES)
// The base64 + JSON envelope is gone entirely: nothing reads the bundle as one string.
expect(commands.some((command) => command.script.includes('FromBase64String'))).toBe(false)
// `[Console]::In` wedged at 50KB where the stream reader did not, so the mkdir batch — the one
// payload still read as a string — must use the reader the reporter measured surviving.
expect(commands.some((command) => command.script.includes('[Console]::In.ReadToEnd()'))).toBe(
false
)
expect(
commands.filter((command) => command.script.includes('StreamReader([Console]::'))
).toHaveLength(1)
})
it('bounds the single-file upload too, which is the path large files take', async () => {
const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64)
describe('Windows upload over sftp', () => {
it('moves the payload without any remote process reading a stdin', async () => {
const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 60 + 11, 0x64)
const localPath = join(localDir, 'big.node')
writeFileSync(localPath, contents)
await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/big.node`, { hostPlatform })
const writes = fileWrites()
expect(writes).toHaveLength(4)
expect(Math.max(...writes.map((write) => write.stdin.length))).toBe(
WINDOWS_STDIN_WRITE_CHUNK_BYTES
)
expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true)
// A wedged PowerShell never closes on its own, so no wait on this path may be unbounded.
expect(
waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS)
).toBe(true)
// The defect is a remote stdin read; the fix is that there is not one.
expect(fileWrites()).toHaveLength(0)
expect(putLines()).toHaveLength(1)
// One transfer, not 61 execs: the whole point of the change.
expect(sftpBatches).toHaveLength(1)
})
it('writes every byte of every artifact across the chunked writes', async () => {
const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 2 + 17, 0x63)
writeFileSync(join(localDir, 'relay.js'), contents)
it('creates the parent chain and sends the payload in one round trip', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform })
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/a/b/relay.js`, {
hostPlatform
})
const writes = fileWrites()
expect(writes).toHaveLength(3)
expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true)
// Only the first write creates the staging file; the rest must extend it or it is truncated.
expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append'])
expect(sftpBatches).toHaveLength(1)
expect(sftpBatches[0]!.script.split('\n').filter(Boolean)).toEqual([
'-mkdir "/C:/Users"',
'-mkdir "/C:/Users/dev"',
'-mkdir "/C:/Users/dev/.orca-remote"',
'-mkdir "/C:/Users/dev/.orca-remote/a"',
'-mkdir "/C:/Users/dev/.orca-remote/a/b"',
expect.stringContaining('put ') as unknown as string
])
})
it('addresses the destination in the drive-rooted namespace sftp exposes', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
// A backslash destination silently writes a file named `C` and still exits 0, so the leading
// slash and forward separators are correctness, not style.
expect(putDestination(putLines()[0]!)).toMatch(
/^\/C:\/Users\/dev\/\.orca-remote\/relay\.js\.orca-partial-[0-9a-f]{12}$/
)
})
it('never lands a partial under the real name, and publishes by rename', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
const remotePath = `${remoteRoot}/relay.js`
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), remotePath, { hostPlatform })
const destination = putDestination(putLines()[0]!)
// Assert the positive first: an unmatched regex yields '', which would satisfy the `not.toBe`
// below without this test ever having seen a destination.
expect(destination).toContain(WINDOWS_STAGED_WRITE_SUFFIX)
expect(destination).not.toBe(`/C:${remotePath.slice(2)}`)
const publish = commands.at(-1)!
expect(publish.script).toContain(
'[System.IO.File]::Replace($staging, $path, [NullString]::Value)'
)
// The publish reads the staged file, never a pipe, so it is safe on PowerShell 5.1.
expect(publish.script).not.toContain('OpenStandardInput')
})
it('never deletes the destination it is replacing', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
const publish = commands.at(-1)!
// Delete-then-move destroys the user's existing file outright if the move then fails, and
// exposes a window where a reader sees no file at all — worse than the truncated partial the
// staging discipline exists to prevent. `File.Replace` is the atomic swap.
expect(publish.script).not.toContain('[System.IO.File]::Delete($path)')
expect(publish.script).toContain(
'[System.IO.File]::Replace($staging, $path, [NullString]::Value)'
)
// An absent destination cannot be Replaced, so that case falls back to a plain Move.
expect(publish.script).toContain(
'catch [System.IO.FileNotFoundException] { [System.IO.File]::Move($staging, $path) }'
)
})
it('gives every attempt its own staging name, so a retry cannot meet a predecessor lock', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
const [first, second] = putLines().map(putDestination)
expect(first).toContain(WINDOWS_STAGED_WRITE_SUFFIX)
// Losing contact is not evidence the previous writer died, so the name must not be reused.
expect(second).not.toBe(first)
})
it('enforces exclusive at the rename, where it is atomic', async () => {
writeFileSync(join(localDir, 'import.bin'), 'x')
await uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, {
hostPlatform,
exclusive: true
})
const publish = commands.at(-1)!
expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)')
expect(publish.script).not.toContain('[System.IO.File]::Delete($path)')
})
it('appends by concatenating the staged file, not by piping bytes to the remote', async () => {
await writeBufferViaSystemSsh(target, `${remoteRoot}/log.bin`, Buffer.from('tail'), {
hostPlatform,
append: true
})
expect(fileWrites()).toHaveLength(0)
const publish = commands.at(-1)!
expect(publish.script).toContain('FileMode]::Append')
expect(publish.script).toContain('$in.CopyTo($out)')
expect(publish.script).toContain('[System.IO.File]::Delete($staging)')
})
it('still creates an empty artifact on the host', async () => {
@@ -175,80 +338,310 @@ describe('Windows upload stdin framing', () => {
await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform })
expect(fileWrites().map(writtenPath)).toEqual([`${remoteRoot}/empty.txt`])
expect(fileWrites()[0].stdin).toHaveLength(0)
expect(fileMode(fileWrites()[0])).toBe('Create')
expect(putLines()).toHaveLength(1)
expect(commands.at(-1)!.script).toContain('[System.IO.File]::Move($staging, $path)')
})
it('lands a multi-chunk write on a staging path and publishes it by rename', async () => {
const remotePath = `${remoteRoot}/relay.js`
writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1))
it('writes a buffer through a 0600 temp file that does not outlive the transfer', async () => {
const seen: { path: string; contents: Buffer; mode: number }[] = []
runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => {
sftpBatches.push({ args: spec.args, script: spec.input })
for (const line of spec.input.split('\n').filter((entry) => entry.startsWith('put '))) {
const path = putSource(line)
seen.push({
path,
contents: await readFile(path),
mode: (await stat(path)).mode & 0o777
})
}
return { code: 0, signal: null, stdout: '', stderr: '', timedOut: false }
})
await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), {
hostPlatform
})
expect(seen).toHaveLength(1)
expect(seen[0]!.contents.toString()).toBe('1.2.3')
// The payload can be repository content and tmpdir is world-readable on every platform, so the
// window between write and upload must not be group- or world-readable.
expect(seen[0]!.mode).toBe(0o600)
await expect(readFile(seen[0]!.path)).rejects.toThrow()
})
it('creates upload directories over sftp rather than a PowerShell stdin batch', async () => {
mkdirSync(join(localDir, 'node'), { recursive: true })
writeFileSync(join(localDir, 'node', 'relay.js'), 'x')
await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform })
// Nothing touches the real name until every byte is on the host.
expect(fileWrites().map(writtenPath)).toEqual([
`${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`,
`${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`
])
const publish = commands.at(-1)!
expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)')
expect(publish.script).toContain('[System.IO.File]::Delete($path)')
// Anchor on a non-empty observation: `some` is false of an empty list, so this would pass even
// if no command had been recorded at all.
expect(commands.length).toBeGreaterThan(0)
expect(commands.some((command) => command.script.includes('StreamReader([Console]::'))).toBe(
false
)
expect(sftpBatches[0]!.script).toContain('-mkdir "/C:/Users/dev/.orca-remote"')
})
it('sweeps the staged bytes when the publish is the thing that fails', async () => {
writeFileSync(join(localDir, 'import.bin'), 'x')
// An exclusive conflict is the ordinary way to get here: the payload is on the host, and the
// rename that would have given it a name refuses.
spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => {
const script = decodePowerShellCommand(command)
return createFakeChannel((channel) => {
commands.push({ script, executable: command.split(' ')[0] ?? '', stdin: channel.written })
const failed = script.includes('::Move($staging, $path)')
setImmediate(() => channel.emit('close', failed ? 1 : 0, null))
})
})
await expect(
uploadFileViaSystemSsh(target, join(localDir, 'import.bin'), `${remoteRoot}/import.bin`, {
hostPlatform,
exclusive: true
})
).rejects.toThrow()
const sweep = commands.at(-1)!
expect(sweep.script).toContain('[System.IO.File]::Delete($staging)')
// Tolerated, not asserted: the previous writer may still hold the file, and losing contact is
// not evidence it died.
expect(sweep.script).toContain('$ErrorActionPreference = "SilentlyContinue"')
})
it('reports a cancelled transfer as an abort, not as a failed one', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
const controller = new AbortController()
// runProcess reports the kill as a non-zero exit rather than throwing, so without checking the
// signal first a user pressing cancel is indistinguishable from the transfer genuinely failing.
runProcessMock.mockImplementation(async (spec: { args: string[]; input: string }) => {
sftpBatches.push({ args: spec.args, script: spec.input })
controller.abort()
return { code: 255, signal: 'SIGTERM', stdout: '', stderr: '', timedOut: false }
})
let error: Error | undefined
try {
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform,
signal: controller.signal
})
} catch (thrown) {
error = thrown as Error
}
expect(error?.name).toBe('AbortError')
expect(error?.message).not.toContain('sftp batch failed')
// A cancel is also not evidence about the host, so it must not send later writes to the slow
// path, and must not fall through to the defective reader now.
expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true)
expect(fileWrites()).toHaveLength(0)
})
it('does not let one unaddressable path become a verdict about the host', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
// A UNC destination has no settled mapping in sftp's drive-rooted namespace, so this write
// falls back — but the host still serves sftp perfectly well for every other path.
await uploadFileViaSystemSsh(
target,
join(localDir, 'relay.js'),
'//fileserver/share/relay.js',
{ hostPlatform }
)
expect(fileWrites().length).toBeGreaterThan(0)
expect(sftpBatches).toHaveLength(0)
// The 30-minute capability cache is keyed by host; caching this would send every later write
// to the same machine down the defective path on the strength of one odd destination.
expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true)
})
it('keeps using sftp for the next file after one path it could not spell', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), '//fileserver/share/a.js', {
hostPlatform
})
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/b.js`, {
hostPlatform
})
expect(putLines()).toHaveLength(1)
expect(putDestination(putLines()[0]!)).toContain('/C:/Users/dev/.orca-remote/b.js')
})
it('does not let a local filename sftp cannot quote become a verdict either', async () => {
// POSIX clients allow a newline in a filename, and sftp's batch lexer would read it as the end
// of one command and the start of another.
const awkward = join(localDir, 'two\nlines.js')
writeFileSync(awkward, 'x')
await uploadFileViaSystemSsh(target, awkward, `${remoteRoot}/relay.js`, { hostPlatform })
expect(fileWrites().length).toBeGreaterThan(0)
expect(getWindowsRemoteWriteCapabilities(target).shouldTry('sftp-subsystem')).toBe(true)
})
it('translates the ssh argument list rather than passing it to a client that reads it differently', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform,
disableControlMaster: true
})
const args = sftpBatches[0]!.args
// sftp's `-T` does not exist, its `-p` preserves mtime, and its `-S` names a program to run.
expect(args).not.toContain('-T')
expect(args).not.toContain('-p')
expect(args).not.toContain('-S')
expect(args).toContain('ControlPath=none')
expect(args).toContain('ServerAliveInterval=15')
})
})
describe('Windows upload on a host with no sftp subsystem', () => {
beforeEach(() => {
refuseSftp()
})
it('creates a multi-directory tree, which the one-element case never exercised', async () => {
mkdirSync(join(localDir, 'node', 'deep'), { recursive: true })
writeFileSync(join(localDir, 'index.js'), 'a')
writeFileSync(join(localDir, 'node', 'deep', 'x.js'), 'b')
await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform })
const mkdir = commands.find((command) => command.script.includes('ConvertFrom-Json'))!
// `@($json | ConvertFrom-Json)` wraps the parsed array in another array, so the loop variable
// binds to the whole thing and `[string]` of it is the paths joined by spaces — which
// CreateDirectory rejects. It only ever worked for a single directory, where stringifying a
// one-element array happens to yield the element, so no batch of one can catch this.
expect(mkdir.script).toContain('[string[]]($json | ConvertFrom-Json)')
expect(mkdir.script).not.toContain('@($json | ConvertFrom-Json)')
const batch = JSON.parse(mkdir.stdin.toString('utf-8')) as string[]
expect(batch.length).toBeGreaterThan(1)
})
it('falls back rather than failing the transfer', async () => {
const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 5, 0x61)
writeFileSync(join(localDir, 'relay.js'), contents)
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
expect(Buffer.concat(fileWrites().map((write) => write.stdin)).equals(contents)).toBe(true)
})
it('remembers the refusal, so a multi-file upload probes once', async () => {
writeFileSync(join(localDir, 'a.js'), 'a')
writeFileSync(join(localDir, 'b.js'), 'b')
writeFileSync(join(localDir, 'c.js'), 'c')
await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform })
// One refusal is enough; re-probing per file is a wasted round trip on every file.
expect(sftpBatches).toHaveLength(1)
})
it('does not spend a sweep round trip when sftp declined before moving any bytes', async () => {
writeFileSync(join(localDir, 'relay.js'), 'x')
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
// A refused subsystem staged nothing, so there is nothing to delete — and on a host without
// sftp that sweep would otherwise be paid on every single write.
expect(commands.length).toBeGreaterThan(0)
expect(commands.some((command) => command.script.includes('Delete($staging)'))).toBe(false)
})
it('prefers PowerShell 7, which reads a redirected stdin correctly', async () => {
writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3))
await uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
expect(fileWrites().map((write) => write.executable)).toEqual(['pwsh.exe'])
// PowerShell 7 took 2MB through one exec when measured, so chunking it buys nothing.
expect(fileWrites()[0]!.stdin).toHaveLength(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)
})
it('bounds every write when only Windows PowerShell 5.1 is available', async () => {
refusePwsh()
const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64)
writeFileSync(join(localDir, 'big.node'), contents)
await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, {
hostPlatform
})
const writes = fileWrites().filter((write) => write.executable === 'powershell.exe')
expect(writes).toHaveLength(4)
expect(Math.max(...writes.map((write) => write.stdin.length))).toBe(
WINDOWS_STDIN_WRITE_CHUNK_BYTES
)
expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true)
expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append', 'Append'])
// A wedged PowerShell never closes on its own, so no wait on this path may be unbounded.
// Count first: `every` is true of zero calls, so a wait that moved to a different helper would
// pass this silently.
expect(waitForChannelCloseSpy.mock.calls.length).toBeGreaterThan(0)
expect(
waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS)
).toBe(true)
})
it('remembers that PowerShell 7 is absent instead of re-probing per chunk', async () => {
refusePwsh()
writeFileSync(join(localDir, 'big.node'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3))
await uploadFileViaSystemSsh(target, join(localDir, 'big.node'), `${remoteRoot}/big.node`, {
hostPlatform
})
expect(fileWrites().filter((write) => write.executable === 'pwsh.exe')).toHaveLength(1)
})
it('leaves no truncated file under the real name when a chunk fails mid-file', async () => {
writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3))
// Spawns: 0 = mkdir batch, 1..3 = chunk writes. Fail the second chunk.
failAtSpawn = 2
// Spawn 0 is the pwsh write; fail it and every retry beneath it.
failAtSpawn = 0
await expect(
uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform })
uploadFileViaSystemSsh(target, join(localDir, 'relay.js'), `${remoteRoot}/relay.js`, {
hostPlatform
})
).rejects.toThrow()
expect(fileWrites().length).toBeGreaterThan(0)
expect(fileWrites().map(writtenPath)).not.toContain(`${remoteRoot}/relay.js`)
expect(commands.some((command) => command.script.includes('::Move('))).toBe(false)
})
})
it('enforces exclusive once at the rename, so a retry is not blocked by its own leftovers', async () => {
const localPath = join(localDir, 'import.bin')
writeFileSync(localPath, Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1))
describe('last-resort Windows PowerShell failure reporting', () => {
it('names the host limitation and its remedy, not just the timeout', () => {
const timeout = new Error('write C:/x at offset 0 timed out after 60000ms with no response')
await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/import.bin`, {
hostPlatform,
exclusive: true
})
const explained = explainWindowsPowerShellStdinFailure(timeout) as Error
// CreateNew on chunk one would fail against a leftover staging file from a failed attempt;
// `File::Move` raising on an existing destination is what carries the exclusive contract.
expect(fileWrites().map(fileMode)).toEqual(['Create', 'Append'])
const publish = commands.at(-1)!
expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)')
expect(publish.script).not.toContain('[System.IO.File]::Delete($path)')
// "timed out" alone sends the user to retry a network they cannot fix; the fix is host-side.
expect(explained.message).toContain('Windows PowerShell 5.1')
expect(explained.message).toContain('Subsystem sftp sftp-server.exe')
expect(explained.cause).toBe(timeout)
})
it('keeps a single-chunk write on the destination, with the caller mode intact', async () => {
await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), {
hostPlatform,
exclusive: true
})
it('leaves a real failure alone, so a permission error is not reported as a host limitation', () => {
const denied = new Error('write C:/x at offset 0 failed (exit 1): Access to the path is denied')
expect(fileWrites()).toHaveLength(1)
expect(writtenPath(fileWrites()[0])).toBe(`${remoteRoot}/version`)
expect(fileMode(fileWrites()[0])).toBe('CreateNew')
expect(commands.some((command) => command.script.includes('::Move('))).toBe(false)
})
it('appends onto the destination rather than staging, since append cannot be staged', async () => {
const remotePath = `${remoteRoot}/log.bin`
await writeBufferViaSystemSsh(
target,
remotePath,
Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1),
{ hostPlatform, append: true }
)
expect(fileWrites().map(writtenPath)).toEqual([remotePath, remotePath])
expect(fileWrites().map(fileMode)).toEqual(['Append', 'Append'])
expect(explainWindowsPowerShellStdinFailure(denied)).toBe(denied)
})
})
@@ -0,0 +1,79 @@
/**
* Whether a Windows host has an sftp subsystem is a fact about that host, so the cache is keyed by
* the endpoint that executes rather than by Orca's target id — otherwise a hardened host is
* re-probed once per file, and two targets pointing at one machine learn the same fact twice.
*/
import { afterEach, describe, expect, it } from 'vitest'
import type { SshTarget } from '../../shared/ssh-types'
import {
clearWindowsRemoteWriteCapabilitiesForTests,
getWindowsRemoteWriteCapabilities,
getWindowsRemoteWriteExecutionHostKey
} from './system-ssh-windows-write-capabilities'
const asTarget = (fields: Partial<SshTarget>): SshTarget => fields as SshTarget
afterEach(() => {
clearWindowsRemoteWriteCapabilitiesForTests()
})
describe('getWindowsRemoteWriteExecutionHostKey', () => {
it('gives two targets on one endpoint the same key', () => {
const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 })
const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 })
// A target re-created under a new id has not changed what the host supports.
expect(getWindowsRemoteWriteExecutionHostKey(first)).toBe(
getWindowsRemoteWriteExecutionHostKey(second)
)
})
it('separates hosts, ports and users', () => {
const base = { id: 'a', host: 'win.example', username: 'dev', port: 22 }
const keys = [
asTarget(base),
asTarget({ ...base, host: 'other.example' }),
asTarget({ ...base, port: 2222 }),
asTarget({ ...base, username: 'ops' })
].map(getWindowsRemoteWriteExecutionHostKey)
expect(new Set(keys).size).toBe(4)
})
it('keys a config alias by the alias, since ssh_config decides where it lands', () => {
const alias = asTarget({ id: 'a', host: 'stale.example', configHost: 'winbox' })
expect(getWindowsRemoteWriteExecutionHostKey(alias)).toBe('config:winbox')
})
})
describe('getWindowsRemoteWriteCapabilities', () => {
it('shares one cache across targets that reach the same host', () => {
const first = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 })
const second = asTarget({ id: 'b', host: 'win.example', username: 'dev', port: 22 })
getWindowsRemoteWriteCapabilities(first).rememberUnsupported('sftp-subsystem')
expect(getWindowsRemoteWriteCapabilities(second).shouldTry('sftp-subsystem')).toBe(false)
})
it('does not let one host answer for another', () => {
const hardened = asTarget({ id: 'a', host: 'hardened.example', username: 'dev', port: 22 })
const ordinary = asTarget({ id: 'b', host: 'ordinary.example', username: 'dev', port: 22 })
getWindowsRemoteWriteCapabilities(hardened).rememberUnsupported('sftp-subsystem')
expect(getWindowsRemoteWriteCapabilities(ordinary).shouldTry('sftp-subsystem')).toBe(true)
})
it('keeps the two capabilities independent', () => {
const target = asTarget({ id: 'a', host: 'win.example', username: 'dev', port: 22 })
const capabilities = getWindowsRemoteWriteCapabilities(target)
capabilities.rememberUnsupported('pwsh')
// No PowerShell 7 says nothing about whether the host will serve sftp.
expect(capabilities.shouldTry('sftp-subsystem')).toBe(true)
expect(capabilities.shouldTry('pwsh')).toBe(false)
})
})
@@ -0,0 +1,52 @@
import type { SshTarget } from '../../shared/ssh-types'
import { CapabilityProbeCache } from '../../shared/capability-probe-cache'
/**
* Whether a Windows host can take a file write over the sftp subsystem, and whether it has a
* PowerShell 7 to fall back to. Both are host facts, so they are cached per execution host rather
* than per transfer — a hardened host with `Subsystem sftp` removed must not be re-probed on every
* file of a multi-file upload.
*/
export type WindowsRemoteWriteCapability = 'sftp-subsystem' | 'pwsh'
// Why re-probe at all: an admin can enable the subsystem, or install PowerShell 7, without the
// user restarting Orca. Long enough that a hardened host costs one failed probe per half hour.
export const WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS = 30 * 60_000
const capabilitiesByExecutionHost = new Map<
string,
CapabilityProbeCache<WindowsRemoteWriteCapability>
>()
/**
* Keyed by the endpoint that executes, not by target id: two Orca targets pointing at one host
* describe the same sshd, and a target re-created under a new id has not changed what that host
* supports. A config alias is its own key because ssh_config, not Orca, resolves where it lands.
*/
export function getWindowsRemoteWriteExecutionHostKey(target: SshTarget): string {
if (target.configHost) {
return `config:${target.configHost}`
}
const port = target.port ?? 22
return target.username
? `host:${target.username}@${target.host}:${port}`
: `host:${target.host}:${port}`
}
export function getWindowsRemoteWriteCapabilities(
target: SshTarget
): CapabilityProbeCache<WindowsRemoteWriteCapability> {
const key = getWindowsRemoteWriteExecutionHostKey(target)
let cache = capabilitiesByExecutionHost.get(key)
if (!cache) {
cache = new CapabilityProbeCache<WindowsRemoteWriteCapability>(
WINDOWS_WRITE_CAPABILITY_RETRY_INTERVAL_MS
)
capabilitiesByExecutionHost.set(key, cache)
}
return cache
}
export function clearWindowsRemoteWriteCapabilitiesForTests(): void {
capabilitiesByExecutionHost.clear()
}
@@ -0,0 +1,329 @@
import type { SshTarget } from '../../shared/ssh-types'
import { getSystemSshBuildArgsFromOperationOptions } from './system-ssh-args'
import { spawnSystemSshCommand } from './system-ssh-command'
import {
awaitWithSystemSshAbort,
throwIfAborted,
waitForChannelClose
} from './system-ssh-operation-lifecycle'
import {
isSftpPathUnsupportedError,
isSftpRefusalBeforeStaging,
isSftpUnavailableError,
runSftpBatch
} from './system-ssh-sftp-transfer'
import { quoteSftpBatchArgument, toSftpRemotePath } from './system-ssh-sftp-path'
import { getWindowsRemoteWriteCapabilities } from './system-ssh-windows-write-capabilities'
import {
makeWindowsDiscardStagedFileCommand,
makeWindowsPublishStagedFileCommand,
makeWindowsStagingPath,
makeWindowsWriteFileCommand,
windowsRemoteAncestorDirectories,
type WindowsPublishMode
} from './system-ssh-windows-file-write'
/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */
export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000
/**
* Bound on one stdin write for the last-resort Windows PowerShell 5.1 path.
*
* Measured on Windows 11 26200 / OpenSSH 10.0p2: a 32KB write still hangs 15 times in 120 under
* load, and no smaller value removes the risk. The defect is per blocking read, not per byte, so
* shrinking the chunk trades one risky read for more execs that each carry their own. This is a
* damage bound on a path known to be unreliable, not a safe size.
*/
export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024
export type WindowsWriteOptions = Parameters<
typeof getSystemSshBuildArgsFromOperationOptions
>[0] & {
signal?: AbortSignal
append?: boolean
exclusive?: boolean
}
/** Bytes to write, plus a way to present them to sftp, which can only send a local file. */
export type WindowsWriteSource = {
totalBytes: number
readChunk: (offset: number, maxBytes: number) => Promise<Buffer>
withLocalFile: <T>(send: (localPath: string) => Promise<T>) => Promise<T>
}
function publishMode(options: WindowsWriteOptions): WindowsPublishMode {
return options.append ? 'append' : options.exclusive === true ? 'exclusive' : 'create'
}
/**
* Writes one file to a Windows host, preferring transports that do not push bytes through a remote
* PowerShell's stdin.
*
* Order, and why: sftp carries the whole payload in one transfer and never has a remote process
* read a pipe. Measured on Windows 11 / OpenSSH 10.0p2: 1.9MB in a median 315ms over sftp against
* 0 of 6 completions on the chunked path, whose best case was ~62 execs at ~350ms each. PowerShell
* 7 reads a redirected stdin correctly but is not installed by default. Windows PowerShell 5.1 is
* always present and is the defective reader, so it is last and it is bounded.
*
* Every transport stages under a unique name and publishes by rename, so no partial write is ever
* visible under the real name and no retry inherits a predecessor's lock.
*/
export async function writeWindowsRemoteFile(
target: SshTarget,
remotePath: string,
source: WindowsWriteSource,
options: WindowsWriteOptions
): Promise<void> {
throwIfAborted(options.signal)
const capabilities = getWindowsRemoteWriteCapabilities(target)
await capabilities.runWithFallback(
'sftp-subsystem',
() => writeViaSftp(target, remotePath, source, options),
() => writeViaRemoteStdin(target, remotePath, source, options),
isSftpUnavailableError
)
}
/**
* Stages under a name nothing else can own, publishes it, and sweeps the staging file if either
* step fails.
*
* Shared by both transports so the cleanup contract cannot drift between them: a failed publish —
* an exclusive conflict is the ordinary case — leaves bytes on the host that no longer have a
* purpose, and the sweep is what stops them accumulating.
*/
async function stageThenPublish(
target: SshTarget,
remotePath: string,
options: WindowsWriteOptions,
stage: (stagingPath: string) => Promise<void>,
nothingStaged: (error: unknown) => boolean = () => false
): Promise<void> {
const stagingPath = makeWindowsStagingPath(remotePath)
try {
await stage(stagingPath)
await publishStagedWrite(target, stagingPath, remotePath, options)
} catch (error) {
// A transport that declined before it moved any bytes has nothing to sweep, and sweeping
// anyway would spend a round trip on every write to a host that has no sftp subsystem.
if (!nothingStaged(error)) {
await discardStagedWrite(target, stagingPath, options)
}
throw error
}
}
/**
* A path sftp cannot address falls back for this write alone, without touching the host verdict.
*
* The distinction matters because the capability cache is keyed by host and holds for half an hour:
* routing one UNC destination, or one local filename containing a newline, into
* `rememberUnsupported` would send every later write to that host down the defective path too.
*/
async function writeViaSftp(
target: SshTarget,
remotePath: string,
source: WindowsWriteSource,
options: WindowsWriteOptions
): Promise<void> {
try {
await attemptSftpWrite(target, remotePath, source, options)
} catch (error) {
if (!isSftpPathUnsupportedError(error)) {
throw error
}
await writeViaRemoteStdin(target, remotePath, source, options)
}
}
function attemptSftpWrite(
target: SshTarget,
remotePath: string,
source: WindowsWriteSource,
options: WindowsWriteOptions
): Promise<void> {
const mkdirs = windowsRemoteAncestorDirectories(remotePath).map(
(directory) => `-mkdir ${quoteSftpBatchArgument(toSftpRemotePath(directory))}`
)
return stageThenPublish(
target,
remotePath,
options,
(stagingPath) =>
source.withLocalFile((localPath) =>
// One round trip: the parent chain and the payload travel in the same batch.
runSftpBatch(
target,
[
...mkdirs,
`put ${quoteSftpBatchArgument(localPath)} ${quoteSftpBatchArgument(toSftpRemotePath(stagingPath))}`
],
options
)
),
isSftpRefusalBeforeStaging
)
}
function writeViaRemoteStdin(
target: SshTarget,
remotePath: string,
source: WindowsWriteSource,
options: WindowsWriteOptions
): Promise<void> {
const capabilities = getWindowsRemoteWriteCapabilities(target)
return stageThenPublish(target, remotePath, options, (stagingPath) =>
capabilities.runWithFallback(
'pwsh',
() => writeStdinChunks(target, stagingPath, source, options, 'pwsh.exe'),
() => writeStdinChunks(target, stagingPath, source, options, 'powershell.exe'),
isPwshUnavailableError
)
)
}
/**
* PowerShell 7 takes the whole payload in one exec — measured at 2MB — so only the 5.1 path pays
* for chunking, and only because a bounded write is the most that path can be trusted with.
*/
async function writeStdinChunks(
target: SshTarget,
stagingPath: string,
source: WindowsWriteSource,
options: WindowsWriteOptions,
executable: 'powershell.exe' | 'pwsh.exe'
): Promise<void> {
const chunkBytes =
executable === 'pwsh.exe' ? Math.max(source.totalBytes, 1) : WINDOWS_STDIN_WRITE_CHUNK_BYTES
let offset = 0
// An empty write still has to run: it is what creates the staged file.
do {
const chunk = await source.readChunk(offset, chunkBytes)
if (chunk.length === 0 && offset < source.totalBytes) {
throw new Error(`Source ran short during upload of ${stagingPath}`)
}
await writeOneStdinChunk(
target,
stagingPath,
chunk,
{ ...options, append: offset > 0, exclusive: false },
offset,
executable
)
offset += chunk.length
} while (offset < source.totalBytes)
}
async function writeOneStdinChunk(
target: SshTarget,
stagingPath: string,
chunk: Buffer,
options: WindowsWriteOptions,
offset: number,
executable: 'powershell.exe' | 'pwsh.exe'
): Promise<void> {
throwIfAborted(options.signal)
const channel = spawnSystemSshCommand(
target,
makeWindowsWriteFileCommand(stagingPath, {
append: options.append,
exclusive: options.exclusive,
executable
}),
{ wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) }
)
const closePromise = awaitWithSystemSshAbort(
options.signal,
() => channel.close(),
waitForChannelClose(
channel,
`write ${stagingPath} at offset ${offset}`,
WINDOWS_STDIN_WRITE_TIMEOUT_MS
)
).catch((error: unknown) => {
throw executable === 'powershell.exe' ? explainWindowsPowerShellStdinFailure(error) : error
})
if (!options.signal?.aborted) {
channel.stdin.end(chunk)
}
await closePromise
}
/**
* Names the cause on the one path that can hang, so the failure is not just "timed out".
*
* A user seeing this needs to know it is a host limitation with a host-side remedy, not a network
* fault they should retry into.
*/
export function explainWindowsPowerShellStdinFailure(error: unknown): unknown {
const message = error instanceof Error ? error.message : String(error)
if (!/timed out/i.test(message)) {
return error
}
return new Error(
`${message}\nWindows PowerShell 5.1 can lose a redirected stdin permanently when a read finds it momentarily empty, so this write cannot be made reliable from the client. Enable the sftp subsystem on the host (sshd_config: "Subsystem sftp sftp-server.exe"), or install PowerShell 7, and Orca will use it automatically.`,
{ cause: error instanceof Error ? error : undefined }
)
}
function isPwshUnavailableError(error: unknown): boolean {
const message = error instanceof Error ? error.message : String(error)
// cmd.exe's "not recognized" and sshd's exit 9009 both mean "no pwsh here". A timeout does not:
// that is the stdin defect, and PowerShell 7 does not have it, so it must not be cached as absent.
return /is not recognized as an internal or external command|9009|CommandNotFoundException/i.test(
message
)
}
async function publishStagedWrite(
target: SshTarget,
stagingPath: string,
remotePath: string,
options: WindowsWriteOptions
): Promise<void> {
await runWindowsCommandWithoutStdin(
target,
makeWindowsPublishStagedFileCommand(stagingPath, remotePath, publishMode(options)),
`publish ${remotePath}`,
options
)
}
async function discardStagedWrite(
target: SshTarget,
stagingPath: string,
options: WindowsWriteOptions
): Promise<void> {
try {
await runWindowsCommandWithoutStdin(
target,
makeWindowsDiscardStagedFileCommand(stagingPath),
`discard ${stagingPath}`,
{ ...options, signal: undefined }
)
} catch {
// Housekeeping only. The staging name is unique, so a leftover blocks nothing, and a failure
// here says nothing about whether the abandoned writer is still alive.
}
}
function runWindowsCommandWithoutStdin(
target: SshTarget,
command: string,
label: string,
options: WindowsWriteOptions
): Promise<void> {
const channel = spawnSystemSshCommand(target, command, {
wrapCommand: false,
...getSystemSshBuildArgsFromOperationOptions(options)
})
const closePromise = awaitWithSystemSshAbort(
options.signal,
() => channel.close(),
waitForChannelClose(channel, label, WINDOWS_STDIN_WRITE_TIMEOUT_MS)
)
if (!options.signal?.aborted) {
channel.stdin.end()
}
return closePromise
}
@@ -9,7 +9,7 @@ import { RpcDispatcher } from '../runtime/rpc/dispatcher'
import { browserManager } from '../browser/browser-manager'
import { configureBrowserClientPageAutomationRuntime } from '../browser/browser-client-page-automation-runtime'
import { BrowserClientPageCommandError } from '../browser/browser-client-page-command-failure'
import { startPreGoneProcessMetricsSampling } from '../crash-reporting/process-gone-diagnostics'
import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics'
import { recordProcessGoneCrash } from './main-window-lifecycle-flags'
import { handleGpuChildCrash } from './gpu-lifecycle'
import { isGpuFallbackCrashCandidate } from '../crash-reporting/gpu-crash-fallback-decision'
@@ -130,9 +130,10 @@ export async function initializeReadyRuntimeServices(): Promise<void> {
console.warn('[agent-hooks] failed to reconcile managed hooks on startup:', error)
)
}
// Why: process-gone metrics only see survivors; retain a recent whole-app
// snapshot for comparison in crash reports.
startPreGoneProcessMetricsSampling()
// Why: process-gone metrics only see survivors, and the gone-time host memory
// read lands after the corpse released its pages; both need a live pre-gone
// sample to compare against in crash reports.
startPreGoneCrashSampling()
app.on('child-process-gone', (_event, details) => {
recordProcessGoneCrash('child', details.type, details.reason, details.exitCode ?? null, {
name: details.name,
@@ -0,0 +1,49 @@
import { readFileSync } from 'node:fs'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
/**
* Guards the one line that arms pre-gone crash sampling.
*
* That branch is pure instrumentation, so this line is the whole of its value in
* the shipped app: deleting it left all 691 tests across `src/main/crash-reporting/`
* and `src/main/startup/` green while every crash report silently lost its only
* host reading taken before the dying process returned its pages.
*
* Source-level because that is the property: the sampler is armed once inside the
* ready-phase composition, which has no runtime seam to assert against.
*/
describe('pre-gone crash sampling startup wiring', () => {
// Why normalize: the indent anchors below are `\n`-prefixed, and nothing pins
// src/**/*.ts to LF, so a CRLF Windows checkout would fail them spuriously.
const readSource = (name: string): string =>
readFileSync(join(process.cwd(), 'src/main/startup', name), 'utf8').replace(/\r\n/g, '\n')
const readyRuntimeSource = readSource('main-process-ready-runtime.ts')
const readySource = readSource('main-process-ready.ts')
const READY_ENTRY = 'export async function initializeReadyRuntimeServices('
// Why the entry's body and not the file: the call satisfies a whole-file grep
// just as well from a sibling export nothing calls, which arms nothing.
const readyRuntimeEntryBody = readyRuntimeSource
.slice(readyRuntimeSource.indexOf(READY_ENTRY) + READY_ENTRY.length)
.split('\nexport ')[0]
it('arms the sampler unconditionally inside the function app readiness runs', () => {
expect(readyRuntimeSource).toContain(
"import { startPreGoneCrashSampling } from '../crash-reporting/process-gone-diagnostics'"
)
expect(readyRuntimeSource).toContain(READY_ENTRY)
expect(readyRuntimeEntryBody.split('startPreGoneCrashSampling()').length - 1).toBe(1)
// Why pin the indent: the call also matches as the body of an added
// `if (...)` guard, which keeps every other assertion here true while the
// sampler silently stops arming on most startups.
expect(readyRuntimeEntryBody).toContain('\n startPreGoneCrashSampling()')
// ...and that this really is the function app readiness runs.
expect(readySource).toContain(
"import { initializeReadyRuntimeServices } from './main-process-ready-runtime'"
)
expect(readySource).toContain('\n await initializeReadyRuntimeServices()')
})
})
@@ -24,6 +24,7 @@ export function TerminalWorkspaceDialogs({
saveDialogFile,
saveDialogFileId,
setWindowCloseDialogOpen,
windowCloseDialogKind,
windowCloseDialogOpen
} = controller
return (
@@ -82,10 +83,15 @@ export function TerminalWorkspaceDialogs({
{translate('auto.components.Terminal.2fa9c69ff3', 'Close Window?')}
</DialogTitle>
<DialogDescription className="text-xs">
{translate(
'auto.components.Terminal.7958465754',
'There are local terminals with running processes. Close the window anyway?'
)}
{windowCloseDialogKind === 'unverifiable'
? translate(
'auto.components.Terminal.b7c1f0a934',
'A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?'
)
: translate(
'auto.components.Terminal.7958465754',
'There are terminals with running processes. Close the window anyway?'
)}
</DialogDescription>
</DialogHeader>
<DialogFooter className="gap-2">
@@ -493,6 +493,37 @@ describe('WorkspacePortScanner', () => {
expect(useAppStore.getState().workspacePortScansByKey['environment:env-3:all']).toBeUndefined()
})
// Why: a manual publish (the ports popover) can resolve after the host-set
// change already pruned its key, re-adding it. Per-key writes never delete, so
// that removed host would otherwise hold its ports and a permanent
// unavailable notice until the next host-set change.
it('drops a stale host re-added after pruning on the next poll', async () => {
await act(async () => {
root?.render(<WorkspacePortScanner />)
await flushPromises()
})
const staleKey = 'environment:env-removed:all'
act(() => {
const state = useAppStore.getState()
state.replaceWorkspacePortScans(
{
...state.workspacePortScansByKey,
[staleKey]: { ...emptyScan, unavailableReason: 'gone' }
},
state.workspacePortScan
)
})
expect(useAppStore.getState().workspacePortScansByKey[staleKey]).toBeDefined()
await act(async () => {
vi.advanceTimersByTime(30_000)
await flushPromises()
})
expect(useAppStore.getState().workspacePortScansByKey[staleKey]).toBeUndefined()
})
it('clears ports immediately when the final worktree is removed', async () => {
runtimeEnvironmentCall.mockImplementation(({ method }) => {
if (method === 'workspacePorts.scan') {
@@ -4,10 +4,11 @@ import { getHasAnyWorktreesFromState } from '@/store/selectors'
import { getActiveRuntimeTarget, type RuntimeClientTarget } from '@/runtime/runtime-rpc-client'
import {
mergeWorkspacePortScans,
runtimeTargetForExecutionHostId,
WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY,
scanWorkspacePortsForTarget,
workspacePortScanKeyForTarget
} from '@/lib/workspace-port-actions'
import { runtimeTargetForExecutionHostId } from '@/runtime/runtime-client-target'
import { installWindowVisibilityInterval, isWindowVisible } from '@/lib/window-visibility-interval'
import {
reconcileTransientPortScanFailures,
@@ -41,7 +42,6 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }):
const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan)
const setWorkspacePortScanProjection = useAppStore((s) => s.setWorkspacePortScanProjection)
const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans)
const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey)
const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing)
const inFlightRef = useRef<Promise<void> | null>(null)
const generationRef = useRef(0)
@@ -124,40 +124,40 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }):
const activeTargetKeys = new Set(
allTargets.map((target) => workspacePortScanKeyForTarget(target))
)
const publishedScans = useAppStore.getState().workspacePortScansByKey
const reconciled = reconcileTransientPortScanFailures(
results,
useAppStore.getState().workspacePortScansByKey,
publishedScans,
portScanDebounceRef.current,
WORKSPACE_PORT_SCAN_FAILURE_THRESHOLD,
activeTargetKeys
)
const scansByKey = Object.fromEntries(
Object.entries(useAppStore.getState().workspacePortScansByKey).filter(([key]) =>
activeTargetKeys.has(key)
)
Object.entries(publishedScans).filter(([key]) => activeTargetKeys.has(key))
)
let sourceChanged = false
// Why: a manual publish that lands after a host is pruned re-adds its key,
// and per-key writes never delete. Dropping the inactive keys here is what
// stops a removed host from holding a permanent unavailable notice.
let sourceChanged =
Object.keys(scansByKey).length !== Object.keys(publishedScans).length
for (const { key, result } of reconciled) {
sourceChanged ||= scansByKey[key] !== result
scansByKey[key] = result
setWorkspacePortScanForKey(key, result)
}
const activeScan = scansByKey[scanKey]
const merged = mergeWorkspacePortScans(scansByKey)
const projectionKey =
allTargets.length > 1
? 'all-hosts:all'
? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY
: activeScan
? scanKey
: workspacePortScanKeyForTarget(allTargets[0])
if (sourceChanged || useAppStore.getState().workspacePortScan?.key !== projectionKey) {
setWorkspacePortScanProjection(
merged
? {
key: projectionKey,
result: merged
}
: null
// Why: one store update for the whole poll — a large host set must not
// fan out a notification to every subscriber per host.
replaceWorkspacePortScans(
sourceChanged ? scansByKey : publishedScans,
merged ? { key: projectionKey, result: merged } : null
)
}
}
@@ -177,8 +177,7 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }):
hasWorktrees,
scanKey,
setWorkspacePortScan,
setWorkspacePortScanProjection,
setWorkspacePortScanForKey,
replaceWorkspacePortScans,
setWorkspacePortScanRefreshing
]
)
@@ -215,7 +214,7 @@ export function WorkspacePortScanner({ enabled = true }: { enabled?: boolean }):
: Object.fromEntries(retainedEntries)
const retainedProjection = mergeWorkspacePortScans(retainedScans)
const retainedProjectionKey =
targetKeys.size > 1 ? 'all-hosts:all' : Object.keys(retainedScans)[0]
targetKeys.size > 1 ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : Object.keys(retainedScans)[0]
// Why: unchanged hosts stay visible while the replacement RPC runs; removed
// hosts and the old synthetic aggregate are excluded immediately.
const nextProjection =
@@ -451,25 +451,26 @@ describe('PortsPanel runtime routing', () => {
})
it('returns post-stop refresh failures without throwing', async () => {
const setWorkspacePortScan = vi.fn()
const replaceWorkspacePortScans = vi.fn()
const setWorkspacePortScanRefreshing = vi.fn()
localScan.mockRejectedValueOnce(new Error('scan failed'))
await expect(
refreshWorkspacePortScanAfterStop({
runtimeTarget: { kind: 'local' },
setWorkspacePortScan: setWorkspacePortScan as never,
replaceWorkspacePortScans: replaceWorkspacePortScans as never,
getWorkspacePortScansByKey: () => ({}),
setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never
})
).resolves.toEqual({ ok: false, reason: 'scan failed' })
expect(setWorkspacePortScan).not.toHaveBeenCalled()
expect(replaceWorkspacePortScans).not.toHaveBeenCalled()
expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(1, true)
expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(2, false)
})
it('ignores settled remote post-stop refresh failures after updating state', async () => {
const setWorkspacePortScan = vi.fn()
const replaceWorkspacePortScans = vi.fn()
const setWorkspacePortScanRefreshing = vi.fn()
const firstScan = { ...emptyScan, scannedAt: 2 }
let scanCalls = 0
@@ -500,7 +501,8 @@ describe('PortsPanel runtime routing', () => {
await expect(
refreshWorkspacePortScanAfterStop({
runtimeTarget: { kind: 'environment', environmentId: 'env-1' },
setWorkspacePortScan: setWorkspacePortScan as never,
replaceWorkspacePortScans: replaceWorkspacePortScans as never,
getWorkspacePortScansByKey: () => ({}),
setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never
})
).resolves.toEqual({ ok: true })
@@ -510,18 +512,20 @@ describe('PortsPanel runtime routing', () => {
'workspacePorts.scan',
'workspacePorts.scan'
])
expect(setWorkspacePortScan).toHaveBeenCalledTimes(1)
expect(setWorkspacePortScan).toHaveBeenCalledWith({
key: 'environment:env-1:all',
result: firstScan
})
expect(replaceWorkspacePortScans).toHaveBeenCalledTimes(1)
expect(replaceWorkspacePortScans).toHaveBeenCalledWith(
{ 'environment:env-1:all': firstScan },
{
key: 'environment:env-1:all',
result: firstScan
}
)
expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(1, true)
expect(setWorkspacePortScanRefreshing).toHaveBeenNthCalledWith(2, false)
})
it('preserves an all-host projection after refreshing one host post-stop', async () => {
const setWorkspacePortScan = vi.fn()
const setWorkspacePortScanForKey = vi.fn()
const replaceWorkspacePortScans = vi.fn()
const setWorkspacePortScanRefreshing = vi.fn()
const localPort: WorkspacePort = { ...workspacePort, id: 'local-port', port: 5173 }
const refreshedRemotePort: WorkspacePort = {
@@ -571,23 +575,24 @@ describe('PortsPanel runtime routing', () => {
await expect(
refreshWorkspacePortScanAfterStop({
runtimeTarget: { kind: 'environment', environmentId: 'env-1' },
setWorkspacePortScan: setWorkspacePortScan as never,
setWorkspacePortScanForKey: setWorkspacePortScanForKey as never,
replaceWorkspacePortScans: replaceWorkspacePortScans as never,
getWorkspacePortScansByKey: () => ({ 'local:all': localHostScan }),
setWorkspacePortScanRefreshing: setWorkspacePortScanRefreshing as never
})
).resolves.toEqual({ ok: true })
expect(setWorkspacePortScanForKey).toHaveBeenCalledWith('environment:env-1:all', remoteHostScan)
expect(setWorkspacePortScan).toHaveBeenLastCalledWith({
key: 'all-hosts:all',
result: expect.objectContaining({
ports: expect.arrayContaining([
expect.objectContaining({ port: 5173 }),
expect.objectContaining({ port: 3000 })
])
})
})
expect(replaceWorkspacePortScans).toHaveBeenLastCalledWith(
{ 'local:all': localHostScan, 'environment:env-1:all': remoteHostScan },
{
key: 'all-hosts:all',
result: expect.objectContaining({
ports: expect.arrayContaining([
expect.objectContaining({ port: 5173 }),
expect.objectContaining({ port: 3000 })
])
})
}
)
expect(scanCalls).toBe(2)
})
@@ -0,0 +1,30 @@
import { describe, expect, it } from 'vitest'
import { shouldShowLocalWorkspacePortSections } from './local-workspace-port-sections'
const empty = { activePorts: [], otherWorkspacePorts: [], externalPorts: [] }
describe('shouldShowLocalWorkspacePortSections', () => {
it('shows the sections whenever the scan succeeded', () => {
expect(shouldShowLocalWorkspacePortSections(null, empty)).toBe(true)
expect(shouldShowLocalWorkspacePortSections({}, empty)).toBe(true)
})
// Why: a failed scan keeps the host's last-good ports, and the status bar
// still counts and lists them — hiding the sections here would strip the
// stop and open actions for ports the user can still see elsewhere.
it.each([
['activePorts', { ...empty, activePorts: [{}] }],
['otherWorkspacePorts', { ...empty, otherWorkspacePorts: [{}] }],
['externalPorts', { ...empty, externalPorts: [{}] }]
])('keeps the sections when a failed scan retained %s', (_section, sections) => {
expect(shouldShowLocalWorkspacePortSections({ unavailableReason: 'dropped' }, sections)).toBe(
true
)
})
it('lets the notice stand alone when a failed scan has nothing left to list', () => {
expect(shouldShowLocalWorkspacePortSections({ unavailableReason: 'dropped' }, empty)).toBe(
false
)
})
})
@@ -35,6 +35,26 @@ export function getLocalWorkspacePortSections(
}
}
/**
* Whether the panel still renders its port sections under a failure notice.
* Why: a failed scan retains the host's last-good ports, so hiding every
* section would drop the stop and open actions for ports the status bar still
* counts and lists.
*/
export function shouldShowLocalWorkspacePortSections(
scan: { unavailableReason?: string } | null | undefined,
sections: { activePorts: unknown[]; otherWorkspacePorts: unknown[]; externalPorts: unknown[] }
): boolean {
if (!scan?.unavailableReason) {
return true
}
return (
sections.activePorts.length > 0 ||
sections.otherWorkspacePorts.length > 0 ||
sections.externalPorts.length > 0
)
}
function workspacePortAsExternal(port: WorkspacePort & { kind: 'workspace' }): WorkspacePort {
return {
id: port.id,
@@ -4,11 +4,11 @@ import { toast } from 'sonner'
import { useAppStore } from '@/store'
import { useActiveWorktree, useRepoById } from '@/store/selectors'
import { cn } from '@/lib/utils'
import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client'
import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner'
import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target'
import {
killWorkspacePortForTarget,
openWorkspacePortInBrowser,
publishWorkspacePortScanForHost,
refreshWorkspacePortScanAfterStop,
resolvePortOpenInOrcaBrowser,
scanWorkspacePortsForTarget,
@@ -19,10 +19,14 @@ import { Button } from '@/components/ui/button'
import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip'
import type { WorkspacePort } from '../../../../shared/workspace-ports'
import { translate } from '@/i18n/i18n'
import { getLocalWorkspacePortSections } from './local-workspace-port-sections'
import {
getLocalWorkspacePortSections,
shouldShowLocalWorkspacePortSections
} from './local-workspace-port-sections'
import { LocalPortSection } from './local-port-section'
import { LocalPortDetailsDialog } from './local-port-details-dialog'
/** Right-sidebar Ports panel scoped to the active workspace's owner host. */
export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }): React.JSX.Element {
const activeWorktree = useActiveWorktree()
const activeRepo = useRepoById(activeWorktree?.repoId ?? null)
@@ -31,8 +35,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle)
const scansByKey = useAppStore((s) => s.workspacePortScansByKey)
const refreshing = useAppStore((s) => s.workspacePortScanRefreshing)
const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan)
const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey)
const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans)
const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing)
const [detailsPort, setDetailsPort] = useState<WorkspacePort | null>(null)
const [collapsedSections, setCollapsedSections] = useState<Record<string, boolean>>({
@@ -40,26 +43,24 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
external: true
})
const runtimeTarget = useMemo(() => {
const activeRuntimeEnvironmentId = getRuntimeEnvironmentIdForWorktree(
useAppStore.getState(),
activeWorktree?.id
)
// Why: the Ports panel acts on the active workspace; use that workspace's
// host owner even if the sidebar is focused elsewhere.
return getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId })
}, [activeWorktree?.id, settings])
const scanKey = `${workspacePortRuntimeTargetKey(runtimeTarget)}:all`
// Why: the Ports panel acts on the active workspace; use that workspace's
// host owner even if the sidebar is focused elsewhere.
const runtimeTarget = useWorktreeRuntimeTarget(activeWorktree?.id)
const scanKey = runtimeTarget ? `${workspacePortRuntimeTargetKey(runtimeTarget)}:all` : null
const refresh = useCallback(() => {
if (!activeRepo) {
if (!activeRepo || !runtimeTarget || !scanKey) {
return Promise.resolve()
}
setWorkspacePortScanRefreshing(true)
const promise = scanWorkspacePortsForTarget(runtimeTarget)
.then((nextScan) => {
setWorkspacePortScanForKey(scanKey, nextScan)
setWorkspacePortScan({ key: scanKey, result: nextScan })
publishWorkspacePortScanForHost({
scanKey,
scan: nextScan,
replaceWorkspacePortScans,
getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey
})
})
.catch((error) => {
const message = error instanceof Error ? error.message : String(error)
@@ -86,14 +87,13 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
activeRepo,
runtimeTarget,
scanKey,
setWorkspacePortScan,
setWorkspacePortScanForKey,
replaceWorkspacePortScans,
setWorkspacePortScanRefreshing
])
// Why: WorkspacePortScanner already owns the 30s all-worktree poll. The
// panel scopes that shared result instead of starting a second scan loop.
const displayScan = isVisible ? (scansByKey[scanKey] ?? null) : null
const displayScan = isVisible && scanKey ? (scansByKey[scanKey] ?? null) : null
const toggleSection = useCallback((sectionId: string) => {
setCollapsedSections((current) => ({ ...current, [sectionId]: !current[sectionId] }))
@@ -122,8 +122,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
)
const refreshResult = await refreshWorkspacePortScanAfterStop({
runtimeTarget,
setWorkspacePortScan,
setWorkspacePortScanForKey,
replaceWorkspacePortScans,
getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey,
setWorkspacePortScanRefreshing
})
@@ -139,13 +138,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
)
}
},
[
activeRepo,
runtimeTarget,
setWorkspacePortScan,
setWorkspacePortScanForKey,
setWorkspacePortScanRefreshing
]
[activeRepo, runtimeTarget, replaceWorkspacePortScans, setWorkspacePortScanRefreshing]
)
const handleOpenPortInBrowser = useCallback(
@@ -181,6 +174,12 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
[activeRepo?.id, activeWorktree?.id, displayScan]
)
const showPortSections = shouldShowLocalWorkspacePortSections(displayScan, {
activePorts,
otherWorkspacePorts,
externalPorts
})
if (!activeRepo) {
return (
<div className="flex flex-col items-center justify-center h-full px-4 text-center text-muted-foreground">
@@ -209,7 +208,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
size="icon-xs"
className="text-muted-foreground hover:text-foreground"
onClick={() => void refresh()}
disabled={refreshing}
disabled={refreshing || !runtimeTarget}
aria-label={translate(
'auto.components.right.sidebar.PortsPanel.7822e3edc6',
'Refresh Ports'
@@ -237,7 +236,7 @@ export function LocalWorkspacePortsPanel({ isVisible }: { isVisible: boolean }):
</div>
)}
{!displayScan?.unavailableReason && (
{showPortSections && (
<>
<LocalPortSection
id="active"
@@ -16,7 +16,7 @@ const openModal = vi.fn()
const openTaskPage = vi.fn()
const updateWorktreeMeta = vi.fn()
const recordFeatureInteraction = vi.fn()
const setWorkspacePortScan = vi.fn()
const replaceWorkspacePortScans = vi.fn()
const setWorkspacePortScanRefreshing = vi.fn()
const cacheTimerMocks = vi.hoisted(() => ({
usePromptCacheCountdownStartedAt: vi.fn()
@@ -52,7 +52,7 @@ vi.mock('@/store', () => ({
recordFeatureInteraction,
remoteBranchConflictByWorktreeId: {},
setRemoteBrowserPageHandle: vi.fn(),
setWorkspacePortScan,
replaceWorkspacePortScans,
setWorkspacePortScanRefreshing,
settings,
sshConnectionStates: new Map(),
@@ -13,7 +13,7 @@ import type { WorkspacePortScanResult } from '../../../../shared/workspace-ports
const fetchHostedReviewForBranch = vi.fn()
const fetchIssue = vi.fn()
const fetchLinearIssue = vi.fn()
const setWorkspacePortScan = vi.fn()
const replaceWorkspacePortScans = vi.fn()
const setWorkspacePortScanRefreshing = vi.fn()
const cacheTimerMocks = vi.hoisted(() => ({
usePromptCacheCountdownStartedAt: vi.fn()
@@ -43,7 +43,7 @@ vi.mock('@/store', () => ({
recordFeatureInteraction: vi.fn(),
remoteBranchConflictByWorktreeId: {},
setRemoteBrowserPageHandle: vi.fn(),
setWorkspacePortScan,
replaceWorkspacePortScans,
setWorkspacePortScanRefreshing,
settings,
sshConnectionStates: new Map(),
@@ -8,7 +8,7 @@ vi.mock('@/store', () => ({
selector({
createBrowserTab: vi.fn(),
setRemoteBrowserPageHandle: vi.fn(),
setWorkspacePortScan: vi.fn(),
replaceWorkspacePortScans: vi.fn(),
setWorkspacePortScanRefreshing: vi.fn(),
settings: null
})
@@ -1,11 +1,10 @@
import React, { useCallback, useMemo } from 'react'
import React, { useCallback } from 'react'
import { Plug, Copy, ExternalLink, Trash2 } from 'lucide-react'
import { toast } from 'sonner'
import { useAppStore } from '@/store'
import { Button } from '@/components/ui/button'
import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip'
import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client'
import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner'
import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target'
import {
canStopWorkspacePort,
getPortOpenBrowserTooltipLabel,
@@ -97,21 +96,18 @@ function PortAction({
)
}
/** One port row on a sidebar worktree card, with open/copy/stop actions on its owner host. */
function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element {
const settings = useAppStore((s) => s.settings)
const localhostLabelRoute = useLocalhostLabelRouteForPort(port)
const runtimeEnvironmentId = useAppStore((s) =>
getRuntimeEnvironmentIdForWorktree(s, port.kind === 'workspace' ? port.owner.worktreeId : null)
)
const createBrowserTab = useAppStore((s) => s.createBrowserTab)
const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle)
const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan)
const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey)
const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans)
const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing)
const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction)
const runtimeTarget = useMemo(
() => getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId: runtimeEnvironmentId }),
[runtimeEnvironmentId, settings]
const runtimeTarget = useWorktreeRuntimeTarget(
port.kind === 'workspace' ? port.owner.worktreeId : null
)
const processLabel = port.processName ?? (port.pid ? `PID ${port.pid}` : 'Unknown process')
const address = addressForPort(port)
@@ -203,8 +199,7 @@ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element {
)
const refreshResult = await refreshWorkspacePortScanAfterStop({
runtimeTarget,
setWorkspacePortScan,
setWorkspacePortScanForKey,
replaceWorkspacePortScans,
getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey,
setWorkspacePortScanRefreshing
})
@@ -226,8 +221,7 @@ function WorktreePortRow({ port }: { port: WorkspacePort }): React.JSX.Element {
port,
recordFeatureInteraction,
runtimeTarget,
setWorkspacePortScan,
setWorkspacePortScanForKey,
replaceWorkspacePortScans,
setWorkspacePortScanRefreshing
]
)
@@ -0,0 +1,407 @@
// @vitest-environment happy-dom
import React, { act } from 'react'
import { createRoot, type Root } from 'react-dom/client'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type { WorkspacePort, WorkspacePortScanResult } from '../../../../shared/workspace-ports'
const { popoverHandle, runWorkspacePortScanForTargetMock, storeState } = vi.hoisted(() => {
const storeState = {
settings: { activeRuntimeEnvironmentId: null as string | null },
activeWorktreeId: 'runtime-repo::/srv/app',
workspacePortScan: null as { key: string; result: WorkspacePortScanResult } | null,
workspacePortScansByKey: {} as Record<string, WorkspacePortScanResult>,
workspacePortScanRefreshing: false,
runtimeEnvironments: [] as { id: string; name: string }[],
recordFeatureInteraction: vi.fn(),
replaceWorkspacePortScans:
vi.fn<
(
scansByKey: Record<string, WorkspacePortScanResult>,
projection: { key: string; result: WorkspacePortScanResult } | null
) => void
>()
}
// Why: the real store writes back. A bare spy lets a publish and the notice
// that reads it drift onto different scan keys with every assertion green.
storeState.replaceWorkspacePortScans.mockImplementation((scansByKey, projection) => {
storeState.workspacePortScansByKey = scansByKey
storeState.workspacePortScan = projection
})
return {
popoverHandle: { onOpenChange: null as ((open: boolean) => void) | null },
runWorkspacePortScanForTargetMock: vi.fn(),
storeState
}
})
vi.mock('@/store', () => {
const useAppStore = Object.assign(
(selector: (state: typeof storeState) => unknown) => selector(storeState),
{ getState: () => storeState }
)
return { useAppStore }
})
vi.mock('@/lib/worktree-runtime-owner', () => ({
getExecutionHostIdForWorktree: (_state: unknown, worktreeId: string | null | undefined) => {
if (worktreeId === 'runtime-repo::/srv/app') {
return 'runtime:env-1'
}
if (worktreeId === 'ssh-repo::/srv/app') {
return 'ssh:server-1'
}
return 'local'
}
}))
vi.mock('@/runtime/runtime-rpc-client', async () => {
const actual = await import('@/runtime/runtime-client-target')
return {
getActiveRuntimeTarget: actual.getActiveRuntimeTarget,
callRuntimeRpc: vi.fn(),
assertRuntimeEnvironmentCapability: vi.fn(),
RuntimeRpcCallError: class RuntimeRpcCallError extends Error {
code?: string
}
}
})
vi.mock('@/lib/workspace-port-scan-client', () => ({
runWorkspacePortScanForTarget: runWorkspacePortScanForTargetMock
}))
vi.mock('@/lib/worktree-activation', () => ({
activateAndRevealWorktree: vi.fn()
}))
vi.mock('@/components/ui/popover', () => ({
Popover: ({
children,
onOpenChange
}: {
children: React.ReactNode
onOpenChange: (open: boolean) => void
}) => {
popoverHandle.onOpenChange = onOpenChange
return <>{children}</>
},
PopoverContent: ({ children }: { children: React.ReactNode }) => <>{children}</>,
PopoverTrigger: ({ children }: { children: React.ReactNode }) => <>{children}</>
}))
vi.mock('@/components/ui/tooltip', () => ({
Tooltip: ({ children }: { children: React.ReactNode }) => <>{children}</>,
TooltipContent: ({ children }: { children: React.ReactNode }) => <>{children}</>,
TooltipTrigger: ({ children }: { children: React.ReactNode }) => <>{children}</>
}))
vi.mock('@/components/SelectedTextCopyMenu', () => ({
SelectedTextCopyMenu: ({ children }: { children: React.ReactNode }) => <>{children}</>
}))
vi.mock('./ports-status-popover-rows', () => ({
PortRow: () => <div data-testid="port-row" />,
WorkspaceGroupRows: () => <div data-testid="workspace-group-rows" />
}))
vi.mock('@/i18n/i18n', () => ({
translate: (_key: string, fallback: string, options?: Record<string, unknown>) =>
options
? fallback.replace(/{{(\w+)}}/g, (_match, name: string) => String(options[name] ?? ''))
: fallback
}))
import { PortsStatusSegment } from './PortsStatusSegment'
function workspacePort(overrides: Partial<WorkspacePort> & { port: number; id: string }) {
return {
bindHost: '0.0.0.0',
connectHost: '127.0.0.1',
port: overrides.port,
id: overrides.id,
pid: 4321,
processName: 'node',
protocol: 'http' as const,
kind: 'workspace' as const,
owner: {
worktreeId: 'runtime-repo::/srv/app',
repoId: 'runtime-repo',
displayName: 'runtime app',
path: '/srv/app',
confidence: 'cwd' as const
}
}
}
const localHostScan: WorkspacePortScanResult = {
platform: 'linux',
scannedAt: 10,
ports: [workspacePort({ id: 'local-5173', port: 5173 })]
}
const remoteHostScan: WorkspacePortScanResult = {
platform: 'linux',
scannedAt: 20,
ports: [workspacePort({ id: 'remote-3000', port: 3000 })]
}
describe('PortsStatusSegment popover host routing', () => {
let container: HTMLDivElement
let root: Root
beforeEach(() => {
popoverHandle.onOpenChange = null
storeState.settings = { activeRuntimeEnvironmentId: null }
storeState.activeWorktreeId = 'runtime-repo::/srv/app'
storeState.workspacePortScan = null
storeState.workspacePortScansByKey = { 'local:all': localHostScan }
storeState.runtimeEnvironments = [{ id: 'env-1', name: 'linux-box' }]
storeState.recordFeatureInteraction.mockClear()
storeState.replaceWorkspacePortScans.mockClear()
runWorkspacePortScanForTargetMock.mockReset()
runWorkspacePortScanForTargetMock.mockResolvedValue(remoteHostScan)
container = document.createElement('div')
document.body.appendChild(container)
root = createRoot(container)
act(() => {
root.render(<PortsStatusSegment iconOnly={false} />)
})
})
afterEach(() => {
act(() => {
root.unmount()
})
container.remove()
})
async function openPopover(): Promise<void> {
await act(async () => {
popoverHandle.onOpenChange?.(true)
await Promise.resolve()
await Promise.resolve()
})
}
it("scans the active workspace's host, not the globally focused runtime", async () => {
await openPopover()
expect(runWorkspacePortScanForTargetMock).toHaveBeenCalledWith(
{ kind: 'environment', environmentId: 'env-1' },
undefined
)
expect(storeState.replaceWorkspacePortScans).toHaveBeenCalledTimes(1)
expect(storeState.workspacePortScansByKey['environment:env-1:all']).toBe(remoteHostScan)
})
it('keeps other hosts in the projection instead of overwriting it with one host', async () => {
await openPopover()
const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [
Record<string, WorkspacePortScanResult>,
{ key: string; result: WorkspacePortScanResult }
]
expect(projection).toEqual({
key: 'all-hosts:all',
result: expect.objectContaining({
ports: expect.arrayContaining([
expect.objectContaining({ port: 5173 }),
expect.objectContaining({ port: 3000 })
])
})
})
})
it('publishes a failed scan under its own host without dropping other hosts', async () => {
runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed'))
await openPopover()
const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [
Record<string, WorkspacePortScanResult>,
{ key: string; result: WorkspacePortScanResult }
]
expect(projection.key).toBe('all-hosts:all')
expect(projection.result.ports).toEqual([expect.objectContaining({ port: 5173 })])
expect(storeState.workspacePortScansByKey['environment:env-1:all']).toEqual(
expect.objectContaining({ unavailableReason: 'remote scan failed' })
)
})
it('keeps the failed host last-good ports while naming the failure', async () => {
storeState.workspacePortScansByKey = {
'local:all': localHostScan,
'environment:env-1:all': remoteHostScan
}
runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed'))
await openPopover()
// Why: one dropped scan must not clear the host's ports the way the
// background poll's debounce does not — the notice names the failure
// while the projection keeps serving the last-good rows.
const failed = storeState.workspacePortScansByKey['environment:env-1:all']
expect(failed.unavailableReason).toBe('remote scan failed')
expect(failed.platform).toBe('linux')
expect(failed.ports).toEqual([expect.objectContaining({ port: 3000 })])
const [, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [
Record<string, WorkspacePortScanResult>,
{ key: string; result: WorkspacePortScanResult }
]
expect(projection.key).toBe('all-hosts:all')
expect(projection.result.ports.map((port) => port.port).sort()).toEqual([3000, 5173])
})
// Why: separate tests already cover "the failure is stored" and "a stored
// failure renders". Only this one proves both halves name the same scan key.
it('surfaces the host it just failed to scan on the next render', async () => {
runWorkspacePortScanForTargetMock.mockRejectedValueOnce(new Error('remote scan failed'))
await openPopover()
act(() => {
root.render(<PortsStatusSegment iconOnly={false} />)
})
expect(container.textContent).toContain(
'Port scan unavailable on linux-box: remote scan failed'
)
})
it('names the host whose scan failed while another host still reports ports', () => {
act(() => {
root.unmount()
})
storeState.workspacePortScansByKey = {
'local:all': localHostScan,
'environment:env-1:all': {
platform: 'linux',
scannedAt: 30,
ports: [],
unavailableReason: 'Remote connection dropped'
}
}
storeState.workspacePortScan = { key: 'all-hosts:all', result: localHostScan }
root = createRoot(container)
act(() => {
root.render(<PortsStatusSegment iconOnly={false} />)
})
expect(container.textContent).toContain(
'Port scan unavailable on linux-box: Remote connection dropped'
)
// The notice sits above the list rather than replacing it: a reachable
// host's count still renders.
expect(container.textContent).toContain('1 workspace')
})
// Why: a failed scan keeps the host's last-good ports, and the badge and
// header count them. Replacing the list with the notice left the popover
// claiming N ports over an empty body.
it('keeps the list under the notice when a failed scan retained its ports', () => {
act(() => {
root.unmount()
})
storeState.activeWorktreeId = 'local-repo::/home/dev/app'
const retained: WorkspacePortScanResult = {
...localHostScan,
unavailableReason: 'lsof is unavailable'
}
storeState.workspacePortScansByKey = { 'local:all': retained }
storeState.workspacePortScan = { key: 'local:all', result: retained }
root = createRoot(container)
act(() => {
root.render(<PortsStatusSegment iconOnly={false} />)
})
expect(container.textContent).toContain('1 workspace · 0 external')
expect(container.querySelectorAll('[data-testid="workspace-group-rows"]')).toHaveLength(1)
expect(container.textContent).toContain('Port scan unavailable on Local Linux')
})
// Why: total loss of contact is where naming the host matters most, and the
// merged projection can only offer platform 'unknown' and raw scan keys.
it('names every host when all of them failed with nothing left to list', () => {
act(() => {
root.unmount()
})
const merged: WorkspacePortScanResult = {
platform: 'unknown',
scannedAt: 30,
ports: [],
unavailableReason: 'local:all: lsof is unavailable; environment:env-1:all: dropped'
}
storeState.workspacePortScansByKey = {
'local:all': {
platform: 'darwin',
scannedAt: 30,
ports: [],
unavailableReason: 'lsof is unavailable'
},
'environment:env-1:all': {
platform: 'linux',
scannedAt: 30,
ports: [],
unavailableReason: 'dropped'
}
}
storeState.workspacePortScan = { key: 'all-hosts:all', result: merged }
root = createRoot(container)
act(() => {
root.render(<PortsStatusSegment iconOnly={false} />)
})
// Local label comes from the scan's own platform, not the renderer's
// userAgent — a paired web client is not the Orca host.
expect(container.textContent).toContain(
'Port scan unavailable on Local Mac: lsof is unavailable'
)
expect(container.textContent).toContain('Port scan unavailable on linux-box: dropped')
expect(container.textContent).not.toContain('unavailable on unknown')
expect(container.textContent).not.toContain('environment:env-1:all:')
// The notice takes over the body only when there is nothing left to list.
expect(container.querySelectorAll('[data-testid="workspace-group-rows"]')).toHaveLength(0)
expect(container.textContent).not.toContain('No workspace ports detected')
})
it('stays on the local host when the active workspace has no runtime owner', async () => {
act(() => {
root.unmount()
})
storeState.activeWorktreeId = 'local-repo::/home/dev/app'
root = createRoot(container)
act(() => {
root.render(<PortsStatusSegment iconOnly={false} />)
})
await openPopover()
expect(runWorkspacePortScanForTargetMock).toHaveBeenCalledWith({ kind: 'local' }, undefined)
const [nextScans, projection] = storeState.replaceWorkspacePortScans.mock.calls.at(-1) as [
Record<string, WorkspacePortScanResult>,
{ key: string; result: WorkspacePortScanResult }
]
expect(nextScans['local:all']).toBe(remoteHostScan)
expect(projection).toEqual({
key: 'local:all',
result: remoteHostScan
})
})
it('does not substitute the local host for a direct-SSH workspace', async () => {
act(() => {
root.unmount()
})
storeState.activeWorktreeId = 'ssh-repo::/srv/app'
root = createRoot(container)
act(() => {
root.render(<PortsStatusSegment iconOnly={false} />)
})
await openPopover()
expect(runWorkspacePortScanForTargetMock).not.toHaveBeenCalled()
expect(storeState.replaceWorkspacePortScans).not.toHaveBeenCalled()
expect(storeState.recordFeatureInteraction).toHaveBeenCalledWith('ports')
})
})
@@ -3,39 +3,83 @@ import { Plug, ChevronDown, ChevronRight, LoaderCircle } from 'lucide-react'
import { Popover, PopoverContent, PopoverTrigger } from '@/components/ui/popover'
import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip'
import { useAppStore } from '@/store'
import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client'
import {
publishWorkspacePortScanForHost,
scanWorkspacePortsForTarget,
workspacePortScanKeyForTarget
} from '@/lib/workspace-port-actions'
import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target'
import {
getUnavailableWorkspacePortHosts,
type WorkspacePortHostRef
} from '@/lib/workspace-port-host-availability'
import { getLocalExecutionHostLabel } from '../../../../shared/execution-host'
import { getExternalWorkspacePorts, getWorkspacePortGroups } from '@/lib/workspace-port-groups'
import { SelectedTextCopyMenu } from '@/components/SelectedTextCopyMenu'
import { STATUS_BAR_CONTEXT_MENU_EXEMPT_PROPS } from './status-bar-context-menu-policy'
import { PortRow, WorkspaceGroupRows } from './ports-status-popover-rows'
import { translate } from '@/i18n/i18n'
import type { WorkspacePortScanResult } from '../../../../shared/workspace-ports'
type PortsStatusSegmentProps = {
compact?: boolean
iconOnly: boolean
}
/** Status-bar plug icon with the workspace port count and a per-host ports popover. */
export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React.JSX.Element {
const settings = useAppStore((s) => s.settings)
const scan = useAppStore((s) => s.workspacePortScan?.result ?? null)
const refreshing = useAppStore((s) => s.workspacePortScanRefreshing)
const activeWorktreeId = useAppStore((s) => s.activeWorktreeId)
const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan)
const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey)
const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans)
const scansByKey = useAppStore((s) => s.workspacePortScansByKey)
const runtimeEnvironments = useAppStore((s) => s.runtimeEnvironments)
const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction)
const [open, setOpen] = useState(false)
const [externalOpen, setExternalOpen] = useState(false)
const runtimeTarget = useMemo(() => getActiveRuntimeTarget(settings), [settings])
const scanKey = workspacePortScanKeyForTarget(runtimeTarget)
const runtimeTarget = useWorktreeRuntimeTarget(activeWorktreeId)
const scanKey = runtimeTarget ? workspacePortScanKeyForTarget(runtimeTarget) : null
const workspaceGroups = useMemo(() => getWorkspacePortGroups(scan), [scan])
const externalPorts = useMemo(() => getExternalWorkspacePorts(scan), [scan])
const unavailableHosts = useMemo(() => getUnavailableWorkspacePortHosts(scansByKey), [scansByKey])
const hostLabel = useCallback(
(host: WorkspacePortHostRef, hostScanKey: string, platform: NodeJS.Platform | null) => {
if (host.kind === 'local') {
// Why: a paired web client's own userAgent is not the Orca host's
// platform, so name the machine the scan actually ran on.
return getLocalExecutionHostLabel(platform)
}
if (host.kind === 'unknown') {
return hostScanKey
}
return (
runtimeEnvironments.find((environment) => environment.id === host.environmentId)?.name ??
host.environmentId
)
},
[runtimeEnvironments]
)
const workspacePortCount = workspaceGroups.reduce((count, group) => count + group.ports.length, 0)
const totalCount = workspacePortCount + externalPorts.length
const unavailableNotices = useMemo<PortScanUnavailableNotice[]>(() => {
if (unavailableHosts.length > 0) {
return unavailableHosts.map((entry) => ({
id: entry.scanKey,
host: hostLabel(entry.host, entry.scanKey, entry.platform),
reason: entry.reason
}))
}
// Why: a projection published without per-host scans has no host to name.
return scan?.unavailableReason
? [{ id: 'projection', host: scan.platform, reason: scan.unavailableReason }]
: []
}, [hostLabel, scan?.platform, scan?.unavailableReason, unavailableHosts])
// Why: a failed scan keeps the host's last-good ports, and those ports are
// counted in the badge and header — replacing the list with the notice would
// leave the popover claiming N ports over an empty body. Only take over the
// body when there is genuinely nothing left to list.
const noticeReplacesList = Boolean(scan?.unavailableReason) && totalCount === 0
const handleOpenChange = useCallback(
(nextOpen: boolean) => {
setOpen(nextOpen)
@@ -43,33 +87,36 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React
return
}
recordFeatureInteraction('ports')
if (!runtimeTarget || !scanKey) {
return
}
// Why: the 30s background poll is intentionally quiet; opening the
// popover should still collapse that stale window without flashing icons.
void scanWorkspacePortsForTarget(runtimeTarget)
.then((result) => {
setWorkspacePortScanForKey(scanKey, result)
setWorkspacePortScan({ key: scanKey, result })
const publish = (result: WorkspacePortScanResult): void => {
publishWorkspacePortScanForHost({
scanKey,
scan: result,
replaceWorkspacePortScans,
getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey
})
}
void scanWorkspacePortsForTarget(runtimeTarget)
.then(publish)
.catch((error) => {
const message = error instanceof Error ? error.message : String(error)
setWorkspacePortScan({
key: scanKey,
result: {
platform: 'unknown',
scannedAt: Date.now(),
ports: [],
unavailableReason: message || 'Workspace port scan failed.'
}
// Why: one dropped scan must not clear the host's last-good ports the
// way the background poll's debounce does not; the failure is still
// recorded so the host is named by the unavailable notice below.
const previous = useAppStore.getState().workspacePortScansByKey[scanKey]
publish({
platform: previous?.platform ?? 'unknown',
scannedAt: Date.now(),
ports: previous?.ports ?? [],
unavailableReason: message || 'Workspace port scan failed.'
})
})
},
[
recordFeatureInteraction,
runtimeTarget,
scanKey,
setWorkspacePortScan,
setWorkspacePortScanForKey
]
[recordFeatureInteraction, runtimeTarget, scanKey, replaceWorkspacePortScans]
)
return (
@@ -153,14 +200,18 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React
</span>
</div>
{scan?.unavailableReason ? (
<div className="px-3 py-3 text-xs text-muted-foreground">
{translate(
'auto.components.status.bar.PortsStatusSegment.95495019ed',
'Port scan unavailable on {{value0}}: {{value1}}',
{ value0: scan.platform, value1: scan.unavailableReason }
)}
</div>
{unavailableNotices.length > 0 && !noticeReplacesList && (
<PortScanUnavailableNotices
notices={unavailableNotices}
className="border-b border-border/40 px-3 py-1.5 text-[11px] text-muted-foreground"
/>
)}
{noticeReplacesList ? (
<PortScanUnavailableNotices
notices={unavailableNotices}
className="px-3 py-3 text-xs text-muted-foreground"
/>
) : (
<div className="max-h-[28rem] overflow-y-auto scrollbar-sleek">
{workspaceGroups.length > 0 ? (
@@ -237,3 +288,27 @@ export function PortsStatusSegment({ iconOnly }: PortsStatusSegmentProps): React
</Popover>
)
}
type PortScanUnavailableNotice = { id: string; host: string; reason: string }
function PortScanUnavailableNotices({
notices,
className
}: {
notices: PortScanUnavailableNotice[]
className: string
}): React.JSX.Element {
return (
<div className={className}>
{notices.map((notice) => (
<div key={notice.id} className="truncate">
{translate(
'auto.components.status.bar.PortsStatusSegment.95495019ed',
'Port scan unavailable on {{value0}}: {{value1}}',
{ value0: notice.host, value1: notice.reason }
)}
</div>
))}
</div>
)
}
@@ -17,8 +17,7 @@ const {
settings: { openLinksInApp: true },
createBrowserTab: vi.fn(),
setRemoteBrowserPageHandle: vi.fn(),
setWorkspacePortScan: vi.fn(),
setWorkspacePortScanForKey: vi.fn(),
replaceWorkspacePortScans: vi.fn(),
setWorkspacePortScanRefreshing: vi.fn(),
recordFeatureInteraction: vi.fn(),
workspacePortScansByKey: {}
@@ -46,7 +45,7 @@ vi.mock('@/lib/worktree-activation', () => ({
}))
vi.mock('@/lib/worktree-runtime-owner', () => ({
getRuntimeEnvironmentIdForWorktree: () => null
getExecutionHostIdForWorktree: () => 'local'
}))
vi.mock('@/runtime/runtime-rpc-client', () => ({
@@ -1,4 +1,4 @@
import React, { useCallback, useMemo } from 'react'
import React, { useCallback } from 'react'
import { Copy, ExternalLink, FolderOpen, Trash2 } from 'lucide-react'
import { toast } from 'sonner'
import { Button } from '@/components/ui/button'
@@ -15,9 +15,8 @@ import {
} from '@/lib/workspace-port-actions'
import type { WorkspacePortGroup } from '@/lib/workspace-port-groups'
import { useLocalhostLabelRouteForPort } from '@/lib/workspace-port-localhost-label-selector'
import { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client'
import { useWorktreeRuntimeTarget } from '@/runtime/use-worktree-runtime-target'
import { useAppStore } from '@/store'
import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner'
import type { WorkspacePort } from '../../../../shared/workspace-ports'
import { translate } from '@/i18n/i18n'
@@ -67,6 +66,7 @@ function PortAction({
)
}
/** One port row in the status-bar popover, with open/copy/stop actions on its owner host. */
export function PortRow({
port,
activeWorktreeId,
@@ -78,21 +78,13 @@ export function PortRow({
}): React.JSX.Element {
const settings = useAppStore((s) => s.settings)
const localhostLabelRoute = useLocalhostLabelRouteForPort(port)
const runtimeEnvironmentId = useAppStore((s) =>
getRuntimeEnvironmentIdForWorktree(
s,
port.kind === 'workspace' ? port.owner.worktreeId : activeWorktreeId
)
)
const createBrowserTab = useAppStore((s) => s.createBrowserTab)
const setRemoteBrowserPageHandle = useAppStore((s) => s.setRemoteBrowserPageHandle)
const setWorkspacePortScan = useAppStore((s) => s.setWorkspacePortScan)
const setWorkspacePortScanForKey = useAppStore((s) => s.setWorkspacePortScanForKey)
const replaceWorkspacePortScans = useAppStore((s) => s.replaceWorkspacePortScans)
const setWorkspacePortScanRefreshing = useAppStore((s) => s.setWorkspacePortScanRefreshing)
const recordFeatureInteraction = useAppStore((s) => s.recordFeatureInteraction)
const runtimeTarget = useMemo(
() => getActiveRuntimeTarget({ ...settings, activeRuntimeEnvironmentId: runtimeEnvironmentId }),
[runtimeEnvironmentId, settings]
const runtimeTarget = useWorktreeRuntimeTarget(
port.kind === 'workspace' ? port.owner.worktreeId : activeWorktreeId
)
const processLabel = port.processName ?? (port.pid ? `PID ${port.pid}` : 'Unknown process')
const canStop = canStopWorkspacePort(port)
@@ -187,8 +179,7 @@ export function PortRow({
)
const refreshResult = await refreshWorkspacePortScanAfterStop({
runtimeTarget,
setWorkspacePortScan,
setWorkspacePortScanForKey,
replaceWorkspacePortScans,
getWorkspacePortScansByKey: () => useAppStore.getState().workspacePortScansByKey,
setWorkspacePortScanRefreshing
})
@@ -210,8 +201,7 @@ export function PortRow({
port,
recordFeatureInteraction,
runtimeTarget,
setWorkspacePortScan,
setWorkspacePortScanForKey,
replaceWorkspacePortScans,
setWorkspacePortScanRefreshing
]
)
@@ -0,0 +1,95 @@
/**
* Why `pty.inspectProcess` is not in-flight coalesced (#18419). The host mints one
* `observationEpoch` per request and this reader commits that epoch per read, so a reply shared by
* two overlapping probes reads as a stale replay to the second reader to settle and its would-be
* `live` identity read degrades to `unverifiable`. The pane foreground tracker overlaps its own
* probes on purpose (`cancelPendingRead` bumps the generation but lets the in-flight probe finish,
* then reissues after a 350 ms settle), so that path is reachable. The provider-side ratchet that
* fails if the dedupe returns lives in `src/main/providers/ssh-pty-inspect-observation-identity.test.ts`.
*/
import { describe, expect, it } from 'vitest'
import { createPaneForegroundProcessReader } from './pane-foreground-process-reader'
const CONNECTION_ID = 'conn-1'
const RELAY_PTY_ID = 'pty-1'
const APP_PTY_ID = `ssh:${CONNECTION_ID}@@${RELAY_PTY_ID}`
const INCARNATION_ID = 'inc-1'
/** One host scan per request => one epoch per request. */
const hostObservation = (observationEpoch: number): unknown => ({
foregroundProcess: 'claude',
hasChildProcesses: true,
foregroundProcessEvidence: {
verdict: 'live',
processName: 'claude',
ptyId: RELAY_PTY_ID,
ptyIncarnationId: INCARNATION_ID,
authorityGeneration: 'gen-1',
observationEpoch,
capturedAgeMs: 0,
fence: {
platform: 'posix',
shellPid: 100,
shellStartTime: '1000',
tty: '/dev/pts/3',
foregroundPgid: 200
}
}
})
/** Holds every probe open so the tracker's supersede-and-reissue pair really overlaps. */
function createOverlappingReader(replies: { shared: boolean }): {
readProcess: ReturnType<typeof createPaneForegroundProcessReader>
settle: (index: number) => void
} {
const resolvers: ((value: unknown) => void)[] = []
return {
// One reader instance per pane, exactly as the foreground tracker holds it.
readProcess: createPaneForegroundProcessReader({
readForegroundProcess: () => new Promise((resolve) => resolvers.push(resolve)) as never,
isRemotePtyId: () => true,
getExpectedIncarnationId: () => INCARNATION_ID
}),
// `shared` models what an in-flight dedupe would do: every joiner gets one host observation.
settle: (index) => resolvers[index]?.(hostObservation(replies.shared ? 1 : index + 1))
}
}
const flush = (): Promise<void> => new Promise((resolve) => setTimeout(resolve, 0))
describe('pane foreground inspect observation identity', () => {
it('keeps a reissued read `live` when it overlaps the probe it superseded', async () => {
const { readProcess, settle } = createOverlappingReader({ shared: false })
// The tracker cancels the first read (generation bump) but lets it run to completion, then
// reissues after the settle window — so both are in flight against the same pane.
const superseded = readProcess(APP_PTY_ID, false)
const reissued = readProcess(APP_PTY_ID, false)
await flush()
// The superseded read's continuation commits its epoch first.
settle(0)
expect((await superseded).remoteEvidenceVerdict).toBe('live')
settle(1)
const result = await reissued
expect(result.remoteEvidenceVerdict).toBe('live')
expect(result.processName).toBe('claude')
})
it('degrades the second overlapping read to `unverifiable` when one observation is shared', async () => {
const { readProcess, settle } = createOverlappingReader({ shared: true })
const superseded = readProcess(APP_PTY_ID, false)
const reissued = readProcess(APP_PTY_ID, false)
await flush()
settle(0)
expect((await superseded).remoteEvidenceVerdict).toBe('live')
settle(1)
const result = await reissued
expect(result.remoteEvidenceVerdict).toBe('unverifiable')
expect(result.processName).toBeNull()
})
})
@@ -0,0 +1,88 @@
import type { GlobalSettings } from '../../../../shared/global-settings-types'
import { inspectRuntimeTerminalProcess } from '@/runtime/runtime-terminal-inspection'
import { isRemoteExecutionHostPtyId } from '../../../../shared/remote-execution-host-pty-id'
import { isClientOnlyUnverifiableInspection } from '../../../../shared/terminal-process-inspection'
/**
* One probe answer in the fixed `live` / `unverifiable` / `exited` vocabulary of
* `docs/reference/ssh-execution-boundary.md`. `exited` is only ever produced by a host that
* answered; every failure to reach the owner — a rejection, a closed transport, or a deadline
* that expired first — stays `unverifiable`, because loss of contact is not evidence of death.
*/
export type PtyRunningWorkVerdict = 'live' | 'unverifiable' | 'exited'
export type PtyRunningWorkProbe = {
ptyId: string
verdict: PtyRunningWorkVerdict
/** Why the owner could not be observed. Only set for `unverifiable`. */
reason?: string
/** The deadline expired before this pty's probe answered at all. */
timedOut: boolean
/** The pty is owned by a remote execution host (relay runtime or app SSH). */
remote: boolean
}
type ProbeSettings = Pick<GlobalSettings, 'activeRuntimeEnvironmentId'> | null | undefined
/**
* Probes every pty for running work and resolves at whichever comes first: every answer, or the
* deadline. Never rejects, and never reports a pty it did not hear back about as idle.
*
* Callers own the policy. This owns only the measurement, so the tab-close guard and the
* window-close guard cannot drift apart on what an unanswered remote host means.
*/
export async function probePtyRunningWork(
settings: ProbeSettings,
ptyIds: readonly string[],
options: { timeoutMs: number }
): Promise<PtyRunningWorkProbe[]> {
if (ptyIds.length === 0) {
return []
}
const probes: PtyRunningWorkProbe[] = ptyIds.map((ptyId) => ({
ptyId,
verdict: 'unverifiable',
reason: 'probe_deadline',
timedOut: true,
remote: isRemoteExecutionHostPtyId(ptyId)
}))
const settle = Promise.all(
ptyIds.map(async (ptyId, index) => {
const probe = probes[index]
if (!probe) {
return
}
try {
const inspection = await inspectRuntimeTerminalProcess(settings, ptyId)
probe.timedOut = false
if (isClientOnlyUnverifiableInspection(inspection)) {
probe.verdict = 'unverifiable'
probe.reason = inspection.reason
return
}
probe.verdict = inspection.hasChildProcesses ? 'live' : 'exited'
delete probe.reason
} catch {
// Why: `inspectRuntimeTerminalProcess` already maps every failure it can classify onto a
// reason; an unclassified throw is still a failure to observe, so it stays unverifiable.
probe.timedOut = false
probe.verdict = 'unverifiable'
probe.reason = 'probe_failed'
}
})
)
let deadline: ReturnType<typeof setTimeout> | undefined
try {
await Promise.race([
settle,
new Promise<void>((resolve) => {
deadline = setTimeout(resolve, options.timeoutMs)
})
])
} finally {
clearTimeout(deadline)
}
return probes
}
@@ -46,10 +46,12 @@ function visibleRequest() {
return useRunningTerminalCloseConfirmStore.getState().runningTerminalCloseConfirm
}
// Drains pending microtasks. The probe resolves through several await points (per-pty inspect,
// the batch join, the deadline race), so this flushes generously rather than counting ticks.
async function settleProbe(): Promise<void> {
await Promise.resolve()
await Promise.resolve()
await Promise.resolve()
for (let tick = 0; tick < 12; tick += 1) {
await Promise.resolve()
}
}
describe('shouldConfirmRunningTerminalClose', () => {
@@ -329,6 +331,7 @@ describe('guardRunningTerminalClose', () => {
vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS)
vi.useRealTimers()
await settleProbe()
expect(onClose).not.toHaveBeenCalled()
expect(visibleRequest()).toMatchObject({ terminalTabId: 'tab-1', tabLabel: 'npm run dev' })
@@ -351,6 +354,7 @@ describe('guardRunningTerminalClose', () => {
guard()
vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS)
vi.useRealTimers()
await settleProbe()
expect(visibleRequest()?.copyKind).toBe('agent')
})
@@ -368,6 +372,7 @@ describe('guardRunningTerminalClose', () => {
guard(onClose)
vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS)
vi.useRealTimers()
await settleProbe()
requestSpy.mockRestore()
expect(onClose).toHaveBeenCalledTimes(1)
@@ -385,6 +390,7 @@ describe('guardRunningTerminalClose', () => {
vi.advanceTimersByTime(RUNNING_CLOSE_PROBE_TIMEOUT_MS)
vi.useRealTimers()
await settleProbe()
await settleProbe()
expect(onClose).not.toHaveBeenCalled()
useRunningTerminalCloseConfirmStore.getState().confirmRunningTerminalClose()
@@ -1,10 +1,9 @@
import { useAppStore } from '@/store'
import { inspectRuntimeTerminalProcess } from '@/runtime/runtime-terminal-inspection'
import { useRunningTerminalCloseConfirmStore } from '@/store/running-terminal-close-confirm'
import type { TerminalTabCloseReason } from '@/store/slices/terminal-tab-retirement'
import type { AppState } from '@/store/types'
import { resolveBusyPtyCloseCopyKind } from './terminal-close-copy-kind'
import { isClientOnlyUnverifiableInspection } from '../../../../shared/terminal-process-inspection'
import { probePtyRunningWork } from './pty-running-work-probe'
export type RunningTerminalCloseGuardOptions = {
force?: boolean
@@ -44,7 +43,7 @@ export function shouldConfirmRunningTerminalClose(
* the store's own teardown collector unions both for exactly that reason — reading only
* the map would let a close slip through the window with no prompt. A stale id costs
* nothing: its probe fails and the guard falls open. */
function collectTabPtyIds(
export function collectTabPtyIds(
state: Pick<AppState, 'ptyIdsByTabId' | 'terminalLayoutsByTabId'>,
terminalTabId: string
): string[] {
@@ -112,44 +111,33 @@ export function guardRunningTerminalClose(params: {
decided = true
}
const probeTimeout = setTimeout(() => {
try {
void probePtyRunningWork(settings, ptyIds, { timeoutMs: RUNNING_CLOSE_PROBE_TIMEOUT_MS })
.then((probes) => {
if (decided) {
return
}
// Why: a probe that has not answered yet is unknown, not idle. Ask, treating every pty
// as a candidate, so a degraded relay costs a click instead of a killed remote command.
confirmClose(ptyIds)
} catch {
closeNow()
}
}, RUNNING_CLOSE_PROBE_TIMEOUT_MS)
void Promise.allSettled(ptyIds.map((ptyId) => inspectRuntimeTerminalProcess(settings, ptyId)))
.then((results) => {
clearTimeout(probeTimeout)
if (decided) {
if (probes.some((probe) => probe.timedOut)) {
confirmClose(ptyIds)
return
}
// Why: fail open on an *answered* probe, matching the Cmd+W pane path — a rejection
// (wedged relay, legacy provider) or a stale remote handle is not evidence of a live
// child, and a close button that silently does nothing is worse than closing a busy tab.
const busyPtyIds = ptyIds.filter((_, index) => {
const result = results[index]
return (
result?.status === 'fulfilled' &&
!isClientOnlyUnverifiableInspection(result.value) &&
result.value.hasChildProcesses
)
})
const busyPtyIds = probes
.filter((probe) => probe.verdict === 'live')
.map((probe) => probe.ptyId)
if (busyPtyIds.length === 0) {
closeNow()
return
}
confirmClose(busyPtyIds)
})
// Why: allSettled never rejects, so this only fires when the decision above throws (a
// Why: the probe never rejects, so this only fires when the decision above throws (a
// copy-kind lookup, a store subscriber). Without it the tab would silently never close
// and the user would get no feedback at all; the pane path it replaced had this catch.
.catch(() => {
clearTimeout(probeTimeout)
closeNow()
})
}
@@ -87,10 +87,12 @@ function visibleRequest() {
return useRunningTerminalCloseConfirmStore.getState().runningTerminalCloseConfirm
}
// Drains pending microtasks. The probe resolves through several await points (per-pty inspect,
// the batch join, the deadline race), so this flushes generously rather than counting ticks.
async function settleProbe(): Promise<void> {
await Promise.resolve()
await Promise.resolve()
await Promise.resolve()
for (let tick = 0; tick < 12; tick += 1) {
await Promise.resolve()
}
}
describe('closeTerminalTab running-process confirmation', () => {
@@ -0,0 +1,231 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
const { getStateMock, inspectRuntimeTerminalProcessMock } = vi.hoisted(() => ({
getStateMock: vi.fn(),
inspectRuntimeTerminalProcessMock: vi.fn()
}))
vi.mock('@/store', () => ({
useAppStore: { getState: getStateMock }
}))
vi.mock('@/runtime/runtime-terminal-inspection', () => ({
inspectRuntimeTerminalProcess: inspectRuntimeTerminalProcessMock
}))
import {
assessWindowCloseRunningWork,
WINDOW_CLOSE_PROBE_TIMEOUT_MS
} from './window-close-running-work'
const LOCAL_PTY = 'pty-local'
const SSH_PTY = 'ssh:openclaw@@pty-7'
const RUNTIME_PTY = 'remote:env-1@@handle-1'
/** A runtime pty minted without an owner id. Still someone else's machine. */
const OWNERLESS_RUNTIME_PTY = 'remote:handle-2'
const BUSY = {
foregroundProcess: 'pnpm build',
hasChildProcesses: true,
foregroundProcessEvidence: {}
}
const IDLE = { foregroundProcess: 'bash', hasChildProcesses: false, foregroundProcessEvidence: {} }
const UNVERIFIABLE = {
foregroundProcess: null,
hasChildProcesses: false,
verdict: 'unverifiable',
reason: 'transport_loss'
}
/** One worktree, one tab, owning `ptyIds`. */
function setState(ptyIds: string[]): void {
getStateMock.mockReturnValue({
settings: { activeRuntimeEnvironmentId: null },
tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] },
ptyIdsByTabId: { 'tab-1': ptyIds },
terminalLayoutsByTabId: {}
})
}
/** Answers each pty id from `byPtyId`; anything unlisted never settles. */
function answerWith(byPtyId: Record<string, unknown>): void {
inspectRuntimeTerminalProcessMock.mockImplementation((_settings: unknown, ptyId: string) =>
ptyId in byPtyId ? Promise.resolve(byPtyId[ptyId]) : new Promise(() => {})
)
}
beforeEach(() => {
vi.clearAllMocks()
})
afterEach(() => {
vi.useRealTimers()
})
describe('assessWindowCloseRunningWork', () => {
it('warns about a live process on an SSH host (F15: remote work was filtered out entirely)', async () => {
setState([SSH_PTY])
answerWith({ [SSH_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({
kind: 'running'
})
})
it('warns on quit about a live process on an SSH host', async () => {
setState([SSH_PTY])
answerWith({ [SSH_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'running'
})
})
it('warns on quit about a live process on a paired runtime host', async () => {
setState([RUNTIME_PTY])
answerWith({ [RUNTIME_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'running'
})
})
it('counts an owner-less remote pty as remote work', async () => {
setState([OWNERLESS_RUNTIME_PTY])
answerWith({ [OWNERLESS_RUNTIME_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'running'
})
})
// The crux of docs/reference/ssh-execution-boundary.md: an unreachable host is `unverifiable`,
// and quitting on `unverifiable` as though it were `exited` is what orphans live remote work.
it('warns rather than quitting silently when a remote host answers unverifiable', async () => {
setState([SSH_PTY])
answerWith({ [SSH_PTY]: UNVERIFIABLE })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'unverifiable'
})
})
it('warns rather than quitting silently when a remote probe throws', async () => {
setState([SSH_PTY])
inspectRuntimeTerminalProcessMock.mockRejectedValue(new Error('relay wedged'))
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'unverifiable'
})
})
it('stops waiting at the budget and warns, so an unreachable host cannot hang the quit', async () => {
setState([SSH_PTY])
answerWith({})
vi.useFakeTimers()
const pending = assessWindowCloseRunningWork({ isQuitting: true })
await vi.advanceTimersByTimeAsync(WINDOW_CLOSE_PROBE_TIMEOUT_MS)
await expect(pending).resolves.toEqual({ kind: 'unverifiable' })
})
it('does not resolve before the budget expires', async () => {
setState([SSH_PTY])
answerWith({})
vi.useFakeTimers()
const settled = vi.fn()
void assessWindowCloseRunningWork({ isQuitting: true }).then(settled)
await vi.advanceTimersByTimeAsync(WINDOW_CLOSE_PROBE_TIMEOUT_MS - 1)
expect(settled).not.toHaveBeenCalled()
})
it('does not warn when the owning remote host reports an idle shell', async () => {
setState([SSH_PTY])
answerWith({ [SSH_PTY]: IDLE })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'none'
})
})
it('reports a live process even when a sibling remote pane is only unverifiable', async () => {
setState([SSH_PTY, RUNTIME_PTY])
answerWith({ [SSH_PTY]: UNVERIFIABLE, [RUNTIME_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'running'
})
})
it('still warns about a live local process when closing the window', async () => {
setState([LOCAL_PTY])
answerWith({ [LOCAL_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({
kind: 'running'
})
})
// A local probe has no transport to lose, so its failure means the pty is gone — unlike a
// remote host going quiet, it is not a reason to hold up the close.
it('does not warn when only a local probe is unverifiable', async () => {
setState([LOCAL_PTY])
answerWith({ [LOCAL_PTY]: UNVERIFIABLE })
await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({
kind: 'none'
})
})
// #524 decided quitting is an unambiguous instruction to end this machine's processes. It is
// not an instruction to end execution on someone else's, which is why remote still warns above.
it('leaves local-only quit unprompted, and never probes for it', async () => {
setState([LOCAL_PTY])
answerWith({ [LOCAL_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'none'
})
expect(inspectRuntimeTerminalProcessMock).not.toHaveBeenCalled()
})
it('probes a pane the layout has bound before the liveness map caught up', async () => {
getStateMock.mockReturnValue({
settings: { activeRuntimeEnvironmentId: null },
tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] },
ptyIdsByTabId: {},
terminalLayoutsByTabId: { 'tab-1': { ptyIdsByLeafId: { leaf: SSH_PTY } } }
})
answerWith({ [SSH_PTY]: BUSY })
await expect(assessWindowCloseRunningWork({ isQuitting: true })).resolves.toEqual({
kind: 'running'
})
})
it('probes each pty once when the map and the layout name the same one', async () => {
getStateMock.mockReturnValue({
settings: { activeRuntimeEnvironmentId: null },
tabsByWorktree: { 'worktree-1': [{ id: 'tab-1' }] },
ptyIdsByTabId: { 'tab-1': [SSH_PTY] },
terminalLayoutsByTabId: { 'tab-1': { ptyIdsByLeafId: { leaf: SSH_PTY } } }
})
answerWith({ [SSH_PTY]: IDLE })
await assessWindowCloseRunningWork({ isQuitting: true })
expect(inspectRuntimeTerminalProcessMock).toHaveBeenCalledTimes(1)
})
it('closes without probing when no workspace owns a pty', async () => {
setState([])
await expect(assessWindowCloseRunningWork({ isQuitting: false })).resolves.toEqual({
kind: 'none'
})
expect(inspectRuntimeTerminalProcessMock).not.toHaveBeenCalled()
})
})
@@ -0,0 +1,70 @@
import { useAppStore } from '@/store'
import { isRemoteExecutionHostPtyId } from '../../../../shared/remote-execution-host-pty-id'
import { collectTabPtyIds } from './running-terminal-close-guard'
import { probePtyRunningWork } from './pty-running-work-probe'
/**
* Upper bound on how long closing the window or quitting may wait on the probes.
*
* Shorter than the tab-close guard's 4s because quit is time-sensitive in a way one tab close is
* not: the user has already asked to leave, and a quit that stalls on an unreachable host is its
* own bug. A healthy local inspect answers in single-digit milliseconds and a healthy remote one
* is a single RPC round-trip on an already-open mux channel, so this leaves roughly 3x headroom
* over a slow-but-live transcontinental host while capping the worst case — a host that is simply
* gone — at ~1.5s instead of the 15s RPC timeout the probe would otherwise inherit.
*
* Expiry raises the prompt rather than quitting silently: an unanswered probe is `unverifiable`,
* and `unverifiable` is never evidence that remote work has stopped.
*/
export const WINDOW_CLOSE_PROBE_TIMEOUT_MS = 1_500
/** Which warning the close should raise, if any. */
export type WindowCloseRunningWork =
/** Every pty that mattered answered, and none had children. */
| { kind: 'none' }
/** An owning host reported a live child process. */
| { kind: 'running' }
/** A remote execution host could not be observed, so its work may still be live. */
| { kind: 'unverifiable' }
/**
* Decides whether a window close or quit should stop and ask.
*
* Two deliberate asymmetries:
*
* - **Quit only considers remote ptys.** Quitting is an unambiguous instruction to end this
* machine's processes (#524), but it is not an instruction to end execution on someone else's:
* the client detaches while the relay keeps running, and a target with a bounded grace period
* then SIGKILLs that work once the countdown expires.
* - **Only a remote `unverifiable` warns.** A local probe has no transport to lose, so its failure
* means the pty is gone. A remote one that cannot be reached is the case
* `docs/reference/ssh-execution-boundary.md` exists to protect: loss of contact is not evidence
* of `exited`, so it must fail toward asking rather than toward a silent quit.
*/
export async function assessWindowCloseRunningWork(params: {
isQuitting: boolean
}): Promise<WindowCloseRunningWork> {
const state = useAppStore.getState()
const ptyIds = new Set(
Object.values(state.tabsByWorktree)
.flatMap((worktreeTabs) => worktreeTabs ?? [])
.flatMap((tab) => collectTabPtyIds(state, tab.id))
)
const candidatePtyIds = params.isQuitting
? [...ptyIds].filter(isRemoteExecutionHostPtyId)
: [...ptyIds]
if (candidatePtyIds.length === 0) {
return { kind: 'none' }
}
const probes = await probePtyRunningWork(state.settings, candidatePtyIds, {
timeoutMs: WINDOW_CLOSE_PROBE_TIMEOUT_MS
})
if (probes.some((probe) => probe.verdict === 'live')) {
return { kind: 'running' }
}
if (probes.some((probe) => probe.remote && probe.verdict === 'unverifiable')) {
return { kind: 'unverifiable' }
}
return { kind: 'none' }
}
@@ -1,8 +1,9 @@
import { useCallback, useRef, useState } from 'react'
import { useAppStore } from '../store'
import { getConnectionId } from '../lib/connection-context'
import { isRemoteRuntimePtyId } from '@/runtime/runtime-terminal-inspection'
import { CLOSE_DIALOG_DEBOUNCE_MS } from './terminal-workspace-model'
import {
assessWindowCloseRunningWork,
type WindowCloseRunningWork
} from './terminal/window-close-running-work'
import type { TerminalWorkspaceProjectionController } from './use-terminal-workspace-projection'
import { runWithWindowCloseCheckpointScope } from './window-close-request-coordinator'
import { showShutdownCheckpointFailureToast } from '@/lib/shutdown-checkpoint-failure-toast'
@@ -27,6 +28,11 @@ export function useTerminalEditorCloseFoundation(
closeDialogDebounceTimersRef.current.add(timer)
}, [])
const [windowCloseDialogOpen, setWindowCloseDialogOpen] = useState(false)
// Why: "running" and "could not reach the host" are different claims, and telling the user
// processes are running when the truth is that a host went quiet is the fabricated certainty
// docs/reference/ssh-execution-boundary.md forbids.
const [windowCloseDialogKind, setWindowCloseDialogKind] =
useState<Exclude<WindowCloseRunningWork['kind'], 'none'>>('running')
const windowCloseAfterDirtyRef = useRef<{ isQuitting: boolean } | null>(null)
const confirmNativeWindowClose = useCallback(() => {
@@ -46,33 +52,21 @@ export function useTerminalEditorCloseFoundation(
const proceedToNativeWindowClose = useCallback(
(isQuitting: boolean) => {
if (!isQuitting) {
const state = useAppStore.getState()
const localPtyIds = Object.entries(state.tabsByWorktree).flatMap(
([worktreeId, worktreeTabs]) => {
const connectionId = getConnectionId(worktreeId)
if (connectionId !== null) {
return []
}
return worktreeTabs
.flatMap((tab) => state.ptyIdsByTabId[tab.id] ?? [])
.filter((ptyId) => !isRemoteRuntimePtyId(ptyId))
void assessWindowCloseRunningWork({ isQuitting })
.then((runningWork) => {
if (runningWork.kind === 'none') {
confirmNativeWindowClose()
return
}
)
if (localPtyIds.length > 0) {
void Promise.all(localPtyIds.map((id) => window.api.pty.hasChildProcesses(id))).then(
(results) => {
if (results.some(Boolean)) {
setWindowCloseDialogOpen(true)
} else {
confirmNativeWindowClose()
}
}
)
return
}
}
confirmNativeWindowClose()
setWindowCloseDialogKind(runningWork.kind)
setWindowCloseDialogOpen(true)
})
// Why: the assessment must never be able to trap the window. A thrown store read is
// not evidence either way, and a close that silently does nothing is unrecoverable
// without SIGKILL, so fall through to the close the user actually asked for.
.catch(() => {
confirmNativeWindowClose()
})
},
[confirmNativeWindowClose]
)
@@ -88,6 +82,7 @@ export function useTerminalEditorCloseFoundation(
releaseCloseDialogGuardAfterDebounce,
windowCloseDialogOpen,
setWindowCloseDialogOpen,
windowCloseDialogKind,
windowCloseAfterDirtyRef,
confirmNativeWindowClose,
proceedToNativeWindowClose
@@ -0,0 +1,103 @@
// @vitest-environment happy-dom
/**
* Wiring for the window-close/quit running-work warning. The policy in
* `terminal/window-close-running-work.ts` is inert unless `proceedToNativeWindowClose` actually
* consults it, so pin that it does — and that a warning stops the native close rather than
* confirming it.
*/
import { act, cleanup, renderHook } from '@testing-library/react'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
const { assessWindowCloseRunningWorkMock, confirmWindowCloseMock } = vi.hoisted(() => ({
assessWindowCloseRunningWorkMock: vi.fn(),
confirmWindowCloseMock: vi.fn()
}))
vi.mock('./terminal/window-close-running-work', () => ({
assessWindowCloseRunningWork: assessWindowCloseRunningWorkMock
}))
vi.mock('./window-close-request-coordinator', () => ({
runWithWindowCloseCheckpointScope: (fn: () => unknown) => fn()
}))
vi.mock('@/lib/shutdown-checkpoint-failure-toast', () => ({
showShutdownCheckpointFailureToast: vi.fn()
}))
const { useTerminalEditorCloseFoundation } = await import('./use-terminal-editor-close-foundation')
const controller = { openFiles: [] } as unknown as Parameters<
typeof useTerminalEditorCloseFoundation
>[0]
function mountFoundation() {
return renderHook(() => useTerminalEditorCloseFoundation(controller))
}
beforeEach(() => {
vi.clearAllMocks()
Object.assign(globalThis, {
window: Object.assign(globalThis.window, {
api: { ui: { confirmWindowClose: confirmWindowCloseMock } }
})
})
})
afterEach(() => {
cleanup()
})
describe('proceedToNativeWindowClose', () => {
it('asks the running-work policy about the quit rather than assuming it is safe', async () => {
assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'none' })
const { result } = mountFoundation()
await act(async () => {
result.current.proceedToNativeWindowClose(true)
})
expect(assessWindowCloseRunningWorkMock).toHaveBeenCalledWith({ isQuitting: true })
expect(confirmWindowCloseMock).toHaveBeenCalledTimes(1)
expect(result.current.windowCloseDialogOpen).toBe(false)
})
it('raises the dialog and does not close when a host reports live work', async () => {
assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'running' })
const { result } = mountFoundation()
await act(async () => {
result.current.proceedToNativeWindowClose(true)
})
expect(result.current.windowCloseDialogOpen).toBe(true)
expect(result.current.windowCloseDialogKind).toBe('running')
expect(confirmWindowCloseMock).not.toHaveBeenCalled()
})
it('raises the unverifiable copy when a remote host could not be reached', async () => {
assessWindowCloseRunningWorkMock.mockResolvedValue({ kind: 'unverifiable' })
const { result } = mountFoundation()
await act(async () => {
result.current.proceedToNativeWindowClose(true)
})
expect(result.current.windowCloseDialogOpen).toBe(true)
expect(result.current.windowCloseDialogKind).toBe('unverifiable')
expect(confirmWindowCloseMock).not.toHaveBeenCalled()
})
// Why: a thrown assessment is not evidence either way, and a close that silently does nothing
// leaves SIGKILL as the user's only exit.
it('falls through to the close when the assessment throws', async () => {
assessWindowCloseRunningWorkMock.mockRejectedValue(new Error('store blew up'))
const { result } = mountFoundation()
await act(async () => {
result.current.proceedToNativeWindowClose(false)
})
expect(confirmWindowCloseMock).toHaveBeenCalledTimes(1)
expect(result.current.windowCloseDialogOpen).toBe(false)
})
})
+2 -1
View File
@@ -2251,7 +2251,8 @@
"Terminal": {
"73768427cf": "Close",
"f82e9f02df": "Cancel",
"7958465754": "There are local terminals with running processes. Close the window anyway?",
"7958465754": "There are terminals with running processes. Close the window anyway?",
"b7c1f0a934": "A remote host could not be reached, so Orca cannot tell whether work is still running there. Close the window anyway?",
"2fa9c69ff3": "Close Window?",
"cd51e28d8b": "Save",
"0037b21794": "Don't Save",
+1 -1
View File
@@ -1924,7 +1924,7 @@
"Terminal": {
"73768427cf": "Cerrar",
"f82e9f02df": "Cancelar",
"7958465754": "Hay terminales locales con procesos en ejecución. ¿Cerrar la ventana de todos modos?",
"7958465754": "Hay terminales con procesos en ejecución. ¿Cerrar la ventana de todos modos?",
"2fa9c69ff3": "¿Cerrar ventana?",
"cd51e28d8b": "Guardar",
"0037b21794": "No guardar",
+1 -1
View File
@@ -2087,7 +2087,7 @@
"Terminal": {
"73768427cf": "Fermer",
"f82e9f02df": "Annuler",
"7958465754": "Des terminaux locaux exécutent des processus. Fermer quand même la fenêtre ?",
"7958465754": "Des terminaux exécutent des processus. Fermer quand même la fenêtre ?",
"2fa9c69ff3": "Fermer la fenêtre ?",
"cd51e28d8b": "Enregistrer",
"0037b21794": "Ne pas enregistrer",
+1 -1
View File
@@ -1924,7 +1924,7 @@
"Terminal": {
"73768427cf": "閉じる",
"f82e9f02df": "キャンセル",
"7958465754": "プロセスが実行中のローカルターミナルがあります。このままウィンドウを閉じますか?",
"7958465754": "プロセスが実行中のターミナルがあります。このままウィンドウを閉じますか?",
"2fa9c69ff3": "ウィンドウを閉じますか?",
"cd51e28d8b": "保存",
"0037b21794": "保存しないでください",
+1 -1
View File
@@ -1929,7 +1929,7 @@
"Terminal": {
"73768427cf": "닫기",
"f82e9f02df": "취소",
"7958465754": "실행 중인 프로세스가 있는 로컬 terminals이 있습니다. 그래도 창을 닫으시겠습니까?",
"7958465754": "실행 중인 프로세스가 있는 terminals이 있습니다. 그래도 창을 닫으시겠습니까?",
"2fa9c69ff3": "창을 닫으시겠습니까?",
"cd51e28d8b": "저장",
"0037b21794": "저장하지 않음",
+1 -1
View File
@@ -1927,7 +1927,7 @@
"Terminal": {
"73768427cf": "关闭",
"f82e9f02df": "取消",
"7958465754": "有正在运行的进程的本地终端。还是关窗吧?",
"7958465754": "有正在运行的进程的终端。还是关窗吧?",
"2fa9c69ff3": "关闭窗口?",
"cd51e28d8b": "保存",
"0037b21794": "不保存",
+51 -34
View File
@@ -7,7 +7,6 @@ import {
type RuntimeClientTarget
} from '@/runtime/runtime-rpc-client'
import { toRuntimeWorktreeSelector } from '@/runtime/runtime-worktree-selector'
import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host'
import type {
WorkspacePort,
WorkspacePortKillResult,
@@ -22,6 +21,11 @@ import { RUNTIME_BROWSER_UNAVAILABLE_MESSAGE } from './client-creation-action-po
export { addressForPort } from './workspace-port-urls'
const WORKSPACE_PORT_STOP_SETTLE_MS = 500
const WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON =
'Workspace ports are unavailable for this execution host.'
/** Projection key for the merged multi-host view; never a per-host scan key. */
export const WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY = 'all-hosts:all'
export function canStopWorkspacePort(
port: WorkspacePort
@@ -33,13 +37,17 @@ type BrowserTabCreator = ReturnType<typeof useAppStore.getState>['createBrowserT
type RemoteBrowserPageHandleSetter = ReturnType<
typeof useAppStore.getState
>['setRemoteBrowserPageHandle']
type WorkspacePortScanSetter = ReturnType<typeof useAppStore.getState>['setWorkspacePortScan']
type WorkspacePortScanByKeySetter = ReturnType<
typeof useAppStore.getState
>['setWorkspacePortScanForKey']
type WorkspacePortScanRefreshingSetter = ReturnType<
typeof useAppStore.getState
>['setWorkspacePortScanRefreshing']
type ReplaceWorkspacePortScansSetter = ReturnType<
typeof useAppStore.getState
>['replaceWorkspacePortScans']
export type WorkspacePortScanPublisher = {
replaceWorkspacePortScans: ReplaceWorkspacePortScansSetter
getWorkspacePortScansByKey: () => Record<string, WorkspacePortScanResult>
}
function delay(ms: number): Promise<void> {
return new Promise((resolve) => window.setTimeout(resolve, ms))
@@ -94,12 +102,15 @@ export function goToWorkspacePortOwner(port: WorkspacePort): boolean {
export async function openWorkspacePortInBrowser(args: {
port: WorkspacePort
activeWorktreeId?: string | null
runtimeTarget: RuntimeClientTarget
runtimeTarget: RuntimeClientTarget | null
createBrowserTab: BrowserTabCreator
setRemoteBrowserPageHandle: RemoteBrowserPageHandleSetter
openInOrcaBrowser?: boolean
localhostLabelRoute?: LocalhostWorktreeLabelRoute | null
}): Promise<{ ok: true } | { ok: false; reason: string }> {
if (!args.runtimeTarget) {
return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON }
}
const rawUrl = browserUrlForPort(args.port)
let url = rawUrl
if (args.runtimeTarget.kind === 'local' && args.localhostLabelRoute) {
@@ -166,22 +177,38 @@ export async function openWorkspacePortInBrowser(args: {
}
}
export async function refreshWorkspacePortScanAfterStop(args: {
runtimeTarget: RuntimeClientTarget
setWorkspacePortScan: WorkspacePortScanSetter
setWorkspacePortScanForKey?: WorkspacePortScanByKeySetter
setWorkspacePortScanRefreshing: WorkspacePortScanRefreshingSetter
getWorkspacePortScansByKey?: () => Record<string, WorkspacePortScanResult>
}): Promise<{ ok: true } | { ok: false; reason: string }> {
/**
* Stores one host's scan and republishes the aggregate the status bar reads.
* Why: a single-host publish used to overwrite that aggregate, so every other
* host's ports vanished from the count until the next background poll. One
* replaceWorkspacePortScans update (not setWorkspacePortScan) keeps the synthetic
* all-hosts key out of workspacePortScansByKey, where re-merging it would
* duplicate rows — and notifies subscribers once instead of twice for one scan.
*/
export function publishWorkspacePortScanForHost(
args: WorkspacePortScanPublisher & { scanKey: string; scan: WorkspacePortScanResult }
): void {
const scansByKey = { ...args.getWorkspacePortScansByKey(), [args.scanKey]: args.scan }
const merged = mergeWorkspacePortScans(scansByKey)
args.replaceWorkspacePortScans(scansByKey, {
key: Object.keys(scansByKey).length > 1 ? WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY : args.scanKey,
result: merged ?? args.scan
})
}
/** Re-scans one host after a port stop (immediately, then settled) and republishes the aggregate. */
export async function refreshWorkspacePortScanAfterStop(
args: WorkspacePortScanPublisher & {
runtimeTarget: RuntimeClientTarget | null
setWorkspacePortScanRefreshing: WorkspacePortScanRefreshingSetter
}
): Promise<{ ok: true } | { ok: false; reason: string }> {
if (!args.runtimeTarget) {
return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON }
}
const scanKey = workspacePortScanKeyForTarget(args.runtimeTarget)
const publishScan = (scan: WorkspacePortScanResult): void => {
args.setWorkspacePortScanForKey?.(scanKey, scan)
const currentScans = args.getWorkspacePortScansByKey?.() ?? {}
const merged = mergeWorkspacePortScans({ ...currentScans, [scanKey]: scan })
args.setWorkspacePortScan({
key: merged && Object.keys(currentScans).length > 0 ? 'all-hosts:all' : scanKey,
result: merged ?? scan
})
publishWorkspacePortScanForHost({ ...args, scanKey, scan })
}
args.setWorkspacePortScanRefreshing(true)
try {
@@ -216,19 +243,6 @@ export function workspacePortRuntimeTargetKey(target: RuntimeClientTarget): stri
return target.kind === 'local' ? 'local' : `environment:${target.environmentId}`
}
export function runtimeTargetForExecutionHostId(
hostId: ExecutionHostId
): RuntimeClientTarget | null {
const parsed = parseExecutionHostId(hostId)
if (parsed?.kind === 'local') {
return { kind: 'local' }
}
if (parsed?.kind === 'runtime') {
return { kind: 'environment', environmentId: parsed.environmentId }
}
return null
}
export function workspacePortScanKeyForTarget(target: RuntimeClientTarget): string {
return `${workspacePortRuntimeTargetKey(target)}:all`
}
@@ -295,9 +309,12 @@ export async function scanWorkspacePortsForTarget(
}
export async function killWorkspacePortForTarget(
target: RuntimeClientTarget,
target: RuntimeClientTarget | null,
args: { repoId: string; pid: number; port: number }
): Promise<WorkspacePortKillResult> {
if (!target) {
return { ok: false, reason: WORKSPACE_PORT_TARGET_UNAVAILABLE_REASON }
}
if (target.kind === 'local') {
return window.api.workspacePorts.kill(args)
}
@@ -0,0 +1,148 @@
import { describe, expect, it } from 'vitest'
import type { WorkspacePortScanResult } from '../../../shared/workspace-ports'
import {
getUnavailableWorkspacePortHosts,
workspacePortHostForScanKey
} from './workspace-port-host-availability'
function scan(overrides: Partial<WorkspacePortScanResult> = {}): WorkspacePortScanResult {
return { platform: 'linux', scannedAt: 1, ports: [], ...overrides }
}
describe('getUnavailableWorkspacePortHosts', () => {
it('reports the failed host while another host still answers', () => {
expect(
getUnavailableWorkspacePortHosts({
'local:all': scan(),
'environment:env-1:all': scan({ unavailableReason: 'Remote connection dropped' })
})
).toEqual([
{
scanKey: 'environment:env-1:all',
host: { kind: 'environment', environmentId: 'env-1' },
platform: 'linux',
reason: 'Remote connection dropped'
}
])
})
it('reports the local host as a local host ref, not an absent environment id', () => {
expect(
getUnavailableWorkspacePortHosts({
'local:all': scan({ unavailableReason: 'lsof is unavailable' }),
'environment:env-1:all': scan()
})
).toEqual([
{
scanKey: 'local:all',
host: { kind: 'local' },
platform: 'linux',
reason: 'lsof is unavailable'
}
])
})
it('keeps colons inside an environment id when parsing the scan key', () => {
// Why: keys are `${targetKey}:all`, so the id runs to the last `:all` —
// splitting on the first colon would truncate ids that contain colons.
expect(
getUnavailableWorkspacePortHosts({
'local:all': scan(),
'environment:weird:id:all': scan({ unavailableReason: 'Remote connection dropped' })
})
).toEqual([
{
scanKey: 'environment:weird:id:all',
host: { kind: 'environment', environmentId: 'weird:id' },
platform: 'linux',
reason: 'Remote connection dropped'
}
])
})
// Why: total loss of contact is where naming the host matters most — the merged
// projection joins raw internal scan keys, so it cannot name them itself.
it('names every host when all of them failed', () => {
expect(
getUnavailableWorkspacePortHosts({
'local:all': scan({ unavailableReason: 'lsof is unavailable' }),
'environment:env-1:all': scan({ unavailableReason: 'Remote connection dropped' })
})
).toEqual([
{
scanKey: 'local:all',
host: { kind: 'local' },
platform: 'linux',
reason: 'lsof is unavailable'
},
{
scanKey: 'environment:env-1:all',
host: { kind: 'environment', environmentId: 'env-1' },
platform: 'linux',
reason: 'Remote connection dropped'
}
])
})
it('names a single failed host', () => {
expect(
getUnavailableWorkspacePortHosts({
'local:all': scan({ unavailableReason: 'lsof is unavailable' })
})
).toEqual([
{
scanKey: 'local:all',
host: { kind: 'local' },
platform: 'linux',
reason: 'lsof is unavailable'
}
])
})
// Why: the synthetic all-hosts projection key must never be labelled as the
// local machine — that would blame the wrong host for a remote failure.
it('marks an unrecognised scan key as an unknown host', () => {
expect(
getUnavailableWorkspacePortHosts({
'all-hosts:all': scan({ unavailableReason: 'Remote connection dropped' })
})
).toEqual([
{
scanKey: 'all-hosts:all',
host: { kind: 'unknown' },
platform: 'linux',
reason: 'Remote connection dropped'
}
])
})
// Why: a paired web client's userAgent is not the Orca host's platform, so the
// caller labels the local host from the scan's own platform.
it("carries the failed scan's platform, and null when it is unknown", () => {
expect(
getUnavailableWorkspacePortHosts({
'local:all': scan({ platform: 'win32', unavailableReason: 'netstat failed' }),
'environment:env-1:all': scan({ platform: 'unknown', unavailableReason: 'dropped' })
}).map((entry) => entry.platform)
).toEqual(['win32', null])
})
it('stays silent when nothing failed', () => {
expect(getUnavailableWorkspacePortHosts({ 'local:all': scan() })).toEqual([])
expect(getUnavailableWorkspacePortHosts({})).toEqual([])
})
})
describe('workspacePortHostForScanKey', () => {
it.each([
['local:all', { kind: 'local' }],
['environment:env-1:all', { kind: 'environment', environmentId: 'env-1' }],
['environment:weird:id:all', { kind: 'environment', environmentId: 'weird:id' }],
['all-hosts:all', { kind: 'unknown' }],
['environment::all', { kind: 'unknown' }],
['environment:env-1', { kind: 'unknown' }],
['local', { kind: 'unknown' }]
])('maps %s', (scanKey, expected) => {
expect(workspacePortHostForScanKey(scanKey)).toEqual(expected)
})
})
@@ -0,0 +1,65 @@
import type { WorkspacePortScanResult } from '../../../shared/workspace-ports'
/**
* Host a per-host scan key points at. `unknown` is kept distinct from `local` so
* an unrecognised key (a synthetic projection key that leaked into the per-host
* map, say) is never mislabelled as a local failure.
*/
export type WorkspacePortHostRef =
| { kind: 'local' }
| { kind: 'environment'; environmentId: string }
| { kind: 'unknown' }
export type UnavailableWorkspacePortHost = {
scanKey: string
host: WorkspacePortHostRef
/** Platform the failed scan last ran on; drives the local host's label. */
platform: NodeJS.Platform | null
reason: string
}
// Why: mirrors workspacePortScanKeyForTarget (`${targetKey}:all`, where the
// target key is `local` or `environment:<id>`) without importing the heavier
// workspace-port-actions module into this pure helper. Splitting on the last
// `:all` keeps environment ids that themselves contain colons intact.
const ENVIRONMENT_SCAN_KEY_PREFIX = 'environment:'
const SCAN_KEY_SUFFIX = ':all'
const LOCAL_SCAN_KEY = `local${SCAN_KEY_SUFFIX}`
/** Host a per-host scan key names; `unknown` for any other key shape. */
export function workspacePortHostForScanKey(scanKey: string): WorkspacePortHostRef {
if (scanKey === LOCAL_SCAN_KEY) {
return { kind: 'local' }
}
if (!scanKey.endsWith(SCAN_KEY_SUFFIX) || !scanKey.startsWith(ENVIRONMENT_SCAN_KEY_PREFIX)) {
return { kind: 'unknown' }
}
const environmentId = scanKey.slice(
ENVIRONMENT_SCAN_KEY_PREFIX.length,
scanKey.length - SCAN_KEY_SUFFIX.length
)
return environmentId ? { kind: 'environment', environmentId } : { kind: 'unknown' }
}
/**
* Every host whose latest scan failed, named by host rather than by scan key.
* Why: on a remote host "none listening" and "could not look" are different
* answers, and the merged projection collapses both the partial case (no reason
* at all) and the total case (reasons joined with raw internal keys).
*/
export function getUnavailableWorkspacePortHosts(
scansByKey: Record<string, WorkspacePortScanResult>
): UnavailableWorkspacePortHost[] {
return Object.entries(scansByKey).flatMap(([scanKey, scan]) =>
scan?.unavailableReason
? [
{
scanKey,
host: workspacePortHostForScanKey(scanKey),
platform: scan.platform === 'unknown' ? null : scan.platform,
reason: scan.unavailableReason
}
]
: []
)
}
@@ -25,6 +25,11 @@ function unavailable(): WorkspacePortScanResult {
return { platform: 'unknown', scannedAt: 1, ports: [], unavailableReason: 'scan failed' }
}
/** What the ports popover publishes when its own scan fails: reason + last-good ports. */
function unavailableWithRetainedPorts(portIds: string[]): WorkspacePortScanResult {
return { ...good(portIds), unavailableReason: 'scan failed' }
}
const FAILURE_THRESHOLD = 2
function createHarness(): {
@@ -125,6 +130,33 @@ describe('reconcileTransientPortScanFailures', () => {
expect(state.has('flaky:all')).toBe(false)
})
// Why: the popover publishes reason + last-good ports the moment its own scan
// fails. Counting that as a spent grace period would drop those ports on the
// very next poll, so the retention would never survive one poll interval.
it('still grants the grace period after a failure that retained its ports', () => {
const { apply, publish } = createHarness()
apply([{ key: 'h:all', result: good(['tcp:3000']) }])
const popoverResult = unavailableWithRetainedPorts(['tcp:3000'])
publish('h:all', popoverResult)
const next = apply([{ key: 'h:all', result: unavailable() }])
expect(next[0].result).toBe(popoverResult)
expect(next[0].result.ports).toHaveLength(1)
})
it('still drops retained ports once failures reach the tolerance', () => {
const { apply, publish } = createHarness()
apply([{ key: 'h:all', result: good(['tcp:3000']) }])
publish('h:all', unavailableWithRetainedPorts(['tcp:3000']))
apply([{ key: 'h:all', result: unavailable() }])
const next = apply([{ key: 'h:all', result: unavailable() }])
expect(next[0].result.ports).toHaveLength(0)
expect(next[0].result.unavailableReason).toBe('scan failed')
})
it('uses a newer manual result instead of resurrecting stale ports', () => {
const { apply, publish } = createHarness()
apply([{ key: 'h:all', result: good(['tcp:3000']) }])
@@ -34,10 +34,14 @@ export function reconcileTransientPortScanFailures(
return { key, result }
}
const failures = previousFailures + 1
// Why: a surface that hit the same failure first (the ports popover) republishes
// the host's last-good ports alongside the reason. Treating that as a spent grace
// period would drop those ports on the very next poll, undoing the retention.
const publishedIsRetainable =
Boolean(publishedResult) &&
(!publishedResult.unavailableReason || publishedResult.ports.length > 0)
const nextResult =
failures < failureThreshold && publishedResult && !publishedResult.unavailableReason
? publishedResult
: result
failures < failureThreshold && publishedIsRetainable ? publishedResult : result
state.set(key, { consecutiveFailures: failures, publishedResult: nextResult })
return { key, result: nextResult }
})
@@ -0,0 +1,153 @@
// @vitest-environment happy-dom
import { beforeEach, describe, expect, it, vi } from 'vitest'
import type { WorkspacePortScanResult } from '../../../shared/workspace-ports'
vi.mock('@/lib/worktree-activation', () => ({
activateAndRevealWorktree: vi.fn()
}))
vi.mock('@/runtime/runtime-rpc-client', () => ({
getActiveRuntimeTarget: vi.fn(),
callRuntimeRpc: vi.fn(),
assertRuntimeEnvironmentCapability: vi.fn(),
RuntimeRpcCallError: class RuntimeRpcCallError extends Error {
code?: string
}
}))
vi.mock('./workspace-port-scan-client', () => ({
runWorkspacePortScanForTarget: vi.fn()
}))
const { publishWorkspacePortScanForHost, WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY } =
await import('./workspace-port-actions')
type WorkspacePortScanPublisher = Parameters<typeof publishWorkspacePortScanForHost>[0]
function scanWithPort(port: number, scannedAt: number): WorkspacePortScanResult {
return {
platform: 'linux',
scannedAt,
ports: [
{
id: `tcp:${port}`,
bindHost: '0.0.0.0',
connectHost: '127.0.0.1',
port,
pid: 100 + port,
processName: 'node',
protocol: 'http',
kind: 'external'
}
]
}
}
/** Mirrors the store's replaceWorkspacePortScans semantics: one atomic update. */
function makeStoreHarness(initial: Record<string, WorkspacePortScanResult> = {}): {
scansByKey: Record<string, WorkspacePortScanResult>
projections: { key: string; result: WorkspacePortScanResult }[]
publisher: Omit<WorkspacePortScanPublisher, 'scanKey' | 'scan'>
} {
let scansByKey: Record<string, WorkspacePortScanResult> = { ...initial }
const projections: { key: string; result: WorkspacePortScanResult }[] = []
return {
get scansByKey() {
return scansByKey
},
projections,
publisher: {
replaceWorkspacePortScans: (
nextScansByKey: Record<string, WorkspacePortScanResult>,
projection: { key: string; result: WorkspacePortScanResult } | null
) => {
scansByKey = nextScansByKey
if (projection) {
projections.push(projection)
}
},
getWorkspacePortScansByKey: () => scansByKey
}
}
}
describe('publishWorkspacePortScanForHost', () => {
let localScan: WorkspacePortScanResult
let remoteScan: WorkspacePortScanResult
beforeEach(() => {
localScan = scanWithPort(5173, 10)
remoteScan = scanWithPort(3000, 20)
})
it('publishes the single tracked host under its own key', () => {
const harness = makeStoreHarness()
publishWorkspacePortScanForHost({
...harness.publisher,
scanKey: 'local:all',
scan: localScan
})
expect(harness.projections).toEqual([{ key: 'local:all', result: localScan }])
expect(Object.keys(harness.scansByKey)).toEqual(['local:all'])
})
it('keeps the other host in the projection when one host refreshes', () => {
const harness = makeStoreHarness({ 'local:all': localScan })
publishWorkspacePortScanForHost({
...harness.publisher,
scanKey: 'environment:env-1:all',
scan: remoteScan
})
const projection = harness.projections.at(-1)
expect(projection?.key).toBe(WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY)
expect(projection?.result.ports.map((port) => port.port).sort()).toEqual([3000, 5173])
expect(Object.keys(harness.scansByKey).sort()).toEqual(['environment:env-1:all', 'local:all'])
})
it('publishes map and projection in a single store update', () => {
const harness = makeStoreHarness({ 'local:all': localScan })
const replaceSpy = vi.spyOn(harness.publisher, 'replaceWorkspacePortScans')
publishWorkspacePortScanForHost({
...harness.publisher,
scanKey: 'environment:env-1:all',
scan: remoteScan
})
// Why: two sequential setter calls notify subscribers twice for one scan;
// one atomic replace keeps map and projection from ever disagreeing.
expect(replaceSpy).toHaveBeenCalledTimes(1)
const [nextScans, projection] = replaceSpy.mock.calls[0]
expect(Object.keys(nextScans).sort()).toEqual(['environment:env-1:all', 'local:all'])
expect(projection?.key).toBe(WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY)
})
it('does not accumulate duplicate rows across repeated publishes', () => {
const harness = makeStoreHarness({ 'local:all': localScan })
publishWorkspacePortScanForHost({
...harness.publisher,
scanKey: 'environment:env-1:all',
scan: remoteScan
})
publishWorkspacePortScanForHost({
...harness.publisher,
scanKey: 'environment:env-1:all',
scan: { ...remoteScan, scannedAt: 30 }
})
// Why: the aggregate must never land in the per-host map, or the next merge
// folds the merged result back into itself and rows multiply.
expect(harness.scansByKey[WORKSPACE_PORT_ALL_HOSTS_SCAN_KEY]).toBeUndefined()
expect(
harness.projections
.at(-1)
?.result.ports.map((port) => port.port)
.sort()
).toEqual([3000, 5173])
})
})
@@ -1,4 +1,5 @@
import type { GlobalSettings } from '../../../shared/global-settings-types'
import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host'
export type RuntimeClientTarget = { kind: 'local' } | { kind: 'environment'; environmentId: string }
@@ -9,6 +10,20 @@ export function getActiveRuntimeTarget(
return environmentId ? { kind: 'environment', environmentId } : { kind: 'local' }
}
/** RPC target for a dispatchable host; direct SSH cannot use this client path. */
export function runtimeTargetForExecutionHostId(
hostId: ExecutionHostId
): RuntimeClientTarget | null {
const parsed = parseExecutionHostId(hostId)
if (parsed?.kind === 'local') {
return { kind: 'local' }
}
if (parsed?.kind === 'runtime') {
return { kind: 'environment', environmentId: parsed.environmentId }
}
return null
}
export function settingsForRuntimeOwner(
settings: Pick<GlobalSettings, 'activeRuntimeEnvironmentId'> | null | undefined,
runtimeEnvironmentId: string | null | undefined
@@ -0,0 +1,16 @@
import { useAppStore } from '@/store'
import { getExecutionHostIdForWorktree } from '@/lib/worktree-runtime-owner'
import { runtimeTargetForExecutionHostId, type RuntimeClientTarget } from './runtime-client-target'
/**
* Runtime target that owns `worktreeId`, which is not always the globally
* focused runtime — acting on the focused one scans the wrong host and reports
* that workspace as having no ports. Direct-SSH owners return null.
*/
export function useWorktreeRuntimeTarget(
worktreeId: string | null | undefined
): RuntimeClientTarget | null {
return useAppStore((state) =>
runtimeTargetForExecutionHostId(getExecutionHostIdForWorktree(state, worktreeId))
)
}
@@ -43,15 +43,20 @@ export function createTerminalLayoutActions(
}
})
},
// Why: pane mount/unmount re-asserts the same booleans; bailing like setTabLayout keeps map subscribers asleep.
setTabPaneExpanded: (tabId, expanded) => {
set((s) => ({
expandedPaneByTabId: { ...s.expandedPaneByTabId, [tabId]: expanded }
}))
set((s) =>
s.expandedPaneByTabId[tabId] === expanded
? s
: { expandedPaneByTabId: { ...s.expandedPaneByTabId, [tabId]: expanded } }
)
},
setTabCanExpandPane: (tabId, canExpand) => {
set((s) => ({
canExpandPaneByTabId: { ...s.canExpandPaneByTabId, [tabId]: canExpand }
}))
set((s) =>
s.canExpandPaneByTabId[tabId] === canExpand
? s
: { canExpandPaneByTabId: { ...s.canExpandPaneByTabId, [tabId]: canExpand } }
)
},
setTabLayout: (tabId, layout) => {
let ownershipTransfers: ReturnType<typeof resolveTerminalLayoutPtyOwnershipTransfers> = []
@@ -0,0 +1,152 @@
// @vitest-environment happy-dom
import { Profiler } from 'react'
import { act, cleanup, render } from '@testing-library/react'
import { afterEach, describe, expect, it } from 'vitest'
import { createTestStore } from '../slices/store-test-helpers'
afterEach(cleanup)
type TestStore = ReturnType<typeof createTestStore>
const TAB_ID = 'tab-1'
const NO_OP_WRITES = 25
function recordPublishedMapKeys(store: TestStore): string[] {
const published: string[] = []
store.subscribe((next, previous) => {
if (next.expandedPaneByTabId !== previous.expandedPaneByTabId) {
published.push('expandedPaneByTabId')
}
if (next.canExpandPaneByTabId !== previous.canExpandPaneByTabId) {
published.push('canExpandPaneByTabId')
}
})
return published
}
// Mirrors use-terminal-workspace-store-bindings.ts:17, which subscribes to the raw map.
function ExpandedPaneSubscriber({ store }: { store: TestStore }): React.JSX.Element {
const expandedPaneByTabId = store((s) => s.expandedPaneByTabId)
return <span>{String(expandedPaneByTabId[TAB_ID] === true)}</span>
}
function CanExpandPaneSubscriber({ store }: { store: TestStore }): React.JSX.Element {
const canExpandPaneByTabId = store((s) => s.canExpandPaneByTabId)
return <span>{String(canExpandPaneByTabId[TAB_ID] === true)}</span>
}
function renderCommitCounter(subscriber: React.JSX.Element): () => number {
let commits = 0
render(
<Profiler
id="terminal-pane-expansion"
onRender={() => {
commits += 1
}}
>
{subscriber}
</Profiler>
)
const mountCommits = commits
return () => commits - mountCommits
}
describe('setTabPaneExpanded', () => {
it('publishes the first write for an unseen tab and a real toggle', () => {
const store = createTestStore()
const published = recordPublishedMapKeys(store)
store.getState().setTabPaneExpanded(TAB_ID, false)
expect(published).toEqual(['expandedPaneByTabId'])
expect(store.getState().expandedPaneByTabId[TAB_ID]).toBe(false)
store.getState().setTabPaneExpanded(TAB_ID, true)
expect(published).toEqual(['expandedPaneByTabId', 'expandedPaneByTabId'])
expect(store.getState().expandedPaneByTabId[TAB_ID]).toBe(true)
})
it('bails out when the value is unchanged', () => {
const store = createTestStore()
store.getState().setTabPaneExpanded(TAB_ID, false)
const before = store.getState().expandedPaneByTabId
// Root identity too: returning `{}` keeps the map but allocates a new root, so zustand still walks every listener.
const rootBefore = store.getState()
const published = recordPublishedMapKeys(store)
for (let i = 0; i < NO_OP_WRITES; i += 1) {
store.getState().setTabPaneExpanded(TAB_ID, false)
}
expect(published).toEqual([])
expect(store.getState().expandedPaneByTabId).toBe(before)
expect(store.getState()).toBe(rootBefore)
})
it('costs no React commit in a map subscriber when the value is unchanged', () => {
const store = createTestStore()
store.getState().setTabPaneExpanded(TAB_ID, false)
const commitsSinceMount = renderCommitCounter(<ExpandedPaneSubscriber store={store} />)
for (let i = 0; i < NO_OP_WRITES; i += 1) {
act(() => {
store.getState().setTabPaneExpanded(TAB_ID, false)
})
}
expect(commitsSinceMount()).toBe(0)
act(() => {
store.getState().setTabPaneExpanded(TAB_ID, true)
})
expect(commitsSinceMount()).toBe(1)
})
})
describe('setTabCanExpandPane', () => {
it('publishes the first write for an unseen tab and a real toggle', () => {
const store = createTestStore()
const published = recordPublishedMapKeys(store)
store.getState().setTabCanExpandPane(TAB_ID, false)
expect(published).toEqual(['canExpandPaneByTabId'])
expect(store.getState().canExpandPaneByTabId[TAB_ID]).toBe(false)
store.getState().setTabCanExpandPane(TAB_ID, true)
expect(published).toEqual(['canExpandPaneByTabId', 'canExpandPaneByTabId'])
expect(store.getState().canExpandPaneByTabId[TAB_ID]).toBe(true)
})
it('bails out when the value is unchanged', () => {
const store = createTestStore()
store.getState().setTabCanExpandPane(TAB_ID, false)
const before = store.getState().canExpandPaneByTabId
const rootBefore = store.getState()
const published = recordPublishedMapKeys(store)
for (let i = 0; i < NO_OP_WRITES; i += 1) {
store.getState().setTabCanExpandPane(TAB_ID, false)
}
expect(published).toEqual([])
expect(store.getState().canExpandPaneByTabId).toBe(before)
expect(store.getState()).toBe(rootBefore)
})
it('costs no React commit in a map subscriber when the value is unchanged', () => {
const store = createTestStore()
store.getState().setTabCanExpandPane(TAB_ID, false)
const commitsSinceMount = renderCommitCounter(<CanExpandPaneSubscriber store={store} />)
for (let i = 0; i < NO_OP_WRITES; i += 1) {
act(() => {
store.getState().setTabCanExpandPane(TAB_ID, false)
})
}
expect(commitsSinceMount()).toBe(0)
act(() => {
store.getState().setTabCanExpandPane(TAB_ID, true)
})
expect(commitsSinceMount()).toBe(1)
})
})
+19 -4
View File
@@ -109,10 +109,12 @@ export function clearMissingProjectGroupMemberships(repos: Repo[], groups: Proje
)
}
export function getProjectGroupSubtreeIds(
groups: readonly Pick<ProjectGroup, 'id' | 'parentGroupId'>[],
rootGroupId: string
): Set<string> {
export type ProjectGroupChildIndex = ReadonlyMap<string, string[]>
/** Build once and reuse when collecting subtrees for more than one root. */
export function buildProjectGroupChildIndex(
groups: readonly Pick<ProjectGroup, 'id' | 'parentGroupId'>[]
): ProjectGroupChildIndex {
const childGroupsByParentId = new Map<string, string[]>()
for (const group of groups) {
if (!group.parentGroupId) {
@@ -122,7 +124,20 @@ export function getProjectGroupSubtreeIds(
children.push(group.id)
childGroupsByParentId.set(group.parentGroupId, children)
}
return childGroupsByParentId
}
export function getProjectGroupSubtreeIds(
groups: readonly Pick<ProjectGroup, 'id' | 'parentGroupId'>[],
rootGroupId: string
): Set<string> {
return collectProjectGroupSubtreeIds(buildProjectGroupChildIndex(groups), rootGroupId)
}
export function collectProjectGroupSubtreeIds(
childGroupsByParentId: ProjectGroupChildIndex,
rootGroupId: string
): Set<string> {
const subtreeIds = new Set<string>()
const pending = [rootGroupId]
while (pending.length > 0) {
+16 -2
View File
@@ -37,6 +37,12 @@ export type RelayArtifact = {
* optional one would loop forever redeploying a relay that is already correct.
*/
optional?: boolean
/**
* Forked by the relay daemon as a long-lived child of its own. These are relay
* infrastructure, never user work, and the reap gate subtracts them from a daemon's
* child census; see src/main/ssh/relay-daemon-service-children.ts.
*/
daemonServiceChild?: boolean
}
/** The bare Windows process-table addon; see docs/reference/windows-process-enumeration.md. */
@@ -44,8 +50,8 @@ export const RELAY_WINDOWS_PROCESS_TREE_FILENAME = 'windows-process-tree.node'
export const RELAY_ARTIFACTS: readonly RelayArtifact[] = [
{ filename: 'relay.js' },
{ filename: 'relay-watcher.js' },
{ filename: 'relay-ai-vault-service.js' },
{ filename: 'relay-watcher.js', daemonServiceChild: true },
{ filename: 'relay-ai-vault-service.js', daemonServiceChild: true },
{ filename: 'managed-hook-runtime.js' },
// Forked by the AI Vault title reader; without it a relay answers every WSL
// title request with no title and no error.
@@ -62,6 +68,14 @@ export const RELAY_ARTIFACTS: readonly RelayArtifact[] = [
{ filename: RELAY_WINDOWS_PROCESS_TREE_FILENAME, windowsOnly: true, optional: true }
]
/**
* The daemon's own service children, by entry filename. Anything else under a relay pid is
* either user work or unidentified, and both keep the relay unreapable.
*/
export const RELAY_DAEMON_SERVICE_ENTRY_FILENAMES: readonly string[] = RELAY_ARTIFACTS.filter(
(artifact) => artifact.daemonServiceChild
).map((artifact) => artifact.filename)
/** Written after the artifacts, so it is never an input to its own hash. */
export const RELAY_VERSION_FILENAME = '.version'
@@ -0,0 +1,14 @@
import { parseRemoteRuntimePtyId } from './remote-runtime-pty-id'
import { parseAppSshPtyId } from './ssh-pty-id'
/**
* Whether the process behind this pty runs on an execution host other than this machine —
* a paired runtime environment or an app SSH target.
*
* Deliberately broader than the inspection module's private remote check, which only counts a
* `remote:` id that carries an owner environment id. An owner-less `remote:<handle>` still runs
* somewhere else, and treating it as local is how remote work becomes invisible to a guard.
*/
export function isRemoteExecutionHostPtyId(ptyId: string): boolean {
return parseRemoteRuntimePtyId(ptyId) !== null || parseAppSshPtyId(ptyId) !== null
}