mirror of
https://github.com/stablyai/orca.git
synced 2026-09-29 08:03:20 +00:00
Three failures with one shape: a payload past a fixed capacity was met with silence, with a wait that never ends, or with a prefix presented as a whole. **The workspace snapshot was silently dropped.** `workspace.changed` carries the tab/session list, and a snapshot past the producer frame capacity (12288 B on a Node <=21 remote) was dropped with only a relay stderr line, so the client kept a stale list forever. The relay now publishes per client and, for a client whose sink refused the frame, sends a compact `workspace.stale` marker on the control lane; the client re-reads through `workspace.get`, whose lane is budgeted in megabytes rather than in one producer frame. A new JSON-RPC notification rather than a new field on `workspace.changed`: `normalizeSnapshot(undefined, ns)` yields revision 0 and an empty session, so a Rule-1 field would make an old client replace its tab list with nothing — worse than the drop. An old client ignores the unknown method and is exactly where it is today. The marker retention/retry machinery is extracted from the `fs.changed` overflow path and shared by both. **The Windows upload hung, and the fix for it could truncate.** `#16432` was attributed to `[Console]::In.ReadToEnd()` materializing the base64 bundle. That is not what the reporter measured: he also measured `new IO.StreamReader([Console]::OpenStandardInput())` — an incremental reader — hanging at 1 MB. The limit is in the stdin the host hands PowerShell over a non-pty ssh exec, not in the string the script builds. - `uploadFileViaSystemSsh` — the user file-import path — was piping a whole file into one Windows stdin, unchunked and untimed. That is the path large files take; it now chunks into 32 KB writes and bounds each wait. - The Windows directory upload reuses that single-file path rather than repeating a weaker copy of chunk-read + write-buffer; the `ino`/`dev` TOCTOU verification comes with it. - A Windows write needing more than one exec lands on a `.orca-partial` staging path and is published by rename, so a failed chunk cannot leave a truncated artifact under the real name. `exclusive` is enforced once at the rename, not on the first chunk, where a retry met its own leftovers. - The mkdir batch reads stdin through the stream reader the reporter measured surviving 50 KB, not `[Console]::In`, which he measured wedging at that size. - `waitForChannelClose` takes an optional bound. A wedged PowerShell stays alive at idle CPU and never closes, so without one the promise is simply never settled and the caller waits forever with no error to show. **Quick Open showed a prefix as the whole workspace.** The mechanism "a full page means there is more" only works if the caller named the cap, and the failing UI named none — it hardcoded `truncated: false`. Quick Open now names `QUICK_OPEN_LISTING_MAX_RESULTS` on both the Electron IPC hop and the runtime-RPC hop (the field #17954 added to `files.listAll`), and reads a full page as truncation. The local hop honours the cap too, which it previously ignored. Rebase note on `fs.listFiles`: an earlier revision of this work also clamped the host unconditionally, and #17934 escalated an uncapped request to an explicit error. #17954 has since landed and made an oversized reply streamable, which removes the premise — the host no longer has to choose between a prefix and a refusal, so it returns the whole listing when no limit is named and only clamps a limit it was given. Keeping either would have regressed #17954 and hard-failed three in-tree callers that deliberately pass no options (`runtime-file-commands-search-runtime-files.ts:81`, `filesystem-read-handlers.ts:125`, `runtime-file-commands-constructor.ts:41`).
168 lines
5.0 KiB
TypeScript
168 lines
5.0 KiB
TypeScript
import type { ChildProcess } from 'node:child_process'
|
|
import type { SystemSshCommandChannel } from './system-ssh-command'
|
|
|
|
export type ProcessResult = { label: string; stderr: string }
|
|
|
|
/**
|
|
* `timeoutMs` bounds a remote consumer that never returns. Windows PowerShell 5.1 cannot drain a
|
|
* large redirected stdin over a non-pty ssh exec (#16432): the remote process stays alive at idle
|
|
* CPU, writes nothing, and never closes — so without a bound this promise is simply never settled
|
|
* and the caller waits forever with no error to show.
|
|
*/
|
|
export function waitForChannelClose(
|
|
channel: SystemSshCommandChannel,
|
|
label: string,
|
|
timeoutMs?: number
|
|
): Promise<void> {
|
|
return new Promise((resolve, reject) => {
|
|
let stderr = ''
|
|
let timer: ReturnType<typeof setTimeout> | null = null
|
|
const cleanup = (): void => {
|
|
if (timer) {
|
|
clearTimeout(timer)
|
|
timer = null
|
|
}
|
|
channel.stderr.off('data', onStderrData)
|
|
channel.off('error', onError)
|
|
channel.off('close', onClose)
|
|
}
|
|
const settle = (fn: typeof resolve | typeof reject, val?: unknown): void => {
|
|
cleanup()
|
|
fn(val as never)
|
|
}
|
|
if (timeoutMs !== undefined) {
|
|
timer = setTimeout(() => {
|
|
// Settle before closing: the close we request would otherwise come back as a SIGTERM
|
|
// failure and mask the timeout, which is the only diagnosis a wedged remote gives.
|
|
settle(
|
|
reject,
|
|
new Error(
|
|
`${label} timed out after ${timeoutMs}ms with no response from the remote host: ${stderr.trim()}`
|
|
)
|
|
)
|
|
channel.close()
|
|
}, timeoutMs)
|
|
timer.unref?.()
|
|
}
|
|
const onStderrData = (data: Buffer): void => {
|
|
stderr += data.toString('utf-8')
|
|
}
|
|
const onError = (err: Error): void => {
|
|
settle(reject, err)
|
|
}
|
|
const onClose = (code: number | null, signal?: NodeJS.Signals | null): void => {
|
|
if (code !== 0) {
|
|
const detail = code === null ? `signal ${signal ?? 'unknown'}` : `exit ${code}`
|
|
settle(reject, new Error(`${label} failed (${detail}): ${stderr.trim()}`))
|
|
return
|
|
}
|
|
settle(resolve)
|
|
}
|
|
|
|
channel.stderr.on('data', onStderrData)
|
|
channel.on('error', onError)
|
|
channel.on('close', onClose)
|
|
})
|
|
}
|
|
|
|
export function waitForProcess(proc: ChildProcess, label: string): Promise<ProcessResult> {
|
|
return new Promise((resolve, reject) => {
|
|
let stderr = ''
|
|
const cleanup = (): void => {
|
|
proc.stderr?.off('data', onStderrData)
|
|
proc.off('error', onError)
|
|
proc.off('close', onClose)
|
|
}
|
|
const settle = (fn: typeof resolve | typeof reject, val: ProcessResult | Error): void => {
|
|
cleanup()
|
|
fn(val as never)
|
|
}
|
|
const onStderrData = (data: Buffer): void => {
|
|
stderr += data.toString('utf-8')
|
|
}
|
|
const onError = (err: Error): void => {
|
|
settle(reject, err)
|
|
}
|
|
const onClose = (code: number | null): void => {
|
|
if (code !== 0) {
|
|
settle(reject, new Error(`${label} failed (exit ${code}): ${stderr.trim()}`))
|
|
return
|
|
}
|
|
settle(resolve, { label, stderr })
|
|
}
|
|
|
|
proc.stderr?.on('data', onStderrData)
|
|
proc.on('error', onError)
|
|
proc.on('close', onClose)
|
|
})
|
|
}
|
|
|
|
export function killProcess(proc: ChildProcess): void {
|
|
if (proc.exitCode !== null || proc.killed) {
|
|
return
|
|
}
|
|
try {
|
|
proc.kill('SIGTERM')
|
|
} catch {
|
|
// Process may already be dead
|
|
}
|
|
}
|
|
|
|
export async function awaitWithSystemSshAbort<T>(
|
|
signal: AbortSignal | undefined,
|
|
abortChildren: () => void,
|
|
operation: Promise<T>
|
|
): Promise<T> {
|
|
if (!signal) {
|
|
return operation
|
|
}
|
|
let abortReject: ((error: Error) => void) | null = null
|
|
let suppressLateOperationError = false
|
|
const abortPromise = new Promise<never>((_resolve, reject) => {
|
|
abortReject = reject
|
|
})
|
|
const abort = (): void => {
|
|
// Why: abort is connection teardown; do not wait for stubborn system ssh/tar
|
|
// children to emit close after we've already signaled them.
|
|
abortChildren()
|
|
suppressLateOperationError = true
|
|
abortReject?.(
|
|
Object.assign(createAbortError(), {
|
|
// The child was signaled, but this fast abort path intentionally does
|
|
// not wait for process exit; callers holding remote locks must retain them.
|
|
sshChannelCloseConfirmed: false
|
|
})
|
|
)
|
|
}
|
|
signal.addEventListener('abort', abort, { once: true })
|
|
if (signal.aborted) {
|
|
abort()
|
|
}
|
|
try {
|
|
return await Promise.race([
|
|
operation.catch((error: unknown) => {
|
|
if (suppressLateOperationError) {
|
|
return new Promise<never>(() => {})
|
|
}
|
|
throw error
|
|
}),
|
|
abortPromise
|
|
])
|
|
} finally {
|
|
signal.removeEventListener('abort', abort)
|
|
}
|
|
}
|
|
|
|
export function throwIfAborted(signal: AbortSignal | undefined): void {
|
|
if (!signal?.aborted) {
|
|
return
|
|
}
|
|
throw createAbortError()
|
|
}
|
|
|
|
function createAbortError(): Error & { name: string } {
|
|
const error = new Error('System SSH operation was cancelled') as Error & { name: string }
|
|
error.name = 'AbortError'
|
|
return error
|
|
}
|