Files
orca/src/relay/relay-client-resync-marker.ts
T
Neil 510305e574 fix(relay): signal capacity loss instead of dropping, hanging, or truncating (#17870)
Three failures with one shape: a payload past a fixed capacity was met with
silence, with a wait that never ends, or with a prefix presented as a whole.

**The workspace snapshot was silently dropped.** `workspace.changed` carries the
tab/session list, and a snapshot past the producer frame capacity (12288 B on a
Node <=21 remote) was dropped with only a relay stderr line, so the client kept a
stale list forever. The relay now publishes per client and, for a client whose
sink refused the frame, sends a compact `workspace.stale` marker on the control
lane; the client re-reads through `workspace.get`, whose lane is budgeted in
megabytes rather than in one producer frame. A new JSON-RPC notification rather
than a new field on `workspace.changed`: `normalizeSnapshot(undefined, ns)` yields
revision 0 and an empty session, so a Rule-1 field would make an old client
replace its tab list with nothing — worse than the drop. An old client ignores the
unknown method and is exactly where it is today. The marker retention/retry
machinery is extracted from the `fs.changed` overflow path and shared by both.

**The Windows upload hung, and the fix for it could truncate.** `#16432` was
attributed to `[Console]::In.ReadToEnd()` materializing the base64 bundle. That is
not what the reporter measured: he also measured
`new IO.StreamReader([Console]::OpenStandardInput())` — an incremental reader —
hanging at 1 MB. The limit is in the stdin the host hands PowerShell over a
non-pty ssh exec, not in the string the script builds.

- `uploadFileViaSystemSsh` — the user file-import path — was piping a whole file
  into one Windows stdin, unchunked and untimed. That is the path large files
  take; it now chunks into 32 KB writes and bounds each wait.
- The Windows directory upload reuses that single-file path rather than repeating
  a weaker copy of chunk-read + write-buffer; the `ino`/`dev` TOCTOU verification
  comes with it.
- A Windows write needing more than one exec lands on a `.orca-partial` staging
  path and is published by rename, so a failed chunk cannot leave a truncated
  artifact under the real name. `exclusive` is enforced once at the rename, not on
  the first chunk, where a retry met its own leftovers.
- The mkdir batch reads stdin through the stream reader the reporter measured
  surviving 50 KB, not `[Console]::In`, which he measured wedging at that size.
- `waitForChannelClose` takes an optional bound. A wedged PowerShell stays alive
  at idle CPU and never closes, so without one the promise is simply never
  settled and the caller waits forever with no error to show.

**Quick Open showed a prefix as the whole workspace.** The mechanism "a full page
means there is more" only works if the caller named the cap, and the failing UI
named none — it hardcoded `truncated: false`. Quick Open now names
`QUICK_OPEN_LISTING_MAX_RESULTS` on both the Electron IPC hop and the runtime-RPC
hop (the field #17954 added to `files.listAll`), and reads a full page as
truncation. The local hop honours the cap too, which it previously ignored.

Rebase note on `fs.listFiles`: an earlier revision of this work also clamped the
host unconditionally, and #17934 escalated an uncapped request to an explicit
error. #17954 has since landed and made an oversized reply streamable, which
removes the premise — the host no longer has to choose between a prefix and a
refusal, so it returns the whole listing when no limit is named and only clamps a
limit it was given. Keeping either would have regressed #17954 and hard-failed
three in-tree callers that deliberately pass no options
(`runtime-file-commands-search-runtime-files.ts:81`,
`filesystem-read-handlers.ts:125`, `runtime-file-commands-constructor.ts:41`).
2026-09-02 15:42:08 -07:00

169 lines
6.1 KiB
TypeScript

import type { RelayDispatcher } from './dispatcher'
type RetainedMarker = { params: Record<string, unknown>; estimatedBytes: number }
type MarkerState = {
method: string
// Why: one outstanding marker per (client, key) keeps sustained backpressure bounded.
inFlight: Set<string>
// Rejected markers, retained per client so they can be republished when the control lane frees up.
pending: Map<number, Map<string, RetainedMarker>>
capacityUnsubscribes: Map<number, () => void>
}
/**
* Publishes a small "your view is stale, re-read it" notification on the control lane after a
* producer-lane payload was refused. Shared by every producer whose oversized frame would otherwise
* desync a client silently; the marker is coalesced per key and retried on capacity, never dropped.
*/
export type RelayClientResyncMarkerPublisher = {
emit(clientId: number, markerKey: string, params: Record<string, unknown>): void
forgetClient(clientId: number): void
}
export function createRelayClientResyncMarkerPublisher(
dispatcher: RelayDispatcher,
method: string
): RelayClientResyncMarkerPublisher {
const state: MarkerState = {
method,
inFlight: new Set(),
pending: new Map(),
capacityUnsubscribes: new Map()
}
// In-flight keys need no sweep here: closing a client settles every queued and written frame first.
dispatcher.onClientDetached((clientId) => {
// Not every detach retires the id: invalidateClient() detaches the primary without removing it and
// setWrite() revives it, so dropping the markers here would desync the state the reconnect restores.
if (dispatcher.isClientAttached(clientId)) {
return
}
forgetClientMarkers(state, clientId)
})
return {
emit(clientId, markerKey, params) {
const retained = state.pending.get(clientId)?.get(markerKey)
if (retained) {
// Latest generation wins: a retained marker that has not been sent yet must not replay a
// projection the producer has already moved past.
retained.params = params
retained.estimatedBytes = dispatcher.notificationFrameBytes(method, params)
return
}
// Per key, never per client alone: an outstanding marker for one subject must not suppress another's resync.
if (state.inFlight.has(markerId(clientId, markerKey))) {
return
}
publishMarker(dispatcher, state, clientId, markerKey, params)
},
forgetClient(clientId) {
forgetClientMarkers(state, clientId)
}
}
}
function markerId(clientId: number, markerKey: string): string {
return `${clientId} ${markerKey}`
}
function forgetClientMarkers(state: MarkerState, clientId: number): void {
// Unsubscribe first so no re-entrant flush can observe a half-cleared client.
state.capacityUnsubscribes.get(clientId)?.()
state.capacityUnsubscribes.delete(clientId)
state.pending.delete(clientId)
}
// Why: the control lane — on the producer lane the marker would hit the same full queue that just
// rejected the payload and be dropped, silently desyncing the client.
function publishMarker(
dispatcher: RelayDispatcher,
state: MarkerState,
clientId: number,
markerKey: string,
params: Record<string, unknown>,
estimatedBytes?: number
): void {
const key = markerId(clientId, markerKey)
const frameBytes = estimatedBytes ?? dispatcher.notificationFrameBytes(state.method, params)
state.inFlight.add(key)
let settled = false
const accepted = dispatcher.tryNotifyClient(
clientId,
state.method,
params,
(result) => {
// Settles on write, drop, or client close, so the slot can never leak.
settled = true
state.inFlight.delete(key)
if (result.ok) {
return
}
// A frame the sink never wrote leaves the client just as desynced as a rejected one — setWrite
// fails every queued and in-flight frame this way. Retain unconditionally: a real detach clears
// it through onClientDetached, which fires after this settlement.
retainMarker(dispatcher, state, clientId, markerKey, params, frameBytes)
},
{ controlOverflow: 'reject' }
)
if (accepted || settled) {
return
}
// Admission rejection has no settlement callback: retain the marker instead of desyncing the client.
state.inFlight.delete(key)
retainMarker(dispatcher, state, clientId, markerKey, params, frameBytes)
}
function retainMarker(
dispatcher: RelayDispatcher,
state: MarkerState,
clientId: number,
markerKey: string,
params: Record<string, unknown>,
estimatedBytes: number
): void {
if (!state.capacityUnsubscribes.has(clientId)) {
const unsubscribe = dispatcher.onClientCapacity(clientId, () =>
flushPendingMarkers(dispatcher, state, clientId)
)
if (!unsubscribe) {
// The client went away between admission and arming, so there is nothing left to resync.
return
}
state.capacityUnsubscribes.set(clientId, unsubscribe)
}
const retained = state.pending.get(clientId)
if (retained) {
retained.set(markerKey, { params, estimatedBytes })
return
}
state.pending.set(clientId, new Map([[markerKey, { params, estimatedBytes }]]))
}
function flushPendingMarkers(
dispatcher: RelayDispatcher,
state: MarkerState,
clientId: number
): void {
const retained = state.pending.get(clientId)
if (!retained) {
return
}
for (const [markerKey, marker] of Array.from(retained)) {
// Capacity fires on every lane; retry only frames the control queue can admit now.
if (!dispatcher.canAdmitControlFrame(clientId, marker.estimatedBytes)) {
continue
}
// Drop before republishing so a synchronous settlement cannot see the marker as still pending —
// and skip keys a re-entrant flush already took, which would otherwise send the marker twice.
if (!retained.delete(markerKey)) {
continue
}
publishMarker(dispatcher, state, clientId, markerKey, marker.params, marker.estimatedBytes)
}
// Identity check: a re-entrant flush may have retired this set and armed a fresh one to keep.
if (retained.size > 0 || state.pending.get(clientId) !== retained) {
return
}
forgetClientMarkers(state, clientId)
}