fix(daemon): name the socket probe default and correct the grace-retry rationale

Answers the review round on the budget accounting: the outer-catch endpoint
rescue is deliberately outside it, and the preflight clamp no longer duplicates
probeDaemonSocket's default as a bare literal.
This commit is contained in:
Neil
2026-08-31 02:00:50 -07:00
parent 9a956551ae
commit df1ca9ea33
3 changed files with 24 additions and 5 deletions
+7 -1
View File
@@ -46,8 +46,14 @@ export function daemonLogArgs(): string[] {
return disabled === '1' || disabled === 'true' ? [] : ['--log-file', getDaemonLogFilePath()]
}
/** Named so a caller clamping this probe to a deadline cannot silently decouple from its default. */
export const DAEMON_SOCKET_PROBE_TIMEOUT_MS = 1_000
// Why: a socket that accepts a connection proves a daemon survived a previous app session and can be reused.
export function probeDaemonSocket(socketPath: string, timeoutMs = 1000): Promise<boolean> {
export function probeDaemonSocket(
socketPath: string,
timeoutMs = DAEMON_SOCKET_PROBE_TIMEOUT_MS
): Promise<boolean> {
const { promise, resolve } = Promise.withResolvers<boolean>()
if (process.platform !== 'win32' && !existsSync(socketPath)) {
resolve(false)
@@ -42,6 +42,11 @@ export const DAEMON_RECOVERY_PROBE_MS = 8_000
* timeout, and 5s for the adoption lease and adapter connects — those two run against a daemon
* that has just reported ready over IPC, so they get a realistic allowance, not the client's
* unbudgeted 4x5s. 60 - 25.5 leaves 34.5s; take 32s and keep the rest as margin.
*
* Outside that accounting by design: the launcher's outer-catch endpoint rescue, which only
* runs once the replacement has already failed. Clamping it to the remainder is what turned a
* recoverable degraded adoption into total daemon loss, so it keeps its own probe default and
* the gate may fail open ahead of it — no worse than not rescuing, and better whenever it wins.
*/
export const DAEMON_RECOVERY_BUDGET_MS = 32_000
@@ -3,7 +3,11 @@ import type { DaemonReplaceReason } from '../../shared/daemon-lifecycle-telemetr
import { isDaemonStaleForCurrentBundle } from './daemon-bundle-staleness'
import { DaemonEndpointOwnershipError } from './daemon-endpoint-adoption'
import { checkDaemonHealth, getMacDaemonSystemResolverHealth } from './daemon-health'
import { getAliveDaemonSessionCount, probeDaemonSocket as probeSocket } from './daemon-launch-paths'
import {
DAEMON_SOCKET_PROBE_TIMEOUT_MS,
getAliveDaemonSessionCount,
probeDaemonSocket as probeSocket
} from './daemon-launch-paths'
import { trackDaemonReplaced } from './daemon-lifecycle-event'
import { getDaemonLaunchIdentity } from './daemon-pid-identity'
import { cleanupDaemonForProtocol } from './daemon-protocol-cleanup'
@@ -12,8 +16,9 @@ import { killStaleDaemon } from './daemon-stale-kill'
import { getMacDaemonTccAttributionHealth } from './daemon-tcc-attribution'
import { PROTOCOL_VERSION } from './types'
// Why a count on top of the wall clock: DAEMON_RECOVERY_BUDGET_MS ends the grace, but a probe
// that fails instantly (nothing listening) would spin hot for the whole budget without this.
// Why a count on top of the wall clock: DAEMON_RECOVERY_BUDGET_MS ends the grace, but a socket
// that accepts and then resets the hello answers both probes instantly, so without this the loop
// would spin hot for the whole budget.
export const WEDGED_DAEMON_GRACE_RETRIES = 11
type PreserveDaemon = (mode?: 'degraded-new-pty-fallback') => Promise<DaemonProcessHandle>
@@ -157,7 +162,10 @@ export async function prepareDaemonReplacement(
health !== 'rejected' &&
graceRetry < WEDGED_DAEMON_GRACE_RETRIES &&
Date.now() < recoveryDeadlineMs &&
(await probeSocket(socketPath, Math.max(1, Math.min(1000, recoveryDeadlineMs - Date.now()))))
(await probeSocket(
socketPath,
Math.max(1, Math.min(DAEMON_SOCKET_PROBE_TIMEOUT_MS, recoveryDeadlineMs - Date.now()))
))
) {
liveSessionCount = await getAliveDaemonSessionCount(socketPath, tokenPath, recoveryDeadlineMs)
graceRetry++