Files
orca/src/main/github/pr-refresh-coordinator.ts
T
Brennan BensonandMerge Sim b5a85890ac perf(git): bound git subprocess execution with an atomic admission scheduler (#16874)
* perf(git): bound git subprocess execution with an atomic admission scheduler

Field traces (#16038, #11363) show Windows freeze storms driven by unbounded
concurrent git children (12+ at once, 50-65s status convoys for 25+ minutes).
Admit every main-process git child against atomic per-budget base+headroom
counters (general / network / per-route), with reserved interactive capacity,
ordering-only aging, close-bound permit release, a 120s fail-safe read timeout
that feeds scheduler backoff, tier plumbing through every option carrier, and
coalesced+jittered visibility pollers. Killswitch: ORCA_GIT_ADMISSION_DISABLED=1.

Storm harness A/B: max concurrent children 65 -> 6, interactive p95 791ms -> 88ms;
output-parity battery byte-identical with admission on vs off.

* test(git): run the admission output-parity battery on every platform

Parity needs real git, not the storm harness's PATH stub, so it must not share
that file's POSIX gate - Windows is the platform where parity evidence matters.

* fix(git): preserve interactive admission invariants

* perf(git): keep admission queue drains linear

* fix(git): close final admission gaps

* perf(git): bound eligible route selection

* fix(merge): remove unrelated stale snapshot changes

* fix(git): preserve refresh lifecycle authority

* test(git): align admission lifetime contracts

* fix(git): harden admission across runtime paths

* fix(git): restore freshness for bulk status reads

* test(git): repoint delete-dialog source pins after admission plumbing

The hydration effect now orders its targets through
orderDeleteWorktreeStatusHydrationTargets and passes includeLineStats
alongside the abort signal, so both literal anchors stopped matching.
The invariants are unchanged and still pinned: dropping the signal, the
main-worktree/folder filter, or getState-instead-of-subscribe each
still reddens this test.

* Fix git admission tier propagation and lock ordering

Decode optional Git status tiers permissively and default runtime RPC status reads to the status lane while preserving renderer caller intent.

Acquire the FETCH_HEAD mutex before atomic admission so same-repository fetch waiters hold no global or route permits.

Preserve automatic pull-request refresh reasons, keep explicit hosted-review refreshes interactive, remove the dead candidate tier, and keep relay scheduling unchanged.

Use tier-aware status lease keys because a shared lease cannot be safely promoted after its admission request is queued or granted.

* test: align expectations with admission plumbing

* refactor(child-process): move the process contract types to process-spec

run-process.ts crossed its line cap after gaining the termination observer;
the public types and defaults move out with re-exports so no caller changes.

* chore: restore pnpm-lock.yaml to main (unintended local drift)

---------

Co-authored-by: Merge Sim <sim@local>
2026-08-30 14:19:05 -07:00

200 lines
6.4 KiB
TypeScript

import type {
GitHubPRRefreshCandidate,
GitHubPRRefreshReason,
PRRefreshOutcome
} from '../../shared/github/pull-request-refresh-types'
import { getPRForBranchOutcome } from './client'
import {
aliasFromCandidate,
hostedReviewOptionArgs,
MANUAL_MERGEABILITY_PENDING_REFRESH_MS,
refreshKey,
shouldBroadcastQueued,
validateCandidate
} from './pr-refresh-candidate-policy'
import {
PRRefreshEventPublisher,
type PRRefreshOutcomeObserver
} from './pr-refresh-event-publisher'
import { PRRefreshPacing } from './pr-refresh-pacing'
import { PRRefreshQueue } from './pr-refresh-queue'
import { PRRefreshQueueDrainer } from './pr-refresh-queue-drainer'
import { prRefreshRateLimitPausedUntil } from './pr-refresh-rate-limit-gate'
import { PRRefreshRetryState } from './pr-refresh-retry-state'
import { PRRefreshVisibility } from './pr-refresh-visibility'
const retry = new PRRefreshRetryState()
const queue = new PRRefreshQueue((key) => retry.reset(key))
const pacing = new PRRefreshPacing()
const visibility = new PRRefreshVisibility()
const events = new PRRefreshEventPublisher()
const drainer = new PRRefreshQueueDrainer(queue, pacing, visibility, retry, events)
export function setPRRefreshOutcomeObserver(observer: PRRefreshOutcomeObserver | null): void {
events.setOutcomeObserver(observer)
}
function removeInvisibleVisibleRefreshes(): void {
const removed = queue.removeInvisibleVisibleEntries((key) => visibility.has(key))
for (const entry of removed) {
events.broadcast({
aliases: Array.from(entry.aliases.values()),
reason: 'visible',
status: 'skipped',
skippedReason: 'fresh'
})
}
}
export function clearVisiblePRRefreshWindow(windowId: number): void {
const hadVisibleRefreshes = visibility.clearWindow(windowId)
pacing.clearActiveBurstWindow(windowId)
if (hadVisibleRefreshes) {
removeInvisibleVisibleRefreshes()
}
}
export function pruneWorktreePRRefreshAliases(worktreeId: string): void {
queue.pruneWorktreeAliases(worktreeId)
}
export function enqueuePRRefresh(
candidate: GitHubPRRefreshCandidate,
reason: GitHubPRRefreshReason,
priority = 0,
windowId?: number
): void {
const alias = aliasFromCandidate(candidate)
const key = refreshKey(candidate)
const skippedReason = validateCandidate(candidate)
if (skippedReason) {
queue.removeInvalidAlias(key, alias)
events.record('skipped', reason, skippedReason)
events.broadcast({ aliases: [alias], reason, status: 'skipped', skippedReason })
return
}
const enqueued = queue.enqueue(candidate, reason, priority, windowId)
events.record(enqueued.coalesced ? 'coalesced' : 'enqueued', reason)
if (shouldBroadcastQueued(reason, enqueued.dueAt)) {
events.broadcast({ aliases: [enqueued.alias], reason, status: 'queued' })
}
drainer.schedule()
}
export function reportVisiblePRRefreshCandidates(
candidates: GitHubPRRefreshCandidate[],
generation: number,
windowId: number
): void {
if (!visibility.report(candidates, generation, windowId)) {
return
}
removeInvisibleVisibleRefreshes()
for (const candidate of candidates) {
enqueuePRRefresh(candidate, 'visible', 40, windowId)
}
}
export function _getVisiblePRRefreshWindowCountForTests(): number {
return visibility.windowCount
}
export function _getPRRefreshErrorBackoffCountForTests(): number {
return retry.errorBackoffCount
}
export function _getPRRefreshQueueSizeForTests(): number {
return queue.size
}
export function _getPRRefreshAliasCountForTests(key: string): number {
return queue.aliasCount(key)
}
export async function refreshPRNow(
candidate: GitHubPRRefreshCandidate,
reason: GitHubPRRefreshReason = 'manual'
): Promise<PRRefreshOutcome> {
const alias = aliasFromCandidate(candidate)
const key = refreshKey(candidate)
const existing = queue.get(key)
const aliasMap = new Map(existing ? existing.aliases : [])
aliasMap.set(alias.cacheKey, alias)
const aliases = Array.from(aliasMap.values())
const skippedReason = validateCandidate(candidate)
if (skippedReason) {
queue.removeInvalidAlias(key, alias)
const outcome: PRRefreshOutcome = {
kind: 'upstream-error',
errorType: 'unknown',
message: `Cannot refresh PR for this worktree: ${skippedReason}`,
fetchedAt: Date.now()
}
events.broadcast({ aliases: [alias], reason, status: 'skipped', skippedReason })
return outcome
}
const primaryGateUntil = await prRefreshRateLimitPausedUntil(candidate, false)
const gateUntil = Math.max(primaryGateUntil ?? 0, retry.manualGateUntil(key))
if (gateUntil > Date.now()) {
queue.set(key, {
key,
candidate,
aliases: aliasMap,
reason,
priority: 40,
dueAt: gateUntil,
queuedAt: queue.nextOrder()
})
events.broadcast({
aliases,
reason,
status: 'paused',
pausedUntil: gateUntil,
skippedReason: 'rate-limit'
})
drainer.schedule(Math.max(1_000, gateUntil - Date.now()))
return {
kind: 'upstream-error',
errorType: 'rate_limited',
message: 'GitHub is temporarily limiting requests. Try again after the limit resets.',
fetchedAt: Date.now(),
nextAutoRetryAt: gateUntil,
retryDisabledUntil: gateUntil
}
}
queue.delete(key)
const requestSequence = events.nextSequence()
const requestStartedAt = Date.now()
events.broadcast({ aliases, reason, status: 'in-flight', requestStartedAt }, requestSequence)
const outcome = await getPRForBranchOutcome(
candidate.repoPath,
candidate.branch,
candidate.linkedPRNumber ?? null,
candidate.connectionId ?? null,
candidate.linkedPRNumber == null ? (candidate.fallbackPRNumber ?? null) : null,
...hostedReviewOptionArgs(candidate, reason)
)
let plannedRetryAt: number | undefined
let broadcastOutcome = outcome
if (outcome.kind === 'upstream-error' && visibility.has(key)) {
plannedRetryAt = retry.nextVisibleErrorRetryAt(key)
broadcastOutcome = retry.withErrorSchedule(outcome, plannedRetryAt)
}
events.observe(candidate, outcome)
retry.noteManualGate(key, broadcastOutcome)
events.broadcast(
{ aliases, reason, outcome: broadcastOutcome, requestStartedAt },
requestSequence
)
drainer.scheduleVisibleFollowUp(key, candidate, outcome, 40, aliases, undefined, {
plannedRetryAt,
...(reason === 'manual'
? { pendingMergeabilityDelayMs: MANUAL_MERGEABILITY_PENDING_REFRESH_MS }
: {})
})
return broadcastOutcome
}