mirror of
https://github.com/stablyai/orca.git
synced 2026-10-02 00:02:05 +00:00
* fix(relay): wait out a cold proxy at boot instead of exiting the cell A cell container starts its relay process beside a cloud-sql-proxy that is itself still dialling. The first pool acquire therefore competes with a proxy cold start, and the 2s connect timeout that protects the request path fires before the proxy is listening. `openRelayDatabase` rejects out of the region backfill, the top-level await rejects, and the process exits; COS restarts the container and the next boot succeeds 1-3s later. The 2026-09-18 fleet roll saw 0-7 of these per cell, including on cells with zero hosts, so it is a property of the boot sequence rather than of database load. The boot open now retries on transient errors only, inside a 45s wall-clock window with exponential backoff from 250ms to 4s. The classifier is the one the request path already uses, so a rejected credential or a bad URL still exits on the first attempt. Each wait logs `orca_relay_boot_database_retry` and a give-up logs `orca_relay_boot_database_failed`, both with the bounded error category, so a rollout can tell a slow boot from a stuck one without reading container exit codes. The bounded startup retry is lifted out of `reconcileCellAdmissionAtStartup`, which had the same loop; its attempt budget, flat delay, and both log events are unchanged (a flat delay is a cap equal to the base). * fix(relay): retry the boot open only when Postgres is unreachable The boot open re-runs the schema apply, and applyPostgresSchema refuses to repeat a DDL lock timeout on purpose: relation locks are granted in queue order, so a repeat parks every writer behind the same statement again. Gating the boot retry on the full request-path classifier would have re-queued it up to 16 times in 45s on sustained 55P03 - the mechanism behind the 2026-09-16 outage. The boot call site now has its own predicate: pool connect failures (both connect-timeout messages and an acquire-marked early-ended socket) plus 08001 and 08006. Lock and overload SQLSTATEs - 55P03, 57014, 53300 - exit on the first attempt. The retry predicate moves onto the policy because what a step re-runs, not the request path, decides what it may repeat; the startup reconcile keeps the full classifier, which is what lets it wait out 55P03.
65 lines
1.8 KiB
TypeScript
65 lines
1.8 KiB
TypeScript
import { isPostgresPoolConnectTimeout } from './postgres-pool-pressure.js'
|
|
|
|
type QueryFailurePhase = 'acquire' | 'execute'
|
|
|
|
const ERROR_CODES = new Set([
|
|
'57014',
|
|
'55P03',
|
|
'40P01',
|
|
'40001',
|
|
'53300',
|
|
'57P01',
|
|
'57P02',
|
|
'57P03',
|
|
'08000',
|
|
'08001',
|
|
'08003',
|
|
'08006',
|
|
'ECONNRESET',
|
|
'ECONNREFUSED',
|
|
'ETIMEDOUT',
|
|
'EPIPE'
|
|
])
|
|
|
|
// A recognised SQLSTATE or errno, or 'unknown': whatever else a driver attached
|
|
// to `code` is not a bounded log category.
|
|
export function postgresErrorCodeCategory(error: unknown): string {
|
|
const code =
|
|
typeof error === 'object' && error !== null && 'code' in error ? error.code : undefined
|
|
return typeof code === 'string' && ERROR_CODES.has(code) ? code : 'unknown'
|
|
}
|
|
|
|
export function reportPostgresQueryFailure(input: {
|
|
error: unknown
|
|
phase: QueryFailurePhase
|
|
sql: string
|
|
// The routing verdict, supplied by the caller that owns it.
|
|
transient: boolean
|
|
elapsedMs: number
|
|
pool: { totalCount: number; idleCount: number; waitingCount: number }
|
|
}): void {
|
|
// Emit only bounded categories: error messages and SQL can contain credentials or identities.
|
|
try {
|
|
const code = postgresErrorCodeCategory(input.error)
|
|
const connectionTimeout = isPostgresPoolConnectTimeout(input.error)
|
|
console.warn(
|
|
JSON.stringify({
|
|
event: 'orca_relay_postgres_query_failed',
|
|
phase: input.phase,
|
|
operation: /^\s*WITH\s+assignment_state\s+AS\s+MATERIALIZED\b/i.test(input.sql)
|
|
? 'control-renewal'
|
|
: 'other',
|
|
code,
|
|
connectionTimeout,
|
|
transient: input.transient,
|
|
elapsedMs: Math.max(0, Math.round(input.elapsedMs)),
|
|
poolTotal: input.pool.totalCount,
|
|
poolIdle: input.pool.idleCount,
|
|
poolWaiting: input.pool.waitingCount
|
|
})
|
|
)
|
|
} catch {
|
|
// Diagnostics must not replace the original database failure.
|
|
}
|
|
}
|