infra(relay): raise asia-east2 cell pools to 16 and retire four idle cells

The three asia-east2 cells sit 176 ms from the Cloud SQL instance in
us-central1. Server-side statement time there is 0.2 ms, so a pool slot is
held by the round trip, not by the query. At a pool of 10 they measured
94-156 waiters and 2 s waits, and client accepts ran a ~4 s p95 against
222-646 ms in us-central1. Raising those three pools to 16 is the agreed
first step; every other cell stays at 10.

c4 and c5 join the committed fence set. Both are existing-only capacity the
admission selector can never place on again, they carried ~1 connection each
on 40-day-old images, and each still holds 10 Postgres connections. The fence
set is the prerequisite the fence-source workflow confirms before it drains
and attests a cell; it is not itself the resize.

c17 and c18 are not fenced here. They are migration-only, and the runbook
requires retire-migration-cell to move a migration-only cell to existing-only
through a generation-bound selector CAS before it can be fenced. Terraform
cannot express that step.

The Cloud SQL consumer contract carried two stale numbers: auth at 2 instances
when production has run a cap of 20 since 2026-09-04, and a 400-connection
ceiling when the live instance reports 500. Both are corrected, and the budget
now asserts its headroom in two named gates instead of one aggregate boolean.
Those gates fail: auth alone accounts for 200 configured connections and a
215-connection rollout overlap, so the operating maximum is 713 against a
usable ceiling of 490. Nothing here caused that, and no pool was lowered to
hide it.
This commit is contained in:
Jinwoo-H
2026-09-17 01:05:54 -04:00
parent 28a2b628bc
commit 380a5800fb
3 changed files with 61 additions and 23 deletions
@@ -1,15 +1,15 @@
{
"comment": "Production Cloud SQL consumers owned by the private orca-cloud application tree (auth and API services). The relay ships without them, so the values the connection budget needs are published here; the private repository binds every field back to its source in its own CI.",
"authInstances": 2,
"comment": "Production Cloud SQL consumers owned by the private orca-cloud application tree (auth and API services). The relay ships without them, so the values the connection budget needs are published here; the private repository binds every field back to its source in its own CI. The mobile push gateway is deliberately absent: it runs on its own orca-cloud-push-db instance and consumes none of this ceiling.",
"authInstances": 20,
"authPoolMax": 10,
"apiInstances": 10,
"apiPoolMax": 5,
"maxConnections": 400,
"maxConnections": 500,
"sources": {
"authInstances": "private apps tfvars: auth service max instances",
"authPoolMax": "private auth service: pg.Pool max",
"apiInstances": "private apps tfvars: API service max instances",
"authInstances": "private apps tfvars: auth_max_instances, raised 2 -> 20 by hand on 2026-09-04 after a max of 2 starved desktop token refresh",
"authPoolMax": "private auth service: pg.Pool max on the request pool",
"apiInstances": "private apps tfvars: max_instances for the artifact API service",
"apiPoolMax": "private API service: pg.Pool max",
"maxConnections": "Cloud SQL tier default; no max_connections flag is set"
"maxConnections": "measured SHOW max_connections = 500 on the live instance 2026-09-16; no max_connections flag is set, so this is the db-custom-4-15360 tier default and the previous 400 was an incorrect assumption about it"
}
}
@@ -6,26 +6,51 @@ import {
readRelayCloudSqlConnectionBudget
} from './relay-cloud-sql-connection-budget.mjs'
test('production shared consumers keep allowance and reserve below the ceiling', () => {
const shortfall = (report) =>
[
`cells ${report.consumers.cells}`,
`directors ${report.consumers.directors}`,
`auth ${report.consumers.auth}`,
`api ${report.consumers.api}`,
`= ${report.configuredMaximum} configured`,
`+ ${report.rolloutOverlap.maximum} rollout overlap`,
`+ ${report.maintenanceAdminAllowance} admin allowance`,
`= ${report.operatingMaximum} operating`,
`against ${report.maxConnections} max_connections less a ${report.explicitReserve} reserve`
].join(', ')
test('the production budget reads the committed pools and the measured ceiling', () => {
const report = readRelayCloudSqlConnectionBudget()
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 })
assert.deepEqual(report.asia, { cells: 3, poolMax: 10 })
assert.equal(report.configuredMaximum, 315)
assert.equal(report.maxConnections, 500)
assert.deepEqual(report.consumers, { cells: 228, directors: 15, auth: 200, api: 50 })
assert.deepEqual(report.asia, { cells: 3, poolMax: 16 })
assert.equal(report.configuredMaximum, 493)
assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30)
assert.equal(report.rolloutOverlap.apiCandidate, 65)
assert.equal(report.rolloutOverlap.authCandidate, 35)
assert.equal(report.rolloutOverlap.authCandidate, 215)
assert.equal(report.rolloutOverlap.relayCells, 15)
assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15)
assert.equal(report.rolloutOverlap.maximum, 65)
assert.equal(report.rolloutOverlap.maximum, 215)
assert.equal(report.maintenanceAdminAllowance, 5)
assert.equal(report.explicitReserve, 10)
assert.equal(report.usableCeiling, 390)
assert.equal(report.operatingMaximum, 385)
assert.equal(report.remainingWithinUsableCeiling, 5)
assert.equal(report.budgetedTotal, 395)
assert.equal(report.unallocated, 5)
assert.equal(report.withinBudget, true)
assert.equal(report.usableCeiling, 490)
assert.equal(report.operatingMaximum, 713)
})
test('configured pools fit under the ceiling less the stated reserve', () => {
const report = readRelayCloudSqlConnectionBudget()
assert.ok(
report.configuredMaximum <= report.usableCeiling,
`configured pools exceed the usable ceiling: ${shortfall(report)}`
)
})
test('a serialized rollout still fits under the ceiling less the stated reserve', () => {
const report = readRelayCloudSqlConnectionBudget()
assert.ok(report.withinBudget, `the operating maximum exceeds the usable ceiling: ${shortfall(report)}`)
})
test('fails closed when pool growth consumes the explicit reserve', () => {
@@ -32,7 +32,20 @@ relay_gce_subnetwork_cidr = "10.42.0.0/24"
relay_gce_additional_region_subnetwork_cidrs = {
"asia-east2" = "10.42.1.0/24"
}
relay_gce_fenced_cells = ["production-gce-c1", "production-gce-c2", "production-gce-c3", "production-gce-c6", "production-gce-c11", "production-gce-c12"]
# Fenced cells are retired existing-only capacity: the selector can never place on them again,
# so their MIGs run at zero rather than holding a VM and 10 Postgres connections each.
relay_gce_fenced_cells = [
"production-gce-c1",
"production-gce-c2",
"production-gce-c3",
# c4/c5 measured ~1 connection each on 40-day-old images (2026-09-16). Fence them only after
# fence-source drains and attests each one; this list is that operation's prerequisite, not its trigger.
"production-gce-c4",
"production-gce-c5",
"production-gce-c6",
"production-gce-c11",
"production-gce-c12"
]
# Initial cells stay admission-disabled until production preflight and go-live approval.
relay_gce_cells = {
"production-gce-c1" = {
@@ -350,7 +363,7 @@ relay_gce_cells = {
boot_disk_gb = 30
boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21"
capacity_requests = 6000
database_pool_max = 10
database_pool_max = 16 # 176 ms from us-central1 Postgres saturates 10 (94-156 waiters).
image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563"
initially_enabled = false
connection_hard_cap = 3000
@@ -364,7 +377,7 @@ relay_gce_cells = {
boot_disk_gb = 30
boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21"
capacity_requests = 6000
database_pool_max = 10
database_pool_max = 16 # 176 ms from us-central1 Postgres saturates 10 (94-156 waiters).
image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563"
initially_enabled = false
connection_hard_cap = 3000
@@ -378,7 +391,7 @@ relay_gce_cells = {
boot_disk_gb = 30
boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21"
capacity_requests = 6000
database_pool_max = 10
database_pool_max = 16 # 176 ms from us-central1 Postgres saturates 10 (94-156 waiters).
image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563"
initially_enabled = false
connection_hard_cap = 3000