refactor(push): remove redundant cloud storage and rollout coupling

This commit is contained in:
Jinwoo-H
2026-09-09 16:03:49 -04:00
parent a4fead9cb4
commit bba1b53447
28 changed files with 360 additions and 621 deletions
+4 -8
View File
@@ -16,10 +16,9 @@ permissions:
contents: read
id-token: write
# Push uses dedicated Cloud SQL; its schema startup and connection-budget rollout intentionally
# retain the production rollout coordination group and lease shared with Relay.
# Serialize push traffic changes independently of Relay and the shared database.
concurrency:
group: production-cloud-sql-rollout
group: production-push-rollout
cancel-in-progress: false
defaults:
@@ -85,10 +84,7 @@ jobs:
- name: Configure Docker auth
run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet
# Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance,
# and a multi-minute image build inside the lease blocks every relay deploy and rehome for
# its duration. The lease below covers exactly the connection-budget window: deploy, probe,
# shift.
# Building an image does not need the deployment lease.
- name: Build and publish the immutable gateway image
shell: bash
run: |
@@ -122,7 +118,7 @@ jobs:
- uses: ./.github/actions/cloud-sql-rollout-lease
with:
bucket: onorca-cloud-terraform-state
object: terraform/state/cloud-sql-rollout/production.lock
object: terraform/state/push-rollout/production.lock
# Why: the candidate inherits the serving revision's scaling. A serving revision that has
# drifted below the floor would hand the candidate a cold start on every notification, and
+2 -2
View File
@@ -82,8 +82,8 @@ surface: publish and deploy the director, roll GCE cell capacity, operate Asia
admission and regional rehoming, prove staging capacity, monitor production,
power staging up and down, and deploy the mobile push gateway.
`.github/actions/cloud-sql-rollout-lease` is the compare-and-swap lease that
serializes every rollout against the shared Cloud SQL instance, the push
gateway deploy included.
serializes rollouts against the shared Cloud SQL instance. Push reuses that
action with its own lease object and deployment concurrency group.
Every one of them is inert. Each top-level job is gated on
`vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true'`, a repository variable that is
@@ -199,3 +199,58 @@ it('does not resurrect an in-flight alert after a dismissal and transient provid
expect(await store.claim()).toBeNull()
expect(await store.pendingCount('phone')).toBe(0)
})
it.each([false, true])(
'normalizes default alert kind (explicit first: %s)',
async (explicitFirst) => {
const { db, store } = await fixture()
const { kind: _kind, ...implicit } = notification(1)
const explicit = { kind: 'alert' as const, ...implicit }
for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) {
expect(await store.accept('host', 'phone', event)).toBe('queued')
}
expect(await store.pendingCount('phone')).toBe(1)
expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(1)
expect(await store.accept('host', 'phone', { ...explicit, body: 'changed' })).toBe('error')
expect(await store.accept('host', 'phone', { ...implicit, kind: 'dismiss' })).toBe('queued')
expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(2)
}
)
it('fences late renew and finish after an expired claim is dismissed', async () => {
const { db, store, advance } = await fixture()
const alert = notification(1)
await store.accept('host', 'phone', alert)
const stale = (await store.claim())!
advance(DELIVERY_LEASE_MS)
await store.accept('host', 'phone', {
...notification(2, 'dismiss'),
notificationId: alert.notificationId
})
const read = async () =>
(
await db.query(
'SELECT state, payload_json, lease_until FROM push_delivery_batches WHERE batch_id = ?',
[stale.id]
)
)[0]
const cancelled = await read()
expect(cancelled).toMatchObject({ state: 'dismissed', payload_json: '{}' })
await store.renew(stale)
expect(await read()).toEqual(cancelled)
await store.finish(stale, 1000)
expect(await read()).toEqual(cancelled)
await store.finish(stale)
expect(await read()).toEqual(cancelled)
const dismissal = (await store.claim())!
expect(dismissal.notification.kind).toBe('dismiss')
await store.finish(dismissal)
await store.accept('host', 'phone', notification(3))
const fresh = (await store.claim())!
await store.finish(fresh, 1000)
advance(1000)
const retry = (await store.claim())!
expect(retry.id).toBe(fresh.id)
await store.finish(retry)
expect(await store.claim()).toBeNull()
})
+6 -4
View File
@@ -34,8 +34,10 @@ export class DurablePushStore {
JSON.stringify([host, kind, notification.notificationEpoch, notification.notificationSeq])
)
.digest('hex')
const { sound: _sound, ...content } = notification
const fingerprint = createHash('sha256').update(JSON.stringify(content)).digest('hex')
const { sound: _sound, kind: _kind, ...content } = notification
const fingerprint = createHash('sha256')
.update(JSON.stringify({ kind, ...content }))
.digest('hex')
return this.database.transaction(async (tx) => {
await tx.lockQuotaScope(`push-events:${host}`)
const [existing] = await tx.query('SELECT * FROM push_events WHERE event_id = ?', [eventId])
@@ -135,7 +137,7 @@ export class DurablePushStore {
async renew(delivery: QueuedPushDelivery): Promise<void> {
await this.database.query(
'UPDATE push_delivery_batches SET lease_until = ? WHERE batch_id = ? AND lease_token = ?',
"UPDATE push_delivery_batches SET lease_until = ? WHERE batch_id = ? AND lease_token = ? AND state = 'pending'",
[this.now() + DELIVERY_LEASE_MS, delivery.id, delivery.lease]
)
}
@@ -150,7 +152,7 @@ export class DurablePushStore {
const retry = retryAt < delivery.expiresAt
await this.database.query(
`UPDATE push_delivery_batches SET state = ?, payload_json = ?, due_at = ?, lease_until = 0, lease_token = NULL
WHERE batch_id = ? AND lease_token = ?`,
WHERE batch_id = ? AND lease_token = ? AND state = 'pending'`,
[
retry ? 'pending' : retryAfterMs !== undefined ? 'expired' : outcome,
retry ? JSON.stringify(delivery.notification) : '{}',
@@ -43,8 +43,6 @@ describe('push host challenge store', () => {
ok: true,
hostFingerprint: deriveHostFingerprint(host.publicKey)
})
const [hostRow] = await database.query('SELECT host_fingerprint, last_seen_at FROM push_hosts')
expect(hostRow?.host_fingerprint).toBe(deriveHostFingerprint(host.publicKey))
})
it('never stores material that reproduces the proof', async () => {
@@ -183,58 +181,6 @@ describe('push host challenge store', () => {
await expect(store.issue(Buffer.alloc(31, 1).toString('base64'))).resolves.toBeNull()
})
it('creates no host row until a proof succeeds', async () => {
const host = createPushHostKeypair(30)
const challenge = await store.issue(hostPublicKeyB64(host))
const [beforeProof] = await database.query('SELECT COUNT(*) AS hosts FROM push_hosts')
expect(Number(beforeProof?.hosts)).toBe(0)
const proof = answerPushHostChallenge(challenge!, {
gatewayOrigin: GATEWAY_ORIGIN,
keypair: host,
now: () => clock
})!
await expect(store.verify(challenge!.challengeId, proof)).resolves.toMatchObject({ ok: true })
const [row] = await database.query('SELECT host_public_key, last_seen_at FROM push_hosts')
expect(row?.host_public_key).toBe(hostPublicKeyB64(host))
expect(Number(row?.last_seen_at)).toBe(clock)
})
it('leaves no host row behind when a challenge is never answered', async () => {
for (let index = 0; index < 5; index++) {
await store.issue(hostPublicKeyB64(createPushHostKeypair(40 + index)))
}
const [row] = await database.query('SELECT COUNT(*) AS hosts FROM push_hosts')
expect(Number(row?.hosts)).toBe(0)
})
it('prunes a host past retention only when it has no registration left', async () => {
const stale = createPushHostKeypair(50)
const kept = createPushHostKeypair(51)
for (const host of [stale, kept]) {
const challenge = await store.issue(hostPublicKeyB64(host))
const proof = answerPushHostChallenge(challenge!, {
gatewayOrigin: GATEWAY_ORIGIN,
keypair: host,
now: () => clock
})!
await store.verify(challenge!.challengeId, proof)
}
await database.query(
`INSERT INTO push_devices (registration_id, host_fingerprint, device_id, platform, token,
created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, ?)`,
['reg-1', deriveHostFingerprint(kept.publicKey), 'device-1', 'android', 'token', clock, clock]
)
clock += PUSH_LIMITS.hostRetentionMs
expect(await store.pruneStaleHosts()).toBe(0)
clock += 1
expect(await store.pruneStaleHosts()).toBe(1)
const [row] = await database.query('SELECT host_fingerprint FROM push_hosts')
expect(row?.host_fingerprint).toBe(deriveHostFingerprint(kept.publicKey))
})
it('prunes challenges that fell out of the skew window', async () => {
const host = createPushHostKeypair(9)
await store.issue(hostPublicKeyB64(host))
+3 -40
View File
@@ -69,21 +69,17 @@ export class PushHostChallengeStore {
const expectedProof = createHmac('sha256', challengeSecret)
.update(buildPushHostProofMacInput(transcript))
.digest()
// No push_hosts row yet: issuing is unauthenticated, so anyone could
// otherwise fill the table. The key rides the challenge until verify() proves it.
await this.database.query(
`INSERT INTO push_challenges
(challenge_id, host_fingerprint, host_public_key, secret_hash, transcript, expires_at,
(challenge_id, host_fingerprint, secret_hash, expires_at,
consumed_at)
VALUES (?, ?, ?, ?, ?, ?, NULL)`,
VALUES (?, ?, ?, ?, NULL)`,
[
challengeId,
hostFingerprint,
hostPublicKeyB64,
// The stored digest is of the ack the secret produces, never of the
// secret itself: a database reader must not be able to forge a proof.
sha256(expectedProof),
Buffer.from(transcript).toString('base64'),
expiresAt
]
)
@@ -101,7 +97,7 @@ export class PushHostChallengeStore {
const proof = decodeCanonicalBase64(proofB64, 32)
return await this.database.transaction<PushProofVerification>(async (transaction) => {
const [row] = await transaction.query(
`SELECT host_fingerprint, host_public_key, secret_hash, expires_at, consumed_at
`SELECT host_fingerprint, secret_hash, expires_at, consumed_at
FROM push_challenges WHERE challenge_id = ?`,
[challengeId]
)
@@ -123,12 +119,6 @@ export class PushHostChallengeStore {
[now, challengeId]
)
if (Number(consumed?.changes ?? 0) !== 1) return { ok: false, reason: 'already_consumed' }
await this.rememberHost(
transaction,
String(row.host_fingerprint),
String(row.host_public_key),
now
)
return { ok: true, hostFingerprint: String(row.host_fingerprint) }
})
}
@@ -142,31 +132,4 @@ export class PushHostChallengeStore {
])
return Number(result?.changes ?? 0)
}
// A host that stopped proving and has no registration left is dead weight;
// its public key is recoverable from the desktop on the next challenge.
async pruneStaleHosts(): Promise<number> {
const [result] = await this.database.query(
`DELETE FROM push_hosts
WHERE last_seen_at < ?
AND host_fingerprint NOT IN (SELECT host_fingerprint FROM push_devices)`,
[this.now() - PUSH_LIMITS.hostRetentionMs]
)
return Number(result?.changes ?? 0)
}
private async rememberHost(
transaction: PushDatabase,
hostFingerprint: string,
hostPublicKeyB64: string,
now: number
): Promise<void> {
await transaction.query(
`INSERT INTO push_hosts (host_fingerprint, host_public_key, created_at, last_seen_at)
VALUES (?, ?, ?, ?)
ON CONFLICT(host_fingerprint) DO UPDATE SET
host_public_key = excluded.host_public_key, last_seen_at = excluded.last_seen_at`,
[hostFingerprint, hostPublicKeyB64, now, now]
)
}
}
@@ -1,119 +0,0 @@
import { expect, it } from 'vitest'
import { PUSH_LIMITS } from '@orca-cloud/push-contract'
import { openPushDatabase, type PushDatabase } from './push-database.js'
import { PushHostChallengeStore } from './host-challenge-store.js'
import {
answerPushHostChallenge,
createPushHostKeypair,
hostPublicKeyB64
} from './host-challenge-answering.test-fixture.js'
const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL
it.skipIf(!databaseUrl)(
'accepts concurrent first proofs, including after host pruning',
async () => {
if (!process.env.CI && new URL(databaseUrl!).port !== '55440')
throw new Error('isolated_postgres_port_required')
const admin = await openPushDatabase({ databaseUrl, dataDir: '' })
const schema = `first_proof_${Date.now()}`
await admin.query(`CREATE SCHEMA ${schema}`)
const isolatedUrl = new URL(databaseUrl!)
isolatedUrl.searchParams.set('options', `-c search_path=${schema}`)
const database = await openPushDatabase({
databaseUrl: isolatedUrl.toString(),
dataDir: '',
poolMax: 4
})
const host = createPushHostKeypair()
const origin = 'https://push.onorca.dev'
let now = Date.now()
let release!: () => void
let arrivals = 0
let gate: Promise<void>
let concurrent = true
const wrapped: PushDatabase = {
dialect: database.dialect,
query: database.query.bind(database),
close: database.close.bind(database),
lockQuotaScope: database.lockQuotaScope.bind(database),
transaction: (run) =>
database.transaction((tx) =>
run({
dialect: tx.dialect,
close: tx.close.bind(tx),
transaction: tx.transaction.bind(tx),
lockQuotaScope: tx.lockQuotaScope.bind(tx),
query: async (sql, params) => {
// Both transactions reach the host insert before either can commit.
if (concurrent && sql.trimStart().startsWith('INSERT INTO push_hosts')) {
if (++arrivals === 2) release()
await gate
}
return tx.query(sql, params)
}
})
)
}
const store = new PushHostChallengeStore(wrapped, origin, () => now)
let fingerprint = ''
try {
for (let round = 0; round < 2; round++) {
arrivals = 0
concurrent = true
gate = new Promise<void>((resolve) => {
release = resolve
})
const challenges = await Promise.all([
store.issue(hostPublicKeyB64(host)),
store.issue(hostPublicKeyB64(host))
])
fingerprint = challenges[0]!.hostFingerprint
const results = await Promise.allSettled(
challenges.map((challenge) =>
store.verify(
challenge!.challengeId,
answerPushHostChallenge(challenge!, {
gatewayOrigin: origin,
keypair: host,
now: () => now
})!
)
)
)
expect(results).toEqual(
challenges.map(() => ({
status: 'fulfilled',
value: { ok: true, hostFingerprint: fingerprint }
}))
)
const createdAt = now
concurrent = false
now += 1000
const next = (await store.issue(hostPublicKeyB64(host)))!
await store.verify(
next.challengeId,
answerPushHostChallenge(next, { gatewayOrigin: origin, keypair: host, now: () => now })!
)
const rows = await database.query('SELECT * FROM push_hosts WHERE host_fingerprint = ?', [
fingerprint
])
expect(rows).toHaveLength(1)
expect(Number(rows[0]!.created_at)).toBe(createdAt)
expect(Number(rows[0]!.last_seen_at)).toBe(now)
expect(rows[0]!.host_public_key).toBe(hostPublicKeyB64(host))
now += PUSH_LIMITS.hostRetentionMs + 1
await store.pruneStaleHosts()
expect(
await database.query('SELECT 1 FROM push_hosts WHERE host_fingerprint = ?', [fingerprint])
).toEqual([])
}
} finally {
release?.()
await database.query('DELETE FROM push_challenges WHERE host_fingerprint = ?', [fingerprint])
await database.query('DELETE FROM push_hosts WHERE host_fingerprint = ?', [fingerprint])
await database.close()
await admin.query(`DROP SCHEMA ${schema} CASCADE`)
await admin.close()
}
}
)
@@ -0,0 +1,47 @@
import { expect, it } from 'vitest'
import { openInMemoryPushDatabase, openPushDatabase } from './push-database.js'
import { PushHostChallengeStore } from './host-challenge-store.js'
import {
answerPushHostChallenge,
createPushHostKeypair,
hostPublicKeyB64
} from './host-challenge-answering.test-fixture.js'
it('accepts independent proofs but consumes each challenge only once under concurrency', async () => {
const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL
if (databaseUrl && !process.env.CI && new URL(databaseUrl).port !== '55440') {
throw new Error('isolated_postgres_port_required')
}
const db = databaseUrl
? await openPushDatabase({ databaseUrl, dataDir: '', poolMax: 4 })
: await openInMemoryPushDatabase()
const host = createPushHostKeypair()
const origin = 'https://push.onorca.dev'
const store = new PushHostChallengeStore(db, origin)
const challenges = await Promise.all([
store.issue(hostPublicKeyB64(host)),
store.issue(hostPublicKeyB64(host))
])
try {
const proofs = challenges.map((challenge) =>
answerPushHostChallenge(challenge!, { gatewayOrigin: origin, keypair: host })!
)
const results = await Promise.all(
challenges.flatMap((challenge, index) =>
Array.from({ length: 5 }, () => store.verify(challenge!.challengeId, proofs[index]!))
)
)
expect(results.filter((result) => result.ok)).toEqual([
{ ok: true, hostFingerprint: challenges[0]!.hostFingerprint },
{ ok: true, hostFingerprint: challenges[0]!.hostFingerprint }
])
expect(results.filter((result) => !result.ok)).toEqual(
Array.from({ length: 8 }, () => ({ ok: false, reason: 'already_consumed' }))
)
} finally {
for (const challenge of challenges) {
await db.query('DELETE FROM push_challenges WHERE challenge_id = ?', [challenge!.challengeId])
}
await db.close()
}
})
+1 -3
View File
@@ -4,7 +4,6 @@ import type { createPushServer } from './push-server.js'
const CHALLENGE_PRUNE_INTERVAL_MS = 60_000
const SESSION_PRUNE_INTERVAL_MS = 10 * 60_000
const DELIVERY_PRUNE_INTERVAL_MS = 60_000
const STALE_HOST_PRUNE_INTERVAL_MS = 30 * 60_000
function prune(label: string, run: () => Promise<number>, intervalMs: number): NodeJS.Timeout {
const timer = setInterval(() => {
@@ -34,8 +33,7 @@ export function startPushBackground(
const timers = [
prune('challenges', () => challenges.pruneExpired(), CHALLENGE_PRUNE_INTERVAL_MS),
prune('sessions', () => sessions.pruneExpired(), SESSION_PRUNE_INTERVAL_MS),
prune('deliveries', () => deliveryStore.prune(), DELIVERY_PRUNE_INTERVAL_MS),
prune('stale_hosts', () => challenges.pruneStaleHosts(), STALE_HOST_PRUNE_INTERVAL_MS)
prune('deliveries', () => deliveryStore.prune(), DELIVERY_PRUNE_INTERVAL_MS)
]
worker.start()
return async () => {
-15
View File
@@ -2,21 +2,10 @@ import { DURABLE_PUSH_SCHEMA } from './durable-push-schema.js'
// Applied at startup for both dialects, including additive queue tables,
// so every column type has to read the same in SQLite and PostgreSQL.
const PUSH_SCHEMA = `
CREATE TABLE IF NOT EXISTS push_hosts (
host_fingerprint TEXT PRIMARY KEY,
host_public_key TEXT NOT NULL,
created_at BIGINT NOT NULL,
last_seen_at BIGINT NOT NULL
);
CREATE TABLE IF NOT EXISTS push_challenges (
challenge_id TEXT PRIMARY KEY,
host_fingerprint TEXT NOT NULL,
-- Carried here so a host row is only written once a proof succeeds; an
-- unauthenticated challenge must not be able to create one.
host_public_key TEXT NOT NULL,
secret_hash TEXT NOT NULL,
transcript TEXT NOT NULL,
expires_at BIGINT NOT NULL,
consumed_at BIGINT
);
@@ -44,10 +33,6 @@ CREATE TABLE IF NOT EXISTS push_devices (
);
CREATE UNIQUE INDEX IF NOT EXISTS push_devices_host_device
ON push_devices(host_fingerprint, device_id);
-- The stale-host pruner scans by last contact. Its owning-host subquery rides
-- the push_devices_host_device index.
CREATE INDEX IF NOT EXISTS push_hosts_last_seen_at ON push_hosts(last_seen_at);
`
export function pushSchemaStatements(): string[] {
@@ -34,3 +34,31 @@ it('returns queued for concurrent retries without double quota or delivery', asy
await h.flushDeliveries()
expect(h.fcmRequests).toHaveLength(2)
})
it.each([false, true])(
'accepts default alert kind equivalently through the API (explicit first: %s)',
async (explicitFirst) => {
const h = await createPushServerHarness()
harnesses.push(h)
const token = await h.signIn(createPushHostKeypair(3))
const registrationId = await h.registerAndroid(token)
const implicit = notification()
const explicit = { kind: 'alert', ...implicit }
for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) {
const response = await h.post(
'/v1/send',
{ v: 1, registrationIds: [registrationId], notification: event },
token
)
expect(await response.json()).toEqual({ results: [{ registrationId, status: 'queued' }] })
}
await h.flushDeliveries()
expect(h.fcmRequests).toHaveLength(1)
const changed = await h.post(
'/v1/send',
{ v: 1, registrationIds: [registrationId], notification: { ...explicit, body: 'changed' } },
token
)
expect(await changed.json()).toEqual({ results: [{ registrationId, status: 'error' }] })
}
)
+1 -1
View File
@@ -70,7 +70,7 @@ it.skipIf(!databaseUrl)(
expect((await runtime.app.request('/v1/send', { method: 'POST' })).status).toBe(503)
expect(calls.mock.calls.map(([sql]) => sql)).toEqual(['SELECT 1 AS ready'])
await expect(
database.query('DELETE FROM public.push_hosts WHERE false')
database.query('DELETE FROM public.push_challenges WHERE false')
).rejects.toMatchObject({ code: '25006' })
} finally {
vi.useRealTimers()
@@ -188,13 +188,11 @@
"google_project_iam_member.relay_runtime_artifact_reader",
"google_project_iam_member.relay_runtime_cloudsql_client",
"google_project_iam_member.relay_runtime_log_writer",
"google_secret_manager_secret.push_database_url",
"google_secret_manager_secret.push_dedicated_database_url",
"google_secret_manager_secret.push_provider",
"google_secret_manager_secret.relay_assignment_signing_key",
"google_secret_manager_secret.relay_database_url",
"google_secret_manager_secret.relay_regional_placement_enabled",
"google_secret_manager_secret_iam_member.push_database_url_runtime_accessor",
"google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor",
"google_secret_manager_secret_iam_member.push_provider_runtime_accessor",
"google_secret_manager_secret_iam_member.relay_assignment_signing_key_accessor",
@@ -206,7 +204,6 @@
"google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer",
"google_secret_manager_secret_iam_member.relay_regional_placement_director_accessor",
"google_secret_manager_secret_iam_member.relay_regional_placement_runtime_accessor",
"google_secret_manager_secret_version.push_database_url",
"google_secret_manager_secret_version.push_dedicated_database_url",
"google_secret_manager_secret_version.relay_assignment_signing_key",
"google_secret_manager_secret_version.relay_database_url",
@@ -242,14 +239,13 @@
"google_service_account_iam_member.github_staging_relay_deploy_auth_runtime_user",
"google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user",
"google_service_account_iam_member.relay_fence_broker_requester_token_creator",
"google_sql_database.push",
"google_sql_database.push_dedicated",
"google_sql_database.relay",
"google_sql_database_instance.push_dedicated",
"google_sql_user.push",
"google_sql_user.push_dedicated",
"google_sql_user.relay",
"google_storage_bucket_iam_member.github_production_relay_capacity_state",
"google_storage_bucket_iam_member.github_push_rollout_lease",
"google_storage_bucket_iam_member.github_relay_asia_topology_state",
"google_storage_bucket_iam_member.github_relay_asia_topology_state_list",
"google_storage_bucket_iam_member.github_staging_relay_capacity_state",
@@ -257,7 +253,6 @@
"google_storage_bucket_iam_member.github_staging_relay_deploy_state_list",
"google_storage_bucket_iam_member.relay_fence_broker_bucket_reader",
"google_storage_bucket_iam_member.relay_fence_broker_state_objects",
"random_password.push_database",
"random_password.push_dedicated_database",
"random_password.relay_assignment_signing_key",
"random_password.relay_database"
@@ -283,8 +283,6 @@ export const LEASED_WORKFLOWS = named([
'operate-relay-production-rehome.yml',
production({ leaseFiles: ['operate-relay-production-rehome-job.yml'] })
],
// The gateway applies its schema at startup, so its deploy revision is the schema step.
['push-deploy.yml', production()],
['deploy-relay-asia-topology.yml', eitherEnvironment()],
['operate-relay-asia-admission.yml', eitherEnvironment()],
['deploy-relay-staging.yml', staging()],
@@ -300,6 +298,10 @@ export const LEASED_WORKFLOWS = named([
])
export const NOT_A_CLOUD_SQL_CANDIDATE = named([
[
'push-deploy.yml',
'Push attaches only to its dedicated database and serializes its own traffic changes under production-push-rollout and the durable push-rollout lease; push-gateway-workflow tests verify both.'
],
[
'monitor-relay-production.yml',
'Read-only. Its identity holds monitoring, logging, Cloud SQL and compute viewer roles only, and it runs `gcloud sql instances describe`, never a mutation. It consumes no connection budget, so the durable lease would only let monitoring block a rollout and a rollout block monitoring.'
@@ -12,7 +12,7 @@ import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs'
// Why: the push gateway holds the APNs key and is the only thing standing between a paired
// phone and a silent notification pipeline. Its deploy is a blue/green rollout against the
// shared Cloud SQL instance, and each of the guarantees below is one careless edit from gone.
// dedicated Cloud SQL instance, and each of the guarantees below is one careless edit from gone.
const WORKFLOW = 'push-deploy.yml'
const workflow = readRelayWorkflow(WORKFLOW)
const deploy = () => {
@@ -60,15 +60,15 @@ test('Terraform trusts this exact workflow file on the production deploy provide
assert.equal(relayWorkflowFile(WORKFLOW), 'cloud-push-deploy.yml')
})
test('the rollout is serialized and leases the production Cloud SQL rollout lock', () => {
test('the rollout is serialized and leases its dedicated push rollout lock', () => {
const blocks = concurrencyBlocks(workflow)
assert.equal(blocks.length, 1)
assert.equal(blocks[0].group, 'production-cloud-sql-rollout')
assert.equal(blocks[0].group, 'production-push-rollout')
assert.equal(blocks[0].cancelInProgress, 'false')
const steps = leaseSteps(workflow)
assert.equal(steps.length, 1, 'exactly one lease step, held for the whole run')
assert.equal(steps[0].bucket, 'onorca-cloud-terraform-state')
assert.equal(steps[0].object, 'terraform/state/cloud-sql-rollout/production.lock')
assert.equal(steps[0].object, 'terraform/state/push-rollout/production.lock')
assert.equal(steps[0].release, undefined, 'release stays at its default for a single-job run')
})
@@ -138,7 +138,7 @@ test('the database pool size is Terraform-owned and bounded at plan time', () =>
assert.ok(block, 'the push service no longer declares a lifecycle block')
assert.match(
block[1],
/var\.push_max_instances \* var\.push_database_pool_max <= 4/,
/var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/,
'instances x pool must be bounded at plan time'
)
assert.match(
@@ -313,3 +313,22 @@ test('push credentials cannot assume the shared Relay deploy identity', () => {
test('dedicated database admits three simultaneous revision pools', () => {
assert.match(terraform('push-gateway.tf'), /var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/)
})
test('push has only a dedicated database attachment and a narrowly scoped deployment lease', () => {
const service = terraform('push-gateway.tf')
const database = terraform('push-dedicated-database.tf')
assert.match(service, /instances = \[google_sql_database_instance\.push_dedicated\[0\]\.connection_name\]/)
assert.match(service, /secret\s*= google_secret_manager_secret\.push_dedicated_database_url\[0\]\.secret_id/)
assert.match(service, /version = google_secret_manager_secret_version\.push_dedicated_database_url\[0\]\.version/)
assert.doesNotMatch(service + database, /push_dedicated_database_(?:active|enabled)|local\.relay_database_connection_name|resource "google_sql_database" "push"/)
assert.match(database, /tier\s*= "db-custom-2-7680"/)
assert.match(database, /availability_type = "REGIONAL"/)
assert.match(database, /deletion_protection\s*= true/)
assert.match(database, /deletion_protection_enabled = true/)
const identity = terraform('push-deploy-identity.tf')
const lease = identity.match(/resource "google_storage_bucket_iam_member" "github_push_rollout_lease" \{([\s\S]*?)\n\}/)?.[1]
assert.ok(lease)
assert.match(lease, /member = local\.push_deploy_member/)
assert.match(lease, /role\s*= "roles\/storage.objectAdmin"/)
assert.match(lease, /resource.name == 'projects\/_\/buckets\/\$\{var.project_id\}-terraform-state\/objects\/terraform\/state\/push-rollout\/production.lock'/)
})
@@ -33,13 +33,6 @@ function requiredInteger(source, pattern, label) {
return value
}
// A tfvars file states only what it overrides, so an absent key means the variable default holds.
// Reading the default as the fallback keeps this honest either way.
function overriddenInteger(override, overridePattern, source, pattern, label) {
if (!overridePattern.test(override)) return requiredInteger(source, pattern, label)
return requiredInteger(override, overridePattern, label)
}
function productionCells(source, defaultPoolMax) {
const fencedMatch = source.match(/relay_gce_fenced_cells\s*=\s*\[([^\]]*)\]/)
if (!fencedMatch) throw new Error('could not read fenced Relay cells')
@@ -59,13 +52,11 @@ function productionCells(source, defaultPoolMax) {
}
export function calculateRelayCloudSqlConnectionBudget(inputs) {
const pushDraw = inputs.pushInstances * inputs.pushPoolMax
const consumers = {
cells: inputs.cellPoolTotal + inputs.asiaCellCount * inputs.asiaPoolMax,
directors: inputs.directorInstances * inputs.directorPoolMax,
auth: inputs.authInstances * inputs.authPoolMax,
api: inputs.apiInstances * inputs.apiPoolMax,
push: pushDraw
api: inputs.apiInstances * inputs.apiPoolMax
}
const configuredMaximum = Object.values(consumers).reduce((total, value) => total + value, 0)
const retainedDirectorRollback = inputs.directorInstances * inputs.directorPoolMax
@@ -73,8 +64,6 @@ export function calculateRelayCloudSqlConnectionBudget(inputs) {
relayDirectorCandidate: retainedDirectorRollback * 2,
apiCandidate: retainedDirectorRollback + inputs.apiInstances * inputs.apiPoolMax,
authCandidate: retainedDirectorRollback + inputs.authInstances * inputs.authPoolMax,
// Serving is already counted; validation/rejected and its successor add two pools.
pushCandidate: retainedDirectorRollback + pushDraw * 2,
relayCells: retainedDirectorRollback
}
const rolloutOverlap = Math.max(...Object.values(candidateOverlap))
@@ -142,20 +131,6 @@ export function readRelayCloudSqlConnectionBudget({
/variable\s+"relay_director_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
'director pool maximum'
),
// The mobile push gateway shares this instance. Its draw was invisible here until Terraform
// declared the pool: docs/push-gateway.md, "Shape".
pushInstances: overriddenInteger(
productionTfvars,
/^\s*push_max_instances\s*=\s*(\d+)/m,
terraformVariables,
/variable\s+"push_max_instances"[\s\S]*?default\s*=\s*(\d+)/,
'push gateway instances'
),
pushPoolMax: requiredInteger(
terraformVariables,
/variable\s+"push_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
'push gateway pool maximum'
),
authInstances: apps.authInstances,
authPoolMax: apps.authPoolMax,
apiInstances: apps.apiInstances,
@@ -6,91 +6,28 @@ import {
readRelayCloudSqlConnectionBudget
} from './relay-cloud-sql-connection-budget.mjs'
// Why these numbers are this tight: the shared instance's 400 connections were already spoken
// for, and the relay shape below leaves exactly five. The gateway is sized to fit in four, two
// instances times a two-connection pool, and its rollout overlap of 23 stays under the API
// candidate's 65, so the Math.max is the API candidate rather than the gateway.
//
// `Deploy Relay Asia Topology` gates on `withinBudget == true`, so the single remaining
// connection is the whole margin. Anything that raises a pool or an instance count moves it.
test('production plus the push gateway keeps allowance and reserve below the ceiling', () => {
test('production shared consumers keep allowance and reserve below the ceiling', () => {
const report = readRelayCloudSqlConnectionBudget()
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50, push: 4 })
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 })
assert.deepEqual(report.asia, { cells: 3, poolMax: 10 })
assert.equal(report.configuredMaximum, 319)
assert.equal(report.configuredMaximum, 315)
assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30)
assert.equal(report.rolloutOverlap.apiCandidate, 65)
assert.equal(report.rolloutOverlap.authCandidate, 35)
assert.equal(report.rolloutOverlap.pushCandidate, 23)
assert.equal(report.rolloutOverlap.relayCells, 15)
assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15)
// The gateway does not set the maximum; the API candidate does, as it did before it existed.
assert.equal(report.rolloutOverlap.maximum, 65)
assert.equal(report.maintenanceAdminAllowance, 5)
assert.equal(report.explicitReserve, 10)
assert.equal(report.usableCeiling, 390)
assert.equal(report.operatingMaximum, 389)
assert.equal(report.remainingWithinUsableCeiling, 1)
assert.equal(report.budgetedTotal, 399)
assert.equal(report.unallocated, 1)
assert.equal(report.withinBudget, true)
})
// Why: the same relay shape without a push gateway is the before picture, and it stood at five
// connections clear. Holding it here keeps the gateway's cost visible as the four it takes,
// rather than letting drift elsewhere in the budget hide inside the same margin.
test('the same relay shape without the gateway stays inside the ceiling', () => {
const report = calculateRelayCloudSqlConnectionBudget({
cellPoolTotal: 200,
asiaCellCount: 3,
asiaPoolMax: 10,
directorInstances: 5,
directorPoolMax: 3,
authInstances: 2,
authPoolMax: 10,
apiInstances: 10,
apiPoolMax: 5,
pushInstances: 0,
pushPoolMax: 0,
maxConnections: 400,
maintenanceAdminAllowance: 5,
explicitReserve: 10
})
assert.equal(report.consumers.push, 0)
assert.equal(report.rolloutOverlap.maximum, 65)
assert.equal(report.operatingMaximum, 385)
assert.equal(report.remainingWithinUsableCeiling, 5)
assert.equal(report.budgetedTotal, 395)
assert.equal(report.unallocated, 5)
assert.equal(report.withinBudget, true)
})
// Why: a tagged candidate is directly addressable and sits outside the service-wide cap, so all three
// push revisions can reach the ceiling at once. The API and auth candidates add one copy; this
// one adds two, like the director candidate.
test('the push rollout scenario triples the gateway draw over the retained director', () => {
const report = calculateRelayCloudSqlConnectionBudget({
cellPoolTotal: 0,
asiaCellCount: 0,
asiaPoolMax: 0,
directorInstances: 5,
directorPoolMax: 3,
authInstances: 0,
authPoolMax: 0,
apiInstances: 0,
apiPoolMax: 0,
pushInstances: 2,
pushPoolMax: 2,
maxConnections: 400,
maintenanceAdminAllowance: 5,
explicitReserve: 10
})
assert.equal(report.consumers.push, 4)
// Serving is in the base; overlap adds 15 retained director plus two 4-connection pools.
assert.equal(report.rolloutOverlap.pushCandidate, 23)
})
test('fails closed when pool growth consumes the explicit reserve', () => {
const report = calculateRelayCloudSqlConnectionBudget({
cellPoolTotal: 200,
@@ -102,14 +39,12 @@ test('fails closed when pool growth consumes the explicit reserve', () => {
authPoolMax: 10,
apiInstances: 20,
apiPoolMax: 5,
pushInstances: 4,
pushPoolMax: 10,
maxConnections: 400,
maintenanceAdminAllowance: 5,
explicitReserve: 10
})
assert.equal(report.operatingMaximum, 555)
assert.equal(report.operatingMaximum, 515)
assert.equal(report.withinBudget, false)
})
@@ -141,15 +76,11 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => {
})
assert.equal(report.consumers.cells, 14)
// No push_max_instances in this tfvars, so the variable default of one instance holds.
assert.equal(report.consumers.push, 2)
assert.equal(report.operatingMaximum, 48)
assert.equal(report.budgetedTotal, 49)
assert.equal(report.operatingMaximum, 46)
assert.equal(report.budgetedTotal, 47)
})
// Why: production.tfvars overrides push_max_instances down to 2 while variables.tf still defaults
// to 4, so reading the default instead of the override would overstate the live draw by half.
test('a tfvars push_max_instances override wins over the variable default', () => {
test('dedicated push scaling does not consume shared capacity', () => {
const report = readRelayCloudSqlConnectionBudget({
proposedAsiaCellCount: 1,
appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 },
@@ -175,8 +106,9 @@ test('a tfvars push_max_instances override wins over the variable default', () =
explicitReserve: 1
})
assert.equal(report.consumers.push, 6)
assert.equal(report.rolloutOverlap.pushCandidate, 15)
assert.equal(report.consumers.push, undefined)
assert.equal(report.rolloutOverlap.pushCandidate, undefined)
assert.equal(report.operatingMaximum, 46)
})
test('requires strict headroom below the physical ceiling', () => {
@@ -190,14 +122,12 @@ test('requires strict headroom below the physical ceiling', () => {
authPoolMax: 10,
apiInstances: 1,
apiPoolMax: 5,
pushInstances: 1,
pushPoolMax: 2,
maxConnections: 50,
maintenanceAdminAllowance: 9,
explicitReserve: 3
})
assert.equal(report.budgetedTotal, 65)
assert.equal(report.budgetedTotal, 63)
assert.equal(report.withinBudget, false)
})
+100 -72
View File
@@ -1,84 +1,112 @@
# Dedicated push database cutover
# Dedicated push database operations
The gateway has a dedicated PostgreSQL 17 instance configured in
`infra/terraform/push-dedicated-database.tf`: regional HA, 2 vCPU, 7.5 GiB RAM, 50 GiB SSD
with automatic growth, seven retained backups and seven-day point-in-time recovery.
Cloud SQL and Terraform both protect it from deletion. Connections use the Cloud SQL
connector and a separate Secret Manager secret pinned to its Terraform-managed version.
Push attaches only to its dedicated PostgreSQL 17 instance: regional HA, 2 vCPU,
7.5 GiB RAM, 50 GiB SSD with automatic growth, seven retained backups and seven-day
point-in-time recovery. Cloud SQL and Terraform deletion protections remain enabled.
The Cloud SQL connector uses the dedicated URL secret pinned to its managed version.
There is no shared-storage fallback or provision/activate switch.
Provisioning (`push_dedicated_database_enabled`) and attachment
(`push_dedicated_database_active`) are separate switches. Neither changes the existing
shared database or its credentials. The production feature is still in internal testing;
its owner approved an empty database and discarding the previous test state. No data
transfer or application maintenance mechanism is needed for this initial activation.
Test phones must register again; pending notifications and old sessions do not transfer.
This reset procedure is not suitable after public launch without explicit data-loss approval.
## Existing-resource cleanup: operator prerequisite
## Provision
This is a plan/runbook, not authorization to apply or delete resources. Preserve the
shared Orca instance, dedicated push instance, all dedicated data and identities, and
unrelated resources. The dedicated resource addresses remain unchanged:
Use the production relay backend and environment variables documented in
`infra/terraform/README.md`. Save and review a targeted Terraform plan containing only:
- `google_sql_database_instance.push_dedicated[0]`
- `google_sql_database.push_dedicated[0]`
- `random_password.push_dedicated_database[0]`
- `google_sql_user.push_dedicated[0]`
- `google_secret_manager_secret.push_dedicated_database_url[0]`
- `google_secret_manager_secret_version.push_dedicated_database_url[0]`
- `google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor[0]`
- `google_sql_database_instance.push_dedicated`
- `google_sql_database.push_dedicated`
- `random_password.push_dedicated_database`
- `google_sql_user.push_dedicated`
- `google_secret_manager_secret.push_dedicated_database_url`
- `google_secret_manager_secret_version.push_dedicated_database_url`
- `google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor`
The relay state may still own these six obsolete shared-store resources, whose
configuration is removed. An untargeted plan would propose deleting them; do not apply it:
Require exactly seven additions and no updates or deletions for initial provisioning.
Apply that saved plan with backend locking, then require the same targeted plan to be
empty. Keep sensitive Terraform plans access-restricted; never print secret values or
upload raw state/plan JSON. Verify the instance is RUNNABLE with the expected version,
tier, backup policy, and regional availability. No Cloud Run service changes in this step.
- `google_sql_database.push[0]`
- `google_sql_user.push[0]`
- `random_password.push_database[0]`
- `google_secret_manager_secret.push_database_url[0]`
- `google_secret_manager_secret_version.push_database_url[0]`
- `google_secret_manager_secret_iam_member.push_database_url_runtime_accessor[0]`
## Activate the empty store
1. Use the production backend in `infra/terraform/README.md`. Inspect state addresses and
the live service attachment, pinned secret reference and revision resources without
printing credentials. Require the dedicated attachment and no old shared-store consumers;
source connection drain needs an authorized operator's read-only observation.
2. Have the shared database owner adopt the six legacy resources in an explicitly owned
archival configuration before retiring their relay-state ownership. Retain the former
database's `prevent_destroy` protection and secret versions; do not disable protection,
drop databases, rotate passwords or introduce a second runtime attachment. A reviewed
exact-address state transfer must preserve remote IDs and secret material in approved
Terraform storage, with no credential exports to local files or terminal output.
3. Require the owner's import/ownership plan to preserve existing resources and then an
empty plan for those addresses. Only after adoption is proven may the operator remove
precisely the six former addresses from relay state under backend locking. Do not
automate this via `removed` blocks, broad `state rm`, force, or an untargeted apply.
4. Review a fresh relay plan. Reject every delete or replace affecting either SQL instance,
dedicated databases/users/secrets, or unrelated resources. Target only the intended push
service and lease IAM grant for rollout; review their dependency closure too. Existing
unrelated drift must be handled by its owner, outside this cleanup.
1. Record the current immutable serving image, revision, SQL attachment, and database
secret reference. Confirm traffic is pinned to that revision rather than LATEST.
2. Set `push_dedicated_database_active = true` in production. Save a targeted plan for
`google_cloud_run_v2_service.push`. Inspect its dependency closure and reject unrelated
changes. Require only the SQL attachment and database secret reference to change.
3. Hold the existing Cloud SQL rollout lease while applying that saved service-shape plan.
Terraform ignores traffic and image; verify traffic remains pinned to the old revision.
Do not apply a plan that would shift traffic or revert runtime configuration.
4. Dispatch `cloud-push-deploy.yml` from main with the exact reviewed source SHA. It creates
an inert, read-only candidate inheriting the dedicated attachment, probes readiness and provider
access, then deliberately creates an active successor of the same digest before deleting validation.
Activation starts schema writes and workers before HTTP promotion. Verify the candidate's SQL
attachment and pinned secret reference as well as its image and health.
5. Register a test phone against the deployed origin and prove real APNs delivery. Check
database errors and confirm the old revisions have no traffic or tags and source SQL
connections have drained. Leave the old database intact; do not delete shared resources.
No data transfer, dedicated database reset, or phone re-registration is part of this cleanup.
If activation fails before promotion, the existing HTTP serving revision is unchanged, but
activated workers may already have sent notifications or mutated the queue. Cloud Run will not
delete the latest created revision, even untagged at zero traffic. Recovery creates a known-good
successor first, verifies its template/runtime/secret shape and health, promotes and verifies it,
then deletes rejected and previous consumers. The recovery successor remains serving; it can run
known-good schema/workers before promotion and does not undo earlier queue or schema changes.
A partial activation leaving three resources must retire non-latest inert validation before
recovery creates another; failed retirement stops automation. Every deploy requires a single
serving revision resource at admission, so review and retire historical/leftover revisions under
the lease before dispatch. A Terraform attachment update can itself create such a revision:
verify/promote that known-good image and attachment and retire the former revision before dispatch.
The deploy workflow can roll traffic back on failure; in this internal reset rollout,
that may discard registrations created during the probe window. After successful activation,
application rollback should retain the dedicated attachment and deploy an older compatible
image through the workflow. Returning to the shared store is another explicit state reset,
not a lossless rollback. Future public migrations require a separately rehearsed transfer.
## Schema prerequisite for existing internal test databases
## Capacity and resizing
New schemas omit `push_hosts` and the unused `host_public_key` and `transcript` columns
on `push_challenges`. Authentication still verifies the encrypted transcript and consumes
its challenge digest once; sessions and device ownership are unchanged. No compatibility
migration for unpublished builds runs at application startup.
The initial gateway keeps its existing two-connection pool and two-instance maximum.
Dedicated database rollout pools are capped at 64 total configured pool connections across three simultaneous
revision resources (serving, validation/rejected, and active/recovery successor), leaving room for maintenance and operators; this is an admission
budget, not a throughput claim. Increase the pool only after measuring deployed contention.
Keep the shared database allocation reserved until source connections have drained.
Before deploying onto an older internal schema, an operator must arrange a separately
reviewed schema-preparation job through the approved database execution path. Its entire
scope is dropping `push_hosts` (including its index) and those two unused challenge columns;
preserve challenge digest/expiry/consumption fields and every session, device and delivery
table. Verify that the old NOT NULL columns are absent before admitting the new image.
Do not hand-edit production SQL or reset the dedicated database to satisfy this prerequisite.
Until that job is reviewed and executed, the new image is not ready for an existing schema.
Cloud SQL CPU/RAM resizing is an in-place infrastructure change but can interrupt database
connections. HA does not make a resize interruption-free. Durable accepted events remain in
SQL and workers retry after recovery within their five-minute expiry; requests that never
reach durable acceptance depend on client retries. Schedule resizes and verify reconnection,
queue recovery, readiness, and real delivery afterward.
## Deployment serialization transition
Finish all old push workflow runs before changing the workflow's lock namespace. An old
shared-lock push run and a new push-lock run do not exclude each other. Hold off new push
dispatches while preparing the following exact changes:
1. Review the relay-root plan for
`google_storage_bucket_iam_member.github_push_rollout_lease[0]`. It grants only
`roles/storage.objectAdmin` on
`projects/_/buckets/onorca-cloud-terraform-state/objects/terraform/state/push-rollout/production.lock`
to the dedicated push deploy account. The lease action uses object GET/upload/delete,
so no bucket-wide listing or Terraform-state access is needed.
2. After approval, apply only the reviewed IAM/dependency plan. Verify the exact condition
and principal independently. If foundation still grants push membership in the old
`cloud_sql_rollout_lease_members`, its owner removes only that push member; keep Relay's
existing members and permissions. Do not mutate foundation through the relay root.
3. Publish the reviewed workflow on main with `production-push-rollout`, cancellation
disabled, and the existing lease action pointed at the dedicated object. The durable
lease covers admission, candidate validation, activation, traffic changes and recovery.
A stale/conflicting lease stops the run; it is never stolen or force-deleted.
4. Deploy the reviewed image through `cloud-push-deploy.yml`. Preserve candidate readiness,
runtime-provider validation, exact digest/configuration checks, and explicit activation.
Verify the public origin and real notification delivery/dismissal afterward.
## Recovery and capacity
Activation starts schema writes and workers before HTTP promotion. Traffic rollback cannot
undo queue or schema changes. Cloud Run cannot delete its latest revision, so failed
activation creates a known-good successor, verifies it, promotes it, then retires rejected
and previous revisions. When partial activation leaves three resources, retire non-latest
inert validation before creating recovery. Failed retirement stops automation. Admission
requires one serving revision resource; retire historical leftovers under the push lease.
Keep the dedicated attachment for application rollback and retain an immutable compatible
image. An image requiring the removed challenge columns needs separate schema review.
The two-instance ceiling and two-connection pool draw four configured connections, twelve
across three simultaneous revision resources. Terraform caps instances × pool × 3 at 64
for serving, validation/rejected and active/recovery pools. Push does not draw from Relay's
shared connection budget. Source connections must have drained before treating that old
allocation as free. Increase capacity only after measuring deployed contention.
Cloud SQL resizing can interrupt connections despite HA. Durable accepted events remain in
SQL; workers retry within each event's original five-minute deadline. Schedule resizes and
verify reconnection, queue recovery, readiness and real delivery afterward.
+27 -38
View File
@@ -29,7 +29,7 @@ edit plus a second set of Apple credentials.
| Ingress | all | `INGRESS_TRAFFIC_ALL` |
| Invoker | IAM disabled | `invoker_iam_disabled = true` on the service |
| Runtime identity | `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com` | `google_service_account.push_runtime` |
| Database | `orca_push` on the shared Cloud SQL instance | `google_sql_database.push` |
| Database | `orca_push` on dedicated HA PostgreSQL 17 | `google_sql_database.push_dedicated` |
| Hostname | `push.onorca.dev` | `push_base_url` |
The minimum of one instance is deliberate and did not move when the ceiling came down to two. A
@@ -37,19 +37,11 @@ cold start delays a notification past the point where it is worth showing, so th
keeps a notification prompt. The
ceiling is a different question, answered below.
The maximum and the pool are set by the connection budget, not by the gateway's own appetite. Two
instances times a two-connection pool is a draw of 4, and a rollout triples it to 12, because the
tagged candidate is directly addressable and sits outside the service-wide cap. The shared Cloud
SQL instance's 400 connections were already spoken for by the relay cells, the directors, auth,
and the API, which left five. Four is the whole of the room there was, and the gateway fits in
it.
Two connections per instance is enough for the work. A send runs two or three short queries, so
at concurrency 80 requests queue against the pool for microseconds rather than holding it. A
`lifecycle` precondition refuses a plan whose instances times pool exceeds 4, because a fifth
connection puts the checked budget over its ceiling and blocks `Deploy Relay Asia Topology`,
which gates on it. `dev/scripts/relay-cloud-sql-connection-budget.mjs` counts the gateway and
prints the whole picture.
Push uses its approved dedicated two-vCPU HA database. Two instances with a two-connection
pool draw four connections; three simultaneous revision resources draw twelve. Tagged
candidates can run outside the service-wide cap, so Terraform bounds instances × pool × 3
at 64 connections, leaving dedicated capacity for maintenance and operators. Increase pool
sizes only after measuring contention. The shared Relay budget excludes push entirely.
Authentication is the host proof in `POST /v1/host/challenge`, not Cloud Run IAM, so the service
opts out of invoker IAM with `invoker_iam_disabled = true`, exactly as the relay director does.
@@ -65,7 +57,7 @@ Set on the container by Terraform:
| `PORT` | Cloud Run, container port 8080 |
| `ORCA_PUSH_PUBLIC_URL` | `push_base_url` |
| `ORCA_PUSH_FCM_PROJECT_ID` | `push_fcm_project_id`, empty means `project_id` |
| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-database-url`, version `latest` |
| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-dedicated-database-url`, pinned version |
| `ORCA_PUSH_DATABASE_POOL_MAX` | `push_database_pool_max`, 2 per instance |
| `ORCA_PUSH_APNS_KEY` | Secret `orca-cloud-push-apns-key`, version `latest` |
| `ORCA_PUSH_APNS_KEY_ID` | Secret `orca-cloud-push-apns-key-id`, version `latest` |
@@ -129,11 +121,9 @@ terraform -chdir=infra/terraform import -var-file=environments/production.tfvars
'projects/onorca-cloud/secrets/orca-cloud-push-apple-team-id roles/secretmanager.secretAccessor serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com'
```
Everything else in `push-gateway.tf` is new and is created by the apply: the `orca_push`
database and user, the database-URL secret and its accessor, the `roles/cloudsql.client` binding
on the runtime account, the Cloud Run service, the domain mapping, and the
three deploy-identity bindings. Save that plan and review it before applying; this root carries
unrelated standing drift, so an untargeted apply is never automatic.
The push resources already exist in production. Preserve their addresses, dedicated database
and identities; review the [database cleanup runbook](./push-database-cutover.md) before applying
changes. This root has unrelated standing drift, so an untargeted apply is never automatic.
Two things this root does **not** declare, because the carve assigns them elsewhere. Neither
affects whether this root's plan is clean, since an undeclared resource is invisible to it.
@@ -156,16 +146,17 @@ It authenticates as the dedicated `orca-cloud-gha-push` identity through
to this exact dispatch workflow on main in the production environment. Its distinct principal
attribute cannot assume the shared Relay deploy identity.
The account can write images to the existing Artifact Registry repository, deploy the push service,
and impersonate only the push runtime account. Foundation separately grants access to the shared
rollout-lock prefix and bucket metadata; it grants no Terraform-state object access.
The account can write images to Artifact Registry, deploy the push service, impersonate only
the push runtime account, and manage exactly `terraform/state/push-rollout/production.lock`
in the production state bucket. The relay root owns that conditional lease grant. It grants
no Terraform-state object access. Publish `github_push_workload_identity_provider` and
`github_push_deploy_service_account` as the production-environment variables above.
Before the next deployment, apply the reviewed identity changes in the relay root, add
`serviceAccount:orca-cloud-gha-push@onorca-cloud.iam.gserviceaccount.com` to production foundation's
`cloud_sql_rollout_lease_members`, and apply foundation. Publish the relay outputs
`github_push_workload_identity_provider` and `github_push_deploy_service_account` as the two
GitHub production-environment variables above. Keep the shared identity's existing lease grant
for Relay. Do not fall back to that identity if push setup is incomplete.
The workflow uses the `production-push-rollout` concurrency group with cancellation disabled
and the existing durable lease action on the push-specific object. Push and Relay deploy
independently; two push deploys cannot race traffic changes. Finish every old shared-lock push
run before enabling the new workflow and lease grant. See the cleanup runbook for the bounded
IAM transition and removal of any obsolete foundation-owned push membership.
The run builds the reviewed `source_sha` while the workflow stays on `main`. Buildx returns
its own pushed digest (no mutable-tag lookup); every subsequent check and deployment uses that
@@ -173,11 +164,11 @@ same digest. Before any production boot, a network-isolated container checks tha
recognizes `ORCA_PUSH_MODE=validation` and rejects invalid modes. Older images that lack this
capability are refused before they can connect to production.
Under the production Cloud SQL rollout lease, it records the serving rollback revision and
Under the production push rollout lease, it records the serving rollback revision and
asserts Terraform-owned scaling. It deploys a tagged, zero-traffic validation revision:
- Validation opens PostgreSQL with `default_transaction_read_only=on` and skips schema setup.
- No delivery worker or challenge, session, delivery, or stale-host pruner starts.
- No delivery worker or challenge, session, or delivery pruner starts.
- Only `/health` and `/ready` are available; all application routes return 503.
- `/health` attests `mode: validation`; `/ready` checks database connectivity only. It does not
prove schema compatibility, provider delivery, or active-worker readiness. Container probes
@@ -189,7 +180,7 @@ The workflow verifies the exact image and scaling, probes readiness and mode, an
runtime identity with a validate-only FCM request. Cloud Run rejects deletion of the latest
created revision even when it has no tag or traffic. Activation therefore creates a successor
before removing the validation tag and deleting validation. The dedicated 64-connection budget
and legacy shared budget reserve three simultaneous revision pools: serving, validation/rejected,
reserves three simultaneous revision pools: serving, validation/rejected,
and active/recovery successor (12 configured pool connections at the current two-by-two shape).
Revision deletion is not proof of physical SQL session drain; verify termination and SQL sessions
in controlled rollout acceptance. There is no shutdown sleep used as a drain gate.
@@ -393,10 +384,8 @@ small number of concurrent collapse keys per device, so excess pending messages
every offline alert is not guaranteed to appear. Socket reconnect reconciles dismissals against the
current native tray; it has no stored replay watermark and never recovers a missed OS banner.
### Dedicated database preparation
### Dedicated database operations
`push_dedicated_database_enabled` provisions an independent HA PostgreSQL instance without changing
the live gateway attachment. It defaults to false. Follow [the database cutover runbook](./push-database-cutover.md)
before enabling it or switching stores. The current pre-release activation discards registrations
and queued deliveries; phones re-register on foreground use. A future public-service migration
requires a separate preservation procedure.
Push has one dedicated database attachment, with stable Terraform addresses and deletion
protection. There is no switch to shared storage. Follow the [database operations runbook](./push-database-cutover.md)
for deployment prerequisites, legacy resource ownership, capacity and recovery.
@@ -416,11 +416,8 @@ relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels
# Mobile push gateway. Production is the only environment that runs one; the runtime account,
# the three Apple secrets, and their accessor bindings already exist and are imported once
# (see docs/push-gateway.md).
push_gateway_enabled = true
push_dedicated_database_enabled = true
push_dedicated_database_active = true
push_base_url = "https://push.onorca.dev"
# Sized so the gateway's rollout overlap, the retained director rollback plus its doubled draw,
# stays under the API candidate's, which keeps the checked Cloud SQL connection budget green.
push_gateway_enabled = true
push_base_url = "https://push.onorca.dev"
# Dedicated push pools allow three revision resources during validation and recovery.
push_max_instances = 2
manage_push_domain_mapping = true
+1 -1
View File
@@ -201,7 +201,7 @@ output "push_runtime_service_account" {
}
output "push_database_name" {
value = try(google_sql_database.push[0].name, null)
value = try(google_sql_database.push_dedicated[0].name, null)
description = "Database isolated for durable push gateway state."
}
@@ -1,25 +1,5 @@
# Provision independently of the gateway's SQL attachment switch.
variable "push_dedicated_database_enabled" {
type = bool
description = "Provision the dedicated push database without switching live gateway traffic."
default = false
}
variable "push_dedicated_database_active" {
type = bool
description = "Attach the gateway to the provisioned dedicated database; does not copy existing state."
default = false
}
locals {
push_dedicated_database_count = var.push_gateway_enabled && var.push_dedicated_database_enabled ? 1 : 0
push_database_connection_name = var.push_dedicated_database_active ? google_sql_database_instance.push_dedicated[0].connection_name : local.relay_database_connection_name
push_database_secret_id = var.push_dedicated_database_active ? google_secret_manager_secret.push_dedicated_database_url[0].secret_id : google_secret_manager_secret.push_database_url[0].secret_id
push_database_secret_version = var.push_dedicated_database_active ? google_secret_manager_secret_version.push_dedicated_database_url[0].version : "latest"
}
resource "google_sql_database_instance" "push_dedicated" {
count = local.push_dedicated_database_count
count = local.push_gateway_count
project = var.project_id
name = "${var.name_prefix}-push-db"
@@ -66,7 +46,7 @@ resource "google_sql_database_instance" "push_dedicated" {
}
resource "google_sql_database" "push_dedicated" {
count = local.push_dedicated_database_count
count = local.push_gateway_count
project = var.project_id
name = "orca_push"
@@ -78,13 +58,13 @@ resource "google_sql_database" "push_dedicated" {
}
resource "random_password" "push_dedicated_database" {
count = local.push_dedicated_database_count
count = local.push_gateway_count
length = 32
special = false
}
resource "google_sql_user" "push_dedicated" {
count = local.push_dedicated_database_count
count = local.push_gateway_count
project = var.project_id
name = "orca_push"
@@ -93,7 +73,7 @@ resource "google_sql_user" "push_dedicated" {
}
resource "google_secret_manager_secret" "push_dedicated_database_url" {
count = local.push_dedicated_database_count
count = local.push_gateway_count
project = var.project_id
secret_id = "${var.name_prefix}-push-dedicated-database-url"
@@ -109,7 +89,7 @@ resource "google_secret_manager_secret" "push_dedicated_database_url" {
}
resource "google_secret_manager_secret_version" "push_dedicated_database_url" {
count = local.push_dedicated_database_count
count = local.push_gateway_count
secret = google_secret_manager_secret.push_dedicated_database_url[0].id
secret_data = format(
@@ -122,7 +102,7 @@ resource "google_secret_manager_secret_version" "push_dedicated_database_url" {
}
resource "google_secret_manager_secret_iam_member" "push_dedicated_database_url_accessor" {
count = local.push_dedicated_database_count
count = local.push_gateway_count
project = var.project_id
secret_id = google_secret_manager_secret.push_dedicated_database_url[0].secret_id
@@ -65,3 +65,18 @@ output "github_push_workload_identity_provider" {
output "github_push_deploy_service_account" {
value = try(google_service_account.github_push_deploy[0].email, null)
}
resource "google_storage_bucket_iam_member" "github_push_rollout_lease" {
count = local.push_gateway_deploy_count
bucket = "${var.project_id}-terraform-state"
role = "roles/storage.objectAdmin"
member = local.push_deploy_member
condition {
title = "push_rollout_lease"
description = "Limits push deployment coordination to its own lease object."
expression = "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/push-rollout/production.lock'"
}
}
+5 -88
View File
@@ -80,74 +80,6 @@ resource "google_project_iam_member" "push_runtime_cloudsql_client" {
member = google_service_account.push_runtime[0].member
}
# --- Database -----------------------------------------------------------------------------
# Gateway state shares the foundation-owned Cloud SQL instance with auth and the relay, and uses
# an isolated database and principal, exactly as relay-database.tf does. The application applies
# its own schema at startup.
resource "google_sql_database" "push" {
count = local.push_gateway_count
project = var.project_id
name = "orca_push"
instance = local.relay_database_instance_name
# Why: this database holds every live device token. Disabling the gateway must not drop it.
lifecycle {
prevent_destroy = true
}
}
resource "random_password" "push_database" {
count = local.push_gateway_count
length = 32
special = false
}
resource "google_sql_user" "push" {
count = local.push_gateway_count
project = var.project_id
name = "orca_push"
instance = local.relay_database_instance_name
password = random_password.push_database[0].result
}
resource "google_secret_manager_secret" "push_database_url" {
count = local.push_gateway_count
project = var.project_id
secret_id = "${var.name_prefix}-push-database-url"
labels = local.relay_shared_labels
replication {
auto {}
}
}
resource "google_secret_manager_secret_version" "push_database_url" {
count = local.push_gateway_count
secret = google_secret_manager_secret.push_database_url[0].id
secret_data = format(
"postgresql://%s:%s@/%s?host=/cloudsql/%s",
google_sql_user.push[0].name,
random_password.push_database[0].result,
google_sql_database.push[0].name,
local.relay_database_connection_name
)
}
resource "google_secret_manager_secret_iam_member" "push_database_url_runtime_accessor" {
count = local.push_gateway_count
project = var.project_id
secret_id = google_secret_manager_secret.push_database_url[0].secret_id
role = "roles/secretmanager.secretAccessor"
member = google_service_account.push_runtime[0].member
}
# --- Apple credentials ----------------------------------------------------------------------
resource "google_secret_manager_secret" "push_provider" {
@@ -207,7 +139,7 @@ resource "google_cloud_run_v2_service" "push" {
name = "cloudsql"
cloud_sql_instance {
instances = [local.push_database_connection_name]
instances = [google_sql_database_instance.push_dedicated[0].connection_name]
}
}
@@ -233,9 +165,7 @@ resource "google_cloud_run_v2_service" "push" {
value = local.push_fcm_project_id
}
# Declared rather than left to the application default, so the gateway's share of the
# shared Cloud SQL connection budget is a value this root states and the precondition
# below can bound.
# Bound the declared pool against the dedicated database rollout budget.
env {
name = "ORCA_PUSH_DATABASE_POOL_MAX"
value = tostring(var.push_database_pool_max)
@@ -246,8 +176,8 @@ resource "google_cloud_run_v2_service" "push" {
value_source {
secret_key_ref {
secret = local.push_database_secret_id
version = local.push_database_secret_version
secret = google_secret_manager_secret.push_dedicated_database_url[0].secret_id
version = google_secret_manager_secret_version.push_dedicated_database_url[0].version
}
}
}
@@ -298,19 +228,8 @@ resource "google_cloud_run_v2_service" "push" {
# 100% LATEST would silently undo either, and this root carries unrelated standing drift, so
# that apply need not be a push change at all.
lifecycle {
# Shared SQL retains its four-connection allocation until the dedicated attachment is active.
precondition {
condition = var.push_dedicated_database_active || var.push_max_instances * var.push_database_pool_max <= 4
error_message = "Push gateway instances x database pool must stay within its 4-connection share while attached to shared Cloud SQL."
}
precondition {
condition = !var.push_dedicated_database_active || var.push_dedicated_database_enabled
error_message = "Provision the dedicated push database before activating it."
}
precondition {
condition = !var.push_dedicated_database_active || var.push_max_instances * var.push_database_pool_max * 3 <= 64
condition = var.push_max_instances * var.push_database_pool_max * 3 <= 64
error_message = "Dedicated push serving, validation/rejected and successor pools must fit the 64-connection rollout budget."
}
@@ -325,9 +244,7 @@ resource "google_cloud_run_v2_service" "push" {
depends_on = [
data.google_artifact_registry_repository.relay_images,
google_project_iam_member.push_runtime_cloudsql_client,
google_secret_manager_secret_iam_member.push_database_url_runtime_accessor,
google_secret_manager_secret_iam_member.push_provider_runtime_accessor,
google_secret_manager_secret_version.push_database_url,
google_secret_manager_secret_version.push_dedicated_database_url,
google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor
]
+1 -7
View File
@@ -548,13 +548,7 @@ variable "push_max_instances" {
}
}
# Why: the gateway's draw on the shared Cloud SQL instance is instances x pool, and the rollout
# lease is taken for twice that, because a tagged candidate is directly addressable and sits
# outside the service-wide cap. Leaving the pool at its application default made that draw
# invisible to this root, so it is declared here and set on the container.
#
# Two is sized to the work, not to the default: a send runs two or three short queries, and at
# concurrency 80 those queue against the pool for microseconds rather than holding it.
# The dedicated database budget counts pools across all three rollout revision resources.
variable "push_database_pool_max" {
type = number
description = "Push gateway database pool size per instance; instances x pool is its Cloud SQL draw."
@@ -48,7 +48,6 @@ describe('push contract limits', () => {
clockSkewToleranceMs: 30_000,
sessionTtlMs: 86_400_000,
notificationTtlSeconds: 300,
hostRetentionMs: 3_600_000,
unauthenticatedRequestsPerMinutePerIp: 30,
authenticatedRequestsPerMinutePerHost: 600
})
@@ -172,7 +171,6 @@ describe('device registration schemas', () => {
).toBe(false)
})
it('shapes the registration and list responses', () => {
expect(
PushDeviceRegistrationResponseSchema.safeParse({ registrationId: 'reg-1' }).success
@@ -16,9 +16,6 @@ export const PUSH_LIMITS = {
clockSkewToleranceMs: 30_000,
sessionTtlMs: 24 * 60 * 60 * 1000,
notificationTtlSeconds: 5 * 60,
// Nothing reads a host row, and any keypair mints one for free, so a host
// with no registration left is kept only long enough to survive a phone swap.
hostRetentionMs: 60 * 60 * 1000,
// The challenge and session routes are the only unauthenticated writes, so
// they are capped per client IP before any key material is generated.
unauthenticatedRequestsPerMinutePerIp: 30,
+11 -9
View File
@@ -57,12 +57,11 @@ The host keypair is X25519 (box), so it cannot sign. Reuse the relay's challenge
- Challenge TTL 10 s, and 10 s is the whole window the gateway honours. The 30 s clock skew tolerance
is the host's alone: it validates a timestamp the gateway chose, so it needs the allowance and the
gateway does not. A gateway that subtracted the tolerance from its own check would run a 40 s TTL.
Store challenge (id, secret hash, host fingerprint, host public key, expiry) in DB so any Cloud Run
Store challenge (id, expected-proof digest, host fingerprint, expiry, consumption time) in DB so any Cloud Run
instance can verify. Expired rows are pruned 30 s late so a slow proof reads as expired rather than
as an unknown challenge.
- Issuing a challenge writes no `push_hosts` row. It is unauthenticated, so a `push_hosts` row would be
a free permanent write for any caller. The row is upserted in `POST /v1/host/session` once the proof
verifies, from the public key the challenge row carries.
- No host registry or public-key/transcript copy is stored. The encrypted challenge and expected-proof
digest provide proof verification; session and device rows retain host ownership.
`POST /v1/host/session`
@@ -232,17 +231,15 @@ metadata server or `GOOGLE_APPLICATION_CREDENTIALS` locally):
### Gateway storage (Postgres in prod, SQLite in tests, same pattern as `cloud/apps/relay/src/database.ts`)
- `push_hosts(host_fingerprint pk, host_public_key, created_at, last_seen_at)`, written only on a
verified proof and pruned after 1 h of no contact when no `push_devices` row still names the host.
Nothing reads it, and any keypair mints a host for free, so it is not allowed to accumulate.
- `push_sessions` holds one row per host, enforced by a unique index and transaction lock. Minting a
session deletes the host's earlier one, since a desktop holds a single session and only re-proves once it is gone.
- `push_challenges(challenge_id pk, host_fingerprint, host_public_key, secret_hash, transcript,
- `push_challenges(challenge_id pk, host_fingerprint, secret_hash,
expires_at, consumed_at)`
- `push_sessions(token_hash pk, host_fingerprint, expires_at, created_at)`
- `push_devices(registration_id pk, host_fingerprint, device_id, platform, token, apns_environment,
dead_at, created_at, updated_at, unique(host_fingerprint, device_id))`
- `push_events` holds logical event identity, content fingerprint, quota timestamp, and expiry.
Omitted `kind` and explicit `kind: "alert"` have the same fingerprint; changed content still conflicts.
- `push_event_recipients` records accepted event/phone pairs for idempotent fanout.
- `push_delivery_batches` holds individual payload envelopes, retry deadlines, and renewable worker
leases.
@@ -368,7 +365,12 @@ Secret Manager names (already exist in `onorca-cloud`): `orca-cloud-push-apns-ke
- Workflow `.github/workflows/cloud-push-deploy.yml`: gated on `vars.ORCA_CLOUD_OPERATIONS_ENABLED`,
Workload Identity like `cloud-relay-*`, builds a reviewed full `source_sha`, deploys with `--no-traffic`, probes the new
revision's `/ready` and a validate-only FCM send, then shifts 100% traffic. Uses
`.github/actions/cloud-sql-rollout-lease` around the schema step.
`.github/actions/cloud-sql-rollout-lease` throughout rollout using
`terraform/state/push-rollout/production.lock` and concurrency group `production-push-rollout`.
The deploy identity can manage only that lease object. Push uses only its dedicated 2-vCPU HA
database and is excluded from Relay's shared database budget. See
[database operations](../../cloud/docs/push-database-cutover.md) for existing-schema preparation,
obsolete resource ownership, and the lock transition before deployment.
- Add the new root files to `cloud/dev/contracts` and `cloud/dev/fixtures` partitions so
`terraform-root-partition.test.mjs` and `Cloud Verify` pass.