From bba1b53447666405d4146d8a6bf914d098e8a4e5 Mon Sep 17 00:00:00 2001 From: Jinwoo-H Date: Wed, 9 Sep 2026 16:03:49 -0400 Subject: [PATCH] refactor(push): remove redundant cloud storage and rollout coupling --- .github/workflows/cloud-push-deploy.yml | 12 +- cloud/README.md | 4 +- .../apps/push/src/durable-push-store.test.ts | 55 ++++++ cloud/apps/push/src/durable-push-store.ts | 10 +- .../push/src/host-challenge-store.test.ts | 54 ------ cloud/apps/push/src/host-challenge-store.ts | 43 +---- .../push/src/host-first-proof-race.test.ts | 119 ------------ .../push/src/host-proof-concurrency.test.ts | 47 +++++ cloud/apps/push/src/push-background.ts | 4 +- cloud/apps/push/src/push-schema.ts | 15 -- .../push/src/push-send-idempotency.test.ts | 28 +++ cloud/apps/push/src/push-validation.test.ts | 2 +- .../terraform-root-partition/families.json | 7 +- .../scripts/cloud-sql-rollout-lock-census.mjs | 6 +- .../scripts/push-gateway-workflow.test.mjs | 29 ++- .../relay-cloud-sql-connection-budget.mjs | 27 +-- ...relay-cloud-sql-connection-budget.test.mjs | 96 ++-------- cloud/docs/push-database-cutover.md | 172 ++++++++++-------- cloud/docs/push-gateway.md | 65 +++---- .../terraform/environments/production.tfvars | 9 +- cloud/infra/terraform/outputs.tf | 2 +- .../terraform/push-dedicated-database.tf | 34 +--- cloud/infra/terraform/push-deploy-identity.tf | 15 ++ cloud/infra/terraform/push-gateway.tf | 93 +--------- cloud/infra/terraform/variables.tf | 8 +- .../push-contract/src/contract.test.ts | 2 - .../packages/push-contract/src/push-limits.ts | 3 - docs/reference/mobile-push-contract.md | 20 +- 28 files changed, 360 insertions(+), 621 deletions(-) delete mode 100644 cloud/apps/push/src/host-first-proof-race.test.ts create mode 100644 cloud/apps/push/src/host-proof-concurrency.test.ts diff --git a/.github/workflows/cloud-push-deploy.yml b/.github/workflows/cloud-push-deploy.yml index d92e2eb3df8..6014681372d 100644 --- a/.github/workflows/cloud-push-deploy.yml +++ b/.github/workflows/cloud-push-deploy.yml @@ -16,10 +16,9 @@ permissions: contents: read id-token: write -# Push uses dedicated Cloud SQL; its schema startup and connection-budget rollout intentionally -# retain the production rollout coordination group and lease shared with Relay. +# Serialize push traffic changes independently of Relay and the shared database. concurrency: - group: production-cloud-sql-rollout + group: production-push-rollout cancel-in-progress: false defaults: @@ -85,10 +84,7 @@ jobs: - name: Configure Docker auth run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet - # Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance, - # and a multi-minute image build inside the lease blocks every relay deploy and rehome for - # its duration. The lease below covers exactly the connection-budget window: deploy, probe, - # shift. + # Building an image does not need the deployment lease. - name: Build and publish the immutable gateway image shell: bash run: | @@ -122,7 +118,7 @@ jobs: - uses: ./.github/actions/cloud-sql-rollout-lease with: bucket: onorca-cloud-terraform-state - object: terraform/state/cloud-sql-rollout/production.lock + object: terraform/state/push-rollout/production.lock # Why: the candidate inherits the serving revision's scaling. A serving revision that has # drifted below the floor would hand the candidate a cold start on every notification, and diff --git a/cloud/README.md b/cloud/README.md index 326135bed14..f183d1647e4 100644 --- a/cloud/README.md +++ b/cloud/README.md @@ -82,8 +82,8 @@ surface: publish and deploy the director, roll GCE cell capacity, operate Asia admission and regional rehoming, prove staging capacity, monitor production, power staging up and down, and deploy the mobile push gateway. `.github/actions/cloud-sql-rollout-lease` is the compare-and-swap lease that -serializes every rollout against the shared Cloud SQL instance, the push -gateway deploy included. +serializes rollouts against the shared Cloud SQL instance. Push reuses that +action with its own lease object and deployment concurrency group. Every one of them is inert. Each top-level job is gated on `vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true'`, a repository variable that is diff --git a/cloud/apps/push/src/durable-push-store.test.ts b/cloud/apps/push/src/durable-push-store.test.ts index 1475e8cd2fa..9fe8bc3035a 100644 --- a/cloud/apps/push/src/durable-push-store.test.ts +++ b/cloud/apps/push/src/durable-push-store.test.ts @@ -199,3 +199,58 @@ it('does not resurrect an in-flight alert after a dismissal and transient provid expect(await store.claim()).toBeNull() expect(await store.pendingCount('phone')).toBe(0) }) + +it.each([false, true])( + 'normalizes default alert kind (explicit first: %s)', + async (explicitFirst) => { + const { db, store } = await fixture() + const { kind: _kind, ...implicit } = notification(1) + const explicit = { kind: 'alert' as const, ...implicit } + for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) { + expect(await store.accept('host', 'phone', event)).toBe('queued') + } + expect(await store.pendingCount('phone')).toBe(1) + expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(1) + expect(await store.accept('host', 'phone', { ...explicit, body: 'changed' })).toBe('error') + expect(await store.accept('host', 'phone', { ...implicit, kind: 'dismiss' })).toBe('queued') + expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(2) + } +) + +it('fences late renew and finish after an expired claim is dismissed', async () => { + const { db, store, advance } = await fixture() + const alert = notification(1) + await store.accept('host', 'phone', alert) + const stale = (await store.claim())! + advance(DELIVERY_LEASE_MS) + await store.accept('host', 'phone', { + ...notification(2, 'dismiss'), + notificationId: alert.notificationId + }) + const read = async () => + ( + await db.query( + 'SELECT state, payload_json, lease_until FROM push_delivery_batches WHERE batch_id = ?', + [stale.id] + ) + )[0] + const cancelled = await read() + expect(cancelled).toMatchObject({ state: 'dismissed', payload_json: '{}' }) + await store.renew(stale) + expect(await read()).toEqual(cancelled) + await store.finish(stale, 1000) + expect(await read()).toEqual(cancelled) + await store.finish(stale) + expect(await read()).toEqual(cancelled) + const dismissal = (await store.claim())! + expect(dismissal.notification.kind).toBe('dismiss') + await store.finish(dismissal) + await store.accept('host', 'phone', notification(3)) + const fresh = (await store.claim())! + await store.finish(fresh, 1000) + advance(1000) + const retry = (await store.claim())! + expect(retry.id).toBe(fresh.id) + await store.finish(retry) + expect(await store.claim()).toBeNull() +}) diff --git a/cloud/apps/push/src/durable-push-store.ts b/cloud/apps/push/src/durable-push-store.ts index 2a58c704b10..3003cc6d384 100644 --- a/cloud/apps/push/src/durable-push-store.ts +++ b/cloud/apps/push/src/durable-push-store.ts @@ -34,8 +34,10 @@ export class DurablePushStore { JSON.stringify([host, kind, notification.notificationEpoch, notification.notificationSeq]) ) .digest('hex') - const { sound: _sound, ...content } = notification - const fingerprint = createHash('sha256').update(JSON.stringify(content)).digest('hex') + const { sound: _sound, kind: _kind, ...content } = notification + const fingerprint = createHash('sha256') + .update(JSON.stringify({ kind, ...content })) + .digest('hex') return this.database.transaction(async (tx) => { await tx.lockQuotaScope(`push-events:${host}`) const [existing] = await tx.query('SELECT * FROM push_events WHERE event_id = ?', [eventId]) @@ -135,7 +137,7 @@ export class DurablePushStore { async renew(delivery: QueuedPushDelivery): Promise { await this.database.query( - 'UPDATE push_delivery_batches SET lease_until = ? WHERE batch_id = ? AND lease_token = ?', + "UPDATE push_delivery_batches SET lease_until = ? WHERE batch_id = ? AND lease_token = ? AND state = 'pending'", [this.now() + DELIVERY_LEASE_MS, delivery.id, delivery.lease] ) } @@ -150,7 +152,7 @@ export class DurablePushStore { const retry = retryAt < delivery.expiresAt await this.database.query( `UPDATE push_delivery_batches SET state = ?, payload_json = ?, due_at = ?, lease_until = 0, lease_token = NULL - WHERE batch_id = ? AND lease_token = ?`, + WHERE batch_id = ? AND lease_token = ? AND state = 'pending'`, [ retry ? 'pending' : retryAfterMs !== undefined ? 'expired' : outcome, retry ? JSON.stringify(delivery.notification) : '{}', diff --git a/cloud/apps/push/src/host-challenge-store.test.ts b/cloud/apps/push/src/host-challenge-store.test.ts index b5b85781c0b..c5aa7928ba9 100644 --- a/cloud/apps/push/src/host-challenge-store.test.ts +++ b/cloud/apps/push/src/host-challenge-store.test.ts @@ -43,8 +43,6 @@ describe('push host challenge store', () => { ok: true, hostFingerprint: deriveHostFingerprint(host.publicKey) }) - const [hostRow] = await database.query('SELECT host_fingerprint, last_seen_at FROM push_hosts') - expect(hostRow?.host_fingerprint).toBe(deriveHostFingerprint(host.publicKey)) }) it('never stores material that reproduces the proof', async () => { @@ -183,58 +181,6 @@ describe('push host challenge store', () => { await expect(store.issue(Buffer.alloc(31, 1).toString('base64'))).resolves.toBeNull() }) - it('creates no host row until a proof succeeds', async () => { - const host = createPushHostKeypair(30) - const challenge = await store.issue(hostPublicKeyB64(host)) - const [beforeProof] = await database.query('SELECT COUNT(*) AS hosts FROM push_hosts') - expect(Number(beforeProof?.hosts)).toBe(0) - - const proof = answerPushHostChallenge(challenge!, { - gatewayOrigin: GATEWAY_ORIGIN, - keypair: host, - now: () => clock - })! - await expect(store.verify(challenge!.challengeId, proof)).resolves.toMatchObject({ ok: true }) - const [row] = await database.query('SELECT host_public_key, last_seen_at FROM push_hosts') - expect(row?.host_public_key).toBe(hostPublicKeyB64(host)) - expect(Number(row?.last_seen_at)).toBe(clock) - }) - - it('leaves no host row behind when a challenge is never answered', async () => { - for (let index = 0; index < 5; index++) { - await store.issue(hostPublicKeyB64(createPushHostKeypair(40 + index))) - } - const [row] = await database.query('SELECT COUNT(*) AS hosts FROM push_hosts') - expect(Number(row?.hosts)).toBe(0) - }) - - it('prunes a host past retention only when it has no registration left', async () => { - const stale = createPushHostKeypair(50) - const kept = createPushHostKeypair(51) - for (const host of [stale, kept]) { - const challenge = await store.issue(hostPublicKeyB64(host)) - const proof = answerPushHostChallenge(challenge!, { - gatewayOrigin: GATEWAY_ORIGIN, - keypair: host, - now: () => clock - })! - await store.verify(challenge!.challengeId, proof) - } - await database.query( - `INSERT INTO push_devices (registration_id, host_fingerprint, device_id, platform, token, - created_at, updated_at) - VALUES (?, ?, ?, ?, ?, ?, ?)`, - ['reg-1', deriveHostFingerprint(kept.publicKey), 'device-1', 'android', 'token', clock, clock] - ) - - clock += PUSH_LIMITS.hostRetentionMs - expect(await store.pruneStaleHosts()).toBe(0) - clock += 1 - expect(await store.pruneStaleHosts()).toBe(1) - const [row] = await database.query('SELECT host_fingerprint FROM push_hosts') - expect(row?.host_fingerprint).toBe(deriveHostFingerprint(kept.publicKey)) - }) - it('prunes challenges that fell out of the skew window', async () => { const host = createPushHostKeypair(9) await store.issue(hostPublicKeyB64(host)) diff --git a/cloud/apps/push/src/host-challenge-store.ts b/cloud/apps/push/src/host-challenge-store.ts index f72bbed8d62..fd12af85983 100644 --- a/cloud/apps/push/src/host-challenge-store.ts +++ b/cloud/apps/push/src/host-challenge-store.ts @@ -69,21 +69,17 @@ export class PushHostChallengeStore { const expectedProof = createHmac('sha256', challengeSecret) .update(buildPushHostProofMacInput(transcript)) .digest() - // No push_hosts row yet: issuing is unauthenticated, so anyone could - // otherwise fill the table. The key rides the challenge until verify() proves it. await this.database.query( `INSERT INTO push_challenges - (challenge_id, host_fingerprint, host_public_key, secret_hash, transcript, expires_at, + (challenge_id, host_fingerprint, secret_hash, expires_at, consumed_at) - VALUES (?, ?, ?, ?, ?, ?, NULL)`, + VALUES (?, ?, ?, ?, NULL)`, [ challengeId, hostFingerprint, - hostPublicKeyB64, // The stored digest is of the ack the secret produces, never of the // secret itself: a database reader must not be able to forge a proof. sha256(expectedProof), - Buffer.from(transcript).toString('base64'), expiresAt ] ) @@ -101,7 +97,7 @@ export class PushHostChallengeStore { const proof = decodeCanonicalBase64(proofB64, 32) return await this.database.transaction(async (transaction) => { const [row] = await transaction.query( - `SELECT host_fingerprint, host_public_key, secret_hash, expires_at, consumed_at + `SELECT host_fingerprint, secret_hash, expires_at, consumed_at FROM push_challenges WHERE challenge_id = ?`, [challengeId] ) @@ -123,12 +119,6 @@ export class PushHostChallengeStore { [now, challengeId] ) if (Number(consumed?.changes ?? 0) !== 1) return { ok: false, reason: 'already_consumed' } - await this.rememberHost( - transaction, - String(row.host_fingerprint), - String(row.host_public_key), - now - ) return { ok: true, hostFingerprint: String(row.host_fingerprint) } }) } @@ -142,31 +132,4 @@ export class PushHostChallengeStore { ]) return Number(result?.changes ?? 0) } - - // A host that stopped proving and has no registration left is dead weight; - // its public key is recoverable from the desktop on the next challenge. - async pruneStaleHosts(): Promise { - const [result] = await this.database.query( - `DELETE FROM push_hosts - WHERE last_seen_at < ? - AND host_fingerprint NOT IN (SELECT host_fingerprint FROM push_devices)`, - [this.now() - PUSH_LIMITS.hostRetentionMs] - ) - return Number(result?.changes ?? 0) - } - - private async rememberHost( - transaction: PushDatabase, - hostFingerprint: string, - hostPublicKeyB64: string, - now: number - ): Promise { - await transaction.query( - `INSERT INTO push_hosts (host_fingerprint, host_public_key, created_at, last_seen_at) - VALUES (?, ?, ?, ?) - ON CONFLICT(host_fingerprint) DO UPDATE SET - host_public_key = excluded.host_public_key, last_seen_at = excluded.last_seen_at`, - [hostFingerprint, hostPublicKeyB64, now, now] - ) - } } diff --git a/cloud/apps/push/src/host-first-proof-race.test.ts b/cloud/apps/push/src/host-first-proof-race.test.ts deleted file mode 100644 index baf5823d08f..00000000000 --- a/cloud/apps/push/src/host-first-proof-race.test.ts +++ /dev/null @@ -1,119 +0,0 @@ -import { expect, it } from 'vitest' -import { PUSH_LIMITS } from '@orca-cloud/push-contract' -import { openPushDatabase, type PushDatabase } from './push-database.js' -import { PushHostChallengeStore } from './host-challenge-store.js' -import { - answerPushHostChallenge, - createPushHostKeypair, - hostPublicKeyB64 -} from './host-challenge-answering.test-fixture.js' - -const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL -it.skipIf(!databaseUrl)( - 'accepts concurrent first proofs, including after host pruning', - async () => { - if (!process.env.CI && new URL(databaseUrl!).port !== '55440') - throw new Error('isolated_postgres_port_required') - const admin = await openPushDatabase({ databaseUrl, dataDir: '' }) - const schema = `first_proof_${Date.now()}` - await admin.query(`CREATE SCHEMA ${schema}`) - const isolatedUrl = new URL(databaseUrl!) - isolatedUrl.searchParams.set('options', `-c search_path=${schema}`) - const database = await openPushDatabase({ - databaseUrl: isolatedUrl.toString(), - dataDir: '', - poolMax: 4 - }) - const host = createPushHostKeypair() - const origin = 'https://push.onorca.dev' - let now = Date.now() - let release!: () => void - let arrivals = 0 - let gate: Promise - let concurrent = true - const wrapped: PushDatabase = { - dialect: database.dialect, - query: database.query.bind(database), - close: database.close.bind(database), - lockQuotaScope: database.lockQuotaScope.bind(database), - transaction: (run) => - database.transaction((tx) => - run({ - dialect: tx.dialect, - close: tx.close.bind(tx), - transaction: tx.transaction.bind(tx), - lockQuotaScope: tx.lockQuotaScope.bind(tx), - query: async (sql, params) => { - // Both transactions reach the host insert before either can commit. - if (concurrent && sql.trimStart().startsWith('INSERT INTO push_hosts')) { - if (++arrivals === 2) release() - await gate - } - return tx.query(sql, params) - } - }) - ) - } - const store = new PushHostChallengeStore(wrapped, origin, () => now) - let fingerprint = '' - try { - for (let round = 0; round < 2; round++) { - arrivals = 0 - concurrent = true - gate = new Promise((resolve) => { - release = resolve - }) - const challenges = await Promise.all([ - store.issue(hostPublicKeyB64(host)), - store.issue(hostPublicKeyB64(host)) - ]) - fingerprint = challenges[0]!.hostFingerprint - const results = await Promise.allSettled( - challenges.map((challenge) => - store.verify( - challenge!.challengeId, - answerPushHostChallenge(challenge!, { - gatewayOrigin: origin, - keypair: host, - now: () => now - })! - ) - ) - ) - expect(results).toEqual( - challenges.map(() => ({ - status: 'fulfilled', - value: { ok: true, hostFingerprint: fingerprint } - })) - ) - const createdAt = now - concurrent = false - now += 1000 - const next = (await store.issue(hostPublicKeyB64(host)))! - await store.verify( - next.challengeId, - answerPushHostChallenge(next, { gatewayOrigin: origin, keypair: host, now: () => now })! - ) - const rows = await database.query('SELECT * FROM push_hosts WHERE host_fingerprint = ?', [ - fingerprint - ]) - expect(rows).toHaveLength(1) - expect(Number(rows[0]!.created_at)).toBe(createdAt) - expect(Number(rows[0]!.last_seen_at)).toBe(now) - expect(rows[0]!.host_public_key).toBe(hostPublicKeyB64(host)) - now += PUSH_LIMITS.hostRetentionMs + 1 - await store.pruneStaleHosts() - expect( - await database.query('SELECT 1 FROM push_hosts WHERE host_fingerprint = ?', [fingerprint]) - ).toEqual([]) - } - } finally { - release?.() - await database.query('DELETE FROM push_challenges WHERE host_fingerprint = ?', [fingerprint]) - await database.query('DELETE FROM push_hosts WHERE host_fingerprint = ?', [fingerprint]) - await database.close() - await admin.query(`DROP SCHEMA ${schema} CASCADE`) - await admin.close() - } - } -) diff --git a/cloud/apps/push/src/host-proof-concurrency.test.ts b/cloud/apps/push/src/host-proof-concurrency.test.ts new file mode 100644 index 00000000000..031ca61c7fd --- /dev/null +++ b/cloud/apps/push/src/host-proof-concurrency.test.ts @@ -0,0 +1,47 @@ +import { expect, it } from 'vitest' +import { openInMemoryPushDatabase, openPushDatabase } from './push-database.js' +import { PushHostChallengeStore } from './host-challenge-store.js' +import { + answerPushHostChallenge, + createPushHostKeypair, + hostPublicKeyB64 +} from './host-challenge-answering.test-fixture.js' + +it('accepts independent proofs but consumes each challenge only once under concurrency', async () => { + const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL + if (databaseUrl && !process.env.CI && new URL(databaseUrl).port !== '55440') { + throw new Error('isolated_postgres_port_required') + } + const db = databaseUrl + ? await openPushDatabase({ databaseUrl, dataDir: '', poolMax: 4 }) + : await openInMemoryPushDatabase() + const host = createPushHostKeypair() + const origin = 'https://push.onorca.dev' + const store = new PushHostChallengeStore(db, origin) + const challenges = await Promise.all([ + store.issue(hostPublicKeyB64(host)), + store.issue(hostPublicKeyB64(host)) + ]) + try { + const proofs = challenges.map((challenge) => + answerPushHostChallenge(challenge!, { gatewayOrigin: origin, keypair: host })! + ) + const results = await Promise.all( + challenges.flatMap((challenge, index) => + Array.from({ length: 5 }, () => store.verify(challenge!.challengeId, proofs[index]!)) + ) + ) + expect(results.filter((result) => result.ok)).toEqual([ + { ok: true, hostFingerprint: challenges[0]!.hostFingerprint }, + { ok: true, hostFingerprint: challenges[0]!.hostFingerprint } + ]) + expect(results.filter((result) => !result.ok)).toEqual( + Array.from({ length: 8 }, () => ({ ok: false, reason: 'already_consumed' })) + ) + } finally { + for (const challenge of challenges) { + await db.query('DELETE FROM push_challenges WHERE challenge_id = ?', [challenge!.challengeId]) + } + await db.close() + } +}) diff --git a/cloud/apps/push/src/push-background.ts b/cloud/apps/push/src/push-background.ts index 6b9bac1b3d1..8925fd41fed 100644 --- a/cloud/apps/push/src/push-background.ts +++ b/cloud/apps/push/src/push-background.ts @@ -4,7 +4,6 @@ import type { createPushServer } from './push-server.js' const CHALLENGE_PRUNE_INTERVAL_MS = 60_000 const SESSION_PRUNE_INTERVAL_MS = 10 * 60_000 const DELIVERY_PRUNE_INTERVAL_MS = 60_000 -const STALE_HOST_PRUNE_INTERVAL_MS = 30 * 60_000 function prune(label: string, run: () => Promise, intervalMs: number): NodeJS.Timeout { const timer = setInterval(() => { @@ -34,8 +33,7 @@ export function startPushBackground( const timers = [ prune('challenges', () => challenges.pruneExpired(), CHALLENGE_PRUNE_INTERVAL_MS), prune('sessions', () => sessions.pruneExpired(), SESSION_PRUNE_INTERVAL_MS), - prune('deliveries', () => deliveryStore.prune(), DELIVERY_PRUNE_INTERVAL_MS), - prune('stale_hosts', () => challenges.pruneStaleHosts(), STALE_HOST_PRUNE_INTERVAL_MS) + prune('deliveries', () => deliveryStore.prune(), DELIVERY_PRUNE_INTERVAL_MS) ] worker.start() return async () => { diff --git a/cloud/apps/push/src/push-schema.ts b/cloud/apps/push/src/push-schema.ts index f0f2a453bef..7375d475703 100644 --- a/cloud/apps/push/src/push-schema.ts +++ b/cloud/apps/push/src/push-schema.ts @@ -2,21 +2,10 @@ import { DURABLE_PUSH_SCHEMA } from './durable-push-schema.js' // Applied at startup for both dialects, including additive queue tables, // so every column type has to read the same in SQLite and PostgreSQL. const PUSH_SCHEMA = ` -CREATE TABLE IF NOT EXISTS push_hosts ( - host_fingerprint TEXT PRIMARY KEY, - host_public_key TEXT NOT NULL, - created_at BIGINT NOT NULL, - last_seen_at BIGINT NOT NULL -); - CREATE TABLE IF NOT EXISTS push_challenges ( challenge_id TEXT PRIMARY KEY, host_fingerprint TEXT NOT NULL, - -- Carried here so a host row is only written once a proof succeeds; an - -- unauthenticated challenge must not be able to create one. - host_public_key TEXT NOT NULL, secret_hash TEXT NOT NULL, - transcript TEXT NOT NULL, expires_at BIGINT NOT NULL, consumed_at BIGINT ); @@ -44,10 +33,6 @@ CREATE TABLE IF NOT EXISTS push_devices ( ); CREATE UNIQUE INDEX IF NOT EXISTS push_devices_host_device ON push_devices(host_fingerprint, device_id); - --- The stale-host pruner scans by last contact. Its owning-host subquery rides --- the push_devices_host_device index. -CREATE INDEX IF NOT EXISTS push_hosts_last_seen_at ON push_hosts(last_seen_at); ` export function pushSchemaStatements(): string[] { diff --git a/cloud/apps/push/src/push-send-idempotency.test.ts b/cloud/apps/push/src/push-send-idempotency.test.ts index 3610050a273..1a687a35449 100644 --- a/cloud/apps/push/src/push-send-idempotency.test.ts +++ b/cloud/apps/push/src/push-send-idempotency.test.ts @@ -34,3 +34,31 @@ it('returns queued for concurrent retries without double quota or delivery', asy await h.flushDeliveries() expect(h.fcmRequests).toHaveLength(2) }) + +it.each([false, true])( + 'accepts default alert kind equivalently through the API (explicit first: %s)', + async (explicitFirst) => { + const h = await createPushServerHarness() + harnesses.push(h) + const token = await h.signIn(createPushHostKeypair(3)) + const registrationId = await h.registerAndroid(token) + const implicit = notification() + const explicit = { kind: 'alert', ...implicit } + for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) { + const response = await h.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: event }, + token + ) + expect(await response.json()).toEqual({ results: [{ registrationId, status: 'queued' }] }) + } + await h.flushDeliveries() + expect(h.fcmRequests).toHaveLength(1) + const changed = await h.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: { ...explicit, body: 'changed' } }, + token + ) + expect(await changed.json()).toEqual({ results: [{ registrationId, status: 'error' }] }) + } +) diff --git a/cloud/apps/push/src/push-validation.test.ts b/cloud/apps/push/src/push-validation.test.ts index ff49e046d60..24a351a7133 100644 --- a/cloud/apps/push/src/push-validation.test.ts +++ b/cloud/apps/push/src/push-validation.test.ts @@ -70,7 +70,7 @@ it.skipIf(!databaseUrl)( expect((await runtime.app.request('/v1/send', { method: 'POST' })).status).toBe(503) expect(calls.mock.calls.map(([sql]) => sql)).toEqual(['SELECT 1 AS ready']) await expect( - database.query('DELETE FROM public.push_hosts WHERE false') + database.query('DELETE FROM public.push_challenges WHERE false') ).rejects.toMatchObject({ code: '25006' }) } finally { vi.useRealTimers() diff --git a/cloud/dev/fixtures/terraform-root-partition/families.json b/cloud/dev/fixtures/terraform-root-partition/families.json index f6be51db1f7..da91ef44a19 100644 --- a/cloud/dev/fixtures/terraform-root-partition/families.json +++ b/cloud/dev/fixtures/terraform-root-partition/families.json @@ -188,13 +188,11 @@ "google_project_iam_member.relay_runtime_artifact_reader", "google_project_iam_member.relay_runtime_cloudsql_client", "google_project_iam_member.relay_runtime_log_writer", - "google_secret_manager_secret.push_database_url", "google_secret_manager_secret.push_dedicated_database_url", "google_secret_manager_secret.push_provider", "google_secret_manager_secret.relay_assignment_signing_key", "google_secret_manager_secret.relay_database_url", "google_secret_manager_secret.relay_regional_placement_enabled", - "google_secret_manager_secret_iam_member.push_database_url_runtime_accessor", "google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor", "google_secret_manager_secret_iam_member.push_provider_runtime_accessor", "google_secret_manager_secret_iam_member.relay_assignment_signing_key_accessor", @@ -206,7 +204,6 @@ "google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer", "google_secret_manager_secret_iam_member.relay_regional_placement_director_accessor", "google_secret_manager_secret_iam_member.relay_regional_placement_runtime_accessor", - "google_secret_manager_secret_version.push_database_url", "google_secret_manager_secret_version.push_dedicated_database_url", "google_secret_manager_secret_version.relay_assignment_signing_key", "google_secret_manager_secret_version.relay_database_url", @@ -242,14 +239,13 @@ "google_service_account_iam_member.github_staging_relay_deploy_auth_runtime_user", "google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user", "google_service_account_iam_member.relay_fence_broker_requester_token_creator", - "google_sql_database.push", "google_sql_database.push_dedicated", "google_sql_database.relay", "google_sql_database_instance.push_dedicated", - "google_sql_user.push", "google_sql_user.push_dedicated", "google_sql_user.relay", "google_storage_bucket_iam_member.github_production_relay_capacity_state", + "google_storage_bucket_iam_member.github_push_rollout_lease", "google_storage_bucket_iam_member.github_relay_asia_topology_state", "google_storage_bucket_iam_member.github_relay_asia_topology_state_list", "google_storage_bucket_iam_member.github_staging_relay_capacity_state", @@ -257,7 +253,6 @@ "google_storage_bucket_iam_member.github_staging_relay_deploy_state_list", "google_storage_bucket_iam_member.relay_fence_broker_bucket_reader", "google_storage_bucket_iam_member.relay_fence_broker_state_objects", - "random_password.push_database", "random_password.push_dedicated_database", "random_password.relay_assignment_signing_key", "random_password.relay_database" diff --git a/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs b/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs index 2f7157d823c..c74473723ab 100644 --- a/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs +++ b/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs @@ -283,8 +283,6 @@ export const LEASED_WORKFLOWS = named([ 'operate-relay-production-rehome.yml', production({ leaseFiles: ['operate-relay-production-rehome-job.yml'] }) ], - // The gateway applies its schema at startup, so its deploy revision is the schema step. - ['push-deploy.yml', production()], ['deploy-relay-asia-topology.yml', eitherEnvironment()], ['operate-relay-asia-admission.yml', eitherEnvironment()], ['deploy-relay-staging.yml', staging()], @@ -300,6 +298,10 @@ export const LEASED_WORKFLOWS = named([ ]) export const NOT_A_CLOUD_SQL_CANDIDATE = named([ + [ + 'push-deploy.yml', + 'Push attaches only to its dedicated database and serializes its own traffic changes under production-push-rollout and the durable push-rollout lease; push-gateway-workflow tests verify both.' + ], [ 'monitor-relay-production.yml', 'Read-only. Its identity holds monitoring, logging, Cloud SQL and compute viewer roles only, and it runs `gcloud sql instances describe`, never a mutation. It consumes no connection budget, so the durable lease would only let monitoring block a rollout and a rollout block monitoring.' diff --git a/cloud/dev/scripts/push-gateway-workflow.test.mjs b/cloud/dev/scripts/push-gateway-workflow.test.mjs index d0a7efd3386..3ee4fabe410 100644 --- a/cloud/dev/scripts/push-gateway-workflow.test.mjs +++ b/cloud/dev/scripts/push-gateway-workflow.test.mjs @@ -12,7 +12,7 @@ import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs' // Why: the push gateway holds the APNs key and is the only thing standing between a paired // phone and a silent notification pipeline. Its deploy is a blue/green rollout against the -// shared Cloud SQL instance, and each of the guarantees below is one careless edit from gone. +// dedicated Cloud SQL instance, and each of the guarantees below is one careless edit from gone. const WORKFLOW = 'push-deploy.yml' const workflow = readRelayWorkflow(WORKFLOW) const deploy = () => { @@ -60,15 +60,15 @@ test('Terraform trusts this exact workflow file on the production deploy provide assert.equal(relayWorkflowFile(WORKFLOW), 'cloud-push-deploy.yml') }) -test('the rollout is serialized and leases the production Cloud SQL rollout lock', () => { +test('the rollout is serialized and leases its dedicated push rollout lock', () => { const blocks = concurrencyBlocks(workflow) assert.equal(blocks.length, 1) - assert.equal(blocks[0].group, 'production-cloud-sql-rollout') + assert.equal(blocks[0].group, 'production-push-rollout') assert.equal(blocks[0].cancelInProgress, 'false') const steps = leaseSteps(workflow) assert.equal(steps.length, 1, 'exactly one lease step, held for the whole run') assert.equal(steps[0].bucket, 'onorca-cloud-terraform-state') - assert.equal(steps[0].object, 'terraform/state/cloud-sql-rollout/production.lock') + assert.equal(steps[0].object, 'terraform/state/push-rollout/production.lock') assert.equal(steps[0].release, undefined, 'release stays at its default for a single-job run') }) @@ -138,7 +138,7 @@ test('the database pool size is Terraform-owned and bounded at plan time', () => assert.ok(block, 'the push service no longer declares a lifecycle block') assert.match( block[1], - /var\.push_max_instances \* var\.push_database_pool_max <= 4/, + /var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/, 'instances x pool must be bounded at plan time' ) assert.match( @@ -313,3 +313,22 @@ test('push credentials cannot assume the shared Relay deploy identity', () => { test('dedicated database admits three simultaneous revision pools', () => { assert.match(terraform('push-gateway.tf'), /var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/) }) + +test('push has only a dedicated database attachment and a narrowly scoped deployment lease', () => { + const service = terraform('push-gateway.tf') + const database = terraform('push-dedicated-database.tf') + assert.match(service, /instances = \[google_sql_database_instance\.push_dedicated\[0\]\.connection_name\]/) + assert.match(service, /secret\s*= google_secret_manager_secret\.push_dedicated_database_url\[0\]\.secret_id/) + assert.match(service, /version = google_secret_manager_secret_version\.push_dedicated_database_url\[0\]\.version/) + assert.doesNotMatch(service + database, /push_dedicated_database_(?:active|enabled)|local\.relay_database_connection_name|resource "google_sql_database" "push"/) + assert.match(database, /tier\s*= "db-custom-2-7680"/) + assert.match(database, /availability_type = "REGIONAL"/) + assert.match(database, /deletion_protection\s*= true/) + assert.match(database, /deletion_protection_enabled = true/) + const identity = terraform('push-deploy-identity.tf') + const lease = identity.match(/resource "google_storage_bucket_iam_member" "github_push_rollout_lease" \{([\s\S]*?)\n\}/)?.[1] + assert.ok(lease) + assert.match(lease, /member = local\.push_deploy_member/) + assert.match(lease, /role\s*= "roles\/storage.objectAdmin"/) + assert.match(lease, /resource.name == 'projects\/_\/buckets\/\$\{var.project_id\}-terraform-state\/objects\/terraform\/state\/push-rollout\/production.lock'/) +}) diff --git a/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs b/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs index 9f0ee3d7eaa..79036918f23 100644 --- a/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs +++ b/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs @@ -33,13 +33,6 @@ function requiredInteger(source, pattern, label) { return value } -// A tfvars file states only what it overrides, so an absent key means the variable default holds. -// Reading the default as the fallback keeps this honest either way. -function overriddenInteger(override, overridePattern, source, pattern, label) { - if (!overridePattern.test(override)) return requiredInteger(source, pattern, label) - return requiredInteger(override, overridePattern, label) -} - function productionCells(source, defaultPoolMax) { const fencedMatch = source.match(/relay_gce_fenced_cells\s*=\s*\[([^\]]*)\]/) if (!fencedMatch) throw new Error('could not read fenced Relay cells') @@ -59,13 +52,11 @@ function productionCells(source, defaultPoolMax) { } export function calculateRelayCloudSqlConnectionBudget(inputs) { - const pushDraw = inputs.pushInstances * inputs.pushPoolMax const consumers = { cells: inputs.cellPoolTotal + inputs.asiaCellCount * inputs.asiaPoolMax, directors: inputs.directorInstances * inputs.directorPoolMax, auth: inputs.authInstances * inputs.authPoolMax, - api: inputs.apiInstances * inputs.apiPoolMax, - push: pushDraw + api: inputs.apiInstances * inputs.apiPoolMax } const configuredMaximum = Object.values(consumers).reduce((total, value) => total + value, 0) const retainedDirectorRollback = inputs.directorInstances * inputs.directorPoolMax @@ -73,8 +64,6 @@ export function calculateRelayCloudSqlConnectionBudget(inputs) { relayDirectorCandidate: retainedDirectorRollback * 2, apiCandidate: retainedDirectorRollback + inputs.apiInstances * inputs.apiPoolMax, authCandidate: retainedDirectorRollback + inputs.authInstances * inputs.authPoolMax, - // Serving is already counted; validation/rejected and its successor add two pools. - pushCandidate: retainedDirectorRollback + pushDraw * 2, relayCells: retainedDirectorRollback } const rolloutOverlap = Math.max(...Object.values(candidateOverlap)) @@ -142,20 +131,6 @@ export function readRelayCloudSqlConnectionBudget({ /variable\s+"relay_director_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/, 'director pool maximum' ), - // The mobile push gateway shares this instance. Its draw was invisible here until Terraform - // declared the pool: docs/push-gateway.md, "Shape". - pushInstances: overriddenInteger( - productionTfvars, - /^\s*push_max_instances\s*=\s*(\d+)/m, - terraformVariables, - /variable\s+"push_max_instances"[\s\S]*?default\s*=\s*(\d+)/, - 'push gateway instances' - ), - pushPoolMax: requiredInteger( - terraformVariables, - /variable\s+"push_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/, - 'push gateway pool maximum' - ), authInstances: apps.authInstances, authPoolMax: apps.authPoolMax, apiInstances: apps.apiInstances, diff --git a/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs index 40f46cdf4cf..89d40954f7d 100644 --- a/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs +++ b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs @@ -6,91 +6,28 @@ import { readRelayCloudSqlConnectionBudget } from './relay-cloud-sql-connection-budget.mjs' -// Why these numbers are this tight: the shared instance's 400 connections were already spoken -// for, and the relay shape below leaves exactly five. The gateway is sized to fit in four, two -// instances times a two-connection pool, and its rollout overlap of 23 stays under the API -// candidate's 65, so the Math.max is the API candidate rather than the gateway. -// -// `Deploy Relay Asia Topology` gates on `withinBudget == true`, so the single remaining -// connection is the whole margin. Anything that raises a pool or an instance count moves it. -test('production plus the push gateway keeps allowance and reserve below the ceiling', () => { +test('production shared consumers keep allowance and reserve below the ceiling', () => { const report = readRelayCloudSqlConnectionBudget() - assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50, push: 4 }) + assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 }) assert.deepEqual(report.asia, { cells: 3, poolMax: 10 }) - assert.equal(report.configuredMaximum, 319) + assert.equal(report.configuredMaximum, 315) assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30) assert.equal(report.rolloutOverlap.apiCandidate, 65) assert.equal(report.rolloutOverlap.authCandidate, 35) - assert.equal(report.rolloutOverlap.pushCandidate, 23) assert.equal(report.rolloutOverlap.relayCells, 15) assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15) - // The gateway does not set the maximum; the API candidate does, as it did before it existed. assert.equal(report.rolloutOverlap.maximum, 65) assert.equal(report.maintenanceAdminAllowance, 5) assert.equal(report.explicitReserve, 10) assert.equal(report.usableCeiling, 390) - assert.equal(report.operatingMaximum, 389) - assert.equal(report.remainingWithinUsableCeiling, 1) - assert.equal(report.budgetedTotal, 399) - assert.equal(report.unallocated, 1) - assert.equal(report.withinBudget, true) -}) - -// Why: the same relay shape without a push gateway is the before picture, and it stood at five -// connections clear. Holding it here keeps the gateway's cost visible as the four it takes, -// rather than letting drift elsewhere in the budget hide inside the same margin. -test('the same relay shape without the gateway stays inside the ceiling', () => { - const report = calculateRelayCloudSqlConnectionBudget({ - cellPoolTotal: 200, - asiaCellCount: 3, - asiaPoolMax: 10, - directorInstances: 5, - directorPoolMax: 3, - authInstances: 2, - authPoolMax: 10, - apiInstances: 10, - apiPoolMax: 5, - pushInstances: 0, - pushPoolMax: 0, - maxConnections: 400, - maintenanceAdminAllowance: 5, - explicitReserve: 10 - }) - - assert.equal(report.consumers.push, 0) - assert.equal(report.rolloutOverlap.maximum, 65) assert.equal(report.operatingMaximum, 385) assert.equal(report.remainingWithinUsableCeiling, 5) + assert.equal(report.budgetedTotal, 395) + assert.equal(report.unallocated, 5) assert.equal(report.withinBudget, true) }) -// Why: a tagged candidate is directly addressable and sits outside the service-wide cap, so all three -// push revisions can reach the ceiling at once. The API and auth candidates add one copy; this -// one adds two, like the director candidate. -test('the push rollout scenario triples the gateway draw over the retained director', () => { - const report = calculateRelayCloudSqlConnectionBudget({ - cellPoolTotal: 0, - asiaCellCount: 0, - asiaPoolMax: 0, - directorInstances: 5, - directorPoolMax: 3, - authInstances: 0, - authPoolMax: 0, - apiInstances: 0, - apiPoolMax: 0, - pushInstances: 2, - pushPoolMax: 2, - maxConnections: 400, - maintenanceAdminAllowance: 5, - explicitReserve: 10 - }) - - assert.equal(report.consumers.push, 4) - // Serving is in the base; overlap adds 15 retained director plus two 4-connection pools. - assert.equal(report.rolloutOverlap.pushCandidate, 23) -}) - test('fails closed when pool growth consumes the explicit reserve', () => { const report = calculateRelayCloudSqlConnectionBudget({ cellPoolTotal: 200, @@ -102,14 +39,12 @@ test('fails closed when pool growth consumes the explicit reserve', () => { authPoolMax: 10, apiInstances: 20, apiPoolMax: 5, - pushInstances: 4, - pushPoolMax: 10, maxConnections: 400, maintenanceAdminAllowance: 5, explicitReserve: 10 }) - assert.equal(report.operatingMaximum, 555) + assert.equal(report.operatingMaximum, 515) assert.equal(report.withinBudget, false) }) @@ -141,15 +76,11 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => { }) assert.equal(report.consumers.cells, 14) - // No push_max_instances in this tfvars, so the variable default of one instance holds. - assert.equal(report.consumers.push, 2) - assert.equal(report.operatingMaximum, 48) - assert.equal(report.budgetedTotal, 49) + assert.equal(report.operatingMaximum, 46) + assert.equal(report.budgetedTotal, 47) }) -// Why: production.tfvars overrides push_max_instances down to 2 while variables.tf still defaults -// to 4, so reading the default instead of the override would overstate the live draw by half. -test('a tfvars push_max_instances override wins over the variable default', () => { +test('dedicated push scaling does not consume shared capacity', () => { const report = readRelayCloudSqlConnectionBudget({ proposedAsiaCellCount: 1, appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 }, @@ -175,8 +106,9 @@ test('a tfvars push_max_instances override wins over the variable default', () = explicitReserve: 1 }) - assert.equal(report.consumers.push, 6) - assert.equal(report.rolloutOverlap.pushCandidate, 15) + assert.equal(report.consumers.push, undefined) + assert.equal(report.rolloutOverlap.pushCandidate, undefined) + assert.equal(report.operatingMaximum, 46) }) test('requires strict headroom below the physical ceiling', () => { @@ -190,14 +122,12 @@ test('requires strict headroom below the physical ceiling', () => { authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, - pushInstances: 1, - pushPoolMax: 2, maxConnections: 50, maintenanceAdminAllowance: 9, explicitReserve: 3 }) - assert.equal(report.budgetedTotal, 65) + assert.equal(report.budgetedTotal, 63) assert.equal(report.withinBudget, false) }) diff --git a/cloud/docs/push-database-cutover.md b/cloud/docs/push-database-cutover.md index 3bc60d96dca..e2b0be3970d 100644 --- a/cloud/docs/push-database-cutover.md +++ b/cloud/docs/push-database-cutover.md @@ -1,84 +1,112 @@ -# Dedicated push database cutover +# Dedicated push database operations -The gateway has a dedicated PostgreSQL 17 instance configured in -`infra/terraform/push-dedicated-database.tf`: regional HA, 2 vCPU, 7.5 GiB RAM, 50 GiB SSD -with automatic growth, seven retained backups and seven-day point-in-time recovery. -Cloud SQL and Terraform both protect it from deletion. Connections use the Cloud SQL -connector and a separate Secret Manager secret pinned to its Terraform-managed version. +Push attaches only to its dedicated PostgreSQL 17 instance: regional HA, 2 vCPU, +7.5 GiB RAM, 50 GiB SSD with automatic growth, seven retained backups and seven-day +point-in-time recovery. Cloud SQL and Terraform deletion protections remain enabled. +The Cloud SQL connector uses the dedicated URL secret pinned to its managed version. +There is no shared-storage fallback or provision/activate switch. -Provisioning (`push_dedicated_database_enabled`) and attachment -(`push_dedicated_database_active`) are separate switches. Neither changes the existing -shared database or its credentials. The production feature is still in internal testing; -its owner approved an empty database and discarding the previous test state. No data -transfer or application maintenance mechanism is needed for this initial activation. -Test phones must register again; pending notifications and old sessions do not transfer. -This reset procedure is not suitable after public launch without explicit data-loss approval. +## Existing-resource cleanup: operator prerequisite -## Provision +This is a plan/runbook, not authorization to apply or delete resources. Preserve the +shared Orca instance, dedicated push instance, all dedicated data and identities, and +unrelated resources. The dedicated resource addresses remain unchanged: -Use the production relay backend and environment variables documented in -`infra/terraform/README.md`. Save and review a targeted Terraform plan containing only: +- `google_sql_database_instance.push_dedicated[0]` +- `google_sql_database.push_dedicated[0]` +- `random_password.push_dedicated_database[0]` +- `google_sql_user.push_dedicated[0]` +- `google_secret_manager_secret.push_dedicated_database_url[0]` +- `google_secret_manager_secret_version.push_dedicated_database_url[0]` +- `google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor[0]` -- `google_sql_database_instance.push_dedicated` -- `google_sql_database.push_dedicated` -- `random_password.push_dedicated_database` -- `google_sql_user.push_dedicated` -- `google_secret_manager_secret.push_dedicated_database_url` -- `google_secret_manager_secret_version.push_dedicated_database_url` -- `google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor` +The relay state may still own these six obsolete shared-store resources, whose +configuration is removed. An untargeted plan would propose deleting them; do not apply it: -Require exactly seven additions and no updates or deletions for initial provisioning. -Apply that saved plan with backend locking, then require the same targeted plan to be -empty. Keep sensitive Terraform plans access-restricted; never print secret values or -upload raw state/plan JSON. Verify the instance is RUNNABLE with the expected version, -tier, backup policy, and regional availability. No Cloud Run service changes in this step. +- `google_sql_database.push[0]` +- `google_sql_user.push[0]` +- `random_password.push_database[0]` +- `google_secret_manager_secret.push_database_url[0]` +- `google_secret_manager_secret_version.push_database_url[0]` +- `google_secret_manager_secret_iam_member.push_database_url_runtime_accessor[0]` -## Activate the empty store +1. Use the production backend in `infra/terraform/README.md`. Inspect state addresses and + the live service attachment, pinned secret reference and revision resources without + printing credentials. Require the dedicated attachment and no old shared-store consumers; + source connection drain needs an authorized operator's read-only observation. +2. Have the shared database owner adopt the six legacy resources in an explicitly owned + archival configuration before retiring their relay-state ownership. Retain the former + database's `prevent_destroy` protection and secret versions; do not disable protection, + drop databases, rotate passwords or introduce a second runtime attachment. A reviewed + exact-address state transfer must preserve remote IDs and secret material in approved + Terraform storage, with no credential exports to local files or terminal output. +3. Require the owner's import/ownership plan to preserve existing resources and then an + empty plan for those addresses. Only after adoption is proven may the operator remove + precisely the six former addresses from relay state under backend locking. Do not + automate this via `removed` blocks, broad `state rm`, force, or an untargeted apply. +4. Review a fresh relay plan. Reject every delete or replace affecting either SQL instance, + dedicated databases/users/secrets, or unrelated resources. Target only the intended push + service and lease IAM grant for rollout; review their dependency closure too. Existing + unrelated drift must be handled by its owner, outside this cleanup. -1. Record the current immutable serving image, revision, SQL attachment, and database - secret reference. Confirm traffic is pinned to that revision rather than LATEST. -2. Set `push_dedicated_database_active = true` in production. Save a targeted plan for - `google_cloud_run_v2_service.push`. Inspect its dependency closure and reject unrelated - changes. Require only the SQL attachment and database secret reference to change. -3. Hold the existing Cloud SQL rollout lease while applying that saved service-shape plan. - Terraform ignores traffic and image; verify traffic remains pinned to the old revision. - Do not apply a plan that would shift traffic or revert runtime configuration. -4. Dispatch `cloud-push-deploy.yml` from main with the exact reviewed source SHA. It creates - an inert, read-only candidate inheriting the dedicated attachment, probes readiness and provider - access, then deliberately creates an active successor of the same digest before deleting validation. - Activation starts schema writes and workers before HTTP promotion. Verify the candidate's SQL - attachment and pinned secret reference as well as its image and health. -5. Register a test phone against the deployed origin and prove real APNs delivery. Check - database errors and confirm the old revisions have no traffic or tags and source SQL - connections have drained. Leave the old database intact; do not delete shared resources. +No data transfer, dedicated database reset, or phone re-registration is part of this cleanup. -If activation fails before promotion, the existing HTTP serving revision is unchanged, but -activated workers may already have sent notifications or mutated the queue. Cloud Run will not -delete the latest created revision, even untagged at zero traffic. Recovery creates a known-good -successor first, verifies its template/runtime/secret shape and health, promotes and verifies it, -then deletes rejected and previous consumers. The recovery successor remains serving; it can run -known-good schema/workers before promotion and does not undo earlier queue or schema changes. -A partial activation leaving three resources must retire non-latest inert validation before -recovery creates another; failed retirement stops automation. Every deploy requires a single -serving revision resource at admission, so review and retire historical/leftover revisions under -the lease before dispatch. A Terraform attachment update can itself create such a revision: -verify/promote that known-good image and attachment and retire the former revision before dispatch. -The deploy workflow can roll traffic back on failure; in this internal reset rollout, -that may discard registrations created during the probe window. After successful activation, -application rollback should retain the dedicated attachment and deploy an older compatible -image through the workflow. Returning to the shared store is another explicit state reset, -not a lossless rollback. Future public migrations require a separately rehearsed transfer. +## Schema prerequisite for existing internal test databases -## Capacity and resizing +New schemas omit `push_hosts` and the unused `host_public_key` and `transcript` columns +on `push_challenges`. Authentication still verifies the encrypted transcript and consumes +its challenge digest once; sessions and device ownership are unchanged. No compatibility +migration for unpublished builds runs at application startup. -The initial gateway keeps its existing two-connection pool and two-instance maximum. -Dedicated database rollout pools are capped at 64 total configured pool connections across three simultaneous -revision resources (serving, validation/rejected, and active/recovery successor), leaving room for maintenance and operators; this is an admission -budget, not a throughput claim. Increase the pool only after measuring deployed contention. -Keep the shared database allocation reserved until source connections have drained. +Before deploying onto an older internal schema, an operator must arrange a separately +reviewed schema-preparation job through the approved database execution path. Its entire +scope is dropping `push_hosts` (including its index) and those two unused challenge columns; +preserve challenge digest/expiry/consumption fields and every session, device and delivery +table. Verify that the old NOT NULL columns are absent before admitting the new image. +Do not hand-edit production SQL or reset the dedicated database to satisfy this prerequisite. +Until that job is reviewed and executed, the new image is not ready for an existing schema. -Cloud SQL CPU/RAM resizing is an in-place infrastructure change but can interrupt database -connections. HA does not make a resize interruption-free. Durable accepted events remain in -SQL and workers retry after recovery within their five-minute expiry; requests that never -reach durable acceptance depend on client retries. Schedule resizes and verify reconnection, -queue recovery, readiness, and real delivery afterward. +## Deployment serialization transition + +Finish all old push workflow runs before changing the workflow's lock namespace. An old +shared-lock push run and a new push-lock run do not exclude each other. Hold off new push +dispatches while preparing the following exact changes: + +1. Review the relay-root plan for + `google_storage_bucket_iam_member.github_push_rollout_lease[0]`. It grants only + `roles/storage.objectAdmin` on + `projects/_/buckets/onorca-cloud-terraform-state/objects/terraform/state/push-rollout/production.lock` + to the dedicated push deploy account. The lease action uses object GET/upload/delete, + so no bucket-wide listing or Terraform-state access is needed. +2. After approval, apply only the reviewed IAM/dependency plan. Verify the exact condition + and principal independently. If foundation still grants push membership in the old + `cloud_sql_rollout_lease_members`, its owner removes only that push member; keep Relay's + existing members and permissions. Do not mutate foundation through the relay root. +3. Publish the reviewed workflow on main with `production-push-rollout`, cancellation + disabled, and the existing lease action pointed at the dedicated object. The durable + lease covers admission, candidate validation, activation, traffic changes and recovery. + A stale/conflicting lease stops the run; it is never stolen or force-deleted. +4. Deploy the reviewed image through `cloud-push-deploy.yml`. Preserve candidate readiness, + runtime-provider validation, exact digest/configuration checks, and explicit activation. + Verify the public origin and real notification delivery/dismissal afterward. + +## Recovery and capacity + +Activation starts schema writes and workers before HTTP promotion. Traffic rollback cannot +undo queue or schema changes. Cloud Run cannot delete its latest revision, so failed +activation creates a known-good successor, verifies it, promotes it, then retires rejected +and previous revisions. When partial activation leaves three resources, retire non-latest +inert validation before creating recovery. Failed retirement stops automation. Admission +requires one serving revision resource; retire historical leftovers under the push lease. +Keep the dedicated attachment for application rollback and retain an immutable compatible +image. An image requiring the removed challenge columns needs separate schema review. + +The two-instance ceiling and two-connection pool draw four configured connections, twelve +across three simultaneous revision resources. Terraform caps instances × pool × 3 at 64 +for serving, validation/rejected and active/recovery pools. Push does not draw from Relay's +shared connection budget. Source connections must have drained before treating that old +allocation as free. Increase capacity only after measuring deployed contention. + +Cloud SQL resizing can interrupt connections despite HA. Durable accepted events remain in +SQL; workers retry within each event's original five-minute deadline. Schedule resizes and +verify reconnection, queue recovery, readiness and real delivery afterward. diff --git a/cloud/docs/push-gateway.md b/cloud/docs/push-gateway.md index a4710e5ccec..3e5f0ec14c3 100644 --- a/cloud/docs/push-gateway.md +++ b/cloud/docs/push-gateway.md @@ -29,7 +29,7 @@ edit plus a second set of Apple credentials. | Ingress | all | `INGRESS_TRAFFIC_ALL` | | Invoker | IAM disabled | `invoker_iam_disabled = true` on the service | | Runtime identity | `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com` | `google_service_account.push_runtime` | -| Database | `orca_push` on the shared Cloud SQL instance | `google_sql_database.push` | +| Database | `orca_push` on dedicated HA PostgreSQL 17 | `google_sql_database.push_dedicated` | | Hostname | `push.onorca.dev` | `push_base_url` | The minimum of one instance is deliberate and did not move when the ceiling came down to two. A @@ -37,19 +37,11 @@ cold start delays a notification past the point where it is worth showing, so th keeps a notification prompt. The ceiling is a different question, answered below. -The maximum and the pool are set by the connection budget, not by the gateway's own appetite. Two -instances times a two-connection pool is a draw of 4, and a rollout triples it to 12, because the -tagged candidate is directly addressable and sits outside the service-wide cap. The shared Cloud -SQL instance's 400 connections were already spoken for by the relay cells, the directors, auth, -and the API, which left five. Four is the whole of the room there was, and the gateway fits in -it. - -Two connections per instance is enough for the work. A send runs two or three short queries, so -at concurrency 80 requests queue against the pool for microseconds rather than holding it. A -`lifecycle` precondition refuses a plan whose instances times pool exceeds 4, because a fifth -connection puts the checked budget over its ceiling and blocks `Deploy Relay Asia Topology`, -which gates on it. `dev/scripts/relay-cloud-sql-connection-budget.mjs` counts the gateway and -prints the whole picture. +Push uses its approved dedicated two-vCPU HA database. Two instances with a two-connection +pool draw four connections; three simultaneous revision resources draw twelve. Tagged +candidates can run outside the service-wide cap, so Terraform bounds instances × pool × 3 +at 64 connections, leaving dedicated capacity for maintenance and operators. Increase pool +sizes only after measuring contention. The shared Relay budget excludes push entirely. Authentication is the host proof in `POST /v1/host/challenge`, not Cloud Run IAM, so the service opts out of invoker IAM with `invoker_iam_disabled = true`, exactly as the relay director does. @@ -65,7 +57,7 @@ Set on the container by Terraform: | `PORT` | Cloud Run, container port 8080 | | `ORCA_PUSH_PUBLIC_URL` | `push_base_url` | | `ORCA_PUSH_FCM_PROJECT_ID` | `push_fcm_project_id`, empty means `project_id` | -| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-database-url`, version `latest` | +| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-dedicated-database-url`, pinned version | | `ORCA_PUSH_DATABASE_POOL_MAX` | `push_database_pool_max`, 2 per instance | | `ORCA_PUSH_APNS_KEY` | Secret `orca-cloud-push-apns-key`, version `latest` | | `ORCA_PUSH_APNS_KEY_ID` | Secret `orca-cloud-push-apns-key-id`, version `latest` | @@ -129,11 +121,9 @@ terraform -chdir=infra/terraform import -var-file=environments/production.tfvars 'projects/onorca-cloud/secrets/orca-cloud-push-apple-team-id roles/secretmanager.secretAccessor serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com' ``` -Everything else in `push-gateway.tf` is new and is created by the apply: the `orca_push` -database and user, the database-URL secret and its accessor, the `roles/cloudsql.client` binding -on the runtime account, the Cloud Run service, the domain mapping, and the -three deploy-identity bindings. Save that plan and review it before applying; this root carries -unrelated standing drift, so an untargeted apply is never automatic. +The push resources already exist in production. Preserve their addresses, dedicated database +and identities; review the [database cleanup runbook](./push-database-cutover.md) before applying +changes. This root has unrelated standing drift, so an untargeted apply is never automatic. Two things this root does **not** declare, because the carve assigns them elsewhere. Neither affects whether this root's plan is clean, since an undeclared resource is invisible to it. @@ -156,16 +146,17 @@ It authenticates as the dedicated `orca-cloud-gha-push` identity through to this exact dispatch workflow on main in the production environment. Its distinct principal attribute cannot assume the shared Relay deploy identity. -The account can write images to the existing Artifact Registry repository, deploy the push service, -and impersonate only the push runtime account. Foundation separately grants access to the shared -rollout-lock prefix and bucket metadata; it grants no Terraform-state object access. +The account can write images to Artifact Registry, deploy the push service, impersonate only +the push runtime account, and manage exactly `terraform/state/push-rollout/production.lock` +in the production state bucket. The relay root owns that conditional lease grant. It grants +no Terraform-state object access. Publish `github_push_workload_identity_provider` and +`github_push_deploy_service_account` as the production-environment variables above. -Before the next deployment, apply the reviewed identity changes in the relay root, add -`serviceAccount:orca-cloud-gha-push@onorca-cloud.iam.gserviceaccount.com` to production foundation's -`cloud_sql_rollout_lease_members`, and apply foundation. Publish the relay outputs -`github_push_workload_identity_provider` and `github_push_deploy_service_account` as the two -GitHub production-environment variables above. Keep the shared identity's existing lease grant -for Relay. Do not fall back to that identity if push setup is incomplete. +The workflow uses the `production-push-rollout` concurrency group with cancellation disabled +and the existing durable lease action on the push-specific object. Push and Relay deploy +independently; two push deploys cannot race traffic changes. Finish every old shared-lock push +run before enabling the new workflow and lease grant. See the cleanup runbook for the bounded +IAM transition and removal of any obsolete foundation-owned push membership. The run builds the reviewed `source_sha` while the workflow stays on `main`. Buildx returns its own pushed digest (no mutable-tag lookup); every subsequent check and deployment uses that @@ -173,11 +164,11 @@ same digest. Before any production boot, a network-isolated container checks tha recognizes `ORCA_PUSH_MODE=validation` and rejects invalid modes. Older images that lack this capability are refused before they can connect to production. -Under the production Cloud SQL rollout lease, it records the serving rollback revision and +Under the production push rollout lease, it records the serving rollback revision and asserts Terraform-owned scaling. It deploys a tagged, zero-traffic validation revision: - Validation opens PostgreSQL with `default_transaction_read_only=on` and skips schema setup. -- No delivery worker or challenge, session, delivery, or stale-host pruner starts. +- No delivery worker or challenge, session, or delivery pruner starts. - Only `/health` and `/ready` are available; all application routes return 503. - `/health` attests `mode: validation`; `/ready` checks database connectivity only. It does not prove schema compatibility, provider delivery, or active-worker readiness. Container probes @@ -189,7 +180,7 @@ The workflow verifies the exact image and scaling, probes readiness and mode, an runtime identity with a validate-only FCM request. Cloud Run rejects deletion of the latest created revision even when it has no tag or traffic. Activation therefore creates a successor before removing the validation tag and deleting validation. The dedicated 64-connection budget -and legacy shared budget reserve three simultaneous revision pools: serving, validation/rejected, +reserves three simultaneous revision pools: serving, validation/rejected, and active/recovery successor (12 configured pool connections at the current two-by-two shape). Revision deletion is not proof of physical SQL session drain; verify termination and SQL sessions in controlled rollout acceptance. There is no shutdown sleep used as a drain gate. @@ -393,10 +384,8 @@ small number of concurrent collapse keys per device, so excess pending messages every offline alert is not guaranteed to appear. Socket reconnect reconciles dismissals against the current native tray; it has no stored replay watermark and never recovers a missed OS banner. -### Dedicated database preparation +### Dedicated database operations -`push_dedicated_database_enabled` provisions an independent HA PostgreSQL instance without changing -the live gateway attachment. It defaults to false. Follow [the database cutover runbook](./push-database-cutover.md) -before enabling it or switching stores. The current pre-release activation discards registrations -and queued deliveries; phones re-register on foreground use. A future public-service migration -requires a separate preservation procedure. +Push has one dedicated database attachment, with stable Terraform addresses and deletion +protection. There is no switch to shared storage. Follow the [database operations runbook](./push-database-cutover.md) +for deployment prerequisites, legacy resource ownership, capacity and recovery. diff --git a/cloud/infra/terraform/environments/production.tfvars b/cloud/infra/terraform/environments/production.tfvars index 23eb646efc3..925a171719d 100644 --- a/cloud/infra/terraform/environments/production.tfvars +++ b/cloud/infra/terraform/environments/production.tfvars @@ -416,11 +416,8 @@ relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels # Mobile push gateway. Production is the only environment that runs one; the runtime account, # the three Apple secrets, and their accessor bindings already exist and are imported once # (see docs/push-gateway.md). -push_gateway_enabled = true -push_dedicated_database_enabled = true -push_dedicated_database_active = true -push_base_url = "https://push.onorca.dev" -# Sized so the gateway's rollout overlap, the retained director rollback plus its doubled draw, -# stays under the API candidate's, which keeps the checked Cloud SQL connection budget green. +push_gateway_enabled = true +push_base_url = "https://push.onorca.dev" +# Dedicated push pools allow three revision resources during validation and recovery. push_max_instances = 2 manage_push_domain_mapping = true diff --git a/cloud/infra/terraform/outputs.tf b/cloud/infra/terraform/outputs.tf index 184b3be61f7..5fd3b647b76 100644 --- a/cloud/infra/terraform/outputs.tf +++ b/cloud/infra/terraform/outputs.tf @@ -201,7 +201,7 @@ output "push_runtime_service_account" { } output "push_database_name" { - value = try(google_sql_database.push[0].name, null) + value = try(google_sql_database.push_dedicated[0].name, null) description = "Database isolated for durable push gateway state." } diff --git a/cloud/infra/terraform/push-dedicated-database.tf b/cloud/infra/terraform/push-dedicated-database.tf index eac8f675030..9a4b9e4c69f 100644 --- a/cloud/infra/terraform/push-dedicated-database.tf +++ b/cloud/infra/terraform/push-dedicated-database.tf @@ -1,25 +1,5 @@ -# Provision independently of the gateway's SQL attachment switch. -variable "push_dedicated_database_enabled" { - type = bool - description = "Provision the dedicated push database without switching live gateway traffic." - default = false -} - -variable "push_dedicated_database_active" { - type = bool - description = "Attach the gateway to the provisioned dedicated database; does not copy existing state." - default = false -} - -locals { - push_dedicated_database_count = var.push_gateway_enabled && var.push_dedicated_database_enabled ? 1 : 0 - push_database_connection_name = var.push_dedicated_database_active ? google_sql_database_instance.push_dedicated[0].connection_name : local.relay_database_connection_name - push_database_secret_id = var.push_dedicated_database_active ? google_secret_manager_secret.push_dedicated_database_url[0].secret_id : google_secret_manager_secret.push_database_url[0].secret_id - push_database_secret_version = var.push_dedicated_database_active ? google_secret_manager_secret_version.push_dedicated_database_url[0].version : "latest" -} - resource "google_sql_database_instance" "push_dedicated" { - count = local.push_dedicated_database_count + count = local.push_gateway_count project = var.project_id name = "${var.name_prefix}-push-db" @@ -66,7 +46,7 @@ resource "google_sql_database_instance" "push_dedicated" { } resource "google_sql_database" "push_dedicated" { - count = local.push_dedicated_database_count + count = local.push_gateway_count project = var.project_id name = "orca_push" @@ -78,13 +58,13 @@ resource "google_sql_database" "push_dedicated" { } resource "random_password" "push_dedicated_database" { - count = local.push_dedicated_database_count + count = local.push_gateway_count length = 32 special = false } resource "google_sql_user" "push_dedicated" { - count = local.push_dedicated_database_count + count = local.push_gateway_count project = var.project_id name = "orca_push" @@ -93,7 +73,7 @@ resource "google_sql_user" "push_dedicated" { } resource "google_secret_manager_secret" "push_dedicated_database_url" { - count = local.push_dedicated_database_count + count = local.push_gateway_count project = var.project_id secret_id = "${var.name_prefix}-push-dedicated-database-url" @@ -109,7 +89,7 @@ resource "google_secret_manager_secret" "push_dedicated_database_url" { } resource "google_secret_manager_secret_version" "push_dedicated_database_url" { - count = local.push_dedicated_database_count + count = local.push_gateway_count secret = google_secret_manager_secret.push_dedicated_database_url[0].id secret_data = format( @@ -122,7 +102,7 @@ resource "google_secret_manager_secret_version" "push_dedicated_database_url" { } resource "google_secret_manager_secret_iam_member" "push_dedicated_database_url_accessor" { - count = local.push_dedicated_database_count + count = local.push_gateway_count project = var.project_id secret_id = google_secret_manager_secret.push_dedicated_database_url[0].secret_id diff --git a/cloud/infra/terraform/push-deploy-identity.tf b/cloud/infra/terraform/push-deploy-identity.tf index 9e8eae0ea54..8d04e4155f5 100644 --- a/cloud/infra/terraform/push-deploy-identity.tf +++ b/cloud/infra/terraform/push-deploy-identity.tf @@ -65,3 +65,18 @@ output "github_push_workload_identity_provider" { output "github_push_deploy_service_account" { value = try(google_service_account.github_push_deploy[0].email, null) } + + +resource "google_storage_bucket_iam_member" "github_push_rollout_lease" { + count = local.push_gateway_deploy_count + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.objectAdmin" + member = local.push_deploy_member + + condition { + title = "push_rollout_lease" + description = "Limits push deployment coordination to its own lease object." + expression = "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/push-rollout/production.lock'" + } +} diff --git a/cloud/infra/terraform/push-gateway.tf b/cloud/infra/terraform/push-gateway.tf index cfb27bce0a6..ae6bf6b809a 100644 --- a/cloud/infra/terraform/push-gateway.tf +++ b/cloud/infra/terraform/push-gateway.tf @@ -80,74 +80,6 @@ resource "google_project_iam_member" "push_runtime_cloudsql_client" { member = google_service_account.push_runtime[0].member } -# --- Database ----------------------------------------------------------------------------- -# Gateway state shares the foundation-owned Cloud SQL instance with auth and the relay, and uses -# an isolated database and principal, exactly as relay-database.tf does. The application applies -# its own schema at startup. - -resource "google_sql_database" "push" { - count = local.push_gateway_count - - project = var.project_id - name = "orca_push" - instance = local.relay_database_instance_name - - # Why: this database holds every live device token. Disabling the gateway must not drop it. - lifecycle { - prevent_destroy = true - } -} - -resource "random_password" "push_database" { - count = local.push_gateway_count - - length = 32 - special = false -} - -resource "google_sql_user" "push" { - count = local.push_gateway_count - - project = var.project_id - name = "orca_push" - instance = local.relay_database_instance_name - password = random_password.push_database[0].result -} - -resource "google_secret_manager_secret" "push_database_url" { - count = local.push_gateway_count - - project = var.project_id - secret_id = "${var.name_prefix}-push-database-url" - labels = local.relay_shared_labels - - replication { - auto {} - } -} - -resource "google_secret_manager_secret_version" "push_database_url" { - count = local.push_gateway_count - - secret = google_secret_manager_secret.push_database_url[0].id - secret_data = format( - "postgresql://%s:%s@/%s?host=/cloudsql/%s", - google_sql_user.push[0].name, - random_password.push_database[0].result, - google_sql_database.push[0].name, - local.relay_database_connection_name - ) -} - -resource "google_secret_manager_secret_iam_member" "push_database_url_runtime_accessor" { - count = local.push_gateway_count - - project = var.project_id - secret_id = google_secret_manager_secret.push_database_url[0].secret_id - role = "roles/secretmanager.secretAccessor" - member = google_service_account.push_runtime[0].member -} - # --- Apple credentials ---------------------------------------------------------------------- resource "google_secret_manager_secret" "push_provider" { @@ -207,7 +139,7 @@ resource "google_cloud_run_v2_service" "push" { name = "cloudsql" cloud_sql_instance { - instances = [local.push_database_connection_name] + instances = [google_sql_database_instance.push_dedicated[0].connection_name] } } @@ -233,9 +165,7 @@ resource "google_cloud_run_v2_service" "push" { value = local.push_fcm_project_id } - # Declared rather than left to the application default, so the gateway's share of the - # shared Cloud SQL connection budget is a value this root states and the precondition - # below can bound. + # Bound the declared pool against the dedicated database rollout budget. env { name = "ORCA_PUSH_DATABASE_POOL_MAX" value = tostring(var.push_database_pool_max) @@ -246,8 +176,8 @@ resource "google_cloud_run_v2_service" "push" { value_source { secret_key_ref { - secret = local.push_database_secret_id - version = local.push_database_secret_version + secret = google_secret_manager_secret.push_dedicated_database_url[0].secret_id + version = google_secret_manager_secret_version.push_dedicated_database_url[0].version } } } @@ -298,19 +228,8 @@ resource "google_cloud_run_v2_service" "push" { # 100% LATEST would silently undo either, and this root carries unrelated standing drift, so # that apply need not be a push change at all. lifecycle { - # Shared SQL retains its four-connection allocation until the dedicated attachment is active. precondition { - condition = var.push_dedicated_database_active || var.push_max_instances * var.push_database_pool_max <= 4 - error_message = "Push gateway instances x database pool must stay within its 4-connection share while attached to shared Cloud SQL." - } - - precondition { - condition = !var.push_dedicated_database_active || var.push_dedicated_database_enabled - error_message = "Provision the dedicated push database before activating it." - } - - precondition { - condition = !var.push_dedicated_database_active || var.push_max_instances * var.push_database_pool_max * 3 <= 64 + condition = var.push_max_instances * var.push_database_pool_max * 3 <= 64 error_message = "Dedicated push serving, validation/rejected and successor pools must fit the 64-connection rollout budget." } @@ -325,9 +244,7 @@ resource "google_cloud_run_v2_service" "push" { depends_on = [ data.google_artifact_registry_repository.relay_images, google_project_iam_member.push_runtime_cloudsql_client, - google_secret_manager_secret_iam_member.push_database_url_runtime_accessor, google_secret_manager_secret_iam_member.push_provider_runtime_accessor, - google_secret_manager_secret_version.push_database_url, google_secret_manager_secret_version.push_dedicated_database_url, google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor ] diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf index 3beb0965c5c..68584f64120 100644 --- a/cloud/infra/terraform/variables.tf +++ b/cloud/infra/terraform/variables.tf @@ -548,13 +548,7 @@ variable "push_max_instances" { } } -# Why: the gateway's draw on the shared Cloud SQL instance is instances x pool, and the rollout -# lease is taken for twice that, because a tagged candidate is directly addressable and sits -# outside the service-wide cap. Leaving the pool at its application default made that draw -# invisible to this root, so it is declared here and set on the container. -# -# Two is sized to the work, not to the default: a send runs two or three short queries, and at -# concurrency 80 those queue against the pool for microseconds rather than holding it. +# The dedicated database budget counts pools across all three rollout revision resources. variable "push_database_pool_max" { type = number description = "Push gateway database pool size per instance; instances x pool is its Cloud SQL draw." diff --git a/cloud/packages/push-contract/src/contract.test.ts b/cloud/packages/push-contract/src/contract.test.ts index 7b6d5caf4f3..6986c5c8058 100644 --- a/cloud/packages/push-contract/src/contract.test.ts +++ b/cloud/packages/push-contract/src/contract.test.ts @@ -48,7 +48,6 @@ describe('push contract limits', () => { clockSkewToleranceMs: 30_000, sessionTtlMs: 86_400_000, notificationTtlSeconds: 300, - hostRetentionMs: 3_600_000, unauthenticatedRequestsPerMinutePerIp: 30, authenticatedRequestsPerMinutePerHost: 600 }) @@ -172,7 +171,6 @@ describe('device registration schemas', () => { ).toBe(false) }) - it('shapes the registration and list responses', () => { expect( PushDeviceRegistrationResponseSchema.safeParse({ registrationId: 'reg-1' }).success diff --git a/cloud/packages/push-contract/src/push-limits.ts b/cloud/packages/push-contract/src/push-limits.ts index ef7126e84c5..752b15e6f8f 100644 --- a/cloud/packages/push-contract/src/push-limits.ts +++ b/cloud/packages/push-contract/src/push-limits.ts @@ -16,9 +16,6 @@ export const PUSH_LIMITS = { clockSkewToleranceMs: 30_000, sessionTtlMs: 24 * 60 * 60 * 1000, notificationTtlSeconds: 5 * 60, - // Nothing reads a host row, and any keypair mints one for free, so a host - // with no registration left is kept only long enough to survive a phone swap. - hostRetentionMs: 60 * 60 * 1000, // The challenge and session routes are the only unauthenticated writes, so // they are capped per client IP before any key material is generated. unauthenticatedRequestsPerMinutePerIp: 30, diff --git a/docs/reference/mobile-push-contract.md b/docs/reference/mobile-push-contract.md index 3b3d9cf91d9..c5ff8955dbf 100644 --- a/docs/reference/mobile-push-contract.md +++ b/docs/reference/mobile-push-contract.md @@ -57,12 +57,11 @@ The host keypair is X25519 (box), so it cannot sign. Reuse the relay's challenge - Challenge TTL 10 s, and 10 s is the whole window the gateway honours. The 30 s clock skew tolerance is the host's alone: it validates a timestamp the gateway chose, so it needs the allowance and the gateway does not. A gateway that subtracted the tolerance from its own check would run a 40 s TTL. - Store challenge (id, secret hash, host fingerprint, host public key, expiry) in DB so any Cloud Run + Store challenge (id, expected-proof digest, host fingerprint, expiry, consumption time) in DB so any Cloud Run instance can verify. Expired rows are pruned 30 s late so a slow proof reads as expired rather than as an unknown challenge. -- Issuing a challenge writes no `push_hosts` row. It is unauthenticated, so a `push_hosts` row would be - a free permanent write for any caller. The row is upserted in `POST /v1/host/session` once the proof - verifies, from the public key the challenge row carries. +- No host registry or public-key/transcript copy is stored. The encrypted challenge and expected-proof + digest provide proof verification; session and device rows retain host ownership. `POST /v1/host/session` @@ -232,17 +231,15 @@ metadata server or `GOOGLE_APPLICATION_CREDENTIALS` locally): ### Gateway storage (Postgres in prod, SQLite in tests, same pattern as `cloud/apps/relay/src/database.ts`) -- `push_hosts(host_fingerprint pk, host_public_key, created_at, last_seen_at)`, written only on a - verified proof and pruned after 1 h of no contact when no `push_devices` row still names the host. - Nothing reads it, and any keypair mints a host for free, so it is not allowed to accumulate. - `push_sessions` holds one row per host, enforced by a unique index and transaction lock. Minting a session deletes the host's earlier one, since a desktop holds a single session and only re-proves once it is gone. -- `push_challenges(challenge_id pk, host_fingerprint, host_public_key, secret_hash, transcript, +- `push_challenges(challenge_id pk, host_fingerprint, secret_hash, expires_at, consumed_at)` - `push_sessions(token_hash pk, host_fingerprint, expires_at, created_at)` - `push_devices(registration_id pk, host_fingerprint, device_id, platform, token, apns_environment, dead_at, created_at, updated_at, unique(host_fingerprint, device_id))` - `push_events` holds logical event identity, content fingerprint, quota timestamp, and expiry. + Omitted `kind` and explicit `kind: "alert"` have the same fingerprint; changed content still conflicts. - `push_event_recipients` records accepted event/phone pairs for idempotent fanout. - `push_delivery_batches` holds individual payload envelopes, retry deadlines, and renewable worker leases. @@ -368,7 +365,12 @@ Secret Manager names (already exist in `onorca-cloud`): `orca-cloud-push-apns-ke - Workflow `.github/workflows/cloud-push-deploy.yml`: gated on `vars.ORCA_CLOUD_OPERATIONS_ENABLED`, Workload Identity like `cloud-relay-*`, builds a reviewed full `source_sha`, deploys with `--no-traffic`, probes the new revision's `/ready` and a validate-only FCM send, then shifts 100% traffic. Uses - `.github/actions/cloud-sql-rollout-lease` around the schema step. + `.github/actions/cloud-sql-rollout-lease` throughout rollout using + `terraform/state/push-rollout/production.lock` and concurrency group `production-push-rollout`. + The deploy identity can manage only that lease object. Push uses only its dedicated 2-vCPU HA + database and is excluded from Relay's shared database budget. See + [database operations](../../cloud/docs/push-database-cutover.md) for existing-schema preparation, + obsolete resource ownership, and the lock transition before deployment. - Add the new root files to `cloud/dev/contracts` and `cloud/dev/fixtures` partitions so `terraform-root-partition.test.mjs` and `Cloud Verify` pass.