mirror of
https://github.com/stablyai/orca.git
synced 2026-09-29 00:02:56 +00:00
refactor(push): remove redundant cloud storage and rollout coupling
This commit is contained in:
@@ -16,10 +16,9 @@ permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
# Push uses dedicated Cloud SQL; its schema startup and connection-budget rollout intentionally
|
||||
# retain the production rollout coordination group and lease shared with Relay.
|
||||
# Serialize push traffic changes independently of Relay and the shared database.
|
||||
concurrency:
|
||||
group: production-cloud-sql-rollout
|
||||
group: production-push-rollout
|
||||
cancel-in-progress: false
|
||||
|
||||
defaults:
|
||||
@@ -85,10 +84,7 @@ jobs:
|
||||
- name: Configure Docker auth
|
||||
run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet
|
||||
|
||||
# Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance,
|
||||
# and a multi-minute image build inside the lease blocks every relay deploy and rehome for
|
||||
# its duration. The lease below covers exactly the connection-budget window: deploy, probe,
|
||||
# shift.
|
||||
# Building an image does not need the deployment lease.
|
||||
- name: Build and publish the immutable gateway image
|
||||
shell: bash
|
||||
run: |
|
||||
@@ -122,7 +118,7 @@ jobs:
|
||||
- uses: ./.github/actions/cloud-sql-rollout-lease
|
||||
with:
|
||||
bucket: onorca-cloud-terraform-state
|
||||
object: terraform/state/cloud-sql-rollout/production.lock
|
||||
object: terraform/state/push-rollout/production.lock
|
||||
|
||||
# Why: the candidate inherits the serving revision's scaling. A serving revision that has
|
||||
# drifted below the floor would hand the candidate a cold start on every notification, and
|
||||
|
||||
+2
-2
@@ -82,8 +82,8 @@ surface: publish and deploy the director, roll GCE cell capacity, operate Asia
|
||||
admission and regional rehoming, prove staging capacity, monitor production,
|
||||
power staging up and down, and deploy the mobile push gateway.
|
||||
`.github/actions/cloud-sql-rollout-lease` is the compare-and-swap lease that
|
||||
serializes every rollout against the shared Cloud SQL instance, the push
|
||||
gateway deploy included.
|
||||
serializes rollouts against the shared Cloud SQL instance. Push reuses that
|
||||
action with its own lease object and deployment concurrency group.
|
||||
|
||||
Every one of them is inert. Each top-level job is gated on
|
||||
`vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true'`, a repository variable that is
|
||||
|
||||
@@ -199,3 +199,58 @@ it('does not resurrect an in-flight alert after a dismissal and transient provid
|
||||
expect(await store.claim()).toBeNull()
|
||||
expect(await store.pendingCount('phone')).toBe(0)
|
||||
})
|
||||
|
||||
it.each([false, true])(
|
||||
'normalizes default alert kind (explicit first: %s)',
|
||||
async (explicitFirst) => {
|
||||
const { db, store } = await fixture()
|
||||
const { kind: _kind, ...implicit } = notification(1)
|
||||
const explicit = { kind: 'alert' as const, ...implicit }
|
||||
for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) {
|
||||
expect(await store.accept('host', 'phone', event)).toBe('queued')
|
||||
}
|
||||
expect(await store.pendingCount('phone')).toBe(1)
|
||||
expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(1)
|
||||
expect(await store.accept('host', 'phone', { ...explicit, body: 'changed' })).toBe('error')
|
||||
expect(await store.accept('host', 'phone', { ...implicit, kind: 'dismiss' })).toBe('queued')
|
||||
expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(2)
|
||||
}
|
||||
)
|
||||
|
||||
it('fences late renew and finish after an expired claim is dismissed', async () => {
|
||||
const { db, store, advance } = await fixture()
|
||||
const alert = notification(1)
|
||||
await store.accept('host', 'phone', alert)
|
||||
const stale = (await store.claim())!
|
||||
advance(DELIVERY_LEASE_MS)
|
||||
await store.accept('host', 'phone', {
|
||||
...notification(2, 'dismiss'),
|
||||
notificationId: alert.notificationId
|
||||
})
|
||||
const read = async () =>
|
||||
(
|
||||
await db.query(
|
||||
'SELECT state, payload_json, lease_until FROM push_delivery_batches WHERE batch_id = ?',
|
||||
[stale.id]
|
||||
)
|
||||
)[0]
|
||||
const cancelled = await read()
|
||||
expect(cancelled).toMatchObject({ state: 'dismissed', payload_json: '{}' })
|
||||
await store.renew(stale)
|
||||
expect(await read()).toEqual(cancelled)
|
||||
await store.finish(stale, 1000)
|
||||
expect(await read()).toEqual(cancelled)
|
||||
await store.finish(stale)
|
||||
expect(await read()).toEqual(cancelled)
|
||||
const dismissal = (await store.claim())!
|
||||
expect(dismissal.notification.kind).toBe('dismiss')
|
||||
await store.finish(dismissal)
|
||||
await store.accept('host', 'phone', notification(3))
|
||||
const fresh = (await store.claim())!
|
||||
await store.finish(fresh, 1000)
|
||||
advance(1000)
|
||||
const retry = (await store.claim())!
|
||||
expect(retry.id).toBe(fresh.id)
|
||||
await store.finish(retry)
|
||||
expect(await store.claim()).toBeNull()
|
||||
})
|
||||
|
||||
@@ -34,8 +34,10 @@ export class DurablePushStore {
|
||||
JSON.stringify([host, kind, notification.notificationEpoch, notification.notificationSeq])
|
||||
)
|
||||
.digest('hex')
|
||||
const { sound: _sound, ...content } = notification
|
||||
const fingerprint = createHash('sha256').update(JSON.stringify(content)).digest('hex')
|
||||
const { sound: _sound, kind: _kind, ...content } = notification
|
||||
const fingerprint = createHash('sha256')
|
||||
.update(JSON.stringify({ kind, ...content }))
|
||||
.digest('hex')
|
||||
return this.database.transaction(async (tx) => {
|
||||
await tx.lockQuotaScope(`push-events:${host}`)
|
||||
const [existing] = await tx.query('SELECT * FROM push_events WHERE event_id = ?', [eventId])
|
||||
@@ -135,7 +137,7 @@ export class DurablePushStore {
|
||||
|
||||
async renew(delivery: QueuedPushDelivery): Promise<void> {
|
||||
await this.database.query(
|
||||
'UPDATE push_delivery_batches SET lease_until = ? WHERE batch_id = ? AND lease_token = ?',
|
||||
"UPDATE push_delivery_batches SET lease_until = ? WHERE batch_id = ? AND lease_token = ? AND state = 'pending'",
|
||||
[this.now() + DELIVERY_LEASE_MS, delivery.id, delivery.lease]
|
||||
)
|
||||
}
|
||||
@@ -150,7 +152,7 @@ export class DurablePushStore {
|
||||
const retry = retryAt < delivery.expiresAt
|
||||
await this.database.query(
|
||||
`UPDATE push_delivery_batches SET state = ?, payload_json = ?, due_at = ?, lease_until = 0, lease_token = NULL
|
||||
WHERE batch_id = ? AND lease_token = ?`,
|
||||
WHERE batch_id = ? AND lease_token = ? AND state = 'pending'`,
|
||||
[
|
||||
retry ? 'pending' : retryAfterMs !== undefined ? 'expired' : outcome,
|
||||
retry ? JSON.stringify(delivery.notification) : '{}',
|
||||
|
||||
@@ -43,8 +43,6 @@ describe('push host challenge store', () => {
|
||||
ok: true,
|
||||
hostFingerprint: deriveHostFingerprint(host.publicKey)
|
||||
})
|
||||
const [hostRow] = await database.query('SELECT host_fingerprint, last_seen_at FROM push_hosts')
|
||||
expect(hostRow?.host_fingerprint).toBe(deriveHostFingerprint(host.publicKey))
|
||||
})
|
||||
|
||||
it('never stores material that reproduces the proof', async () => {
|
||||
@@ -183,58 +181,6 @@ describe('push host challenge store', () => {
|
||||
await expect(store.issue(Buffer.alloc(31, 1).toString('base64'))).resolves.toBeNull()
|
||||
})
|
||||
|
||||
it('creates no host row until a proof succeeds', async () => {
|
||||
const host = createPushHostKeypair(30)
|
||||
const challenge = await store.issue(hostPublicKeyB64(host))
|
||||
const [beforeProof] = await database.query('SELECT COUNT(*) AS hosts FROM push_hosts')
|
||||
expect(Number(beforeProof?.hosts)).toBe(0)
|
||||
|
||||
const proof = answerPushHostChallenge(challenge!, {
|
||||
gatewayOrigin: GATEWAY_ORIGIN,
|
||||
keypair: host,
|
||||
now: () => clock
|
||||
})!
|
||||
await expect(store.verify(challenge!.challengeId, proof)).resolves.toMatchObject({ ok: true })
|
||||
const [row] = await database.query('SELECT host_public_key, last_seen_at FROM push_hosts')
|
||||
expect(row?.host_public_key).toBe(hostPublicKeyB64(host))
|
||||
expect(Number(row?.last_seen_at)).toBe(clock)
|
||||
})
|
||||
|
||||
it('leaves no host row behind when a challenge is never answered', async () => {
|
||||
for (let index = 0; index < 5; index++) {
|
||||
await store.issue(hostPublicKeyB64(createPushHostKeypair(40 + index)))
|
||||
}
|
||||
const [row] = await database.query('SELECT COUNT(*) AS hosts FROM push_hosts')
|
||||
expect(Number(row?.hosts)).toBe(0)
|
||||
})
|
||||
|
||||
it('prunes a host past retention only when it has no registration left', async () => {
|
||||
const stale = createPushHostKeypair(50)
|
||||
const kept = createPushHostKeypair(51)
|
||||
for (const host of [stale, kept]) {
|
||||
const challenge = await store.issue(hostPublicKeyB64(host))
|
||||
const proof = answerPushHostChallenge(challenge!, {
|
||||
gatewayOrigin: GATEWAY_ORIGIN,
|
||||
keypair: host,
|
||||
now: () => clock
|
||||
})!
|
||||
await store.verify(challenge!.challengeId, proof)
|
||||
}
|
||||
await database.query(
|
||||
`INSERT INTO push_devices (registration_id, host_fingerprint, device_id, platform, token,
|
||||
created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)`,
|
||||
['reg-1', deriveHostFingerprint(kept.publicKey), 'device-1', 'android', 'token', clock, clock]
|
||||
)
|
||||
|
||||
clock += PUSH_LIMITS.hostRetentionMs
|
||||
expect(await store.pruneStaleHosts()).toBe(0)
|
||||
clock += 1
|
||||
expect(await store.pruneStaleHosts()).toBe(1)
|
||||
const [row] = await database.query('SELECT host_fingerprint FROM push_hosts')
|
||||
expect(row?.host_fingerprint).toBe(deriveHostFingerprint(kept.publicKey))
|
||||
})
|
||||
|
||||
it('prunes challenges that fell out of the skew window', async () => {
|
||||
const host = createPushHostKeypair(9)
|
||||
await store.issue(hostPublicKeyB64(host))
|
||||
|
||||
@@ -69,21 +69,17 @@ export class PushHostChallengeStore {
|
||||
const expectedProof = createHmac('sha256', challengeSecret)
|
||||
.update(buildPushHostProofMacInput(transcript))
|
||||
.digest()
|
||||
// No push_hosts row yet: issuing is unauthenticated, so anyone could
|
||||
// otherwise fill the table. The key rides the challenge until verify() proves it.
|
||||
await this.database.query(
|
||||
`INSERT INTO push_challenges
|
||||
(challenge_id, host_fingerprint, host_public_key, secret_hash, transcript, expires_at,
|
||||
(challenge_id, host_fingerprint, secret_hash, expires_at,
|
||||
consumed_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?, NULL)`,
|
||||
VALUES (?, ?, ?, ?, NULL)`,
|
||||
[
|
||||
challengeId,
|
||||
hostFingerprint,
|
||||
hostPublicKeyB64,
|
||||
// The stored digest is of the ack the secret produces, never of the
|
||||
// secret itself: a database reader must not be able to forge a proof.
|
||||
sha256(expectedProof),
|
||||
Buffer.from(transcript).toString('base64'),
|
||||
expiresAt
|
||||
]
|
||||
)
|
||||
@@ -101,7 +97,7 @@ export class PushHostChallengeStore {
|
||||
const proof = decodeCanonicalBase64(proofB64, 32)
|
||||
return await this.database.transaction<PushProofVerification>(async (transaction) => {
|
||||
const [row] = await transaction.query(
|
||||
`SELECT host_fingerprint, host_public_key, secret_hash, expires_at, consumed_at
|
||||
`SELECT host_fingerprint, secret_hash, expires_at, consumed_at
|
||||
FROM push_challenges WHERE challenge_id = ?`,
|
||||
[challengeId]
|
||||
)
|
||||
@@ -123,12 +119,6 @@ export class PushHostChallengeStore {
|
||||
[now, challengeId]
|
||||
)
|
||||
if (Number(consumed?.changes ?? 0) !== 1) return { ok: false, reason: 'already_consumed' }
|
||||
await this.rememberHost(
|
||||
transaction,
|
||||
String(row.host_fingerprint),
|
||||
String(row.host_public_key),
|
||||
now
|
||||
)
|
||||
return { ok: true, hostFingerprint: String(row.host_fingerprint) }
|
||||
})
|
||||
}
|
||||
@@ -142,31 +132,4 @@ export class PushHostChallengeStore {
|
||||
])
|
||||
return Number(result?.changes ?? 0)
|
||||
}
|
||||
|
||||
// A host that stopped proving and has no registration left is dead weight;
|
||||
// its public key is recoverable from the desktop on the next challenge.
|
||||
async pruneStaleHosts(): Promise<number> {
|
||||
const [result] = await this.database.query(
|
||||
`DELETE FROM push_hosts
|
||||
WHERE last_seen_at < ?
|
||||
AND host_fingerprint NOT IN (SELECT host_fingerprint FROM push_devices)`,
|
||||
[this.now() - PUSH_LIMITS.hostRetentionMs]
|
||||
)
|
||||
return Number(result?.changes ?? 0)
|
||||
}
|
||||
|
||||
private async rememberHost(
|
||||
transaction: PushDatabase,
|
||||
hostFingerprint: string,
|
||||
hostPublicKeyB64: string,
|
||||
now: number
|
||||
): Promise<void> {
|
||||
await transaction.query(
|
||||
`INSERT INTO push_hosts (host_fingerprint, host_public_key, created_at, last_seen_at)
|
||||
VALUES (?, ?, ?, ?)
|
||||
ON CONFLICT(host_fingerprint) DO UPDATE SET
|
||||
host_public_key = excluded.host_public_key, last_seen_at = excluded.last_seen_at`,
|
||||
[hostFingerprint, hostPublicKeyB64, now, now]
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,119 +0,0 @@
|
||||
import { expect, it } from 'vitest'
|
||||
import { PUSH_LIMITS } from '@orca-cloud/push-contract'
|
||||
import { openPushDatabase, type PushDatabase } from './push-database.js'
|
||||
import { PushHostChallengeStore } from './host-challenge-store.js'
|
||||
import {
|
||||
answerPushHostChallenge,
|
||||
createPushHostKeypair,
|
||||
hostPublicKeyB64
|
||||
} from './host-challenge-answering.test-fixture.js'
|
||||
|
||||
const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL
|
||||
it.skipIf(!databaseUrl)(
|
||||
'accepts concurrent first proofs, including after host pruning',
|
||||
async () => {
|
||||
if (!process.env.CI && new URL(databaseUrl!).port !== '55440')
|
||||
throw new Error('isolated_postgres_port_required')
|
||||
const admin = await openPushDatabase({ databaseUrl, dataDir: '' })
|
||||
const schema = `first_proof_${Date.now()}`
|
||||
await admin.query(`CREATE SCHEMA ${schema}`)
|
||||
const isolatedUrl = new URL(databaseUrl!)
|
||||
isolatedUrl.searchParams.set('options', `-c search_path=${schema}`)
|
||||
const database = await openPushDatabase({
|
||||
databaseUrl: isolatedUrl.toString(),
|
||||
dataDir: '',
|
||||
poolMax: 4
|
||||
})
|
||||
const host = createPushHostKeypair()
|
||||
const origin = 'https://push.onorca.dev'
|
||||
let now = Date.now()
|
||||
let release!: () => void
|
||||
let arrivals = 0
|
||||
let gate: Promise<void>
|
||||
let concurrent = true
|
||||
const wrapped: PushDatabase = {
|
||||
dialect: database.dialect,
|
||||
query: database.query.bind(database),
|
||||
close: database.close.bind(database),
|
||||
lockQuotaScope: database.lockQuotaScope.bind(database),
|
||||
transaction: (run) =>
|
||||
database.transaction((tx) =>
|
||||
run({
|
||||
dialect: tx.dialect,
|
||||
close: tx.close.bind(tx),
|
||||
transaction: tx.transaction.bind(tx),
|
||||
lockQuotaScope: tx.lockQuotaScope.bind(tx),
|
||||
query: async (sql, params) => {
|
||||
// Both transactions reach the host insert before either can commit.
|
||||
if (concurrent && sql.trimStart().startsWith('INSERT INTO push_hosts')) {
|
||||
if (++arrivals === 2) release()
|
||||
await gate
|
||||
}
|
||||
return tx.query(sql, params)
|
||||
}
|
||||
})
|
||||
)
|
||||
}
|
||||
const store = new PushHostChallengeStore(wrapped, origin, () => now)
|
||||
let fingerprint = ''
|
||||
try {
|
||||
for (let round = 0; round < 2; round++) {
|
||||
arrivals = 0
|
||||
concurrent = true
|
||||
gate = new Promise<void>((resolve) => {
|
||||
release = resolve
|
||||
})
|
||||
const challenges = await Promise.all([
|
||||
store.issue(hostPublicKeyB64(host)),
|
||||
store.issue(hostPublicKeyB64(host))
|
||||
])
|
||||
fingerprint = challenges[0]!.hostFingerprint
|
||||
const results = await Promise.allSettled(
|
||||
challenges.map((challenge) =>
|
||||
store.verify(
|
||||
challenge!.challengeId,
|
||||
answerPushHostChallenge(challenge!, {
|
||||
gatewayOrigin: origin,
|
||||
keypair: host,
|
||||
now: () => now
|
||||
})!
|
||||
)
|
||||
)
|
||||
)
|
||||
expect(results).toEqual(
|
||||
challenges.map(() => ({
|
||||
status: 'fulfilled',
|
||||
value: { ok: true, hostFingerprint: fingerprint }
|
||||
}))
|
||||
)
|
||||
const createdAt = now
|
||||
concurrent = false
|
||||
now += 1000
|
||||
const next = (await store.issue(hostPublicKeyB64(host)))!
|
||||
await store.verify(
|
||||
next.challengeId,
|
||||
answerPushHostChallenge(next, { gatewayOrigin: origin, keypair: host, now: () => now })!
|
||||
)
|
||||
const rows = await database.query('SELECT * FROM push_hosts WHERE host_fingerprint = ?', [
|
||||
fingerprint
|
||||
])
|
||||
expect(rows).toHaveLength(1)
|
||||
expect(Number(rows[0]!.created_at)).toBe(createdAt)
|
||||
expect(Number(rows[0]!.last_seen_at)).toBe(now)
|
||||
expect(rows[0]!.host_public_key).toBe(hostPublicKeyB64(host))
|
||||
now += PUSH_LIMITS.hostRetentionMs + 1
|
||||
await store.pruneStaleHosts()
|
||||
expect(
|
||||
await database.query('SELECT 1 FROM push_hosts WHERE host_fingerprint = ?', [fingerprint])
|
||||
).toEqual([])
|
||||
}
|
||||
} finally {
|
||||
release?.()
|
||||
await database.query('DELETE FROM push_challenges WHERE host_fingerprint = ?', [fingerprint])
|
||||
await database.query('DELETE FROM push_hosts WHERE host_fingerprint = ?', [fingerprint])
|
||||
await database.close()
|
||||
await admin.query(`DROP SCHEMA ${schema} CASCADE`)
|
||||
await admin.close()
|
||||
}
|
||||
}
|
||||
)
|
||||
@@ -0,0 +1,47 @@
|
||||
import { expect, it } from 'vitest'
|
||||
import { openInMemoryPushDatabase, openPushDatabase } from './push-database.js'
|
||||
import { PushHostChallengeStore } from './host-challenge-store.js'
|
||||
import {
|
||||
answerPushHostChallenge,
|
||||
createPushHostKeypair,
|
||||
hostPublicKeyB64
|
||||
} from './host-challenge-answering.test-fixture.js'
|
||||
|
||||
it('accepts independent proofs but consumes each challenge only once under concurrency', async () => {
|
||||
const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL
|
||||
if (databaseUrl && !process.env.CI && new URL(databaseUrl).port !== '55440') {
|
||||
throw new Error('isolated_postgres_port_required')
|
||||
}
|
||||
const db = databaseUrl
|
||||
? await openPushDatabase({ databaseUrl, dataDir: '', poolMax: 4 })
|
||||
: await openInMemoryPushDatabase()
|
||||
const host = createPushHostKeypair()
|
||||
const origin = 'https://push.onorca.dev'
|
||||
const store = new PushHostChallengeStore(db, origin)
|
||||
const challenges = await Promise.all([
|
||||
store.issue(hostPublicKeyB64(host)),
|
||||
store.issue(hostPublicKeyB64(host))
|
||||
])
|
||||
try {
|
||||
const proofs = challenges.map((challenge) =>
|
||||
answerPushHostChallenge(challenge!, { gatewayOrigin: origin, keypair: host })!
|
||||
)
|
||||
const results = await Promise.all(
|
||||
challenges.flatMap((challenge, index) =>
|
||||
Array.from({ length: 5 }, () => store.verify(challenge!.challengeId, proofs[index]!))
|
||||
)
|
||||
)
|
||||
expect(results.filter((result) => result.ok)).toEqual([
|
||||
{ ok: true, hostFingerprint: challenges[0]!.hostFingerprint },
|
||||
{ ok: true, hostFingerprint: challenges[0]!.hostFingerprint }
|
||||
])
|
||||
expect(results.filter((result) => !result.ok)).toEqual(
|
||||
Array.from({ length: 8 }, () => ({ ok: false, reason: 'already_consumed' }))
|
||||
)
|
||||
} finally {
|
||||
for (const challenge of challenges) {
|
||||
await db.query('DELETE FROM push_challenges WHERE challenge_id = ?', [challenge!.challengeId])
|
||||
}
|
||||
await db.close()
|
||||
}
|
||||
})
|
||||
@@ -4,7 +4,6 @@ import type { createPushServer } from './push-server.js'
|
||||
const CHALLENGE_PRUNE_INTERVAL_MS = 60_000
|
||||
const SESSION_PRUNE_INTERVAL_MS = 10 * 60_000
|
||||
const DELIVERY_PRUNE_INTERVAL_MS = 60_000
|
||||
const STALE_HOST_PRUNE_INTERVAL_MS = 30 * 60_000
|
||||
|
||||
function prune(label: string, run: () => Promise<number>, intervalMs: number): NodeJS.Timeout {
|
||||
const timer = setInterval(() => {
|
||||
@@ -34,8 +33,7 @@ export function startPushBackground(
|
||||
const timers = [
|
||||
prune('challenges', () => challenges.pruneExpired(), CHALLENGE_PRUNE_INTERVAL_MS),
|
||||
prune('sessions', () => sessions.pruneExpired(), SESSION_PRUNE_INTERVAL_MS),
|
||||
prune('deliveries', () => deliveryStore.prune(), DELIVERY_PRUNE_INTERVAL_MS),
|
||||
prune('stale_hosts', () => challenges.pruneStaleHosts(), STALE_HOST_PRUNE_INTERVAL_MS)
|
||||
prune('deliveries', () => deliveryStore.prune(), DELIVERY_PRUNE_INTERVAL_MS)
|
||||
]
|
||||
worker.start()
|
||||
return async () => {
|
||||
|
||||
@@ -2,21 +2,10 @@ import { DURABLE_PUSH_SCHEMA } from './durable-push-schema.js'
|
||||
// Applied at startup for both dialects, including additive queue tables,
|
||||
// so every column type has to read the same in SQLite and PostgreSQL.
|
||||
const PUSH_SCHEMA = `
|
||||
CREATE TABLE IF NOT EXISTS push_hosts (
|
||||
host_fingerprint TEXT PRIMARY KEY,
|
||||
host_public_key TEXT NOT NULL,
|
||||
created_at BIGINT NOT NULL,
|
||||
last_seen_at BIGINT NOT NULL
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS push_challenges (
|
||||
challenge_id TEXT PRIMARY KEY,
|
||||
host_fingerprint TEXT NOT NULL,
|
||||
-- Carried here so a host row is only written once a proof succeeds; an
|
||||
-- unauthenticated challenge must not be able to create one.
|
||||
host_public_key TEXT NOT NULL,
|
||||
secret_hash TEXT NOT NULL,
|
||||
transcript TEXT NOT NULL,
|
||||
expires_at BIGINT NOT NULL,
|
||||
consumed_at BIGINT
|
||||
);
|
||||
@@ -44,10 +33,6 @@ CREATE TABLE IF NOT EXISTS push_devices (
|
||||
);
|
||||
CREATE UNIQUE INDEX IF NOT EXISTS push_devices_host_device
|
||||
ON push_devices(host_fingerprint, device_id);
|
||||
|
||||
-- The stale-host pruner scans by last contact. Its owning-host subquery rides
|
||||
-- the push_devices_host_device index.
|
||||
CREATE INDEX IF NOT EXISTS push_hosts_last_seen_at ON push_hosts(last_seen_at);
|
||||
`
|
||||
|
||||
export function pushSchemaStatements(): string[] {
|
||||
|
||||
@@ -34,3 +34,31 @@ it('returns queued for concurrent retries without double quota or delivery', asy
|
||||
await h.flushDeliveries()
|
||||
expect(h.fcmRequests).toHaveLength(2)
|
||||
})
|
||||
|
||||
it.each([false, true])(
|
||||
'accepts default alert kind equivalently through the API (explicit first: %s)',
|
||||
async (explicitFirst) => {
|
||||
const h = await createPushServerHarness()
|
||||
harnesses.push(h)
|
||||
const token = await h.signIn(createPushHostKeypair(3))
|
||||
const registrationId = await h.registerAndroid(token)
|
||||
const implicit = notification()
|
||||
const explicit = { kind: 'alert', ...implicit }
|
||||
for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) {
|
||||
const response = await h.post(
|
||||
'/v1/send',
|
||||
{ v: 1, registrationIds: [registrationId], notification: event },
|
||||
token
|
||||
)
|
||||
expect(await response.json()).toEqual({ results: [{ registrationId, status: 'queued' }] })
|
||||
}
|
||||
await h.flushDeliveries()
|
||||
expect(h.fcmRequests).toHaveLength(1)
|
||||
const changed = await h.post(
|
||||
'/v1/send',
|
||||
{ v: 1, registrationIds: [registrationId], notification: { ...explicit, body: 'changed' } },
|
||||
token
|
||||
)
|
||||
expect(await changed.json()).toEqual({ results: [{ registrationId, status: 'error' }] })
|
||||
}
|
||||
)
|
||||
|
||||
@@ -70,7 +70,7 @@ it.skipIf(!databaseUrl)(
|
||||
expect((await runtime.app.request('/v1/send', { method: 'POST' })).status).toBe(503)
|
||||
expect(calls.mock.calls.map(([sql]) => sql)).toEqual(['SELECT 1 AS ready'])
|
||||
await expect(
|
||||
database.query('DELETE FROM public.push_hosts WHERE false')
|
||||
database.query('DELETE FROM public.push_challenges WHERE false')
|
||||
).rejects.toMatchObject({ code: '25006' })
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
|
||||
@@ -188,13 +188,11 @@
|
||||
"google_project_iam_member.relay_runtime_artifact_reader",
|
||||
"google_project_iam_member.relay_runtime_cloudsql_client",
|
||||
"google_project_iam_member.relay_runtime_log_writer",
|
||||
"google_secret_manager_secret.push_database_url",
|
||||
"google_secret_manager_secret.push_dedicated_database_url",
|
||||
"google_secret_manager_secret.push_provider",
|
||||
"google_secret_manager_secret.relay_assignment_signing_key",
|
||||
"google_secret_manager_secret.relay_database_url",
|
||||
"google_secret_manager_secret.relay_regional_placement_enabled",
|
||||
"google_secret_manager_secret_iam_member.push_database_url_runtime_accessor",
|
||||
"google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor",
|
||||
"google_secret_manager_secret_iam_member.push_provider_runtime_accessor",
|
||||
"google_secret_manager_secret_iam_member.relay_assignment_signing_key_accessor",
|
||||
@@ -206,7 +204,6 @@
|
||||
"google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer",
|
||||
"google_secret_manager_secret_iam_member.relay_regional_placement_director_accessor",
|
||||
"google_secret_manager_secret_iam_member.relay_regional_placement_runtime_accessor",
|
||||
"google_secret_manager_secret_version.push_database_url",
|
||||
"google_secret_manager_secret_version.push_dedicated_database_url",
|
||||
"google_secret_manager_secret_version.relay_assignment_signing_key",
|
||||
"google_secret_manager_secret_version.relay_database_url",
|
||||
@@ -242,14 +239,13 @@
|
||||
"google_service_account_iam_member.github_staging_relay_deploy_auth_runtime_user",
|
||||
"google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user",
|
||||
"google_service_account_iam_member.relay_fence_broker_requester_token_creator",
|
||||
"google_sql_database.push",
|
||||
"google_sql_database.push_dedicated",
|
||||
"google_sql_database.relay",
|
||||
"google_sql_database_instance.push_dedicated",
|
||||
"google_sql_user.push",
|
||||
"google_sql_user.push_dedicated",
|
||||
"google_sql_user.relay",
|
||||
"google_storage_bucket_iam_member.github_production_relay_capacity_state",
|
||||
"google_storage_bucket_iam_member.github_push_rollout_lease",
|
||||
"google_storage_bucket_iam_member.github_relay_asia_topology_state",
|
||||
"google_storage_bucket_iam_member.github_relay_asia_topology_state_list",
|
||||
"google_storage_bucket_iam_member.github_staging_relay_capacity_state",
|
||||
@@ -257,7 +253,6 @@
|
||||
"google_storage_bucket_iam_member.github_staging_relay_deploy_state_list",
|
||||
"google_storage_bucket_iam_member.relay_fence_broker_bucket_reader",
|
||||
"google_storage_bucket_iam_member.relay_fence_broker_state_objects",
|
||||
"random_password.push_database",
|
||||
"random_password.push_dedicated_database",
|
||||
"random_password.relay_assignment_signing_key",
|
||||
"random_password.relay_database"
|
||||
|
||||
@@ -283,8 +283,6 @@ export const LEASED_WORKFLOWS = named([
|
||||
'operate-relay-production-rehome.yml',
|
||||
production({ leaseFiles: ['operate-relay-production-rehome-job.yml'] })
|
||||
],
|
||||
// The gateway applies its schema at startup, so its deploy revision is the schema step.
|
||||
['push-deploy.yml', production()],
|
||||
['deploy-relay-asia-topology.yml', eitherEnvironment()],
|
||||
['operate-relay-asia-admission.yml', eitherEnvironment()],
|
||||
['deploy-relay-staging.yml', staging()],
|
||||
@@ -300,6 +298,10 @@ export const LEASED_WORKFLOWS = named([
|
||||
])
|
||||
|
||||
export const NOT_A_CLOUD_SQL_CANDIDATE = named([
|
||||
[
|
||||
'push-deploy.yml',
|
||||
'Push attaches only to its dedicated database and serializes its own traffic changes under production-push-rollout and the durable push-rollout lease; push-gateway-workflow tests verify both.'
|
||||
],
|
||||
[
|
||||
'monitor-relay-production.yml',
|
||||
'Read-only. Its identity holds monitoring, logging, Cloud SQL and compute viewer roles only, and it runs `gcloud sql instances describe`, never a mutation. It consumes no connection budget, so the durable lease would only let monitoring block a rollout and a rollout block monitoring.'
|
||||
|
||||
@@ -12,7 +12,7 @@ import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs'
|
||||
|
||||
// Why: the push gateway holds the APNs key and is the only thing standing between a paired
|
||||
// phone and a silent notification pipeline. Its deploy is a blue/green rollout against the
|
||||
// shared Cloud SQL instance, and each of the guarantees below is one careless edit from gone.
|
||||
// dedicated Cloud SQL instance, and each of the guarantees below is one careless edit from gone.
|
||||
const WORKFLOW = 'push-deploy.yml'
|
||||
const workflow = readRelayWorkflow(WORKFLOW)
|
||||
const deploy = () => {
|
||||
@@ -60,15 +60,15 @@ test('Terraform trusts this exact workflow file on the production deploy provide
|
||||
assert.equal(relayWorkflowFile(WORKFLOW), 'cloud-push-deploy.yml')
|
||||
})
|
||||
|
||||
test('the rollout is serialized and leases the production Cloud SQL rollout lock', () => {
|
||||
test('the rollout is serialized and leases its dedicated push rollout lock', () => {
|
||||
const blocks = concurrencyBlocks(workflow)
|
||||
assert.equal(blocks.length, 1)
|
||||
assert.equal(blocks[0].group, 'production-cloud-sql-rollout')
|
||||
assert.equal(blocks[0].group, 'production-push-rollout')
|
||||
assert.equal(blocks[0].cancelInProgress, 'false')
|
||||
const steps = leaseSteps(workflow)
|
||||
assert.equal(steps.length, 1, 'exactly one lease step, held for the whole run')
|
||||
assert.equal(steps[0].bucket, 'onorca-cloud-terraform-state')
|
||||
assert.equal(steps[0].object, 'terraform/state/cloud-sql-rollout/production.lock')
|
||||
assert.equal(steps[0].object, 'terraform/state/push-rollout/production.lock')
|
||||
assert.equal(steps[0].release, undefined, 'release stays at its default for a single-job run')
|
||||
})
|
||||
|
||||
@@ -138,7 +138,7 @@ test('the database pool size is Terraform-owned and bounded at plan time', () =>
|
||||
assert.ok(block, 'the push service no longer declares a lifecycle block')
|
||||
assert.match(
|
||||
block[1],
|
||||
/var\.push_max_instances \* var\.push_database_pool_max <= 4/,
|
||||
/var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/,
|
||||
'instances x pool must be bounded at plan time'
|
||||
)
|
||||
assert.match(
|
||||
@@ -313,3 +313,22 @@ test('push credentials cannot assume the shared Relay deploy identity', () => {
|
||||
test('dedicated database admits three simultaneous revision pools', () => {
|
||||
assert.match(terraform('push-gateway.tf'), /var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/)
|
||||
})
|
||||
|
||||
test('push has only a dedicated database attachment and a narrowly scoped deployment lease', () => {
|
||||
const service = terraform('push-gateway.tf')
|
||||
const database = terraform('push-dedicated-database.tf')
|
||||
assert.match(service, /instances = \[google_sql_database_instance\.push_dedicated\[0\]\.connection_name\]/)
|
||||
assert.match(service, /secret\s*= google_secret_manager_secret\.push_dedicated_database_url\[0\]\.secret_id/)
|
||||
assert.match(service, /version = google_secret_manager_secret_version\.push_dedicated_database_url\[0\]\.version/)
|
||||
assert.doesNotMatch(service + database, /push_dedicated_database_(?:active|enabled)|local\.relay_database_connection_name|resource "google_sql_database" "push"/)
|
||||
assert.match(database, /tier\s*= "db-custom-2-7680"/)
|
||||
assert.match(database, /availability_type = "REGIONAL"/)
|
||||
assert.match(database, /deletion_protection\s*= true/)
|
||||
assert.match(database, /deletion_protection_enabled = true/)
|
||||
const identity = terraform('push-deploy-identity.tf')
|
||||
const lease = identity.match(/resource "google_storage_bucket_iam_member" "github_push_rollout_lease" \{([\s\S]*?)\n\}/)?.[1]
|
||||
assert.ok(lease)
|
||||
assert.match(lease, /member = local\.push_deploy_member/)
|
||||
assert.match(lease, /role\s*= "roles\/storage.objectAdmin"/)
|
||||
assert.match(lease, /resource.name == 'projects\/_\/buckets\/\$\{var.project_id\}-terraform-state\/objects\/terraform\/state\/push-rollout\/production.lock'/)
|
||||
})
|
||||
|
||||
@@ -33,13 +33,6 @@ function requiredInteger(source, pattern, label) {
|
||||
return value
|
||||
}
|
||||
|
||||
// A tfvars file states only what it overrides, so an absent key means the variable default holds.
|
||||
// Reading the default as the fallback keeps this honest either way.
|
||||
function overriddenInteger(override, overridePattern, source, pattern, label) {
|
||||
if (!overridePattern.test(override)) return requiredInteger(source, pattern, label)
|
||||
return requiredInteger(override, overridePattern, label)
|
||||
}
|
||||
|
||||
function productionCells(source, defaultPoolMax) {
|
||||
const fencedMatch = source.match(/relay_gce_fenced_cells\s*=\s*\[([^\]]*)\]/)
|
||||
if (!fencedMatch) throw new Error('could not read fenced Relay cells')
|
||||
@@ -59,13 +52,11 @@ function productionCells(source, defaultPoolMax) {
|
||||
}
|
||||
|
||||
export function calculateRelayCloudSqlConnectionBudget(inputs) {
|
||||
const pushDraw = inputs.pushInstances * inputs.pushPoolMax
|
||||
const consumers = {
|
||||
cells: inputs.cellPoolTotal + inputs.asiaCellCount * inputs.asiaPoolMax,
|
||||
directors: inputs.directorInstances * inputs.directorPoolMax,
|
||||
auth: inputs.authInstances * inputs.authPoolMax,
|
||||
api: inputs.apiInstances * inputs.apiPoolMax,
|
||||
push: pushDraw
|
||||
api: inputs.apiInstances * inputs.apiPoolMax
|
||||
}
|
||||
const configuredMaximum = Object.values(consumers).reduce((total, value) => total + value, 0)
|
||||
const retainedDirectorRollback = inputs.directorInstances * inputs.directorPoolMax
|
||||
@@ -73,8 +64,6 @@ export function calculateRelayCloudSqlConnectionBudget(inputs) {
|
||||
relayDirectorCandidate: retainedDirectorRollback * 2,
|
||||
apiCandidate: retainedDirectorRollback + inputs.apiInstances * inputs.apiPoolMax,
|
||||
authCandidate: retainedDirectorRollback + inputs.authInstances * inputs.authPoolMax,
|
||||
// Serving is already counted; validation/rejected and its successor add two pools.
|
||||
pushCandidate: retainedDirectorRollback + pushDraw * 2,
|
||||
relayCells: retainedDirectorRollback
|
||||
}
|
||||
const rolloutOverlap = Math.max(...Object.values(candidateOverlap))
|
||||
@@ -142,20 +131,6 @@ export function readRelayCloudSqlConnectionBudget({
|
||||
/variable\s+"relay_director_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
|
||||
'director pool maximum'
|
||||
),
|
||||
// The mobile push gateway shares this instance. Its draw was invisible here until Terraform
|
||||
// declared the pool: docs/push-gateway.md, "Shape".
|
||||
pushInstances: overriddenInteger(
|
||||
productionTfvars,
|
||||
/^\s*push_max_instances\s*=\s*(\d+)/m,
|
||||
terraformVariables,
|
||||
/variable\s+"push_max_instances"[\s\S]*?default\s*=\s*(\d+)/,
|
||||
'push gateway instances'
|
||||
),
|
||||
pushPoolMax: requiredInteger(
|
||||
terraformVariables,
|
||||
/variable\s+"push_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
|
||||
'push gateway pool maximum'
|
||||
),
|
||||
authInstances: apps.authInstances,
|
||||
authPoolMax: apps.authPoolMax,
|
||||
apiInstances: apps.apiInstances,
|
||||
|
||||
@@ -6,91 +6,28 @@ import {
|
||||
readRelayCloudSqlConnectionBudget
|
||||
} from './relay-cloud-sql-connection-budget.mjs'
|
||||
|
||||
// Why these numbers are this tight: the shared instance's 400 connections were already spoken
|
||||
// for, and the relay shape below leaves exactly five. The gateway is sized to fit in four, two
|
||||
// instances times a two-connection pool, and its rollout overlap of 23 stays under the API
|
||||
// candidate's 65, so the Math.max is the API candidate rather than the gateway.
|
||||
//
|
||||
// `Deploy Relay Asia Topology` gates on `withinBudget == true`, so the single remaining
|
||||
// connection is the whole margin. Anything that raises a pool or an instance count moves it.
|
||||
test('production plus the push gateway keeps allowance and reserve below the ceiling', () => {
|
||||
test('production shared consumers keep allowance and reserve below the ceiling', () => {
|
||||
const report = readRelayCloudSqlConnectionBudget()
|
||||
|
||||
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50, push: 4 })
|
||||
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 })
|
||||
assert.deepEqual(report.asia, { cells: 3, poolMax: 10 })
|
||||
assert.equal(report.configuredMaximum, 319)
|
||||
assert.equal(report.configuredMaximum, 315)
|
||||
assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30)
|
||||
assert.equal(report.rolloutOverlap.apiCandidate, 65)
|
||||
assert.equal(report.rolloutOverlap.authCandidate, 35)
|
||||
assert.equal(report.rolloutOverlap.pushCandidate, 23)
|
||||
assert.equal(report.rolloutOverlap.relayCells, 15)
|
||||
assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15)
|
||||
// The gateway does not set the maximum; the API candidate does, as it did before it existed.
|
||||
assert.equal(report.rolloutOverlap.maximum, 65)
|
||||
assert.equal(report.maintenanceAdminAllowance, 5)
|
||||
assert.equal(report.explicitReserve, 10)
|
||||
assert.equal(report.usableCeiling, 390)
|
||||
assert.equal(report.operatingMaximum, 389)
|
||||
assert.equal(report.remainingWithinUsableCeiling, 1)
|
||||
assert.equal(report.budgetedTotal, 399)
|
||||
assert.equal(report.unallocated, 1)
|
||||
assert.equal(report.withinBudget, true)
|
||||
})
|
||||
|
||||
// Why: the same relay shape without a push gateway is the before picture, and it stood at five
|
||||
// connections clear. Holding it here keeps the gateway's cost visible as the four it takes,
|
||||
// rather than letting drift elsewhere in the budget hide inside the same margin.
|
||||
test('the same relay shape without the gateway stays inside the ceiling', () => {
|
||||
const report = calculateRelayCloudSqlConnectionBudget({
|
||||
cellPoolTotal: 200,
|
||||
asiaCellCount: 3,
|
||||
asiaPoolMax: 10,
|
||||
directorInstances: 5,
|
||||
directorPoolMax: 3,
|
||||
authInstances: 2,
|
||||
authPoolMax: 10,
|
||||
apiInstances: 10,
|
||||
apiPoolMax: 5,
|
||||
pushInstances: 0,
|
||||
pushPoolMax: 0,
|
||||
maxConnections: 400,
|
||||
maintenanceAdminAllowance: 5,
|
||||
explicitReserve: 10
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.push, 0)
|
||||
assert.equal(report.rolloutOverlap.maximum, 65)
|
||||
assert.equal(report.operatingMaximum, 385)
|
||||
assert.equal(report.remainingWithinUsableCeiling, 5)
|
||||
assert.equal(report.budgetedTotal, 395)
|
||||
assert.equal(report.unallocated, 5)
|
||||
assert.equal(report.withinBudget, true)
|
||||
})
|
||||
|
||||
// Why: a tagged candidate is directly addressable and sits outside the service-wide cap, so all three
|
||||
// push revisions can reach the ceiling at once. The API and auth candidates add one copy; this
|
||||
// one adds two, like the director candidate.
|
||||
test('the push rollout scenario triples the gateway draw over the retained director', () => {
|
||||
const report = calculateRelayCloudSqlConnectionBudget({
|
||||
cellPoolTotal: 0,
|
||||
asiaCellCount: 0,
|
||||
asiaPoolMax: 0,
|
||||
directorInstances: 5,
|
||||
directorPoolMax: 3,
|
||||
authInstances: 0,
|
||||
authPoolMax: 0,
|
||||
apiInstances: 0,
|
||||
apiPoolMax: 0,
|
||||
pushInstances: 2,
|
||||
pushPoolMax: 2,
|
||||
maxConnections: 400,
|
||||
maintenanceAdminAllowance: 5,
|
||||
explicitReserve: 10
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.push, 4)
|
||||
// Serving is in the base; overlap adds 15 retained director plus two 4-connection pools.
|
||||
assert.equal(report.rolloutOverlap.pushCandidate, 23)
|
||||
})
|
||||
|
||||
test('fails closed when pool growth consumes the explicit reserve', () => {
|
||||
const report = calculateRelayCloudSqlConnectionBudget({
|
||||
cellPoolTotal: 200,
|
||||
@@ -102,14 +39,12 @@ test('fails closed when pool growth consumes the explicit reserve', () => {
|
||||
authPoolMax: 10,
|
||||
apiInstances: 20,
|
||||
apiPoolMax: 5,
|
||||
pushInstances: 4,
|
||||
pushPoolMax: 10,
|
||||
maxConnections: 400,
|
||||
maintenanceAdminAllowance: 5,
|
||||
explicitReserve: 10
|
||||
})
|
||||
|
||||
assert.equal(report.operatingMaximum, 555)
|
||||
assert.equal(report.operatingMaximum, 515)
|
||||
assert.equal(report.withinBudget, false)
|
||||
})
|
||||
|
||||
@@ -141,15 +76,11 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => {
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.cells, 14)
|
||||
// No push_max_instances in this tfvars, so the variable default of one instance holds.
|
||||
assert.equal(report.consumers.push, 2)
|
||||
assert.equal(report.operatingMaximum, 48)
|
||||
assert.equal(report.budgetedTotal, 49)
|
||||
assert.equal(report.operatingMaximum, 46)
|
||||
assert.equal(report.budgetedTotal, 47)
|
||||
})
|
||||
|
||||
// Why: production.tfvars overrides push_max_instances down to 2 while variables.tf still defaults
|
||||
// to 4, so reading the default instead of the override would overstate the live draw by half.
|
||||
test('a tfvars push_max_instances override wins over the variable default', () => {
|
||||
test('dedicated push scaling does not consume shared capacity', () => {
|
||||
const report = readRelayCloudSqlConnectionBudget({
|
||||
proposedAsiaCellCount: 1,
|
||||
appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 },
|
||||
@@ -175,8 +106,9 @@ test('a tfvars push_max_instances override wins over the variable default', () =
|
||||
explicitReserve: 1
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.push, 6)
|
||||
assert.equal(report.rolloutOverlap.pushCandidate, 15)
|
||||
assert.equal(report.consumers.push, undefined)
|
||||
assert.equal(report.rolloutOverlap.pushCandidate, undefined)
|
||||
assert.equal(report.operatingMaximum, 46)
|
||||
})
|
||||
|
||||
test('requires strict headroom below the physical ceiling', () => {
|
||||
@@ -190,14 +122,12 @@ test('requires strict headroom below the physical ceiling', () => {
|
||||
authPoolMax: 10,
|
||||
apiInstances: 1,
|
||||
apiPoolMax: 5,
|
||||
pushInstances: 1,
|
||||
pushPoolMax: 2,
|
||||
maxConnections: 50,
|
||||
maintenanceAdminAllowance: 9,
|
||||
explicitReserve: 3
|
||||
})
|
||||
|
||||
assert.equal(report.budgetedTotal, 65)
|
||||
assert.equal(report.budgetedTotal, 63)
|
||||
assert.equal(report.withinBudget, false)
|
||||
})
|
||||
|
||||
|
||||
@@ -1,84 +1,112 @@
|
||||
# Dedicated push database cutover
|
||||
# Dedicated push database operations
|
||||
|
||||
The gateway has a dedicated PostgreSQL 17 instance configured in
|
||||
`infra/terraform/push-dedicated-database.tf`: regional HA, 2 vCPU, 7.5 GiB RAM, 50 GiB SSD
|
||||
with automatic growth, seven retained backups and seven-day point-in-time recovery.
|
||||
Cloud SQL and Terraform both protect it from deletion. Connections use the Cloud SQL
|
||||
connector and a separate Secret Manager secret pinned to its Terraform-managed version.
|
||||
Push attaches only to its dedicated PostgreSQL 17 instance: regional HA, 2 vCPU,
|
||||
7.5 GiB RAM, 50 GiB SSD with automatic growth, seven retained backups and seven-day
|
||||
point-in-time recovery. Cloud SQL and Terraform deletion protections remain enabled.
|
||||
The Cloud SQL connector uses the dedicated URL secret pinned to its managed version.
|
||||
There is no shared-storage fallback or provision/activate switch.
|
||||
|
||||
Provisioning (`push_dedicated_database_enabled`) and attachment
|
||||
(`push_dedicated_database_active`) are separate switches. Neither changes the existing
|
||||
shared database or its credentials. The production feature is still in internal testing;
|
||||
its owner approved an empty database and discarding the previous test state. No data
|
||||
transfer or application maintenance mechanism is needed for this initial activation.
|
||||
Test phones must register again; pending notifications and old sessions do not transfer.
|
||||
This reset procedure is not suitable after public launch without explicit data-loss approval.
|
||||
## Existing-resource cleanup: operator prerequisite
|
||||
|
||||
## Provision
|
||||
This is a plan/runbook, not authorization to apply or delete resources. Preserve the
|
||||
shared Orca instance, dedicated push instance, all dedicated data and identities, and
|
||||
unrelated resources. The dedicated resource addresses remain unchanged:
|
||||
|
||||
Use the production relay backend and environment variables documented in
|
||||
`infra/terraform/README.md`. Save and review a targeted Terraform plan containing only:
|
||||
- `google_sql_database_instance.push_dedicated[0]`
|
||||
- `google_sql_database.push_dedicated[0]`
|
||||
- `random_password.push_dedicated_database[0]`
|
||||
- `google_sql_user.push_dedicated[0]`
|
||||
- `google_secret_manager_secret.push_dedicated_database_url[0]`
|
||||
- `google_secret_manager_secret_version.push_dedicated_database_url[0]`
|
||||
- `google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor[0]`
|
||||
|
||||
- `google_sql_database_instance.push_dedicated`
|
||||
- `google_sql_database.push_dedicated`
|
||||
- `random_password.push_dedicated_database`
|
||||
- `google_sql_user.push_dedicated`
|
||||
- `google_secret_manager_secret.push_dedicated_database_url`
|
||||
- `google_secret_manager_secret_version.push_dedicated_database_url`
|
||||
- `google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor`
|
||||
The relay state may still own these six obsolete shared-store resources, whose
|
||||
configuration is removed. An untargeted plan would propose deleting them; do not apply it:
|
||||
|
||||
Require exactly seven additions and no updates or deletions for initial provisioning.
|
||||
Apply that saved plan with backend locking, then require the same targeted plan to be
|
||||
empty. Keep sensitive Terraform plans access-restricted; never print secret values or
|
||||
upload raw state/plan JSON. Verify the instance is RUNNABLE with the expected version,
|
||||
tier, backup policy, and regional availability. No Cloud Run service changes in this step.
|
||||
- `google_sql_database.push[0]`
|
||||
- `google_sql_user.push[0]`
|
||||
- `random_password.push_database[0]`
|
||||
- `google_secret_manager_secret.push_database_url[0]`
|
||||
- `google_secret_manager_secret_version.push_database_url[0]`
|
||||
- `google_secret_manager_secret_iam_member.push_database_url_runtime_accessor[0]`
|
||||
|
||||
## Activate the empty store
|
||||
1. Use the production backend in `infra/terraform/README.md`. Inspect state addresses and
|
||||
the live service attachment, pinned secret reference and revision resources without
|
||||
printing credentials. Require the dedicated attachment and no old shared-store consumers;
|
||||
source connection drain needs an authorized operator's read-only observation.
|
||||
2. Have the shared database owner adopt the six legacy resources in an explicitly owned
|
||||
archival configuration before retiring their relay-state ownership. Retain the former
|
||||
database's `prevent_destroy` protection and secret versions; do not disable protection,
|
||||
drop databases, rotate passwords or introduce a second runtime attachment. A reviewed
|
||||
exact-address state transfer must preserve remote IDs and secret material in approved
|
||||
Terraform storage, with no credential exports to local files or terminal output.
|
||||
3. Require the owner's import/ownership plan to preserve existing resources and then an
|
||||
empty plan for those addresses. Only after adoption is proven may the operator remove
|
||||
precisely the six former addresses from relay state under backend locking. Do not
|
||||
automate this via `removed` blocks, broad `state rm`, force, or an untargeted apply.
|
||||
4. Review a fresh relay plan. Reject every delete or replace affecting either SQL instance,
|
||||
dedicated databases/users/secrets, or unrelated resources. Target only the intended push
|
||||
service and lease IAM grant for rollout; review their dependency closure too. Existing
|
||||
unrelated drift must be handled by its owner, outside this cleanup.
|
||||
|
||||
1. Record the current immutable serving image, revision, SQL attachment, and database
|
||||
secret reference. Confirm traffic is pinned to that revision rather than LATEST.
|
||||
2. Set `push_dedicated_database_active = true` in production. Save a targeted plan for
|
||||
`google_cloud_run_v2_service.push`. Inspect its dependency closure and reject unrelated
|
||||
changes. Require only the SQL attachment and database secret reference to change.
|
||||
3. Hold the existing Cloud SQL rollout lease while applying that saved service-shape plan.
|
||||
Terraform ignores traffic and image; verify traffic remains pinned to the old revision.
|
||||
Do not apply a plan that would shift traffic or revert runtime configuration.
|
||||
4. Dispatch `cloud-push-deploy.yml` from main with the exact reviewed source SHA. It creates
|
||||
an inert, read-only candidate inheriting the dedicated attachment, probes readiness and provider
|
||||
access, then deliberately creates an active successor of the same digest before deleting validation.
|
||||
Activation starts schema writes and workers before HTTP promotion. Verify the candidate's SQL
|
||||
attachment and pinned secret reference as well as its image and health.
|
||||
5. Register a test phone against the deployed origin and prove real APNs delivery. Check
|
||||
database errors and confirm the old revisions have no traffic or tags and source SQL
|
||||
connections have drained. Leave the old database intact; do not delete shared resources.
|
||||
No data transfer, dedicated database reset, or phone re-registration is part of this cleanup.
|
||||
|
||||
If activation fails before promotion, the existing HTTP serving revision is unchanged, but
|
||||
activated workers may already have sent notifications or mutated the queue. Cloud Run will not
|
||||
delete the latest created revision, even untagged at zero traffic. Recovery creates a known-good
|
||||
successor first, verifies its template/runtime/secret shape and health, promotes and verifies it,
|
||||
then deletes rejected and previous consumers. The recovery successor remains serving; it can run
|
||||
known-good schema/workers before promotion and does not undo earlier queue or schema changes.
|
||||
A partial activation leaving three resources must retire non-latest inert validation before
|
||||
recovery creates another; failed retirement stops automation. Every deploy requires a single
|
||||
serving revision resource at admission, so review and retire historical/leftover revisions under
|
||||
the lease before dispatch. A Terraform attachment update can itself create such a revision:
|
||||
verify/promote that known-good image and attachment and retire the former revision before dispatch.
|
||||
The deploy workflow can roll traffic back on failure; in this internal reset rollout,
|
||||
that may discard registrations created during the probe window. After successful activation,
|
||||
application rollback should retain the dedicated attachment and deploy an older compatible
|
||||
image through the workflow. Returning to the shared store is another explicit state reset,
|
||||
not a lossless rollback. Future public migrations require a separately rehearsed transfer.
|
||||
## Schema prerequisite for existing internal test databases
|
||||
|
||||
## Capacity and resizing
|
||||
New schemas omit `push_hosts` and the unused `host_public_key` and `transcript` columns
|
||||
on `push_challenges`. Authentication still verifies the encrypted transcript and consumes
|
||||
its challenge digest once; sessions and device ownership are unchanged. No compatibility
|
||||
migration for unpublished builds runs at application startup.
|
||||
|
||||
The initial gateway keeps its existing two-connection pool and two-instance maximum.
|
||||
Dedicated database rollout pools are capped at 64 total configured pool connections across three simultaneous
|
||||
revision resources (serving, validation/rejected, and active/recovery successor), leaving room for maintenance and operators; this is an admission
|
||||
budget, not a throughput claim. Increase the pool only after measuring deployed contention.
|
||||
Keep the shared database allocation reserved until source connections have drained.
|
||||
Before deploying onto an older internal schema, an operator must arrange a separately
|
||||
reviewed schema-preparation job through the approved database execution path. Its entire
|
||||
scope is dropping `push_hosts` (including its index) and those two unused challenge columns;
|
||||
preserve challenge digest/expiry/consumption fields and every session, device and delivery
|
||||
table. Verify that the old NOT NULL columns are absent before admitting the new image.
|
||||
Do not hand-edit production SQL or reset the dedicated database to satisfy this prerequisite.
|
||||
Until that job is reviewed and executed, the new image is not ready for an existing schema.
|
||||
|
||||
Cloud SQL CPU/RAM resizing is an in-place infrastructure change but can interrupt database
|
||||
connections. HA does not make a resize interruption-free. Durable accepted events remain in
|
||||
SQL and workers retry after recovery within their five-minute expiry; requests that never
|
||||
reach durable acceptance depend on client retries. Schedule resizes and verify reconnection,
|
||||
queue recovery, readiness, and real delivery afterward.
|
||||
## Deployment serialization transition
|
||||
|
||||
Finish all old push workflow runs before changing the workflow's lock namespace. An old
|
||||
shared-lock push run and a new push-lock run do not exclude each other. Hold off new push
|
||||
dispatches while preparing the following exact changes:
|
||||
|
||||
1. Review the relay-root plan for
|
||||
`google_storage_bucket_iam_member.github_push_rollout_lease[0]`. It grants only
|
||||
`roles/storage.objectAdmin` on
|
||||
`projects/_/buckets/onorca-cloud-terraform-state/objects/terraform/state/push-rollout/production.lock`
|
||||
to the dedicated push deploy account. The lease action uses object GET/upload/delete,
|
||||
so no bucket-wide listing or Terraform-state access is needed.
|
||||
2. After approval, apply only the reviewed IAM/dependency plan. Verify the exact condition
|
||||
and principal independently. If foundation still grants push membership in the old
|
||||
`cloud_sql_rollout_lease_members`, its owner removes only that push member; keep Relay's
|
||||
existing members and permissions. Do not mutate foundation through the relay root.
|
||||
3. Publish the reviewed workflow on main with `production-push-rollout`, cancellation
|
||||
disabled, and the existing lease action pointed at the dedicated object. The durable
|
||||
lease covers admission, candidate validation, activation, traffic changes and recovery.
|
||||
A stale/conflicting lease stops the run; it is never stolen or force-deleted.
|
||||
4. Deploy the reviewed image through `cloud-push-deploy.yml`. Preserve candidate readiness,
|
||||
runtime-provider validation, exact digest/configuration checks, and explicit activation.
|
||||
Verify the public origin and real notification delivery/dismissal afterward.
|
||||
|
||||
## Recovery and capacity
|
||||
|
||||
Activation starts schema writes and workers before HTTP promotion. Traffic rollback cannot
|
||||
undo queue or schema changes. Cloud Run cannot delete its latest revision, so failed
|
||||
activation creates a known-good successor, verifies it, promotes it, then retires rejected
|
||||
and previous revisions. When partial activation leaves three resources, retire non-latest
|
||||
inert validation before creating recovery. Failed retirement stops automation. Admission
|
||||
requires one serving revision resource; retire historical leftovers under the push lease.
|
||||
Keep the dedicated attachment for application rollback and retain an immutable compatible
|
||||
image. An image requiring the removed challenge columns needs separate schema review.
|
||||
|
||||
The two-instance ceiling and two-connection pool draw four configured connections, twelve
|
||||
across three simultaneous revision resources. Terraform caps instances × pool × 3 at 64
|
||||
for serving, validation/rejected and active/recovery pools. Push does not draw from Relay's
|
||||
shared connection budget. Source connections must have drained before treating that old
|
||||
allocation as free. Increase capacity only after measuring deployed contention.
|
||||
|
||||
Cloud SQL resizing can interrupt connections despite HA. Durable accepted events remain in
|
||||
SQL; workers retry within each event's original five-minute deadline. Schedule resizes and
|
||||
verify reconnection, queue recovery, readiness and real delivery afterward.
|
||||
|
||||
+27
-38
@@ -29,7 +29,7 @@ edit plus a second set of Apple credentials.
|
||||
| Ingress | all | `INGRESS_TRAFFIC_ALL` |
|
||||
| Invoker | IAM disabled | `invoker_iam_disabled = true` on the service |
|
||||
| Runtime identity | `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com` | `google_service_account.push_runtime` |
|
||||
| Database | `orca_push` on the shared Cloud SQL instance | `google_sql_database.push` |
|
||||
| Database | `orca_push` on dedicated HA PostgreSQL 17 | `google_sql_database.push_dedicated` |
|
||||
| Hostname | `push.onorca.dev` | `push_base_url` |
|
||||
|
||||
The minimum of one instance is deliberate and did not move when the ceiling came down to two. A
|
||||
@@ -37,19 +37,11 @@ cold start delays a notification past the point where it is worth showing, so th
|
||||
keeps a notification prompt. The
|
||||
ceiling is a different question, answered below.
|
||||
|
||||
The maximum and the pool are set by the connection budget, not by the gateway's own appetite. Two
|
||||
instances times a two-connection pool is a draw of 4, and a rollout triples it to 12, because the
|
||||
tagged candidate is directly addressable and sits outside the service-wide cap. The shared Cloud
|
||||
SQL instance's 400 connections were already spoken for by the relay cells, the directors, auth,
|
||||
and the API, which left five. Four is the whole of the room there was, and the gateway fits in
|
||||
it.
|
||||
|
||||
Two connections per instance is enough for the work. A send runs two or three short queries, so
|
||||
at concurrency 80 requests queue against the pool for microseconds rather than holding it. A
|
||||
`lifecycle` precondition refuses a plan whose instances times pool exceeds 4, because a fifth
|
||||
connection puts the checked budget over its ceiling and blocks `Deploy Relay Asia Topology`,
|
||||
which gates on it. `dev/scripts/relay-cloud-sql-connection-budget.mjs` counts the gateway and
|
||||
prints the whole picture.
|
||||
Push uses its approved dedicated two-vCPU HA database. Two instances with a two-connection
|
||||
pool draw four connections; three simultaneous revision resources draw twelve. Tagged
|
||||
candidates can run outside the service-wide cap, so Terraform bounds instances × pool × 3
|
||||
at 64 connections, leaving dedicated capacity for maintenance and operators. Increase pool
|
||||
sizes only after measuring contention. The shared Relay budget excludes push entirely.
|
||||
|
||||
Authentication is the host proof in `POST /v1/host/challenge`, not Cloud Run IAM, so the service
|
||||
opts out of invoker IAM with `invoker_iam_disabled = true`, exactly as the relay director does.
|
||||
@@ -65,7 +57,7 @@ Set on the container by Terraform:
|
||||
| `PORT` | Cloud Run, container port 8080 |
|
||||
| `ORCA_PUSH_PUBLIC_URL` | `push_base_url` |
|
||||
| `ORCA_PUSH_FCM_PROJECT_ID` | `push_fcm_project_id`, empty means `project_id` |
|
||||
| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-database-url`, version `latest` |
|
||||
| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-dedicated-database-url`, pinned version |
|
||||
| `ORCA_PUSH_DATABASE_POOL_MAX` | `push_database_pool_max`, 2 per instance |
|
||||
| `ORCA_PUSH_APNS_KEY` | Secret `orca-cloud-push-apns-key`, version `latest` |
|
||||
| `ORCA_PUSH_APNS_KEY_ID` | Secret `orca-cloud-push-apns-key-id`, version `latest` |
|
||||
@@ -129,11 +121,9 @@ terraform -chdir=infra/terraform import -var-file=environments/production.tfvars
|
||||
'projects/onorca-cloud/secrets/orca-cloud-push-apple-team-id roles/secretmanager.secretAccessor serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com'
|
||||
```
|
||||
|
||||
Everything else in `push-gateway.tf` is new and is created by the apply: the `orca_push`
|
||||
database and user, the database-URL secret and its accessor, the `roles/cloudsql.client` binding
|
||||
on the runtime account, the Cloud Run service, the domain mapping, and the
|
||||
three deploy-identity bindings. Save that plan and review it before applying; this root carries
|
||||
unrelated standing drift, so an untargeted apply is never automatic.
|
||||
The push resources already exist in production. Preserve their addresses, dedicated database
|
||||
and identities; review the [database cleanup runbook](./push-database-cutover.md) before applying
|
||||
changes. This root has unrelated standing drift, so an untargeted apply is never automatic.
|
||||
|
||||
Two things this root does **not** declare, because the carve assigns them elsewhere. Neither
|
||||
affects whether this root's plan is clean, since an undeclared resource is invisible to it.
|
||||
@@ -156,16 +146,17 @@ It authenticates as the dedicated `orca-cloud-gha-push` identity through
|
||||
to this exact dispatch workflow on main in the production environment. Its distinct principal
|
||||
attribute cannot assume the shared Relay deploy identity.
|
||||
|
||||
The account can write images to the existing Artifact Registry repository, deploy the push service,
|
||||
and impersonate only the push runtime account. Foundation separately grants access to the shared
|
||||
rollout-lock prefix and bucket metadata; it grants no Terraform-state object access.
|
||||
The account can write images to Artifact Registry, deploy the push service, impersonate only
|
||||
the push runtime account, and manage exactly `terraform/state/push-rollout/production.lock`
|
||||
in the production state bucket. The relay root owns that conditional lease grant. It grants
|
||||
no Terraform-state object access. Publish `github_push_workload_identity_provider` and
|
||||
`github_push_deploy_service_account` as the production-environment variables above.
|
||||
|
||||
Before the next deployment, apply the reviewed identity changes in the relay root, add
|
||||
`serviceAccount:orca-cloud-gha-push@onorca-cloud.iam.gserviceaccount.com` to production foundation's
|
||||
`cloud_sql_rollout_lease_members`, and apply foundation. Publish the relay outputs
|
||||
`github_push_workload_identity_provider` and `github_push_deploy_service_account` as the two
|
||||
GitHub production-environment variables above. Keep the shared identity's existing lease grant
|
||||
for Relay. Do not fall back to that identity if push setup is incomplete.
|
||||
The workflow uses the `production-push-rollout` concurrency group with cancellation disabled
|
||||
and the existing durable lease action on the push-specific object. Push and Relay deploy
|
||||
independently; two push deploys cannot race traffic changes. Finish every old shared-lock push
|
||||
run before enabling the new workflow and lease grant. See the cleanup runbook for the bounded
|
||||
IAM transition and removal of any obsolete foundation-owned push membership.
|
||||
|
||||
The run builds the reviewed `source_sha` while the workflow stays on `main`. Buildx returns
|
||||
its own pushed digest (no mutable-tag lookup); every subsequent check and deployment uses that
|
||||
@@ -173,11 +164,11 @@ same digest. Before any production boot, a network-isolated container checks tha
|
||||
recognizes `ORCA_PUSH_MODE=validation` and rejects invalid modes. Older images that lack this
|
||||
capability are refused before they can connect to production.
|
||||
|
||||
Under the production Cloud SQL rollout lease, it records the serving rollback revision and
|
||||
Under the production push rollout lease, it records the serving rollback revision and
|
||||
asserts Terraform-owned scaling. It deploys a tagged, zero-traffic validation revision:
|
||||
|
||||
- Validation opens PostgreSQL with `default_transaction_read_only=on` and skips schema setup.
|
||||
- No delivery worker or challenge, session, delivery, or stale-host pruner starts.
|
||||
- No delivery worker or challenge, session, or delivery pruner starts.
|
||||
- Only `/health` and `/ready` are available; all application routes return 503.
|
||||
- `/health` attests `mode: validation`; `/ready` checks database connectivity only. It does not
|
||||
prove schema compatibility, provider delivery, or active-worker readiness. Container probes
|
||||
@@ -189,7 +180,7 @@ The workflow verifies the exact image and scaling, probes readiness and mode, an
|
||||
runtime identity with a validate-only FCM request. Cloud Run rejects deletion of the latest
|
||||
created revision even when it has no tag or traffic. Activation therefore creates a successor
|
||||
before removing the validation tag and deleting validation. The dedicated 64-connection budget
|
||||
and legacy shared budget reserve three simultaneous revision pools: serving, validation/rejected,
|
||||
reserves three simultaneous revision pools: serving, validation/rejected,
|
||||
and active/recovery successor (12 configured pool connections at the current two-by-two shape).
|
||||
Revision deletion is not proof of physical SQL session drain; verify termination and SQL sessions
|
||||
in controlled rollout acceptance. There is no shutdown sleep used as a drain gate.
|
||||
@@ -393,10 +384,8 @@ small number of concurrent collapse keys per device, so excess pending messages
|
||||
every offline alert is not guaranteed to appear. Socket reconnect reconciles dismissals against the
|
||||
current native tray; it has no stored replay watermark and never recovers a missed OS banner.
|
||||
|
||||
### Dedicated database preparation
|
||||
### Dedicated database operations
|
||||
|
||||
`push_dedicated_database_enabled` provisions an independent HA PostgreSQL instance without changing
|
||||
the live gateway attachment. It defaults to false. Follow [the database cutover runbook](./push-database-cutover.md)
|
||||
before enabling it or switching stores. The current pre-release activation discards registrations
|
||||
and queued deliveries; phones re-register on foreground use. A future public-service migration
|
||||
requires a separate preservation procedure.
|
||||
Push has one dedicated database attachment, with stable Terraform addresses and deletion
|
||||
protection. There is no switch to shared storage. Follow the [database operations runbook](./push-database-cutover.md)
|
||||
for deployment prerequisites, legacy resource ownership, capacity and recovery.
|
||||
|
||||
@@ -416,11 +416,8 @@ relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels
|
||||
# Mobile push gateway. Production is the only environment that runs one; the runtime account,
|
||||
# the three Apple secrets, and their accessor bindings already exist and are imported once
|
||||
# (see docs/push-gateway.md).
|
||||
push_gateway_enabled = true
|
||||
push_dedicated_database_enabled = true
|
||||
push_dedicated_database_active = true
|
||||
push_base_url = "https://push.onorca.dev"
|
||||
# Sized so the gateway's rollout overlap, the retained director rollback plus its doubled draw,
|
||||
# stays under the API candidate's, which keeps the checked Cloud SQL connection budget green.
|
||||
push_gateway_enabled = true
|
||||
push_base_url = "https://push.onorca.dev"
|
||||
# Dedicated push pools allow three revision resources during validation and recovery.
|
||||
push_max_instances = 2
|
||||
manage_push_domain_mapping = true
|
||||
|
||||
@@ -201,7 +201,7 @@ output "push_runtime_service_account" {
|
||||
}
|
||||
|
||||
output "push_database_name" {
|
||||
value = try(google_sql_database.push[0].name, null)
|
||||
value = try(google_sql_database.push_dedicated[0].name, null)
|
||||
description = "Database isolated for durable push gateway state."
|
||||
}
|
||||
|
||||
|
||||
@@ -1,25 +1,5 @@
|
||||
# Provision independently of the gateway's SQL attachment switch.
|
||||
variable "push_dedicated_database_enabled" {
|
||||
type = bool
|
||||
description = "Provision the dedicated push database without switching live gateway traffic."
|
||||
default = false
|
||||
}
|
||||
|
||||
variable "push_dedicated_database_active" {
|
||||
type = bool
|
||||
description = "Attach the gateway to the provisioned dedicated database; does not copy existing state."
|
||||
default = false
|
||||
}
|
||||
|
||||
locals {
|
||||
push_dedicated_database_count = var.push_gateway_enabled && var.push_dedicated_database_enabled ? 1 : 0
|
||||
push_database_connection_name = var.push_dedicated_database_active ? google_sql_database_instance.push_dedicated[0].connection_name : local.relay_database_connection_name
|
||||
push_database_secret_id = var.push_dedicated_database_active ? google_secret_manager_secret.push_dedicated_database_url[0].secret_id : google_secret_manager_secret.push_database_url[0].secret_id
|
||||
push_database_secret_version = var.push_dedicated_database_active ? google_secret_manager_secret_version.push_dedicated_database_url[0].version : "latest"
|
||||
}
|
||||
|
||||
resource "google_sql_database_instance" "push_dedicated" {
|
||||
count = local.push_dedicated_database_count
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
name = "${var.name_prefix}-push-db"
|
||||
@@ -66,7 +46,7 @@ resource "google_sql_database_instance" "push_dedicated" {
|
||||
}
|
||||
|
||||
resource "google_sql_database" "push_dedicated" {
|
||||
count = local.push_dedicated_database_count
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
name = "orca_push"
|
||||
@@ -78,13 +58,13 @@ resource "google_sql_database" "push_dedicated" {
|
||||
}
|
||||
|
||||
resource "random_password" "push_dedicated_database" {
|
||||
count = local.push_dedicated_database_count
|
||||
count = local.push_gateway_count
|
||||
length = 32
|
||||
special = false
|
||||
}
|
||||
|
||||
resource "google_sql_user" "push_dedicated" {
|
||||
count = local.push_dedicated_database_count
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
name = "orca_push"
|
||||
@@ -93,7 +73,7 @@ resource "google_sql_user" "push_dedicated" {
|
||||
}
|
||||
|
||||
resource "google_secret_manager_secret" "push_dedicated_database_url" {
|
||||
count = local.push_dedicated_database_count
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
secret_id = "${var.name_prefix}-push-dedicated-database-url"
|
||||
@@ -109,7 +89,7 @@ resource "google_secret_manager_secret" "push_dedicated_database_url" {
|
||||
}
|
||||
|
||||
resource "google_secret_manager_secret_version" "push_dedicated_database_url" {
|
||||
count = local.push_dedicated_database_count
|
||||
count = local.push_gateway_count
|
||||
|
||||
secret = google_secret_manager_secret.push_dedicated_database_url[0].id
|
||||
secret_data = format(
|
||||
@@ -122,7 +102,7 @@ resource "google_secret_manager_secret_version" "push_dedicated_database_url" {
|
||||
}
|
||||
|
||||
resource "google_secret_manager_secret_iam_member" "push_dedicated_database_url_accessor" {
|
||||
count = local.push_dedicated_database_count
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
secret_id = google_secret_manager_secret.push_dedicated_database_url[0].secret_id
|
||||
|
||||
@@ -65,3 +65,18 @@ output "github_push_workload_identity_provider" {
|
||||
output "github_push_deploy_service_account" {
|
||||
value = try(google_service_account.github_push_deploy[0].email, null)
|
||||
}
|
||||
|
||||
|
||||
resource "google_storage_bucket_iam_member" "github_push_rollout_lease" {
|
||||
count = local.push_gateway_deploy_count
|
||||
|
||||
bucket = "${var.project_id}-terraform-state"
|
||||
role = "roles/storage.objectAdmin"
|
||||
member = local.push_deploy_member
|
||||
|
||||
condition {
|
||||
title = "push_rollout_lease"
|
||||
description = "Limits push deployment coordination to its own lease object."
|
||||
expression = "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/push-rollout/production.lock'"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -80,74 +80,6 @@ resource "google_project_iam_member" "push_runtime_cloudsql_client" {
|
||||
member = google_service_account.push_runtime[0].member
|
||||
}
|
||||
|
||||
# --- Database -----------------------------------------------------------------------------
|
||||
# Gateway state shares the foundation-owned Cloud SQL instance with auth and the relay, and uses
|
||||
# an isolated database and principal, exactly as relay-database.tf does. The application applies
|
||||
# its own schema at startup.
|
||||
|
||||
resource "google_sql_database" "push" {
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
name = "orca_push"
|
||||
instance = local.relay_database_instance_name
|
||||
|
||||
# Why: this database holds every live device token. Disabling the gateway must not drop it.
|
||||
lifecycle {
|
||||
prevent_destroy = true
|
||||
}
|
||||
}
|
||||
|
||||
resource "random_password" "push_database" {
|
||||
count = local.push_gateway_count
|
||||
|
||||
length = 32
|
||||
special = false
|
||||
}
|
||||
|
||||
resource "google_sql_user" "push" {
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
name = "orca_push"
|
||||
instance = local.relay_database_instance_name
|
||||
password = random_password.push_database[0].result
|
||||
}
|
||||
|
||||
resource "google_secret_manager_secret" "push_database_url" {
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
secret_id = "${var.name_prefix}-push-database-url"
|
||||
labels = local.relay_shared_labels
|
||||
|
||||
replication {
|
||||
auto {}
|
||||
}
|
||||
}
|
||||
|
||||
resource "google_secret_manager_secret_version" "push_database_url" {
|
||||
count = local.push_gateway_count
|
||||
|
||||
secret = google_secret_manager_secret.push_database_url[0].id
|
||||
secret_data = format(
|
||||
"postgresql://%s:%s@/%s?host=/cloudsql/%s",
|
||||
google_sql_user.push[0].name,
|
||||
random_password.push_database[0].result,
|
||||
google_sql_database.push[0].name,
|
||||
local.relay_database_connection_name
|
||||
)
|
||||
}
|
||||
|
||||
resource "google_secret_manager_secret_iam_member" "push_database_url_runtime_accessor" {
|
||||
count = local.push_gateway_count
|
||||
|
||||
project = var.project_id
|
||||
secret_id = google_secret_manager_secret.push_database_url[0].secret_id
|
||||
role = "roles/secretmanager.secretAccessor"
|
||||
member = google_service_account.push_runtime[0].member
|
||||
}
|
||||
|
||||
# --- Apple credentials ----------------------------------------------------------------------
|
||||
|
||||
resource "google_secret_manager_secret" "push_provider" {
|
||||
@@ -207,7 +139,7 @@ resource "google_cloud_run_v2_service" "push" {
|
||||
name = "cloudsql"
|
||||
|
||||
cloud_sql_instance {
|
||||
instances = [local.push_database_connection_name]
|
||||
instances = [google_sql_database_instance.push_dedicated[0].connection_name]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -233,9 +165,7 @@ resource "google_cloud_run_v2_service" "push" {
|
||||
value = local.push_fcm_project_id
|
||||
}
|
||||
|
||||
# Declared rather than left to the application default, so the gateway's share of the
|
||||
# shared Cloud SQL connection budget is a value this root states and the precondition
|
||||
# below can bound.
|
||||
# Bound the declared pool against the dedicated database rollout budget.
|
||||
env {
|
||||
name = "ORCA_PUSH_DATABASE_POOL_MAX"
|
||||
value = tostring(var.push_database_pool_max)
|
||||
@@ -246,8 +176,8 @@ resource "google_cloud_run_v2_service" "push" {
|
||||
|
||||
value_source {
|
||||
secret_key_ref {
|
||||
secret = local.push_database_secret_id
|
||||
version = local.push_database_secret_version
|
||||
secret = google_secret_manager_secret.push_dedicated_database_url[0].secret_id
|
||||
version = google_secret_manager_secret_version.push_dedicated_database_url[0].version
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -298,19 +228,8 @@ resource "google_cloud_run_v2_service" "push" {
|
||||
# 100% LATEST would silently undo either, and this root carries unrelated standing drift, so
|
||||
# that apply need not be a push change at all.
|
||||
lifecycle {
|
||||
# Shared SQL retains its four-connection allocation until the dedicated attachment is active.
|
||||
precondition {
|
||||
condition = var.push_dedicated_database_active || var.push_max_instances * var.push_database_pool_max <= 4
|
||||
error_message = "Push gateway instances x database pool must stay within its 4-connection share while attached to shared Cloud SQL."
|
||||
}
|
||||
|
||||
precondition {
|
||||
condition = !var.push_dedicated_database_active || var.push_dedicated_database_enabled
|
||||
error_message = "Provision the dedicated push database before activating it."
|
||||
}
|
||||
|
||||
precondition {
|
||||
condition = !var.push_dedicated_database_active || var.push_max_instances * var.push_database_pool_max * 3 <= 64
|
||||
condition = var.push_max_instances * var.push_database_pool_max * 3 <= 64
|
||||
error_message = "Dedicated push serving, validation/rejected and successor pools must fit the 64-connection rollout budget."
|
||||
}
|
||||
|
||||
@@ -325,9 +244,7 @@ resource "google_cloud_run_v2_service" "push" {
|
||||
depends_on = [
|
||||
data.google_artifact_registry_repository.relay_images,
|
||||
google_project_iam_member.push_runtime_cloudsql_client,
|
||||
google_secret_manager_secret_iam_member.push_database_url_runtime_accessor,
|
||||
google_secret_manager_secret_iam_member.push_provider_runtime_accessor,
|
||||
google_secret_manager_secret_version.push_database_url,
|
||||
google_secret_manager_secret_version.push_dedicated_database_url,
|
||||
google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor
|
||||
]
|
||||
|
||||
@@ -548,13 +548,7 @@ variable "push_max_instances" {
|
||||
}
|
||||
}
|
||||
|
||||
# Why: the gateway's draw on the shared Cloud SQL instance is instances x pool, and the rollout
|
||||
# lease is taken for twice that, because a tagged candidate is directly addressable and sits
|
||||
# outside the service-wide cap. Leaving the pool at its application default made that draw
|
||||
# invisible to this root, so it is declared here and set on the container.
|
||||
#
|
||||
# Two is sized to the work, not to the default: a send runs two or three short queries, and at
|
||||
# concurrency 80 those queue against the pool for microseconds rather than holding it.
|
||||
# The dedicated database budget counts pools across all three rollout revision resources.
|
||||
variable "push_database_pool_max" {
|
||||
type = number
|
||||
description = "Push gateway database pool size per instance; instances x pool is its Cloud SQL draw."
|
||||
|
||||
@@ -48,7 +48,6 @@ describe('push contract limits', () => {
|
||||
clockSkewToleranceMs: 30_000,
|
||||
sessionTtlMs: 86_400_000,
|
||||
notificationTtlSeconds: 300,
|
||||
hostRetentionMs: 3_600_000,
|
||||
unauthenticatedRequestsPerMinutePerIp: 30,
|
||||
authenticatedRequestsPerMinutePerHost: 600
|
||||
})
|
||||
@@ -172,7 +171,6 @@ describe('device registration schemas', () => {
|
||||
).toBe(false)
|
||||
})
|
||||
|
||||
|
||||
it('shapes the registration and list responses', () => {
|
||||
expect(
|
||||
PushDeviceRegistrationResponseSchema.safeParse({ registrationId: 'reg-1' }).success
|
||||
|
||||
@@ -16,9 +16,6 @@ export const PUSH_LIMITS = {
|
||||
clockSkewToleranceMs: 30_000,
|
||||
sessionTtlMs: 24 * 60 * 60 * 1000,
|
||||
notificationTtlSeconds: 5 * 60,
|
||||
// Nothing reads a host row, and any keypair mints one for free, so a host
|
||||
// with no registration left is kept only long enough to survive a phone swap.
|
||||
hostRetentionMs: 60 * 60 * 1000,
|
||||
// The challenge and session routes are the only unauthenticated writes, so
|
||||
// they are capped per client IP before any key material is generated.
|
||||
unauthenticatedRequestsPerMinutePerIp: 30,
|
||||
|
||||
@@ -57,12 +57,11 @@ The host keypair is X25519 (box), so it cannot sign. Reuse the relay's challenge
|
||||
- Challenge TTL 10 s, and 10 s is the whole window the gateway honours. The 30 s clock skew tolerance
|
||||
is the host's alone: it validates a timestamp the gateway chose, so it needs the allowance and the
|
||||
gateway does not. A gateway that subtracted the tolerance from its own check would run a 40 s TTL.
|
||||
Store challenge (id, secret hash, host fingerprint, host public key, expiry) in DB so any Cloud Run
|
||||
Store challenge (id, expected-proof digest, host fingerprint, expiry, consumption time) in DB so any Cloud Run
|
||||
instance can verify. Expired rows are pruned 30 s late so a slow proof reads as expired rather than
|
||||
as an unknown challenge.
|
||||
- Issuing a challenge writes no `push_hosts` row. It is unauthenticated, so a `push_hosts` row would be
|
||||
a free permanent write for any caller. The row is upserted in `POST /v1/host/session` once the proof
|
||||
verifies, from the public key the challenge row carries.
|
||||
- No host registry or public-key/transcript copy is stored. The encrypted challenge and expected-proof
|
||||
digest provide proof verification; session and device rows retain host ownership.
|
||||
|
||||
`POST /v1/host/session`
|
||||
|
||||
@@ -232,17 +231,15 @@ metadata server or `GOOGLE_APPLICATION_CREDENTIALS` locally):
|
||||
|
||||
### Gateway storage (Postgres in prod, SQLite in tests, same pattern as `cloud/apps/relay/src/database.ts`)
|
||||
|
||||
- `push_hosts(host_fingerprint pk, host_public_key, created_at, last_seen_at)`, written only on a
|
||||
verified proof and pruned after 1 h of no contact when no `push_devices` row still names the host.
|
||||
Nothing reads it, and any keypair mints a host for free, so it is not allowed to accumulate.
|
||||
- `push_sessions` holds one row per host, enforced by a unique index and transaction lock. Minting a
|
||||
session deletes the host's earlier one, since a desktop holds a single session and only re-proves once it is gone.
|
||||
- `push_challenges(challenge_id pk, host_fingerprint, host_public_key, secret_hash, transcript,
|
||||
- `push_challenges(challenge_id pk, host_fingerprint, secret_hash,
|
||||
expires_at, consumed_at)`
|
||||
- `push_sessions(token_hash pk, host_fingerprint, expires_at, created_at)`
|
||||
- `push_devices(registration_id pk, host_fingerprint, device_id, platform, token, apns_environment,
|
||||
dead_at, created_at, updated_at, unique(host_fingerprint, device_id))`
|
||||
- `push_events` holds logical event identity, content fingerprint, quota timestamp, and expiry.
|
||||
Omitted `kind` and explicit `kind: "alert"` have the same fingerprint; changed content still conflicts.
|
||||
- `push_event_recipients` records accepted event/phone pairs for idempotent fanout.
|
||||
- `push_delivery_batches` holds individual payload envelopes, retry deadlines, and renewable worker
|
||||
leases.
|
||||
@@ -368,7 +365,12 @@ Secret Manager names (already exist in `onorca-cloud`): `orca-cloud-push-apns-ke
|
||||
- Workflow `.github/workflows/cloud-push-deploy.yml`: gated on `vars.ORCA_CLOUD_OPERATIONS_ENABLED`,
|
||||
Workload Identity like `cloud-relay-*`, builds a reviewed full `source_sha`, deploys with `--no-traffic`, probes the new
|
||||
revision's `/ready` and a validate-only FCM send, then shifts 100% traffic. Uses
|
||||
`.github/actions/cloud-sql-rollout-lease` around the schema step.
|
||||
`.github/actions/cloud-sql-rollout-lease` throughout rollout using
|
||||
`terraform/state/push-rollout/production.lock` and concurrency group `production-push-rollout`.
|
||||
The deploy identity can manage only that lease object. Push uses only its dedicated 2-vCPU HA
|
||||
database and is excluded from Relay's shared database budget. See
|
||||
[database operations](../../cloud/docs/push-database-cutover.md) for existing-schema preparation,
|
||||
obsolete resource ownership, and the lock transition before deployment.
|
||||
- Add the new root files to `cloud/dev/contracts` and `cloud/dev/fixtures` partitions so
|
||||
`terraform-root-partition.test.mjs` and `Cloud Verify` pass.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user