diff --git a/.github/workflows/cloud-push-deploy.yml b/.github/workflows/cloud-push-deploy.yml index a483e984c4f..b165aa4a38c 100644 --- a/.github/workflows/cloud-push-deploy.yml +++ b/.github/workflows/cloud-push-deploy.yml @@ -37,12 +37,15 @@ jobs: IMAGE_NAME: push PUSH_ORIGIN: https://push.onorca.dev PUSH_RUNTIME_SERVICE_ACCOUNT: orca-cloud-push@onorca-cloud.iam.gserviceaccount.com - # Ceiling the tagged candidate, matching push_max_instances in - # environments/production.tfvars. A tagged revision is directly addressable and so sits - # outside the service-wide cap: without this the candidate and the serving revision could - # each reach the ceiling and double the gateway's Cloud SQL draw during the probe window. - # Terraform still owns the value; this only fails a deploy that would exceed it. - PUSH_MAX_INSTANCES: 4 + # Scaling the serving revision must already hold, matching push_min_instances and + # push_max_instances. Terraform owns both, and the candidate inherits them from the + # service, so this deploy never passes a scaling flag: doing so would write a + # Terraform-owned field that `lifecycle.ignore_changes` does not cover, and a later + # `push_max_instances` raise would then be reverted by every deploy. These two values + # are the expected shape, asserted before the candidate is created and again on the + # candidate itself, so a deploy that would change the gateway's Cloud SQL draw fails. + PUSH_MIN_INSTANCES: 1 + PUSH_MAX_INSTANCES: 2 CONFIRMATION: ${{ inputs.confirmation }} steps: - uses: actions/checkout@v4 @@ -60,18 +63,15 @@ jobs: - uses: google-github-actions/setup-gcloud@v2 - # Held across the deploy, not just a separate schema step: the gateway opens its pool and - # applies its schema while the new revision starts, so the revision is the schema step. - - uses: ./.github/actions/cloud-sql-rollout-lease - with: - bucket: onorca-cloud-terraform-state - object: terraform/state/cloud-sql-rollout/production.lock - - uses: docker/setup-buildx-action@v3 - name: Configure Docker auth run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet + # Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance, + # and a multi-minute image build inside the lease blocks every relay deploy and rehome for + # its duration. The lease below covers exactly the connection-budget window: deploy, probe, + # shift. - name: Build and publish the immutable gateway image shell: bash run: | @@ -86,7 +86,18 @@ jobs: >> "${GITHUB_ENV}" echo "IMAGE_DIGEST=${digest}" >> "${GITHUB_ENV}" - - name: Record the serving revision before the rollout + # Held across the deploy, not just a separate schema step: the gateway opens its pool and + # applies its schema while the new revision starts, so the revision is the schema step. + - uses: ./.github/actions/cloud-sql-rollout-lease + with: + bucket: onorca-cloud-terraform-state + object: terraform/state/cloud-sql-rollout/production.lock + + # Why: the candidate inherits the serving revision's scaling. A serving revision that has + # drifted below the floor would hand the candidate a cold start on every notification, and + # one that has drifted above the ceiling would hand it a larger Cloud SQL draw than the + # rollout lease was taken for. Refuse to inherit either rather than latch it. + - name: Record the serving revision and require its Terraform-owned scaling shell: bash run: | set -euo pipefail @@ -95,6 +106,21 @@ jobs: | jq -r '[.status.traffic[] | select((.percent // 0) > 0)] | if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')" test -n "${serving}" + floor="$(gcloud run revisions describe "${serving}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --format="value(metadata.annotations['autoscaling.knative.dev/minScale'])")" + if [[ "${floor:-0}" -lt "${PUSH_MIN_INSTANCES}" ]]; then + echo "serving revision ${serving} holds ${floor:-0} minimum instances," \ + "below ${PUSH_MIN_INSTANCES}; deploying would inherit and latch it." >&2 + echo "Restore the floor first: gcloud run services update ${SERVICE_NAME}" \ + "--region ${GCP_REGION} --min-instances=${PUSH_MIN_INSTANCES}" >&2 + exit 1 + fi + ceiling="$(gcloud run revisions describe "${serving}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --format="value(metadata.annotations['autoscaling.knative.dev/maxScale'])")" + test "${ceiling}" = "${PUSH_MAX_INSTANCES}" + echo "serving revision ${serving} holds ${floor} minimum and ${ceiling} maximum instances" echo "ROLLBACK_REVISION=${serving}" >> "${GITHUB_ENV}" # No traffic and a per-revision tag: the candidate boots, applies schema, and is probed on @@ -110,7 +136,6 @@ jobs: --image "${IMAGE}" \ --tag "${tag}" \ --no-traffic \ - --max-instances "${PUSH_MAX_INSTANCES}" \ --quiet candidate="$(gcloud run services describe "${SERVICE_NAME}" \ --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ @@ -121,7 +146,11 @@ jobs: echo "CANDIDATE_REVISION=$(jq -r '.revisionName' <<< "${candidate}")" >> "${GITHUB_ENV}" echo "CANDIDATE_URL=$(jq -r '.url' <<< "${candidate}")" >> "${GITHUB_ENV}" - - name: Require the candidate to serve the exact image + # A tagged revision is directly addressable and sits outside the service-wide cap, so the + # candidate and the serving revision each draw up to the ceiling during the probe window. + # The lease is taken for exactly that doubling; a candidate that inherited a wider ceiling + # would exceed it, so the inherited scaling is asserted here too. + - name: Require the candidate to serve the exact image and inherited scaling shell: bash run: | set -euo pipefail @@ -130,6 +159,10 @@ jobs: --format='value(spec.containers[0].image)')" test "${served}" = "${IMAGE}" test "${CANDIDATE_REVISION}" != "${ROLLBACK_REVISION}" + candidate_ceiling="$(gcloud run revisions describe "${CANDIDATE_REVISION}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --format="value(metadata.annotations['autoscaling.knative.dev/maxScale'])")" + test "${candidate_ceiling}" = "${PUSH_MAX_INSTANCES}" - name: Probe the candidate readiness endpoint shell: bash @@ -154,6 +187,10 @@ jobs: # runtime account's FCM grant end to end without delivering anything: validate_only stops # Google before any push, and the deliberately invalid token means a healthy credential # answers INVALID_ARGUMENT. PERMISSION_DENIED is the failure this step exists to catch. + # + # Only the four verdicts below are conclusive. A 429, a 5xx, or a transport failure says + # nothing about the credential, so it is retried rather than treated as either answer; a + # denied credential still fails on the first attempt, without burning the retries. - name: Prove the runtime identity can reach FCM shell: bash run: | @@ -162,13 +199,20 @@ jobs: --impersonate-service-account "${PUSH_RUNTIME_SERVICE_ACCOUNT}")" test -n "${token}" body='{"validate_only":true,"message":{"token":"orca-push-deploy-probe-invalid-token","notification":{"title":"Orca","body":"deploy probe"}}}' - code="$(curl -sS -o "${RUNNER_TEMP}/push-fcm.json" -w '%{http_code}' --max-time 20 \ - -X POST "https://fcm.googleapis.com/v1/projects/${GCP_PROJECT_ID}/messages:send" \ - -H "Authorization: Bearer ${token}" \ - -H 'Content-Type: application/json' \ - --data "${body}" || true)" - status="$(jq -r '.error.status // empty' < "${RUNNER_TEMP}/push-fcm.json")" - echo "FCM validate-only send returned HTTP ${code} status ${status:-OK}" + for attempt in $(seq 1 5); do + code="$(curl -sS -o "${RUNNER_TEMP}/push-fcm.json" -w '%{http_code}' --max-time 20 \ + -X POST "https://fcm.googleapis.com/v1/projects/${GCP_PROJECT_ID}/messages:send" \ + -H "Authorization: Bearer ${token}" \ + -H 'Content-Type: application/json' \ + --data "${body}" || true)" + status="$(jq -r '.error.status // empty' < "${RUNNER_TEMP}/push-fcm.json" || true)" + echo "attempt ${attempt}: FCM validate-only send returned HTTP ${code} status ${status:-OK}" + if test "${status}" = PERMISSION_DENIED || test "${status}" = INVALID_ARGUMENT || + test "${code}" = 401 || test "${code}" = 403; then + break + fi + sleep 5 + done if test "${status}" = PERMISSION_DENIED || test "${code}" = 401 || test "${code}" = 403; then echo "the push runtime identity cannot send through FCM" >&2 exit 1 @@ -189,13 +233,15 @@ jobs: | jq -r '[.status.traffic[] | select((.percent // 0) > 0)] | if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')" test "${serving}" = "${CANDIDATE_REVISION}" + echo "TRAFFIC_SHIFTED=true" >> "${GITHUB_ENV}" - - name: Verify the public origin after the shift + # Why: the summary is written before the origin check, not after it. Once traffic has + # moved, the rollback target is the single thing an operator needs, and a summary that only + # appeared on success would be missing in exactly the run that needs it. + - name: Publish the rollout summary shell: bash run: | set -euo pipefail - code="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 10 "${PUSH_ORIGIN}/ready")" - test "${code}" = 200 { echo '### Push gateway deployed' echo @@ -207,6 +253,74 @@ jobs: "--region ${GCP_REGION} --to-revisions ${ROLLBACK_REVISION}=100\`" } >> "${GITHUB_STEP_SUMMARY}" + - name: Verify the public origin after the shift + shell: bash + run: | + set -euo pipefail + for attempt in $(seq 1 30); do + code="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 10 \ + "${PUSH_ORIGIN}/ready" || true)" + if test "${code}" = 200; then + echo "${PUSH_ORIGIN} is ready after ${attempt} attempt(s)" + exit 0 + fi + echo "attempt ${attempt}: ${PUSH_ORIGIN}/ready returned ${code}" + sleep 5 + done + echo "${PUSH_ORIGIN} never reported ready after the shift" >&2 + exit 1 + + # Why: everything after the shift runs with production on the candidate. A failure there + # is not a failure to deploy, it is a live gateway that has to go back, so the traffic move + # is undone here rather than left to whoever reads the run. + - name: Roll traffic back to the previous revision + if: ${{ failure() && env.TRAFFIC_SHIFTED == 'true' }} + shell: bash + run: | + set -euo pipefail + test -n "${ROLLBACK_REVISION:-}" + gcloud run services update-traffic "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" \ + --region "${GCP_REGION}" \ + --to-revisions "${ROLLBACK_REVISION}=100" \ + --quiet + serving="$(gcloud run services describe "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ + | jq -r '[.status.traffic[] | select((.percent // 0) > 0)] + | if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')" + test "${serving}" = "${ROLLBACK_REVISION}" + { + echo + echo '### Push gateway rolled back' + echo + echo "Traffic returned to \`${ROLLBACK_REVISION}\`; the candidate" \ + "\`${CANDIDATE_REVISION}\` no longer serves." + } >> "${GITHUB_STEP_SUMMARY}" + + # Why: a candidate that never took traffic is a revision holding a warm floor and a Cloud + # SQL pool for nothing. Its tag comes off first, because Cloud Run refuses to delete a + # revision a traffic target still names, and clearing CANDIDATE_TAG makes the always() tag + # step below a no-op rather than a second failure. + - name: Delete the candidate revision that never took traffic + if: ${{ failure() && env.TRAFFIC_SHIFTED != 'true' }} + shell: bash + run: | + set -euo pipefail + test -n "${CANDIDATE_REVISION:-}" || exit 0 + if test -n "${CANDIDATE_TAG:-}"; then + gcloud run services update-traffic "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" \ + --region "${GCP_REGION}" \ + --remove-tags "${CANDIDATE_TAG}" \ + --quiet + echo "CANDIDATE_TAG=" >> "${GITHUB_ENV}" + fi + gcloud run revisions delete "${CANDIDATE_REVISION}" \ + --project "${GCP_PROJECT_ID}" \ + --region "${GCP_REGION}" \ + --quiet + echo "deleted the candidate revision ${CANDIDATE_REVISION}" + - name: Drop the candidate traffic tag if: always() shell: bash diff --git a/cloud/dev/scripts/push-gateway-workflow.test.mjs b/cloud/dev/scripts/push-gateway-workflow.test.mjs index 4525c5d581e..d7aacedf3fe 100644 --- a/cloud/dev/scripts/push-gateway-workflow.test.mjs +++ b/cloud/dev/scripts/push-gateway-workflow.test.mjs @@ -1,7 +1,13 @@ import assert from 'node:assert/strict' import { readFileSync } from 'node:fs' import test from 'node:test' -import { concurrencyBlocks, jobIf, jobs, leaseSteps } from './cloud-sql-rollout-lock-census.mjs' +import { + concurrencyBlocks, + jobIf, + jobs, + LEASE_ACTION, + leaseSteps +} from './cloud-sql-rollout-lock-census.mjs' import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs' // Why: the push gateway holds the APNs key and is the only thing standing between a paired @@ -81,18 +87,66 @@ test('the candidate revision takes no traffic and is addressed by its own tag', assert.match(workflow, /^ {12}--no-traffic \\$/m) assert.match(workflow, /--tag "\$\{tag\}"/) assert.match(workflow, /test "\$\{CANDIDATE_REVISION\}" != "\$\{ROLLBACK_REVISION\}"/) - // A tagged revision is directly addressable and sits outside the service-wide cap, so the - // candidate needs its own ceiling or it doubles the gateway's Cloud SQL draw while probing. - assert.match(workflow, /--max-instances "\$\{PUSH_MAX_INSTANCES\}"/) - assert.match(workflow, /PUSH_MAX_INSTANCES: 4$/m) - assert.match(terraform('variables.tf'), /variable "push_max_instances"[\s\S]*?default {5}= 4/) assert.ok( - indexOfStep('Record the serving revision before the rollout') < + indexOfStep('Record the serving revision and require its Terraform-owned scaling') < indexOfStep('Deploy the candidate revision with no traffic'), 'the rollback target must be captured before the candidate exists' ) }) +// Why: scaling is a Terraform-owned field that `lifecycle.ignore_changes` does not cover, so a +// deploy that passed --max-instances would revert a later push_max_instances raise on every run. +// The workflow asserts the shape instead of writing it, on the serving revision before the +// candidate exists and on the candidate that inherits it. +test('the deploy asserts the Terraform-owned scaling instead of mutating it', () => { + assert.doesNotMatch(workflow, /--max-instances/, 'the deploy must not write a scaling field') + assert.doesNotMatch(workflow, /--min-instances "/, 'the deploy must not write a scaling field') + // The floor is the variables.tf default; production.tfvars overrides only the ceiling, down to + // the two instances the Cloud SQL connection budget leaves room for. + assert.match(workflow, /PUSH_MIN_INSTANCES: 1$/m) + assert.match(workflow, /PUSH_MAX_INSTANCES: 2$/m) + assert.match(terraform('variables.tf'), /variable "push_min_instances"[\s\S]*?default {5}= 1/) + assert.match(terraform('environments/production.tfvars'), /^push_max_instances {9}= 2$/m) + const gate = indexOfStep('Record the serving revision and require its Terraform-owned scaling') + assert.ok(gate < indexOfStep('Deploy the candidate revision with no traffic')) + assert.match(workflow, /autoscaling\.knative\.dev\/minScale/) + assert.match(workflow, /\[\[ "\$\{floor:-0\}" -lt "\$\{PUSH_MIN_INSTANCES\}" \]\]/) + assert.match(workflow, /test "\$\{ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/) + assert.match(workflow, /test "\$\{candidate_ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/) +}) + +// Why: the image build is not a Cloud SQL operation, and the lease is a global serialization +// point. A build inside it blocks every relay deploy and rehome for its duration. +test('the image is built before the rollout lease is taken', () => { + const lease = workflow.indexOf(`- uses: ${LEASE_ACTION}`) + assert.notEqual(lease, -1) + const build = workflow.indexOf('- name: Build and publish the immutable gateway image') + const deployCandidate = workflow.indexOf('- name: Deploy the candidate revision with no traffic') + assert.ok(build < lease, 'the build must finish before the run takes the lease') + assert.ok(lease < deployCandidate, 'the lease must still cover the deploy, probe, and shift') +}) + +// Why: the gateway's Cloud SQL draw is instances x pool, and the root that takes the rollout +// lease can only account for a pool it declares. Leaving it at the application default hid it. +test('the database pool size is Terraform-owned and bounded at plan time', () => { + const source = terraform('push-gateway.tf') + assert.match(source, /name {2}= "ORCA_PUSH_DATABASE_POOL_MAX"/) + assert.match(source, /value = tostring\(var\.push_database_pool_max\)/) + assert.match(terraform('variables.tf'), /variable "push_database_pool_max"[\s\S]*?default {5}= 2/) + const block = /resource "google_cloud_run_v2_service" "push"[\s\S]*?\n lifecycle \{([\s\S]*?)\n \}/.exec(source) + assert.ok(block, 'the push service no longer declares a lifecycle block') + assert.match( + block[1], + /var\.push_max_instances \* var\.push_database_pool_max <= 4/, + 'instances x pool must be bounded at plan time' + ) + assert.match( + readFileSync(new URL('../../apps/push/src/config.ts', import.meta.url), 'utf8'), + /ORCA_PUSH_DATABASE_POOL_MAX/, + 'the gateway must read the variable Terraform sets' + ) +}) + test('the candidate is probed on its own URL before any traffic moves', () => { const probe = indexOfStep('Probe the candidate readiness endpoint') assert.ok(probe > indexOfStep('Deploy the candidate revision with no traffic')) @@ -115,6 +169,15 @@ test('the FCM probe is validate-only and separates a bad token from a bad creden assert.match(workflow, /orca-push-deploy-probe-invalid-token/) assert.match(workflow, /test "\$\{status\}" = INVALID_ARGUMENT/) assert.match(workflow, /test "\$\{status\}" = PERMISSION_DENIED/) + // Only those four answers are conclusive; a 429 or a 5xx says nothing about the credential, so + // it is retried rather than read as either verdict. A denied credential still fails at once. + assert.match(workflow, /for attempt in \$\(seq 1 5\); do/) + const probe = workflow.slice( + workflow.indexOf('- name: Prove the runtime identity can reach FCM'), + workflow.indexOf('- name: Shift all traffic to the verified candidate') + ) + assert.match(probe, /for attempt in \$\(seq 1 5\); do/) + assert.match(probe, /test "\$\{code\}" = 401 \|\| test "\$\{code\}" = 403; then\n {14}break/) assert.match( workflow, /--impersonate-service-account "\$\{PUSH_RUNTIME_SERVICE_ACCOUNT\}"/, @@ -150,11 +213,80 @@ test('the traffic shift is all-or-nothing and is verified after the fact', () => assert.match(workflow, /"\$\{PUSH_ORIGIN\}\/ready"/) }) -test('the run reports a rollback target and always drops its traffic tag', () => { +// Why: the origin can lag the traffic move by seconds, and a single unlucky curl would otherwise +// roll a healthy deploy back. It retries on the same schedule as the candidate probe. +test('the post-shift origin check retries like the candidate probe', () => { + const check = workflow.slice( + workflow.indexOf('- name: Verify the public origin after the shift'), + workflow.indexOf('- name: Roll traffic back to the previous revision') + ) + assert.match(check, /for attempt in \$\(seq 1 30\); do/) + assert.match(check, /sleep 5/) + assert.match(check, /test "\$\{code\}" = 200/) +}) + +// Why: the summary carries the rollback target. Writing it after the origin check meant the one +// run that needed it, the run whose check failed, was the one run that never got it. +test('the summary is written before anything that can fail after the shift', () => { + const summary = indexOfStep('Publish the rollout summary') + assert.ok(summary > indexOfStep('Shift all traffic to the verified candidate')) + assert.ok(summary < indexOfStep('Verify the public origin after the shift')) assert.match(workflow, /--to-revisions \$\{ROLLBACK_REVISION\}=100/) + assert.match(workflow, /GITHUB_STEP_SUMMARY/) +}) + +// Why: everything after the shift runs with production on the candidate, so a failure there is a +// live gateway that has to go back. The marker is what separates that case from a failure before +// the shift, where production never moved and the candidate is the thing to clean up. +test('a failure after the shift rolls production back automatically', () => { + const rollback = indexOfStep('Roll traffic back to the previous revision') + assert.ok(rollback > indexOfStep('Verify the public origin after the shift')) + assert.match(workflow, /echo "TRAFFIC_SHIFTED=true" >> "\$\{GITHUB_ENV\}"/) + const shift = workflow.indexOf('- name: Shift all traffic to the verified candidate') + assert.ok( + workflow.indexOf('echo "TRAFFIC_SHIFTED=true"') > shift, + 'the marker must be set only once the shift has been verified' + ) + const body = workflow.slice( + workflow.indexOf('- name: Roll traffic back to the previous revision'), + workflow.indexOf('- name: Delete the candidate revision that never took traffic') + ) + assert.match( + body, + /if: \$\{\{ failure\(\) && env\.TRAFFIC_SHIFTED == 'true' \}\}/, + 'the rollback must be conditioned on both failure and the shift marker' + ) + assert.match(body, /test -n "\$\{ROLLBACK_REVISION:-\}"/) + assert.match(body, /--to-revisions "\$\{ROLLBACK_REVISION\}=100"/) + assert.match(body, /test "\$\{serving\}" = "\$\{ROLLBACK_REVISION\}"/) + assert.match(body, /GITHUB_STEP_SUMMARY/, 'the rollback must be reported in the summary') +}) + +// Why: a candidate that never took traffic still holds a warm instance and a Cloud SQL pool. Its +// tag comes off first, because Cloud Run refuses to delete a revision a traffic target names. +test('a failure before the shift deletes the candidate it created', () => { + const body = workflow.slice( + workflow.indexOf('- name: Delete the candidate revision that never took traffic'), + workflow.indexOf('- name: Drop the candidate traffic tag') + ) + assert.match( + body, + /if: \$\{\{ failure\(\) && env\.TRAFFIC_SHIFTED != 'true' \}\}/, + 'the cleanup must be conditioned on both failure and the absence of the shift marker' + ) + assert.match(body, /test -n "\$\{CANDIDATE_REVISION:-\}" \|\| exit 0/) + assert.ok( + body.indexOf('--remove-tags') < body.indexOf('gcloud run revisions delete'), + 'the tag must come off before the revision is deleted' + ) + assert.match(body, /echo "CANDIDATE_TAG=" >> "\$\{GITHUB_ENV\}"/) +}) + +test('the run always drops its traffic tag', () => { const cleanup = indexOfStep('Drop the candidate traffic tag') assert.equal(cleanup, stepNames().length - 1, 'tag cleanup must be the last step') assert.match(workflow, /--remove-tags "\$\{CANDIDATE_TAG\}"/) const body = workflow.slice(workflow.indexOf('- name: Drop the candidate traffic tag')) assert.match(body, /if: always\(\)/) + assert.match(body, /test -n "\$\{CANDIDATE_TAG:-\}" \|\| exit 0/) }) diff --git a/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs b/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs index 79036918f23..6965986845c 100644 --- a/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs +++ b/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs @@ -33,6 +33,13 @@ function requiredInteger(source, pattern, label) { return value } +// A tfvars file states only what it overrides, so an absent key means the variable default holds. +// Reading the default as the fallback keeps this honest either way. +function overriddenInteger(override, overridePattern, source, pattern, label) { + if (!overridePattern.test(override)) return requiredInteger(source, pattern, label) + return requiredInteger(override, overridePattern, label) +} + function productionCells(source, defaultPoolMax) { const fencedMatch = source.match(/relay_gce_fenced_cells\s*=\s*\[([^\]]*)\]/) if (!fencedMatch) throw new Error('could not read fenced Relay cells') @@ -52,11 +59,13 @@ function productionCells(source, defaultPoolMax) { } export function calculateRelayCloudSqlConnectionBudget(inputs) { + const pushDraw = inputs.pushInstances * inputs.pushPoolMax const consumers = { cells: inputs.cellPoolTotal + inputs.asiaCellCount * inputs.asiaPoolMax, directors: inputs.directorInstances * inputs.directorPoolMax, auth: inputs.authInstances * inputs.authPoolMax, - api: inputs.apiInstances * inputs.apiPoolMax + api: inputs.apiInstances * inputs.apiPoolMax, + push: pushDraw } const configuredMaximum = Object.values(consumers).reduce((total, value) => total + value, 0) const retainedDirectorRollback = inputs.directorInstances * inputs.directorPoolMax @@ -64,6 +73,11 @@ export function calculateRelayCloudSqlConnectionBudget(inputs) { relayDirectorCandidate: retainedDirectorRollback * 2, apiCandidate: retainedDirectorRollback + inputs.apiInstances * inputs.apiPoolMax, authCandidate: retainedDirectorRollback + inputs.authInstances * inputs.authPoolMax, + // The push candidate doubles rather than adding one copy, like the director candidate and + // unlike the API and auth ones: cloud-push-deploy.yml probes a *tagged* revision, which is + // directly addressable and so sits outside the service-wide instance cap, letting the + // candidate and the serving revision each reach push_max_instances at the same time. + pushCandidate: retainedDirectorRollback + pushDraw * 2, relayCells: retainedDirectorRollback } const rolloutOverlap = Math.max(...Object.values(candidateOverlap)) @@ -131,6 +145,20 @@ export function readRelayCloudSqlConnectionBudget({ /variable\s+"relay_director_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/, 'director pool maximum' ), + // The mobile push gateway shares this instance. Its draw was invisible here until Terraform + // declared the pool: docs/push-gateway.md, "Shape". + pushInstances: overriddenInteger( + productionTfvars, + /^\s*push_max_instances\s*=\s*(\d+)/m, + terraformVariables, + /variable\s+"push_max_instances"[\s\S]*?default\s*=\s*(\d+)/, + 'push gateway instances' + ), + pushPoolMax: requiredInteger( + terraformVariables, + /variable\s+"push_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/, + 'push gateway pool maximum' + ), authInstances: apps.authInstances, authPoolMax: apps.authPoolMax, apiInstances: apps.apiInstances, diff --git a/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs index a26d24c274d..4e3536c0e2b 100644 --- a/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs +++ b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs @@ -6,28 +6,91 @@ import { readRelayCloudSqlConnectionBudget } from './relay-cloud-sql-connection-budget.mjs' -test('production plus three Asia pools preserves allowance and reserve below the ceiling', () => { +// Why these numbers are this tight: the shared instance's 400 connections were already spoken +// for, and the relay shape below leaves exactly five. The gateway is sized to fit in four, two +// instances times a two-connection pool, and its rollout overlap of 23 stays under the API +// candidate's 65, so the Math.max is the API candidate rather than the gateway. +// +// `Deploy Relay Asia Topology` gates on `withinBudget == true`, so the single remaining +// connection is the whole margin. Anything that raises a pool or an instance count moves it. +test('production plus the push gateway keeps allowance and reserve below the ceiling', () => { const report = readRelayCloudSqlConnectionBudget() - assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 }) + assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50, push: 4 }) assert.deepEqual(report.asia, { cells: 3, poolMax: 10 }) - assert.equal(report.configuredMaximum, 315) + assert.equal(report.configuredMaximum, 319) assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30) assert.equal(report.rolloutOverlap.apiCandidate, 65) assert.equal(report.rolloutOverlap.authCandidate, 35) + assert.equal(report.rolloutOverlap.pushCandidate, 23) assert.equal(report.rolloutOverlap.relayCells, 15) assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15) + // The gateway does not set the maximum; the API candidate does, as it did before it existed. assert.equal(report.rolloutOverlap.maximum, 65) assert.equal(report.maintenanceAdminAllowance, 5) assert.equal(report.explicitReserve, 10) assert.equal(report.usableCeiling, 390) + assert.equal(report.operatingMaximum, 389) + assert.equal(report.remainingWithinUsableCeiling, 1) + assert.equal(report.budgetedTotal, 399) + assert.equal(report.unallocated, 1) + assert.equal(report.withinBudget, true) +}) + +// Why: the same relay shape without a push gateway is the before picture, and it stood at five +// connections clear. Holding it here keeps the gateway's cost visible as the four it takes, +// rather than letting drift elsewhere in the budget hide inside the same margin. +test('the same relay shape without the gateway stays inside the ceiling', () => { + const report = calculateRelayCloudSqlConnectionBudget({ + cellPoolTotal: 200, + asiaCellCount: 3, + asiaPoolMax: 10, + directorInstances: 5, + directorPoolMax: 3, + authInstances: 2, + authPoolMax: 10, + apiInstances: 10, + apiPoolMax: 5, + pushInstances: 0, + pushPoolMax: 0, + maxConnections: 400, + maintenanceAdminAllowance: 5, + explicitReserve: 10 + }) + + assert.equal(report.consumers.push, 0) + assert.equal(report.rolloutOverlap.maximum, 65) assert.equal(report.operatingMaximum, 385) assert.equal(report.remainingWithinUsableCeiling, 5) - assert.equal(report.budgetedTotal, 395) - assert.equal(report.unallocated, 5) assert.equal(report.withinBudget, true) }) +// Why: a tagged candidate is directly addressable and sits outside the service-wide cap, so both +// push revisions can reach the ceiling at once. The API and auth candidates add one copy; this +// one adds two, like the director candidate. +test('the push rollout scenario doubles the gateway draw over the retained director', () => { + const report = calculateRelayCloudSqlConnectionBudget({ + cellPoolTotal: 0, + asiaCellCount: 0, + asiaPoolMax: 0, + directorInstances: 5, + directorPoolMax: 3, + authInstances: 0, + authPoolMax: 0, + apiInstances: 0, + apiPoolMax: 0, + pushInstances: 2, + pushPoolMax: 2, + maxConnections: 400, + maintenanceAdminAllowance: 5, + explicitReserve: 10 + }) + + assert.equal(report.consumers.push, 4) + // 15 retained director rollback, plus the 4-connection draw counted twice. + assert.equal(report.rolloutOverlap.pushCandidate, 23) +}) + test('fails closed when pool growth consumes the explicit reserve', () => { const report = calculateRelayCloudSqlConnectionBudget({ cellPoolTotal: 200, @@ -39,12 +102,14 @@ test('fails closed when pool growth consumes the explicit reserve', () => { authPoolMax: 10, apiInstances: 20, apiPoolMax: 5, + pushInstances: 4, + pushPoolMax: 10, maxConnections: 400, maintenanceAdminAllowance: 5, explicitReserve: 10 }) - assert.equal(report.operatingMaximum, 515) + assert.equal(report.operatingMaximum, 555) assert.equal(report.withinBudget, false) }) @@ -63,7 +128,11 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => { } } `, - terraformVariables: 'variable "relay_director_database_pool_max" { default = 3 }', + terraformVariables: [ + 'variable "relay_director_database_pool_max" { default = 3 }', + 'variable "push_max_instances" { default = 1 }', + 'variable "push_database_pool_max" { default = 2 }' + ].join('\n'), relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10' }, maxConnections: 100, @@ -72,8 +141,42 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => { }) assert.equal(report.consumers.cells, 14) - assert.equal(report.operatingMaximum, 46) - assert.equal(report.budgetedTotal, 47) + // No push_max_instances in this tfvars, so the variable default of one instance holds. + assert.equal(report.consumers.push, 2) + assert.equal(report.operatingMaximum, 48) + assert.equal(report.budgetedTotal, 49) +}) + +// Why: production.tfvars overrides push_max_instances down to 2 while variables.tf still defaults +// to 4, so reading the default instead of the override would overstate the live draw by half. +test('a tfvars push_max_instances override wins over the variable default', () => { + const report = readRelayCloudSqlConnectionBudget({ + proposedAsiaCellCount: 1, + appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 }, + sources: { + productionTfvars: ` + relay_max_instances = 1 + push_max_instances = 3 + relay_gce_fenced_cells = [] + relay_gce_cells = { + "production-gce-c2" = { database_pool_max = 4 + } + } + `, + terraformVariables: [ + 'variable "relay_director_database_pool_max" { default = 3 }', + 'variable "push_max_instances" { default = 1 }', + 'variable "push_database_pool_max" { default = 2 }' + ].join('\n'), + relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10' + }, + maxConnections: 100, + maintenanceAdminAllowance: 1, + explicitReserve: 1 + }) + + assert.equal(report.consumers.push, 6) + assert.equal(report.rolloutOverlap.pushCandidate, 15) }) test('requires strict headroom below the physical ceiling', () => { @@ -87,12 +190,14 @@ test('requires strict headroom below the physical ceiling', () => { authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, + pushInstances: 1, + pushPoolMax: 2, maxConnections: 50, maintenanceAdminAllowance: 9, explicitReserve: 3 }) - assert.equal(report.budgetedTotal, 63) + assert.equal(report.budgetedTotal, 65) assert.equal(report.withinBudget, false) }) diff --git a/cloud/docs/push-gateway.md b/cloud/docs/push-gateway.md index cc003cd0d52..0d70d8fb757 100644 --- a/cloud/docs/push-gateway.md +++ b/cloud/docs/push-gateway.md @@ -21,7 +21,8 @@ edit plus a second set of Apple credentials. | --- | --- | --- | | Cloud Run service | `orca-cloud-push` | `push_cloud_run_service_name` | | Region | `us-central1` | `region` | -| Instances | min 1, max 4 | `push_min_instances`, `push_max_instances` | +| Instances | min 1, max 2 | `push_min_instances`, `push_max_instances` | +| Database pool | 2 per instance | `push_database_pool_max` | | Concurrency | 80 | `push_concurrency` | | Ingress | all | `INGRESS_TRAFFIC_ALL` | | Invoker | IAM disabled | `invoker_iam_disabled = true` on the service | @@ -29,8 +30,24 @@ edit plus a second set of Apple credentials. | Database | `orca_push` on the shared Cloud SQL instance | `google_sql_database.push` | | Hostname | `push.onorca.dev` | `push_base_url` | -The minimum of one instance is deliberate. A cold start delays a notification past the point -where it is worth showing, and the three-second coalescing window lives in instance memory. +The minimum of one instance is deliberate and did not move when the ceiling came down to two. A +cold start delays a notification past the point where it is worth showing, and the three-second +coalescing window lives in instance memory, so the floor is what keeps a notification prompt. The +ceiling is a different question, answered below. + +The maximum and the pool are set by the connection budget, not by the gateway's own appetite. Two +instances times a two-connection pool is a draw of 4, and a rollout doubles it to 8, because the +tagged candidate is directly addressable and sits outside the service-wide cap. The shared Cloud +SQL instance's 400 connections were already spoken for by the relay cells, the directors, auth, +and the API, which left five. Four is the whole of the room there was, and the gateway fits in +it. + +Two connections per instance is enough for the work. A send runs two or three short queries, so +at concurrency 80 requests queue against the pool for microseconds rather than holding it. A +`lifecycle` precondition refuses a plan whose instances times pool exceeds 4, because a fifth +connection puts the checked budget over its ceiling and blocks `Deploy Relay Asia Topology`, +which gates on it. `dev/scripts/relay-cloud-sql-connection-budget.mjs` counts the gateway and +prints the whole picture. Authentication is the host proof in `POST /v1/host/challenge`, not Cloud Run IAM, so the service opts out of invoker IAM with `invoker_iam_disabled = true`, exactly as the relay director does. @@ -47,6 +64,7 @@ Set on the container by Terraform: | `ORCA_PUSH_PUBLIC_URL` | `push_base_url` | | `ORCA_PUSH_FCM_PROJECT_ID` | `push_fcm_project_id`, empty means `project_id` | | `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-database-url`, version `latest` | +| `ORCA_PUSH_DATABASE_POOL_MAX` | `push_database_pool_max`, 2 per instance | | `ORCA_PUSH_APNS_KEY` | Secret `orca-cloud-push-apns-key`, version `latest` | | `ORCA_PUSH_APNS_KEY_ID` | Secret `orca-cloud-push-apns-key-id`, version `latest` | | `ORCA_PUSH_APPLE_TEAM_ID` | Secret `orca-cloud-push-apple-team-id`, version `latest` | @@ -130,28 +148,55 @@ It authenticates as the shared production deploy identity through `PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER` and `PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT`, which are already published. No new GitHub variable is required. That account was chosen because the Cloud SQL rollout lease grant is -foundation-owned and already names it; a dedicated identity could not take that lease from this -root. Its authority over the gateway is exactly three bindings in `push-gateway.tf`: Cloud Run -developer on this one service, service-account user on the runtime account, and token creator on -the runtime account. +foundation-owned and names only that account; a dedicated identity could not take that lease from +this root, and the gateway's schema rollout has to serialize against the relay's. + +**That choice widens what this workflow can reach, and the widening is deliberate.** Adding +`push-deploy.yml` to the provider allowlist gives the run the account's whole existing authority, +not only the push bindings: Artifact Registry writer on `orca-cloud`, `roles/run.developer` on +the relay director and the fence broker, accessor and version-adder on the relay +regional-placement secret, and service-account user on the relay runtime identities. It was +accepted as the price of the lease. What `push-gateway.tf` adds on top is three bindings scoped +to the gateway alone: Cloud Run developer on this one service, and service-account user plus +token creator on the runtime account. The bound on the rest is the provider condition, which +admits this exact workflow file on `main` in the `production` environment only, and the workflow +itself, which is dispatch-only behind a typed confirmation. The run, in order: -1. Takes the production Cloud SQL rollout lease and holds it for the whole run. The gateway - applies its schema while the new revision starts, so the revision **is** the schema step; - there is no separate migration command to wrap. -2. Builds `apps/push/Dockerfile` with the `cloud/` build context and pushes to the existing +1. Builds `apps/push/Dockerfile` with the `cloud/` build context and pushes to the existing `orca-cloud` Artifact Registry repository as `push:sha-`, then resolves the digest. -3. Records the currently serving revision as the rollback target. + This happens **before** the lease is taken. Artifact Registry is not the Cloud SQL instance, + and a multi-minute build inside the lease would block every relay deploy and rehome for its + duration. +2. Takes the production Cloud SQL rollout lease and holds it from here to the end. The gateway + applies its schema while the new revision starts, so the revision **is** the schema step; + there is no separate migration command to wrap. The lease therefore covers exactly the + connection-budget window: deploy, probe, shift. +3. Records the currently serving revision as the rollback target, and requires it to still hold + the Terraform-owned floor and ceiling. The candidate inherits that scaling, so a drifted + serving revision would be latched rather than corrected. 4. `gcloud run deploy --no-traffic` with a per-run traffic tag, so the candidate boots and - applies schema while every phone still reaches the previous revision. + applies schema while every phone still reaches the previous revision. The deploy passes no + scaling flag: the shape is Terraform's, and the candidate's inherited ceiling is asserted + instead. 5. Probes the tagged candidate's own `/ready`, up to 30 times at five-second intervals. 6. Sends a validate-only FCM message as the runtime identity, by impersonation. See below. -7. Shifts 100% of traffic to the candidate, verifies it is the only revision serving, and - checks `https://push.onorca.dev/ready`. -8. Always removes the traffic tag, so tags do not accumulate across runs. +7. Shifts 100% of traffic to the candidate and verifies it is the only revision serving. +8. Writes the run summary, including the rollback command, before checking the public origin, so + the summary exists even when the check that follows does not pass. +9. Checks `https://push.onorca.dev/ready`, up to 30 times at five-second intervals, since the + origin can lag the traffic move by a few seconds. +10. Always removes the traffic tag, so tags do not accumulate across runs. -Rolling back is a traffic move, printed in the run summary: +**Failure after the shift rolls itself back.** Everything from step 8 on runs with production +already on the candidate, so a failure there is not a failed deploy, it is a live gateway that +has to go back. The run returns traffic to the recorded rollback revision, verifies the move, and +reports it in the summary. A failure *before* the shift leaves production untouched and deletes +the candidate revision, which otherwise sits holding a warm instance and a Cloud SQL pool for +nothing. + +To move traffic by hand, from the revision named in the run summary: ```sh gcloud run services update-traffic orca-cloud-push \ @@ -167,7 +212,9 @@ mints an access token for `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com` `validate_only: true` with a token that cannot exist. `validate_only` stops Google before any delivery, and a healthy credential answers `INVALID_ARGUMENT` because the device token is garbage. `PERMISSION_DENIED`, `401`, and `403` are the failures the step exists to catch, and -they fail the run before traffic moves. Probing as the deploy identity instead would prove +they fail the run immediately, before traffic moves. Those four answers are the only conclusive +ones: a `429`, a `5xx`, or a transport failure says nothing about the credential, so the send is +retried up to five times at five-second intervals rather than read as either verdict. Probing as the deploy identity instead would prove something true about the wrong account. ## Rotating the APNs key diff --git a/cloud/docs/relay-workflows.md b/cloud/docs/relay-workflows.md index 01e8229a3be..88c574f3206 100644 --- a/cloud/docs/relay-workflows.md +++ b/cloud/docs/relay-workflows.md @@ -411,20 +411,31 @@ Artifact Registry repository, and its rollout lease. It needs **no new GitHub environment variable.** It authenticates as the shared production deploy identity through the already-published `PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER` and `PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT`, and reads `PRODUCTION_GCP_REGION` like the -rest. That account holds the foundation-owned Cloud SQL rollout lease grant, which a dedicated -identity could not be given from this root. Its authority over the gateway is three bindings in -`infra/terraform/push-gateway.tf` and nothing wider: Cloud Run developer on that one service, and -service-account user plus token creator on the gateway's runtime account. +rest. That account holds the foundation-owned Cloud SQL rollout lease grant, which names it and nothing +else, so a dedicated identity could not be given that lease from this root. + +`infra/terraform/push-gateway.tf` adds three bindings scoped to the gateway: Cloud Run developer +on that one service, and service-account user plus token creator on the gateway's runtime +account. Those three are not the workflow's whole authority. Running as the shared account gives +the run every role that account already holds for the relay: Artifact Registry writer on +`orca-cloud`, `roles/run.developer` on the relay director and the fence broker, accessor and +version-adder on the relay regional-placement secret, and service-account user on the relay +runtime identities. That widening was accepted as the price of the lease, and it is bounded by +the provider condition and by the workflow being dispatch-only behind a typed confirmation. The provider's workflow allowlist gained exactly one entry, `cloud-push-deploy.yml`, on `main` in the `production` environment. That entry is required: the allowlist compares complete workflow refs by equality, so the `cloud-` filename prefix alone does not admit a new file. -The run takes the production rollout lease and holds it across the deploy, because the gateway -applies its schema while the new revision starts. It builds `apps/push/Dockerfile`, deploys with -`--no-traffic` behind a per-run traffic tag, probes the candidate's own `/ready`, proves the -runtime identity can reach FCM with a validate-only send, and only then shifts 100% of traffic. -There is no staging gateway, so there is no staging counterpart to run first. +The run builds `apps/push/Dockerfile` **before** taking the lease, so an image build never blocks +a relay deploy or rehome, then holds the production rollout lease across the deploy itself, +because the gateway applies its schema while the new revision starts. Under the lease it checks +the serving revision's Terraform-owned scaling, deploys with `--no-traffic` behind a per-run +traffic tag and no scaling flag of its own, probes the candidate's own `/ready`, proves the +runtime identity can reach FCM with a validate-only send, and only then shifts 100% of traffic. A +failure after the shift returns traffic to the recorded rollback revision; a failure before it +deletes the candidate. There is no staging gateway, so there is no staging counterpart to run +first. Full runbook, including the APNs key rotation and the DNS record the `stablyai/orca-cloud` apps root still owes, is in `docs/push-gateway.md`. diff --git a/cloud/infra/terraform/environments/production.tfvars b/cloud/infra/terraform/environments/production.tfvars index 6abdc29f91e..79db1904ee6 100644 --- a/cloud/infra/terraform/environments/production.tfvars +++ b/cloud/infra/terraform/environments/production.tfvars @@ -412,6 +412,9 @@ relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels # Mobile push gateway. Production is the only environment that runs one; the runtime account, # the three Apple secrets, and their accessor bindings already exist and are imported once # (see docs/push-gateway.md). -push_gateway_enabled = true -push_base_url = "https://push.onorca.dev" +push_gateway_enabled = true +push_base_url = "https://push.onorca.dev" +# Sized so the gateway's rollout overlap, the retained director rollback plus its doubled draw, +# stays under the API candidate's, which keeps the checked Cloud SQL connection budget green. +push_max_instances = 2 manage_push_domain_mapping = true diff --git a/cloud/infra/terraform/push-gateway.tf b/cloud/infra/terraform/push-gateway.tf index a4d013463fc..9fc3e4bc6fe 100644 --- a/cloud/infra/terraform/push-gateway.tf +++ b/cloud/infra/terraform/push-gateway.tf @@ -40,9 +40,10 @@ locals { push_fqdn = replace(replace(var.push_base_url, "https://", ""), "http://", "") - # The shared production deploy identity runs `cloud-push-deploy.yml`. Its Cloud Run and - # service-account grants are scoped to this service alone; the account itself is declared in - # relay-github-actions.tf and is production-only. + # The shared production deploy identity runs `cloud-push-deploy.yml`. The grants this file adds + # are scoped to this service and its runtime account alone, but the workflow inherits every + # other grant that account already holds for the relay; see the deploy-identity section below. + # The account itself is declared in relay-github-actions.tf and is production-only. push_gateway_deploy_count = ( var.push_gateway_enabled && local.relay_create_production_ops_identity ? 1 : 0 ) @@ -227,6 +228,14 @@ resource "google_cloud_run_v2_service" "push" { value = local.push_fcm_project_id } + # Declared rather than left to the application default, so the gateway's share of the + # shared Cloud SQL connection budget is a value this root states and the precondition + # below can bound. + env { + name = "ORCA_PUSH_DATABASE_POOL_MAX" + value = tostring(var.push_database_pool_max) + } + env { name = "ORCA_PUSH_DATABASE_URL" @@ -284,6 +293,19 @@ resource "google_cloud_run_v2_service" "push" { # 100% LATEST would silently undo either, and this root carries unrelated standing drift, so # that apply need not be a push change at all. lifecycle { + # Why: the gateway draws instances x pool from the shared Cloud SQL instance, and a rollout + # doubles it, because the tagged candidate is directly addressable and sits outside the + # service-wide cap. The instance's 400 connections were already spoken for by the relay + # cells, directors, auth, and API, which left five: 4 is the whole of the gateway's share and + # it fits, with the doubled 8 still under the API candidate's rollout overlap, the term + # dev/scripts/relay-cloud-sql-connection-budget.mjs maximizes over. A fifth connection here + # puts the checked budget over its ceiling and blocks Deploy Relay Asia Topology, which gates + # on it, so catch a raise at plan time rather than in someone else's rollout. + precondition { + condition = var.push_max_instances * var.push_database_pool_max <= 4 + error_message = "Push gateway instances x database pool must stay within its 4-connection share of the shared Cloud SQL instance." + } + ignore_changes = [ client, client_version, @@ -327,8 +349,19 @@ resource "google_cloud_run_domain_mapping" "push" { # --- Deploy identity grants ------------------------------------------------------------------- # `cloud-push-deploy.yml` authenticates as the shared production deploy account, because that -# account is the one the foundation root grants the Cloud SQL rollout lease to. These three -# bindings are the whole of its authority over the push gateway. +# account is the one the foundation root grants the Cloud SQL rollout lease to; the grant names +# that account and nothing else, so a dedicated push identity could not take the lease from this +# root and the gateway's schema rollout could not be serialized against the relay's. +# +# The three bindings below are the whole of that account's authority over the *push gateway*, but +# they are not the whole of what the workflow can do. Adding `push-deploy.yml` to the provider's +# allowlist in relay-github-actions.tf gives the run the account's entire existing authority: +# Artifact Registry writer on `orca-cloud`, `roles/run.developer` on the relay director and the +# fence broker, accessor and version-adder on the relay regional-placement secret, and +# service-account user on the relay runtime identities. That widening was accepted deliberately +# as the price of the lease. It is bounded by the provider condition, which admits this exact +# workflow file on `main` in the `production` environment only, and by the workflow itself, which +# is dispatch-only behind a typed confirmation. resource "google_cloud_run_v2_service_iam_member" "github_production_push_developer" { count = local.push_gateway_deploy_count diff --git a/cloud/infra/terraform/relay-github-actions.tf b/cloud/infra/terraform/relay-github-actions.tf index c6057357a37..a73e8f511e5 100644 --- a/cloud/infra/terraform/relay-github-actions.tf +++ b/cloud/infra/terraform/relay-github-actions.tf @@ -21,8 +21,13 @@ locals { "operate-relay-asia-admission.yml", "publish-relay-production.yml", # The push gateway deploy runs as this account because the Cloud SQL rollout lease grant is - # foundation-owned and already names it; a dedicated identity could not take that lease. - # Its authority over the gateway is three bindings in push-gateway.tf and nothing more. + # foundation-owned and names only this account; a dedicated identity could not take that + # lease, and the gateway's schema rollout has to serialize against the relay's. + # + # This entry therefore grants that workflow every role the account already holds, not just + # the three push bindings in push-gateway.tf: Artifact Registry writer, run.developer on the + # relay director and fence broker, relay secret accessor and version-adder, and + # serviceAccountUser on the relay runtime identities. Accepted as the price of the lease. "push-deploy.yml" ] github_production_relay_capacity_workflow_file = "deploy-relay-production-capacity.yml" diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf index d8d1af1927b..1ef74bbc40f 100644 --- a/cloud/infra/terraform/variables.tf +++ b/cloud/infra/terraform/variables.tf @@ -548,6 +548,24 @@ variable "push_max_instances" { } } +# Why: the gateway's draw on the shared Cloud SQL instance is instances x pool, and the rollout +# lease is taken for twice that, because a tagged candidate is directly addressable and sits +# outside the service-wide cap. Leaving the pool at its application default made that draw +# invisible to this root, so it is declared here and set on the container. +# +# Two is sized to the work, not to the default: a send runs two or three short queries, and at +# concurrency 80 those queue against the pool for microseconds rather than holding it. +variable "push_database_pool_max" { + type = number + description = "Push gateway database pool size per instance; instances x pool is its Cloud SQL draw." + default = 2 + + validation { + condition = var.push_database_pool_max >= 1 && var.push_database_pool_max <= 100 + error_message = "The push gateway pool must hold at least one connection and stay under the per-service bound." + } +} + variable "push_concurrency" { type = number description = "Cloud Run concurrency for short-lived push gateway HTTP requests."