fix(cloud): harden the push deploy workflow and size the gateway to the budget (#8129)

- Roll traffic back on a failed post-shift check; delete a candidate that
  never took traffic; retry the origin probe and the FCM probe.
- Assert Terraform-owned scaling instead of mutating it from the workflow.
- Build before taking the Cloud SQL rollout lease.
- Declare the database pool in Terraform (2 per instance, max 2 instances)
  and add the gateway to the connection budget; the previous default put the
  shared instance 65 connections over its ceiling.
- State plainly that the shared deploy identity's relay authority is inherited.
This commit is contained in:
Jinwoo-H
2026-09-06 15:19:21 -04:00
parent e5ca0336f7
commit 41b9754877
10 changed files with 577 additions and 81 deletions
+140 -26
View File
@@ -37,12 +37,15 @@ jobs:
IMAGE_NAME: push
PUSH_ORIGIN: https://push.onorca.dev
PUSH_RUNTIME_SERVICE_ACCOUNT: orca-cloud-push@onorca-cloud.iam.gserviceaccount.com
# Ceiling the tagged candidate, matching push_max_instances in
# environments/production.tfvars. A tagged revision is directly addressable and so sits
# outside the service-wide cap: without this the candidate and the serving revision could
# each reach the ceiling and double the gateway's Cloud SQL draw during the probe window.
# Terraform still owns the value; this only fails a deploy that would exceed it.
PUSH_MAX_INSTANCES: 4
# Scaling the serving revision must already hold, matching push_min_instances and
# push_max_instances. Terraform owns both, and the candidate inherits them from the
# service, so this deploy never passes a scaling flag: doing so would write a
# Terraform-owned field that `lifecycle.ignore_changes` does not cover, and a later
# `push_max_instances` raise would then be reverted by every deploy. These two values
# are the expected shape, asserted before the candidate is created and again on the
# candidate itself, so a deploy that would change the gateway's Cloud SQL draw fails.
PUSH_MIN_INSTANCES: 1
PUSH_MAX_INSTANCES: 2
CONFIRMATION: ${{ inputs.confirmation }}
steps:
- uses: actions/checkout@v4
@@ -60,18 +63,15 @@ jobs:
- uses: google-github-actions/setup-gcloud@v2
# Held across the deploy, not just a separate schema step: the gateway opens its pool and
# applies its schema while the new revision starts, so the revision is the schema step.
- uses: ./.github/actions/cloud-sql-rollout-lease
with:
bucket: onorca-cloud-terraform-state
object: terraform/state/cloud-sql-rollout/production.lock
- uses: docker/setup-buildx-action@v3
- name: Configure Docker auth
run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet
# Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance,
# and a multi-minute image build inside the lease blocks every relay deploy and rehome for
# its duration. The lease below covers exactly the connection-budget window: deploy, probe,
# shift.
- name: Build and publish the immutable gateway image
shell: bash
run: |
@@ -86,7 +86,18 @@ jobs:
>> "${GITHUB_ENV}"
echo "IMAGE_DIGEST=${digest}" >> "${GITHUB_ENV}"
- name: Record the serving revision before the rollout
# Held across the deploy, not just a separate schema step: the gateway opens its pool and
# applies its schema while the new revision starts, so the revision is the schema step.
- uses: ./.github/actions/cloud-sql-rollout-lease
with:
bucket: onorca-cloud-terraform-state
object: terraform/state/cloud-sql-rollout/production.lock
# Why: the candidate inherits the serving revision's scaling. A serving revision that has
# drifted below the floor would hand the candidate a cold start on every notification, and
# one that has drifted above the ceiling would hand it a larger Cloud SQL draw than the
# rollout lease was taken for. Refuse to inherit either rather than latch it.
- name: Record the serving revision and require its Terraform-owned scaling
shell: bash
run: |
set -euo pipefail
@@ -95,6 +106,21 @@ jobs:
| jq -r '[.status.traffic[] | select((.percent // 0) > 0)]
| if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
test -n "${serving}"
floor="$(gcloud run revisions describe "${serving}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--format="value(metadata.annotations['autoscaling.knative.dev/minScale'])")"
if [[ "${floor:-0}" -lt "${PUSH_MIN_INSTANCES}" ]]; then
echo "serving revision ${serving} holds ${floor:-0} minimum instances," \
"below ${PUSH_MIN_INSTANCES}; deploying would inherit and latch it." >&2
echo "Restore the floor first: gcloud run services update ${SERVICE_NAME}" \
"--region ${GCP_REGION} --min-instances=${PUSH_MIN_INSTANCES}" >&2
exit 1
fi
ceiling="$(gcloud run revisions describe "${serving}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--format="value(metadata.annotations['autoscaling.knative.dev/maxScale'])")"
test "${ceiling}" = "${PUSH_MAX_INSTANCES}"
echo "serving revision ${serving} holds ${floor} minimum and ${ceiling} maximum instances"
echo "ROLLBACK_REVISION=${serving}" >> "${GITHUB_ENV}"
# No traffic and a per-revision tag: the candidate boots, applies schema, and is probed on
@@ -110,7 +136,6 @@ jobs:
--image "${IMAGE}" \
--tag "${tag}" \
--no-traffic \
--max-instances "${PUSH_MAX_INSTANCES}" \
--quiet
candidate="$(gcloud run services describe "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
@@ -121,7 +146,11 @@ jobs:
echo "CANDIDATE_REVISION=$(jq -r '.revisionName' <<< "${candidate}")" >> "${GITHUB_ENV}"
echo "CANDIDATE_URL=$(jq -r '.url' <<< "${candidate}")" >> "${GITHUB_ENV}"
- name: Require the candidate to serve the exact image
# A tagged revision is directly addressable and sits outside the service-wide cap, so the
# candidate and the serving revision each draw up to the ceiling during the probe window.
# The lease is taken for exactly that doubling; a candidate that inherited a wider ceiling
# would exceed it, so the inherited scaling is asserted here too.
- name: Require the candidate to serve the exact image and inherited scaling
shell: bash
run: |
set -euo pipefail
@@ -130,6 +159,10 @@ jobs:
--format='value(spec.containers[0].image)')"
test "${served}" = "${IMAGE}"
test "${CANDIDATE_REVISION}" != "${ROLLBACK_REVISION}"
candidate_ceiling="$(gcloud run revisions describe "${CANDIDATE_REVISION}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--format="value(metadata.annotations['autoscaling.knative.dev/maxScale'])")"
test "${candidate_ceiling}" = "${PUSH_MAX_INSTANCES}"
- name: Probe the candidate readiness endpoint
shell: bash
@@ -154,6 +187,10 @@ jobs:
# runtime account's FCM grant end to end without delivering anything: validate_only stops
# Google before any push, and the deliberately invalid token means a healthy credential
# answers INVALID_ARGUMENT. PERMISSION_DENIED is the failure this step exists to catch.
#
# Only the four verdicts below are conclusive. A 429, a 5xx, or a transport failure says
# nothing about the credential, so it is retried rather than treated as either answer; a
# denied credential still fails on the first attempt, without burning the retries.
- name: Prove the runtime identity can reach FCM
shell: bash
run: |
@@ -162,13 +199,20 @@ jobs:
--impersonate-service-account "${PUSH_RUNTIME_SERVICE_ACCOUNT}")"
test -n "${token}"
body='{"validate_only":true,"message":{"token":"orca-push-deploy-probe-invalid-token","notification":{"title":"Orca","body":"deploy probe"}}}'
code="$(curl -sS -o "${RUNNER_TEMP}/push-fcm.json" -w '%{http_code}' --max-time 20 \
-X POST "https://fcm.googleapis.com/v1/projects/${GCP_PROJECT_ID}/messages:send" \
-H "Authorization: Bearer ${token}" \
-H 'Content-Type: application/json' \
--data "${body}" || true)"
status="$(jq -r '.error.status // empty' < "${RUNNER_TEMP}/push-fcm.json")"
echo "FCM validate-only send returned HTTP ${code} status ${status:-OK}"
for attempt in $(seq 1 5); do
code="$(curl -sS -o "${RUNNER_TEMP}/push-fcm.json" -w '%{http_code}' --max-time 20 \
-X POST "https://fcm.googleapis.com/v1/projects/${GCP_PROJECT_ID}/messages:send" \
-H "Authorization: Bearer ${token}" \
-H 'Content-Type: application/json' \
--data "${body}" || true)"
status="$(jq -r '.error.status // empty' < "${RUNNER_TEMP}/push-fcm.json" || true)"
echo "attempt ${attempt}: FCM validate-only send returned HTTP ${code} status ${status:-OK}"
if test "${status}" = PERMISSION_DENIED || test "${status}" = INVALID_ARGUMENT ||
test "${code}" = 401 || test "${code}" = 403; then
break
fi
sleep 5
done
if test "${status}" = PERMISSION_DENIED || test "${code}" = 401 || test "${code}" = 403; then
echo "the push runtime identity cannot send through FCM" >&2
exit 1
@@ -189,13 +233,15 @@ jobs:
| jq -r '[.status.traffic[] | select((.percent // 0) > 0)]
| if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
test "${serving}" = "${CANDIDATE_REVISION}"
echo "TRAFFIC_SHIFTED=true" >> "${GITHUB_ENV}"
- name: Verify the public origin after the shift
# Why: the summary is written before the origin check, not after it. Once traffic has
# moved, the rollback target is the single thing an operator needs, and a summary that only
# appeared on success would be missing in exactly the run that needs it.
- name: Publish the rollout summary
shell: bash
run: |
set -euo pipefail
code="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 10 "${PUSH_ORIGIN}/ready")"
test "${code}" = 200
{
echo '### Push gateway deployed'
echo
@@ -207,6 +253,74 @@ jobs:
"--region ${GCP_REGION} --to-revisions ${ROLLBACK_REVISION}=100\`"
} >> "${GITHUB_STEP_SUMMARY}"
- name: Verify the public origin after the shift
shell: bash
run: |
set -euo pipefail
for attempt in $(seq 1 30); do
code="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 10 \
"${PUSH_ORIGIN}/ready" || true)"
if test "${code}" = 200; then
echo "${PUSH_ORIGIN} is ready after ${attempt} attempt(s)"
exit 0
fi
echo "attempt ${attempt}: ${PUSH_ORIGIN}/ready returned ${code}"
sleep 5
done
echo "${PUSH_ORIGIN} never reported ready after the shift" >&2
exit 1
# Why: everything after the shift runs with production on the candidate. A failure there
# is not a failure to deploy, it is a live gateway that has to go back, so the traffic move
# is undone here rather than left to whoever reads the run.
- name: Roll traffic back to the previous revision
if: ${{ failure() && env.TRAFFIC_SHIFTED == 'true' }}
shell: bash
run: |
set -euo pipefail
test -n "${ROLLBACK_REVISION:-}"
gcloud run services update-traffic "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" \
--region "${GCP_REGION}" \
--to-revisions "${ROLLBACK_REVISION}=100" \
--quiet
serving="$(gcloud run services describe "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
| jq -r '[.status.traffic[] | select((.percent // 0) > 0)]
| if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
test "${serving}" = "${ROLLBACK_REVISION}"
{
echo
echo '### Push gateway rolled back'
echo
echo "Traffic returned to \`${ROLLBACK_REVISION}\`; the candidate" \
"\`${CANDIDATE_REVISION}\` no longer serves."
} >> "${GITHUB_STEP_SUMMARY}"
# Why: a candidate that never took traffic is a revision holding a warm floor and a Cloud
# SQL pool for nothing. Its tag comes off first, because Cloud Run refuses to delete a
# revision a traffic target still names, and clearing CANDIDATE_TAG makes the always() tag
# step below a no-op rather than a second failure.
- name: Delete the candidate revision that never took traffic
if: ${{ failure() && env.TRAFFIC_SHIFTED != 'true' }}
shell: bash
run: |
set -euo pipefail
test -n "${CANDIDATE_REVISION:-}" || exit 0
if test -n "${CANDIDATE_TAG:-}"; then
gcloud run services update-traffic "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" \
--region "${GCP_REGION}" \
--remove-tags "${CANDIDATE_TAG}" \
--quiet
echo "CANDIDATE_TAG=" >> "${GITHUB_ENV}"
fi
gcloud run revisions delete "${CANDIDATE_REVISION}" \
--project "${GCP_PROJECT_ID}" \
--region "${GCP_REGION}" \
--quiet
echo "deleted the candidate revision ${CANDIDATE_REVISION}"
- name: Drop the candidate traffic tag
if: always()
shell: bash
@@ -1,7 +1,13 @@
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import test from 'node:test'
import { concurrencyBlocks, jobIf, jobs, leaseSteps } from './cloud-sql-rollout-lock-census.mjs'
import {
concurrencyBlocks,
jobIf,
jobs,
LEASE_ACTION,
leaseSteps
} from './cloud-sql-rollout-lock-census.mjs'
import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs'
// Why: the push gateway holds the APNs key and is the only thing standing between a paired
@@ -81,18 +87,66 @@ test('the candidate revision takes no traffic and is addressed by its own tag',
assert.match(workflow, /^ {12}--no-traffic \\$/m)
assert.match(workflow, /--tag "\$\{tag\}"/)
assert.match(workflow, /test "\$\{CANDIDATE_REVISION\}" != "\$\{ROLLBACK_REVISION\}"/)
// A tagged revision is directly addressable and sits outside the service-wide cap, so the
// candidate needs its own ceiling or it doubles the gateway's Cloud SQL draw while probing.
assert.match(workflow, /--max-instances "\$\{PUSH_MAX_INSTANCES\}"/)
assert.match(workflow, /PUSH_MAX_INSTANCES: 4$/m)
assert.match(terraform('variables.tf'), /variable "push_max_instances"[\s\S]*?default {5}= 4/)
assert.ok(
indexOfStep('Record the serving revision before the rollout') <
indexOfStep('Record the serving revision and require its Terraform-owned scaling') <
indexOfStep('Deploy the candidate revision with no traffic'),
'the rollback target must be captured before the candidate exists'
)
})
// Why: scaling is a Terraform-owned field that `lifecycle.ignore_changes` does not cover, so a
// deploy that passed --max-instances would revert a later push_max_instances raise on every run.
// The workflow asserts the shape instead of writing it, on the serving revision before the
// candidate exists and on the candidate that inherits it.
test('the deploy asserts the Terraform-owned scaling instead of mutating it', () => {
assert.doesNotMatch(workflow, /--max-instances/, 'the deploy must not write a scaling field')
assert.doesNotMatch(workflow, /--min-instances "/, 'the deploy must not write a scaling field')
// The floor is the variables.tf default; production.tfvars overrides only the ceiling, down to
// the two instances the Cloud SQL connection budget leaves room for.
assert.match(workflow, /PUSH_MIN_INSTANCES: 1$/m)
assert.match(workflow, /PUSH_MAX_INSTANCES: 2$/m)
assert.match(terraform('variables.tf'), /variable "push_min_instances"[\s\S]*?default {5}= 1/)
assert.match(terraform('environments/production.tfvars'), /^push_max_instances {9}= 2$/m)
const gate = indexOfStep('Record the serving revision and require its Terraform-owned scaling')
assert.ok(gate < indexOfStep('Deploy the candidate revision with no traffic'))
assert.match(workflow, /autoscaling\.knative\.dev\/minScale/)
assert.match(workflow, /\[\[ "\$\{floor:-0\}" -lt "\$\{PUSH_MIN_INSTANCES\}" \]\]/)
assert.match(workflow, /test "\$\{ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/)
assert.match(workflow, /test "\$\{candidate_ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/)
})
// Why: the image build is not a Cloud SQL operation, and the lease is a global serialization
// point. A build inside it blocks every relay deploy and rehome for its duration.
test('the image is built before the rollout lease is taken', () => {
const lease = workflow.indexOf(`- uses: ${LEASE_ACTION}`)
assert.notEqual(lease, -1)
const build = workflow.indexOf('- name: Build and publish the immutable gateway image')
const deployCandidate = workflow.indexOf('- name: Deploy the candidate revision with no traffic')
assert.ok(build < lease, 'the build must finish before the run takes the lease')
assert.ok(lease < deployCandidate, 'the lease must still cover the deploy, probe, and shift')
})
// Why: the gateway's Cloud SQL draw is instances x pool, and the root that takes the rollout
// lease can only account for a pool it declares. Leaving it at the application default hid it.
test('the database pool size is Terraform-owned and bounded at plan time', () => {
const source = terraform('push-gateway.tf')
assert.match(source, /name {2}= "ORCA_PUSH_DATABASE_POOL_MAX"/)
assert.match(source, /value = tostring\(var\.push_database_pool_max\)/)
assert.match(terraform('variables.tf'), /variable "push_database_pool_max"[\s\S]*?default {5}= 2/)
const block = /resource "google_cloud_run_v2_service" "push"[\s\S]*?\n lifecycle \{([\s\S]*?)\n \}/.exec(source)
assert.ok(block, 'the push service no longer declares a lifecycle block')
assert.match(
block[1],
/var\.push_max_instances \* var\.push_database_pool_max <= 4/,
'instances x pool must be bounded at plan time'
)
assert.match(
readFileSync(new URL('../../apps/push/src/config.ts', import.meta.url), 'utf8'),
/ORCA_PUSH_DATABASE_POOL_MAX/,
'the gateway must read the variable Terraform sets'
)
})
test('the candidate is probed on its own URL before any traffic moves', () => {
const probe = indexOfStep('Probe the candidate readiness endpoint')
assert.ok(probe > indexOfStep('Deploy the candidate revision with no traffic'))
@@ -115,6 +169,15 @@ test('the FCM probe is validate-only and separates a bad token from a bad creden
assert.match(workflow, /orca-push-deploy-probe-invalid-token/)
assert.match(workflow, /test "\$\{status\}" = INVALID_ARGUMENT/)
assert.match(workflow, /test "\$\{status\}" = PERMISSION_DENIED/)
// Only those four answers are conclusive; a 429 or a 5xx says nothing about the credential, so
// it is retried rather than read as either verdict. A denied credential still fails at once.
assert.match(workflow, /for attempt in \$\(seq 1 5\); do/)
const probe = workflow.slice(
workflow.indexOf('- name: Prove the runtime identity can reach FCM'),
workflow.indexOf('- name: Shift all traffic to the verified candidate')
)
assert.match(probe, /for attempt in \$\(seq 1 5\); do/)
assert.match(probe, /test "\$\{code\}" = 401 \|\| test "\$\{code\}" = 403; then\n {14}break/)
assert.match(
workflow,
/--impersonate-service-account "\$\{PUSH_RUNTIME_SERVICE_ACCOUNT\}"/,
@@ -150,11 +213,80 @@ test('the traffic shift is all-or-nothing and is verified after the fact', () =>
assert.match(workflow, /"\$\{PUSH_ORIGIN\}\/ready"/)
})
test('the run reports a rollback target and always drops its traffic tag', () => {
// Why: the origin can lag the traffic move by seconds, and a single unlucky curl would otherwise
// roll a healthy deploy back. It retries on the same schedule as the candidate probe.
test('the post-shift origin check retries like the candidate probe', () => {
const check = workflow.slice(
workflow.indexOf('- name: Verify the public origin after the shift'),
workflow.indexOf('- name: Roll traffic back to the previous revision')
)
assert.match(check, /for attempt in \$\(seq 1 30\); do/)
assert.match(check, /sleep 5/)
assert.match(check, /test "\$\{code\}" = 200/)
})
// Why: the summary carries the rollback target. Writing it after the origin check meant the one
// run that needed it, the run whose check failed, was the one run that never got it.
test('the summary is written before anything that can fail after the shift', () => {
const summary = indexOfStep('Publish the rollout summary')
assert.ok(summary > indexOfStep('Shift all traffic to the verified candidate'))
assert.ok(summary < indexOfStep('Verify the public origin after the shift'))
assert.match(workflow, /--to-revisions \$\{ROLLBACK_REVISION\}=100/)
assert.match(workflow, /GITHUB_STEP_SUMMARY/)
})
// Why: everything after the shift runs with production on the candidate, so a failure there is a
// live gateway that has to go back. The marker is what separates that case from a failure before
// the shift, where production never moved and the candidate is the thing to clean up.
test('a failure after the shift rolls production back automatically', () => {
const rollback = indexOfStep('Roll traffic back to the previous revision')
assert.ok(rollback > indexOfStep('Verify the public origin after the shift'))
assert.match(workflow, /echo "TRAFFIC_SHIFTED=true" >> "\$\{GITHUB_ENV\}"/)
const shift = workflow.indexOf('- name: Shift all traffic to the verified candidate')
assert.ok(
workflow.indexOf('echo "TRAFFIC_SHIFTED=true"') > shift,
'the marker must be set only once the shift has been verified'
)
const body = workflow.slice(
workflow.indexOf('- name: Roll traffic back to the previous revision'),
workflow.indexOf('- name: Delete the candidate revision that never took traffic')
)
assert.match(
body,
/if: \$\{\{ failure\(\) && env\.TRAFFIC_SHIFTED == 'true' \}\}/,
'the rollback must be conditioned on both failure and the shift marker'
)
assert.match(body, /test -n "\$\{ROLLBACK_REVISION:-\}"/)
assert.match(body, /--to-revisions "\$\{ROLLBACK_REVISION\}=100"/)
assert.match(body, /test "\$\{serving\}" = "\$\{ROLLBACK_REVISION\}"/)
assert.match(body, /GITHUB_STEP_SUMMARY/, 'the rollback must be reported in the summary')
})
// Why: a candidate that never took traffic still holds a warm instance and a Cloud SQL pool. Its
// tag comes off first, because Cloud Run refuses to delete a revision a traffic target names.
test('a failure before the shift deletes the candidate it created', () => {
const body = workflow.slice(
workflow.indexOf('- name: Delete the candidate revision that never took traffic'),
workflow.indexOf('- name: Drop the candidate traffic tag')
)
assert.match(
body,
/if: \$\{\{ failure\(\) && env\.TRAFFIC_SHIFTED != 'true' \}\}/,
'the cleanup must be conditioned on both failure and the absence of the shift marker'
)
assert.match(body, /test -n "\$\{CANDIDATE_REVISION:-\}" \|\| exit 0/)
assert.ok(
body.indexOf('--remove-tags') < body.indexOf('gcloud run revisions delete'),
'the tag must come off before the revision is deleted'
)
assert.match(body, /echo "CANDIDATE_TAG=" >> "\$\{GITHUB_ENV\}"/)
})
test('the run always drops its traffic tag', () => {
const cleanup = indexOfStep('Drop the candidate traffic tag')
assert.equal(cleanup, stepNames().length - 1, 'tag cleanup must be the last step')
assert.match(workflow, /--remove-tags "\$\{CANDIDATE_TAG\}"/)
const body = workflow.slice(workflow.indexOf('- name: Drop the candidate traffic tag'))
assert.match(body, /if: always\(\)/)
assert.match(body, /test -n "\$\{CANDIDATE_TAG:-\}" \|\| exit 0/)
})
@@ -33,6 +33,13 @@ function requiredInteger(source, pattern, label) {
return value
}
// A tfvars file states only what it overrides, so an absent key means the variable default holds.
// Reading the default as the fallback keeps this honest either way.
function overriddenInteger(override, overridePattern, source, pattern, label) {
if (!overridePattern.test(override)) return requiredInteger(source, pattern, label)
return requiredInteger(override, overridePattern, label)
}
function productionCells(source, defaultPoolMax) {
const fencedMatch = source.match(/relay_gce_fenced_cells\s*=\s*\[([^\]]*)\]/)
if (!fencedMatch) throw new Error('could not read fenced Relay cells')
@@ -52,11 +59,13 @@ function productionCells(source, defaultPoolMax) {
}
export function calculateRelayCloudSqlConnectionBudget(inputs) {
const pushDraw = inputs.pushInstances * inputs.pushPoolMax
const consumers = {
cells: inputs.cellPoolTotal + inputs.asiaCellCount * inputs.asiaPoolMax,
directors: inputs.directorInstances * inputs.directorPoolMax,
auth: inputs.authInstances * inputs.authPoolMax,
api: inputs.apiInstances * inputs.apiPoolMax
api: inputs.apiInstances * inputs.apiPoolMax,
push: pushDraw
}
const configuredMaximum = Object.values(consumers).reduce((total, value) => total + value, 0)
const retainedDirectorRollback = inputs.directorInstances * inputs.directorPoolMax
@@ -64,6 +73,11 @@ export function calculateRelayCloudSqlConnectionBudget(inputs) {
relayDirectorCandidate: retainedDirectorRollback * 2,
apiCandidate: retainedDirectorRollback + inputs.apiInstances * inputs.apiPoolMax,
authCandidate: retainedDirectorRollback + inputs.authInstances * inputs.authPoolMax,
// The push candidate doubles rather than adding one copy, like the director candidate and
// unlike the API and auth ones: cloud-push-deploy.yml probes a *tagged* revision, which is
// directly addressable and so sits outside the service-wide instance cap, letting the
// candidate and the serving revision each reach push_max_instances at the same time.
pushCandidate: retainedDirectorRollback + pushDraw * 2,
relayCells: retainedDirectorRollback
}
const rolloutOverlap = Math.max(...Object.values(candidateOverlap))
@@ -131,6 +145,20 @@ export function readRelayCloudSqlConnectionBudget({
/variable\s+"relay_director_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
'director pool maximum'
),
// The mobile push gateway shares this instance. Its draw was invisible here until Terraform
// declared the pool: docs/push-gateway.md, "Shape".
pushInstances: overriddenInteger(
productionTfvars,
/^\s*push_max_instances\s*=\s*(\d+)/m,
terraformVariables,
/variable\s+"push_max_instances"[\s\S]*?default\s*=\s*(\d+)/,
'push gateway instances'
),
pushPoolMax: requiredInteger(
terraformVariables,
/variable\s+"push_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
'push gateway pool maximum'
),
authInstances: apps.authInstances,
authPoolMax: apps.authPoolMax,
apiInstances: apps.apiInstances,
@@ -6,28 +6,91 @@ import {
readRelayCloudSqlConnectionBudget
} from './relay-cloud-sql-connection-budget.mjs'
test('production plus three Asia pools preserves allowance and reserve below the ceiling', () => {
// Why these numbers are this tight: the shared instance's 400 connections were already spoken
// for, and the relay shape below leaves exactly five. The gateway is sized to fit in four, two
// instances times a two-connection pool, and its rollout overlap of 23 stays under the API
// candidate's 65, so the Math.max is the API candidate rather than the gateway.
//
// `Deploy Relay Asia Topology` gates on `withinBudget == true`, so the single remaining
// connection is the whole margin. Anything that raises a pool or an instance count moves it.
test('production plus the push gateway keeps allowance and reserve below the ceiling', () => {
const report = readRelayCloudSqlConnectionBudget()
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 })
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50, push: 4 })
assert.deepEqual(report.asia, { cells: 3, poolMax: 10 })
assert.equal(report.configuredMaximum, 315)
assert.equal(report.configuredMaximum, 319)
assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30)
assert.equal(report.rolloutOverlap.apiCandidate, 65)
assert.equal(report.rolloutOverlap.authCandidate, 35)
assert.equal(report.rolloutOverlap.pushCandidate, 23)
assert.equal(report.rolloutOverlap.relayCells, 15)
assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15)
// The gateway does not set the maximum; the API candidate does, as it did before it existed.
assert.equal(report.rolloutOverlap.maximum, 65)
assert.equal(report.maintenanceAdminAllowance, 5)
assert.equal(report.explicitReserve, 10)
assert.equal(report.usableCeiling, 390)
assert.equal(report.operatingMaximum, 389)
assert.equal(report.remainingWithinUsableCeiling, 1)
assert.equal(report.budgetedTotal, 399)
assert.equal(report.unallocated, 1)
assert.equal(report.withinBudget, true)
})
// Why: the same relay shape without a push gateway is the before picture, and it stood at five
// connections clear. Holding it here keeps the gateway's cost visible as the four it takes,
// rather than letting drift elsewhere in the budget hide inside the same margin.
test('the same relay shape without the gateway stays inside the ceiling', () => {
const report = calculateRelayCloudSqlConnectionBudget({
cellPoolTotal: 200,
asiaCellCount: 3,
asiaPoolMax: 10,
directorInstances: 5,
directorPoolMax: 3,
authInstances: 2,
authPoolMax: 10,
apiInstances: 10,
apiPoolMax: 5,
pushInstances: 0,
pushPoolMax: 0,
maxConnections: 400,
maintenanceAdminAllowance: 5,
explicitReserve: 10
})
assert.equal(report.consumers.push, 0)
assert.equal(report.rolloutOverlap.maximum, 65)
assert.equal(report.operatingMaximum, 385)
assert.equal(report.remainingWithinUsableCeiling, 5)
assert.equal(report.budgetedTotal, 395)
assert.equal(report.unallocated, 5)
assert.equal(report.withinBudget, true)
})
// Why: a tagged candidate is directly addressable and sits outside the service-wide cap, so both
// push revisions can reach the ceiling at once. The API and auth candidates add one copy; this
// one adds two, like the director candidate.
test('the push rollout scenario doubles the gateway draw over the retained director', () => {
const report = calculateRelayCloudSqlConnectionBudget({
cellPoolTotal: 0,
asiaCellCount: 0,
asiaPoolMax: 0,
directorInstances: 5,
directorPoolMax: 3,
authInstances: 0,
authPoolMax: 0,
apiInstances: 0,
apiPoolMax: 0,
pushInstances: 2,
pushPoolMax: 2,
maxConnections: 400,
maintenanceAdminAllowance: 5,
explicitReserve: 10
})
assert.equal(report.consumers.push, 4)
// 15 retained director rollback, plus the 4-connection draw counted twice.
assert.equal(report.rolloutOverlap.pushCandidate, 23)
})
test('fails closed when pool growth consumes the explicit reserve', () => {
const report = calculateRelayCloudSqlConnectionBudget({
cellPoolTotal: 200,
@@ -39,12 +102,14 @@ test('fails closed when pool growth consumes the explicit reserve', () => {
authPoolMax: 10,
apiInstances: 20,
apiPoolMax: 5,
pushInstances: 4,
pushPoolMax: 10,
maxConnections: 400,
maintenanceAdminAllowance: 5,
explicitReserve: 10
})
assert.equal(report.operatingMaximum, 515)
assert.equal(report.operatingMaximum, 555)
assert.equal(report.withinBudget, false)
})
@@ -63,7 +128,11 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => {
}
}
`,
terraformVariables: 'variable "relay_director_database_pool_max" { default = 3 }',
terraformVariables: [
'variable "relay_director_database_pool_max" { default = 3 }',
'variable "push_max_instances" { default = 1 }',
'variable "push_database_pool_max" { default = 2 }'
].join('\n'),
relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10'
},
maxConnections: 100,
@@ -72,8 +141,42 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => {
})
assert.equal(report.consumers.cells, 14)
assert.equal(report.operatingMaximum, 46)
assert.equal(report.budgetedTotal, 47)
// No push_max_instances in this tfvars, so the variable default of one instance holds.
assert.equal(report.consumers.push, 2)
assert.equal(report.operatingMaximum, 48)
assert.equal(report.budgetedTotal, 49)
})
// Why: production.tfvars overrides push_max_instances down to 2 while variables.tf still defaults
// to 4, so reading the default instead of the override would overstate the live draw by half.
test('a tfvars push_max_instances override wins over the variable default', () => {
const report = readRelayCloudSqlConnectionBudget({
proposedAsiaCellCount: 1,
appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 },
sources: {
productionTfvars: `
relay_max_instances = 1
push_max_instances = 3
relay_gce_fenced_cells = []
relay_gce_cells = {
"production-gce-c2" = { database_pool_max = 4
}
}
`,
terraformVariables: [
'variable "relay_director_database_pool_max" { default = 3 }',
'variable "push_max_instances" { default = 1 }',
'variable "push_database_pool_max" { default = 2 }'
].join('\n'),
relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10'
},
maxConnections: 100,
maintenanceAdminAllowance: 1,
explicitReserve: 1
})
assert.equal(report.consumers.push, 6)
assert.equal(report.rolloutOverlap.pushCandidate, 15)
})
test('requires strict headroom below the physical ceiling', () => {
@@ -87,12 +190,14 @@ test('requires strict headroom below the physical ceiling', () => {
authPoolMax: 10,
apiInstances: 1,
apiPoolMax: 5,
pushInstances: 1,
pushPoolMax: 2,
maxConnections: 50,
maintenanceAdminAllowance: 9,
explicitReserve: 3
})
assert.equal(report.budgetedTotal, 63)
assert.equal(report.budgetedTotal, 65)
assert.equal(report.withinBudget, false)
})
+65 -18
View File
@@ -21,7 +21,8 @@ edit plus a second set of Apple credentials.
| --- | --- | --- |
| Cloud Run service | `orca-cloud-push` | `push_cloud_run_service_name` |
| Region | `us-central1` | `region` |
| Instances | min 1, max 4 | `push_min_instances`, `push_max_instances` |
| Instances | min 1, max 2 | `push_min_instances`, `push_max_instances` |
| Database pool | 2 per instance | `push_database_pool_max` |
| Concurrency | 80 | `push_concurrency` |
| Ingress | all | `INGRESS_TRAFFIC_ALL` |
| Invoker | IAM disabled | `invoker_iam_disabled = true` on the service |
@@ -29,8 +30,24 @@ edit plus a second set of Apple credentials.
| Database | `orca_push` on the shared Cloud SQL instance | `google_sql_database.push` |
| Hostname | `push.onorca.dev` | `push_base_url` |
The minimum of one instance is deliberate. A cold start delays a notification past the point
where it is worth showing, and the three-second coalescing window lives in instance memory.
The minimum of one instance is deliberate and did not move when the ceiling came down to two. A
cold start delays a notification past the point where it is worth showing, and the three-second
coalescing window lives in instance memory, so the floor is what keeps a notification prompt. The
ceiling is a different question, answered below.
The maximum and the pool are set by the connection budget, not by the gateway's own appetite. Two
instances times a two-connection pool is a draw of 4, and a rollout doubles it to 8, because the
tagged candidate is directly addressable and sits outside the service-wide cap. The shared Cloud
SQL instance's 400 connections were already spoken for by the relay cells, the directors, auth,
and the API, which left five. Four is the whole of the room there was, and the gateway fits in
it.
Two connections per instance is enough for the work. A send runs two or three short queries, so
at concurrency 80 requests queue against the pool for microseconds rather than holding it. A
`lifecycle` precondition refuses a plan whose instances times pool exceeds 4, because a fifth
connection puts the checked budget over its ceiling and blocks `Deploy Relay Asia Topology`,
which gates on it. `dev/scripts/relay-cloud-sql-connection-budget.mjs` counts the gateway and
prints the whole picture.
Authentication is the host proof in `POST /v1/host/challenge`, not Cloud Run IAM, so the service
opts out of invoker IAM with `invoker_iam_disabled = true`, exactly as the relay director does.
@@ -47,6 +64,7 @@ Set on the container by Terraform:
| `ORCA_PUSH_PUBLIC_URL` | `push_base_url` |
| `ORCA_PUSH_FCM_PROJECT_ID` | `push_fcm_project_id`, empty means `project_id` |
| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-database-url`, version `latest` |
| `ORCA_PUSH_DATABASE_POOL_MAX` | `push_database_pool_max`, 2 per instance |
| `ORCA_PUSH_APNS_KEY` | Secret `orca-cloud-push-apns-key`, version `latest` |
| `ORCA_PUSH_APNS_KEY_ID` | Secret `orca-cloud-push-apns-key-id`, version `latest` |
| `ORCA_PUSH_APPLE_TEAM_ID` | Secret `orca-cloud-push-apple-team-id`, version `latest` |
@@ -130,28 +148,55 @@ It authenticates as the shared production deploy identity through
`PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER` and
`PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT`, which are already published. No new GitHub
variable is required. That account was chosen because the Cloud SQL rollout lease grant is
foundation-owned and already names it; a dedicated identity could not take that lease from this
root. Its authority over the gateway is exactly three bindings in `push-gateway.tf`: Cloud Run
developer on this one service, service-account user on the runtime account, and token creator on
the runtime account.
foundation-owned and names only that account; a dedicated identity could not take that lease from
this root, and the gateway's schema rollout has to serialize against the relay's.
**That choice widens what this workflow can reach, and the widening is deliberate.** Adding
`push-deploy.yml` to the provider allowlist gives the run the account's whole existing authority,
not only the push bindings: Artifact Registry writer on `orca-cloud`, `roles/run.developer` on
the relay director and the fence broker, accessor and version-adder on the relay
regional-placement secret, and service-account user on the relay runtime identities. It was
accepted as the price of the lease. What `push-gateway.tf` adds on top is three bindings scoped
to the gateway alone: Cloud Run developer on this one service, and service-account user plus
token creator on the runtime account. The bound on the rest is the provider condition, which
admits this exact workflow file on `main` in the `production` environment only, and the workflow
itself, which is dispatch-only behind a typed confirmation.
The run, in order:
1. Takes the production Cloud SQL rollout lease and holds it for the whole run. The gateway
applies its schema while the new revision starts, so the revision **is** the schema step;
there is no separate migration command to wrap.
2. Builds `apps/push/Dockerfile` with the `cloud/` build context and pushes to the existing
1. Builds `apps/push/Dockerfile` with the `cloud/` build context and pushes to the existing
`orca-cloud` Artifact Registry repository as `push:sha-<commit>`, then resolves the digest.
3. Records the currently serving revision as the rollback target.
This happens **before** the lease is taken. Artifact Registry is not the Cloud SQL instance,
and a multi-minute build inside the lease would block every relay deploy and rehome for its
duration.
2. Takes the production Cloud SQL rollout lease and holds it from here to the end. The gateway
applies its schema while the new revision starts, so the revision **is** the schema step;
there is no separate migration command to wrap. The lease therefore covers exactly the
connection-budget window: deploy, probe, shift.
3. Records the currently serving revision as the rollback target, and requires it to still hold
the Terraform-owned floor and ceiling. The candidate inherits that scaling, so a drifted
serving revision would be latched rather than corrected.
4. `gcloud run deploy --no-traffic` with a per-run traffic tag, so the candidate boots and
applies schema while every phone still reaches the previous revision.
applies schema while every phone still reaches the previous revision. The deploy passes no
scaling flag: the shape is Terraform's, and the candidate's inherited ceiling is asserted
instead.
5. Probes the tagged candidate's own `/ready`, up to 30 times at five-second intervals.
6. Sends a validate-only FCM message as the runtime identity, by impersonation. See below.
7. Shifts 100% of traffic to the candidate, verifies it is the only revision serving, and
checks `https://push.onorca.dev/ready`.
8. Always removes the traffic tag, so tags do not accumulate across runs.
7. Shifts 100% of traffic to the candidate and verifies it is the only revision serving.
8. Writes the run summary, including the rollback command, before checking the public origin, so
the summary exists even when the check that follows does not pass.
9. Checks `https://push.onorca.dev/ready`, up to 30 times at five-second intervals, since the
origin can lag the traffic move by a few seconds.
10. Always removes the traffic tag, so tags do not accumulate across runs.
Rolling back is a traffic move, printed in the run summary:
**Failure after the shift rolls itself back.** Everything from step 8 on runs with production
already on the candidate, so a failure there is not a failed deploy, it is a live gateway that
has to go back. The run returns traffic to the recorded rollback revision, verifies the move, and
reports it in the summary. A failure *before* the shift leaves production untouched and deletes
the candidate revision, which otherwise sits holding a warm instance and a Cloud SQL pool for
nothing.
To move traffic by hand, from the revision named in the run summary:
```sh
gcloud run services update-traffic orca-cloud-push \
@@ -167,7 +212,9 @@ mints an access token for `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com`
`validate_only: true` with a token that cannot exist. `validate_only` stops Google before any
delivery, and a healthy credential answers `INVALID_ARGUMENT` because the device token is
garbage. `PERMISSION_DENIED`, `401`, and `403` are the failures the step exists to catch, and
they fail the run before traffic moves. Probing as the deploy identity instead would prove
they fail the run immediately, before traffic moves. Those four answers are the only conclusive
ones: a `429`, a `5xx`, or a transport failure says nothing about the credential, so the send is
retried up to five times at five-second intervals rather than read as either verdict. Probing as the deploy identity instead would prove
something true about the wrong account.
## Rotating the APNs key
+20 -9
View File
@@ -411,20 +411,31 @@ Artifact Registry repository, and its rollout lease.
It needs **no new GitHub environment variable.** It authenticates as the shared production deploy
identity through the already-published `PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER`
and `PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT`, and reads `PRODUCTION_GCP_REGION` like the
rest. That account holds the foundation-owned Cloud SQL rollout lease grant, which a dedicated
identity could not be given from this root. Its authority over the gateway is three bindings in
`infra/terraform/push-gateway.tf` and nothing wider: Cloud Run developer on that one service, and
service-account user plus token creator on the gateway's runtime account.
rest. That account holds the foundation-owned Cloud SQL rollout lease grant, which names it and nothing
else, so a dedicated identity could not be given that lease from this root.
`infra/terraform/push-gateway.tf` adds three bindings scoped to the gateway: Cloud Run developer
on that one service, and service-account user plus token creator on the gateway's runtime
account. Those three are not the workflow's whole authority. Running as the shared account gives
the run every role that account already holds for the relay: Artifact Registry writer on
`orca-cloud`, `roles/run.developer` on the relay director and the fence broker, accessor and
version-adder on the relay regional-placement secret, and service-account user on the relay
runtime identities. That widening was accepted as the price of the lease, and it is bounded by
the provider condition and by the workflow being dispatch-only behind a typed confirmation.
The provider's workflow allowlist gained exactly one entry, `cloud-push-deploy.yml`, on `main` in
the `production` environment. That entry is required: the allowlist compares complete workflow
refs by equality, so the `cloud-` filename prefix alone does not admit a new file.
The run takes the production rollout lease and holds it across the deploy, because the gateway
applies its schema while the new revision starts. It builds `apps/push/Dockerfile`, deploys with
`--no-traffic` behind a per-run traffic tag, probes the candidate's own `/ready`, proves the
runtime identity can reach FCM with a validate-only send, and only then shifts 100% of traffic.
There is no staging gateway, so there is no staging counterpart to run first.
The run builds `apps/push/Dockerfile` **before** taking the lease, so an image build never blocks
a relay deploy or rehome, then holds the production rollout lease across the deploy itself,
because the gateway applies its schema while the new revision starts. Under the lease it checks
the serving revision's Terraform-owned scaling, deploys with `--no-traffic` behind a per-run
traffic tag and no scaling flag of its own, probes the candidate's own `/ready`, proves the
runtime identity can reach FCM with a validate-only send, and only then shifts 100% of traffic. A
failure after the shift returns traffic to the recorded rollback revision; a failure before it
deletes the candidate. There is no staging gateway, so there is no staging counterpart to run
first.
Full runbook, including the APNs key rotation and the DNS record the `stablyai/orca-cloud` apps
root still owes, is in `docs/push-gateway.md`.
@@ -412,6 +412,9 @@ relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels
# Mobile push gateway. Production is the only environment that runs one; the runtime account,
# the three Apple secrets, and their accessor bindings already exist and are imported once
# (see docs/push-gateway.md).
push_gateway_enabled = true
push_base_url = "https://push.onorca.dev"
push_gateway_enabled = true
push_base_url = "https://push.onorca.dev"
# Sized so the gateway's rollout overlap, the retained director rollback plus its doubled draw,
# stays under the API candidate's, which keeps the checked Cloud SQL connection budget green.
push_max_instances = 2
manage_push_domain_mapping = true
+38 -5
View File
@@ -40,9 +40,10 @@ locals {
push_fqdn = replace(replace(var.push_base_url, "https://", ""), "http://", "")
# The shared production deploy identity runs `cloud-push-deploy.yml`. Its Cloud Run and
# service-account grants are scoped to this service alone; the account itself is declared in
# relay-github-actions.tf and is production-only.
# The shared production deploy identity runs `cloud-push-deploy.yml`. The grants this file adds
# are scoped to this service and its runtime account alone, but the workflow inherits every
# other grant that account already holds for the relay; see the deploy-identity section below.
# The account itself is declared in relay-github-actions.tf and is production-only.
push_gateway_deploy_count = (
var.push_gateway_enabled && local.relay_create_production_ops_identity ? 1 : 0
)
@@ -227,6 +228,14 @@ resource "google_cloud_run_v2_service" "push" {
value = local.push_fcm_project_id
}
# Declared rather than left to the application default, so the gateway's share of the
# shared Cloud SQL connection budget is a value this root states and the precondition
# below can bound.
env {
name = "ORCA_PUSH_DATABASE_POOL_MAX"
value = tostring(var.push_database_pool_max)
}
env {
name = "ORCA_PUSH_DATABASE_URL"
@@ -284,6 +293,19 @@ resource "google_cloud_run_v2_service" "push" {
# 100% LATEST would silently undo either, and this root carries unrelated standing drift, so
# that apply need not be a push change at all.
lifecycle {
# Why: the gateway draws instances x pool from the shared Cloud SQL instance, and a rollout
# doubles it, because the tagged candidate is directly addressable and sits outside the
# service-wide cap. The instance's 400 connections were already spoken for by the relay
# cells, directors, auth, and API, which left five: 4 is the whole of the gateway's share and
# it fits, with the doubled 8 still under the API candidate's rollout overlap, the term
# dev/scripts/relay-cloud-sql-connection-budget.mjs maximizes over. A fifth connection here
# puts the checked budget over its ceiling and blocks Deploy Relay Asia Topology, which gates
# on it, so catch a raise at plan time rather than in someone else's rollout.
precondition {
condition = var.push_max_instances * var.push_database_pool_max <= 4
error_message = "Push gateway instances x database pool must stay within its 4-connection share of the shared Cloud SQL instance."
}
ignore_changes = [
client,
client_version,
@@ -327,8 +349,19 @@ resource "google_cloud_run_domain_mapping" "push" {
# --- Deploy identity grants -------------------------------------------------------------------
# `cloud-push-deploy.yml` authenticates as the shared production deploy account, because that
# account is the one the foundation root grants the Cloud SQL rollout lease to. These three
# bindings are the whole of its authority over the push gateway.
# account is the one the foundation root grants the Cloud SQL rollout lease to; the grant names
# that account and nothing else, so a dedicated push identity could not take the lease from this
# root and the gateway's schema rollout could not be serialized against the relay's.
#
# The three bindings below are the whole of that account's authority over the *push gateway*, but
# they are not the whole of what the workflow can do. Adding `push-deploy.yml` to the provider's
# allowlist in relay-github-actions.tf gives the run the account's entire existing authority:
# Artifact Registry writer on `orca-cloud`, `roles/run.developer` on the relay director and the
# fence broker, accessor and version-adder on the relay regional-placement secret, and
# service-account user on the relay runtime identities. That widening was accepted deliberately
# as the price of the lease. It is bounded by the provider condition, which admits this exact
# workflow file on `main` in the `production` environment only, and by the workflow itself, which
# is dispatch-only behind a typed confirmation.
resource "google_cloud_run_v2_service_iam_member" "github_production_push_developer" {
count = local.push_gateway_deploy_count
@@ -21,8 +21,13 @@ locals {
"operate-relay-asia-admission.yml",
"publish-relay-production.yml",
# The push gateway deploy runs as this account because the Cloud SQL rollout lease grant is
# foundation-owned and already names it; a dedicated identity could not take that lease.
# Its authority over the gateway is three bindings in push-gateway.tf and nothing more.
# foundation-owned and names only this account; a dedicated identity could not take that
# lease, and the gateway's schema rollout has to serialize against the relay's.
#
# This entry therefore grants that workflow every role the account already holds, not just
# the three push bindings in push-gateway.tf: Artifact Registry writer, run.developer on the
# relay director and fence broker, relay secret accessor and version-adder, and
# serviceAccountUser on the relay runtime identities. Accepted as the price of the lease.
"push-deploy.yml"
]
github_production_relay_capacity_workflow_file = "deploy-relay-production-capacity.yml"
+18
View File
@@ -548,6 +548,24 @@ variable "push_max_instances" {
}
}
# Why: the gateway's draw on the shared Cloud SQL instance is instances x pool, and the rollout
# lease is taken for twice that, because a tagged candidate is directly addressable and sits
# outside the service-wide cap. Leaving the pool at its application default made that draw
# invisible to this root, so it is declared here and set on the container.
#
# Two is sized to the work, not to the default: a send runs two or three short queries, and at
# concurrency 80 those queue against the pool for microseconds rather than holding it.
variable "push_database_pool_max" {
type = number
description = "Push gateway database pool size per instance; instances x pool is its Cloud SQL draw."
default = 2
validation {
condition = var.push_database_pool_max >= 1 && var.push_database_pool_max <= 100
error_message = "The push gateway pool must hold at least one connection and stay under the per-service bound."
}
}
variable "push_concurrency" {
type = number
description = "Cloud Run concurrency for short-lived push gateway HTTP requests."