mirror of
https://github.com/stablyai/orca.git
synced 2026-09-29 08:03:20 +00:00
fix(cloud): harden the push deploy workflow and size the gateway to the budget (#8129)
- Roll traffic back on a failed post-shift check; delete a candidate that never took traffic; retry the origin probe and the FCM probe. - Assert Terraform-owned scaling instead of mutating it from the workflow. - Build before taking the Cloud SQL rollout lease. - Declare the database pool in Terraform (2 per instance, max 2 instances) and add the gateway to the connection budget; the previous default put the shared instance 65 connections over its ceiling. - State plainly that the shared deploy identity's relay authority is inherited.
This commit is contained in:
@@ -37,12 +37,15 @@ jobs:
|
||||
IMAGE_NAME: push
|
||||
PUSH_ORIGIN: https://push.onorca.dev
|
||||
PUSH_RUNTIME_SERVICE_ACCOUNT: orca-cloud-push@onorca-cloud.iam.gserviceaccount.com
|
||||
# Ceiling the tagged candidate, matching push_max_instances in
|
||||
# environments/production.tfvars. A tagged revision is directly addressable and so sits
|
||||
# outside the service-wide cap: without this the candidate and the serving revision could
|
||||
# each reach the ceiling and double the gateway's Cloud SQL draw during the probe window.
|
||||
# Terraform still owns the value; this only fails a deploy that would exceed it.
|
||||
PUSH_MAX_INSTANCES: 4
|
||||
# Scaling the serving revision must already hold, matching push_min_instances and
|
||||
# push_max_instances. Terraform owns both, and the candidate inherits them from the
|
||||
# service, so this deploy never passes a scaling flag: doing so would write a
|
||||
# Terraform-owned field that `lifecycle.ignore_changes` does not cover, and a later
|
||||
# `push_max_instances` raise would then be reverted by every deploy. These two values
|
||||
# are the expected shape, asserted before the candidate is created and again on the
|
||||
# candidate itself, so a deploy that would change the gateway's Cloud SQL draw fails.
|
||||
PUSH_MIN_INSTANCES: 1
|
||||
PUSH_MAX_INSTANCES: 2
|
||||
CONFIRMATION: ${{ inputs.confirmation }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
@@ -60,18 +63,15 @@ jobs:
|
||||
|
||||
- uses: google-github-actions/setup-gcloud@v2
|
||||
|
||||
# Held across the deploy, not just a separate schema step: the gateway opens its pool and
|
||||
# applies its schema while the new revision starts, so the revision is the schema step.
|
||||
- uses: ./.github/actions/cloud-sql-rollout-lease
|
||||
with:
|
||||
bucket: onorca-cloud-terraform-state
|
||||
object: terraform/state/cloud-sql-rollout/production.lock
|
||||
|
||||
- uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Configure Docker auth
|
||||
run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet
|
||||
|
||||
# Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance,
|
||||
# and a multi-minute image build inside the lease blocks every relay deploy and rehome for
|
||||
# its duration. The lease below covers exactly the connection-budget window: deploy, probe,
|
||||
# shift.
|
||||
- name: Build and publish the immutable gateway image
|
||||
shell: bash
|
||||
run: |
|
||||
@@ -86,7 +86,18 @@ jobs:
|
||||
>> "${GITHUB_ENV}"
|
||||
echo "IMAGE_DIGEST=${digest}" >> "${GITHUB_ENV}"
|
||||
|
||||
- name: Record the serving revision before the rollout
|
||||
# Held across the deploy, not just a separate schema step: the gateway opens its pool and
|
||||
# applies its schema while the new revision starts, so the revision is the schema step.
|
||||
- uses: ./.github/actions/cloud-sql-rollout-lease
|
||||
with:
|
||||
bucket: onorca-cloud-terraform-state
|
||||
object: terraform/state/cloud-sql-rollout/production.lock
|
||||
|
||||
# Why: the candidate inherits the serving revision's scaling. A serving revision that has
|
||||
# drifted below the floor would hand the candidate a cold start on every notification, and
|
||||
# one that has drifted above the ceiling would hand it a larger Cloud SQL draw than the
|
||||
# rollout lease was taken for. Refuse to inherit either rather than latch it.
|
||||
- name: Record the serving revision and require its Terraform-owned scaling
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
@@ -95,6 +106,21 @@ jobs:
|
||||
| jq -r '[.status.traffic[] | select((.percent // 0) > 0)]
|
||||
| if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
|
||||
test -n "${serving}"
|
||||
floor="$(gcloud run revisions describe "${serving}" \
|
||||
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
|
||||
--format="value(metadata.annotations['autoscaling.knative.dev/minScale'])")"
|
||||
if [[ "${floor:-0}" -lt "${PUSH_MIN_INSTANCES}" ]]; then
|
||||
echo "serving revision ${serving} holds ${floor:-0} minimum instances," \
|
||||
"below ${PUSH_MIN_INSTANCES}; deploying would inherit and latch it." >&2
|
||||
echo "Restore the floor first: gcloud run services update ${SERVICE_NAME}" \
|
||||
"--region ${GCP_REGION} --min-instances=${PUSH_MIN_INSTANCES}" >&2
|
||||
exit 1
|
||||
fi
|
||||
ceiling="$(gcloud run revisions describe "${serving}" \
|
||||
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
|
||||
--format="value(metadata.annotations['autoscaling.knative.dev/maxScale'])")"
|
||||
test "${ceiling}" = "${PUSH_MAX_INSTANCES}"
|
||||
echo "serving revision ${serving} holds ${floor} minimum and ${ceiling} maximum instances"
|
||||
echo "ROLLBACK_REVISION=${serving}" >> "${GITHUB_ENV}"
|
||||
|
||||
# No traffic and a per-revision tag: the candidate boots, applies schema, and is probed on
|
||||
@@ -110,7 +136,6 @@ jobs:
|
||||
--image "${IMAGE}" \
|
||||
--tag "${tag}" \
|
||||
--no-traffic \
|
||||
--max-instances "${PUSH_MAX_INSTANCES}" \
|
||||
--quiet
|
||||
candidate="$(gcloud run services describe "${SERVICE_NAME}" \
|
||||
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
|
||||
@@ -121,7 +146,11 @@ jobs:
|
||||
echo "CANDIDATE_REVISION=$(jq -r '.revisionName' <<< "${candidate}")" >> "${GITHUB_ENV}"
|
||||
echo "CANDIDATE_URL=$(jq -r '.url' <<< "${candidate}")" >> "${GITHUB_ENV}"
|
||||
|
||||
- name: Require the candidate to serve the exact image
|
||||
# A tagged revision is directly addressable and sits outside the service-wide cap, so the
|
||||
# candidate and the serving revision each draw up to the ceiling during the probe window.
|
||||
# The lease is taken for exactly that doubling; a candidate that inherited a wider ceiling
|
||||
# would exceed it, so the inherited scaling is asserted here too.
|
||||
- name: Require the candidate to serve the exact image and inherited scaling
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
@@ -130,6 +159,10 @@ jobs:
|
||||
--format='value(spec.containers[0].image)')"
|
||||
test "${served}" = "${IMAGE}"
|
||||
test "${CANDIDATE_REVISION}" != "${ROLLBACK_REVISION}"
|
||||
candidate_ceiling="$(gcloud run revisions describe "${CANDIDATE_REVISION}" \
|
||||
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
|
||||
--format="value(metadata.annotations['autoscaling.knative.dev/maxScale'])")"
|
||||
test "${candidate_ceiling}" = "${PUSH_MAX_INSTANCES}"
|
||||
|
||||
- name: Probe the candidate readiness endpoint
|
||||
shell: bash
|
||||
@@ -154,6 +187,10 @@ jobs:
|
||||
# runtime account's FCM grant end to end without delivering anything: validate_only stops
|
||||
# Google before any push, and the deliberately invalid token means a healthy credential
|
||||
# answers INVALID_ARGUMENT. PERMISSION_DENIED is the failure this step exists to catch.
|
||||
#
|
||||
# Only the four verdicts below are conclusive. A 429, a 5xx, or a transport failure says
|
||||
# nothing about the credential, so it is retried rather than treated as either answer; a
|
||||
# denied credential still fails on the first attempt, without burning the retries.
|
||||
- name: Prove the runtime identity can reach FCM
|
||||
shell: bash
|
||||
run: |
|
||||
@@ -162,13 +199,20 @@ jobs:
|
||||
--impersonate-service-account "${PUSH_RUNTIME_SERVICE_ACCOUNT}")"
|
||||
test -n "${token}"
|
||||
body='{"validate_only":true,"message":{"token":"orca-push-deploy-probe-invalid-token","notification":{"title":"Orca","body":"deploy probe"}}}'
|
||||
code="$(curl -sS -o "${RUNNER_TEMP}/push-fcm.json" -w '%{http_code}' --max-time 20 \
|
||||
-X POST "https://fcm.googleapis.com/v1/projects/${GCP_PROJECT_ID}/messages:send" \
|
||||
-H "Authorization: Bearer ${token}" \
|
||||
-H 'Content-Type: application/json' \
|
||||
--data "${body}" || true)"
|
||||
status="$(jq -r '.error.status // empty' < "${RUNNER_TEMP}/push-fcm.json")"
|
||||
echo "FCM validate-only send returned HTTP ${code} status ${status:-OK}"
|
||||
for attempt in $(seq 1 5); do
|
||||
code="$(curl -sS -o "${RUNNER_TEMP}/push-fcm.json" -w '%{http_code}' --max-time 20 \
|
||||
-X POST "https://fcm.googleapis.com/v1/projects/${GCP_PROJECT_ID}/messages:send" \
|
||||
-H "Authorization: Bearer ${token}" \
|
||||
-H 'Content-Type: application/json' \
|
||||
--data "${body}" || true)"
|
||||
status="$(jq -r '.error.status // empty' < "${RUNNER_TEMP}/push-fcm.json" || true)"
|
||||
echo "attempt ${attempt}: FCM validate-only send returned HTTP ${code} status ${status:-OK}"
|
||||
if test "${status}" = PERMISSION_DENIED || test "${status}" = INVALID_ARGUMENT ||
|
||||
test "${code}" = 401 || test "${code}" = 403; then
|
||||
break
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
if test "${status}" = PERMISSION_DENIED || test "${code}" = 401 || test "${code}" = 403; then
|
||||
echo "the push runtime identity cannot send through FCM" >&2
|
||||
exit 1
|
||||
@@ -189,13 +233,15 @@ jobs:
|
||||
| jq -r '[.status.traffic[] | select((.percent // 0) > 0)]
|
||||
| if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
|
||||
test "${serving}" = "${CANDIDATE_REVISION}"
|
||||
echo "TRAFFIC_SHIFTED=true" >> "${GITHUB_ENV}"
|
||||
|
||||
- name: Verify the public origin after the shift
|
||||
# Why: the summary is written before the origin check, not after it. Once traffic has
|
||||
# moved, the rollback target is the single thing an operator needs, and a summary that only
|
||||
# appeared on success would be missing in exactly the run that needs it.
|
||||
- name: Publish the rollout summary
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
code="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 10 "${PUSH_ORIGIN}/ready")"
|
||||
test "${code}" = 200
|
||||
{
|
||||
echo '### Push gateway deployed'
|
||||
echo
|
||||
@@ -207,6 +253,74 @@ jobs:
|
||||
"--region ${GCP_REGION} --to-revisions ${ROLLBACK_REVISION}=100\`"
|
||||
} >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Verify the public origin after the shift
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
for attempt in $(seq 1 30); do
|
||||
code="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 10 \
|
||||
"${PUSH_ORIGIN}/ready" || true)"
|
||||
if test "${code}" = 200; then
|
||||
echo "${PUSH_ORIGIN} is ready after ${attempt} attempt(s)"
|
||||
exit 0
|
||||
fi
|
||||
echo "attempt ${attempt}: ${PUSH_ORIGIN}/ready returned ${code}"
|
||||
sleep 5
|
||||
done
|
||||
echo "${PUSH_ORIGIN} never reported ready after the shift" >&2
|
||||
exit 1
|
||||
|
||||
# Why: everything after the shift runs with production on the candidate. A failure there
|
||||
# is not a failure to deploy, it is a live gateway that has to go back, so the traffic move
|
||||
# is undone here rather than left to whoever reads the run.
|
||||
- name: Roll traffic back to the previous revision
|
||||
if: ${{ failure() && env.TRAFFIC_SHIFTED == 'true' }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
test -n "${ROLLBACK_REVISION:-}"
|
||||
gcloud run services update-traffic "${SERVICE_NAME}" \
|
||||
--project "${GCP_PROJECT_ID}" \
|
||||
--region "${GCP_REGION}" \
|
||||
--to-revisions "${ROLLBACK_REVISION}=100" \
|
||||
--quiet
|
||||
serving="$(gcloud run services describe "${SERVICE_NAME}" \
|
||||
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
|
||||
| jq -r '[.status.traffic[] | select((.percent // 0) > 0)]
|
||||
| if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
|
||||
test "${serving}" = "${ROLLBACK_REVISION}"
|
||||
{
|
||||
echo
|
||||
echo '### Push gateway rolled back'
|
||||
echo
|
||||
echo "Traffic returned to \`${ROLLBACK_REVISION}\`; the candidate" \
|
||||
"\`${CANDIDATE_REVISION}\` no longer serves."
|
||||
} >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
# Why: a candidate that never took traffic is a revision holding a warm floor and a Cloud
|
||||
# SQL pool for nothing. Its tag comes off first, because Cloud Run refuses to delete a
|
||||
# revision a traffic target still names, and clearing CANDIDATE_TAG makes the always() tag
|
||||
# step below a no-op rather than a second failure.
|
||||
- name: Delete the candidate revision that never took traffic
|
||||
if: ${{ failure() && env.TRAFFIC_SHIFTED != 'true' }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
test -n "${CANDIDATE_REVISION:-}" || exit 0
|
||||
if test -n "${CANDIDATE_TAG:-}"; then
|
||||
gcloud run services update-traffic "${SERVICE_NAME}" \
|
||||
--project "${GCP_PROJECT_ID}" \
|
||||
--region "${GCP_REGION}" \
|
||||
--remove-tags "${CANDIDATE_TAG}" \
|
||||
--quiet
|
||||
echo "CANDIDATE_TAG=" >> "${GITHUB_ENV}"
|
||||
fi
|
||||
gcloud run revisions delete "${CANDIDATE_REVISION}" \
|
||||
--project "${GCP_PROJECT_ID}" \
|
||||
--region "${GCP_REGION}" \
|
||||
--quiet
|
||||
echo "deleted the candidate revision ${CANDIDATE_REVISION}"
|
||||
|
||||
- name: Drop the candidate traffic tag
|
||||
if: always()
|
||||
shell: bash
|
||||
|
||||
@@ -1,7 +1,13 @@
|
||||
import assert from 'node:assert/strict'
|
||||
import { readFileSync } from 'node:fs'
|
||||
import test from 'node:test'
|
||||
import { concurrencyBlocks, jobIf, jobs, leaseSteps } from './cloud-sql-rollout-lock-census.mjs'
|
||||
import {
|
||||
concurrencyBlocks,
|
||||
jobIf,
|
||||
jobs,
|
||||
LEASE_ACTION,
|
||||
leaseSteps
|
||||
} from './cloud-sql-rollout-lock-census.mjs'
|
||||
import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs'
|
||||
|
||||
// Why: the push gateway holds the APNs key and is the only thing standing between a paired
|
||||
@@ -81,18 +87,66 @@ test('the candidate revision takes no traffic and is addressed by its own tag',
|
||||
assert.match(workflow, /^ {12}--no-traffic \\$/m)
|
||||
assert.match(workflow, /--tag "\$\{tag\}"/)
|
||||
assert.match(workflow, /test "\$\{CANDIDATE_REVISION\}" != "\$\{ROLLBACK_REVISION\}"/)
|
||||
// A tagged revision is directly addressable and sits outside the service-wide cap, so the
|
||||
// candidate needs its own ceiling or it doubles the gateway's Cloud SQL draw while probing.
|
||||
assert.match(workflow, /--max-instances "\$\{PUSH_MAX_INSTANCES\}"/)
|
||||
assert.match(workflow, /PUSH_MAX_INSTANCES: 4$/m)
|
||||
assert.match(terraform('variables.tf'), /variable "push_max_instances"[\s\S]*?default {5}= 4/)
|
||||
assert.ok(
|
||||
indexOfStep('Record the serving revision before the rollout') <
|
||||
indexOfStep('Record the serving revision and require its Terraform-owned scaling') <
|
||||
indexOfStep('Deploy the candidate revision with no traffic'),
|
||||
'the rollback target must be captured before the candidate exists'
|
||||
)
|
||||
})
|
||||
|
||||
// Why: scaling is a Terraform-owned field that `lifecycle.ignore_changes` does not cover, so a
|
||||
// deploy that passed --max-instances would revert a later push_max_instances raise on every run.
|
||||
// The workflow asserts the shape instead of writing it, on the serving revision before the
|
||||
// candidate exists and on the candidate that inherits it.
|
||||
test('the deploy asserts the Terraform-owned scaling instead of mutating it', () => {
|
||||
assert.doesNotMatch(workflow, /--max-instances/, 'the deploy must not write a scaling field')
|
||||
assert.doesNotMatch(workflow, /--min-instances "/, 'the deploy must not write a scaling field')
|
||||
// The floor is the variables.tf default; production.tfvars overrides only the ceiling, down to
|
||||
// the two instances the Cloud SQL connection budget leaves room for.
|
||||
assert.match(workflow, /PUSH_MIN_INSTANCES: 1$/m)
|
||||
assert.match(workflow, /PUSH_MAX_INSTANCES: 2$/m)
|
||||
assert.match(terraform('variables.tf'), /variable "push_min_instances"[\s\S]*?default {5}= 1/)
|
||||
assert.match(terraform('environments/production.tfvars'), /^push_max_instances {9}= 2$/m)
|
||||
const gate = indexOfStep('Record the serving revision and require its Terraform-owned scaling')
|
||||
assert.ok(gate < indexOfStep('Deploy the candidate revision with no traffic'))
|
||||
assert.match(workflow, /autoscaling\.knative\.dev\/minScale/)
|
||||
assert.match(workflow, /\[\[ "\$\{floor:-0\}" -lt "\$\{PUSH_MIN_INSTANCES\}" \]\]/)
|
||||
assert.match(workflow, /test "\$\{ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/)
|
||||
assert.match(workflow, /test "\$\{candidate_ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/)
|
||||
})
|
||||
|
||||
// Why: the image build is not a Cloud SQL operation, and the lease is a global serialization
|
||||
// point. A build inside it blocks every relay deploy and rehome for its duration.
|
||||
test('the image is built before the rollout lease is taken', () => {
|
||||
const lease = workflow.indexOf(`- uses: ${LEASE_ACTION}`)
|
||||
assert.notEqual(lease, -1)
|
||||
const build = workflow.indexOf('- name: Build and publish the immutable gateway image')
|
||||
const deployCandidate = workflow.indexOf('- name: Deploy the candidate revision with no traffic')
|
||||
assert.ok(build < lease, 'the build must finish before the run takes the lease')
|
||||
assert.ok(lease < deployCandidate, 'the lease must still cover the deploy, probe, and shift')
|
||||
})
|
||||
|
||||
// Why: the gateway's Cloud SQL draw is instances x pool, and the root that takes the rollout
|
||||
// lease can only account for a pool it declares. Leaving it at the application default hid it.
|
||||
test('the database pool size is Terraform-owned and bounded at plan time', () => {
|
||||
const source = terraform('push-gateway.tf')
|
||||
assert.match(source, /name {2}= "ORCA_PUSH_DATABASE_POOL_MAX"/)
|
||||
assert.match(source, /value = tostring\(var\.push_database_pool_max\)/)
|
||||
assert.match(terraform('variables.tf'), /variable "push_database_pool_max"[\s\S]*?default {5}= 2/)
|
||||
const block = /resource "google_cloud_run_v2_service" "push"[\s\S]*?\n lifecycle \{([\s\S]*?)\n \}/.exec(source)
|
||||
assert.ok(block, 'the push service no longer declares a lifecycle block')
|
||||
assert.match(
|
||||
block[1],
|
||||
/var\.push_max_instances \* var\.push_database_pool_max <= 4/,
|
||||
'instances x pool must be bounded at plan time'
|
||||
)
|
||||
assert.match(
|
||||
readFileSync(new URL('../../apps/push/src/config.ts', import.meta.url), 'utf8'),
|
||||
/ORCA_PUSH_DATABASE_POOL_MAX/,
|
||||
'the gateway must read the variable Terraform sets'
|
||||
)
|
||||
})
|
||||
|
||||
test('the candidate is probed on its own URL before any traffic moves', () => {
|
||||
const probe = indexOfStep('Probe the candidate readiness endpoint')
|
||||
assert.ok(probe > indexOfStep('Deploy the candidate revision with no traffic'))
|
||||
@@ -115,6 +169,15 @@ test('the FCM probe is validate-only and separates a bad token from a bad creden
|
||||
assert.match(workflow, /orca-push-deploy-probe-invalid-token/)
|
||||
assert.match(workflow, /test "\$\{status\}" = INVALID_ARGUMENT/)
|
||||
assert.match(workflow, /test "\$\{status\}" = PERMISSION_DENIED/)
|
||||
// Only those four answers are conclusive; a 429 or a 5xx says nothing about the credential, so
|
||||
// it is retried rather than read as either verdict. A denied credential still fails at once.
|
||||
assert.match(workflow, /for attempt in \$\(seq 1 5\); do/)
|
||||
const probe = workflow.slice(
|
||||
workflow.indexOf('- name: Prove the runtime identity can reach FCM'),
|
||||
workflow.indexOf('- name: Shift all traffic to the verified candidate')
|
||||
)
|
||||
assert.match(probe, /for attempt in \$\(seq 1 5\); do/)
|
||||
assert.match(probe, /test "\$\{code\}" = 401 \|\| test "\$\{code\}" = 403; then\n {14}break/)
|
||||
assert.match(
|
||||
workflow,
|
||||
/--impersonate-service-account "\$\{PUSH_RUNTIME_SERVICE_ACCOUNT\}"/,
|
||||
@@ -150,11 +213,80 @@ test('the traffic shift is all-or-nothing and is verified after the fact', () =>
|
||||
assert.match(workflow, /"\$\{PUSH_ORIGIN\}\/ready"/)
|
||||
})
|
||||
|
||||
test('the run reports a rollback target and always drops its traffic tag', () => {
|
||||
// Why: the origin can lag the traffic move by seconds, and a single unlucky curl would otherwise
|
||||
// roll a healthy deploy back. It retries on the same schedule as the candidate probe.
|
||||
test('the post-shift origin check retries like the candidate probe', () => {
|
||||
const check = workflow.slice(
|
||||
workflow.indexOf('- name: Verify the public origin after the shift'),
|
||||
workflow.indexOf('- name: Roll traffic back to the previous revision')
|
||||
)
|
||||
assert.match(check, /for attempt in \$\(seq 1 30\); do/)
|
||||
assert.match(check, /sleep 5/)
|
||||
assert.match(check, /test "\$\{code\}" = 200/)
|
||||
})
|
||||
|
||||
// Why: the summary carries the rollback target. Writing it after the origin check meant the one
|
||||
// run that needed it, the run whose check failed, was the one run that never got it.
|
||||
test('the summary is written before anything that can fail after the shift', () => {
|
||||
const summary = indexOfStep('Publish the rollout summary')
|
||||
assert.ok(summary > indexOfStep('Shift all traffic to the verified candidate'))
|
||||
assert.ok(summary < indexOfStep('Verify the public origin after the shift'))
|
||||
assert.match(workflow, /--to-revisions \$\{ROLLBACK_REVISION\}=100/)
|
||||
assert.match(workflow, /GITHUB_STEP_SUMMARY/)
|
||||
})
|
||||
|
||||
// Why: everything after the shift runs with production on the candidate, so a failure there is a
|
||||
// live gateway that has to go back. The marker is what separates that case from a failure before
|
||||
// the shift, where production never moved and the candidate is the thing to clean up.
|
||||
test('a failure after the shift rolls production back automatically', () => {
|
||||
const rollback = indexOfStep('Roll traffic back to the previous revision')
|
||||
assert.ok(rollback > indexOfStep('Verify the public origin after the shift'))
|
||||
assert.match(workflow, /echo "TRAFFIC_SHIFTED=true" >> "\$\{GITHUB_ENV\}"/)
|
||||
const shift = workflow.indexOf('- name: Shift all traffic to the verified candidate')
|
||||
assert.ok(
|
||||
workflow.indexOf('echo "TRAFFIC_SHIFTED=true"') > shift,
|
||||
'the marker must be set only once the shift has been verified'
|
||||
)
|
||||
const body = workflow.slice(
|
||||
workflow.indexOf('- name: Roll traffic back to the previous revision'),
|
||||
workflow.indexOf('- name: Delete the candidate revision that never took traffic')
|
||||
)
|
||||
assert.match(
|
||||
body,
|
||||
/if: \$\{\{ failure\(\) && env\.TRAFFIC_SHIFTED == 'true' \}\}/,
|
||||
'the rollback must be conditioned on both failure and the shift marker'
|
||||
)
|
||||
assert.match(body, /test -n "\$\{ROLLBACK_REVISION:-\}"/)
|
||||
assert.match(body, /--to-revisions "\$\{ROLLBACK_REVISION\}=100"/)
|
||||
assert.match(body, /test "\$\{serving\}" = "\$\{ROLLBACK_REVISION\}"/)
|
||||
assert.match(body, /GITHUB_STEP_SUMMARY/, 'the rollback must be reported in the summary')
|
||||
})
|
||||
|
||||
// Why: a candidate that never took traffic still holds a warm instance and a Cloud SQL pool. Its
|
||||
// tag comes off first, because Cloud Run refuses to delete a revision a traffic target names.
|
||||
test('a failure before the shift deletes the candidate it created', () => {
|
||||
const body = workflow.slice(
|
||||
workflow.indexOf('- name: Delete the candidate revision that never took traffic'),
|
||||
workflow.indexOf('- name: Drop the candidate traffic tag')
|
||||
)
|
||||
assert.match(
|
||||
body,
|
||||
/if: \$\{\{ failure\(\) && env\.TRAFFIC_SHIFTED != 'true' \}\}/,
|
||||
'the cleanup must be conditioned on both failure and the absence of the shift marker'
|
||||
)
|
||||
assert.match(body, /test -n "\$\{CANDIDATE_REVISION:-\}" \|\| exit 0/)
|
||||
assert.ok(
|
||||
body.indexOf('--remove-tags') < body.indexOf('gcloud run revisions delete'),
|
||||
'the tag must come off before the revision is deleted'
|
||||
)
|
||||
assert.match(body, /echo "CANDIDATE_TAG=" >> "\$\{GITHUB_ENV\}"/)
|
||||
})
|
||||
|
||||
test('the run always drops its traffic tag', () => {
|
||||
const cleanup = indexOfStep('Drop the candidate traffic tag')
|
||||
assert.equal(cleanup, stepNames().length - 1, 'tag cleanup must be the last step')
|
||||
assert.match(workflow, /--remove-tags "\$\{CANDIDATE_TAG\}"/)
|
||||
const body = workflow.slice(workflow.indexOf('- name: Drop the candidate traffic tag'))
|
||||
assert.match(body, /if: always\(\)/)
|
||||
assert.match(body, /test -n "\$\{CANDIDATE_TAG:-\}" \|\| exit 0/)
|
||||
})
|
||||
|
||||
@@ -33,6 +33,13 @@ function requiredInteger(source, pattern, label) {
|
||||
return value
|
||||
}
|
||||
|
||||
// A tfvars file states only what it overrides, so an absent key means the variable default holds.
|
||||
// Reading the default as the fallback keeps this honest either way.
|
||||
function overriddenInteger(override, overridePattern, source, pattern, label) {
|
||||
if (!overridePattern.test(override)) return requiredInteger(source, pattern, label)
|
||||
return requiredInteger(override, overridePattern, label)
|
||||
}
|
||||
|
||||
function productionCells(source, defaultPoolMax) {
|
||||
const fencedMatch = source.match(/relay_gce_fenced_cells\s*=\s*\[([^\]]*)\]/)
|
||||
if (!fencedMatch) throw new Error('could not read fenced Relay cells')
|
||||
@@ -52,11 +59,13 @@ function productionCells(source, defaultPoolMax) {
|
||||
}
|
||||
|
||||
export function calculateRelayCloudSqlConnectionBudget(inputs) {
|
||||
const pushDraw = inputs.pushInstances * inputs.pushPoolMax
|
||||
const consumers = {
|
||||
cells: inputs.cellPoolTotal + inputs.asiaCellCount * inputs.asiaPoolMax,
|
||||
directors: inputs.directorInstances * inputs.directorPoolMax,
|
||||
auth: inputs.authInstances * inputs.authPoolMax,
|
||||
api: inputs.apiInstances * inputs.apiPoolMax
|
||||
api: inputs.apiInstances * inputs.apiPoolMax,
|
||||
push: pushDraw
|
||||
}
|
||||
const configuredMaximum = Object.values(consumers).reduce((total, value) => total + value, 0)
|
||||
const retainedDirectorRollback = inputs.directorInstances * inputs.directorPoolMax
|
||||
@@ -64,6 +73,11 @@ export function calculateRelayCloudSqlConnectionBudget(inputs) {
|
||||
relayDirectorCandidate: retainedDirectorRollback * 2,
|
||||
apiCandidate: retainedDirectorRollback + inputs.apiInstances * inputs.apiPoolMax,
|
||||
authCandidate: retainedDirectorRollback + inputs.authInstances * inputs.authPoolMax,
|
||||
// The push candidate doubles rather than adding one copy, like the director candidate and
|
||||
// unlike the API and auth ones: cloud-push-deploy.yml probes a *tagged* revision, which is
|
||||
// directly addressable and so sits outside the service-wide instance cap, letting the
|
||||
// candidate and the serving revision each reach push_max_instances at the same time.
|
||||
pushCandidate: retainedDirectorRollback + pushDraw * 2,
|
||||
relayCells: retainedDirectorRollback
|
||||
}
|
||||
const rolloutOverlap = Math.max(...Object.values(candidateOverlap))
|
||||
@@ -131,6 +145,20 @@ export function readRelayCloudSqlConnectionBudget({
|
||||
/variable\s+"relay_director_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
|
||||
'director pool maximum'
|
||||
),
|
||||
// The mobile push gateway shares this instance. Its draw was invisible here until Terraform
|
||||
// declared the pool: docs/push-gateway.md, "Shape".
|
||||
pushInstances: overriddenInteger(
|
||||
productionTfvars,
|
||||
/^\s*push_max_instances\s*=\s*(\d+)/m,
|
||||
terraformVariables,
|
||||
/variable\s+"push_max_instances"[\s\S]*?default\s*=\s*(\d+)/,
|
||||
'push gateway instances'
|
||||
),
|
||||
pushPoolMax: requiredInteger(
|
||||
terraformVariables,
|
||||
/variable\s+"push_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/,
|
||||
'push gateway pool maximum'
|
||||
),
|
||||
authInstances: apps.authInstances,
|
||||
authPoolMax: apps.authPoolMax,
|
||||
apiInstances: apps.apiInstances,
|
||||
|
||||
@@ -6,28 +6,91 @@ import {
|
||||
readRelayCloudSqlConnectionBudget
|
||||
} from './relay-cloud-sql-connection-budget.mjs'
|
||||
|
||||
test('production plus three Asia pools preserves allowance and reserve below the ceiling', () => {
|
||||
// Why these numbers are this tight: the shared instance's 400 connections were already spoken
|
||||
// for, and the relay shape below leaves exactly five. The gateway is sized to fit in four, two
|
||||
// instances times a two-connection pool, and its rollout overlap of 23 stays under the API
|
||||
// candidate's 65, so the Math.max is the API candidate rather than the gateway.
|
||||
//
|
||||
// `Deploy Relay Asia Topology` gates on `withinBudget == true`, so the single remaining
|
||||
// connection is the whole margin. Anything that raises a pool or an instance count moves it.
|
||||
test('production plus the push gateway keeps allowance and reserve below the ceiling', () => {
|
||||
const report = readRelayCloudSqlConnectionBudget()
|
||||
|
||||
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 })
|
||||
assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50, push: 4 })
|
||||
assert.deepEqual(report.asia, { cells: 3, poolMax: 10 })
|
||||
assert.equal(report.configuredMaximum, 315)
|
||||
assert.equal(report.configuredMaximum, 319)
|
||||
assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30)
|
||||
assert.equal(report.rolloutOverlap.apiCandidate, 65)
|
||||
assert.equal(report.rolloutOverlap.authCandidate, 35)
|
||||
assert.equal(report.rolloutOverlap.pushCandidate, 23)
|
||||
assert.equal(report.rolloutOverlap.relayCells, 15)
|
||||
assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15)
|
||||
// The gateway does not set the maximum; the API candidate does, as it did before it existed.
|
||||
assert.equal(report.rolloutOverlap.maximum, 65)
|
||||
assert.equal(report.maintenanceAdminAllowance, 5)
|
||||
assert.equal(report.explicitReserve, 10)
|
||||
assert.equal(report.usableCeiling, 390)
|
||||
assert.equal(report.operatingMaximum, 389)
|
||||
assert.equal(report.remainingWithinUsableCeiling, 1)
|
||||
assert.equal(report.budgetedTotal, 399)
|
||||
assert.equal(report.unallocated, 1)
|
||||
assert.equal(report.withinBudget, true)
|
||||
})
|
||||
|
||||
// Why: the same relay shape without a push gateway is the before picture, and it stood at five
|
||||
// connections clear. Holding it here keeps the gateway's cost visible as the four it takes,
|
||||
// rather than letting drift elsewhere in the budget hide inside the same margin.
|
||||
test('the same relay shape without the gateway stays inside the ceiling', () => {
|
||||
const report = calculateRelayCloudSqlConnectionBudget({
|
||||
cellPoolTotal: 200,
|
||||
asiaCellCount: 3,
|
||||
asiaPoolMax: 10,
|
||||
directorInstances: 5,
|
||||
directorPoolMax: 3,
|
||||
authInstances: 2,
|
||||
authPoolMax: 10,
|
||||
apiInstances: 10,
|
||||
apiPoolMax: 5,
|
||||
pushInstances: 0,
|
||||
pushPoolMax: 0,
|
||||
maxConnections: 400,
|
||||
maintenanceAdminAllowance: 5,
|
||||
explicitReserve: 10
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.push, 0)
|
||||
assert.equal(report.rolloutOverlap.maximum, 65)
|
||||
assert.equal(report.operatingMaximum, 385)
|
||||
assert.equal(report.remainingWithinUsableCeiling, 5)
|
||||
assert.equal(report.budgetedTotal, 395)
|
||||
assert.equal(report.unallocated, 5)
|
||||
assert.equal(report.withinBudget, true)
|
||||
})
|
||||
|
||||
// Why: a tagged candidate is directly addressable and sits outside the service-wide cap, so both
|
||||
// push revisions can reach the ceiling at once. The API and auth candidates add one copy; this
|
||||
// one adds two, like the director candidate.
|
||||
test('the push rollout scenario doubles the gateway draw over the retained director', () => {
|
||||
const report = calculateRelayCloudSqlConnectionBudget({
|
||||
cellPoolTotal: 0,
|
||||
asiaCellCount: 0,
|
||||
asiaPoolMax: 0,
|
||||
directorInstances: 5,
|
||||
directorPoolMax: 3,
|
||||
authInstances: 0,
|
||||
authPoolMax: 0,
|
||||
apiInstances: 0,
|
||||
apiPoolMax: 0,
|
||||
pushInstances: 2,
|
||||
pushPoolMax: 2,
|
||||
maxConnections: 400,
|
||||
maintenanceAdminAllowance: 5,
|
||||
explicitReserve: 10
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.push, 4)
|
||||
// 15 retained director rollback, plus the 4-connection draw counted twice.
|
||||
assert.equal(report.rolloutOverlap.pushCandidate, 23)
|
||||
})
|
||||
|
||||
test('fails closed when pool growth consumes the explicit reserve', () => {
|
||||
const report = calculateRelayCloudSqlConnectionBudget({
|
||||
cellPoolTotal: 200,
|
||||
@@ -39,12 +102,14 @@ test('fails closed when pool growth consumes the explicit reserve', () => {
|
||||
authPoolMax: 10,
|
||||
apiInstances: 20,
|
||||
apiPoolMax: 5,
|
||||
pushInstances: 4,
|
||||
pushPoolMax: 10,
|
||||
maxConnections: 400,
|
||||
maintenanceAdminAllowance: 5,
|
||||
explicitReserve: 10
|
||||
})
|
||||
|
||||
assert.equal(report.operatingMaximum, 515)
|
||||
assert.equal(report.operatingMaximum, 555)
|
||||
assert.equal(report.withinBudget, false)
|
||||
})
|
||||
|
||||
@@ -63,7 +128,11 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => {
|
||||
}
|
||||
}
|
||||
`,
|
||||
terraformVariables: 'variable "relay_director_database_pool_max" { default = 3 }',
|
||||
terraformVariables: [
|
||||
'variable "relay_director_database_pool_max" { default = 3 }',
|
||||
'variable "push_max_instances" { default = 1 }',
|
||||
'variable "push_database_pool_max" { default = 2 }'
|
||||
].join('\n'),
|
||||
relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10'
|
||||
},
|
||||
maxConnections: 100,
|
||||
@@ -72,8 +141,42 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => {
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.cells, 14)
|
||||
assert.equal(report.operatingMaximum, 46)
|
||||
assert.equal(report.budgetedTotal, 47)
|
||||
// No push_max_instances in this tfvars, so the variable default of one instance holds.
|
||||
assert.equal(report.consumers.push, 2)
|
||||
assert.equal(report.operatingMaximum, 48)
|
||||
assert.equal(report.budgetedTotal, 49)
|
||||
})
|
||||
|
||||
// Why: production.tfvars overrides push_max_instances down to 2 while variables.tf still defaults
|
||||
// to 4, so reading the default instead of the override would overstate the live draw by half.
|
||||
test('a tfvars push_max_instances override wins over the variable default', () => {
|
||||
const report = readRelayCloudSqlConnectionBudget({
|
||||
proposedAsiaCellCount: 1,
|
||||
appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 },
|
||||
sources: {
|
||||
productionTfvars: `
|
||||
relay_max_instances = 1
|
||||
push_max_instances = 3
|
||||
relay_gce_fenced_cells = []
|
||||
relay_gce_cells = {
|
||||
"production-gce-c2" = { database_pool_max = 4
|
||||
}
|
||||
}
|
||||
`,
|
||||
terraformVariables: [
|
||||
'variable "relay_director_database_pool_max" { default = 3 }',
|
||||
'variable "push_max_instances" { default = 1 }',
|
||||
'variable "push_database_pool_max" { default = 2 }'
|
||||
].join('\n'),
|
||||
relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10'
|
||||
},
|
||||
maxConnections: 100,
|
||||
maintenanceAdminAllowance: 1,
|
||||
explicitReserve: 1
|
||||
})
|
||||
|
||||
assert.equal(report.consumers.push, 6)
|
||||
assert.equal(report.rolloutOverlap.pushCandidate, 15)
|
||||
})
|
||||
|
||||
test('requires strict headroom below the physical ceiling', () => {
|
||||
@@ -87,12 +190,14 @@ test('requires strict headroom below the physical ceiling', () => {
|
||||
authPoolMax: 10,
|
||||
apiInstances: 1,
|
||||
apiPoolMax: 5,
|
||||
pushInstances: 1,
|
||||
pushPoolMax: 2,
|
||||
maxConnections: 50,
|
||||
maintenanceAdminAllowance: 9,
|
||||
explicitReserve: 3
|
||||
})
|
||||
|
||||
assert.equal(report.budgetedTotal, 63)
|
||||
assert.equal(report.budgetedTotal, 65)
|
||||
assert.equal(report.withinBudget, false)
|
||||
})
|
||||
|
||||
|
||||
+65
-18
@@ -21,7 +21,8 @@ edit plus a second set of Apple credentials.
|
||||
| --- | --- | --- |
|
||||
| Cloud Run service | `orca-cloud-push` | `push_cloud_run_service_name` |
|
||||
| Region | `us-central1` | `region` |
|
||||
| Instances | min 1, max 4 | `push_min_instances`, `push_max_instances` |
|
||||
| Instances | min 1, max 2 | `push_min_instances`, `push_max_instances` |
|
||||
| Database pool | 2 per instance | `push_database_pool_max` |
|
||||
| Concurrency | 80 | `push_concurrency` |
|
||||
| Ingress | all | `INGRESS_TRAFFIC_ALL` |
|
||||
| Invoker | IAM disabled | `invoker_iam_disabled = true` on the service |
|
||||
@@ -29,8 +30,24 @@ edit plus a second set of Apple credentials.
|
||||
| Database | `orca_push` on the shared Cloud SQL instance | `google_sql_database.push` |
|
||||
| Hostname | `push.onorca.dev` | `push_base_url` |
|
||||
|
||||
The minimum of one instance is deliberate. A cold start delays a notification past the point
|
||||
where it is worth showing, and the three-second coalescing window lives in instance memory.
|
||||
The minimum of one instance is deliberate and did not move when the ceiling came down to two. A
|
||||
cold start delays a notification past the point where it is worth showing, and the three-second
|
||||
coalescing window lives in instance memory, so the floor is what keeps a notification prompt. The
|
||||
ceiling is a different question, answered below.
|
||||
|
||||
The maximum and the pool are set by the connection budget, not by the gateway's own appetite. Two
|
||||
instances times a two-connection pool is a draw of 4, and a rollout doubles it to 8, because the
|
||||
tagged candidate is directly addressable and sits outside the service-wide cap. The shared Cloud
|
||||
SQL instance's 400 connections were already spoken for by the relay cells, the directors, auth,
|
||||
and the API, which left five. Four is the whole of the room there was, and the gateway fits in
|
||||
it.
|
||||
|
||||
Two connections per instance is enough for the work. A send runs two or three short queries, so
|
||||
at concurrency 80 requests queue against the pool for microseconds rather than holding it. A
|
||||
`lifecycle` precondition refuses a plan whose instances times pool exceeds 4, because a fifth
|
||||
connection puts the checked budget over its ceiling and blocks `Deploy Relay Asia Topology`,
|
||||
which gates on it. `dev/scripts/relay-cloud-sql-connection-budget.mjs` counts the gateway and
|
||||
prints the whole picture.
|
||||
|
||||
Authentication is the host proof in `POST /v1/host/challenge`, not Cloud Run IAM, so the service
|
||||
opts out of invoker IAM with `invoker_iam_disabled = true`, exactly as the relay director does.
|
||||
@@ -47,6 +64,7 @@ Set on the container by Terraform:
|
||||
| `ORCA_PUSH_PUBLIC_URL` | `push_base_url` |
|
||||
| `ORCA_PUSH_FCM_PROJECT_ID` | `push_fcm_project_id`, empty means `project_id` |
|
||||
| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-database-url`, version `latest` |
|
||||
| `ORCA_PUSH_DATABASE_POOL_MAX` | `push_database_pool_max`, 2 per instance |
|
||||
| `ORCA_PUSH_APNS_KEY` | Secret `orca-cloud-push-apns-key`, version `latest` |
|
||||
| `ORCA_PUSH_APNS_KEY_ID` | Secret `orca-cloud-push-apns-key-id`, version `latest` |
|
||||
| `ORCA_PUSH_APPLE_TEAM_ID` | Secret `orca-cloud-push-apple-team-id`, version `latest` |
|
||||
@@ -130,28 +148,55 @@ It authenticates as the shared production deploy identity through
|
||||
`PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER` and
|
||||
`PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT`, which are already published. No new GitHub
|
||||
variable is required. That account was chosen because the Cloud SQL rollout lease grant is
|
||||
foundation-owned and already names it; a dedicated identity could not take that lease from this
|
||||
root. Its authority over the gateway is exactly three bindings in `push-gateway.tf`: Cloud Run
|
||||
developer on this one service, service-account user on the runtime account, and token creator on
|
||||
the runtime account.
|
||||
foundation-owned and names only that account; a dedicated identity could not take that lease from
|
||||
this root, and the gateway's schema rollout has to serialize against the relay's.
|
||||
|
||||
**That choice widens what this workflow can reach, and the widening is deliberate.** Adding
|
||||
`push-deploy.yml` to the provider allowlist gives the run the account's whole existing authority,
|
||||
not only the push bindings: Artifact Registry writer on `orca-cloud`, `roles/run.developer` on
|
||||
the relay director and the fence broker, accessor and version-adder on the relay
|
||||
regional-placement secret, and service-account user on the relay runtime identities. It was
|
||||
accepted as the price of the lease. What `push-gateway.tf` adds on top is three bindings scoped
|
||||
to the gateway alone: Cloud Run developer on this one service, and service-account user plus
|
||||
token creator on the runtime account. The bound on the rest is the provider condition, which
|
||||
admits this exact workflow file on `main` in the `production` environment only, and the workflow
|
||||
itself, which is dispatch-only behind a typed confirmation.
|
||||
|
||||
The run, in order:
|
||||
|
||||
1. Takes the production Cloud SQL rollout lease and holds it for the whole run. The gateway
|
||||
applies its schema while the new revision starts, so the revision **is** the schema step;
|
||||
there is no separate migration command to wrap.
|
||||
2. Builds `apps/push/Dockerfile` with the `cloud/` build context and pushes to the existing
|
||||
1. Builds `apps/push/Dockerfile` with the `cloud/` build context and pushes to the existing
|
||||
`orca-cloud` Artifact Registry repository as `push:sha-<commit>`, then resolves the digest.
|
||||
3. Records the currently serving revision as the rollback target.
|
||||
This happens **before** the lease is taken. Artifact Registry is not the Cloud SQL instance,
|
||||
and a multi-minute build inside the lease would block every relay deploy and rehome for its
|
||||
duration.
|
||||
2. Takes the production Cloud SQL rollout lease and holds it from here to the end. The gateway
|
||||
applies its schema while the new revision starts, so the revision **is** the schema step;
|
||||
there is no separate migration command to wrap. The lease therefore covers exactly the
|
||||
connection-budget window: deploy, probe, shift.
|
||||
3. Records the currently serving revision as the rollback target, and requires it to still hold
|
||||
the Terraform-owned floor and ceiling. The candidate inherits that scaling, so a drifted
|
||||
serving revision would be latched rather than corrected.
|
||||
4. `gcloud run deploy --no-traffic` with a per-run traffic tag, so the candidate boots and
|
||||
applies schema while every phone still reaches the previous revision.
|
||||
applies schema while every phone still reaches the previous revision. The deploy passes no
|
||||
scaling flag: the shape is Terraform's, and the candidate's inherited ceiling is asserted
|
||||
instead.
|
||||
5. Probes the tagged candidate's own `/ready`, up to 30 times at five-second intervals.
|
||||
6. Sends a validate-only FCM message as the runtime identity, by impersonation. See below.
|
||||
7. Shifts 100% of traffic to the candidate, verifies it is the only revision serving, and
|
||||
checks `https://push.onorca.dev/ready`.
|
||||
8. Always removes the traffic tag, so tags do not accumulate across runs.
|
||||
7. Shifts 100% of traffic to the candidate and verifies it is the only revision serving.
|
||||
8. Writes the run summary, including the rollback command, before checking the public origin, so
|
||||
the summary exists even when the check that follows does not pass.
|
||||
9. Checks `https://push.onorca.dev/ready`, up to 30 times at five-second intervals, since the
|
||||
origin can lag the traffic move by a few seconds.
|
||||
10. Always removes the traffic tag, so tags do not accumulate across runs.
|
||||
|
||||
Rolling back is a traffic move, printed in the run summary:
|
||||
**Failure after the shift rolls itself back.** Everything from step 8 on runs with production
|
||||
already on the candidate, so a failure there is not a failed deploy, it is a live gateway that
|
||||
has to go back. The run returns traffic to the recorded rollback revision, verifies the move, and
|
||||
reports it in the summary. A failure *before* the shift leaves production untouched and deletes
|
||||
the candidate revision, which otherwise sits holding a warm instance and a Cloud SQL pool for
|
||||
nothing.
|
||||
|
||||
To move traffic by hand, from the revision named in the run summary:
|
||||
|
||||
```sh
|
||||
gcloud run services update-traffic orca-cloud-push \
|
||||
@@ -167,7 +212,9 @@ mints an access token for `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com`
|
||||
`validate_only: true` with a token that cannot exist. `validate_only` stops Google before any
|
||||
delivery, and a healthy credential answers `INVALID_ARGUMENT` because the device token is
|
||||
garbage. `PERMISSION_DENIED`, `401`, and `403` are the failures the step exists to catch, and
|
||||
they fail the run before traffic moves. Probing as the deploy identity instead would prove
|
||||
they fail the run immediately, before traffic moves. Those four answers are the only conclusive
|
||||
ones: a `429`, a `5xx`, or a transport failure says nothing about the credential, so the send is
|
||||
retried up to five times at five-second intervals rather than read as either verdict. Probing as the deploy identity instead would prove
|
||||
something true about the wrong account.
|
||||
|
||||
## Rotating the APNs key
|
||||
|
||||
@@ -411,20 +411,31 @@ Artifact Registry repository, and its rollout lease.
|
||||
It needs **no new GitHub environment variable.** It authenticates as the shared production deploy
|
||||
identity through the already-published `PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER`
|
||||
and `PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT`, and reads `PRODUCTION_GCP_REGION` like the
|
||||
rest. That account holds the foundation-owned Cloud SQL rollout lease grant, which a dedicated
|
||||
identity could not be given from this root. Its authority over the gateway is three bindings in
|
||||
`infra/terraform/push-gateway.tf` and nothing wider: Cloud Run developer on that one service, and
|
||||
service-account user plus token creator on the gateway's runtime account.
|
||||
rest. That account holds the foundation-owned Cloud SQL rollout lease grant, which names it and nothing
|
||||
else, so a dedicated identity could not be given that lease from this root.
|
||||
|
||||
`infra/terraform/push-gateway.tf` adds three bindings scoped to the gateway: Cloud Run developer
|
||||
on that one service, and service-account user plus token creator on the gateway's runtime
|
||||
account. Those three are not the workflow's whole authority. Running as the shared account gives
|
||||
the run every role that account already holds for the relay: Artifact Registry writer on
|
||||
`orca-cloud`, `roles/run.developer` on the relay director and the fence broker, accessor and
|
||||
version-adder on the relay regional-placement secret, and service-account user on the relay
|
||||
runtime identities. That widening was accepted as the price of the lease, and it is bounded by
|
||||
the provider condition and by the workflow being dispatch-only behind a typed confirmation.
|
||||
|
||||
The provider's workflow allowlist gained exactly one entry, `cloud-push-deploy.yml`, on `main` in
|
||||
the `production` environment. That entry is required: the allowlist compares complete workflow
|
||||
refs by equality, so the `cloud-` filename prefix alone does not admit a new file.
|
||||
|
||||
The run takes the production rollout lease and holds it across the deploy, because the gateway
|
||||
applies its schema while the new revision starts. It builds `apps/push/Dockerfile`, deploys with
|
||||
`--no-traffic` behind a per-run traffic tag, probes the candidate's own `/ready`, proves the
|
||||
runtime identity can reach FCM with a validate-only send, and only then shifts 100% of traffic.
|
||||
There is no staging gateway, so there is no staging counterpart to run first.
|
||||
The run builds `apps/push/Dockerfile` **before** taking the lease, so an image build never blocks
|
||||
a relay deploy or rehome, then holds the production rollout lease across the deploy itself,
|
||||
because the gateway applies its schema while the new revision starts. Under the lease it checks
|
||||
the serving revision's Terraform-owned scaling, deploys with `--no-traffic` behind a per-run
|
||||
traffic tag and no scaling flag of its own, probes the candidate's own `/ready`, proves the
|
||||
runtime identity can reach FCM with a validate-only send, and only then shifts 100% of traffic. A
|
||||
failure after the shift returns traffic to the recorded rollback revision; a failure before it
|
||||
deletes the candidate. There is no staging gateway, so there is no staging counterpart to run
|
||||
first.
|
||||
|
||||
Full runbook, including the APNs key rotation and the DNS record the `stablyai/orca-cloud` apps
|
||||
root still owes, is in `docs/push-gateway.md`.
|
||||
|
||||
@@ -412,6 +412,9 @@ relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels
|
||||
# Mobile push gateway. Production is the only environment that runs one; the runtime account,
|
||||
# the three Apple secrets, and their accessor bindings already exist and are imported once
|
||||
# (see docs/push-gateway.md).
|
||||
push_gateway_enabled = true
|
||||
push_base_url = "https://push.onorca.dev"
|
||||
push_gateway_enabled = true
|
||||
push_base_url = "https://push.onorca.dev"
|
||||
# Sized so the gateway's rollout overlap, the retained director rollback plus its doubled draw,
|
||||
# stays under the API candidate's, which keeps the checked Cloud SQL connection budget green.
|
||||
push_max_instances = 2
|
||||
manage_push_domain_mapping = true
|
||||
|
||||
@@ -40,9 +40,10 @@ locals {
|
||||
|
||||
push_fqdn = replace(replace(var.push_base_url, "https://", ""), "http://", "")
|
||||
|
||||
# The shared production deploy identity runs `cloud-push-deploy.yml`. Its Cloud Run and
|
||||
# service-account grants are scoped to this service alone; the account itself is declared in
|
||||
# relay-github-actions.tf and is production-only.
|
||||
# The shared production deploy identity runs `cloud-push-deploy.yml`. The grants this file adds
|
||||
# are scoped to this service and its runtime account alone, but the workflow inherits every
|
||||
# other grant that account already holds for the relay; see the deploy-identity section below.
|
||||
# The account itself is declared in relay-github-actions.tf and is production-only.
|
||||
push_gateway_deploy_count = (
|
||||
var.push_gateway_enabled && local.relay_create_production_ops_identity ? 1 : 0
|
||||
)
|
||||
@@ -227,6 +228,14 @@ resource "google_cloud_run_v2_service" "push" {
|
||||
value = local.push_fcm_project_id
|
||||
}
|
||||
|
||||
# Declared rather than left to the application default, so the gateway's share of the
|
||||
# shared Cloud SQL connection budget is a value this root states and the precondition
|
||||
# below can bound.
|
||||
env {
|
||||
name = "ORCA_PUSH_DATABASE_POOL_MAX"
|
||||
value = tostring(var.push_database_pool_max)
|
||||
}
|
||||
|
||||
env {
|
||||
name = "ORCA_PUSH_DATABASE_URL"
|
||||
|
||||
@@ -284,6 +293,19 @@ resource "google_cloud_run_v2_service" "push" {
|
||||
# 100% LATEST would silently undo either, and this root carries unrelated standing drift, so
|
||||
# that apply need not be a push change at all.
|
||||
lifecycle {
|
||||
# Why: the gateway draws instances x pool from the shared Cloud SQL instance, and a rollout
|
||||
# doubles it, because the tagged candidate is directly addressable and sits outside the
|
||||
# service-wide cap. The instance's 400 connections were already spoken for by the relay
|
||||
# cells, directors, auth, and API, which left five: 4 is the whole of the gateway's share and
|
||||
# it fits, with the doubled 8 still under the API candidate's rollout overlap, the term
|
||||
# dev/scripts/relay-cloud-sql-connection-budget.mjs maximizes over. A fifth connection here
|
||||
# puts the checked budget over its ceiling and blocks Deploy Relay Asia Topology, which gates
|
||||
# on it, so catch a raise at plan time rather than in someone else's rollout.
|
||||
precondition {
|
||||
condition = var.push_max_instances * var.push_database_pool_max <= 4
|
||||
error_message = "Push gateway instances x database pool must stay within its 4-connection share of the shared Cloud SQL instance."
|
||||
}
|
||||
|
||||
ignore_changes = [
|
||||
client,
|
||||
client_version,
|
||||
@@ -327,8 +349,19 @@ resource "google_cloud_run_domain_mapping" "push" {
|
||||
|
||||
# --- Deploy identity grants -------------------------------------------------------------------
|
||||
# `cloud-push-deploy.yml` authenticates as the shared production deploy account, because that
|
||||
# account is the one the foundation root grants the Cloud SQL rollout lease to. These three
|
||||
# bindings are the whole of its authority over the push gateway.
|
||||
# account is the one the foundation root grants the Cloud SQL rollout lease to; the grant names
|
||||
# that account and nothing else, so a dedicated push identity could not take the lease from this
|
||||
# root and the gateway's schema rollout could not be serialized against the relay's.
|
||||
#
|
||||
# The three bindings below are the whole of that account's authority over the *push gateway*, but
|
||||
# they are not the whole of what the workflow can do. Adding `push-deploy.yml` to the provider's
|
||||
# allowlist in relay-github-actions.tf gives the run the account's entire existing authority:
|
||||
# Artifact Registry writer on `orca-cloud`, `roles/run.developer` on the relay director and the
|
||||
# fence broker, accessor and version-adder on the relay regional-placement secret, and
|
||||
# service-account user on the relay runtime identities. That widening was accepted deliberately
|
||||
# as the price of the lease. It is bounded by the provider condition, which admits this exact
|
||||
# workflow file on `main` in the `production` environment only, and by the workflow itself, which
|
||||
# is dispatch-only behind a typed confirmation.
|
||||
|
||||
resource "google_cloud_run_v2_service_iam_member" "github_production_push_developer" {
|
||||
count = local.push_gateway_deploy_count
|
||||
|
||||
@@ -21,8 +21,13 @@ locals {
|
||||
"operate-relay-asia-admission.yml",
|
||||
"publish-relay-production.yml",
|
||||
# The push gateway deploy runs as this account because the Cloud SQL rollout lease grant is
|
||||
# foundation-owned and already names it; a dedicated identity could not take that lease.
|
||||
# Its authority over the gateway is three bindings in push-gateway.tf and nothing more.
|
||||
# foundation-owned and names only this account; a dedicated identity could not take that
|
||||
# lease, and the gateway's schema rollout has to serialize against the relay's.
|
||||
#
|
||||
# This entry therefore grants that workflow every role the account already holds, not just
|
||||
# the three push bindings in push-gateway.tf: Artifact Registry writer, run.developer on the
|
||||
# relay director and fence broker, relay secret accessor and version-adder, and
|
||||
# serviceAccountUser on the relay runtime identities. Accepted as the price of the lease.
|
||||
"push-deploy.yml"
|
||||
]
|
||||
github_production_relay_capacity_workflow_file = "deploy-relay-production-capacity.yml"
|
||||
|
||||
@@ -548,6 +548,24 @@ variable "push_max_instances" {
|
||||
}
|
||||
}
|
||||
|
||||
# Why: the gateway's draw on the shared Cloud SQL instance is instances x pool, and the rollout
|
||||
# lease is taken for twice that, because a tagged candidate is directly addressable and sits
|
||||
# outside the service-wide cap. Leaving the pool at its application default made that draw
|
||||
# invisible to this root, so it is declared here and set on the container.
|
||||
#
|
||||
# Two is sized to the work, not to the default: a send runs two or three short queries, and at
|
||||
# concurrency 80 those queue against the pool for microseconds rather than holding it.
|
||||
variable "push_database_pool_max" {
|
||||
type = number
|
||||
description = "Push gateway database pool size per instance; instances x pool is its Cloud SQL draw."
|
||||
default = 2
|
||||
|
||||
validation {
|
||||
condition = var.push_database_pool_max >= 1 && var.push_database_pool_max <= 100
|
||||
error_message = "The push gateway pool must hold at least one connection and stay under the per-service bound."
|
||||
}
|
||||
}
|
||||
|
||||
variable "push_concurrency" {
|
||||
type = number
|
||||
description = "Cloud Run concurrency for short-lived push gateway HTTP requests."
|
||||
|
||||
Reference in New Issue
Block a user