Increase Cloud Deploy analysis step durations plus other QoL changes (#3230)

- Adjusts SLA analysis soak durations to 30 minutes per phase for sandbox and 1 hour per phase for production.
- Updates non-prod automations to auto-advance through canary-1 and canary-5 in crash and sandbox.
- Adds an automation rule to auto-promote releases from crash to sandbox (still need to manually approve the sandbox release).
- Adds documentation in skaffold.yaml clarifying dual deployment.
This commit is contained in:
Juan Celhay
2026-09-10 20:40:41 +00:00
committed by GitHub
parent d272288ac1
commit b6397ae2e4
2 changed files with 42 additions and 14 deletions
+30 -14
View File
@@ -99,8 +99,8 @@ serialPipeline:
gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-schema-verify-${TARGET_ID}.yaml . gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-schema-verify-${TARGET_ID}.yaml .
gcloud builds submit --no-source --config=cloudbuild-schema-verify-${TARGET_ID}.yaml gcloud builds submit --no-source --config=cloudbuild-schema-verify-${TARGET_ID}.yaml
analysis: analysis:
# 10 minutes. # 30 minutes.
duration: 600s duration: 1800s
googleCloud: googleCloud:
alertPolicyChecks: alertPolicyChecks:
sandboxPartialDeploymentAlertPolicyChecks sandboxPartialDeploymentAlertPolicyChecks
@@ -108,8 +108,8 @@ serialPipeline:
profiles: ["sandbox-partial-phase-5"] profiles: ["sandbox-partial-phase-5"]
percentage: 50 percentage: 50
analysis: analysis:
# 10 minutes. # 30 minutes.
duration: 600s duration: 1800s
googleCloud: googleCloud:
alertPolicyChecks: alertPolicyChecks:
sandboxPartialDeploymentAlertPolicyChecks sandboxPartialDeploymentAlertPolicyChecks
@@ -155,8 +155,8 @@ serialPipeline:
gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-deploy-gke-${TARGET_ID}.yaml . gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-deploy-gke-${TARGET_ID}.yaml .
gcloud builds submit --no-source --config=cloudbuild-deploy-gke-${TARGET_ID}.yaml gcloud builds submit --no-source --config=cloudbuild-deploy-gke-${TARGET_ID}.yaml
analysis: analysis:
# 10 minutes. # 30 minutes.
duration: 600s duration: 1800s
googleCloud: googleCloud:
alertPolicyChecks: alertPolicyChecks:
sandboxStableDeploymentAlertPolicyChecks sandboxStableDeploymentAlertPolicyChecks
@@ -183,8 +183,8 @@ serialPipeline:
gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-schema-verify-${TARGET_ID}.yaml . gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-schema-verify-${TARGET_ID}.yaml .
gcloud builds submit --no-source --config=cloudbuild-schema-verify-${TARGET_ID}.yaml gcloud builds submit --no-source --config=cloudbuild-schema-verify-${TARGET_ID}.yaml
analysis: analysis:
# 10 minutes. # 1 hour.
duration: 600s duration: 3600s
googleCloud: googleCloud:
alertPolicyChecks: alertPolicyChecks:
productionPartialDeploymentAlertPolicyChecks productionPartialDeploymentAlertPolicyChecks
@@ -192,8 +192,8 @@ serialPipeline:
profiles: ["production-partial-phase-5"] profiles: ["production-partial-phase-5"]
percentage: 50 percentage: 50
analysis: analysis:
# 10 minutes. # 1 hour.
duration: 600s duration: 3600s
googleCloud: googleCloud:
alertPolicyChecks: alertPolicyChecks:
productionPartialDeploymentAlertPolicyChecks productionPartialDeploymentAlertPolicyChecks
@@ -250,8 +250,8 @@ serialPipeline:
gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-deploy-gke-${TARGET_ID}.yaml . gcloud storage cp gs://${PROJECT_ID}-deploy/${TAG_NAME}/cloudbuild-deploy-gke-${TARGET_ID}.yaml .
gcloud builds submit --no-source --config=cloudbuild-deploy-gke-${TARGET_ID}.yaml gcloud builds submit --no-source --config=cloudbuild-deploy-gke-${TARGET_ID}.yaml
analysis: analysis:
# 10 minutes. # 1 hour.
duration: 600s duration: 3600s
googleCloud: googleCloud:
alertPolicyChecks: alertPolicyChecks:
productionStableDeploymentAlertPolicyChecks productionStableDeploymentAlertPolicyChecks
@@ -260,18 +260,34 @@ apiVersion: deploy.cloud.google.com/v1
kind: Automation kind: Automation
metadata: metadata:
name: deploy-nomulus/auto-advance-canary name: deploy-nomulus/auto-advance-canary
description: Automatically advances rollouts through canary-1 phase after successful deployment and analysis. description: Automatically advances rollouts through canary-1 and canary-5 phases in non-prod.
# Placeholder: Replace with project service account. # Placeholder: Replace with project service account.
serviceAccount: serviceAccount serviceAccount: serviceAccount
selector: selector:
targets: targets:
- id: crash - id: crash
- id: sandbox - id: sandbox
- id: production
rules: rules:
- advanceRolloutRule: - advanceRolloutRule:
id: advance-canary-phases id: advance-canary-phases
sourcePhases: sourcePhases:
- "canary-1" - "canary-1"
- "canary-5"
wait: 0m
---
apiVersion: deploy.cloud.google.com/v1
kind: Automation
metadata:
name: deploy-nomulus/auto-promote-crash-to-sandbox
description: Automatically promotes release to sandbox after successful rollout in crash.
# Placeholder: Replace with project service account.
serviceAccount: serviceAccount
selector:
targets:
- id: crash
rules:
- promoteReleaseRule:
id: promote-crash-to-sandbox
destinationTargetId: "sandbox"
wait: 0m wait: 0m
+12
View File
@@ -5,6 +5,10 @@ metadata:
name: nomulus-skaffold name: nomulus-skaffold
profiles: profiles:
- name: crash - name: crash
# Note: The stable profile intentionally includes partial-phase-5 manifests alongside
# primary workloads. This ensures that the candidate image is deployed across both deployments,
# maintaining stable pod and GKE node allocation to prevent churn on nodes hosting persistent
# registrar sessions.
manifests: manifests:
rawYaml: rawYaml:
- ../../jetty/kubernetes/nomulus-crash-backend.yaml - ../../jetty/kubernetes/nomulus-crash-backend.yaml
@@ -40,6 +44,10 @@ profiles:
deploy: deploy:
kubectl: { } kubectl: { }
- name: sandbox - name: sandbox
# Note: The stable profile intentionally includes partial-phase-5 manifests alongside
# primary workloads. This ensures that the candidate image is deployed across both deployments,
# maintaining stable pod and GKE node allocation to prevent churn on nodes hosting persistent
# registrar sessions.
manifests: manifests:
rawYaml: rawYaml:
- ../../jetty/kubernetes/nomulus-sandbox-backend.yaml - ../../jetty/kubernetes/nomulus-sandbox-backend.yaml
@@ -75,6 +83,10 @@ profiles:
deploy: deploy:
kubectl: { } kubectl: { }
- name: production - name: production
# Note: The stable profile intentionally includes partial-phase-5 manifests alongside
# primary workloads. This ensures that the candidate image is deployed across both deployments,
# maintaining stable pod and GKE node allocation to prevent churn on nodes hosting persistent
# registrar sessions.
manifests: manifests:
rawYaml: rawYaml:
- ../../jetty/kubernetes/nomulus-production-backend.yaml - ../../jetty/kubernetes/nomulus-production-backend.yaml