id: chaos-game-day-orchestrator
namespace: company.team
description: |
Run a controlled pod-loss game day against a Kubernetes deployment: capture
an error-rate baseline, scale the target down one replica as the fault, watch
error rate and ready replicas in parallel until the system stabilizes, then
restore capacity and publish an evidence report with the observed blast
radius - all from one auditable execution.
triggers:
- id: game_day
type: io.kestra.plugin.core.trigger.Webhook
description: Fire the drill from CI, a dashboard button, or curl.
- id: weekly_drill
type: io.kestra.plugin.core.trigger.Schedule
description: Recurring Saturday resilience drill (enable by flipping disabled).
cron: "0 10 * * 6"
disabled: true
inputs:
- id: target_namespace
type: STRING
displayName: Target namespace
description: Namespace of the deployment under test.
defaults: chaos-lab
- id: target_deployment
type: STRING
displayName: Target deployment
description: Deployment that plays the victim for this drill.
defaults: chaos-victim
- id: replicas
type: INT
displayName: Steady-state replicas
description: Healthy replica count; the drill scales down to replicas minus one
and restores this value.
defaults: 3
- id: error_budget_pct
type: FLOAT
displayName: Error budget (5xx ratio)
description: Maximum acceptable 5xx ratio while the system absorbs the fault.
defaults: 0.05
- id: max_watch
type: STRING
displayName: Max stabilization time
description: ISO-8601 ceiling for the blast-radius watch before the drill fails.
defaults: PT10M
- id: dry_run
type: BOOL
displayName: Dry run
description: Skip fault injection and only capture baseline plus readiness checks.
defaults: false
tasks:
- id: capture_baseline
type: io.kestra.plugin.prometheus.Query
description: Record the pre-fault 5xx ratio so the evidence report has a
comparison point.
url: "{{ secret('PROMETHEUS_URL') }}"
query: 'sum(rate(http_requests_total{namespace="{{ inputs.target_namespace }}",
status=~"5.."}[3m])) / sum(rate(http_requests_total{namespace="{{
inputs.target_namespace }}"}[3m]))'
fetchType: FETCH_ONE
- id: inject_gate
type: io.kestra.plugin.core.flow.If
description: Skip fault injection on dry runs so observers can rehearse the
drill safely.
condition: "{{ inputs.dry_run == false }}"
then:
- id: scale_down
type: io.kestra.plugin.kubernetes.kubectl.Patch
description: Drop one replica - the abrupt loss the system must absorb.
connection:
masterUrl: "{{ secret('K8S_MASTER_URL') }}"
oauthToken: "{{ secret('K8S_TOKEN') }}"
trustCerts: true
namespace: "{{ inputs.target_namespace }}"
resourceType: deployment
resourceName: "{{ inputs.target_deployment }}"
patchStrategy: JSON_MERGE
patch: |
{"spec": {"replicas": {{ inputs.replicas - 1 }}}}
else:
- id: log_dry_run
type: io.kestra.plugin.core.log.Log
description: Dry run announced - baseline and watches still execute, no cluster
change.
message: "Dry run for {{ inputs.target_deployment }}: skipping scale-down fault
injection."
- id: blast_radius_watch
type: io.kestra.plugin.core.flow.Parallel
description: Watch both sides of the blast radius concurrently - client-visible
errors and backend readiness.
tasks:
- id: watch_error_rate
type: io.kestra.plugin.core.flow.LoopUntil
description: Poll the 5xx ratio every 15 seconds until it sits inside the error
budget again.
condition: "{{ (outputs.probe_error_rate.metric.value | number) <=
inputs.error_budget_pct }}"
checkFrequency:
interval: PT15S
maxDuration: "{{ inputs.max_watch }}"
failOnMaxReached: true
tasks:
- id: probe_error_rate
type: io.kestra.plugin.prometheus.Query
description: Current 5xx ratio sample used for the stabilization check.
url: "{{ secret('PROMETHEUS_URL') }}"
query: 'sum(rate(http_requests_total{namespace="{{ inputs.target_namespace }}",
status=~"5.."}[2m])) / sum(rate(http_requests_total{namespace="{{
inputs.target_namespace }}"}[2m]))'
fetchType: FETCH_ONE
- id: watch_ready_replicas
type: io.kestra.plugin.core.flow.LoopUntil
description: Poll ready replica count until the deployment is back to its
degraded-but-serving size.
condition: "{{ (outputs.probe_ready.metric.value | number) >= (inputs.replicas -
1) }}"
checkFrequency:
interval: PT15S
maxDuration: "{{ inputs.max_watch }}"
failOnMaxReached: true
tasks:
- id: probe_ready
type: io.kestra.plugin.prometheus.Query
description: Ready replica count for the target deployment.
url: "{{ secret('PROMETHEUS_URL') }}"
query: 'kube_deployment_status_replicas_ready{deployment="{{
inputs.target_deployment }}", namespace="{{
inputs.target_namespace }}"}'
fetchType: FETCH_ONE
- id: restore_capacity
type: io.kestra.plugin.kubernetes.kubectl.Patch
description: Put the deployment back to its steady-state replica count now that
it stabilized.
connection:
masterUrl: "{{ secret('K8S_MASTER_URL') }}"
oauthToken: "{{ secret('K8S_TOKEN') }}"
trustCerts: true
namespace: "{{ inputs.target_namespace }}"
resourceType: deployment
resourceName: "{{ inputs.target_deployment }}"
patchStrategy: JSON_MERGE
patch: |
{"spec": {"replicas": {{ inputs.replicas }}}}
- id: sample_recovery
type: io.kestra.plugin.prometheus.Query
description: Post-restore 5xx ratio for the evidence report.
url: "{{ secret('PROMETHEUS_URL') }}"
query: 'sum(rate(http_requests_total{namespace="{{ inputs.target_namespace }}",
status=~"5.."}[3m])) / sum(rate(http_requests_total{namespace="{{
inputs.target_namespace }}"}[3m]))'
fetchType: FETCH_ONE
- id: sample_ready
type: io.kestra.plugin.prometheus.Query
description: Post-restore ready replica count to prove full recovery.
url: "{{ secret('PROMETHEUS_URL') }}"
query: 'kube_deployment_status_replicas_ready{deployment="{{
inputs.target_deployment }}", namespace="{{ inputs.target_namespace }}"}'
fetchType: FETCH_ONE
- id: write_evidence
type: io.kestra.plugin.scripts.python.Script
description: Compute the verdict and the blast radius actually observed from
baseline and recovery samples.
env:
BASELINE_ERR: "{{ outputs.capture_baseline.metric.value | number }}"
RECOVERY_ERR: "{{ outputs.sample_recovery.metric.value | number }}"
RECOVERY_READY: "{{ outputs.sample_ready.metric.value | number }}"
ERROR_BUDGET: "{{ inputs.error_budget_pct }}"
REPLICAS: "{{ inputs.replicas }}"
DRY_RUN: "{{ inputs.dry_run }}"
DEPLOYMENT: "{{ inputs.target_deployment }}"
NAMESPACE: "{{ inputs.target_namespace }}"
script: |
import json
import os
def num(name, default=0.0):
try:
return float(os.environ.get(name) or default)
except Exception:
return default
baseline = num("BASELINE_ERR")
recovery = num("RECOVERY_ERR")
ready = num("RECOVERY_READY")
budget = num("ERROR_BUDGET", 0.05)
replicas = int(num("REPLICAS", 1))
dry_run = str(os.environ.get("DRY_RUN") or "false").lower() == "true"
deployment = os.environ.get("DEPLOYMENT") or "unknown"
namespace = os.environ.get("NAMESPACE") or "unknown"
within_budget = recovery <= budget
ready_restored = ready >= replicas
if within_budget and ready_restored:
verdict = "STABLE"
elif ready_restored:
verdict = "DEGRADED"
else:
verdict = "NOT_RESTORED"
print(
"game day {}: baseline={} recovery={} budget={} ready={}/{}".format(
verdict, round(baseline, 5), round(recovery, 5), budget, int(ready), replicas
)
)
print(
"::"
+ json.dumps(
{
"outputs": {
"verdict": verdict,
"baseline_error_rate": round(baseline, 5),
"recovery_error_rate": round(recovery, 5),
"error_budget": budget,
"recovery_ready_replicas": int(ready),
"expected_replicas": replicas,
"headroom": round(budget - recovery, 5),
"fault_injected": not dry_run,
"target": "{}/{}".format(namespace, deployment),
}
}
)
+ "::"
)
- id: announce_evidence
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Publish the evidence summary so the drill result is visible without
opening Kestra.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
messageText: |
*Chaos game day complete* - {{ outputs.write_evidence.vars.target }}
• Verdict: {{ outputs.write_evidence.vars.verdict }}
• Fault injected: {{ outputs.write_evidence.vars.fault_injected }}
• 5xx ratio baseline -> recovery: {{ outputs.write_evidence.vars.baseline_error_rate }} -> {{ outputs.write_evidence.vars.recovery_error_rate }} (budget {{ outputs.write_evidence.vars.error_budget }})
• Ready replicas after restore: {{ outputs.write_evidence.vars.recovery_ready_replicas }}/{{ outputs.write_evidence.vars.expected_replicas }}
• Evidence execution: {{ execution.id }}
outputs:
- id: verdict
type: STRING
description: STABLE, DEGRADED, or NOT_RESTORED after the drill.
value: "{{ outputs.write_evidence.vars.verdict }}"
- id: baseline_error_rate
type: FLOAT
description: Pre-fault 5xx ratio.
value: "{{ outputs.write_evidence.vars.baseline_error_rate }}"
- id: recovery_error_rate
type: FLOAT
description: Post-restore 5xx ratio.
value: "{{ outputs.write_evidence.vars.recovery_error_rate }}"
- id: headroom
type: FLOAT
description: Error budget remaining after recovery; negative means the budget
was breached.
value: "{{ outputs.write_evidence.vars.headroom }}"
- id: fault_injected
type: BOOL
description: Whether the scale-down fault actually ran (false on dry runs).
value: "{{ outputs.write_evidence.vars.fault_injected }}"
errors:
- id: alert_on_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the drill itself fails - a timeout means the system did
not stabilize in time.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
messageText: "Chaos game day FAILED in flow {{ flow.id }} (execution {{
execution.id }}) - the blast-radius watch likely hit maxDuration or the
cluster rejected the patch. Check execution logs before assuming
recovery."