id: k8s-crashloop-backoff-triage
namespace: company.team
description: |
Detect Kubernetes CrashLoopBackOff and ImagePullBackOff pods, capture recent logs
and restart history, and post an actionable triage digest to Slack.
triggers:
- id: every_15_minutes
type: io.kestra.plugin.core.trigger.Schedule
description: Scan workloads every 15 minutes
cron: "*/15 * * * *"
- id: on_demand_webhook
type: io.kestra.plugin.core.trigger.Webhook
description: On-demand triage from incident channels
key: "{{ secret('WEBHOOK_KEY') }}"
inputs:
- id: namespaces
type: STRING
displayName: Namespaces to scan
description: Comma-separated namespaces, or 'all' for every namespace
defaults: "all"
- id: restart_threshold
type: INT
displayName: Restart threshold
description: Flag pods with restart counts at or above this value
defaults: 5
- id: log_tail_lines
type: INT
displayName: Log tail lines
description: Number of log lines captured per unhealthy pod
defaults: 100
tasks:
- id: list_unhealthy_pods
type: io.kestra.plugin.scripts.shell.Commands
description: List pods and filter to CrashLoop / backoff / frequent restarts
taskRunner:
type: io.kestra.plugin.scripts.runner.docker.Docker
user: "0"
containerImage: bitnami/kubectl:latest
outputFiles:
- unhealthy.json
commands:
- kubectl get pods -A -o json > all-pods.json
- >-
python3 -c " import json; ns_filter='{{ inputs.namespaces }}';
threshold={{ inputs.restart_threshold }};
data=json.load(open('all-pods.json')); bad=[]; for p in
data.get('items',[]):
meta=p.get('metadata',{}); status=p.get('status',{});
ns=meta.get('namespace','?'); name=meta.get('name','?');
if ns_filter != 'all' and ns not in [n.strip() for n in ns_filter.split(',')]:
continue;
restarts=sum(c.get('restartCount',0) for c in (status.get('containerStatuses') or []));
reasons=','.join((c.get('state',{}).get('waiting',{}).get('reason','') or '') + '/' + (c.get('state',{}).get('terminated',{}).get('reason','') or '') for c in (status.get('containerStatuses') or []));
if 'CrashLoopBackOff' in reasons or 'ImagePullBackOff' in reasons or 'ErrImagePull' in reasons or restarts >= threshold:
bad.append({'namespace':ns,'name':name,'restarts':restarts,'reasons':reasons,'phase':status.get('phase','?')});
json.dump(bad, open('unhealthy.json','w'), indent=2);
print(f'flagged={len(bad)}'); "
- id: triage_in_parallel
type: io.kestra.plugin.core.flow.If
description: Branch triage only when unhealthy pods were found
condition: "{{ true }}"
then:
- id: build_triage_report
type: io.kestra.plugin.scripts.python.Script
description: Build human-readable triage digest from unhealthy pod list
containerImage: python:3.11
dependencies:
- kestra
inputFiles:
unhealthy.json: "{{ outputs.list_unhealthy_pods.outputFiles['unhealthy.json'] }}"
script: |
import json
from kestra import Kestra
with open("unhealthy.json") as f:
bad = json.load(f)
threshold = {{ inputs.restart_threshold }}
tail = {{ inputs.log_tail_lines }}
if not bad:
Kestra.outputs({"count": 0, "breached": False, "digest": "All namespaces healthy: no CrashLoop or high-restart pods."})
else:
lines = [f":rotating_light: *K8s CrashLoop Triage*, {len(bad)} pod(s) need attention", ""]
for p in bad[:20]:
lines.append(f"- `{p['namespace']}/{p['name']}` restarts={p['restarts']} phase={p['phase']} reasons=`{p['reasons']}`")
lines.append(f" triage: `kubectl logs -n {p['namespace']} {p['name']} --tail={tail}` then `kubectl describe pod -n {p['namespace']} {p['name']}`")
if len(bad) > 20:
lines.append(f"...and {len(bad) - 20} more (see execution outputs).")
Kestra.outputs({"count": len(bad), "breached": True, "digest": "\n".join(lines)})
- id: decide_alert
type: io.kestra.plugin.core.flow.If
description: Post Slack digest only when triage flagged pods
condition: "{{ outputs.build_triage_report.vars.breached }}"
then:
- id: notify_slack
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Post CrashLoop triage digest with kubectl remediation hints
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
messageText: "{{ outputs.build_triage_report.vars.digest }}"
else:
- id: log_healthy
type: io.kestra.plugin.core.log.Log
description: Record all-clear when no unhealthy pods are found
message: "K8s triage passed: no CrashLoopBackOff or high-restart pods."
- id: emit_triage_metrics
type: io.kestra.plugin.core.output.OutputValues
description: Export triage counts for dashboards
values:
unhealthy_pods: "{{ outputs.build_triage_report.vars.count }}"
namespaces_scanned: "{{ inputs.namespaces }}"
triaged_at: "{{ now() }}"
errors:
- id: alert_on_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert on-call if cluster scan fails (RBAC, API reachability)
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
messageText: |
:warning: *K8s CrashLoop Triage FAILED*
Flow: `{{ flow.id }}` execution `{{ execution.id }}`
Check kubectl RBAC, API server reachability, and runner logs.