id: autonomous-incident-triage-with-fallback
namespace: company.team
description: |
Autonomous incident triage workflow that ingests production alerts via webhook or schedule,
synthesizes root causes using multi-tier LLM failover from OpenAI to Gemini, assesses blast radius,
pauses for human engineer sign-off, and dispatches actionable Slack notifications.
triggers:
- id: alert_webhook
type: io.kestra.plugin.core.trigger.Webhook
description: Inbound webhook receiving alerts from PagerDuty, Sentry, or
monitoring systems.
key: "{{ secret('INCIDENT_WEBHOOK_KEY') }}"
- id: poll_unassigned_incidents
type: io.kestra.plugin.core.trigger.Schedule
description: Optional recurring poll to sweep for unassigned production incidents.
cron: "*/30 * * * *"
disabled: true
inputs:
- id: incident_id
type: STRING
description: Unique identifier for the incident ticket or alert event.
defaults: "INC-84920"
- id: service
type: STRING
description: Affected microservice or infrastructure component name.
defaults: "checkout-payment-api"
- id: severity
type: SELECT
values:
- CRITICAL
- HIGH
- MEDIUM
- LOW
description: Inbound alert severity level.
defaults: HIGH
- id: error_message
type: STRING
description: Primary error message or anomaly alert description.
defaults: "HTTP 504 Gateway Timeout spike exceeding 35% error rate on POST
/api/v2/charge across us-east-1"
- id: logs
type: STRING
description: Diagnostic logs, stack trace, or observability snippets associated
with the incident.
defaults: "2026-10-02T18:50:12Z ERROR [pool-payment-upstream] Connection pool
exhausted (active=100/100, pending=482). Latency p99 > 8500ms. Upstream
gateway db-replica-02 read timeouts."
- id: environment
type: STRING
description: Target deployment environment.
defaults: "production"
tasks:
- id: log_incident_payload
type: io.kestra.plugin.core.log.Log
description: Record the inbound incident alert details for auditability.
message: |
Incoming incident triage initiated:
- Incident ID: {{ trigger.body.incident_id ?? inputs.incident_id }}
- Service: {{ trigger.body.service ?? inputs.service }}
- Inbound Severity: {{ trigger.body.severity ?? inputs.severity }}
- Environment: {{ trigger.body.environment ?? inputs.environment }}
- Error: {{ trigger.body.error_message ?? inputs.error_message }}
- id: triage_incident_with_fallback
type: io.kestra.plugin.scripts.python.Script
description: Execute autonomous incident triage using primary OpenAI model with
seamless automatic fallback to Google Gemini and deterministic heuristics.
dependencies:
- kestra
- requests
env:
OPENAI_API_KEY: "{{ secret('OPENAI_API_KEY') }}"
GEMINI_API_KEY: "{{ secret('GEMINI_API_KEY') }}"
INCIDENT_ID: "{{ trigger.body.incident_id ?? inputs.incident_id }}"
SERVICE_NAME: "{{ trigger.body.service ?? inputs.service }}"
SEVERITY: "{{ trigger.body.severity ?? inputs.severity }}"
ERROR_MESSAGE: "{{ trigger.body.error_message ?? inputs.error_message }}"
INCIDENT_LOGS: "{{ trigger.body.logs ?? inputs.logs }}"
ENVIRONMENT: "{{ trigger.body.environment ?? inputs.environment }}"
script: |
import os
import json
import time
import requests
from kestra import Kestra
def clean_text(text):
if not text:
return ""
return str(text).replace('"', "'").replace("\n", " ").strip()
incident_id = os.environ.get("INCIDENT_ID", "INC-UNKNOWN")
service = os.environ.get("SERVICE_NAME", "unknown-service")
severity = os.environ.get("SEVERITY", "HIGH")
error_msg = os.environ.get("ERROR_MESSAGE", "")
logs = os.environ.get("INCIDENT_LOGS", "")
env_name = os.environ.get("ENVIRONMENT", "production")
openai_api_key = os.environ.get("OPENAI_API_KEY", "").strip()
gemini_api_key = os.environ.get("GEMINI_API_KEY", "").strip()
triage_prompt = f"""You are a Principal Site Reliability Engineer on-call. Analyze the following production incident:
Incident ID: {incident_id}
Service: {service}
Environment: {env_name}
Reported Severity: {severity}
Error Message: {error_msg}
Recent Logs:
{logs}
Provide a structured JSON triage response with the exact keys:
- "root_cause_hypothesis": concise analysis of likely root cause
- "blast_radius": estimated user impact, services affected, data loss risk
- "severity_level": "CRITICAL", "HIGH", "MEDIUM", or "LOW"
- "mitigation_steps": list of 3-5 concise actionable mitigation steps
- "requires_human_approval": boolean true if blast radius is severe or risk is high, false otherwise
- "summary": 2-3 sentence executive triage summary
"""
triage_result = None
provider_used = None
failover_occurred = False
error_log = []
# Tier 1: Primary provider - OpenAI
if openai_api_key:
print("Attempting primary LLM triage with OpenAI (gpt-4o-mini)...")
try:
start_time = time.time()
response = requests.post(
"https://api.openai.com/v1/chat/completions",
headers={
"Authorization": f"Bearer {openai_api_key}",
"Content-Type": "application/json"
},
json={
"model": "gpt-4o-mini",
"messages": [
{"role": "system", "content": "You are an automated incident triage engine. Respond ONLY with valid JSON."},
{"role": "user", "content": triage_prompt}
],
"temperature": 0.2,
"response_format": {"type": "json_object"}
},
timeout=12
)
if response.status_code == 200:
res_data = response.json()
raw_content = res_data["choices"][0]["message"]["content"]
triage_result = json.loads(raw_content)
provider_used = "OpenAI (gpt-4o-mini)"
latency = round((time.time() - start_time) * 1000, 2)
print(f"OpenAI triage succeeded in {latency}ms.")
else:
error_log.append(f"OpenAI HTTP {response.status_code}")
print(f"OpenAI returned non-200: {response.status_code}")
except Exception as e:
error_log.append(f"OpenAI exception: {str(e)}")
print(f"OpenAI call failed: {str(e)}")
else:
error_log.append("OPENAI_API_KEY not configured.")
# Tier 2: Secondary provider - Google Gemini Fallback
if not triage_result and gemini_api_key:
print("Failing over to secondary LLM provider: Google Gemini (gemini-1.5-flash)...")
failover_occurred = True
try:
start_time = time.time()
gemini_url = f"https://generativelanguage.googleapis.com/v1beta/models/gemini-1.5-flash:generateContent?key={gemini_api_key}"
response = requests.post(
gemini_url,
headers={"Content-Type": "application/json"},
json={
"contents": [
{
"parts": [
{"text": "You are an automated incident triage engine. Respond strictly with valid JSON conforming to keys: root_cause_hypothesis, blast_radius, severity_level, mitigation_steps, requires_human_approval, summary.\n\n" + triage_prompt}
]
}
],
"generationConfig": {
"temperature": 0.2,
"responseMimeType": "application/json"
}
},
timeout=12
)
if response.status_code == 200:
res_data = response.json()
raw_text = res_data["candidates"][0]["content"]["parts"][0]["text"]
triage_result = json.loads(raw_text)
provider_used = "Google Gemini (gemini-1.5-flash)"
latency = round((time.time() - start_time) * 1000, 2)
print(f"Gemini fallback triage succeeded in {latency}ms.")
else:
error_log.append(f"Gemini HTTP {response.status_code}")
print(f"Gemini returned non-200: {response.status_code}")
except Exception as e:
error_log.append(f"Gemini exception: {str(e)}")
print(f"Gemini call failed: {str(e)}")
# Tier 3: Deterministic Rule-Based Fallback Engine
if not triage_result:
print("Engaging Tier 3 deterministic heuristic fallback...")
failover_occurred = True
provider_used = "Deterministic Heuristic Engine (Offline Fallback)"
lower_context = (logs + " " + error_msg).lower()
if "connection pool" in lower_context or "exhausted" in lower_context:
hypothesis = f"Database connection pool exhaustion detected in dependencies for {service}."
radius = "Payment transactions failing; upstream database read replica thread pool saturated."
mitigations = [
"Increase upstream database connection pool max connections",
"Recycle degraded worker pods on checkout-payment-api",
"Enable read replica traffic shedding for non-critical query paths",
"Verify database slow query log and active lock count"
]
elif "timeout" in lower_context or "504" in lower_context:
hypothesis = f"Upstream service or gateway timeout degradation impacting {service}."
radius = "Inbound API latency spike affecting client requests and webhook delivery."
mitigations = [
"Verify health and response latency of downstream gateway dependencies",
"Enable circuit breaker to shed load to fallback cache",
"Drain unhealthy availability zone routing if regional",
"Scale horizontal pod autoscalers (HPA) to absorb queue build-up"
]
else:
hypothesis = f"Application runtime exception spike detected in {service}."
radius = "Degraded transaction throughput across production environment."
mitigations = [
"Review recent deployments within the last 60 minutes",
"Inspect pod memory and CPU utilization metrics",
"Rotate container replicas to clear potential deadlock states"
]
triage_result = {
"root_cause_hypothesis": hypothesis,
"blast_radius": radius,
"severity_level": severity.upper() if severity in ["CRITICAL", "HIGH", "MEDIUM", "LOW"] else "HIGH",
"mitigation_steps": mitigations,
"requires_human_approval": True,
"summary": f"Automated heuristic fallback diagnosed degradation in {service}. Mitigation runbook requires on-call verification."
}
steps_raw = triage_result.get("mitigation_steps", [])
if isinstance(steps_raw, list):
mitigation_formatted = " | ".join([f"Step {i+1}: {clean_text(s)}" for i, s in enumerate(steps_raw)])
else:
mitigation_formatted = clean_text(steps_raw)
requires_approval = bool(triage_result.get("requires_human_approval", True))
assessed_severity = str(triage_result.get("severity_level", severity)).upper()
if assessed_severity in ["CRITICAL", "HIGH"]:
requires_approval = True
Kestra.outputs({
"incident_id": incident_id,
"service": service,
"provider_used": provider_used,
"failover_occurred": failover_occurred,
"severity_level": assessed_severity,
"root_cause_hypothesis": clean_text(triage_result.get("root_cause_hypothesis", "N/A")),
"blast_radius": clean_text(triage_result.get("blast_radius", "N/A")),
"mitigation_steps": mitigation_formatted,
"summary": clean_text(triage_result.get("summary", "Triage completed.")),
"requires_human_approval": requires_approval,
"error_log": "; ".join(error_log) if error_log else "None"
})
- id: notify_triage_channel
type: io.kestra.plugin.notifications.slack.SlackIncomingWebhook
description: Dispatch AI triage report and blast radius analysis to the incident
Slack channel.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Incident Triage [{{ outputs.triage_incident_with_fallback.vars.incident_id }}] - {{ outputs.triage_incident_with_fallback.vars.severity_level }}: {{ outputs.triage_incident_with_fallback.vars.summary }}",
"blocks": [
{
"type": "header",
"text": {
"type": "plain_text",
"text": "Incident Triage: {{ outputs.triage_incident_with_fallback.vars.incident_id }}"
}
},
{
"type": "section",
"fields": [
{
"type": "mrkdwn",
"text": "*Service:*\n{{ outputs.triage_incident_with_fallback.vars.service }}"
},
{
"type": "mrkdwn",
"text": "*Severity:*\n{{ outputs.triage_incident_with_fallback.vars.severity_level }}"
},
{
"type": "mrkdwn",
"text": "*LLM Provider:*\n{{ outputs.triage_incident_with_fallback.vars.provider_used }}"
},
{
"type": "mrkdwn",
"text": "*Failover Active:*\n{{ outputs.triage_incident_with_fallback.vars.failover_occurred }}"
}
]
},
{
"type": "section",
"text": {
"type": "mrkdwn",
"text": "*Root Cause Hypothesis:*\n{{ outputs.triage_incident_with_fallback.vars.root_cause_hypothesis }}\n\n*Blast Radius:*\n{{ outputs.triage_incident_with_fallback.vars.blast_radius }}"
}
},
{
"type": "section",
"text": {
"type": "mrkdwn",
"text": "*Mitigation Runbook:*\n{{ outputs.triage_incident_with_fallback.vars.mitigation_steps }}\n\n*Summary:*\n{{ outputs.triage_incident_with_fallback.vars.summary }}"
}
}
]
}
- id: evaluate_approval_gate
type: io.kestra.plugin.core.flow.If
description: Pause execution for human on-call engineer sign-off if severity is
high or critical, or if LLM triage mandates approval.
condition: "{{
outputs.triage_incident_with_fallback.vars.requires_human_approval == true
or (trigger.body.severity ?? inputs.severity) == 'CRITICAL' or
(trigger.body.severity ?? inputs.severity) == 'HIGH' }}"
then:
- id: notify_approval_requested
type: io.kestra.plugin.notifications.slack.SlackIncomingWebhook
description: Request human confirmation in Slack before initiating automated
remediation.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "ACTION REQUIRED: Human approval needed for incident {{ outputs.triage_incident_with_fallback.vars.incident_id }} (Execution: {{ execution.id }}). Resume execution in Kestra UI to authorize mitigation runbook."
}
- id: pause_for_human_approval
type: io.kestra.plugin.core.flow.Pause
description: Hold execution awaiting on-call engineer approval and remediation
action confirmation.
pauseDuration: PT1H
onResume:
- id: approved
type: BOOL
defaults: false
description: Set to true to authorize automated mitigation runbooks.
- id: approver_email
type: STRING
defaults: "oncall-lead@company.team"
description: Email address of the authorizing responder.
- id: remediation_action
type: STRING
defaults: "restart-upstream-pool"
description: Confirmed runbook remediation action.
- id: execute_mitigation_decision
type: io.kestra.plugin.core.flow.If
description: Branch on human approval decision.
condition: "{{ (outputs.pause_for_human_approval.onResume is defined) and
outputs.pause_for_human_approval.onResume.approved }}"
then:
- id: log_mitigation_approved
type: io.kestra.plugin.core.log.Log
description: Record approved mitigation execution details.
message: "Remediation action [{{
outputs.pause_for_human_approval.onResume.remediation_action }}]
authorized by {{
outputs.pause_for_human_approval.onResume.approver_email }} for
incident {{ outputs.triage_incident_with_fallback.vars.incident_id
}}."
- id: notify_mitigation_commenced
type: io.kestra.plugin.notifications.slack.SlackIncomingWebhook
description: Notify Slack channel that remediation action has commenced.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Mitigation commenced for {{ outputs.triage_incident_with_fallback.vars.incident_id }}: Action [{{ outputs.pause_for_human_approval.onResume.remediation_action }}] approved by {{ outputs.pause_for_human_approval.onResume.approver_email }}."
}
else:
- id: log_mitigation_rejected
type: io.kestra.plugin.core.log.Log
description: Log that human approver declined or timed out the mitigation
action.
message: "Incident mitigation was not approved. Standing down automated
intervention for {{
outputs.triage_incident_with_fallback.vars.incident_id }}."
- id: record_triage_audit
type: io.kestra.plugin.core.log.Log
description: Record final audit log entry summarizing triage execution and
provider routing.
message: |
Incident triage workflow execution {{ execution.id }} completed.
Provider utilized: {{ outputs.triage_incident_with_fallback.vars.provider_used }}
Severity: {{ outputs.triage_incident_with_fallback.vars.severity_level }}
Failover triggered: {{ outputs.triage_incident_with_fallback.vars.failover_occurred }}
errors:
- id: alert_workflow_failure
type: io.kestra.plugin.notifications.slack.SlackIncomingWebhook
description: Alert the SRE on-call channel if the incident triage pipeline fails
unexpectedly.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "CRITICAL: Incident triage workflow failed in execution {{ execution.id }} on namespace {{ flow.namespace }}. Check execution logs in Kestra console."
}