id: ai-incident-postmortem-generator
namespace: company.sre
description: |
Automates blameless incident post-mortem generation by ingesting alert timelines,
error logs, and engineer Slack updates, using Google Gemini to reconstruct the failure chain,
conduct 5 Whys root-cause analysis, and publish markdown incident reports.
triggers:
- id: scheduled_sweep
type: io.kestra.plugin.core.trigger.Schedule
description: Weekly post-mortem review sweep for closed operational incidents.
Shipped disabled by default.
cron: "0 9 * * 1"
disabled: true
- id: webhook_incident_resolved
type: io.kestra.plugin.core.trigger.Webhook
description: Authenticated webhook for incident management platforms (PagerDuty,
Opsgenie, Datadog) upon resolution.
key: "{{ secret('WEBHOOK_KEY') }}"
inputs:
- id: incident_id
type: STRING
defaults: INC-2026-804
description: Unique incident tracking identifier.
- id: service_name
type: STRING
defaults: payments-gateway-api
description: Primary impacted service or infrastructure component.
- id: severity
type: STRING
defaults: SEV-1
description: Incident severity classification (SEV-0, SEV-1, SEV-2, SEV-3).
- id: incident_timeline_log
type: STRING
defaults: |
14:15 UTC - PagerDuty Alert: HTTP 504 Gateway Timeout rate spiked to 38% on checkout endpoints.
14:18 UTC - On-call SRE acknowledged incident and joined war room channel #incident-804.
14:22 UTC - Database telemetry indicated connection pool exhaustion (max_connections=500 reached) on primary PostgreSQL cluster.
14:28 UTC - SRE identified an unindexed table migration query holding an ExclusiveLock on the orders table.
14:35 UTC - SRE terminated the blocking transaction (PID 49102) and recycled API gateway connection pools.
14:42 UTC - Error rates dropped to 0.01%, latency normalized to p95 < 45ms. Incident marked resolved.
description: Raw chronological timeline notes, alert timestamps, and diagnostic
observations.
tasks:
- id: analyze_incident_timeline
type: io.kestra.plugin.ai.agent.AIAgent
description: Analyzes incident logs with Gemini 2.5 Flash to synthesize the
timeline, calculate MTTR, conduct 5 Whys analysis, and draft preventative
action items.
provider:
type: io.kestra.plugin.ai.provider.GoogleGemini
apiKey: "{{ secret('GEMINI_API_KEY') }}"
modelName: gemini-2.5-flash
configuration:
temperature: 0.2
maxToken: 4096
responseFormat:
type: JSON
jsonSchema:
type: object
required:
- incident_id
- service_name
- severity
- time_to_detect_minutes
- time_to_resolve_minutes
- root_cause_summary
- five_whys
- preventative_action_items
- executive_summary
properties:
incident_id:
type: string
service_name:
type: string
severity:
type: string
time_to_detect_minutes:
type: integer
time_to_resolve_minutes:
type: integer
root_cause_summary:
type: string
five_whys:
type: array
items:
type: string
preventative_action_items:
type: array
items:
type: object
required:
- action_id
- description
- priority
- owner_team
properties:
action_id:
type: string
description:
type: string
priority:
type: string
owner_team:
type: string
executive_summary:
type: string
prompt: |
You are an expert Site Reliability Engineer (SRE) post-mortem facilitator.
Analyze the following production incident timeline notes and synthesize a blameless post-mortem report:
Incident ID: {{ inputs.incident_id }}
Service: {{ inputs.service_name }}
Severity: {{ inputs.severity }}
Timeline Log:
\"\"\"
{{ inputs.incident_timeline_log }}
\"\"\"
Requirements:
1. Calculate Time to Detect (TTD) and Time to Resolve (TTR) from initial impact to resolution.
2. Construct a rigorous, blameless "5 Whys" root-cause chain explaining how the trigger cascaded to service outage.
3. Propose actionable, concrete engineering preventative measures (e.g. DDL timeout guards, pool isolation, alert thresholds).
4. Synthesize a concise 2-sentence executive summary suitable for engineering leadership.
- id: render_postmortem_markdown
type: io.kestra.plugin.scripts.python.Script
description: Formats structured AI analysis into an industry-standard blameless
post-mortem markdown artifact.
taskRunner:
type: io.kestra.plugin.core.runner.Process
script: |
import json
raw_payload = """{{ outputs.analyze_incident_timeline.text }}"""
data = json.loads(raw_payload)
inc_id = data.get("incident_id", "{{ inputs.incident_id }}")
service = data.get("service_name", "{{ inputs.service_name }}")
sev = data.get("severity", "{{ inputs.severity }}")
ttd = data.get("time_to_detect_minutes", 0)
ttr = data.get("time_to_resolve_minutes", 0)
summary = data.get("executive_summary", "")
root_cause = data.get("root_cause_summary", "")
whys = "\n".join([f"{i+1}. {why}" for i, why in enumerate(data.get("five_whys", []))])
actions_rows = []
for a in data.get("preventative_action_items", []):
actions_rows.append(f"| {a.get('action_id')} | {a.get('description')} | {a.get('priority')} | {a.get('owner_team')} |")
actions_table = "\n".join(actions_rows) or "| N/A | None defined | Low | SRE |"
markdown = f"""# BLAMELESS POST-MORTEM: {inc_id}
## Executive Overview
- **Impacted Service**: {service}
- **Severity**: {sev}
- **Time to Detect (TTD)**: {ttd} minutes
- **Time to Resolve (TTR)**: {ttr} minutes
### Summary
{summary}
## Root Cause Analysis
{root_cause}
### The 5 Whys Chain
{whys}
## Preventative Action Items
| Action ID | Description | Priority | Owner Team |
|---|---|---|---|
{actions_table}
"""
with open("incident-postmortem.md", "w") as f:
f.write(markdown)
print("Generated incident-postmortem.md")
outputFiles:
- incident-postmortem.md
- id: severity_routing_gate
type: io.kestra.plugin.core.flow.If
description: Routes SEV-0 and SEV-1 incidents directly to engineering leadership
Slack channels.
condition: "{{ inputs.severity == 'SEV-0' || inputs.severity == 'SEV-1' }}"
then:
- id: alert_leadership_channel
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alerts engineering leadership channel with executive post-mortem
summary and action items.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"channel": "#engineering-leadership",
"text": "🚨 *Executive Post-Mortem Published*: `{{ inputs.incident_id }}` ({{ inputs.severity }})\n*Service*: `{{ inputs.service_name }}`\n*Resolution Time (TTR)*: {{ json(outputs.analyze_incident_timeline.text).time_to_resolve_minutes }} minutes\n*Root Cause*: {{ json(outputs.analyze_incident_timeline.text).root_cause_summary }}\nFull post-mortem artifact generated in Kestra storage."
}
else:
- id: notify_devops_team
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Dispatches post-mortem notification to standard DevOps channel.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"channel": "#devops-incidents",
"text": "📋 *Post-Mortem Ready*: `{{ inputs.incident_id }}` ({{ inputs.severity }}) for `{{ inputs.service_name }}`.\n*TTR*: {{ json(outputs.analyze_incident_timeline.text).time_to_resolve_minutes }} mins | Root cause analyzed and action items assigned."
}
errors:
- id: alert_on_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alerts on-call engineering channel if post-mortem generation
encounters an unhandled error.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"channel": "#sre-alerts",
"text": "⚠️ *Pipeline Error*: Incident post-mortem generator `{{ flow.id }}` failed on execution `{{ execution.id }}`."
}