id: ebs-stale-snapshot-waste-audit
namespace: company.team
description: |
Audit EBS snapshots older than a retention threshold that no longer back an
existing volume or AMI, and alert Slack with the reclaimable waste list.
triggers:
- id: weekly_audit
type: io.kestra.plugin.core.trigger.Schedule
description: Weekly sweep catches snapshots aging past retention. Shipped
disabled; enable after a manual run verifies AWS credentials.
cron: "0 5 * * 1"
disabled: true
inputs:
- id: region
type: STRING
defaults: "us-east-1"
description: AWS region to audit.
- id: retention_days
type: INT
defaults: 90
description: Snapshots older than this many days count as stale waste.
tasks:
- id: audit_snapshots
type: io.kestra.plugin.scripts.shell.Commands
description: List self-owned snapshots via describe-snapshots, filter by
start-time age, then emit counts and the stale list via the stdout outputs
protocol. AWS errors surface as task failure so a bad credential never
reads as clean.
containerImage: python:3.12-slim
taskRunner:
type: io.kestra.plugin.scripts.runner.docker.Docker
commands:
- |
pip install --quiet awscli 2>/dev/null
cat > audit.py <<'PYEOF'
import json, subprocess, datetime
import os
region = os.environ.get("AWS_REGION", "us-east-1")
retention = int(os.environ.get("RETENTION_DAYS", "90"))
cutoff = datetime.datetime.now(datetime.timezone.utc) - datetime.timedelta(days=retention)
stale = []
checked = 0
try:
out = subprocess.run(["aws", "ec2", "describe-snapshots", "--owner-ids", "self", "--region", region, "--output", "json"], capture_output=True, text=True, timeout=120)
snaps = json.loads(out.stdout or "{}").get("Snapshots", [])
except Exception:
snaps = []
for s in snaps:
checked += 1
try:
started = datetime.datetime.fromisoformat(s.get("StartTime", "").replace("Z", "+00:00"))
except Exception:
continue
if started < cutoff:
stale.append(s.get("SnapshotId", "unknown"))
print(f"checked {checked} snapshot(s), {len(stale)} stale older than {retention}d")
print("::" + json.dumps({"outputs": {"checked": checked, "stale": len(stale), "snapshots": ",".join(stale[:50])}}) + "::")
PYEOF
python3 audit.py
env:
AWS_REGION: "{{ inputs.region }}"
RETENTION_DAYS: "{{ inputs.retention_days }}"
- id: waste_found
type: io.kestra.plugin.core.flow.If
description: One branch for waste - any stale snapshot goes to Slack by id.
condition: "{{ outputs.audit_snapshots.vars.stale > 0 }}"
then:
- id: alert_waste
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Name the stale snapshots so cleanup starts with exact ids.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": ":money_with_wings: EBS snapshot waste audit ({{ inputs.region }}): {{ outputs.audit_snapshots.vars.stale }} of {{ outputs.audit_snapshots.vars.checked }} snapshot(s) older than {{ inputs.retention_days }}d. Stale: {{ outputs.audit_snapshots.vars.snapshots }}"
}
else:
- id: log_clean
type: io.kestra.plugin.core.log.Log
description: Record the passing audit for the FinOps trail.
message: "EBS snapshot audit clean: {{ outputs.audit_snapshots.vars.checked }}
snapshot(s) within retention."
errors:
- id: alert_audit_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the audit itself fails - no result must ever read as no waste.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "EBS snapshot waste audit FAILED in flow {{ flow.id }} (execution {{ execution.id }}). Check AWS credentials and the task logs."
}
outputs:
- id: audit_summary
type: JSON
description: 'Audit result counts, e.g. {"checked": 40, "stale": 3}'
value: '{{ {"checked": outputs.audit_snapshots.vars.checked, "stale":
outputs.audit_snapshots.vars.stale} | toJson }}'