id: terraform-state-lock-monitor
namespace: company.team
description: |
Detect stuck terraform state locks (lock older than N minutes) and alert
Slack so a dead apply never bricks the next run.
triggers:
- id: every_thirty_minutes
type: io.kestra.plugin.core.trigger.Schedule
description: Poll often enough that a wedged lock is caught before the team
retries and makes it worse.
cron: "*/30 * * * *"
disabled: true
inputs:
- id: lock_id
type: STRING
defaults: ""
description: Known stale lock ID to check (from the `terraform force-unlock`
error message). Leave empty to only log a reminder.
- id: max_age_minutes
type: INT
defaults: 60
description: Lock age in minutes that triggers the alert.
tasks:
- id: check_lock
type: io.kestra.plugin.scripts.shell.Commands
description: Query the configured backend for the lock's creation time (S3
example included), compute its age in minutes, and emit it via the stdout
outputs protocol. Any lookup failure reports worst case.
containerImage: hashicorp/terraform:1.9
taskRunner:
type: io.kestra.plugin.scripts.runner.docker.Docker
commands:
- |
set -u
cat > lock.py <<'PYEOF'
import os, json, subprocess, time, datetime
lock_id = os.environ.get("LOCK_ID", "").strip()
bucket = os.environ.get("STATE_BUCKET", "")
key = os.environ.get("STATE_KEY", "")
if not lock_id or not bucket:
print("no lock configured; treating as fresh probe")
print('::{"outputs": {"lock_age_minutes": 0}}::')
else:
try:
out = subprocess.run(["aws", "s3api", "head-object", "--bucket", bucket, "--key", key + ".tflock"], capture_output=True, text=True, timeout=30)
age = 9999
if out.returncode == 0:
for line in out.stdout.splitlines():
if "LastModified" in line:
ts = line.split('"')[3]
dt = datetime.datetime.strptime(ts, "%a, %d %b %Y %H:%M:%S %Z")
age = int((time.time() - dt.timestamp()) / 60)
print(f"lock {lock_id} age {age} minutes")
print("::" + json.dumps({"outputs": {"lock_age_minutes": age}}) + "::")
except Exception:
print("lock age probe failed; treating as stale")
print('::{"outputs": {"lock_age_minutes": 9999}}::')
PYEOF
apk add --no-cache python3 aws-cli >/dev/null 2>&1 || pip install --quiet awscli 2>/dev/null || true
python3 lock.py 2>/dev/null || echo '::{"outputs": {"lock_age_minutes": 9999}}::'
env:
LOCK_ID: "{{ inputs.lock_id }}"
STATE_BUCKET: "{{ secret('TF_STATE_BUCKET') }}"
STATE_KEY: "{{ secret('TF_STATE_KEY') }}"
- id: stale_lock
type: io.kestra.plugin.core.flow.If
description: One branch for stale locks - alert with the runbook hint; fresh or
none just logs.
condition: "{{ outputs.check_lock.vars.lock_age_minutes >= inputs.max_age_minutes }}"
then:
- id: alert_stale_lock
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: State the lock ID, age, and the safe next step.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": ":lock: Terraform state lock {{ inputs.lock_id }} has been held for {{ outputs.check_lock.vars.lock_age_minutes }} minutes (limit {{ inputs.max_age_minutes }}). Verify no apply is running, then `terraform force-unlock {{ inputs.lock_id }}`. Execution {{ execution.id }}."
}
else:
- id: log_ok
type: io.kestra.plugin.core.log.Log
description: Record the healthy reading.
message: "State lock age {{ outputs.check_lock.vars.lock_age_minutes }} minutes,
under the {{ inputs.max_age_minutes }} minute limit."
errors:
- id: alert_check_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the probe itself fails - a dead probe must not look like
a healthy lock state.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Terraform state lock check FAILED in flow {{ flow.id }} (execution {{ execution.id }}). Check AWS credentials and the backend bucket."
}
outputs:
- id: lock_age_minutes
type: INT
description: 'Age of the current lock in minutes, e.g. 5.'
value: "{{ outputs.check_lock.vars.lock_age_minutes }}"