id: kubernetes-resource-quota-audit
namespace: company.team
description: |
Compare each namespaces' resource quota usage against the hard limit and
alert Slack when any namespace rides above the threshold.
triggers:
- id: daily_quota_audit
type: io.kestra.plugin.core.trigger.Schedule
description: Daily pass so a quota wall never surprises a deploy.
cron: "0 8 * * *"
disabled: true
inputs:
- id: threshold_percent
type: INT
defaults: 90
description: Usage percentage of any quota resource that triggers the alert.
tasks:
- id: audit_quotas
type: io.kestra.plugin.scripts.shell.Commands
description: kubectl get resourcequotas as JSON, compute percent for
CPU/memory/requests, and emit the hot namespaces via the stdout outputs
protocol. A failed kubectl reports worst case.
containerImage: alpine:3.20
taskRunner:
type: io.kestra.plugin.scripts.runner.docker.Docker
commands:
- |
apk add --no-cache kubectl python3 >/dev/null 2>&1
set -u
kubectl get resourcequotas -A -o json > rq.json 2>/dev/null || echo '{"items":[]}' > rq.json
cat > audit.py <<'PYEOF'
import json
items = json.load(open("rq.json")).get("items", [])
hot = []
for q in items:
used = q.get("status", {}).get("used", {})
hard = q.get("status", {}).get("hard", {})
ns = q.get("metadata", {}).get("namespace", "?")
worst = 0
for key, h in hard.items():
try:
pct = int(str(used.get(key, "0")).replace("m", "") or 0) / max(int(str(h).replace("m", "")), 1) * 100
if key in ("requests.cpu", "limits.cpu", "requests.memory", "limits.memory", "pods"):
worst = max(worst, int(pct))
except Exception:
pass
if worst >= int(__import__("os").environ.get("THRESHOLD", "90")):
hot.append(f"{ns}:{worst}%")
print(f"{len(items)} quota(s), {len(hot)} hot")
print("::" + json.dumps({"outputs": {"quotas": len(items), "hot_count": len(hot), "hot": ",".join(hot)}}) + "::")
PYEOF
python3 audit.py 2>/dev/null || echo '::{"outputs": {"quotas": 0, "hot_count": 1, "hot": "probe_failed"}}::'
env:
THRESHOLD: "{{ inputs.threshold_percent }}"
- id: hot_found
type: io.kestra.plugin.core.flow.If
description: Any hot namespace alerts; otherwise log the healthy pass.
condition: "{{ outputs.audit_quotas.vars.hot_count > 0 }}"
then:
- id: alert_hot
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Name the hot namespaces and their worst percentages.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": ":bar_chart: Resource quota pressure: {{ outputs.audit_quotas.vars.hot_count }} namespace(s) at {{ inputs.threshold_percent }}%+ of quota: {{ outputs.audit_quotas.vars.hot }}. Raise quotas or shed workloads before the next deploy. Execution {{ execution.id }}."
}
else:
- id: log_ok
type: io.kestra.plugin.core.log.Log
description: Record the passing audit.
message: "All {{ outputs.audit_quotas.vars.quotas }} quota(s) below {{
inputs.threshold_percent }}%."
errors:
- id: alert_audit_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the audit fails - a dead kubectl must not read as
healthy quotas.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Resource quota audit FAILED in flow {{ flow.id }} (execution {{ execution.id }}). Check kubectl access and the kubeconfig."
}
outputs:
- id: hot_namespaces
type: STRING
description: 'Hot namespaces with their worst percent, e.g. "prod:94%, staging:91%".'
value: "{{ outputs.audit_quotas.vars.hot }}"