id: http-latency-probe-matrix
namespace: company.team
description: |
Probe a URL repeatedly, compute latency percentiles, and alert Slack when
p95 climbs past your budget — a lightweight synthetic check you own.
triggers:
- id: every_15min
type: io.kestra.plugin.core.trigger.Schedule
description: Runs the probe matrix every 15 minutes; shipped disabled until the
URL and budget are set.
cron: "*/15 * * * *"
disabled: true
- id: incident_webhook
type: io.kestra.plugin.core.trigger.Webhook
description: Fire an immediate probe run while debugging a live incident.
key: latency-probe
inputs:
- id: url
type: STRING
defaults: https://example.com
description: Endpoint to probe.
- id: samples
type: INT
defaults: 10
description: Number of requests per run.
- id: p95_budget_ms
type: INT
defaults: 800
description: Budget for the 95th-percentile response time in milliseconds.
tasks:
- id: probe_latency
type: io.kestra.plugin.scripts.python.Script
description: Fire N timed requests at the URL and emit min/mean/p95 plus failure
count via the stdout outputs protocol. HTTP errors count as samples with a
sentinel latency so degradation always shows up in the numbers.
containerImage: python:3.11-slim
warningOnStdErr: false
env:
URL: "{{ inputs.url }}"
SAMPLES: "{{ inputs.samples }}"
script: |
import json, os, statistics, time, urllib.request
url = os.environ["URL"]
samples = int(os.environ["SAMPLES"])
latencies, failures = [], 0
for _ in range(samples):
started = time.perf_counter()
try:
req = urllib.request.Request(url, headers={"User-Agent": "kestra-latency-probe"})
with urllib.request.urlopen(req, timeout=10) as r:
r.read(256)
latencies.append(int((time.perf_counter() - started) * 1000))
except Exception:
failures += 1
latencies.append(10000)
latencies.sort()
p95 = latencies[max(0, int(0.95 * len(latencies)) - 1)]
print(f"p95 {p95}ms over {samples} samples, failures {failures}")
print("::" + json.dumps({"outputs": {"p95": p95, "mean": int(statistics.mean(latencies)), "min": latencies[0], "failures": failures}}) + "::")
- id: check_budget
type: io.kestra.plugin.core.flow.If
description: Breach either the p95 budget or the failure allowance → page Slack
with the numbers; healthy → logged for the trend.
condition: "{{ outputs.probe_latency.vars.p95 > inputs.p95_budget_ms or
outputs.probe_latency.vars.failures > 0 }}"
then:
- id: alert_latency_breach
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Post p95, mean, and failures next to the budget so the message
carries the size of the problem.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": ":warning: Latency breach on {{ inputs.url }} — p95 {{ outputs.probe_latency.vars.p95 }}ms (budget {{ inputs.p95_budget_ms }}ms), mean {{ outputs.probe_latency.vars.mean }}ms, failures {{ outputs.probe_latency.vars.failures }}/{{ inputs.samples }}. Execution {{ execution.id }}."
}
else:
- id: log_within_budget
type: io.kestra.plugin.core.log.Log
description: Record healthy numbers on every run so the execution history is
your latency trend.
message: "Latency within budget on {{ inputs.url }}: p95 {{
outputs.probe_latency.vars.p95 }}ms, mean {{
outputs.probe_latency.vars.mean }}ms, failures {{
outputs.probe_latency.vars.failures }}."
errors:
- id: alert_probe_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the probe script itself fails — no numbers must never
look like a fast page.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Latency probe FAILED in flow {{ flow.id }} (execution {{ execution.id }}) for {{ inputs.url }}."
}
outputs:
- id: latency_summary
type: JSON
description: 'Percentile stats from this run, e.g. {"p95": 412, "mean": 180,
"min": 95, "failures": 0}.'
value: "{{ outputs.probe_latency.vars | toJson }}"