id: k6-load-regression-gate
namespace: company.team
inputs:
- id: service
type: STRING
displayName: Service name
description: Names the baseline in the KV store. Keep it the same across
releases of one service.
defaults: quickpizza
- id: base_url
type: STRING
displayName: Base URL
description: The deployment under test. The default is Grafana's public k6 demo
application.
defaults: https://quickpizza.grafana.com
- id: endpoint
type: STRING
displayName: Endpoint
description: Path to load. Set it to /api/delay/1 to simulate a slow release and
watch the gate block it.
defaults: /api/quotes
- id: release
type: STRING
displayName: Release
description: Version or commit of the deployment under test. Stored with the baseline.
defaults: v1.0.0
- id: vus
type: INT
displayName: Virtual users
description: Keep it small against a public demo target.
defaults: 3
- id: duration
type: STRING
displayName: Duration
description: k6 duration string.
defaults: 20s
- id: max_p95_ms
type: INT
displayName: Absolute p95 limit (ms)
description: Hard k6 threshold. A breach makes k6 exit with code 99.
defaults: 3000
- id: max_error_rate
type: FLOAT
displayName: Max error rate
description: Hard k6 threshold on the share of failed requests.
defaults: 0.01
- id: max_p95_regression_pct
type: INT
displayName: Max p95 regression (%)
description: How much slower than the last good release p95 may get before the
release is blocked.
defaults: 25
- id: notify_slack
type: BOOL
displayName: Notify Slack
description: Post the verdict to Slack. Needs the SLACK_WEBHOOK_URL secret.
defaults: false
variables:
baseline_key: "k6_baseline_{{ inputs.service }}"
triggers:
- id: after_deploy
type: io.kestra.plugin.core.trigger.Webhook
description: >
Call this from the deploy pipeline once a release is live, with the
release in the body, for example {"release": "v1.4.2"}. Replace the key
with a long random value first.
key: replace-with-a-long-random-key
tasks:
- id: load_test
type: io.kestra.plugin.scripts.shell.Commands
description: >
Run k6 with hard thresholds. handleSummary prints p95, p99, error rate and
request count as Kestra outputs, and the k6 exit code is captured instead
of failing the task: 99 means a threshold was breached, which the gate
below handles.
containerImage: grafana/k6:2.3.0
taskRunner:
type: io.kestra.plugin.scripts.runner.docker.Docker
user: root
env:
BASE_URL: "{{ inputs.base_url }}"
ENDPOINT: "{{ trigger.body.endpoint ?? inputs.endpoint }}"
VUS: "{{ inputs.vus }}"
DURATION: "{{ inputs.duration }}"
MAX_P95: "{{ inputs.max_p95_ms }}"
MAX_ERR: "{{ inputs.max_error_rate }}"
inputFiles:
test.js: |
import http from 'k6/http';
import { check, sleep } from 'k6';
export const options = {
vus: Number(__ENV.VUS),
duration: __ENV.DURATION,
thresholds: {
http_req_duration: [`p(95)<${__ENV.MAX_P95}`],
http_req_failed: [`rate<${__ENV.MAX_ERR}`],
},
summaryTrendStats: ['avg', 'med', 'p(95)', 'p(99)', 'max'],
};
export default function () {
const res = http.get(`${__ENV.BASE_URL}${__ENV.ENDPOINT}`);
check(res, { 'status is 2xx': (r) => r.status >= 200 && r.status < 300 });
sleep(0.5);
}
export function handleSummary(data) {
const d = data.metrics.http_req_duration.values;
const out = {
p95_ms: Math.round(d['p(95)'] * 10) / 10,
p99_ms: Math.round(d['p(99)'] * 10) / 10,
median_ms: Math.round(d.med * 10) / 10,
error_rate: Math.round(data.metrics.http_req_failed.values.rate * 10000) / 10000,
requests: data.metrics.http_reqs.values.count,
};
return {
stdout: `::${JSON.stringify({ outputs: out })}::\n`,
'summary.json': JSON.stringify(data, null, 2),
};
}
outputFiles:
- summary.json
commands:
- k6 run --quiet test.js && code=0 || code=$?
- echo "::{\"outputs\":{\"k6_exit\":$code}}::"
- test "$code" -eq 0 -o "$code" -eq 99
- id: read_baseline
type: io.kestra.plugin.core.output.OutputValues
description: The last good release for this service. kv() returns null when
there is none yet.
values:
baseline: "{{ kv(render(vars.baseline_key), errorOnMissing=false) ?? '' }}"
- id: verdict
type: io.kestra.plugin.core.output.OutputValues
description: >
THRESHOLD_BROKEN when k6 itself failed a hard limit, FIRST_RUN when there
is no baseline yet, REGRESSED when p95 got slower than the baseline by
more than the allowed share, PASS otherwise.
values:
release: "{{ trigger.body.release ?? inputs.release }}"
baseline_p95: "{{ outputs.read_baseline.values.baseline == '' ? 0 :
(outputs.read_baseline.values.baseline | jq('.p95_ms') | first) }}"
baseline_release: "{{ outputs.read_baseline.values.baseline == '' ? '' :
(outputs.read_baseline.values.baseline | jq('.release') | first) }}"
limit_p95: "{{ outputs.read_baseline.values.baseline == '' ? 0 :
((outputs.read_baseline.values.baseline | jq('.p95_ms') | first) * (100
+ inputs.max_p95_regression_pct) / 100) | numberFormat('0.0') }}"
state: >-
{%- set p95 = outputs.load_test.vars.p95_ms -%} {%- if
outputs.load_test.vars.k6_exit == 99 -%}THRESHOLD_BROKEN {%- elseif
outputs.read_baseline.values.baseline == '' -%}FIRST_RUN {%- elseif p95
> ((outputs.read_baseline.values.baseline | jq('.p95_ms') | first) *
(100 + inputs.max_p95_regression_pct) / 100) -%}REGRESSED {%- else
-%}PASS{%- endif -%}
- id: log_verdict
type: io.kestra.plugin.core.log.Log
message: >-
{{ outputs.verdict.values.state }} for {{ inputs.service }} {{
outputs.verdict.values.release }}: p95 {{ outputs.load_test.vars.p95_ms }}
ms, p99 {{ outputs.load_test.vars.p99_ms }} ms, error rate {{
outputs.load_test.vars.error_rate }}, {{ outputs.load_test.vars.requests
}} requests. {% if outputs.verdict.values.baseline_release != ''
%}Baseline {{ outputs.verdict.values.baseline_release }} p95 {{
outputs.verdict.values.baseline_p95 }} ms, limit {{
outputs.verdict.values.limit_p95 }} ms.{% else %}No baseline yet.{% endif
%}
- id: gate
type: io.kestra.plugin.core.flow.Switch
value: "{{ outputs.verdict.values.state }}"
cases:
PASS:
- id: promote_baseline
type: io.kestra.plugin.core.kv.Set
description: This release becomes the bar the next one is measured against.
key: "{{ render(vars.baseline_key) }}"
kvType: JSON
value: |
{"release": "{{ outputs.verdict.values.release }}", "p95_ms": {{ outputs.load_test.vars.p95_ms }}, "error_rate": {{ outputs.load_test.vars.error_rate }}, "execution_id": "{{ execution.id }}"}
FIRST_RUN:
- id: seed_baseline
type: io.kestra.plugin.core.kv.Set
description: No baseline yet, so this run sets it. Only absolute thresholds were
applied.
key: "{{ render(vars.baseline_key) }}"
kvType: JSON
value: |
{"release": "{{ outputs.verdict.values.release }}", "p95_ms": {{ outputs.load_test.vars.p95_ms }}, "error_rate": {{ outputs.load_test.vars.error_rate }}, "execution_id": "{{ execution.id }}"}
defaults:
- id: notify_blocked
type: io.kestra.plugin.core.flow.If
condition: "{{ inputs.notify_slack }}"
then:
- id: slack_blocked
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Load gate {{ outputs.verdict.values.state }} for {{ inputs.service }} {{ outputs.verdict.values.release }}: p95 {{ outputs.load_test.vars.p95_ms }} ms (limit {{ outputs.verdict.values.limit_p95 }} ms vs baseline {{ outputs.verdict.values.baseline_release }}), error rate {{ outputs.load_test.vars.error_rate }}. Baseline kept. Execution {{ execution.id }}."
}
- id: block_release
type: io.kestra.plugin.core.execution.Fail
description: End the run as FAILED, so the deploy pipeline that called the
webhook can roll back. The baseline is not touched.
errorMessage: "{{ outputs.verdict.values.state }}: p95 {{
outputs.load_test.vars.p95_ms }} ms, error rate {{
outputs.load_test.vars.error_rate }}. Baseline {{
outputs.verdict.values.baseline_release }} p95 {{
outputs.verdict.values.baseline_p95 }} ms, limit {{
outputs.verdict.values.limit_p95 }} ms."
- id: notify_ok
type: io.kestra.plugin.core.flow.If
condition: "{{ inputs.notify_slack }}"
then:
- id: slack_ok
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Load gate {{ outputs.verdict.values.state }} for {{ inputs.service }} {{ outputs.verdict.values.release }}: p95 {{ outputs.load_test.vars.p95_ms }} ms, error rate {{ outputs.load_test.vars.error_rate }}. Baseline updated. Execution {{ execution.id }}."
}