id: ab-test-readout-srm-guardrail
namespace: company.team
inputs:
- id: experiment
type: STRING
displayName: Experiment
defaults: checkout-one-page
- id: scenario
type: SELECT
displayName: Demo scenario
description: >
WINNER: the treatment lifts conversion. FLAT: no real effect.
GUARDRAIL_HIT: conversion goes up but refunds go up more.
BROKEN_ASSIGNMENT: a bug drops part of the treatment traffic, so the
result looks great but the experiment is invalid.
values:
- WINNER
- FLAT
- GUARDRAIL_HIT
- BROKEN_ASSIGNMENT
defaults: WINNER
- id: alpha
type: FLOAT
displayName: Significance level
defaults: 0.05
- id: srm_alpha
type: FLOAT
displayName: Sample ratio mismatch threshold
description: A p-value below this on the traffic split means the assignment is
broken and nothing else is trusted.
defaults: 0.001
- id: max_guardrail_increase
type: FLOAT
displayName: Max refund-rate increase (relative)
description: Block shipping when the treatment raises the refund rate by more
than this share, for example 0.10 for 10%.
defaults: 0.10
tasks:
- id: readout
type: io.kestra.plugin.scripts.python.Script
description: >
Generate the experiment events for the chosen scenario, then read them out
in the order that protects against wrong decisions: sample ratio mismatch
first, then the primary metric with a two-proportion z-test and a
confidence interval, then the guardrail. Python standard library only.
containerImage: python:3.12-slim
env:
SCENARIO: "{{ inputs.scenario }}"
ALPHA: "{{ inputs.alpha }}"
SRM_ALPHA: "{{ inputs.srm_alpha }}"
MAX_GUARDRAIL: "{{ inputs.max_guardrail_increase }}"
outputFiles:
- readout.md
script: |
import json, math, os, random
scenario = os.environ["SCENARIO"]
alpha, srm_alpha, max_guard = float(os.environ["ALPHA"]), float(os.environ["SRM_ALPHA"]), float(os.environ["MAX_GUARDRAIL"])
random.seed(2026)
# Demo data: 120,000 visitors split 50/50. Replace with a query on your event table.
conv = {"control": 0.100, "treatment": {"WINNER": 0.108, "FLAT": 0.100, "GUARDRAIL_HIT": 0.108, "BROKEN_ASSIGNMENT": 0.100}[scenario]}
refund = {"control": 0.050, "treatment": 0.075 if scenario == "GUARDRAIL_HIT" else 0.050}
arms = {"control": [0, 0, 0], "treatment": [0, 0, 0]} # visitors, conversions, refunds
for _ in range(120000):
arm = "control" if random.random() < 0.5 else "treatment"
converted = random.random() < conv[arm]
# The bug: treatment visitors who do not convert on the first page view are not logged.
if scenario == "BROKEN_ASSIGNMENT" and arm == "treatment" and not converted and random.random() < 0.06:
continue
a = arms[arm]
a[0] += 1
if converted:
a[1] += 1
a[2] += random.random() < refund[arm]
def norm_sf(z):
return 0.5 * math.erfc(z / math.sqrt(2))
(nc, cc, rc), (nt, ct, rt) = arms["control"], arms["treatment"]
# 1. Sample ratio mismatch: chi-square with 1 degree of freedom against the planned 50/50.
expected = (nc + nt) / 2
chi2 = (nc - expected) ** 2 / expected + (nt - expected) ** 2 / expected
srm_p = math.erfc(math.sqrt(chi2 / 2))
# 2. Primary metric: conversion, two-sided two-proportion z-test and a 95% interval on the difference.
pc, pt = cc / nc, ct / nt
pooled = (cc + ct) / (nc + nt)
se_pooled = math.sqrt(pooled * (1 - pooled) * (1 / nc + 1 / nt))
z = (pt - pc) / se_pooled
p_value = 2 * norm_sf(abs(z))
se = math.sqrt(pc * (1 - pc) / nc + pt * (1 - pt) / nt)
lo, hi = (pt - pc) - 1.96 * se, (pt - pc) + 1.96 * se
# 3. Guardrail: refunds per conversion.
gc, gt = rc / max(cc, 1), rt / max(ct, 1)
guard_increase = (gt - gc) / gc if gc else 0.0
if srm_p < srm_alpha:
verdict = "INVALID"
elif p_value < alpha and pt > pc and guard_increase > max_guard:
verdict = "GUARDRAIL_BREACH"
elif p_value < alpha and pt > pc:
verdict = "SHIP"
elif p_value < alpha and pt < pc:
verdict = "ROLL_BACK"
else:
verdict = "INCONCLUSIVE"
out = {
"verdict": verdict, "control_visitors": nc, "treatment_visitors": nt,
"srm_p": float(f"{srm_p:.3g}"), "control_rate": round(pc, 4), "treatment_rate": round(pt, 4),
"relative_lift": round((pt - pc) / pc, 4), "p_value": float(f"{p_value:.3g}"),
"ci_low_pts": round(lo * 100, 2), "ci_high_pts": round(hi * 100, 2),
"refund_control": round(gc, 4), "refund_treatment": round(gt, 4), "refund_increase": round(guard_increase, 4),
}
lines = [
f"# Experiment readout: {scenario}", "", f"Verdict: **{verdict}**", "",
"| check | control | treatment | result |", "|---|---|---|---|",
f"| traffic split (planned 50/50) | {nc} | {nt} | SRM p = {out['srm_p']} |",
f"| conversion | {pc:.2%} | {pt:.2%} | lift {out['relative_lift']:+.1%}, p = {out['p_value']}, 95% CI {out['ci_low_pts']:+.2f} to {out['ci_high_pts']:+.2f} pts |",
f"| refunds per conversion (guardrail) | {gc:.2%} | {gt:.2%} | {guard_increase:+.0%} relative |",
]
open("readout.md", "w").write("\n".join(lines) + "\n")
print("\n".join(lines))
print("::" + json.dumps({"outputs": out}) + "::")
- id: publish_readout
type: io.kestra.plugin.core.namespace.UploadFiles
namespace: "{{ flow.namespace }}"
filesMap:
"experiments/{{ inputs.experiment }}/readout.md": "{{ outputs.readout.outputFiles['readout.md'] }}"
- id: decision
type: io.kestra.plugin.core.flow.Switch
value: "{{ outputs.readout.vars.verdict }}"
cases:
SHIP:
- id: record_ship
type: io.kestra.plugin.core.kv.Set
description: The decision a feature flag service or a deploy pipeline reads.
key: "experiment_{{ inputs.experiment | replace({'-': '_'}) }}"
kvType: JSON
value: |
{"decision": "SHIP", "lift": {{ outputs.readout.vars.relative_lift }}, "p_value": {{ outputs.readout.vars.p_value }}, "execution_id": "{{ execution.id }}"}
INCONCLUSIVE:
- id: keep_running
type: io.kestra.plugin.core.log.Log
message: "No significant difference yet (p = {{ outputs.readout.vars.p_value }},
95% CI {{ outputs.readout.vars.ci_low_pts }} to {{
outputs.readout.vars.ci_high_pts }} pts). Keep the experiment
running or stop it as flat."
INVALID:
- id: invalid_experiment
type: io.kestra.plugin.core.execution.Fail
description: A broken split biases every metric, so the conversion result is not
even reported as a decision.
errorMessage: "Sample ratio mismatch: {{ outputs.readout.vars.control_visitors
}} vs {{ outputs.readout.vars.treatment_visitors }} visitors (p = {{
outputs.readout.vars.srm_p }}). The assignment or the logging is
broken. Do not read the {{ outputs.readout.vars.relative_lift }}
lift. Fix the bug and restart the experiment."
defaults:
- id: do_not_ship
type: io.kestra.plugin.core.execution.Fail
errorMessage: "{{ outputs.readout.vars.verdict }}: conversion {{
outputs.readout.vars.control_rate }} to {{
outputs.readout.vars.treatment_rate }} (p = {{
outputs.readout.vars.p_value }}), refunds per conversion {{
outputs.readout.vars.refund_control }} to {{
outputs.readout.vars.refund_treatment }} ({{
outputs.readout.vars.refund_increase }} relative, max {{
inputs.max_guardrail_increase }})."
triggers:
- id: daily
type: io.kestra.plugin.core.trigger.Schedule
cron: "0 9 * * *"