id: llm-eval-regression-gate
namespace: company.team
description: |
Run a golden set of prompts through your model, grade every answer against
its expected marker, and alert the moment the pass rate drops below your
bar — the release gate for prompt and model changes.
triggers:
- id: nightly_eval
type: io.kestra.plugin.core.trigger.Schedule
description: Run the golden set once a day so quiet regressions surface without
waiting for a release. Shipped disabled; enable it after your first
successful manual run.
cron: "0 6 * * *"
disabled: true
- id: eval_webhook
type: io.kestra.plugin.core.trigger.Webhook
description: Call this from CI when a prompt file or model version changes, so
the gate runs on the same commit that made the change.
key: llm-eval-gate
inputs:
- id: eval_cases
type: STRING
defaults: |
[
{"id": "france-capital", "prompt": "What is the capital of France? Answer with one word.", "expected_contains": "Paris"},
{"id": "add-two", "prompt": "What is 2 + 2? Answer with the digit only.", "expected_contains": "4"},
{"id": "literal-reply", "prompt": "Reply with exactly: ok", "expected_contains": "ok"},
{"id": "say-no", "prompt": "Say NO and nothing else.", "expected_contains": "NO"}
]
description: JSON array of golden cases — each with a stable id, the user
prompt, and the substring the answer must contain (matched
case-insensitively). Keep ids simple; they appear in alerts.
- id: model
type: STRING
defaults: gpt-4o-mini
description: Chat model to evaluate. Any model id the endpoint accepts.
- id: base_url
type: STRING
defaults: https://api.openai.com/v1
description: OpenAI-compatible chat completions endpoint. Point it at Ollama,
vLLM, or any other compatible server to gate local models instead.
- id: pass_rate_threshold
type: FLOAT
defaults: 0.8
description: Minimum passing fraction (0.0-1.0). The alert fires when the
measured pass rate falls below this number.
tasks:
- id: run_eval
type: io.kestra.plugin.scripts.python.Script
description: Call the chat completions endpoint once per golden case with
temperature 0, grade each reply against its expected substring, and
publish totals — total cases, passes, pass rate, and the ids that failed —
through Kestra's stdout outputs protocol. A transient HTTP error is
retried once; authentication errors fail fast so a bad key cannot
masquerade as a zero score.
taskRunner:
type: io.kestra.plugin.core.runner.Process
env:
EVAL_CASES: "{{ inputs.eval_cases }}"
MODEL: "{{ inputs.model }}"
BASE_URL: "{{ inputs.base_url }}"
OPENAI_API_KEY: "{{ secret('OPENAI_API_KEY') }}"
script: |
import json
import os
import time
import urllib.error
import urllib.request
cases = json.loads(os.environ["EVAL_CASES"])
if not cases:
raise SystemExit("eval_cases is empty — the gate has nothing to grade")
base = os.environ["BASE_URL"].rstrip("/")
model = os.environ["MODEL"]
key = os.environ["OPENAI_API_KEY"]
def ask(prompt):
body = json.dumps({
"model": model,
"messages": [
{"role": "system", "content": "You are a precise assistant being evaluated. Follow the user instruction exactly."},
{"role": "user", "content": prompt},
],
"temperature": 0,
}).encode()
req = urllib.request.Request(
base + "/chat/completions",
data=body,
headers={"Content-Type": "application/json", "Authorization": "Bearer " + key},
method="POST",
)
last = "unknown error"
for attempt in range(2):
try:
with urllib.request.urlopen(req, timeout=60) as resp:
data = json.load(resp)
return (data["choices"][0]["message"].get("content") or "")
except urllib.error.HTTPError as e:
detail = e.read()[:300].decode(errors="replace")
last = "HTTP %s: %s" % (e.code, detail)
if e.code in (401, 403):
break
time.sleep(2)
except Exception as e:
last = str(e)
time.sleep(2)
raise RuntimeError("model call failed: " + last)
total = len(cases)
passed = 0
failed = []
for case in cases:
reply = ask(case["prompt"])
if case["expected_contains"].lower() in reply.lower():
passed += 1
else:
failed.append(str(case.get("id", "?")))
rate = round(passed / total, 4)
failed_ids = ",".join(failed)
print("eval: %d/%d passed (rate %s) failed: %s" % (passed, total, rate, failed_ids or "-"))
print('::{"outputs":{"total":%d,"passed":%d,"pass_rate":%s,"failed_ids":"%s"}}::'
% (total, passed, rate, failed_ids))
- id: check_pass_rate
type: io.kestra.plugin.core.flow.If
description: Compare the measured pass rate against the threshold — one branch
for a regression that blocks the release, one for a clean gate.
condition: "{{ outputs.run_eval.vars.pass_rate < inputs.pass_rate_threshold }}"
then:
- id: alert_regression
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Post the score, the bar, the model, and the exact case ids that
failed so the author can open the failing prompt immediately instead
of re-running by hand.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "LLM eval REGRESSION: model {{ inputs.model }} scored {{ outputs.run_eval.vars.pass_rate }} ({{ outputs.run_eval.vars.passed }}/{{ outputs.run_eval.vars.total }}) below the {{ inputs.pass_rate_threshold }} bar. Failing cases: {{ outputs.run_eval.vars.failed_ids }}. Execution {{ execution.id }}."
}
else:
- id: log_pass
type: io.kestra.plugin.core.log.Log
description: Record a passing run so the execution history becomes the score
trend across model and prompt versions.
message: "LLM eval passed: model {{ inputs.model }} scored {{
outputs.run_eval.vars.pass_rate }} ({{ outputs.run_eval.vars.passed
}}/{{ outputs.run_eval.vars.total }}) against threshold {{
inputs.pass_rate_threshold }}."
errors:
- id: alert_eval_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the harness itself breaks — an unreachable endpoint or
rejected key produced no score, and a dead gate must never look like a
passing one.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "LLM eval FAILED to run in flow {{ flow.id }} (execution {{ execution.id }}) for model {{ inputs.model }}. No score was produced — verify base_url, the OPENAI_API_KEY secret, and network access from the Worker."
}
outputs:
- id: eval_summary
type: JSON
description: 'The measured score — e.g. {"total": 4, "passed": 3, "pass_rate":
0.75, "failed_ids": "say-no"} — ready for dashboards or downstream gates.'
value: "{{ outputs.run_eval.vars | toJson }}"