id: osv-new-vulnerability-issue-sync
namespace: company.team
inputs:
- id: scan_repository
type: STRING
displayName: Repository to scan
description: Repository whose manifests and lockfiles are scanned, in owner/repo form.
defaults: OWASP/NodeGoat
- id: scan_commit
type: STRING
displayName: Commit to scan
description: Full or short commit SHA. The same commit is re-scanned on every
run, so a finding can only be new because the advisory database moved.
defaults: c5cb68a7084e4ae7dcc60e6a98768720a81841e8
- id: scan_path
type: STRING
displayName: Path inside the checkout
description: Directory or lockfile osv-scanner walks. Use a subdirectory to scan
one service in a monorepo.
defaults: "."
- id: issue_repository
type: STRING
displayName: Repository that receives the issues
description: Where issues are opened and closed, in owner/repo form. Often the
same as the scanned repository.
defaults: your-org/your-repo
- id: max_new_issues
type: INT
displayName: Max new issues per run
description: Safety cap. Only this many new vulnerabilities are handled per run,
worst severity first. The rest stay out of the baseline so the next run
picks them up.
defaults: 10
- id: dry_run
type: BOOL
displayName: Dry run
description: Report the diff and log every issue it would open or close without
calling GitHub. Leave this on until the counts look right.
defaults: true
variables:
baseline_key: "osv_baseline_{{ inputs.scan_repository | replace({'/': '_'})
}}_{{ inputs.scan_commit | slice(0, 7) }}"
triggers:
- id: daily
type: io.kestra.plugin.core.trigger.Schedule
description: The load-bearing trigger. The lockfile does not change, the
advisory database does, so re-scanning the same commit every morning is
the only way an advisory published against an already shipped dependency
is ever found.
cron: "0 6 * * *"
- id: on_push
type: io.kestra.plugin.core.trigger.Webhook
description: Optional push-time scan. POST the commit you just shipped to get
the diff for it. The key guards the URL, so treat it as a secret.
key: "{{ secret('OSV_SCAN_WEBHOOK_KEY') }}"
tasks:
- id: read_baseline
type: io.kestra.plugin.core.kv.Get
description: The vulnerabilities this flow already tracks for this repository at
this commit. No TTL is set, because an expired key throws instead of
reading as absent.
key: "{{ render(vars.baseline_key) }}"
errorOnMissing: false
- id: checkout_and_scan
type: io.kestra.plugin.core.flow.WorkingDirectory
description: One directory so the checkout, the scanner and the diff all see the
same files.
tasks:
- id: clone_repository
type: io.kestra.plugin.git.Clone
description: Detached checkout of the exact commit. Pinning it is what makes two
runs comparable, since any difference then comes from the advisory
database rather than from the code.
url: "https://github.com/{{ inputs.scan_repository }}.git"
commit: "{{ inputs.scan_commit }}"
- id: scan_dependencies
type: io.kestra.plugin.scripts.shell.Commands
description: >
osv-scanner in its own container. Three things this task has to get
right. The image entry point is the bare binary at /osv-scanner with
no PATH entry for it. Kestra blanks the image entry point, so the
command gives the full path. Exit code 1 means vulnerabilities were
found, which is the normal case on any real repository, so the exit
code is captured rather than allowed to abort the run. Exit code 128
means no supported manifest was found and no report file is written at
all, which is a configuration problem rather than a clean repository,
so it fails loudly.
taskRunner:
type: io.kestra.plugin.scripts.runner.docker.Docker
containerImage: ghcr.io/google/osv-scanner:v2.6.0
commands:
- |
pwd > osv-root.txt
code=0
/osv-scanner scan source --format json --output-file osv.json "{{ inputs.scan_path }}" || code=$?
echo "osv-scanner exit code: $code (0 clean, 1 vulnerabilities found, 127 scanner error, 128 no supported manifest)"
if [ "$code" = "128" ]; then
echo "No supported manifest or lockfile was found under '{{ inputs.scan_path }}'. That is a configuration problem, not a clean repository, so this run stops here. Pass --allow-no-lockfiles to the scanner if an empty scan is expected."
exit 1
fi
if [ "$code" != "0" ] && [ "$code" != "1" ]; then
echo "osv-scanner itself failed with exit code $code. No diff is possible."
exit "$code"
fi
if [ ! -s osv.json ]; then
echo "osv-scanner exited $code but wrote no report. Treating that as a scanner failure rather than a clean result."
exit 1
fi
echo "report bytes: $(wc -c < osv.json)"
echo "::{\"outputs\":{\"scanner_exit\":$code}}::"
- id: diff_against_baseline
type: io.kestra.plugin.scripts.python.Script
description: >
The whole state model, in the standard library. Vulnerabilities are
keyed on the alias group rather than on a raw id, since OSV says two
advisories that alias each other are one vulnerability. A group that
shares any id with a baseline entry is known, which keeps a tracked
vulnerability from looking new when OSV adds an alias to it. Only the
capped selection enters the next baseline, so a cap can never mark an
unhandled finding as known.
taskRunner:
type: io.kestra.plugin.scripts.runner.docker.Docker
containerImage: python:3.12-slim
env:
MAX_NEW_ISSUES: "{{ inputs.max_new_issues }}"
inputFiles:
baseline.json: "{{ outputs.read_baseline.value ?? '' }}"
script: |
import json
import os
MAX_NEW = max(0, int(os.environ.get("MAX_NEW_ISSUES", "10")))
SUMMARY_CHARS = 180
ID_LIST_CHARS = 20
with open("osv.json", encoding="utf-8") as fh:
report = json.load(fh)
root = ""
if os.path.exists("osv-root.txt"):
root = open("osv-root.txt", encoding="utf-8").read().strip()
baseline = {}
if os.path.exists("baseline.json"):
text = open("baseline.json", encoding="utf-8").read().strip()
if text:
baseline = json.loads(text)
if not isinstance(baseline, dict):
baseline = {}
tracked = [e for e in (baseline.get("tracked") or [])
if isinstance(e, dict) and e.get("ids")]
def relative(path):
if root and path.startswith(root):
return path[len(root):].lstrip("/") or os.path.basename(path)
return path.lstrip("/") or "unknown"
def primary_of(ids):
for prefix in ("GHSA-", "CVE-", "GO-", "PYSEC-", "RUSTSEC-", "OSV-"):
hits = sorted(i for i in ids if i.startswith(prefix))
if hits:
return hits[0]
return sorted(ids)[0]
def score(text):
try:
return float(text)
except (TypeError, ValueError):
return -1.0
def one_line(text):
return " ".join((text or "").split())
raw = []
for result in report.get("results") or []:
source = relative((result.get("source") or {}).get("path") or "")
for pkg in result.get("packages") or []:
meta = pkg.get("package") or {}
label = "{}@{}".format(meta.get("name") or "unknown",
meta.get("version") or "unknown")
if meta.get("ecosystem"):
label = "{} ({})".format(label, meta["ecosystem"])
summaries = {}
for vuln in pkg.get("vulnerabilities") or []:
if vuln.get("id"):
summaries[vuln["id"]] = one_line(vuln.get("summary"))
for group in pkg.get("groups") or []:
ids = {i for i in (group.get("ids") or []) + (group.get("aliases") or []) if i}
if not ids:
continue
summary = ""
for gid in group.get("ids") or []:
if summaries.get(gid):
summary = summaries[gid]
break
raw.append({
"ids": ids,
"packages": {label},
"sources": {source},
"severity": score(group.get("max_severity")),
"summary": summary,
})
# One advisory affecting two packages is one vulnerability. Merge on any shared id.
reported = []
for item in raw:
overlapping = [r for r in reported if r["ids"] & item["ids"]]
if not overlapping:
reported.append(item)
continue
head = overlapping[0]
for other in overlapping[1:] + [item]:
head["ids"] |= other["ids"]
head["packages"] |= other["packages"]
head["sources"] |= other["sources"]
head["severity"] = max(head["severity"], other["severity"])
head["summary"] = head["summary"] or other["summary"]
for other in overlapping[1:]:
reported.remove(other)
matched = set()
new_items = []
known_items = []
for record in reported:
hit = None
for index, entry in enumerate(tracked):
if index in matched:
continue
if set(entry.get("ids") or []) & record["ids"]:
hit = entry
matched.add(index)
break
if hit is None:
record["primary_id"] = primary_of(record["ids"])
new_items.append(record)
else:
# Keep the id the open issue was titled with, so an added alias never retitles it.
record["primary_id"] = hit.get("primary_id") or primary_of(record["ids"])
known_items.append(record)
resolved = [e for i, e in enumerate(tracked) if i not in matched]
new_items.sort(key=lambda r: (-r["severity"], r["primary_id"]))
selected = new_items[:MAX_NEW]
deferred = new_items[MAX_NEW:]
def item_json(record):
return json.dumps({
"primary_id": record["primary_id"],
"ids": ", ".join(sorted(record["ids"])),
"packages": ", ".join(sorted(record["packages"])),
"sources": ", ".join(sorted(record["sources"])),
"severity": "{:g}".format(record["severity"]) if record["severity"] >= 0 else "not scored",
"summary": (record["summary"] or "No summary published by OSV.")[:SUMMARY_CHARS],
}, sort_keys=True)
def resolved_json(entry):
ids = sorted(entry.get("ids") or [])
return json.dumps({
"primary_id": entry.get("primary_id") or ids[0],
"ids": ", ".join(ids),
"packages": ", ".join(entry.get("packages") or []) or "unknown",
}, sort_keys=True)
def id_list(records, key="primary_id"):
names = [r[key] if isinstance(r, dict) and key in r else "?" for r in records]
head = names[:ID_LIST_CHARS]
text = ", ".join(head) if head else "none"
if len(names) > len(head):
text = "{} and {} more".format(text, len(names) - len(head))
return text
# Only what this run actually handled enters the baseline.
next_tracked = []
for record in known_items + selected:
next_tracked.append({
"ids": sorted(record["ids"]),
"primary_id": record["primary_id"],
})
next_tracked.sort(key=lambda e: e["primary_id"])
baseline_text = json.dumps({
"scan_repository": os.environ.get("SCAN_REPOSITORY", ""),
"tracked": next_tracked,
}, sort_keys=True)
with open("baseline-next.json", "w", encoding="utf-8") as fh:
fh.write(baseline_text)
verdict = {
"total_reported": len(reported),
"known_count": len(known_items),
"new_count": len(new_items),
"selected_count": len(selected),
"deferred_count": len(deferred),
"resolved_count": len(resolved),
"tracked_after": len(next_tracked),
"first_run": not tracked,
"capped": len(deferred) > 0,
"has_new": len(selected) > 0,
"has_resolved": len(resolved) > 0,
"selected_ids": id_list(selected),
"deferred_ids": id_list(deferred),
"resolved_ids": id_list(resolved),
"selected_json": [item_json(r) for r in selected],
"resolved_json": [resolved_json(e) for e in resolved],
"baseline_json": baseline_text,
}
payload = {"outputs": verdict}
print("::" + json.dumps(payload) + "::")
print("reported={} known={} new={} selected={} deferred={} resolved={} first_run={}".format(
len(reported), len(known_items), len(new_items), len(selected),
len(deferred), len(resolved), not tracked))
- id: report_diff
type: io.kestra.plugin.core.log.Log
description: The full count beside the capped set, in one place, so a run never
reads as having covered more than it did.
message: |
osv-scanner exit {{ outputs.scan_dependencies.vars.scanner_exit }} on {{ inputs.scan_repository }} at {{ inputs.scan_commit | slice(0, 7) }} (first run for this baseline: {{ outputs.diff_against_baseline.vars.first_run }}).
{{ outputs.diff_against_baseline.vars.total_reported }} vulnerability group(s) reported: {{ outputs.diff_against_baseline.vars.known_count }} already tracked, {{ outputs.diff_against_baseline.vars.new_count }} new, {{ outputs.diff_against_baseline.vars.resolved_count }} tracked but no longer reported.
Handled in this run: {{ outputs.diff_against_baseline.vars.selected_count }} of {{ outputs.diff_against_baseline.vars.new_count }} new (max_new_issues = {{ inputs.max_new_issues }}, capped: {{ outputs.diff_against_baseline.vars.capped }}). The other {{ outputs.diff_against_baseline.vars.deferred_count }} stay out of the baseline, so the next run reports them again.
Selected: {{ outputs.diff_against_baseline.vars.selected_ids }}
Deferred: {{ outputs.diff_against_baseline.vars.deferred_ids }}
No longer reported: {{ outputs.diff_against_baseline.vars.resolved_ids }}
Baseline after this run: {{ outputs.diff_against_baseline.vars.tracked_after }} vulnerability group(s) under key {{ render(vars.baseline_key) }}.
- id: handle_new
type: io.kestra.plugin.core.flow.If
description: Open one issue per newly reported vulnerability, worst severity first.
condition: "{{ outputs.diff_against_baseline.vars.has_new }}"
then:
- id: new_mode
type: io.kestra.plugin.core.flow.If
condition: "{{ inputs.dry_run }}"
then:
- id: planned_opens
type: io.kestra.plugin.core.flow.Loop
values: "{{ outputs.diff_against_baseline.vars.selected_json }}"
concurrencyLimit: 1
tasks:
- id: would_open
type: io.kestra.plugin.core.log.Log
message: |
DRY RUN would search {{ inputs.issue_repository }} for an existing issue titled "{{ fromJson(item.value).primary_id }}" (open or closed) and open this one when there is none:
title: {{ fromJson(item.value).primary_id }}: {{ fromJson(item.value).summary }}
labels: security, dependencies
severity: {{ fromJson(item.value).severity }}
ids and aliases: {{ fromJson(item.value).ids }}
packages: {{ fromJson(item.value).packages }}
manifest: {{ fromJson(item.value).sources }}
else:
- id: open_new_issues
type: io.kestra.plugin.core.flow.Loop
values: "{{ outputs.diff_against_baseline.vars.selected_json }}"
concurrencyLimit: 1
tasks:
- id: find_existing_issue
type: io.kestra.plugin.github.issues.Search
description: Second line of defence behind the baseline. A previous run can have
created the issue then failed before the baseline advanced, so
look for the id in a title first. Closed issues count as
found, since a vulnerability somebody already triaged should
not come back.
oauthToken: "{{ secret('GITHUB_TOKEN') }}"
repository: "{{ inputs.issue_repository }}"
query: "in:title \"{{ fromJson(item.value).primary_id }}\""
fetchType: FETCH
- id: create_when_absent
type: io.kestra.plugin.core.flow.If
condition: "{{ outputs.find_existing_issue.size == 0 }}"
then:
- id: create_issue
type: io.kestra.plugin.github.issues.Create
description: One issue per vulnerability, not per affected package, with every
affected package listed in the body.
oauthToken: "{{ secret('GITHUB_TOKEN') }}"
repository: "{{ inputs.issue_repository }}"
title: "{{ fromJson(item.value).primary_id }}: {{ fromJson(item.value).summary
}}"
labels:
- security
- dependencies
body: |
{{ fromJson(item.value).summary }}
osv-scanner reported this advisory against a dependency of `{{ inputs.scan_repository }}` at commit `{{ inputs.scan_commit }}`.
| field | value |
| --- | --- |
| Primary id | `{{ fromJson(item.value).primary_id }}` |
| Ids and aliases | {{ fromJson(item.value).ids }} |
| Max severity | {{ fromJson(item.value).severity }} |
| Affected packages | {{ fromJson(item.value).packages }} |
| Manifest | {{ fromJson(item.value).sources }} |
Advisory: https://osv.dev/vulnerability/{{ fromJson(item.value).primary_id }}
Opened by the Kestra flow `{{ flow.id }}`, execution `{{ execution.id }}`. The same flow closes this issue once osv-scanner stops reporting the advisory for this commit.
else:
- id: already_filed
type: io.kestra.plugin.core.log.Log
message: "{{ fromJson(item.value).primary_id }} already has issue #{{
outputs.find_existing_issue.row.number }} ({{
outputs.find_existing_issue.row.state }}) in {{
inputs.issue_repository }}. Nothing opened."
- id: handle_resolved
type: io.kestra.plugin.core.flow.If
description: Close the issues whose vulnerability stopped being reported, so the
backlog shrinks on its own.
condition: "{{ outputs.diff_against_baseline.vars.has_resolved }}"
then:
- id: resolved_mode
type: io.kestra.plugin.core.flow.If
condition: "{{ inputs.dry_run }}"
then:
- id: planned_closes
type: io.kestra.plugin.core.log.Log
message: |
DRY RUN would close {{ outputs.diff_against_baseline.vars.resolved_count }} issue(s) in {{ inputs.issue_repository }} for vulnerabilities osv-scanner no longer reports: {{ outputs.diff_against_baseline.vars.resolved_ids }}.
For each one it would search open issues for the id in the title, then PATCH https://api.github.com/repos/{{ inputs.issue_repository }}/issues/<number> with {"state": "closed", "state_reason": "completed"}.
else:
- id: close_resolved_issues
type: io.kestra.plugin.core.flow.Loop
values: "{{ outputs.diff_against_baseline.vars.resolved_json }}"
concurrencyLimit: 1
tasks:
- id: find_resolved_issue
type: io.kestra.plugin.github.issues.Search
description: Find the open issue by the id in its title. The baseline keeps the
id the issue was opened under, so this lookup is stable even
after OSV adds an alias.
oauthToken: "{{ secret('GITHUB_TOKEN') }}"
repository: "{{ inputs.issue_repository }}"
query: "in:title \"{{ fromJson(item.value).primary_id }}\""
open: true
fetchType: FETCH
- id: close_when_found
type: io.kestra.plugin.core.flow.If
condition: "{{ outputs.find_resolved_issue.size > 0 }}"
then:
- id: comment_resolution
type: io.kestra.plugin.github.issues.Comment
description: Say why it closed before closing it, so the issue carries its own
evidence.
oauthToken: "{{ secret('GITHUB_TOKEN') }}"
repository: "{{ inputs.issue_repository }}"
issueNumber: "{{ outputs.find_resolved_issue.row.number }}"
updateTag: "osv-resolution"
body: "osv-scanner no longer reports `{{ fromJson(item.value).primary_id }}` ({{
fromJson(item.value).ids }}) for `{{
inputs.scan_repository }}` at commit `{{
inputs.scan_commit }}`. Closing from Kestra execution `{{
execution.id }}`."
- id: close_issue
type: io.kestra.plugin.core.http.Request
description: Closing goes over REST because plugin-github ships only Create,
Search and Comment for issues. There is no close or update
task to call.
method: PATCH
uri: "https://api.github.com/repos/{{ inputs.issue_repository }}/issues/{{
outputs.find_resolved_issue.row.number }}"
contentType: application/json
headers:
Authorization: "Bearer {{ secret('GITHUB_TOKEN') }}"
Accept: application/vnd.github+json
X-GitHub-Api-Version: "2022-11-28"
body: |
{"state": "closed", "state_reason": "completed"}
else:
- id: no_issue_to_close
type: io.kestra.plugin.core.log.Log
message: "{{ fromJson(item.value).primary_id }} is no longer reported but no
open issue was found in {{ inputs.issue_repository }}. It
was closed by hand or the search index has not caught up."
- id: advance_baseline
type: io.kestra.plugin.core.kv.Set
description: Last task on purpose. A failed scan or a failed sync leaves the
baseline where it was, so nothing is ever silently marked known. No TTL,
because kv.Get throws on an expired key.
key: "{{ render(vars.baseline_key) }}"
kvType: JSON
overwrite: true
value: "{{ outputs.diff_against_baseline.vars.baseline_json }}"
errors:
- id: baseline_untouched
type: io.kestra.plugin.core.log.Log
level: ERROR
message: "Run failed for {{ inputs.scan_repository }} at {{ inputs.scan_commit
}}. The baseline under {{ render(vars.baseline_key) }} was not advanced,
so the next run compares against the same tracked set and re-reports
anything that was missed."
outputs:
- id: total_reported
type: INT
description: Alias-grouped vulnerabilities osv-scanner reported for this commit,
e.g. 305.
value: "{{ outputs.diff_against_baseline.vars.total_reported }}"
- id: new_count
type: INT
description: Vulnerabilities not in the baseline, e.g. 3.
value: "{{ outputs.diff_against_baseline.vars.new_count }}"
- id: selected_count
type: INT
description: New vulnerabilities handled in this run after the cap, e.g. 3.
value: "{{ outputs.diff_against_baseline.vars.selected_count }}"
- id: resolved_count
type: INT
description: Tracked vulnerabilities osv-scanner stopped reporting, e.g. 1.
value: "{{ outputs.diff_against_baseline.vars.resolved_count }}"