id: sitemap-broken-link-auditor
namespace: company.team
description: |
Crawl your sitemap, HEAD-check every indexed URL, and turn 404s and
timeouts into a GitHub issue and a Slack alert — the link rot audit
that runs itself.
triggers:
- id: weekly_audit
type: io.kestra.plugin.core.trigger.Schedule
description: Audit the sitemap once a week, early Monday. Shipped disabled;
enable it after your first successful manual run.
cron: "0 5 * * 1"
disabled: true
- id: audit_webhook
type: io.kestra.plugin.core.trigger.Webhook
description: Call this after a deploy or a migration to re-audit immediately
instead of waiting for the weekly slot.
key: sitemap-audit
inputs:
- id: sitemap_url
type: STRING
defaults: https://example.com/sitemap.xml
description: Absolute URL of your sitemap.xml — plain urlset or a sitemap index
(child sitemaps are followed, capped by max_urls). .gz files are
supported.
- id: repository
type: STRING
defaults: your-org/your-site
description: GitHub repository (owner/name) where the broken-link issue will be opened.
- id: max_urls
type: INT
defaults: 50
description: Audit at most this many URLs per run, so one giant sitemap cannot
turn into an hour-long crawl.
- id: request_timeout
type: INT
defaults: 10
description: Seconds to wait per URL before counting it as broken (timeouts are
link rot too).
tasks:
- id: crawl_sitemap
type: io.kestra.plugin.scripts.python.Script
description: Fetch and parse the sitemap (following index files and gzipped
sitemaps), probe every URL with HEAD and a GET fallback in a small thread
pool, and publish checked, broken_count, and the broken list — url with
status — through Kestra's stdout outputs protocol. A sitemap that fetches
but contains zero URLs fails fast, because an empty audit must never read
as a clean one.
taskRunner:
type: io.kestra.plugin.core.runner.Process
env:
SITEMAP_URL: "{{ inputs.sitemap_url }}"
MAX_URLS: "{{ inputs.max_urls }}"
REQUEST_TIMEOUT: "{{ inputs.request_timeout }}"
script: |
import gzip
import json
import os
import urllib.error
import urllib.request
import xml.etree.ElementTree as ET
from concurrent.futures import ThreadPoolExecutor
SITEMAP_URL = os.environ["SITEMAP_URL"]
MAX_URLS = int(os.environ["MAX_URLS"])
TIMEOUT = int(os.environ["REQUEST_TIMEOUT"])
UA = {"User-Agent": "Kestra-sitemap-auditor/1.0"}
NS = "{http://www.sitemaps.org/schemas/sitemap/0.9}"
def fetch(url):
req = urllib.request.Request(url, headers=UA)
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
data = resp.read()
if url.endswith(".gz") or data[:2] == b"\x1f\x8b":
data = gzip.decompress(data)
return data
def load_urls(url, depth=0):
root = ET.fromstring(fetch(url))
tag = root.tag.split("}")[-1]
locs = [e.text.strip() for e in root.iter(NS + "loc") if e.text]
if tag == "sitemapindex" and depth < 2:
out = []
for child in locs[:5]:
out.extend(load_urls(child, depth + 1))
if len(out) >= MAX_URLS:
break
return out
return locs
def probe(url):
for method in ("HEAD", "GET"):
try:
req = urllib.request.Request(url, method=method, headers=UA)
with urllib.request.urlopen(req, timeout=TIMEOUT) as resp:
return {"url": url, "status": resp.status}
except urllib.error.HTTPError as e:
if method == "HEAD":
continue
return {"url": url, "status": e.code}
except Exception as e:
if method == "HEAD":
continue
return {"url": url, "status": "ERR: " + str(e)[:80]}
urls = load_urls(SITEMAP_URL)[:MAX_URLS]
if not urls:
raise SystemExit("sitemap contains no URLs — refusing to report a clean audit")
with ThreadPoolExecutor(max_workers=8) as pool:
results = list(pool.map(probe, urls))
broken = [r for r in results if not isinstance(r["status"], int) or r["status"] >= 400]
print("checked %d url(s), %d broken" % (len(results), len(broken)))
print("::" + json.dumps({"outputs": {
"checked": len(results),
"broken_count": len(broken),
"broken": broken[:25],
}}) + "::")
- id: check_links
type: io.kestra.plugin.core.flow.If
description: Any broken URL is actionable — branch on the count so clean audits
stay quiet and broken ones leave a paper trail engineers already watch.
condition: "{{ outputs.crawl_sitemap.vars.broken_count > 0 }}"
then:
- id: create_issue
type: io.kestra.plugin.github.issues.Create
description: Open a labeled GitHub issue listing the failing URLs and their
statuses, so link rot lands in the same backlog as everything else the
team triages.
jwtToken: "{{ secret('GITHUB_TOKEN') }}"
repository: "{{ inputs.repository }}"
title: "Broken links detected: {{ outputs.crawl_sitemap.vars.broken_count }}
URL(s) failing in sitemap audit"
body: |
## Sitemap audit findings
**{{ outputs.crawl_sitemap.vars.broken_count }} of {{ outputs.crawl_sitemap.vars.checked }} URL(s)** checked from `{{ inputs.sitemap_url }}` are failing:
| URL | Status |
| --- | --- |
{% for item in outputs.crawl_sitemap.vars.broken %}| {{ item.url }} | {{ item.status }} |
{% endfor %}
Execution: {{ execution.id }}
Fix or remove the pages, then re-run the audit (or call the sitemap-audit webhook) to confirm they recover.
labels:
- seo
- broken-links
- id: alert_slack
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Tell the team how many pages are down and that a GitHub issue holds
the full list.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Sitemap audit: {{ outputs.crawl_sitemap.vars.broken_count }} of {{ outputs.crawl_sitemap.vars.checked }} URL(s) from {{ inputs.sitemap_url }} are broken or timing out. A GitHub issue has been opened in {{ inputs.repository }} with the list. Execution {{ execution.id }}."
}
else:
- id: log_clean
type: io.kestra.plugin.core.log.Log
description: Record the all-clear with the URL count, so a healthy audit still
shows up in the execution history.
message: "Sitemap audit clean: {{ outputs.crawl_sitemap.vars.checked }} URL(s)
checked from {{ inputs.sitemap_url }}, zero broken."
errors:
- id: alert_audit_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the audit itself could not run — unreachable sitemap,
parse error, rejected token — because a dead auditor must never be
mistaken for a clean bill of health.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": "Sitemap audit FAILED in flow {{ flow.id }} (execution {{ execution.id }}) for {{ inputs.sitemap_url }}. No link check was performed — verify the sitemap URL, Worker network access, and the GITHUB_TOKEN/SLACK_WEBHOOK_URL secrets."
}
outputs:
- id: audit_summary
type: JSON
description: 'The audit result — e.g. {"checked": 12, "broken_count": 1,
"broken": [{"url": "...", "status": 404}]} — ready for dashboards or a
follow-up gate.'
value: "{{ outputs.crawl_sitemap.vars | toJson }}"