id: huggingface-model-safetensors-integrity-guard
namespace: company.ai
description: Audit Hugging Face model repositories for safe serialization
formats (safetensors) and block insecure pickle weights before deployment.
triggers:
- id: daily_model_registry_audit
type: io.kestra.plugin.core.trigger.Schedule
description: Daily security audit of monitored model repositories.
cron: "0 8 * * 1-5"
- id: on_demand_webhook
type: io.kestra.plugin.core.trigger.Webhook
description: Trigger an on-demand audit from MLOps pipelines or model promotion gates.
key: hf-model-security-check
inputs:
- id: model_id
type: STRING
displayName: Hugging Face Model ID
defaults: google/gemma-2-2b
description: Target Hugging Face repository in owner/model format.
- id: allow_legacy_bin
type: BOOL
displayName: Allow Legacy PyTorch (.bin) Weights
defaults: false
description: When false, any presence of unverified .bin or .pkl pickle files
triggers a security alert.
- id: require_model_card
type: BOOL
displayName: Require Model Card (README.md)
defaults: true
description: Enforce presence of a documented model card.
tasks:
- id: fetch_model_metadata
type: io.kestra.plugin.core.http.Request
description: Query the Hugging Face Hub API for model tree and repository siblings.
uri: "https://huggingface.co/api/models/{{ inputs.model_id }}"
method: GET
headers:
Accept: "application/json"
User-Agent: "Kestra-HF-Safetensors-Guard"
Authorization: "Bearer {{ secret('HF_TOKEN') }}"
- id: audit_weights_security
type: io.kestra.plugin.scripts.python.Script
description: Analyze model files for unsafe serialization formats and missing
documentation.
containerImage: python:3.11-slim
inputFiles:
model_info.json: "{{ outputs.fetch_model_metadata.body | toJson }}"
script: |
import json
with open("model_info.json", "r") as f:
raw = f.read().strip()
data = json.loads(raw) if raw else {}
siblings = data.get("siblings", []) if isinstance(data, dict) else []
all_files = [s.get("rfilename", "") for s in siblings if s.get("rfilename")]
allow_bin = str("{{ inputs.allow_legacy_bin }}").lower() == "true"
require_card = str("{{ inputs.require_model_card }}").lower() == "true"
unsafe_extensions = (".bin", ".pt", ".pth", ".pkl", ".pickle", ".joblib")
safe_extensions = (".safetensors", ".gguf", ".onnx", ".tflite")
unsafe_files = []
safe_weight_files = []
has_config = False
has_readme = False
for filename in all_files:
lower = filename.lower()
if lower == "config.json":
has_config = True
elif lower == "readme.md":
has_readme = True
if lower.endswith(unsafe_extensions):
unsafe_files.append(filename)
elif lower.endswith(safe_extensions):
safe_weight_files.append(filename)
violations = []
if unsafe_files and not allow_bin:
violations.append(f"Unsafe pickle-based weights detected: {', '.join(unsafe_files)}")
if not safe_weight_files:
violations.append("No modern safe serialization weights (safetensors/GGUF/ONNX) found.")
if require_card and not has_readme:
violations.append("Missing README.md model card documentation.")
if not has_config:
violations.append("Missing config.json architecture definition.")
is_secure = len(violations) == 0
sha = data.get("sha", "unknown")
downloads = data.get("downloads", 0)
likes = data.get("likes", 0)
summary = {
"model_id": "{{ inputs.model_id }}",
"commit_sha": sha,
"is_secure": is_secure,
"total_files": len(all_files),
"safe_weight_files_count": len(safe_weight_files),
"unsafe_files_count": len(unsafe_files),
"unsafe_files": unsafe_files,
"safe_weight_files": safe_weight_files[:10],
"violations": violations,
"has_config": has_config,
"has_readme": has_readme,
"downloads": downloads,
"likes": likes
}
print(f"Audited {{ inputs.model_id }}: secure={is_secure}, {len(safe_weight_files)} safe weights, {len(unsafe_files)} unsafe files.")
print("::" + json.dumps({"outputs": summary}) + "::")
- id: evaluate_security_gate
type: io.kestra.plugin.core.flow.If
description: Gate deployment based on model serialization security and integrity.
condition: "{{ outputs.audit_weights_security.vars.is_secure == false }}"
then:
- id: alert_insecure_model
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert Slack SecOps when unverified or pickle-based weights are
detected.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": ":rotating_light: *Hugging Face Model Security Gate Triggered*\n*Model:* {{ inputs.model_id }}\n*Commit SHA:* `{{ outputs.audit_weights_security.vars.commit_sha }}`\n*Security Violations:* {{ outputs.audit_weights_security.vars.violations | toJson }}\n*Unsafe Files Detected:* {{ outputs.audit_weights_security.vars.unsafe_files | toJson }}\n\nAction: Model promotion halted. Verify weights or re-serialize with `safetensors` before deploying to inference clusters."
}
else:
- id: log_model_verified
type: io.kestra.plugin.core.log.Log
description: Log model verification when all safety criteria pass.
message: "Hugging Face model {{ inputs.model_id }} passed security audit (SHA:
{{ outputs.audit_weights_security.vars.commit_sha }}). Safe weights
count: {{ outputs.audit_weights_security.vars.safe_weight_files_count
}}. Ready for production deployment."
errors:
- id: alert_audit_failure
type: io.kestra.plugin.slack.notifications.SlackIncomingWebhook
description: Alert when the Hugging Face API or security analysis fails.
url: "{{ secret('SLACK_WEBHOOK_URL') }}"
payload: |
{
"text": ":warning: *Hugging Face Security Audit Failed*\n*Model:* {{ inputs.model_id }}\n*Execution:* {{ execution.id }}\nCheck HF_TOKEN permissions, model identifier, or rate limits."
}
outputs:
- id: audit_report
type: JSON
description: Complete security audit report containing safe weight count, commit
SHA, and violation status.
value: "{{ outputs.audit_weights_security.vars | toJson }}"