# Armature Core — PrometheusRule alerts (v0.33.0) # Source file for the Helm template. Deploy via: # monitoring.prometheusRule.enabled: true apiVersion: monitoring.coreos.com/v1 kind: PrometheusRule metadata: name: armature-alerts spec: groups: - name: armature.rules rules: # Pod restart (possible OOM) - alert: ArmaturePodRestart expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0 for: 0m labels: severity: warning annotations: summary: "Armature backend pod restarted (possible OOM)" description: "Container {{ $labels.container }} in pod {{ $labels.pod }} restarted." # Provider down for 5+ minutes - alert: ArmatureProviderDown expr: armature_provider_status > 2 for: 5m labels: severity: critical annotations: summary: "Provider {{ $labels.provider_config_id }} is down" # DB pool >80% utilized - alert: ArmatureDBPoolExhaustion expr: armature_db_in_use_connections / armature_db_open_connections > 0.8 for: 5m labels: severity: warning annotations: summary: "DB connection pool >80% utilized" # HTTP 5xx error rate >5% - alert: ArmatureHighErrorRate expr: > sum(rate(armature_http_requests_total{status=~"5.."}[5m])) / sum(rate(armature_http_requests_total[5m])) > 0.05 for: 5m labels: severity: warning annotations: summary: "HTTP 5xx error rate exceeds 5%" # Task failure rate >25% - alert: ArmatureTaskFailureRate expr: > rate(armature_task_executions_total{status="error"}[15m]) / rate(armature_task_executions_total[15m]) > 0.25 for: 10m labels: severity: warning annotations: summary: "Task failure rate exceeds 25% over 15 minutes" # No completions processed in 15 minutes (canary) - alert: ArmatureNoCompletions expr: sum(rate(armature_completions_total[10m])) == 0 for: 15m labels: severity: critical annotations: summary: "No completions processed in 15 minutes"