{{- if .Values.monitoring.prometheusRule.enabled }} apiVersion: monitoring.coreos.com/v1 kind: PrometheusRule metadata: name: {{ include "armature.fullname" . }}-alerts labels: {{- include "armature.labels" . | nindent 4 }} {{- with .Values.monitoring.prometheusRule.labels }} {{- toYaml . | nindent 4 }} {{- end }} spec: groups: - name: armature.rules rules: - alert: ArmaturePodRestart expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0 for: 0m labels: severity: warning annotations: summary: "Armature backend pod restarted (possible OOM)" - alert: ArmatureProviderDown expr: armature_provider_status > 2 for: 5m labels: severity: critical annotations: summary: "Provider {{`{{ $labels.provider_config_id }}`}} is down" - alert: ArmatureDBPoolExhaustion expr: armature_db_in_use_connections / armature_db_open_connections > 0.8 for: 5m labels: severity: warning annotations: summary: "DB connection pool >80% utilized" - alert: ArmatureHighErrorRate expr: | sum(rate(armature_http_requests_total{status=~"5.."}[5m])) / sum(rate(armature_http_requests_total[5m])) > 0.05 for: 5m labels: severity: warning annotations: summary: "HTTP 5xx error rate exceeds 5%" - alert: ArmatureTaskFailureRate expr: | rate(armature_task_executions_total{status="error"}[15m]) / rate(armature_task_executions_total[15m]) > 0.25 for: 10m labels: severity: warning annotations: summary: "Task failure rate exceeds 25%" - alert: ArmatureNoCompletions expr: sum(rate(armature_completions_total[10m])) == 0 for: 15m labels: severity: critical annotations: summary: "No completions processed in 15 minutes" {{- end }}