{{- if .Values.monitoring.prometheusRule.enabled }} apiVersion: monitoring.coreos.com/v1 kind: PrometheusRule metadata: name: {{ include "switchboard.fullname" . }}-alerts labels: {{- include "switchboard.labels" . | nindent 4 }} {{- with .Values.monitoring.prometheusRule.labels }} {{- toYaml . | nindent 4 }} {{- end }} spec: groups: - name: switchboard.rules rules: - alert: SwitchboardPodRestart expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0 for: 0m labels: severity: warning annotations: summary: "Switchboard backend pod restarted (possible OOM)" - alert: SwitchboardProviderDown expr: switchboard_provider_status > 2 for: 5m labels: severity: critical annotations: summary: "Provider {{`{{ $labels.provider_config_id }}`}} is down" - alert: SwitchboardDBPoolExhaustion expr: switchboard_db_in_use_connections / switchboard_db_open_connections > 0.8 for: 5m labels: severity: warning annotations: summary: "DB connection pool >80% utilized" - alert: SwitchboardHighErrorRate expr: | sum(rate(switchboard_http_requests_total{status=~"5.."}[5m])) / sum(rate(switchboard_http_requests_total[5m])) > 0.05 for: 5m labels: severity: warning annotations: summary: "HTTP 5xx error rate exceeds 5%" - alert: SwitchboardTaskFailureRate expr: | rate(switchboard_task_executions_total{status="error"}[15m]) / rate(switchboard_task_executions_total[15m]) > 0.25 for: 10m labels: severity: warning annotations: summary: "Task failure rate exceeds 25%" - alert: SwitchboardNoCompletions expr: sum(rate(switchboard_completions_total[10m])) == 0 for: 15m labels: severity: critical annotations: summary: "No completions processed in 15 minutes" {{- end }}