This repository has been archived on 2026-04-03. You can view files and clone it. You cannot open issues or pull requests or push a commit.
Files
core/chart/alerting/armature-rules.yaml
Jeffrey Smith f0dd43144e rebrand: Switchboard Core → Armature
- Rename Go module switchboard-core → armature (155+ files)
- Rename Docker image → gobha/armature
- Rename K8s resources, secrets, deployments
- Rename Prometheus metrics switchboard_* → armature_*
- Rename env vars SWITCHBOARD_ADMIN_* → ARMATURE_ADMIN_*
- Rename DB names switchboard_core* → armature*
- Update all frontend branding, notification templates, docs
- Update CI scripts, e2e tests, Keycloak realm, nginx conf
- Rename scripts/switchboard-ca.sh → scripts/armature-ca.sh
- Rename k8s/switchboard.yaml → k8s/armature.yaml
- Rename chart alerting/dashboard files
- Fix: DockerHub push uses env: binding for secret injection
- Helm chart updated (name, labels, template functions, dashboard, alerting)
- Replace favicon/icon assets with Armature brand

No functional changes. Pure mechanical rename + CI fix.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-03-31 21:39:58 +00:00

70 lines
2.3 KiB
YAML

# Armature Core — PrometheusRule alerts (v0.33.0)
# Source file for the Helm template. Deploy via:
# monitoring.prometheusRule.enabled: true
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: armature-alerts
spec:
groups:
- name: armature.rules
rules:
# Pod restart (possible OOM)
- alert: ArmaturePodRestart
expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0
for: 0m
labels:
severity: warning
annotations:
summary: "Armature backend pod restarted (possible OOM)"
description: "Container {{ $labels.container }} in pod {{ $labels.pod }} restarted."
# Provider down for 5+ minutes
- alert: ArmatureProviderDown
expr: armature_provider_status > 2
for: 5m
labels:
severity: critical
annotations:
summary: "Provider {{ $labels.provider_config_id }} is down"
# DB pool >80% utilized
- alert: ArmatureDBPoolExhaustion
expr: armature_db_in_use_connections / armature_db_open_connections > 0.8
for: 5m
labels:
severity: warning
annotations:
summary: "DB connection pool >80% utilized"
# HTTP 5xx error rate >5%
- alert: ArmatureHighErrorRate
expr: >
sum(rate(armature_http_requests_total{status=~"5.."}[5m]))
/ sum(rate(armature_http_requests_total[5m])) > 0.05
for: 5m
labels:
severity: warning
annotations:
summary: "HTTP 5xx error rate exceeds 5%"
# Task failure rate >25%
- alert: ArmatureTaskFailureRate
expr: >
rate(armature_task_executions_total{status="error"}[15m])
/ rate(armature_task_executions_total[15m]) > 0.25
for: 10m
labels:
severity: warning
annotations:
summary: "Task failure rate exceeds 25% over 15 minutes"
# No completions processed in 15 minutes (canary)
- alert: ArmatureNoCompletions
expr: sum(rate(armature_completions_total[10m])) == 0
for: 15m
labels:
severity: critical
annotations:
summary: "No completions processed in 15 minutes"