Changeset 0.33.0 (#207)
Co-authored-by: Jeffrey Smith <jasafpro@gmail.com> Co-committed-by: Jeffrey Smith <jasafpro@gmail.com>
This commit is contained in:
@@ -3,7 +3,7 @@ name: switchboard
|
||||
description: Chat Switchboard — self-hosted enterprise AI chat platform
|
||||
type: application
|
||||
version: 0.1.0
|
||||
appVersion: "0.28.6"
|
||||
appVersion: "0.0.0" # Patched at CI/release time from /VERSION
|
||||
home: https://gobha.ai
|
||||
sources:
|
||||
- https://git.gobha.me/xcaliber/chat-switchboard
|
||||
|
||||
69
chart/alerting/switchboard-rules.yaml
Normal file
69
chart/alerting/switchboard-rules.yaml
Normal file
@@ -0,0 +1,69 @@
|
||||
# Chat Switchboard — PrometheusRule alerts (v0.33.0)
|
||||
# Source file for the Helm template. Deploy via:
|
||||
# monitoring.prometheusRule.enabled: true
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: switchboard-alerts
|
||||
spec:
|
||||
groups:
|
||||
- name: switchboard.rules
|
||||
rules:
|
||||
# Pod restart (possible OOM)
|
||||
- alert: SwitchboardPodRestart
|
||||
expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0
|
||||
for: 0m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Switchboard backend pod restarted (possible OOM)"
|
||||
description: "Container {{ $labels.container }} in pod {{ $labels.pod }} restarted."
|
||||
|
||||
# Provider down for 5+ minutes
|
||||
- alert: SwitchboardProviderDown
|
||||
expr: switchboard_provider_status > 2
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Provider {{ $labels.provider_config_id }} is down"
|
||||
|
||||
# DB pool >80% utilized
|
||||
- alert: SwitchboardDBPoolExhaustion
|
||||
expr: switchboard_db_in_use_connections / switchboard_db_open_connections > 0.8
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "DB connection pool >80% utilized"
|
||||
|
||||
# HTTP 5xx error rate >5%
|
||||
- alert: SwitchboardHighErrorRate
|
||||
expr: >
|
||||
sum(rate(switchboard_http_requests_total{status=~"5.."}[5m]))
|
||||
/ sum(rate(switchboard_http_requests_total[5m])) > 0.05
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "HTTP 5xx error rate exceeds 5%"
|
||||
|
||||
# Task failure rate >25%
|
||||
- alert: SwitchboardTaskFailureRate
|
||||
expr: >
|
||||
rate(switchboard_task_executions_total{status="error"}[15m])
|
||||
/ rate(switchboard_task_executions_total[15m]) > 0.25
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Task failure rate exceeds 25% over 15 minutes"
|
||||
|
||||
# No completions processed in 15 minutes (canary)
|
||||
- alert: SwitchboardNoCompletions
|
||||
expr: sum(rate(switchboard_completions_total[10m])) == 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "No completions processed in 15 minutes"
|
||||
207
chart/dashboards/switchboard-overview.json
Normal file
207
chart/dashboards/switchboard-overview.json
Normal file
@@ -0,0 +1,207 @@
|
||||
{
|
||||
"__inputs": [
|
||||
{
|
||||
"name": "DS_PROMETHEUS",
|
||||
"label": "Prometheus",
|
||||
"description": "",
|
||||
"type": "datasource",
|
||||
"pluginId": "prometheus",
|
||||
"pluginName": "Prometheus"
|
||||
}
|
||||
],
|
||||
"__requires": [
|
||||
{ "type": "grafana", "id": "grafana", "name": "Grafana", "version": "9.0.0" },
|
||||
{ "type": "datasource", "id": "prometheus", "name": "Prometheus", "version": "1.0.0" },
|
||||
{ "type": "panel", "id": "timeseries", "name": "Time series", "version": "" },
|
||||
{ "type": "panel", "id": "stat", "name": "Stat", "version": "" },
|
||||
{ "type": "panel", "id": "gauge", "name": "Gauge", "version": "" }
|
||||
],
|
||||
"id": null,
|
||||
"uid": "switchboard-overview",
|
||||
"title": "Chat Switchboard — Overview",
|
||||
"description": "System overview: request rates, latency, provider health, token usage, DB pool.",
|
||||
"tags": ["switchboard"],
|
||||
"timezone": "browser",
|
||||
"refresh": "30s",
|
||||
"schemaVersion": 38,
|
||||
"version": 1,
|
||||
"templating": {
|
||||
"list": [
|
||||
{
|
||||
"name": "datasource",
|
||||
"type": "datasource",
|
||||
"query": "prometheus",
|
||||
"current": { "text": "Prometheus", "value": "Prometheus" }
|
||||
},
|
||||
{
|
||||
"name": "namespace",
|
||||
"type": "query",
|
||||
"datasource": { "type": "prometheus", "uid": "${datasource}" },
|
||||
"query": "label_values(switchboard_http_requests_total, namespace)",
|
||||
"includeAll": true,
|
||||
"current": { "text": "All", "value": "$__all" }
|
||||
},
|
||||
{
|
||||
"name": "pod",
|
||||
"type": "query",
|
||||
"datasource": { "type": "prometheus", "uid": "${datasource}" },
|
||||
"query": "label_values(switchboard_http_requests_total{namespace=~\"$namespace\"}, pod)",
|
||||
"includeAll": true,
|
||||
"current": { "text": "All", "value": "$__all" }
|
||||
}
|
||||
]
|
||||
},
|
||||
"panels": [
|
||||
{
|
||||
"title": "Request Rate",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 6, "x": 0, "y": 0 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "sum(rate(switchboard_http_requests_total{namespace=~\"$namespace\"}[5m]))",
|
||||
"legendFormat": "Total req/s"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Error Rate (5xx)",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 6, "x": 6, "y": 0 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "sum(rate(switchboard_http_requests_total{namespace=~\"$namespace\",status=~\"5..\"}[5m])) / sum(rate(switchboard_http_requests_total{namespace=~\"$namespace\"}[5m]))",
|
||||
"legendFormat": "5xx rate"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "percentunit",
|
||||
"max": 1,
|
||||
"thresholds": {
|
||||
"steps": [
|
||||
{ "color": "green", "value": null },
|
||||
{ "color": "yellow", "value": 0.01 },
|
||||
{ "color": "red", "value": 0.05 }
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"title": "Request Latency (p50 / p95 / p99)",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 6, "x": 12, "y": 0 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "histogram_quantile(0.50, sum(rate(switchboard_http_request_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])) by (le))",
|
||||
"legendFormat": "p50"
|
||||
},
|
||||
{
|
||||
"expr": "histogram_quantile(0.95, sum(rate(switchboard_http_request_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])) by (le))",
|
||||
"legendFormat": "p95"
|
||||
},
|
||||
{
|
||||
"expr": "histogram_quantile(0.99, sum(rate(switchboard_http_request_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])) by (le))",
|
||||
"legendFormat": "p99"
|
||||
}
|
||||
],
|
||||
"fieldConfig": { "defaults": { "unit": "s" } }
|
||||
},
|
||||
{
|
||||
"title": "WebSocket Connections",
|
||||
"type": "stat",
|
||||
"gridPos": { "h": 8, "w": 6, "x": 18, "y": 0 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "sum(switchboard_websocket_connections{namespace=~\"$namespace\"})",
|
||||
"legendFormat": "Active"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Completion Rate by Provider",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 8 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "sum by (provider_config_id) (rate(switchboard_completions_total{namespace=~\"$namespace\"}[5m]))",
|
||||
"legendFormat": "{{provider_config_id}}"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Completion Latency p95 by Provider",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 8 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "histogram_quantile(0.95, sum by (provider_config_id, le) (rate(switchboard_completion_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])))",
|
||||
"legendFormat": "{{provider_config_id}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": { "defaults": { "unit": "s" } }
|
||||
},
|
||||
{
|
||||
"title": "Tokens/min by Model",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 16 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "sum by (model_id) (rate(switchboard_completion_tokens_total{namespace=~\"$namespace\"}[5m])) * 60",
|
||||
"legendFormat": "{{model_id}}"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Provider Status",
|
||||
"type": "stat",
|
||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 16 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "switchboard_provider_status{namespace=~\"$namespace\"}",
|
||||
"legendFormat": "{{provider_config_id}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"mappings": [
|
||||
{ "type": "value", "options": { "0": { "text": "Unknown", "color": "text" } } },
|
||||
{ "type": "value", "options": { "1": { "text": "Healthy", "color": "green" } } },
|
||||
{ "type": "value", "options": { "2": { "text": "Degraded", "color": "yellow" } } },
|
||||
{ "type": "value", "options": { "3": { "text": "Down", "color": "red" } } }
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"title": "DB Connection Pool",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 24 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "switchboard_db_open_connections{namespace=~\"$namespace\"}",
|
||||
"legendFormat": "Open"
|
||||
},
|
||||
{
|
||||
"expr": "switchboard_db_in_use_connections{namespace=~\"$namespace\"}",
|
||||
"legendFormat": "In Use"
|
||||
},
|
||||
{
|
||||
"expr": "switchboard_db_idle_connections{namespace=~\"$namespace\"}",
|
||||
"legendFormat": "Idle"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Task Executions",
|
||||
"type": "timeseries",
|
||||
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 24 },
|
||||
"targets": [
|
||||
{
|
||||
"expr": "sum by (status) (rate(switchboard_task_executions_total{namespace=~\"$namespace\"}[5m]))",
|
||||
"legendFormat": "{{status}}"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -19,6 +19,8 @@ data:
|
||||
EXTRACTION_CONCURRENCY: {{ .Values.extraction.concurrency | quote }}
|
||||
WORKSPACE_INDEXING_ENABLED: {{ .Values.workspace.indexingEnabled | quote }}
|
||||
WORKSPACE_INDEX_CONCURRENCY: {{ .Values.workspace.indexConcurrency | quote }}
|
||||
LOG_FORMAT: {{ .Values.logging.format | default "text" | quote }}
|
||||
LOG_LEVEL: {{ .Values.logging.level | default "info" | quote }}
|
||||
CORS_ALLOWED_ORIGINS: {{ .Values.corsAllowedOrigins | quote }}
|
||||
{{- if .Values.storage.s3.endpoint }}
|
||||
S3_ENDPOINT: {{ .Values.storage.s3.endpoint | quote }}
|
||||
|
||||
14
chart/templates/grafana-dashboard-configmap.yaml
Normal file
14
chart/templates/grafana-dashboard-configmap.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
{{- if .Values.monitoring.grafanaDashboard.enabled }}
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: {{ include "switchboard.fullname" . }}-grafana-dashboard
|
||||
labels:
|
||||
{{- include "switchboard.labels" . | nindent 4 }}
|
||||
{{- with .Values.monitoring.grafanaDashboard.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
data:
|
||||
switchboard-overview.json: |-
|
||||
{{- .Files.Get "dashboards/switchboard-overview.json" | nindent 4 }}
|
||||
{{- end }}
|
||||
61
chart/templates/prometheusrule.yaml
Normal file
61
chart/templates/prometheusrule.yaml
Normal file
@@ -0,0 +1,61 @@
|
||||
{{- if .Values.monitoring.prometheusRule.enabled }}
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: {{ include "switchboard.fullname" . }}-alerts
|
||||
labels:
|
||||
{{- include "switchboard.labels" . | nindent 4 }}
|
||||
{{- with .Values.monitoring.prometheusRule.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
groups:
|
||||
- name: switchboard.rules
|
||||
rules:
|
||||
- alert: SwitchboardPodRestart
|
||||
expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0
|
||||
for: 0m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Switchboard backend pod restarted (possible OOM)"
|
||||
- alert: SwitchboardProviderDown
|
||||
expr: switchboard_provider_status > 2
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Provider {{`{{ $labels.provider_config_id }}`}} is down"
|
||||
- alert: SwitchboardDBPoolExhaustion
|
||||
expr: switchboard_db_in_use_connections / switchboard_db_open_connections > 0.8
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "DB connection pool >80% utilized"
|
||||
- alert: SwitchboardHighErrorRate
|
||||
expr: |
|
||||
sum(rate(switchboard_http_requests_total{status=~"5.."}[5m]))
|
||||
/ sum(rate(switchboard_http_requests_total[5m])) > 0.05
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "HTTP 5xx error rate exceeds 5%"
|
||||
- alert: SwitchboardTaskFailureRate
|
||||
expr: |
|
||||
rate(switchboard_task_executions_total{status="error"}[15m])
|
||||
/ rate(switchboard_task_executions_total[15m]) > 0.25
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Task failure rate exceeds 25%"
|
||||
- alert: SwitchboardNoCompletions
|
||||
expr: sum(rate(switchboard_completions_total[10m])) == 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "No completions processed in 15 minutes"
|
||||
{{- end }}
|
||||
20
chart/templates/servicemonitor.yaml
Normal file
20
chart/templates/servicemonitor.yaml
Normal file
@@ -0,0 +1,20 @@
|
||||
{{- if .Values.monitoring.serviceMonitor.enabled }}
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: ServiceMonitor
|
||||
metadata:
|
||||
name: {{ include "switchboard.fullname" . }}
|
||||
labels:
|
||||
{{- include "switchboard.labels" . | nindent 4 }}
|
||||
{{- with .Values.monitoring.serviceMonitor.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
{{- include "switchboard.selectorLabels" . | nindent 6 }}
|
||||
app.kubernetes.io/component: backend
|
||||
endpoints:
|
||||
- port: http
|
||||
path: {{ .Values.monitoring.serviceMonitor.path | default "/metrics" }}
|
||||
interval: {{ .Values.monitoring.serviceMonitor.interval | default "30s" }}
|
||||
{{- end }}
|
||||
@@ -4,6 +4,10 @@ metadata:
|
||||
name: {{ .Release.Name }}-backend
|
||||
labels:
|
||||
{{- include "switchboard.backend.labels" . | nindent 4 }}
|
||||
annotations:
|
||||
prometheus.io/scrape: "true"
|
||||
prometheus.io/port: {{ .Values.backend.port | quote }}
|
||||
prometheus.io/path: "/metrics"
|
||||
spec:
|
||||
type: ClusterIP
|
||||
ports:
|
||||
|
||||
@@ -132,6 +132,31 @@ workspace:
|
||||
indexingEnabled: true
|
||||
indexConcurrency: 2
|
||||
|
||||
# ── Logging (v0.33.0) ─────────────────────
|
||||
logging:
|
||||
format: text # "text" (human-readable) or "json" (structured)
|
||||
level: info # "debug", "info", "warn", "error"
|
||||
|
||||
# ── Monitoring (v0.33.0) ──────────────────
|
||||
# All monitoring resources are opt-in (disabled by default).
|
||||
monitoring:
|
||||
# ServiceMonitor for Prometheus Operator (kube-prometheus-stack)
|
||||
serviceMonitor:
|
||||
enabled: false
|
||||
interval: 30s
|
||||
path: /metrics
|
||||
labels: {} # match your Prometheus Operator selector
|
||||
# Grafana dashboard ConfigMap (auto-discovered by Grafana sidecar)
|
||||
grafanaDashboard:
|
||||
enabled: false
|
||||
labels:
|
||||
grafana_dashboard: "1" # default sidecar label
|
||||
# PrometheusRule alerts
|
||||
prometheusRule:
|
||||
enabled: false
|
||||
labels:
|
||||
release: prometheus # match kube-prometheus-stack
|
||||
|
||||
# ── CORS ───────────────────────────────────
|
||||
corsAllowedOrigins: "*"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user