Changeset 0.33.0 (#207)

Co-authored-by: Jeffrey Smith <jasafpro@gmail.com>
Co-committed-by: Jeffrey Smith <jasafpro@gmail.com>
This commit is contained in:
2026-03-19 21:37:32 +00:00
committed by xcaliber
parent b1266b0d7c
commit ed3e9363f2
42 changed files with 2527 additions and 129 deletions

View File

@@ -3,7 +3,7 @@ name: switchboard
description: Chat Switchboard — self-hosted enterprise AI chat platform
type: application
version: 0.1.0
appVersion: "0.28.6"
appVersion: "0.0.0" # Patched at CI/release time from /VERSION
home: https://gobha.ai
sources:
- https://git.gobha.me/xcaliber/chat-switchboard

View File

@@ -0,0 +1,69 @@
# Chat Switchboard — PrometheusRule alerts (v0.33.0)
# Source file for the Helm template. Deploy via:
# monitoring.prometheusRule.enabled: true
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: switchboard-alerts
spec:
groups:
- name: switchboard.rules
rules:
# Pod restart (possible OOM)
- alert: SwitchboardPodRestart
expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0
for: 0m
labels:
severity: warning
annotations:
summary: "Switchboard backend pod restarted (possible OOM)"
description: "Container {{ $labels.container }} in pod {{ $labels.pod }} restarted."
# Provider down for 5+ minutes
- alert: SwitchboardProviderDown
expr: switchboard_provider_status > 2
for: 5m
labels:
severity: critical
annotations:
summary: "Provider {{ $labels.provider_config_id }} is down"
# DB pool >80% utilized
- alert: SwitchboardDBPoolExhaustion
expr: switchboard_db_in_use_connections / switchboard_db_open_connections > 0.8
for: 5m
labels:
severity: warning
annotations:
summary: "DB connection pool >80% utilized"
# HTTP 5xx error rate >5%
- alert: SwitchboardHighErrorRate
expr: >
sum(rate(switchboard_http_requests_total{status=~"5.."}[5m]))
/ sum(rate(switchboard_http_requests_total[5m])) > 0.05
for: 5m
labels:
severity: warning
annotations:
summary: "HTTP 5xx error rate exceeds 5%"
# Task failure rate >25%
- alert: SwitchboardTaskFailureRate
expr: >
rate(switchboard_task_executions_total{status="error"}[15m])
/ rate(switchboard_task_executions_total[15m]) > 0.25
for: 10m
labels:
severity: warning
annotations:
summary: "Task failure rate exceeds 25% over 15 minutes"
# No completions processed in 15 minutes (canary)
- alert: SwitchboardNoCompletions
expr: sum(rate(switchboard_completions_total[10m])) == 0
for: 15m
labels:
severity: critical
annotations:
summary: "No completions processed in 15 minutes"

View File

@@ -0,0 +1,207 @@
{
"__inputs": [
{
"name": "DS_PROMETHEUS",
"label": "Prometheus",
"description": "",
"type": "datasource",
"pluginId": "prometheus",
"pluginName": "Prometheus"
}
],
"__requires": [
{ "type": "grafana", "id": "grafana", "name": "Grafana", "version": "9.0.0" },
{ "type": "datasource", "id": "prometheus", "name": "Prometheus", "version": "1.0.0" },
{ "type": "panel", "id": "timeseries", "name": "Time series", "version": "" },
{ "type": "panel", "id": "stat", "name": "Stat", "version": "" },
{ "type": "panel", "id": "gauge", "name": "Gauge", "version": "" }
],
"id": null,
"uid": "switchboard-overview",
"title": "Chat Switchboard — Overview",
"description": "System overview: request rates, latency, provider health, token usage, DB pool.",
"tags": ["switchboard"],
"timezone": "browser",
"refresh": "30s",
"schemaVersion": 38,
"version": 1,
"templating": {
"list": [
{
"name": "datasource",
"type": "datasource",
"query": "prometheus",
"current": { "text": "Prometheus", "value": "Prometheus" }
},
{
"name": "namespace",
"type": "query",
"datasource": { "type": "prometheus", "uid": "${datasource}" },
"query": "label_values(switchboard_http_requests_total, namespace)",
"includeAll": true,
"current": { "text": "All", "value": "$__all" }
},
{
"name": "pod",
"type": "query",
"datasource": { "type": "prometheus", "uid": "${datasource}" },
"query": "label_values(switchboard_http_requests_total{namespace=~\"$namespace\"}, pod)",
"includeAll": true,
"current": { "text": "All", "value": "$__all" }
}
]
},
"panels": [
{
"title": "Request Rate",
"type": "timeseries",
"gridPos": { "h": 8, "w": 6, "x": 0, "y": 0 },
"targets": [
{
"expr": "sum(rate(switchboard_http_requests_total{namespace=~\"$namespace\"}[5m]))",
"legendFormat": "Total req/s"
}
]
},
{
"title": "Error Rate (5xx)",
"type": "timeseries",
"gridPos": { "h": 8, "w": 6, "x": 6, "y": 0 },
"targets": [
{
"expr": "sum(rate(switchboard_http_requests_total{namespace=~\"$namespace\",status=~\"5..\"}[5m])) / sum(rate(switchboard_http_requests_total{namespace=~\"$namespace\"}[5m]))",
"legendFormat": "5xx rate"
}
],
"fieldConfig": {
"defaults": {
"unit": "percentunit",
"max": 1,
"thresholds": {
"steps": [
{ "color": "green", "value": null },
{ "color": "yellow", "value": 0.01 },
{ "color": "red", "value": 0.05 }
]
}
}
}
},
{
"title": "Request Latency (p50 / p95 / p99)",
"type": "timeseries",
"gridPos": { "h": 8, "w": 6, "x": 12, "y": 0 },
"targets": [
{
"expr": "histogram_quantile(0.50, sum(rate(switchboard_http_request_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])) by (le))",
"legendFormat": "p50"
},
{
"expr": "histogram_quantile(0.95, sum(rate(switchboard_http_request_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])) by (le))",
"legendFormat": "p95"
},
{
"expr": "histogram_quantile(0.99, sum(rate(switchboard_http_request_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])) by (le))",
"legendFormat": "p99"
}
],
"fieldConfig": { "defaults": { "unit": "s" } }
},
{
"title": "WebSocket Connections",
"type": "stat",
"gridPos": { "h": 8, "w": 6, "x": 18, "y": 0 },
"targets": [
{
"expr": "sum(switchboard_websocket_connections{namespace=~\"$namespace\"})",
"legendFormat": "Active"
}
]
},
{
"title": "Completion Rate by Provider",
"type": "timeseries",
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 8 },
"targets": [
{
"expr": "sum by (provider_config_id) (rate(switchboard_completions_total{namespace=~\"$namespace\"}[5m]))",
"legendFormat": "{{provider_config_id}}"
}
]
},
{
"title": "Completion Latency p95 by Provider",
"type": "timeseries",
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 8 },
"targets": [
{
"expr": "histogram_quantile(0.95, sum by (provider_config_id, le) (rate(switchboard_completion_duration_seconds_bucket{namespace=~\"$namespace\"}[5m])))",
"legendFormat": "{{provider_config_id}}"
}
],
"fieldConfig": { "defaults": { "unit": "s" } }
},
{
"title": "Tokens/min by Model",
"type": "timeseries",
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 16 },
"targets": [
{
"expr": "sum by (model_id) (rate(switchboard_completion_tokens_total{namespace=~\"$namespace\"}[5m])) * 60",
"legendFormat": "{{model_id}}"
}
]
},
{
"title": "Provider Status",
"type": "stat",
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 16 },
"targets": [
{
"expr": "switchboard_provider_status{namespace=~\"$namespace\"}",
"legendFormat": "{{provider_config_id}}"
}
],
"fieldConfig": {
"defaults": {
"mappings": [
{ "type": "value", "options": { "0": { "text": "Unknown", "color": "text" } } },
{ "type": "value", "options": { "1": { "text": "Healthy", "color": "green" } } },
{ "type": "value", "options": { "2": { "text": "Degraded", "color": "yellow" } } },
{ "type": "value", "options": { "3": { "text": "Down", "color": "red" } } }
]
}
}
},
{
"title": "DB Connection Pool",
"type": "timeseries",
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 24 },
"targets": [
{
"expr": "switchboard_db_open_connections{namespace=~\"$namespace\"}",
"legendFormat": "Open"
},
{
"expr": "switchboard_db_in_use_connections{namespace=~\"$namespace\"}",
"legendFormat": "In Use"
},
{
"expr": "switchboard_db_idle_connections{namespace=~\"$namespace\"}",
"legendFormat": "Idle"
}
]
},
{
"title": "Task Executions",
"type": "timeseries",
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 24 },
"targets": [
{
"expr": "sum by (status) (rate(switchboard_task_executions_total{namespace=~\"$namespace\"}[5m]))",
"legendFormat": "{{status}}"
}
]
}
]
}

View File

@@ -19,6 +19,8 @@ data:
EXTRACTION_CONCURRENCY: {{ .Values.extraction.concurrency | quote }}
WORKSPACE_INDEXING_ENABLED: {{ .Values.workspace.indexingEnabled | quote }}
WORKSPACE_INDEX_CONCURRENCY: {{ .Values.workspace.indexConcurrency | quote }}
LOG_FORMAT: {{ .Values.logging.format | default "text" | quote }}
LOG_LEVEL: {{ .Values.logging.level | default "info" | quote }}
CORS_ALLOWED_ORIGINS: {{ .Values.corsAllowedOrigins | quote }}
{{- if .Values.storage.s3.endpoint }}
S3_ENDPOINT: {{ .Values.storage.s3.endpoint | quote }}

View File

@@ -0,0 +1,14 @@
{{- if .Values.monitoring.grafanaDashboard.enabled }}
apiVersion: v1
kind: ConfigMap
metadata:
name: {{ include "switchboard.fullname" . }}-grafana-dashboard
labels:
{{- include "switchboard.labels" . | nindent 4 }}
{{- with .Values.monitoring.grafanaDashboard.labels }}
{{- toYaml . | nindent 4 }}
{{- end }}
data:
switchboard-overview.json: |-
{{- .Files.Get "dashboards/switchboard-overview.json" | nindent 4 }}
{{- end }}

View File

@@ -0,0 +1,61 @@
{{- if .Values.monitoring.prometheusRule.enabled }}
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: {{ include "switchboard.fullname" . }}-alerts
labels:
{{- include "switchboard.labels" . | nindent 4 }}
{{- with .Values.monitoring.prometheusRule.labels }}
{{- toYaml . | nindent 4 }}
{{- end }}
spec:
groups:
- name: switchboard.rules
rules:
- alert: SwitchboardPodRestart
expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0
for: 0m
labels:
severity: warning
annotations:
summary: "Switchboard backend pod restarted (possible OOM)"
- alert: SwitchboardProviderDown
expr: switchboard_provider_status > 2
for: 5m
labels:
severity: critical
annotations:
summary: "Provider {{`{{ $labels.provider_config_id }}`}} is down"
- alert: SwitchboardDBPoolExhaustion
expr: switchboard_db_in_use_connections / switchboard_db_open_connections > 0.8
for: 5m
labels:
severity: warning
annotations:
summary: "DB connection pool >80% utilized"
- alert: SwitchboardHighErrorRate
expr: |
sum(rate(switchboard_http_requests_total{status=~"5.."}[5m]))
/ sum(rate(switchboard_http_requests_total[5m])) > 0.05
for: 5m
labels:
severity: warning
annotations:
summary: "HTTP 5xx error rate exceeds 5%"
- alert: SwitchboardTaskFailureRate
expr: |
rate(switchboard_task_executions_total{status="error"}[15m])
/ rate(switchboard_task_executions_total[15m]) > 0.25
for: 10m
labels:
severity: warning
annotations:
summary: "Task failure rate exceeds 25%"
- alert: SwitchboardNoCompletions
expr: sum(rate(switchboard_completions_total[10m])) == 0
for: 15m
labels:
severity: critical
annotations:
summary: "No completions processed in 15 minutes"
{{- end }}

View File

@@ -0,0 +1,20 @@
{{- if .Values.monitoring.serviceMonitor.enabled }}
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: {{ include "switchboard.fullname" . }}
labels:
{{- include "switchboard.labels" . | nindent 4 }}
{{- with .Values.monitoring.serviceMonitor.labels }}
{{- toYaml . | nindent 4 }}
{{- end }}
spec:
selector:
matchLabels:
{{- include "switchboard.selectorLabels" . | nindent 6 }}
app.kubernetes.io/component: backend
endpoints:
- port: http
path: {{ .Values.monitoring.serviceMonitor.path | default "/metrics" }}
interval: {{ .Values.monitoring.serviceMonitor.interval | default "30s" }}
{{- end }}

View File

@@ -4,6 +4,10 @@ metadata:
name: {{ .Release.Name }}-backend
labels:
{{- include "switchboard.backend.labels" . | nindent 4 }}
annotations:
prometheus.io/scrape: "true"
prometheus.io/port: {{ .Values.backend.port | quote }}
prometheus.io/path: "/metrics"
spec:
type: ClusterIP
ports:

View File

@@ -132,6 +132,31 @@ workspace:
indexingEnabled: true
indexConcurrency: 2
# ── Logging (v0.33.0) ─────────────────────
logging:
format: text # "text" (human-readable) or "json" (structured)
level: info # "debug", "info", "warn", "error"
# ── Monitoring (v0.33.0) ──────────────────
# All monitoring resources are opt-in (disabled by default).
monitoring:
# ServiceMonitor for Prometheus Operator (kube-prometheus-stack)
serviceMonitor:
enabled: false
interval: 30s
path: /metrics
labels: {} # match your Prometheus Operator selector
# Grafana dashboard ConfigMap (auto-discovered by Grafana sidecar)
grafanaDashboard:
enabled: false
labels:
grafana_dashboard: "1" # default sidecar label
# PrometheusRule alerts
prometheusRule:
enabled: false
labels:
release: prometheus # match kube-prometheus-stack
# ── CORS ───────────────────────────────────
corsAllowedOrigins: "*"