Changeset 0.33.0 (#207)
Co-authored-by: Jeffrey Smith <jasafpro@gmail.com> Co-committed-by: Jeffrey Smith <jasafpro@gmail.com>
This commit is contained in:
@@ -19,6 +19,8 @@ data:
|
||||
EXTRACTION_CONCURRENCY: {{ .Values.extraction.concurrency | quote }}
|
||||
WORKSPACE_INDEXING_ENABLED: {{ .Values.workspace.indexingEnabled | quote }}
|
||||
WORKSPACE_INDEX_CONCURRENCY: {{ .Values.workspace.indexConcurrency | quote }}
|
||||
LOG_FORMAT: {{ .Values.logging.format | default "text" | quote }}
|
||||
LOG_LEVEL: {{ .Values.logging.level | default "info" | quote }}
|
||||
CORS_ALLOWED_ORIGINS: {{ .Values.corsAllowedOrigins | quote }}
|
||||
{{- if .Values.storage.s3.endpoint }}
|
||||
S3_ENDPOINT: {{ .Values.storage.s3.endpoint | quote }}
|
||||
|
||||
14
chart/templates/grafana-dashboard-configmap.yaml
Normal file
14
chart/templates/grafana-dashboard-configmap.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
{{- if .Values.monitoring.grafanaDashboard.enabled }}
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: {{ include "switchboard.fullname" . }}-grafana-dashboard
|
||||
labels:
|
||||
{{- include "switchboard.labels" . | nindent 4 }}
|
||||
{{- with .Values.monitoring.grafanaDashboard.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
data:
|
||||
switchboard-overview.json: |-
|
||||
{{- .Files.Get "dashboards/switchboard-overview.json" | nindent 4 }}
|
||||
{{- end }}
|
||||
61
chart/templates/prometheusrule.yaml
Normal file
61
chart/templates/prometheusrule.yaml
Normal file
@@ -0,0 +1,61 @@
|
||||
{{- if .Values.monitoring.prometheusRule.enabled }}
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: {{ include "switchboard.fullname" . }}-alerts
|
||||
labels:
|
||||
{{- include "switchboard.labels" . | nindent 4 }}
|
||||
{{- with .Values.monitoring.prometheusRule.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
groups:
|
||||
- name: switchboard.rules
|
||||
rules:
|
||||
- alert: SwitchboardPodRestart
|
||||
expr: increase(kube_pod_container_status_restarts_total{container="backend"}[1h]) > 0
|
||||
for: 0m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Switchboard backend pod restarted (possible OOM)"
|
||||
- alert: SwitchboardProviderDown
|
||||
expr: switchboard_provider_status > 2
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Provider {{`{{ $labels.provider_config_id }}`}} is down"
|
||||
- alert: SwitchboardDBPoolExhaustion
|
||||
expr: switchboard_db_in_use_connections / switchboard_db_open_connections > 0.8
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "DB connection pool >80% utilized"
|
||||
- alert: SwitchboardHighErrorRate
|
||||
expr: |
|
||||
sum(rate(switchboard_http_requests_total{status=~"5.."}[5m]))
|
||||
/ sum(rate(switchboard_http_requests_total[5m])) > 0.05
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "HTTP 5xx error rate exceeds 5%"
|
||||
- alert: SwitchboardTaskFailureRate
|
||||
expr: |
|
||||
rate(switchboard_task_executions_total{status="error"}[15m])
|
||||
/ rate(switchboard_task_executions_total[15m]) > 0.25
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Task failure rate exceeds 25%"
|
||||
- alert: SwitchboardNoCompletions
|
||||
expr: sum(rate(switchboard_completions_total[10m])) == 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "No completions processed in 15 minutes"
|
||||
{{- end }}
|
||||
20
chart/templates/servicemonitor.yaml
Normal file
20
chart/templates/servicemonitor.yaml
Normal file
@@ -0,0 +1,20 @@
|
||||
{{- if .Values.monitoring.serviceMonitor.enabled }}
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: ServiceMonitor
|
||||
metadata:
|
||||
name: {{ include "switchboard.fullname" . }}
|
||||
labels:
|
||||
{{- include "switchboard.labels" . | nindent 4 }}
|
||||
{{- with .Values.monitoring.serviceMonitor.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
{{- include "switchboard.selectorLabels" . | nindent 6 }}
|
||||
app.kubernetes.io/component: backend
|
||||
endpoints:
|
||||
- port: http
|
||||
path: {{ .Values.monitoring.serviceMonitor.path | default "/metrics" }}
|
||||
interval: {{ .Values.monitoring.serviceMonitor.interval | default "30s" }}
|
||||
{{- end }}
|
||||
@@ -4,6 +4,10 @@ metadata:
|
||||
name: {{ .Release.Name }}-backend
|
||||
labels:
|
||||
{{- include "switchboard.backend.labels" . | nindent 4 }}
|
||||
annotations:
|
||||
prometheus.io/scrape: "true"
|
||||
prometheus.io/port: {{ .Values.backend.port | quote }}
|
||||
prometheus.io/path: "/metrics"
|
||||
spec:
|
||||
type: ClusterIP
|
||||
ports:
|
||||
|
||||
Reference in New Issue
Block a user