diff options
| author | Paul Buetow <paul@buetow.org> | 2026-01-08 22:22:10 +0200 |
|---|---|---|
| committer | Paul Buetow <paul@buetow.org> | 2026-01-08 22:22:10 +0200 |
| commit | d9671ba9c6ba158cd4516626c4627d38d6478110 (patch) | |
| tree | 715dd6c2f7a4479545ae780bf12fb2cabbea0937 /internal/prometheus_test.go | |
| parent | a10cbd4e27d944464cec88aaf49d8b8c354d26e1 (diff) | |
Add special handling for Prometheus Watchdog alert
- Treat firing Watchdog as OK status to confirm Alertmanager is working
- Treat absent/non-firing Watchdog as CRITICAL to alert on Alertmanager issues
- Add comprehensive tests for both scenarios
Diffstat (limited to 'internal/prometheus_test.go')
| -rw-r--r-- | internal/prometheus_test.go | 102 |
1 files changed, 102 insertions, 0 deletions
diff --git a/internal/prometheus_test.go b/internal/prometheus_test.go index 8d98e7f..150ba52 100644 --- a/internal/prometheus_test.go +++ b/internal/prometheus_test.go @@ -157,3 +157,105 @@ func TestMergePrometheusAlertsNoHosts(t *testing.T) { t.Errorf("expected no checks, got %d", len(result.checks)) } } + +func TestMergePrometheusAlertsWatchdogFiring(t *testing.T) { + resp := prometheusResponse{ + Status: "success", + Data: struct { + Alerts []prometheusAlert `json:"alerts"` + }{ + Alerts: []prometheusAlert{ + { + Labels: map[string]string{"alertname": "Watchdog", "severity": "none"}, + Annotations: map[string]string{"summary": "An alert that should always be firing to certify that Alertmanager is working properly."}, + State: "firing", + }, + { + Labels: map[string]string{"alertname": "HighCPU", "severity": "critical"}, + Annotations: map[string]string{"summary": "CPU usage is high"}, + State: "firing", + }, + }, + }, + } + + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusOK) + _ = json.NewEncoder(w).Encode(resp) + })) + defer server.Close() + + conf := config{ + PrometheusHosts: []string{strings.TrimPrefix(server.URL, "http://")}, + PrometheusTimeoutS: 2, + } + s := state{checks: make(map[string]checkState)} + + result := mergePrometheusAlerts(context.Background(), s, conf) + + watchdog, ok := result.checks["Prometheus: Watchdog"] + if !ok { + t.Fatal("Watchdog check not found in state") + } + + if watchdog.Status != nagiosOk { + t.Errorf("expected Watchdog status OK, got %v", watchdog.Status) + } + + if !strings.Contains(watchdog.output, "working properly") { + t.Errorf("expected working properly message, got: %s", watchdog.output) + } + + // Verify other alerts are still processed + cpu, ok := result.checks["Prometheus: HighCPU"] + if !ok { + t.Fatal("HighCPU check not found in state") + } + if cpu.Status != nagiosCritical { + t.Errorf("expected HighCPU status CRITICAL, got %v", cpu.Status) + } +} + +func TestMergePrometheusAlertsWatchdogNotFiring(t *testing.T) { + resp := prometheusResponse{ + Status: "success", + Data: struct { + Alerts []prometheusAlert `json:"alerts"` + }{ + Alerts: []prometheusAlert{ + { + Labels: map[string]string{"alertname": "HighCPU", "severity": "critical"}, + Annotations: map[string]string{"summary": "CPU usage is high"}, + State: "firing", + }, + }, + }, + } + + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusOK) + _ = json.NewEncoder(w).Encode(resp) + })) + defer server.Close() + + conf := config{ + PrometheusHosts: []string{strings.TrimPrefix(server.URL, "http://")}, + PrometheusTimeoutS: 2, + } + s := state{checks: make(map[string]checkState)} + + result := mergePrometheusAlerts(context.Background(), s, conf) + + watchdog, ok := result.checks["Prometheus: Watchdog"] + if !ok { + t.Fatal("Watchdog check not found in state") + } + + if watchdog.Status != nagiosCritical { + t.Errorf("expected Watchdog status CRITICAL, got %v", watchdog.Status) + } + + if !strings.Contains(watchdog.output, "not firing") { + t.Errorf("expected not firing message, got: %s", watchdog.output) + } +} |
