From 3b212e01c17597264ff6303498329ffa52157f38 Mon Sep 17 00:00:00 2001 From: Tom Hughes Date: Mon, 15 Nov 2021 15:39:26 +0000 Subject: [PATCH] Add alert for failed services --- cookbooks/prometheus/templates/default/alert_rules.yml.erb | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/cookbooks/prometheus/templates/default/alert_rules.yml.erb b/cookbooks/prometheus/templates/default/alert_rules.yml.erb index 1b3afcba5..b9ffa9f44 100644 --- a/cookbooks/prometheus/templates/default/alert_rules.yml.erb +++ b/cookbooks/prometheus/templates/default/alert_rules.yml.erb @@ -262,6 +262,13 @@ groups: for: 0m labels: alertgroup: ssl + - name: systemd + rules: + - alert: systemd failed service + expr: node_systemd_unit_state{state="failed"} == 1 + for: 5m + labels: + alertgroup: "{{ $labels.instance }}" - name: tile rules: - alert: renderd replication delay -- 2.39.5