]> git.openstreetmap.org Git - chef.git/commitdiff
Add alert rules for cisco switches
authorTom Hughes <tom@compton.nu>
Thu, 21 Jul 2022 16:30:14 +0000 (17:30 +0100)
committerTom Hughes <tom@compton.nu>
Thu, 21 Jul 2022 16:30:35 +0000 (17:30 +0100)
cookbooks/prometheus/templates/default/alert_rules.yml.erb

index 0d045af715588109abffb78b141c7c5ed721e477..289b33f572059289ef5f8dd7dc7d8a27b3faee3d 100644 (file)
@@ -68,6 +68,28 @@ groups:
           alertgroup: "{{ $labels.instance }}"
         annotations:
           down_time: "{{ $value | humanizeDuration }}"
           alertgroup: "{{ $labels.instance }}"
         annotations:
           down_time: "{{ $value | humanizeDuration }}"
+  - name: cisco
+    rules:
+      - alert: cisco fan alarm
+        expr: rlPhdUnitEnvParamFan1Status{rlPhdUnitEnvParamFan1Status!="normal"} > 0 or rlPhdUnitEnvParamFan2Status{rlPhdUnitEnvParamFan2Status!="normal"} > 0
+        for: 5m
+        labels:
+          alertgroup: "{{ $labels.site }}"
+      - alert: cisco temperature alarm
+        expr: rlPhdUnitEnvParamTempSensorStatus{rlPhdUnitEnvParamTempSensorStatus!="ok"} > 0
+        for: 5m
+        labels:
+          alertgroup: "{{ $labels.site }}"
+      - alert: cisco main power alarm
+        expr: rlPhdUnitEnvParamMainPSStatus{rlPhdUnitEnvParamMainPSStatus!="normal"} > 0
+        for: 5m
+        labels:
+          alertgroup: "{{ $labels.site }}"
+      - alert: cisco redundant power alarm
+        expr: rlPhdUnitEnvParamRedundantPSStatus{rlPhdUnitEnvParamRedundantPSStatus!="normal"} > 0
+        for: 5m
+        labels:
+          alertgroup: "{{ $labels.site }}"
   - name: cpu
     rules:
       - alert: cpu pressure
   - name: cpu
     rules:
       - alert: cpu pressure