summaryrefslogtreecommitdiff
path: root/roles/monitoring
diff options
context:
space:
mode:
authorChristian Pointner <equinox@spreadspace.org>2021-10-15 23:05:20 +0200
committerChristian Pointner <equinox@spreadspace.org>2021-10-15 23:05:20 +0200
commitbdcafe51f1b40d2dab2d52136e1ce60ab95e2ed5 (patch)
tree43ae0b79a9b2d6f76192eaac8dd6241eb340cf96 /roles/monitoring
parentadd some alerts for smartmon collector (diff)
add alerts for zpool state
Diffstat (limited to 'roles/monitoring')
-rw-r--r--roles/monitoring/prometheus/server/defaults/main/rules_node.yml18
1 files changed, 18 insertions, 0 deletions
diff --git a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml
index 79a474e8..ffe616b7 100644
--- a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml
+++ b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml
@@ -227,6 +227,24 @@ prometheus_server_rules_node:
summary: Host clock not synchronising (instance {{ '{{' }} $labels.instance {{ '}}' }})
description: "Clock not synchronising.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}"
+ - alert: ZpoolStateDegraded
+ expr: node_zfs_zpool_state{state="degraded"} == 1
+ for: 0m
+ labels:
+ severity: warning
+ annotations:
+ summary: ZFS zpool is degraded (instance {{ '{{' }} $labels.instance {{ '}}' }})
+ description: "The ZFS zpool {{ '{{' }} $labels.zpool {{ '}}' }} is degraded.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}"
+
+ - alert: ZpoolStateFaulted
+ expr: node_zfs_zpool_state{state="faulted"} == 1
+ for: 0m
+ labels:
+ severity: critical
+ annotations:
+ summary: ZFS zpool is faulted (instance {{ '{{' }} $labels.instance {{ '}}' }})
+ description: "The ZFS zpool {{ '{{' }} $labels.zpool {{ '}}' }} is faulted.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}"
+
- alert: AptUpgradesPending
expr: sum by (instance) (apt_upgrades_pending) > 0
for: 0m