summaryrefslogtreecommitdiff
path: root/roles/monitoring
diff options
context:
space:
mode:
authorChristian Pointner <equinox@spreadspace.org>2022-01-25 18:31:09 +0100
committerChristian Pointner <equinox@spreadspace.org>2022-01-25 18:31:09 +0100
commit7061267448ee4ba86a7257d350a14feb9bc6fc95 (patch)
tree5f69d5856767eaa8438fc9b3f10dbb51207f02cd /roles/monitoring
parentch-apps: update kubelet (diff)
prometheus: add alert for network bonds
Diffstat (limited to 'roles/monitoring')
-rw-r--r--roles/monitoring/prometheus/server/defaults/main/rules_node.yml9
1 files changed, 9 insertions, 0 deletions
diff --git a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml
index 6d4f763f..8a02e67b 100644
--- a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml
+++ b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml
@@ -209,6 +209,15 @@ prometheus_server_rules_node:
summary: Host conntrack limit (instance {{ '{{' }} $labels.instance {{ '}}' }})
description: "The number of conntrack is approching limit\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}"
+ - alert: HostNetworkBondDegraded
+ expr: (node_bonding_active - node_bonding_slaves) != 0
+ for: 1m
+ labels:
+ severity: warning
+ annotations:
+ title: Bond is degraded on (instance {{ '{{' }} $labels.instance {{ '}}' }})
+ description: "Bond \"{{ '{{' }} $labels.master {{ '}}' }}\" on \"{{ '{{' }} $labels.instance {{ '}}' }}\" is degraded\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}"
+
- alert: HostClockSkew
expr: (node_timex_offset_seconds > 0.05 and deriv(node_timex_offset_seconds[5m]) >= 0) or (node_timex_offset_seconds < -0.05 and deriv(node_timex_offset_seconds[5m]) <= 0)
for: 2m