From 7061267448ee4ba86a7257d350a14feb9bc6fc95 Mon Sep 17 00:00:00 2001 From: Christian Pointner Date: Tue, 25 Jan 2022 18:31:09 +0100 Subject: prometheus: add alert for network bonds --- roles/monitoring/prometheus/server/defaults/main/rules_node.yml | 9 +++++++++ 1 file changed, 9 insertions(+) (limited to 'roles/monitoring/prometheus/server/defaults') diff --git a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml index 6d4f763f..8a02e67b 100644 --- a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml +++ b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml @@ -209,6 +209,15 @@ prometheus_server_rules_node: summary: Host conntrack limit (instance {{ '{{' }} $labels.instance {{ '}}' }}) description: "The number of conntrack is approching limit\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}" + - alert: HostNetworkBondDegraded + expr: (node_bonding_active - node_bonding_slaves) != 0 + for: 1m + labels: + severity: warning + annotations: + title: Bond is degraded on (instance {{ '{{' }} $labels.instance {{ '}}' }}) + description: "Bond \"{{ '{{' }} $labels.master {{ '}}' }}\" on \"{{ '{{' }} $labels.instance {{ '}}' }}\" is degraded\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}" + - alert: HostClockSkew expr: (node_timex_offset_seconds > 0.05 and deriv(node_timex_offset_seconds[5m]) >= 0) or (node_timex_offset_seconds < -0.05 and deriv(node_timex_offset_seconds[5m]) <= 0) for: 2m -- cgit v1.2.3