From 31b8ee126ae339a382b681263e133354463da053 Mon Sep 17 00:00:00 2001 From: Christian Pointner Date: Tue, 9 Nov 2021 11:14:40 +0100 Subject: add alert for missing smartmon metrics --- roles/monitoring/prometheus/server/defaults/main/rules_node.yml | 9 +++++++++ 1 file changed, 9 insertions(+) (limited to 'roles/monitoring/prometheus/server') diff --git a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml index 6a77b105..0a28871d 100644 --- a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml +++ b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml @@ -281,6 +281,15 @@ prometheus_server_rules_node: summary: Some processes still use a deleted library (instance {{ '{{' }} $labels.instance {{ '}}' }}) description: "The deleted library {{ '{{' }} $labels.library_name {{ '}}' }} on host {{ '{{' }} $labels.instance {{ '}}' }} is still in use by {{ '{{' }} $value {{ '}}' }} processes.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}" + - alert: SmartmonMetricsMissing + expr: absent(smartmon_smartctl_run) + for: 30m + labels: + severity: warning + annotations: + summary: Metrics from smartctl are missing (instance {{ '{{' }} $labels.instance {{ '}}' }}) + description: "smartctl on host {{ '{{' }} $labels.instance {{ '}}' }} stopped reporting metrics for device {{ '{{' }} $labels.device {{ '}}' }}.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}" + - alert: SmartmonMetricsOutdated expr: time() - smartmon_smartctl_run > 3600 for: 0m -- cgit v1.2.3