From fd03bf128252d768e51bc1d2cc8eb7bd416e91c4 Mon Sep 17 00:00:00 2001 From: Christian Pointner Date: Tue, 11 Jan 2022 00:33:45 +0100 Subject: prometheus: add alert rules for syncoid pull based backups --- .../prometheus/server/defaults/main/rules_node.yml | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) (limited to 'roles/monitoring') diff --git a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml index e6415bd9..6d4f763f 100644 --- a/roles/monitoring/prometheus/server/defaults/main/rules_node.yml +++ b/roles/monitoring/prometheus/server/defaults/main/rules_node.yml @@ -370,3 +370,21 @@ prometheus_server_rules_node: annotations: summary: Host disk S.M.A.R.T. reports spin-up retries (instance {{ '{{' }} $labels.instance {{ '}}' }}) description: "S.M.A.R.T. reports spin-up retries for disk {{ '{{' }} $labels.device {{ '}}' }} on host {{ '{{' }} $labels.instance {{ '}}' }}, the drive might be failing.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}" + + - alert: SyncoidPullJobTooLongAgo + expr: time() - syncoid_pull_run > (24 * 3600) + for: 0m + labels: + severity: warning + annotations: + summary: The last syncoid pull job was too long ago (instance {{ '{{' }} $labels.instance {{ '}}' }}) + description: "The last syncoid-based backup job of {{ '{{' }} $labels.instance {{ '}}' }} from {{ '{{' }} $labels.backup_server {{ '}}' }} ran more then 24 hours ago.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}" + + - alert: SyncoidPullJobFailed + expr: syncoid_pull_exit_code != 0 + for: 0m + labels: + severity: warning + annotations: + summary: The last syncoid pull job failed (instance {{ '{{' }} $labels.instance {{ '}}' }}) + description: "The last syncoid-based backup job of {{ '{{' }} $labels.instance {{ '}}' }} from {{ '{{' }} $labels.backup_server {{ '}}' }} has failed.\n VALUE = {{ '{{' }} $value {{ '}}' }}\n LABELS = {{ '{{' }} $labels {{ '}}' }}" -- cgit v1.2.3