Prometheus Alertmanager Production Alert Rules Template
plain (yaml)
13 hours ago
·
39 lines
·
6 views
1groups:
2 - name: host_infrastructure_alerts
3 rules:
4 - alert: HostHighCpuUsage
5 expr: 100 - (avg by (instance) (rate(node_cpu_seconds_total{mode="idle"}[5m])) * 100) > 85
6 for: 5m
7 labels:
8 severity: warning
9 annotations:
10 summary: "High CPU load on host {{ $labels.instance }}"
11 description: "CPU load has exceeded 85% for more than 5 minutes (current value: {{ $value }}%)."
13 - alert: HostMemoryExhaustion
14 expr: (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) * 100 < 10
15 for: 3m
16 labels:
17 severity: critical
18 annotations:
19 summary: "Critical RAM exhaustion on {{ $labels.instance }}"
20 description: "Host available memory is below 10% (current free: {{ $value }}%)."
22 - alert: HostDiskFillingFast
23 expr: (node_filesystem_free_bytes{mountpoint="/"} / node_filesystem_size_bytes{mountpoint="/"}) * 100 < 15
24 for: 10m
25 labels:
26 severity: warning
27 annotations:
28 summary: "Root filesystem low on disk space on {{ $labels.instance }}"
29 description: "Remaining disk space on / is below 15%."
31 - alert: InstanceDown
32 expr: up == 0
33 for: 1m
34 labels:
35 severity: critical
36 annotations:
37 summary: "Service target {{ $labels.instance }} is down"
38 description: "Prometheus scraper cannot connect to target for over 1 minute."
Replies 0
No replies yet
Every reply is a note. Start a discussion, ask a question, or attach a code snippet.
Notification