@leverege/build-tools 2.66.0 → 2.66.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/helm-charts/prom-operator/prometheus-stack.yaml.ovh +1 -1
- package/src/helm-charts/prom-operator/rules/awesome-base/elasticsearch-rules.yaml +219 -0
- package/src/helm-charts/prom-operator/rules/awesome-base/etcd-exporter.yaml +153 -0
- package/src/helm-charts/prom-operator/rules/awesome-base/google-cadvisor.yaml +108 -0
- package/src/helm-charts/prom-operator/rules/{new-awesome.yaml → awesome-base/kubestate-exporter.yaml} +2 -1
- package/src/helm-charts/prom-operator/rules/awesome-base/prometheus-rules.yaml +318 -0
- package/src/helm-charts/prom-operator/rules/awesome-base/redis-rules.yaml +142 -0
- package/src/helm-charts/prom-operator/rules/awesome-base/traefik-rules.yaml +43 -0
- package/src/helm-charts/prom-operator/rules/cadvisor-rules.yaml +109 -0
- package/src/helm-charts/prom-operator/rules/elasticsearch-rules.yaml +213 -95
- package/src/helm-charts/prom-operator/rules/etcd-rulte.yaml +153 -0
- package/src/helm-charts/prom-operator/rules/kubestate-rules.yaml +412 -0
- package/src/helm-charts/prom-operator/rules/prometheus-rules.yaml +312 -73
- package/src/helm-charts/prom-operator/rules/redis-rules.yaml +136 -75
- package/src/helm-charts/prom-operator/rules/traefik-rules.yaml +48 -21
- package/src/helmup.sh +1 -1
- package/src/helm-charts/prom-operator/rules/awesome-k8s.yaml +0 -326
- package/src/helm-charts/prom-operator/rules/kubernetes-rules.yaml +0 -109
package/package.json
CHANGED
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
# https://samber.github.io/awesome-prometheus-alerts/rules#elasticsearch
|
|
2
|
+
apiVersion: monitoring.coreos.com/v1
|
|
3
|
+
kind: PrometheusRule
|
|
4
|
+
metadata:
|
|
5
|
+
name: prometheus-community-elasticsearch-exporter
|
|
6
|
+
namespace: prometheus
|
|
7
|
+
spec:
|
|
8
|
+
groups:
|
|
9
|
+
- name: Elasticsearch (awesome) # PrometheusCommunityElasticsearchExporter-rules
|
|
10
|
+
rules:
|
|
11
|
+
- alert: ElasticsearchHeapUsageTooHigh
|
|
12
|
+
expr: (elasticsearch_jvm_memory_used_bytes{area="heap"} / elasticsearch_jvm_memory_max_bytes{area="heap"}) * 100 > 90
|
|
13
|
+
for: 2m
|
|
14
|
+
labels:
|
|
15
|
+
severity: critical
|
|
16
|
+
annotations:
|
|
17
|
+
summary: Elasticsearch Heap Usage Too High (instance {{ $labels.instance }})
|
|
18
|
+
description: |-
|
|
19
|
+
The heap usage is over 90%
|
|
20
|
+
VALUE = {{ $value }}
|
|
21
|
+
LABELS = {{ $labels }}
|
|
22
|
+
- alert: ElasticsearchHeapUsageWarning
|
|
23
|
+
expr: (elasticsearch_jvm_memory_used_bytes{area="heap"} / elasticsearch_jvm_memory_max_bytes{area="heap"}) * 100 > 80
|
|
24
|
+
for: 2m
|
|
25
|
+
labels:
|
|
26
|
+
severity: warning
|
|
27
|
+
annotations:
|
|
28
|
+
summary: Elasticsearch Heap Usage warning (instance {{ $labels.instance }})
|
|
29
|
+
description: |-
|
|
30
|
+
The heap usage is over 80%
|
|
31
|
+
VALUE = {{ $value }}
|
|
32
|
+
LABELS = {{ $labels }}
|
|
33
|
+
- alert: ElasticsearchDiskOutOfSpace
|
|
34
|
+
expr: elasticsearch_filesystem_data_available_bytes / elasticsearch_filesystem_data_size_bytes * 100 < 10
|
|
35
|
+
for: 0m
|
|
36
|
+
labels:
|
|
37
|
+
severity: critical
|
|
38
|
+
annotations:
|
|
39
|
+
summary: Elasticsearch disk out of space (instance {{ $labels.instance }})
|
|
40
|
+
description: |-
|
|
41
|
+
The disk usage is over 90%
|
|
42
|
+
VALUE = {{ $value }}
|
|
43
|
+
LABELS = {{ $labels }}
|
|
44
|
+
- alert: ElasticsearchDiskSpaceLow
|
|
45
|
+
expr: elasticsearch_filesystem_data_available_bytes / elasticsearch_filesystem_data_size_bytes * 100 < 20
|
|
46
|
+
for: 2m
|
|
47
|
+
labels:
|
|
48
|
+
severity: warning
|
|
49
|
+
annotations:
|
|
50
|
+
summary: Elasticsearch disk space low (instance {{ $labels.instance }})
|
|
51
|
+
description: |-
|
|
52
|
+
The disk usage is over 80%
|
|
53
|
+
VALUE = {{ $value }}
|
|
54
|
+
LABELS = {{ $labels }}
|
|
55
|
+
- alert: ElasticsearchClusterRed
|
|
56
|
+
expr: elasticsearch_cluster_health_status{color="red"} == 1
|
|
57
|
+
for: 0m
|
|
58
|
+
labels:
|
|
59
|
+
severity: critical
|
|
60
|
+
annotations:
|
|
61
|
+
summary: Elasticsearch Cluster Red (instance {{ $labels.instance }})
|
|
62
|
+
description: |-
|
|
63
|
+
Elastic Cluster Red status
|
|
64
|
+
VALUE = {{ $value }}
|
|
65
|
+
LABELS = {{ $labels }}
|
|
66
|
+
- alert: ElasticsearchClusterYellow
|
|
67
|
+
expr: elasticsearch_cluster_health_status{color="yellow"} == 1
|
|
68
|
+
for: 0m
|
|
69
|
+
labels:
|
|
70
|
+
severity: warning
|
|
71
|
+
annotations:
|
|
72
|
+
summary: Elasticsearch Cluster Yellow (instance {{ $labels.instance }})
|
|
73
|
+
description: |-
|
|
74
|
+
Elastic Cluster Yellow status
|
|
75
|
+
VALUE = {{ $value }}
|
|
76
|
+
LABELS = {{ $labels }}
|
|
77
|
+
- alert: ElasticsearchHealthyNodes
|
|
78
|
+
expr: elasticsearch_cluster_health_number_of_nodes < 3
|
|
79
|
+
for: 0m
|
|
80
|
+
labels:
|
|
81
|
+
severity: critical
|
|
82
|
+
annotations:
|
|
83
|
+
summary: Elasticsearch Healthy Nodes (instance {{ $labels.instance }})
|
|
84
|
+
description: |-
|
|
85
|
+
Missing node in Elasticsearch cluster
|
|
86
|
+
VALUE = {{ $value }}
|
|
87
|
+
LABELS = {{ $labels }}
|
|
88
|
+
- alert: ElasticsearchHealthyDataNodes
|
|
89
|
+
expr: elasticsearch_cluster_health_number_of_data_nodes < 3
|
|
90
|
+
for: 0m
|
|
91
|
+
labels:
|
|
92
|
+
severity: critical
|
|
93
|
+
annotations:
|
|
94
|
+
summary: Elasticsearch Healthy Data Nodes (instance {{ $labels.instance }})
|
|
95
|
+
description: |-
|
|
96
|
+
Missing data node in Elasticsearch cluster
|
|
97
|
+
VALUE = {{ $value }}
|
|
98
|
+
LABELS = {{ $labels }}
|
|
99
|
+
- alert: ElasticsearchRelocatingShards
|
|
100
|
+
expr: elasticsearch_cluster_health_relocating_shards > 0
|
|
101
|
+
for: 0m
|
|
102
|
+
labels:
|
|
103
|
+
severity: info
|
|
104
|
+
annotations:
|
|
105
|
+
summary: Elasticsearch relocating shards (instance {{ $labels.instance }})
|
|
106
|
+
description: |-
|
|
107
|
+
Elasticsearch is relocating shards
|
|
108
|
+
VALUE = {{ $value }}
|
|
109
|
+
LABELS = {{ $labels }}
|
|
110
|
+
- alert: ElasticsearchRelocatingShardsTooLong
|
|
111
|
+
expr: elasticsearch_cluster_health_relocating_shards > 0
|
|
112
|
+
for: 15m
|
|
113
|
+
labels:
|
|
114
|
+
severity: warning
|
|
115
|
+
annotations:
|
|
116
|
+
summary: Elasticsearch relocating shards too long (instance {{ $labels.instance }})
|
|
117
|
+
description: |-
|
|
118
|
+
Elasticsearch has been relocating shards for 15min
|
|
119
|
+
VALUE = {{ $value }}
|
|
120
|
+
LABELS = {{ $labels }}
|
|
121
|
+
- alert: ElasticsearchInitializingShards
|
|
122
|
+
expr: elasticsearch_cluster_health_initializing_shards > 0
|
|
123
|
+
for: 0m
|
|
124
|
+
labels:
|
|
125
|
+
severity: info
|
|
126
|
+
annotations:
|
|
127
|
+
summary: Elasticsearch initializing shards (instance {{ $labels.instance }})
|
|
128
|
+
description: |-
|
|
129
|
+
Elasticsearch is initializing shards
|
|
130
|
+
VALUE = {{ $value }}
|
|
131
|
+
LABELS = {{ $labels }}
|
|
132
|
+
- alert: ElasticsearchInitializingShardsTooLong
|
|
133
|
+
expr: elasticsearch_cluster_health_initializing_shards > 0
|
|
134
|
+
for: 15m
|
|
135
|
+
labels:
|
|
136
|
+
severity: warning
|
|
137
|
+
annotations:
|
|
138
|
+
summary: Elasticsearch initializing shards too long (instance {{ $labels.instance }})
|
|
139
|
+
description: |-
|
|
140
|
+
Elasticsearch has been initializing shards for 15 min
|
|
141
|
+
VALUE = {{ $value }}
|
|
142
|
+
LABELS = {{ $labels }}
|
|
143
|
+
- alert: ElasticsearchUnassignedShards
|
|
144
|
+
expr: elasticsearch_cluster_health_unassigned_shards > 0
|
|
145
|
+
for: 0m
|
|
146
|
+
labels:
|
|
147
|
+
severity: critical
|
|
148
|
+
annotations:
|
|
149
|
+
summary: Elasticsearch unassigned shards (instance {{ $labels.instance }})
|
|
150
|
+
description: |-
|
|
151
|
+
Elasticsearch has unassigned shards
|
|
152
|
+
VALUE = {{ $value }}
|
|
153
|
+
LABELS = {{ $labels }}
|
|
154
|
+
- alert: ElasticsearchPendingTasks
|
|
155
|
+
expr: elasticsearch_cluster_health_number_of_pending_tasks > 0
|
|
156
|
+
for: 15m
|
|
157
|
+
labels:
|
|
158
|
+
severity: warning
|
|
159
|
+
annotations:
|
|
160
|
+
summary: Elasticsearch pending tasks (instance {{ $labels.instance }})
|
|
161
|
+
description: |-
|
|
162
|
+
Elasticsearch has pending tasks. Cluster works slowly.
|
|
163
|
+
VALUE = {{ $value }}
|
|
164
|
+
LABELS = {{ $labels }}
|
|
165
|
+
- alert: ElasticsearchNoNewDocuments
|
|
166
|
+
expr: increase(elasticsearch_indices_indexing_index_total{es_data_node="true"}[10m]) < 1
|
|
167
|
+
for: 0m
|
|
168
|
+
labels:
|
|
169
|
+
severity: warning
|
|
170
|
+
annotations:
|
|
171
|
+
summary: Elasticsearch no new documents (instance {{ $labels.instance }})
|
|
172
|
+
description: |-
|
|
173
|
+
No new documents for 10 min!
|
|
174
|
+
VALUE = {{ $value }}
|
|
175
|
+
LABELS = {{ $labels }}
|
|
176
|
+
- alert: ElasticsearchHighIndexingLatency
|
|
177
|
+
expr: elasticsearch_indices_indexing_index_time_seconds_total / elasticsearch_indices_indexing_index_total > 0.0005
|
|
178
|
+
for: 10m
|
|
179
|
+
labels:
|
|
180
|
+
severity: warning
|
|
181
|
+
annotations:
|
|
182
|
+
summary: Elasticsearch High Indexing Latency (instance {{ $labels.instance }})
|
|
183
|
+
description: |-
|
|
184
|
+
The indexing latency on Elasticsearch cluster is higher than the threshold.
|
|
185
|
+
VALUE = {{ $value }}
|
|
186
|
+
LABELS = {{ $labels }}
|
|
187
|
+
- alert: ElasticsearchHighIndexingRate
|
|
188
|
+
expr: sum(rate(elasticsearch_indices_indexing_index_total[1m]))> 10000
|
|
189
|
+
for: 5m
|
|
190
|
+
labels:
|
|
191
|
+
severity: warning
|
|
192
|
+
annotations:
|
|
193
|
+
summary: Elasticsearch High Indexing Rate (instance {{ $labels.instance }})
|
|
194
|
+
description: |-
|
|
195
|
+
The indexing rate on Elasticsearch cluster is higher than the threshold.
|
|
196
|
+
VALUE = {{ $value }}
|
|
197
|
+
LABELS = {{ $labels }}
|
|
198
|
+
- alert: ElasticsearchHighQueryRate
|
|
199
|
+
expr: sum(rate(elasticsearch_indices_search_query_total[1m])) > 100
|
|
200
|
+
for: 5m
|
|
201
|
+
labels:
|
|
202
|
+
severity: warning
|
|
203
|
+
annotations:
|
|
204
|
+
summary: Elasticsearch High Query Rate (instance {{ $labels.instance }})
|
|
205
|
+
description: |-
|
|
206
|
+
The query rate on Elasticsearch cluster is higher than the threshold.
|
|
207
|
+
VALUE = {{ $value }}
|
|
208
|
+
LABELS = {{ $labels }}
|
|
209
|
+
- alert: ElasticsearchHighQueryLatency
|
|
210
|
+
expr: elasticsearch_indices_search_fetch_time_seconds / elasticsearch_indices_search_fetch_total > 1
|
|
211
|
+
for: 5m
|
|
212
|
+
labels:
|
|
213
|
+
severity: warning
|
|
214
|
+
annotations:
|
|
215
|
+
summary: Elasticsearch High Query Latency (instance {{ $labels.instance }})
|
|
216
|
+
description: |-
|
|
217
|
+
The query latency on Elasticsearch cluster is higher than the threshold.
|
|
218
|
+
VALUE = {{ $value }}
|
|
219
|
+
LABELS = {{ $labels }}
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# https://samber.github.io/awesome-prometheus-alerts/rules#etcd
|
|
2
|
+
apiVersion: monitoring.coreos.com/v1
|
|
3
|
+
kind: PrometheusRule
|
|
4
|
+
metadata:
|
|
5
|
+
name: etcd-exporter # embedded-exporter
|
|
6
|
+
namespace: prometheus
|
|
7
|
+
spec:
|
|
8
|
+
groups:
|
|
9
|
+
- name: Etcd (awesome) # EmbeddedExporter-rules
|
|
10
|
+
rules:
|
|
11
|
+
- alert: EtcdInsufficientMembers
|
|
12
|
+
expr: count(etcd_server_id) % 2 == 0
|
|
13
|
+
for: 0m
|
|
14
|
+
labels:
|
|
15
|
+
severity: critical
|
|
16
|
+
annotations:
|
|
17
|
+
summary: Etcd insufficient Members (instance {{ $labels.instance }})
|
|
18
|
+
description: |-
|
|
19
|
+
Etcd cluster should have an odd number of members
|
|
20
|
+
VALUE = {{ $value }}
|
|
21
|
+
LABELS = {{ $labels }}
|
|
22
|
+
- alert: EtcdNoLeader
|
|
23
|
+
expr: etcd_server_has_leader == 0
|
|
24
|
+
for: 0m
|
|
25
|
+
labels:
|
|
26
|
+
severity: critical
|
|
27
|
+
annotations:
|
|
28
|
+
summary: Etcd no Leader (instance {{ $labels.instance }})
|
|
29
|
+
description: |-
|
|
30
|
+
Etcd cluster have no leader
|
|
31
|
+
VALUE = {{ $value }}
|
|
32
|
+
LABELS = {{ $labels }}
|
|
33
|
+
- alert: EtcdHighNumberOfLeaderChanges
|
|
34
|
+
expr: increase(etcd_server_leader_changes_seen_total[10m]) > 2
|
|
35
|
+
for: 0m
|
|
36
|
+
labels:
|
|
37
|
+
severity: warning
|
|
38
|
+
annotations:
|
|
39
|
+
summary: Etcd high number of leader changes (instance {{ $labels.instance }})
|
|
40
|
+
description: |-
|
|
41
|
+
Etcd leader changed more than 2 times during 10 minutes
|
|
42
|
+
VALUE = {{ $value }}
|
|
43
|
+
LABELS = {{ $labels }}
|
|
44
|
+
- alert: EtcdHighNumberOfFailedGrpcRequests
|
|
45
|
+
expr: sum(rate(grpc_server_handled_total{grpc_code!="OK"}[1m])) BY (grpc_service, grpc_method) / sum(rate(grpc_server_handled_total[1m])) BY (grpc_service, grpc_method) > 0.01
|
|
46
|
+
for: 2m
|
|
47
|
+
labels:
|
|
48
|
+
severity: warning
|
|
49
|
+
annotations:
|
|
50
|
+
summary: Etcd high number of failed GRPC requests (instance {{ $labels.instance }})
|
|
51
|
+
description: |-
|
|
52
|
+
More than 1% GRPC request failure detected in Etcd
|
|
53
|
+
VALUE = {{ $value }}
|
|
54
|
+
LABELS = {{ $labels }}
|
|
55
|
+
- alert: EtcdHighNumberOfFailedGrpcRequests
|
|
56
|
+
expr: sum(rate(grpc_server_handled_total{grpc_code!="OK"}[1m])) BY (grpc_service, grpc_method) / sum(rate(grpc_server_handled_total[1m])) BY (grpc_service, grpc_method) > 0.05
|
|
57
|
+
for: 2m
|
|
58
|
+
labels:
|
|
59
|
+
severity: critical
|
|
60
|
+
annotations:
|
|
61
|
+
summary: Etcd high number of failed GRPC requests (instance {{ $labels.instance }})
|
|
62
|
+
description: |-
|
|
63
|
+
More than 5% GRPC request failure detected in Etcd
|
|
64
|
+
VALUE = {{ $value }}
|
|
65
|
+
LABELS = {{ $labels }}
|
|
66
|
+
- alert: EtcdGrpcRequestsSlow
|
|
67
|
+
expr: histogram_quantile(0.99, sum(rate(grpc_server_handling_seconds_bucket{grpc_type="unary"}[1m])) by (grpc_service, grpc_method, le)) > 0.15
|
|
68
|
+
for: 2m
|
|
69
|
+
labels:
|
|
70
|
+
severity: warning
|
|
71
|
+
annotations:
|
|
72
|
+
summary: Etcd GRPC requests slow (instance {{ $labels.instance }})
|
|
73
|
+
description: |-
|
|
74
|
+
GRPC requests slowing down, 99th percentile is over 0.15s
|
|
75
|
+
VALUE = {{ $value }}
|
|
76
|
+
LABELS = {{ $labels }}
|
|
77
|
+
- alert: EtcdHighNumberOfFailedHttpRequests
|
|
78
|
+
expr: sum(rate(etcd_http_failed_total[1m])) BY (method) / sum(rate(etcd_http_received_total[1m])) BY (method) > 0.01
|
|
79
|
+
for: 2m
|
|
80
|
+
labels:
|
|
81
|
+
severity: warning
|
|
82
|
+
annotations:
|
|
83
|
+
summary: Etcd high number of failed HTTP requests (instance {{ $labels.instance }})
|
|
84
|
+
description: |-
|
|
85
|
+
More than 1% HTTP failure detected in Etcd
|
|
86
|
+
VALUE = {{ $value }}
|
|
87
|
+
LABELS = {{ $labels }}
|
|
88
|
+
- alert: EtcdHighNumberOfFailedHttpRequests
|
|
89
|
+
expr: sum(rate(etcd_http_failed_total[1m])) BY (method) / sum(rate(etcd_http_received_total[1m])) BY (method) > 0.05
|
|
90
|
+
for: 2m
|
|
91
|
+
labels:
|
|
92
|
+
severity: critical
|
|
93
|
+
annotations:
|
|
94
|
+
summary: Etcd high number of failed HTTP requests (instance {{ $labels.instance }})
|
|
95
|
+
description: |-
|
|
96
|
+
More than 5% HTTP failure detected in Etcd
|
|
97
|
+
VALUE = {{ $value }}
|
|
98
|
+
LABELS = {{ $labels }}
|
|
99
|
+
- alert: EtcdHttpRequestsSlow
|
|
100
|
+
expr: histogram_quantile(0.99, rate(etcd_http_successful_duration_seconds_bucket[1m])) > 0.15
|
|
101
|
+
for: 2m
|
|
102
|
+
labels:
|
|
103
|
+
severity: warning
|
|
104
|
+
annotations:
|
|
105
|
+
summary: Etcd HTTP requests slow (instance {{ $labels.instance }})
|
|
106
|
+
description: |-
|
|
107
|
+
HTTP requests slowing down, 99th percentile is over 0.15s
|
|
108
|
+
VALUE = {{ $value }}
|
|
109
|
+
LABELS = {{ $labels }}
|
|
110
|
+
- alert: EtcdMemberCommunicationSlow
|
|
111
|
+
expr: histogram_quantile(0.99, rate(etcd_network_peer_round_trip_time_seconds_bucket[1m])) > 0.15
|
|
112
|
+
for: 2m
|
|
113
|
+
labels:
|
|
114
|
+
severity: warning
|
|
115
|
+
annotations:
|
|
116
|
+
summary: Etcd member communication slow (instance {{ $labels.instance }})
|
|
117
|
+
description: |-
|
|
118
|
+
Etcd member communication slowing down, 99th percentile is over 0.15s
|
|
119
|
+
VALUE = {{ $value }}
|
|
120
|
+
LABELS = {{ $labels }}
|
|
121
|
+
- alert: EtcdHighNumberOfFailedProposals
|
|
122
|
+
expr: increase(etcd_server_proposals_failed_total[1h]) > 5
|
|
123
|
+
for: 2m
|
|
124
|
+
labels:
|
|
125
|
+
severity: warning
|
|
126
|
+
annotations:
|
|
127
|
+
summary: Etcd high number of failed proposals (instance {{ $labels.instance }})
|
|
128
|
+
description: |-
|
|
129
|
+
Etcd server got more than 5 failed proposals past hour
|
|
130
|
+
VALUE = {{ $value }}
|
|
131
|
+
LABELS = {{ $labels }}
|
|
132
|
+
- alert: EtcdHighFsyncDurations
|
|
133
|
+
expr: histogram_quantile(0.99, rate(etcd_disk_wal_fsync_duration_seconds_bucket[1m])) > 0.5
|
|
134
|
+
for: 2m
|
|
135
|
+
labels:
|
|
136
|
+
severity: warning
|
|
137
|
+
annotations:
|
|
138
|
+
summary: Etcd high fsync durations (instance {{ $labels.instance }})
|
|
139
|
+
description: |-
|
|
140
|
+
Etcd WAL fsync duration increasing, 99th percentile is over 0.5s
|
|
141
|
+
VALUE = {{ $value }}
|
|
142
|
+
LABELS = {{ $labels }}
|
|
143
|
+
- alert: EtcdHighCommitDurations
|
|
144
|
+
expr: histogram_quantile(0.99, rate(etcd_disk_backend_commit_duration_seconds_bucket[1m])) > 0.25
|
|
145
|
+
for: 2m
|
|
146
|
+
labels:
|
|
147
|
+
severity: warning
|
|
148
|
+
annotations:
|
|
149
|
+
summary: Etcd high commit durations (instance {{ $labels.instance }})
|
|
150
|
+
description: |-
|
|
151
|
+
Etcd commit duration increasing, 99th percentile is over 0.25s
|
|
152
|
+
VALUE = {{ $value }}
|
|
153
|
+
LABELS = {{ $labels }}
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
# https://samber.github.io/awesome-prometheus-alerts/rules#docker-containers
|
|
2
|
+
apiVersion: monitoring.coreos.com/v1
|
|
3
|
+
kind: PrometheusRule
|
|
4
|
+
metadata:
|
|
5
|
+
name: google-cadvisor
|
|
6
|
+
spec:
|
|
7
|
+
groups:
|
|
8
|
+
- name: Google Cadvisor (awesome) # GoogleCadvisor-rules
|
|
9
|
+
rules:
|
|
10
|
+
- alert: ContainerKilled
|
|
11
|
+
expr: time() - container_last_seen > 60
|
|
12
|
+
for: 0m
|
|
13
|
+
labels:
|
|
14
|
+
severity: warning
|
|
15
|
+
annotations:
|
|
16
|
+
summary: Container killed (instance {{ $labels.instance }})
|
|
17
|
+
description: |-
|
|
18
|
+
A container has disappeared
|
|
19
|
+
VALUE = {{ $value }}
|
|
20
|
+
LABELS = {{ $labels }}
|
|
21
|
+
- alert: ContainerAbsent
|
|
22
|
+
expr: absent(container_last_seen)
|
|
23
|
+
for: 5m
|
|
24
|
+
labels:
|
|
25
|
+
severity: warning
|
|
26
|
+
annotations:
|
|
27
|
+
summary: Container absent (instance {{ $labels.instance }})
|
|
28
|
+
description: |-
|
|
29
|
+
A container is absent for 5 min
|
|
30
|
+
VALUE = {{ $value }}
|
|
31
|
+
LABELS = {{ $labels }}
|
|
32
|
+
- alert: ContainerHighCpuUtilization
|
|
33
|
+
expr: (sum(rate(container_cpu_usage_seconds_total{container!=""}[5m])) by (pod, container) / sum(container_spec_cpu_quota{container!=""}/container_spec_cpu_period{container!=""}) by (pod, container) * 100) > 80
|
|
34
|
+
for: 2m
|
|
35
|
+
labels:
|
|
36
|
+
severity: warning
|
|
37
|
+
annotations:
|
|
38
|
+
summary: Container High CPU utilization (instance {{ $labels.instance }})
|
|
39
|
+
description: |-
|
|
40
|
+
Container CPU utilization is above 80%
|
|
41
|
+
VALUE = {{ $value }}
|
|
42
|
+
LABELS = {{ $labels }}
|
|
43
|
+
- alert: ContainerHighMemoryUsage
|
|
44
|
+
expr: (sum(container_memory_working_set_bytes{name!=""}) BY (instance, name) / sum(container_spec_memory_limit_bytes > 0) BY (instance, name) * 100) > 80
|
|
45
|
+
for: 2m
|
|
46
|
+
labels:
|
|
47
|
+
severity: warning
|
|
48
|
+
annotations:
|
|
49
|
+
summary: Container High Memory usage (instance {{ $labels.instance }})
|
|
50
|
+
description: |-
|
|
51
|
+
Container Memory usage is above 80%
|
|
52
|
+
VALUE = {{ $value }}
|
|
53
|
+
LABELS = {{ $labels }}
|
|
54
|
+
- alert: ContainerVolumeUsage
|
|
55
|
+
expr: (1 - (sum(container_fs_inodes_free{name!=""}) BY (instance) / sum(container_fs_inodes_total) BY (instance))) * 100 > 80
|
|
56
|
+
for: 2m
|
|
57
|
+
labels:
|
|
58
|
+
severity: warning
|
|
59
|
+
annotations:
|
|
60
|
+
summary: Container Volume usage (instance {{ $labels.instance }})
|
|
61
|
+
description: |-
|
|
62
|
+
Container Volume usage is above 80%
|
|
63
|
+
VALUE = {{ $value }}
|
|
64
|
+
LABELS = {{ $labels }}
|
|
65
|
+
- alert: ContainerHighThrottleRate
|
|
66
|
+
expr: sum(increase(container_cpu_cfs_throttled_periods_total{container!=""}[5m])) by (container, pod, namespace) / sum(increase(container_cpu_cfs_periods_total[5m])) by (container, pod, namespace) > ( 25 / 100 )
|
|
67
|
+
for: 5m
|
|
68
|
+
labels:
|
|
69
|
+
severity: warning
|
|
70
|
+
annotations:
|
|
71
|
+
summary: Container high throttle rate (instance {{ $labels.instance }})
|
|
72
|
+
description: |-
|
|
73
|
+
Container is being throttled
|
|
74
|
+
VALUE = {{ $value }}
|
|
75
|
+
LABELS = {{ $labels }}
|
|
76
|
+
- alert: ContainerHighLowChangeCpuUsage
|
|
77
|
+
expr: (abs((sum by (instance, name) (rate(container_cpu_usage_seconds_total{name!=""}[1m])) * 100) - (sum by (instance, name) (rate(container_cpu_usage_seconds_total{name!=""}[1m] offset 1m)) * 100)) or abs((sum by (instance, name) (rate(container_cpu_usage_seconds_total{name!=""}[1m])) * 100) - (sum by (instance, name) (rate(container_cpu_usage_seconds_total{name!=""}[5m] offset 1m)) * 100))) > 25
|
|
78
|
+
for: 0m
|
|
79
|
+
labels:
|
|
80
|
+
severity: info
|
|
81
|
+
annotations:
|
|
82
|
+
summary: Container high low change CPU usage (instance {{ $labels.instance }})
|
|
83
|
+
description: |-
|
|
84
|
+
This alert rule monitors the absolute change in CPU usage within a time window and triggers an alert when the change exceeds 25%.
|
|
85
|
+
VALUE = {{ $value }}
|
|
86
|
+
LABELS = {{ $labels }}
|
|
87
|
+
- alert: ContainerLowCpuUtilization
|
|
88
|
+
expr: (sum(rate(container_cpu_usage_seconds_total{container!=""}[5m])) by (pod, container) / sum(container_spec_cpu_quota{container!=""}/container_spec_cpu_period{container!=""}) by (pod, container) * 100) < 20
|
|
89
|
+
for: 7d
|
|
90
|
+
labels:
|
|
91
|
+
severity: info
|
|
92
|
+
annotations:
|
|
93
|
+
summary: Container Low CPU utilization (instance {{ $labels.instance }})
|
|
94
|
+
description: |-
|
|
95
|
+
Container CPU utilization is under 20% for 1 week. Consider reducing the allocated CPU.
|
|
96
|
+
VALUE = {{ $value }}
|
|
97
|
+
LABELS = {{ $labels }}
|
|
98
|
+
- alert: ContainerLowMemoryUsage
|
|
99
|
+
expr: (sum(container_memory_working_set_bytes{name!=""}) BY (instance, name) / sum(container_spec_memory_limit_bytes > 0) BY (instance, name) * 100) < 20
|
|
100
|
+
for: 7d
|
|
101
|
+
labels:
|
|
102
|
+
severity: info
|
|
103
|
+
annotations:
|
|
104
|
+
summary: Container Low Memory usage (instance {{ $labels.instance }})
|
|
105
|
+
description: |-
|
|
106
|
+
Container Memory usage is under 20% for 1 week. Consider reducing the allocated memory.
|
|
107
|
+
VALUE = {{ $value }}
|
|
108
|
+
LABELS = {{ $labels }}
|
|
@@ -1,10 +1,11 @@
|
|
|
1
|
+
# https://samber.github.io/awesome-prometheus-alerts/rules#kubernetes
|
|
1
2
|
apiVersion: monitoring.coreos.com/v1
|
|
2
3
|
kind: PrometheusRule
|
|
3
4
|
metadata:
|
|
4
5
|
name: kubestate-exporter
|
|
5
6
|
spec:
|
|
6
7
|
groups:
|
|
7
|
-
- name: KubestateExporter-rules
|
|
8
|
+
- name: Kubestate (awesome) # KubestateExporter-rules
|
|
8
9
|
rules:
|
|
9
10
|
- alert: KubernetesNodeNotReady
|
|
10
11
|
expr: kube_node_status_condition{condition="Ready",status="true"} == 0
|