add some kubelet alarms (#5724)
##### Summary Add some alarms for kubelet module from #5720 ##### Component Name [/health/health.d](https://github.com/netdata/netdata/tree/master/health/health.d)
Ilya Mashchenko committed
Mar 29, 2019 at 13:27 UTC
1e682dc59cf955636e62b1ac0c0a150f3b27943d
2 files changed
+116
health/Makefile.am
+1
@@ -46,6 +46,7 @@ dist_healthconfig_DATA = \
46
health.d/ipfs.conf \
47
health.d/ipmi.conf \
48
health.d/isc_dhcpd.conf \
49
+ health.d/kubelet.conf \
50
health.d/lighttpd.conf \
51
health.d/linux_power_supply.conf \
52
health.d/load.conf \
health/health.d/kubelet.conf
new
+115
@@ -0,0 +1,115 @@
1
+# you can disable an alarm notification by setting the 'to' line to: silent
2
+
3
+# -----------------------------------------------------------------------------
4
+
5
+# True (1) if the node is experiencing a configuration-related error, false (0) otherwise.
6
+
7
+ template: node_config_error
8
+ on: k8s_kubelet.kubelet_node_config_error
9
+ calc: $kubelet_node_config_error
10
+ units: bool
11
+ every: 10s
12
+ warn: $this == 1
13
+ delay: down 1m multiplier 1.5 max 2h
14
+ info: the node is experiencing a configuration-related error
15
+ to: sysadmin
16
+
17
+# Failed Token() requests to the alternate token source
18
+
19
+ template: token_requests
20
+ lookup: sum -10s of token_fail_count
21
+ on: k8s_kubelet.kubelet_token_requests
22
+ units: failed requests
23
+ every: 10s
24
+ warn: $this > 0
25
+ delay: down 1m multiplier 1.5 max 2h
26
+ info: failed token requests to alternate token source
27
+ to: sysadmin
28
+
29
+# Docker and runtime operation errors
30
+
31
+ template: kubelet_operations_error
32
+ lookup: sum -1m
33
+ on: k8s_kubelet.kubelet_operations_errors
34
+ units: errors
35
+ every: 10s
36
+ warn: $this > (($status >= $WARNING) ? (0) : (20))
37
+ delay: up 30s down 1m multiplier 1.5 max 2h
38
+ info: operations error
39
+ to: sysadmin
40
+
41
+# -----------------------------------------------------------------------------
42
+
43
+# Pod Lifecycle Event Generator Relisting Latency
44
+
45
+# 1. calculate the pleg relisting latency for 1m (quantile 0.5, quantile 0.9, quantile 0.99)
46
+# 2. do the same for the last 10s
47
+# 3. raise an alarm if the later is:
48
+# - 2x the first for quantile 0.5
49
+# - 4x the first for quantile 0.9
50
+# - 8x the first for quantile 0.99
51
+#
52
+# we assume the minimum latency is 1000 microseconds
53
+
54
+# quantile 0.5
55
+
56
+template: 1m_kubelet_pleg_relist_latency_quantile_05
57
+ on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
58
+ lookup: average -1m unaligned of kubelet_pleg_relist_latency_05
59
+ units: microseconds
60
+ every: 10s
61
+ info: the average value of pleg relisting latency during the last minute (quantile 0.5)
62
+
63
+template: 10s_kubelet_pleg_relist_latency_quantile_05
64
+ on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
65
+ lookup: average -10s unaligned of kubelet_pleg_relist_latency_05
66
+ calc: $this * 100 / (($1m_kubelet_pleg_relist_latency_quantile_05 < 1000)?(1000):($1m_kubelet_pleg_relist_latency_quantile_05))
67
+ every: 10s
68
+ units: %
69
+ warn: $this > (($status >= $WARNING)?(100):(200))
70
+ crit: $this > (($status >= $WARNING)?(200):(400))
71
+ delay: down 1m multiplier 1.5 max 2h
72
+ info: the % of the pleg relisting latency in the last 10 seconds, compared to the last minute (quantile 0.5)
73
+ to: sysadmin
74
+
75
+# quantile 0.9
76
+
77
+template: 1m_kubelet_pleg_relist_latency_quantile_09
78
+ on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
79
+ lookup: average -1m unaligned of kubelet_pleg_relist_latency_09
80
+ units: microseconds
81
+ every: 10s
82
+ info: the average value of pleg relisting latency during the last minute (quantile 0.9)
83
+
84
+template: 10s_kubelet_pleg_relist_latency_quantile_09
85
+ on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
86
+ lookup: average -10s unaligned of kubelet_pleg_relist_latency_09
87
+ calc: $this * 100 / (($1m_kubelet_pleg_relist_latency_quantile_09 < 1000)?(1000):($1m_kubelet_pleg_relist_latency_quantile_09))
88
+ every: 10s
89
+ units: %
90
+ warn: $this > (($status >= $WARNING)?(200):(400))
91
+ crit: $this > (($status >= $WARNING)?(400):(800))
92
+ delay: down 1m multiplier 1.5 max 2h
93
+ info: the % of the pleg relisting latency in the last 10 seconds, compared to the last minute (quantile 0.9)
94
+ to: sysadmin
95
+
96
+# quantile 0.99
97
+
98
+template: 1m_kubelet_pleg_relist_latency_quantile_099
99
+ on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
100
+ lookup: average -1m unaligned of kubelet_pleg_relist_latency_099
101
+ units: microseconds
102
+ every: 10s
103
+ info: the average value of pleg relisting latency during the last minute (quantile 0.99)
104
+
105
+template: 10s_kubelet_pleg_relist_latency_quantile_099
106
+ on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
107
+ lookup: average -10s unaligned of kubelet_pleg_relist_latency_099
108
+ calc: $this * 100 / (($1m_kubelet_pleg_relist_latency_quantile_099 < 1000)?(1000):($1m_kubelet_pleg_relist_latency_quantile_099))
109
+ every: 10s
110
+ units: %
111
+ warn: $this > (($status >= $WARNING)?(400):(800))
112
+ crit: $this > (($status >= $WARNING)?(800):(1200))
113
+ delay: down 1m multiplier 1.5 max 2h
114
+ info: the % of the pleg relisting latency in the last 10 seconds, compared to the last minute (quantile 0.99)
115
+ to: sysadmin