@cryptotaxi247 / netdata-1 / commits / 1e682dc59

add some kubelet alarms (#5724)

##### Summary Add some alarms for kubelet module from #5720 ##### Component Name [/health/health.d](https://github.com/netdata/netdata/tree/master/health/health.d)

Ilya Mashchenko committed Mar 29, 2019 at 13:27 UTC 1e682dc59cf955636e62b1ac0c0a150f3b27943d
2 files changed +116
health/Makefile.am
+1
@@ -46,6 +46,7 @@ dist_healthconfig_DATA = \
46 health.d/ipfs.conf \
47 health.d/ipmi.conf \
48 health.d/isc_dhcpd.conf \
49 + health.d/kubelet.conf \
50 health.d/lighttpd.conf \
51 health.d/linux_power_supply.conf \
52 health.d/load.conf \
health/health.d/kubelet.conf new
+115
@@ -0,0 +1,115 @@
1 +# you can disable an alarm notification by setting the 'to' line to: silent
2 +
3 +# -----------------------------------------------------------------------------
4 +
5 +# True (1) if the node is experiencing a configuration-related error, false (0) otherwise.
6 +
7 + template: node_config_error
8 + on: k8s_kubelet.kubelet_node_config_error
9 + calc: $kubelet_node_config_error
10 + units: bool
11 + every: 10s
12 + warn: $this == 1
13 + delay: down 1m multiplier 1.5 max 2h
14 + info: the node is experiencing a configuration-related error
15 + to: sysadmin
16 +
17 +# Failed Token() requests to the alternate token source
18 +
19 + template: token_requests
20 + lookup: sum -10s of token_fail_count
21 + on: k8s_kubelet.kubelet_token_requests
22 + units: failed requests
23 + every: 10s
24 + warn: $this > 0
25 + delay: down 1m multiplier 1.5 max 2h
26 + info: failed token requests to alternate token source
27 + to: sysadmin
28 +
29 +# Docker and runtime operation errors
30 +
31 + template: kubelet_operations_error
32 + lookup: sum -1m
33 + on: k8s_kubelet.kubelet_operations_errors
34 + units: errors
35 + every: 10s
36 + warn: $this > (($status >= $WARNING) ? (0) : (20))
37 + delay: up 30s down 1m multiplier 1.5 max 2h
38 + info: operations error
39 + to: sysadmin
40 +
41 +# -----------------------------------------------------------------------------
42 +
43 +# Pod Lifecycle Event Generator Relisting Latency
44 +
45 +# 1. calculate the pleg relisting latency for 1m (quantile 0.5, quantile 0.9, quantile 0.99)
46 +# 2. do the same for the last 10s
47 +# 3. raise an alarm if the later is:
48 +# - 2x the first for quantile 0.5
49 +# - 4x the first for quantile 0.9
50 +# - 8x the first for quantile 0.99
51 +#
52 +# we assume the minimum latency is 1000 microseconds
53 +
54 +# quantile 0.5
55 +
56 +template: 1m_kubelet_pleg_relist_latency_quantile_05
57 + on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
58 + lookup: average -1m unaligned of kubelet_pleg_relist_latency_05
59 + units: microseconds
60 + every: 10s
61 + info: the average value of pleg relisting latency during the last minute (quantile 0.5)
62 +
63 +template: 10s_kubelet_pleg_relist_latency_quantile_05
64 + on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
65 + lookup: average -10s unaligned of kubelet_pleg_relist_latency_05
66 + calc: $this * 100 / (($1m_kubelet_pleg_relist_latency_quantile_05 < 1000)?(1000):($1m_kubelet_pleg_relist_latency_quantile_05))
67 + every: 10s
68 + units: %
69 + warn: $this > (($status >= $WARNING)?(100):(200))
70 + crit: $this > (($status >= $WARNING)?(200):(400))
71 + delay: down 1m multiplier 1.5 max 2h
72 + info: the % of the pleg relisting latency in the last 10 seconds, compared to the last minute (quantile 0.5)
73 + to: sysadmin
74 +
75 +# quantile 0.9
76 +
77 +template: 1m_kubelet_pleg_relist_latency_quantile_09
78 + on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
79 + lookup: average -1m unaligned of kubelet_pleg_relist_latency_09
80 + units: microseconds
81 + every: 10s
82 + info: the average value of pleg relisting latency during the last minute (quantile 0.9)
83 +
84 +template: 10s_kubelet_pleg_relist_latency_quantile_09
85 + on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
86 + lookup: average -10s unaligned of kubelet_pleg_relist_latency_09
87 + calc: $this * 100 / (($1m_kubelet_pleg_relist_latency_quantile_09 < 1000)?(1000):($1m_kubelet_pleg_relist_latency_quantile_09))
88 + every: 10s
89 + units: %
90 + warn: $this > (($status >= $WARNING)?(200):(400))
91 + crit: $this > (($status >= $WARNING)?(400):(800))
92 + delay: down 1m multiplier 1.5 max 2h
93 + info: the % of the pleg relisting latency in the last 10 seconds, compared to the last minute (quantile 0.9)
94 + to: sysadmin
95 +
96 +# quantile 0.99
97 +
98 +template: 1m_kubelet_pleg_relist_latency_quantile_099
99 + on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
100 + lookup: average -1m unaligned of kubelet_pleg_relist_latency_099
101 + units: microseconds
102 + every: 10s
103 + info: the average value of pleg relisting latency during the last minute (quantile 0.99)
104 +
105 +template: 10s_kubelet_pleg_relist_latency_quantile_099
106 + on: k8s_kubelet.kubelet_pleg_relist_latency_microseconds
107 + lookup: average -10s unaligned of kubelet_pleg_relist_latency_099
108 + calc: $this * 100 / (($1m_kubelet_pleg_relist_latency_quantile_099 < 1000)?(1000):($1m_kubelet_pleg_relist_latency_quantile_099))
109 + every: 10s
110 + units: %
111 + warn: $this > (($status >= $WARNING)?(400):(800))
112 + crit: $this > (($status >= $WARNING)?(800):(1200))
113 + delay: down 1m multiplier 1.5 max 2h
114 + info: the % of the pleg relisting latency in the last 10 seconds, compared to the last minute (quantile 0.99)
115 + to: sysadmin