updated alarms to use O/S and hosts filtering based on simple patterns; #2608
Costa Tsaousis (ktsaou) committed
Sep 5, 2017 at 23:33 UTC
344cb4c893da7c52730d546749cd300481b93829
15 files changed
+122
-20
conf.d/Makefile.am
+14
-19
@@ -70,45 +70,40 @@ dist_healthconfig_DATA = \
70
health.d/apache.conf \
71
health.d/backend.conf \
72
health.d/bind_rndc.conf \
73
+ health.d/cpu.conf \
74
+ health.d/disks.conf \
75
health.d/elasticsearch.conf \
76
+ health.d/entropy.conf \
77
health.d/fping.conf \
78
health.d/haproxy.conf \
79
+ health.d/ipc.conf \
80
health.d/ipfs.conf \
81
health.d/ipmi.conf \
82
health.d/isc_dhcpd.conf \
83
health.d/lighttpd.conf \
84
health.d/mdstat.conf \
85
health.d/memcached.conf \
86
+ health.d/memory.conf \
87
+ health.d/mongodb.conf \
88
health.d/mysql.conf \
89
health.d/named.conf \
84
- health.d/mongodb.conf \
85
- health.d/nginx.conf \
86
- health.d/postgres.conf \
87
- health.d/redis.conf \
88
- health.d/retroshare.conf \
89
- health.d/squid.conf \
90
- health.d/varnish.conf \
91
- health.d/web_log.conf \
92
- health.d/zfs.conf \
93
- $(NULL)
94
-
95
-if LINUX
96
-dist_healthconfig_DATA += \
97
- health.d/cpu.conf \
98
- health.d/disks.conf \
99
- health.d/entropy.conf \
100
- health.d/ipc.conf \
101
- health.d/memory.conf \
90
health.d/net.conf \
91
health.d/netfilter.conf \
92
+ health.d/nginx.conf \
93
+ health.d/postgres.conf \
94
health.d/qos.conf \
95
health.d/ram.conf \
96
+ health.d/redis.conf \
97
+ health.d/retroshare.conf \
98
health.d/softnet.conf \
99
+ health.d/squid.conf \
100
health.d/swap.conf \
101
health.d/tcp_resets.conf \
102
health.d/udp_errors.conf \
103
+ health.d/varnish.conf \
104
+ health.d/web_log.conf \
105
+ health.d/zfs.conf \
106
$(NULL)
111
-endif LINUX
107
108
chartsconfigdir=$(configdir)/charts.d
109
dist_chartsconfig_DATA = \
conf.d/health.d/cpu.conf
+8
@@ -1,6 +1,10 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
template: 10min_cpu_usage
5
on: system.cpu
6
+ os: linux
7
+ hosts: *
8
lookup: average -10m unaligned of user,system,softirq,irq,guest
9
units: %
10
every: 1m
@@ -12,6 +16,8 @@ template: 10min_cpu_usage
16
17
template: 10min_cpu_iowait
18
on: system.cpu
19
+ os: linux
20
+ hosts: *
21
lookup: average -10m unaligned of iowait
22
units: %
23
every: 1m
@@ -23,6 +29,8 @@ template: 10min_cpu_iowait
29
30
template: 20min_steal_cpu
31
on: system.cpu
32
+ os: linux
33
+ hosts: *
34
lookup: average -20m unaligned of steal
35
units: %
36
every: 5m
conf.d/health.d/disks.conf
+14
@@ -1,3 +1,7 @@
1
+
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
+
5
# -----------------------------------------------------------------------------
6
# low disk space
7
@@ -7,6 +11,8 @@
11
12
template: disk_space_usage
13
on: disk.space
14
+ os: linux
15
+ hosts: *
16
families: *
17
calc: $used * 100 / ($avail + $used)
18
units: %
@@ -43,6 +49,8 @@ families: *
49
50
template: disk_fill_rate
51
on: disk.space
52
+ os: linux
53
+ hosts: *
54
families: *
55
lookup: min -10m at -50m unaligned of avail
56
calc: ($this - $avail) / (($now - $after) / 3600)
@@ -57,6 +65,8 @@ families: *
65
66
template: out_of_disk_space_time
67
on: disk.space
68
+ os: linux
69
+ hosts: *
70
families: *
71
calc: ($disk_fill_rate > 0) ? ($avail / $disk_fill_rate) : (inf)
72
units: hours
@@ -77,6 +87,8 @@ families: *
87
88
template: 10min_disk_utilization
89
on: disk.util
90
+ os: linux
91
+ hosts: *
92
families: *
93
lookup: average -10m unaligned
94
units: %
@@ -97,6 +109,8 @@ families: *
109
110
template: 10min_disk_backlog
111
on: disk.backlog
112
+ os: linux
113
+ hosts: *
114
families: *
115
lookup: average -10m unaligned
116
units: ms
conf.d/health.d/entropy.conf
+2
@@ -5,6 +5,8 @@
5
6
alarm: lowest_entropy
7
on: system.entropy
8
+ os: linux
9
+ hosts: *
10
lookup: min -10m unaligned
11
units: entries
12
every: 5m
conf.d/health.d/ipc.conf
+6
@@ -1,6 +1,10 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
alarm: semaphores_used
5
on: system.ipc_semaphores
6
+ os: linux
7
+ hosts: *
8
calc: $semaphores * 100 / $ipc.semaphores.max
9
units: %
10
every: 10s
@@ -12,6 +16,8 @@
16
17
alarm: semaphore_arrays_used
18
on: system.ipc_semaphore_arrays
19
+ os: linux
20
+ hosts: *
21
calc: $arrays * 100 / $ipc.semaphores.arrays.max
22
units: %
23
every: 10s
conf.d/health.d/memory.conf
+8
@@ -1,6 +1,10 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
alarm: 1hour_ecc_memory_correctable
5
on: mem.ecc_ce
6
+ os: linux
7
+ hosts: *
8
lookup: sum -10m unaligned
9
units: errors
10
every: 1m
@@ -11,6 +15,8 @@
15
16
alarm: 1hour_ecc_memory_uncorrectable
17
on: mem.ecc_ue
18
+ os: linux
19
+ hosts: *
20
lookup: sum -10m unaligned
21
units: errors
22
every: 1m
@@ -21,6 +27,8 @@
27
28
alarm: 1hour_memory_hw_corrupted
29
on: mem.hwcorrupt
30
+ os: linux
31
+ hosts: *
32
calc: $HardwareCorrupted
33
units: MB
34
every: 10s
conf.d/health.d/net.conf
+16
@@ -1,4 +1,6 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
# -----------------------------------------------------------------------------
5
# dropped packets
6
@@ -8,6 +10,8 @@
10
11
template: inbound_packets_dropped
12
on: net.drops
13
+ os: linux
14
+ hosts: *
15
families: *
16
lookup: sum -10m unaligned absolute of inbound
17
units: packets
@@ -19,6 +23,8 @@ families: *
23
24
template: outbound_packets_dropped
25
on: net.drops
26
+ os: linux
27
+ hosts: *
28
families: *
29
lookup: sum -10m unaligned absolute of outbound
30
units: packets
@@ -30,6 +36,8 @@ families: *
36
37
template: inbound_packets_dropped_ratio
38
on: net.packets
39
+ os: linux
40
+ hosts: *
41
families: *
42
lookup: sum -10m unaligned absolute of received
43
calc: (($inbound_packets_dropped != nan AND $this > 0) ? ($inbound_packets_dropped * 100 / $this) : (0))
@@ -43,6 +51,8 @@ families: *
51
52
template: outbound_packets_dropped_ratio
53
on: net.packets
54
+ os: linux
55
+ hosts: *
56
families: *
57
lookup: sum -10m unaligned absolute of sent
58
calc: (($outbound_packets_dropped != nan AND $this > 0) ? ($outbound_packets_dropped * 100 / $this) : (0))
@@ -65,6 +75,8 @@ families: *
75
76
template: 10min_fifo_errors
77
on: net.fifo
78
+ os: linux
79
+ hosts: *
80
families: *
81
lookup: sum -10m unaligned absolute
82
units: errors
@@ -86,6 +98,8 @@ families: *
98
99
template: 1m_received_packets_rate
100
on: net.packets
101
+ os: linux
102
+ hosts: *
103
families: *
104
lookup: average -1m of received
105
units: packets
@@ -94,6 +108,8 @@ families: *
108
109
template: 10s_received_packets_storm
110
on: net.packets
111
+ os: linux
112
+ hosts: *
113
families: *
114
lookup: average -10s of received
115
calc: $this * 100 / (($1m_received_packets_rate < 1000)?(1000):($1m_received_packets_rate))
conf.d/health.d/netfilter.conf
+6
@@ -1,6 +1,10 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
alarm: netfilter_last_collected_secs
5
on: netfilter.conntrack_sockets
6
+ os: linux
7
+ hosts: *
8
calc: $now - $last_collected_t
9
units: seconds ago
10
every: 10s
@@ -12,6 +16,8 @@
16
17
alarm: netfilter_conntrack_full
18
on: netfilter.conntrack_sockets
19
+ os: linux
20
+ hosts: *
21
lookup: max -10s unaligned of connections
22
calc: $this * 100 / $netfilter.conntrack.max
23
units: %
conf.d/health.d/qos.conf
+4
@@ -1,10 +1,14 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
# check if a QoS class is dropping packets
5
# the alarm is checked every 10 seconds
6
# and examines the last minute of data
7
8
#template: 10min_qos_packet_drops
9
# on: tc.qos_dropped
10
+# os: linux
11
+# hosts: *
12
# lookup: sum -10m unaligned absolute
13
# every: 30s
14
# warn: $this > 0
conf.d/health.d/ram.conf
+6
@@ -1,12 +1,18 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
alarm: used_ram_to_ignore
5
on: system.ram
6
+ os: linux
7
+ hosts: *
8
calc: ($zfs.arc_size.arcsz = nan)?(0):($zfs.arc_size.arcsz)
9
every: 10s
10
info: the amount of memory that is reported as used, but it is actually capable for resizing itself based on the system needs (eg. ZFS ARC)
11
12
alarm: ram_in_use
13
on: system.ram
14
+ os: linux
15
+ hosts: *
16
# calc: $used * 100 / ($used + $cached + $free)
17
calc: ($used - $used_ram_to_ignore) * 100 / ($used - $used_ram_to_ignore + $cached + $free)
18
units: %
conf.d/health.d/softnet.conf
+7
@@ -1,7 +1,12 @@
1
+
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
# check for common /proc/net/softnet_stat errors
5
6
alarm: 10min_netdev_backlog_exceeded
7
on: system.softnet_stat
8
+ os: linux
9
+ hosts: *
10
lookup: sum -10m unaligned absolute of dropped
11
units: packets
12
every: 1m
@@ -12,6 +17,8 @@
17
18
alarm: 10min_netdev_budget_ran_outs
19
on: system.softnet_stat
20
+ os: linux
21
+ hosts: *
22
lookup: sum -10m unaligned absolute of squeezed
23
units: events
24
every: 1m
conf.d/health.d/swap.conf
+8
@@ -1,6 +1,10 @@
1
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
alarm: 30min_ram_swapped_out
5
on: system.swapio
6
+ os: linux
7
+ hosts: *
8
lookup: sum -30m unaligned absolute of out
9
# we have to convert KB to MB by dividing $this (i.e. the result of the lookup) with 1024
10
calc: $this / 1024 * 100 / ( $system.ram.used + $system.ram.cached + $system.ram.free )
@@ -14,6 +18,8 @@
18
19
alarm: ram_in_swap
20
on: system.swap
21
+ os: linux
22
+ hosts: *
23
calc: $used * 100 / ( $system.ram.used + $system.ram.cached + $system.ram.free )
24
units: % of RAM
25
every: 10s
@@ -25,6 +31,8 @@
31
32
alarm: used_swap
33
on: system.swap
34
+ os: linux
35
+ hosts: *
36
calc: $used * 100 / ( $used + $free )
37
units: %
38
every: 10s
conf.d/health.d/tcp_resets.conf
+13
@@ -1,7 +1,12 @@
1
+
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
# -----------------------------------------------------------------------------
5
6
alarm: ipv4_tcphandshake_last_collected_secs
7
on: ipv4.tcphandshake
8
+ os: linux
9
+ hosts: *
10
calc: $now - $last_collected_t
11
units: seconds ago
12
every: 10s
@@ -16,6 +21,8 @@
21
22
alarm: 1m_ipv4_tcp_resets_sent
23
on: ipv4.tcphandshake
24
+ os: linux
25
+ hosts: *
26
lookup: average -1m at -10s unaligned absolute of OutRsts
27
units: tcp resets/s
28
every: 10s
@@ -23,6 +30,8 @@
30
31
alarm: 10s_ipv4_tcp_resets_sent
32
on: ipv4.tcphandshake
33
+ os: linux
34
+ hosts: *
35
lookup: average -10s unaligned absolute of OutRsts
36
units: tcp resets/s
37
every: 10s
@@ -37,6 +46,8 @@ options: no-clear-notification
46
47
alarm: 1m_ipv4_tcp_resets_received
48
on: ipv4.tcphandshake
49
+ os: linux
50
+ hosts: *
51
lookup: average -1m at -10s unaligned absolute of AttemptFails
52
units: tcp resets/s
53
every: 10s
@@ -44,6 +55,8 @@ options: no-clear-notification
55
56
alarm: 10s_ipv4_tcp_resets_received
57
on: ipv4.tcphandshake
58
+ os: linux
59
+ hosts: *
60
lookup: average -10s unaligned absolute of AttemptFails
61
units: tcp resets/s
62
every: 10s
conf.d/health.d/udp_errors.conf
+9
@@ -1,7 +1,12 @@
1
+
2
+# you can disable an alarm notification by setting the 'to' line to: silent
3
+
4
# -----------------------------------------------------------------------------
5
6
alarm: ipv4_udperrors_last_collected_secs
7
on: ipv4.udperrors
8
+ os: linux
9
+ hosts: *
10
calc: $now - $last_collected_t
11
units: seconds ago
12
every: 10s
@@ -16,6 +21,8 @@
21
22
alarm: 1m_ipv4_udp_receive_buffer_errors
23
on: ipv4.udperrors
24
+ os: linux
25
+ hosts: *
26
lookup: sum -1m unaligned absolute of RcvbufErrors
27
units: errors
28
every: 10s
@@ -30,6 +37,8 @@
37
38
alarm: 1m_ipv4_udp_send_buffer_errors
39
on: ipv4.udperrors
40
+ os: linux
41
+ hosts: *
42
lookup: sum -1m unaligned absolute of SndbufErrors
43
units: errors
44
every: 10s
src/health_config.c
+1
-1
@@ -6,7 +6,7 @@
6
#define HEALTH_ALARM_KEY "alarm"
7
#define HEALTH_TEMPLATE_KEY "template"
8
#define HEALTH_ON_KEY "on"
9
-#define HEALTH_HOST_KEY "host"
9
+#define HEALTH_HOST_KEY "hosts"
10
#define HEALTH_OS_KEY "os"
11
#define HEALTH_FAMILIES_KEY "families"
12
#define HEALTH_LOOKUP_KEY "lookup"