@cryptotaxi247 / netdata-1 / commits / ef721290f

Fine tune various alarm values. (#7322)

* Fix formatting in alarm configurations. This makes sure everything is lined up properly so that the alarm definitions are easier to read. * Make TCP Accept Queue alarms much less aggressive. This switches the alarms to use averages instead of sums, and bumps up the trip points to be more aggressive, as both of these may be non-zero even in normal operation of a system. * Make softnet alarms less aggressive. This decreases the sampling window from 10 minutes to 1 minute, switches to using an average instead of a sum, and adjusts the trigger thresholds to be more aggressive. This one will need to be watched, as the resultant values may be too lenient for some systems. * Tweak UDP alarms to work like the TCP alarms. Just to ensure consistency.

Austin S. Hemmelgarn committed Nov 15, 2019 at 13:14 UTC ef721290f03ec9d0292f062672feaad78d7fe3a9
8 files changed +120 -121
health/health.d/dnsmasq_dhcp.conf
+2 -2
@@ -1,6 +1,6 @@
1 - # dhcp-range utilization
1 +# dhcp-range utilization
2
3 - template: dnsmasq_dhcp_dhcp_range_utilization
3 +template: dnsmasq_dhcp_dhcp_range_utilization
4 on: dnsmasq_dhcp.dhcp_range_utilization
5 every: 10s
6 units: %
health/health.d/megacli.conf
+16 -16
@@ -1,48 +1,48 @@
1 alarm: adapter_state
2 on: megacli.adapter_degraded
3 units: is degraded
4 - lookup: sum -10s
5 - every: 10s
4 + lookup: sum -10s
5 + every: 10s
6 crit: $this > 0
7 info: adapter state
8 to: sysadmin
9
10 - template: bbu_relative_charge
10 +template: bbu_relative_charge
11 on: megacli.bbu_relative_charge
12 units: percent
13 - lookup: average -10s
14 - every: 10s
13 + lookup: average -10s
14 + every: 10s
15 warn: $this <= (($status >= $WARNING) ? (85) : (80))
16 crit: $this <= (($status == $CRITICAL) ? (50) : (40))
17 info: BBU relative state of charge
18 to: sysadmin
19
20 - template: bbu_cycle_count
20 +template: bbu_cycle_count
21 on: megacli.bbu_cycle_count
22 units: cycle count
23 - lookup: average -10s
24 - every: 10s
23 + lookup: average -10s
24 + every: 10s
25 warn: $this >= 100
26 crit: $this >= 500
27 info: BBU cycle count
28 to: sysadmin
29
30 - alarm: pd_media_errors
30 + alarm: pd_media_errors
31 on: megacli.pd_media_error
32 units: media errors
33 - lookup: sum -10s
34 - every: 10s
33 + lookup: sum -10s
34 + every: 10s
35 warn: $this > 0
36 - delay: down 1m multiplier 2 max 10m
36 + delay: down 1m multiplier 2 max 10m
37 info: physical drive media errors
38 to: sysadmin
39
40 - alarm: pd_predictive_failures
40 + alarm: pd_predictive_failures
41 on: megacli.pd_predictive_failure
42 units: predictive failures
43 - lookup: sum -10s
44 - every: 10s
43 + lookup: sum -10s
44 + every: 10s
45 warn: $this > 0
46 - delay: down 1m multiplier 2 max 10m
46 + delay: down 1m multiplier 2 max 10m
47 info: physical drive predictive failures
48 to: sysadmin
health/health.d/net.conf
+5 -5
@@ -161,8 +161,8 @@ families: *
161 calc: $this * 100 / (($1m_received_packets_rate < 1000)?(1000):($1m_received_packets_rate))
162 every: 10s
163 units: %
164 - warn: $this > (($status >= $WARNING)?(200):(5000))
165 - crit: $this > (($status >= $WARNING)?(5000):(6000))
166 -options: no-clear-notification
167 - info: the % of the rate of received packets in the last 10 seconds, compared to the rate of the last minute (clear notification for this alarm will not be sent)
168 - to: sysadmin
164 + warn: $this > (($status >= $WARNING)?(200):(5000))
165 + crit: $this > (($status >= $WARNING)?(5000):(6000))
166 + options: no-clear-notification
167 + info: the % of the rate of received packets in the last 10 seconds, compared to the rate of the last minute (clear notification for this alarm will not be sent)
168 + to: sysadmin
health/health.d/pihole.conf
+44 -46
@@ -1,5 +1,5 @@
1
2 - # Make sure Pi-hole is responding.
2 +# Make sure Pi-hole is responding.
3
4 template: pihole_last_collected_secs
5 on: pihole.dns_queries_total
@@ -12,56 +12,54 @@ template: pihole_last_collected_secs
12 info: number of seconds since the last successful data collection
13 to: webmaster
14
15 - # Blocked DNS queries.
15 +# Blocked DNS queries.
16
17 - template: pihole_blocked_queries
18 - on: pihole.dns_queries_percentage
19 - every: 10s
20 - units: %
21 - calc: $blocked
22 - warn: $this > ( ($status >= $WARNING ) ? ( 45 ) : ( 55 ) )
23 - crit: $this > ( ($status >= $CRITICAL) ? ( 55 ) : ( 75 ) )
24 - delay: up 2m down 5m
25 - info: percentage of blocked dns queries for the last 24 hour
26 - to: sysadmin
27 -
28 -
29 - # Blocklist last update time.
30 - # Default update interval is a week.
17 +template: pihole_blocked_queries
18 + on: pihole.dns_queries_percentage
19 + every: 10s
20 + units: %
21 + calc: $blocked
22 + warn: $this > ( ($status >= $WARNING ) ? ( 45 ) : ( 55 ) )
23 + crit: $this > ( ($status >= $CRITICAL) ? ( 55 ) : ( 75 ) )
24 + delay: up 2m down 5m
25 + info: percentage of blocked dns queries for the last 24 hour
26 + to: sysadmin
27
32 - template: pihole_blocklist_last_update
33 - on: pihole.blocklist_last_update
34 - every: 10s
35 - units: seconds
36 - calc: $ago
37 - warn: $this > 60 * 60 * 24 * 8
38 - crit: $this > 60 * 60 * 24 * 8 * 2
39 - info: blocklist last update time
40 - to: sysadmin
28
29 +# Blocklist last update time.
30 +# Default update interval is a week.
31
43 - # Gravity file check (gravity.list).
32 +template: pihole_blocklist_last_update
33 + on: pihole.blocklist_last_update
34 + every: 10s
35 + units: seconds
36 + calc: $ago
37 + warn: $this > 60 * 60 * 24 * 8
38 + crit: $this > 60 * 60 * 24 * 8 * 2
39 + info: blocklist last update time
40 + to: sysadmin
41
45 - template: pihole_blocklist_gravity_file
46 - on: pihole.blocklist_last_update
47 - every: 10s
48 - units: boolean
49 - calc: $file_exists
50 - crit: $this != 1
51 - delay: up 2m down 5m
52 - info: gravity file existence
53 - to: sysadmin
42 +# Gravity file check (gravity.list).
43
44 +template: pihole_blocklist_gravity_file
45 + on: pihole.blocklist_last_update
46 + every: 10s
47 + units: boolean
48 + calc: $file_exists
49 + crit: $this != 1
50 + delay: up 2m down 5m
51 + info: gravity file existence
52 + to: sysadmin
53
56 - # Pi-hole's ability to block unwanted domains.
57 - # Should be enabled. The whole point of Pi-hole!
54 +# Pi-hole's ability to block unwanted domains.
55 +# Should be enabled. The whole point of Pi-hole!
56
59 - template: pihole_status
60 - on: pihole.unwanted_domains_blocking_status
61 - every: 10s
62 - units: boolean
63 - calc: $enabled
64 - warn: $this != 1
65 - delay: up 2m down 5m
66 - info: unwanted domains blocking status
67 - to: sysadmin
57 +template: pihole_status
58 + on: pihole.unwanted_domains_blocking_status
59 + every: 10s
60 + units: boolean
61 + calc: $enabled
62 + warn: $this != 1
63 + delay: up 2m down 5m
64 + info: unwanted domains blocking status
65 + to: sysadmin
health/health.d/ram.conf
+24 -24
@@ -37,28 +37,28 @@
37 to: sysadmin
38
39 ## FreeBSD
40 -alarm: ram_in_use
41 - on: system.ram
42 - os: freebsd
43 -hosts: *
44 - calc: ($active + $wired + $laundry + $buffers - $used_ram_to_ignore) * 100 / ($active + $wired + $laundry + $buffers - $used_ram_to_ignore + $cache + $free + $inactive)
45 -units: %
46 -every: 10s
47 - warn: $this > (($status >= $WARNING) ? (80) : (90))
48 - crit: $this > (($status == $CRITICAL) ? (90) : (98))
49 -delay: down 15m multiplier 1.5 max 1h
50 - info: system RAM usage
51 - to: sysadmin
40 + alarm: ram_in_use
41 + on: system.ram
42 + os: freebsd
43 + hosts: *
44 + calc: ($active + $wired + $laundry + $buffers - $used_ram_to_ignore) * 100 / ($active + $wired + $laundry + $buffers - $used_ram_to_ignore + $cache + $free + $inactive)
45 + units: %
46 + every: 10s
47 + warn: $this > (($status >= $WARNING) ? (80) : (90))
48 + crit: $this > (($status == $CRITICAL) ? (90) : (98))
49 + delay: down 15m multiplier 1.5 max 1h
50 + info: system RAM usage
51 + to: sysadmin
52
53 - alarm: ram_available
54 - on: system.ram
55 - os: freebsd
56 - hosts: *
57 - calc: ($free + $inactive + $used_ram_to_ignore) * 100 / ($free + $active + $inactive + $wired + $cache + $laundry + $buffers)
58 - units: %
59 - every: 10s
60 - warn: $this < (($status >= $WARNING) ? (15) : (10))
61 - crit: $this < (($status == $CRITICAL) ? (10) : ( 5))
62 - delay: down 15m multiplier 1.5 max 1h
63 - info: estimated amount of RAM available for userspace processes, without causing swapping
64 - to: sysadmin
53 + alarm: ram_available
54 + on: system.ram
55 + os: freebsd
56 + hosts: *
57 + calc: ($free + $inactive + $used_ram_to_ignore) * 100 / ($free + $active + $inactive + $wired + $cache + $laundry + $buffers)
58 + units: %
59 + every: 10s
60 + warn: $this < (($status >= $WARNING) ? (15) : (10))
61 + crit: $this < (($status == $CRITICAL) ? (10) : ( 5))
62 + delay: down 15m multiplier 1.5 max 1h
63 + info: estimated amount of RAM available for userspace processes, without causing swapping
64 + to: sysadmin
health/health.d/softnet.conf
+14 -14
@@ -3,38 +3,38 @@
3
4 # check for common /proc/net/softnet_stat errors
5
6 - alarm: 10min_netdev_backlog_exceeded
6 + alarm: 1min_netdev_backlog_exceeded
7 on: system.softnet_stat
8 os: linux
9 hosts: *
10 - lookup: sum -10m unaligned absolute of dropped
10 + lookup: average -1m unaligned absolute of dropped
11 units: packets
12 - every: 1m
13 - warn: $this > 0
12 + every: 10s
13 + warn: $this > (($status >= $WARNING) ? (0) : (10)
14 delay: down 1h multiplier 1.5 max 2h
15 - info: number of packets dropped in the last 10min, because sysctl net.core.netdev_max_backlog was exceeded (this can be a cause for dropped packets)
15 + info: average number of packets dropped in the last 1min, because sysctl net.core.netdev_max_backlog was exceeded (this can be a cause for dropped packets)
16 to: sysadmin
17
18 - alarm: 10min_netdev_budget_ran_outs
18 + alarm: 1min_netdev_budget_ran_outs
19 on: system.softnet_stat
20 os: linux
21 hosts: *
22 - lookup: sum -10m unaligned absolute of squeezed
22 + lookup: average -1m unaligned absolute of squeezed
23 units: events
24 - every: 1m
25 - warn: $this > (($status >= $WARNING) ? (0) : (10))
24 + every: 10s
25 + warn: $this > (($status >= $WARNING) ? (0) : (10))
26 delay: down 1h multiplier 1.5 max 2h
27 - info: number of times, during the last 10min, ksoftirq ran out of sysctl net.core.netdev_budget or net.core.netdev_budget_usecs, with work remaining (this can be a cause for dropped packets)
27 + info: average number of times, during the last 1min, ksoftirq ran out of sysctl net.core.netdev_budget or net.core.netdev_budget_usecs, with work remaining (this can be a cause for dropped packets)
28 to: silent
29
30 alarm: 10min_netisr_backlog_exceeded
31 on: system.softnet_stat
32 os: freebsd
33 hosts: *
34 - lookup: sum -10m unaligned absolute of qdrops
34 + lookup: average -1m unaligned absolute of qdrops
35 units: packets
36 - every: 1m
37 - warn: $this > 0
36 + every: 10s
37 + warn: $this > (($status >+ $WARNING) ? (0) : (10))
38 delay: down 1h multiplier 1.5 max 2h
39 - info: number of drops in the last 10min, because sysctl net.route.netisr_maxqlen was exceeded (this can be a cause for dropped packets)
39 + info: average number of drops in the last 1min, because sysctl net.route.netisr_maxqlen was exceeded (this can be a cause for dropped packets)
40 to: sysadmin
health/health.d/tcp_listen.conf
+8 -7
@@ -22,12 +22,13 @@
22 on: ip.tcp_accept_queue
23 os: linux
24 hosts: *
25 - lookup: sum -60s unaligned absolute of ListenOverflows
25 + lookup: average -60s unaligned absolute of ListenOverflows
26 units: overflows
27 every: 10s
28 - crit: $this > 0
28 + warn: $this > 1
29 + crit: $this > (($status == $CRITICAL) ? (1) : (5))
30 delay: up 0 down 5m multiplier 1.5 max 1h
30 - info: the number of times the TCP accept queue of the kernel overflown, during the last minute
31 + info: the average number of times the TCP accept queue of the kernel overflown, during the last minute
32 to: sysadmin
33
34 # THIS IS TOO GENERIC
@@ -36,13 +37,13 @@
37 on: ip.tcp_accept_queue
38 os: linux
39 hosts: *
39 - lookup: sum -60s unaligned absolute of ListenDrops
40 + lookup: average -60s unaligned absolute of ListenDrops
41 units: drops
42 every: 10s
42 -# warn: $this > 0
43 - crit: $this > (($status == $CRITICAL) ? (0) : (150))
43 + warn: $this > 1
44 + crit: $this > (($status == $CRITICAL) ? (1) : (5))
45 delay: up 0 down 5m multiplier 1.5 max 1h
45 - info: the number of times the TCP accept queue of the kernel dropped packets, during the last minute (includes bogus packets received)
46 + info: the average number of times the TCP accept queue of the kernel dropped packets, during the last minute (includes bogus packets received)
47 to: sysadmin
48
49
health/health.d/udp_errors.conf
+7 -7
@@ -23,12 +23,12 @@
23 on: ipv4.udperrors
24 os: linux freebsd
25 hosts: *
26 - lookup: sum -1m unaligned absolute of RcvbufErrors
26 + lookup: average -1m unaligned absolute of RcvbufErrors
27 units: errors
28 every: 10s
29 - warn: $this > 0
30 - crit: $this > (($status == $CRITICAL) ? (0) : (100))
31 - info: number of UDP receive buffer errors during the last minute
29 + warn: $this > 1
30 + crit: $this > (($status == $CRITICAL) ? (0) : (10))
31 + info: average number of UDP receive buffer errors during the last minute
32 delay: up 0 down 60m multiplier 1.2 max 2h
33 to: sysadmin
34
@@ -39,11 +39,11 @@
39 on: ipv4.udperrors
40 os: linux
41 hosts: *
42 - lookup: sum -1m unaligned absolute of SndbufErrors
42 + lookup: average -1m unaligned absolute of SndbufErrors
43 units: errors
44 every: 10s
45 - warn: $this > 0
46 - crit: $this > (($status == $CRITICAL) ? (0) : (100))
45 + warn: $this > 1
46 + crit: $this > (($status == $CRITICAL) ? (0) : (10))
47 info: number of UDP send buffer errors during the last minute
48 delay: up 0 down 60m multiplier 1.2 max 2h
49 to: sysadmin