web_log fixes and improvements
Costa Tsaousis (ktsaou) committed
Feb 11, 2017 at 21:31 UTC
3b051b5a2e2a09d4c57f7a7b4354150a1acf9ce8
3 files changed
+55
-30
conf.d/health.d/web_log.conf
+44
-24
@@ -13,9 +13,17 @@ families: *
13
info: number of seconds since the last successful data collection
14
to: webmaster
15
16
+
17
# -----------------------------------------------------------------------------
18
# high level response code alarms
19
20
+# the following alarms trigger only when there are enough data.
21
+# we assume there are enough data when:
22
+#
23
+# $1m_requests > 120
24
+#
25
+# i.e. when there are at least 120 requests during the last minute
26
+
27
template: 1m_requests
28
on: web_log.response_codes
29
families: *
@@ -40,58 +48,62 @@ families: *
48
calc: $1m_2xx * 100 / $1m_requests
49
units: %
50
every: 10s
43
- warn: ($1m_requests > 30) ? ($this < (($status >= $WARNING ) ? ( 98 ) : ( 95 )) ) : ( 0 )
44
- crit: ($1m_requests > 30) ? ($this < (($status == $CRITICAL) ? ( 95 ) : ( 90 )) ) : ( 0 )
51
+ warn: ($1m_requests > 120) ? ($this < (($status >= $WARNING ) ? ( 98 ) : ( 95 )) ) : ( 0 )
52
+ crit: ($1m_requests > 120) ? ($this < (($status == $CRITICAL) ? ( 95 ) : ( 90 )) ) : ( 0 )
53
delay: down 15m multiplier 1.5 max 1h
46
- info: the ratio of HTTP redirects (3xx) vs the successful requests, \
47
- over the last minute
54
+ info: the ratio of successful HTTP responses (2xx) over the last minute
55
to: webmaster
56
57
template: 1m_redirects
58
on: web_log.detailed_response_codes
59
families: *
60
lookup: sum -1m unaligned of 301,303,307,308
54
- calc: $this * 100 / ( $1m_2xx + $this )
61
+ calc: $this * 100 / $1m_requests
62
units: %
63
every: 10s
57
- warn: ($1m_requests > 30) ? ($this > (($status >= $WARNING ) ? ( 1 ) : ( 2 )) ) : ( 0 )
58
- crit: ($1m_requests > 30) ? ($this > (($status == $CRITICAL) ? ( 2 ) : ( 5 )) ) : ( 0 )
64
+ warn: ($1m_requests > 120) ? ($this > (($status >= $WARNING ) ? ( 1 ) : ( 2 )) ) : ( 0 )
65
+ crit: ($1m_requests > 120) ? ($this > (($status == $CRITICAL) ? ( 2 ) : ( 5 )) ) : ( 0 )
66
delay: down 15m multiplier 1.5 max 1h
60
- info: the ratio of HTTP redirects (301, 303, 307, 308) vs the successful requests, \
61
- over the last minute
67
+ info: the ratio of HTTP redirects (301, 303, 307, 308) over the last minute
68
to: webmaster
69
70
template: 1m_bad_requests
71
on: web_log.response_codes
72
families: *
73
lookup: sum -1m unaligned of 4xx
68
- calc: $this * 100 / ( $1m_2xx + $this )
74
+ calc: $this * 100 / $1m_requests
75
units: %
76
every: 10s
71
- warn: ($1m_requests > 30) ? ($this > (($status >= $WARNING) ? ( 1 ) : ( 5 )) ) : ( 0 )
72
- crit: ($1m_requests > 30) ? ($this > (($status == $CRITICAL) ? ( 5 ) : ( 10 )) ) : ( 0 )
77
+ warn: ($1m_requests > 120) ? ($this > (($status >= $WARNING) ? ( 1 ) : ( 5 )) ) : ( 0 )
78
+ crit: ($1m_requests > 120) ? ($this > (($status == $CRITICAL) ? ( 5 ) : ( 10 )) ) : ( 0 )
79
delay: down 15m multiplier 1.5 max 1h
74
- info: the ratio of HTTP bad requests (4xx) vs the successful requests, \
75
- over the last minute
80
+ info: the ratio of HTTP bad requests (4xx) over the last minute
81
to: webmaster
82
83
template: 1m_internal_errors
84
on: web_log.response_codes
85
families: *
86
lookup: sum -1m unaligned of 5xx
82
- calc: $this * 100 / ( $1m_2xx + $this )
87
+ calc: $this * 100 / $1m_requests
88
units: %
89
every: 10s
85
- warn: ($1m_requests > 30) ? ($this > (($status >= $WARNING) ? ( 1 ) : ( 2 )) ) : ( 0 )
86
- crit: ($1m_requests > 30) ? ($this > (($status == $CRITICAL) ? ( 2 ) : ( 5 )) ) : ( 0 )
90
+ warn: ($1m_requests > 120) ? ($this > (($status >= $WARNING) ? ( 1 ) : ( 2 )) ) : ( 0 )
91
+ crit: ($1m_requests > 120) ? ($this > (($status == $CRITICAL) ? ( 2 ) : ( 5 )) ) : ( 0 )
92
delay: down 15m multiplier 1.5 max 1h
88
- info: the ratio of HTTP internal server errors (5xx) vs the successful \
89
- requests, over the last minute
93
+ info: the ratio of HTTP internal server errors (5xx), over the last minute
94
to: webmaster
95
96
+
97
# -----------------------------------------------------------------------------
98
# web slow
99
100
+# the following alarms trigger only when there are enough data.
101
+# we assume there are enough data when:
102
+#
103
+# $1m_requests > 120
104
+#
105
+# i.e. when there are at least 120 requests during the last minute
106
+
107
template: 10m_response_time
108
on: web_log.response_time
109
families: *
@@ -108,8 +120,8 @@ families: *
120
every: 10s
121
green: 500
122
red: 1000
111
- warn: ($1m_requests > 30) ? ($this > $green && $this > ($10m_response_time * 2) ) : ( 0 )
112
- crit: ($1m_requests > 30) ? ($this > $red && $this > ($10m_response_time * 4) ) : ( 0 )
123
+ warn: ($1m_requests > 120) ? ($this > $green && $this > ($10m_response_time * 2) ) : ( 0 )
124
+ crit: ($1m_requests > 120) ? ($this > $red && $this > ($10m_response_time * 4) ) : ( 0 )
125
delay: down 15m multiplier 1.5 max 1h
126
info: the average time to respond to HTTP requests, over the last 1 minute
127
to: webmaster
@@ -117,6 +129,14 @@ families: *
129
# -----------------------------------------------------------------------------
130
# web too many or too few requests
131
132
+# the following alarms trigger only when there are enough data.
133
+# we assume there are enough data when:
134
+#
135
+# $5m_2xx_last > 120
136
+#
137
+# i.e. when there were at least 120 requests during the 5 minutes starting
138
+# at -10m and ending at -5m
139
+
140
template: 5m_2xx_last
141
on: web_log.response_codes
142
families: *
@@ -139,11 +159,11 @@ families: *
159
calc: ($5m_2xx_last > 0)?($5m_2xx_now * 100 / $5m_2xx_last):(100)
160
units: %
161
every: 30s
142
- warn: ($1m_requests > 30) ? (($5m_2xx_last > 30) ? ($this > 200 OR $this < 50) : (0) ) : ( 0 )
143
- crit: ($1m_requests > 30) ? (($5m_2xx_last > 30) ? ($this > 400 OR $this < 25) : (0) ) : ( 0 )
162
+ warn: ($5m_2xx_last > 120) ? ($this > 200 OR $this < 50) : (0)
163
+ crit: ($5m_2xx_last > 120) ? ($this > 400 OR $this < 25) : (0)
164
delay: down 15m multiplier 1.5 max 1h
165
options: no-clear-notification
166
info: the percentage of web requests over the last 5 minutes, \
147
- compared with the previous 15 minutes
167
+ compared with the previous 5 minutes
168
to: webmaster
169
web/dashboard_info.js
+10
-5
@@ -857,7 +857,7 @@ netdataDashboard.context = {
857
+ ' data-dimensions="2xx"'
858
+ ' data-chart-library="gauge"'
859
+ ' data-title="Successful"'
860
- + ' data-units="requests"'
860
+ + ' data-units="requests/s"'
861
+ ' data-gauge-adjust="width"'
862
+ ' data-width="12%"'
863
+ ' data-before="0"'
@@ -865,6 +865,7 @@ netdataDashboard.context = {
865
+ ' data-points="CHART_DURATION"'
866
+ ' data-common-max="' + id + '"'
867
+ ' data-colors="' + NETDATA.colors[0] + '"'
868
+ + ' data-decimal-digits="0"'
869
+ ' role="application"></div>';
870
},
871
@@ -874,7 +875,7 @@ netdataDashboard.context = {
875
+ ' data-dimensions="3xx"'
876
+ ' data-chart-library="gauge"'
877
+ ' data-title="Redirects"'
877
- + ' data-units="requests"'
878
+ + ' data-units="requests/s"'
879
+ ' data-gauge-adjust="width"'
880
+ ' data-width="12%"'
881
+ ' data-before="0"'
@@ -882,6 +883,7 @@ netdataDashboard.context = {
883
+ ' data-points="CHART_DURATION"'
884
+ ' data-common-max="' + id + '"'
885
+ ' data-colors="' + NETDATA.colors[2] + '"'
886
+ + ' data-decimal-digits="0"'
887
+ ' role="application"></div>';
888
},
889
@@ -891,7 +893,7 @@ netdataDashboard.context = {
893
+ ' data-dimensions="4xx"'
894
+ ' data-chart-library="gauge"'
895
+ ' data-title="Bad Requests"'
894
- + ' data-units="requests"'
896
+ + ' data-units="requests/s"'
897
+ ' data-gauge-adjust="width"'
898
+ ' data-width="12%"'
899
+ ' data-before="0"'
@@ -899,6 +901,7 @@ netdataDashboard.context = {
901
+ ' data-points="CHART_DURATION"'
902
+ ' data-common-max="' + id + '"'
903
+ ' data-colors="' + NETDATA.colors[3] + '"'
904
+ + ' data-decimal-digits="0"'
905
+ ' role="application"></div>';
906
},
907
@@ -908,7 +911,7 @@ netdataDashboard.context = {
911
+ ' data-dimensions="5xx"'
912
+ ' data-chart-library="gauge"'
913
+ ' data-title="Server Errors"'
911
- + ' data-units="requests"'
914
+ + ' data-units="requests/s"'
915
+ ' data-gauge-adjust="width"'
916
+ ' data-width="12%"'
917
+ ' data-before="0"'
@@ -916,6 +919,7 @@ netdataDashboard.context = {
919
+ ' data-points="CHART_DURATION"'
920
+ ' data-common-max="' + id + '"'
921
+ ' data-colors="' + NETDATA.colors[1] + '"'
922
+ + ' data-decimal-digits="0"'
923
+ ' role="application"></div>';
924
}
925
]
@@ -928,7 +932,7 @@ netdataDashboard.context = {
932
return '<div data-netdata="' + id + '"'
933
+ ' data-dimensions="avg"'
934
+ ' data-chart-library="gauge"'
931
- + ' data-title="Average Response"'
935
+ + ' data-title="Average Response Time"'
936
+ ' data-units="milliseconds"'
937
+ ' data-gauge-adjust="width"'
938
+ ' data-width="12%"'
@@ -936,6 +940,7 @@ netdataDashboard.context = {
940
+ ' data-after="-CHART_DURATION"'
941
+ ' data-points="CHART_DURATION"'
942
+ ' data-colors="' + NETDATA.colors[4] + '"'
943
+ + ' data-decimal-digits="2"'
944
+ ' role="application"></div>';
945
}
946
]
web/index.html
+1
-1
@@ -2800,7 +2800,7 @@
2800
});
2801
2802
NETDATA.requiredJs.push({
2803
- url: NETDATA.serverDefault + 'dashboard_info.js?v20170211-19',
2803
+ url: NETDATA.serverDefault + 'dashboard_info.js?v20170211-20',
2804
async: false,
2805
isAlreadyLoaded: function() { return false; }
2806
});