added charts for traefik reverse proxy
amenezes committed
Mar 18, 2018 at 21:30 UTC
b596491ae3ccb1c6da5e26d9ad4a705125b1eff1
5 files changed
+319
-2
conf.d/Makefile.am
+1
@@ -71,6 +71,7 @@ dist_pythonconfig_DATA = \
71
python.d/squid.conf \
72
python.d/smartd_log.conf \
73
python.d/tomcat.conf \
74
+ python.d/traefik.conf \
75
python.d/varnish.conf \
76
python.d/web_log.conf \
77
$(NULL)
conf.d/python.d/traefik.conf
new
+79
@@ -0,0 +1,79 @@
1
+# netdata python.d.plugin configuration for traefik health data API
2
+#
3
+# This file is in YaML format. Generally the format is:
4
+#
5
+# name: value
6
+#
7
+# There are 2 sections:
8
+# - global variables
9
+# - one or more JOBS
10
+#
11
+# JOBS allow you to collect values from multiple sources.
12
+# Each source will have its own set of charts.
13
+#
14
+# JOB parameters have to be indented (using spaces only, example below).
15
+
16
+# ----------------------------------------------------------------------
17
+# Global Variables
18
+# These variables set the defaults for all JOBs, however each JOB
19
+# may define its own, overriding the defaults.
20
+
21
+# update_every sets the default data collection frequency.
22
+# If unset, the python.d.plugin default is used.
23
+# update_every: 1
24
+
25
+# priority controls the order of charts at the netdata dashboard.
26
+# Lower numbers move the charts towards the top of the page.
27
+# If unset, the default for python.d.plugin is used.
28
+# priority: 60000
29
+
30
+# retries sets the number of retries to be made in case of failures.
31
+# If unset, the default for python.d.plugin is used.
32
+# Attempts to restore the service are made once every update_every
33
+# and only if the module has collected values in the past.
34
+# retries: 60
35
+
36
+# autodetection_retry sets the job re-check interval in seconds.
37
+# The job is not deleted if check fails.
38
+# Attempts to start the job are made once every autodetection_retry.
39
+# This feature is disabled by default.
40
+# autodetection_retry: 0
41
+
42
+# ----------------------------------------------------------------------
43
+# JOBS (data collection sources)
44
+#
45
+# The default JOBS share the same *name*. JOBS with the same name
46
+# are mutually exclusive. Only one of them will be allowed running at
47
+# any time. This allows autodetection to try several alternatives and
48
+# pick the one that works.
49
+#
50
+# Any number of jobs is supported.
51
+#
52
+# All python.d.plugin JOBS (for all its modules) support a set of
53
+# predefined parameters. These are:
54
+#
55
+# job_name:
56
+# name: myname # the JOB's name as it will appear at the
57
+# # dashboard (by default is the job_name)
58
+# # JOBs sharing a name are mutually exclusive
59
+# update_every: 1 # the JOB's data collection frequency
60
+# priority: 60000 # the JOB's order on the dashboard
61
+# retries: 10 # the JOB's number of restoration attempts
62
+# autodetection_retry: 0 # the JOB's re-check interval in seconds
63
+#
64
+# Additionally to the above, traefik plugin also supports the following:
65
+#
66
+# url: '<scheme>://<host>:<port>/<health_page_api>'
67
+# # http://localhost:8080/health
68
+#
69
+# if the URL is password protected, the following are supported:
70
+#
71
+# user: 'username'
72
+# pass: 'password'
73
+#
74
+# ----------------------------------------------------------------------
75
+# AUTO-DETECTION JOBS
76
+# only one of them will run (they have the same name)
77
+#
78
+local:
79
+ url: 'http://localhost:8080/health'
python.d/Makefile.am
+1
@@ -59,6 +59,7 @@ dist_python_DATA = \
59
squid.chart.py \
60
smartd_log.chart.py \
61
tomcat.chart.py \
62
+ traefik.chart.py \
63
varnish.chart.py \
64
web_log.chart.py \
65
$(NULL)
python.d/README.md
+57
-2
@@ -897,10 +897,10 @@ server:
897
898
# icecast
899
900
-This module will monitor number of listeners for active sources.
900
+This module will monitor number of listeners for active sources.
901
902
**Requirements:**
903
- * icecast version >= 2.4.0
903
+ * icecast version >= 2.4.0
904
905
It produces the following charts:
906
@@ -2172,6 +2172,61 @@ So it will probably fail.
2172
2173
---
2174
2175
+# Traefik
2176
+
2177
+Module uses the `health` API to provide statistics.
2178
+
2179
+It produces:
2180
+
2181
+1. **Responses** by statuses
2182
+ * success (1xx, 2xx, 304)
2183
+ * error (5xx)
2184
+ * redirect (3xx except 304)
2185
+ * bad (4xx)
2186
+ * other (all other responses)
2187
+
2188
+2. **Responses** by codes
2189
+ * 2xx (successful)
2190
+ * 5xx (internal server errors)
2191
+ * 3xx (redirect)
2192
+ * 4xx (bad)
2193
+ * 1xx (informational)
2194
+ * other (non-standart responses)
2195
+
2196
+3. **Detailed Response Codes** requests/s (number of responses for each response code family individually)
2197
+
2198
+4. **Requests**/s
2199
+ * request statistics
2200
+
2201
+5. **Total response time**
2202
+ * sum of all response time
2203
+
2204
+6. **Average response time**
2205
+
2206
+7. **Average response time per iteration**
2207
+
2208
+8. **Uptime**
2209
+ * Traefik server uptime
2210
+
2211
+### configuration
2212
+
2213
+Needs only `url` to server's `health`
2214
+
2215
+Here is an example for local server:
2216
+
2217
+```yaml
2218
+update_every : 1
2219
+priority : 60000
2220
+
2221
+local:
2222
+ url : 'http://localhost:8080/health'
2223
+ retries : 10
2224
+```
2225
+
2226
+Without configuration, module attempts to connect to `http://localhost:8080/health`.
2227
+
2228
+---
2229
+
2230
# varnish cache
2231
2232
Module uses the `varnishstat` command to provide varnish cache statistics.
python.d/traefik.chart.py
new
+181
@@ -0,0 +1,181 @@
1
+# -*- coding: utf-8 -*-
2
+# Description: traefik netdata python.d module
3
+# Author: Alexandre Menezes (@ale_menezes)
4
+
5
+from json import loads
6
+from collections import defaultdict
7
+from bases.FrameworkServices.UrlService import UrlService
8
+
9
+# default module values (can be overridden per job in `config`)
10
+update_every = 1
11
+priority = 60000
12
+retries = 10
13
+
14
+# charts order (can be overridden if you want less charts, or different order)
15
+ORDER = [
16
+ 'response_statuses',
17
+ 'response_codes',
18
+ 'detailed_response_codes',
19
+ 'requests',
20
+ 'total_response_time',
21
+ 'average_response_time',
22
+ 'average_response_time_per_iteration',
23
+ 'uptime'
24
+]
25
+
26
+CHARTS = {
27
+ 'response_statuses': {
28
+ 'options': [None, 'Response statuses', 'requests/s', 'responses', 'traefik.response_statuses', 'stacked'],
29
+ 'lines': [
30
+ ['successful_requests', 'success', 'incremental'],
31
+ ['server_errors', 'error', 'incremental'],
32
+ ['redirects', 'redirect', 'incremental'],
33
+ ['bad_requests', 'bad', 'incremental'],
34
+ ['other_requests', 'other', 'incremental']
35
+ ]},
36
+ 'response_codes': {
37
+ 'options': [None, 'Responses by codes', 'requests/s', 'responses', 'traefik.response_codes', 'stacked'],
38
+ 'lines': [
39
+ ['2xx', None, 'incremental'],
40
+ ['5xx', None, 'incremental'],
41
+ ['3xx', None, 'incremental'],
42
+ ['4xx', None, 'incremental'],
43
+ ['1xx', None, 'incremental'],
44
+ ['other', None, 'incremental']
45
+ ]},
46
+ 'detailed_response_codes': {
47
+ 'options': [None, 'Detailed response codes', 'requests/s', 'responses', 'traefik.detailed_response_codes', 'stacked'],
48
+ 'lines': [
49
+ ]},
50
+ 'requests': {
51
+ 'options': [None, 'Requests', 'requests/s', 'requests', 'traefik.requests', 'line'],
52
+ 'lines': [
53
+ ['total_count', 'requests', 'incremental']
54
+ ]},
55
+ 'total_response_time': {
56
+ 'options': [None, 'Total response time', 'seconds', 'timings', 'traefik.total_response_time', 'line'],
57
+ 'lines': [
58
+ ['total_response_time_sec', 'response', 'absolute', 1, 10000]
59
+ ]},
60
+ 'average_response_time': {
61
+ 'options': [None, 'Average response time', 'milliseconds', 'timings', 'traefik.average_response_time', 'line'],
62
+ 'lines': [
63
+ ['average_response_time_sec', 'response', 'absolute', 1, 1000]
64
+ ]},
65
+ 'average_response_time_per_iteration': {
66
+ 'options': [None, 'Average response time per iteration', 'milliseconds', 'timings', 'traefik.average_response_time_per_iteration', 'line'],
67
+ 'lines': [
68
+ ['average_response_time_per_iteration_sec', 'response', 'incremental', 1, 10000]
69
+ ]},
70
+ 'uptime': {
71
+ 'options': [None, 'Uptime', 'seconds', 'uptime', 'traefik.uptime', 'line'],
72
+ 'lines': [
73
+ ['uptime_sec', 'uptime', 'absolute']
74
+ ]}
75
+ }
76
+
77
+HEALTH_STATS = [
78
+ 'uptime_sec',
79
+ 'average_response_time_sec',
80
+ 'total_response_time_sec',
81
+ 'total_count',
82
+ 'total_status_code_count'
83
+]
84
+
85
+class Service(UrlService):
86
+ def __init__(self, configuration=None, name=None):
87
+ UrlService.__init__(self, configuration=configuration, name=name)
88
+ self.url = self.configuration.get('url', 'http://localhost:8080/health')
89
+ self.order = ORDER
90
+ self.definitions = CHARTS
91
+ self.data = {
92
+ 'successful_requests': 0, 'redirects': 0, 'bad_requests': 0,
93
+ 'server_errors': 0, 'other_requests': 0, '1xx': 0, '2xx': 0,
94
+ '3xx': 0, '4xx': 0, '5xx': 0, 'other': 0,
95
+ 'average_response_time_per_iteration_sec': 0
96
+ }
97
+ self.last_total_response_time = 0
98
+ self.last_total_count = 0
99
+
100
+ def _get_data(self):
101
+ data = self._get_raw_data()
102
+
103
+ if not data:
104
+ return None
105
+
106
+ data = loads(data)
107
+
108
+ self.get_data_per_code_status(raw_data=data)
109
+
110
+ self.get_data_per_code_family(raw_data=data)
111
+
112
+ self.get_data_per_code(raw_data=data)
113
+
114
+ self.data.update(fetch_data_(raw_data=data, metrics=HEALTH_STATS))
115
+
116
+ self.data['average_response_time_sec'] *= 1000000
117
+ self.data['total_response_time_sec'] *= 10000
118
+ if data['total_count'] != self.last_total_count:
119
+ self.data['average_response_time_per_iteration_sec'] = (data['total_response_time_sec'] - self.last_total_response_time) * 1000000 / (data['total_count'] - self.last_total_count)
120
+ else:
121
+ self.data['average_response_time_per_iteration_sec'] = 0
122
+ self.last_total_response_time = data['total_response_time_sec']
123
+ self.last_total_count = data['total_count']
124
+
125
+ return self.data or None
126
+
127
+ def get_data_per_code_status(self, raw_data):
128
+ data = defaultdict(int)
129
+ for code, value in raw_data['total_status_code_count'].items():
130
+ code_prefix = code[0]
131
+ if code_prefix == '1' or code_prefix == '2' or code == '304':
132
+ data['successful_requests'] += value
133
+ elif code_prefix == '3':
134
+ data['redirects'] += value
135
+ elif code_prefix == '4':
136
+ data['bad_requests'] += value
137
+ elif code_prefix == '5':
138
+ data['server_errors'] += value
139
+ else:
140
+ data['other_requests'] += value
141
+ self.data.update(data)
142
+
143
+ def get_data_per_code_family(self, raw_data):
144
+ data = defaultdict(int)
145
+ for code, value in raw_data['total_status_code_count'].items():
146
+ code_prefix = code[0]
147
+ if code_prefix == '1':
148
+ data['1xx'] += value
149
+ elif code_prefix == '2':
150
+ data['2xx'] += value
151
+ elif code_prefix == '3':
152
+ data['3xx'] += value
153
+ elif code_prefix == '4':
154
+ data['4xx'] += value
155
+ elif code_prefix == '5':
156
+ data['5xx'] += value
157
+ else:
158
+ data['other'] += value
159
+ self.data.update(data)
160
+
161
+ def get_data_per_code(self, raw_data):
162
+ for code, value in raw_data['total_status_code_count'].items():
163
+ if self.charts:
164
+ if code not in self.data:
165
+ self.charts['detailed_response_codes'].add_dimension([code, code, 'incremental'])
166
+ self.data[code] = value
167
+
168
+def fetch_data_(raw_data, metrics):
169
+ data = dict()
170
+
171
+ for metric in metrics:
172
+ value = raw_data
173
+ metrics_list = metric.split('.')
174
+ try:
175
+ for m in metrics_list:
176
+ value = value[m]
177
+ except KeyError:
178
+ continue
179
+ data['_'.join(metrics_list)] = value
180
+
181
+ return data