@cryptotaxi247 / netdata-1 / commits / b596491ae

added charts for traefik reverse proxy

amenezes committed Mar 18, 2018 at 21:30 UTC b596491ae3ccb1c6da5e26d9ad4a705125b1eff1
5 files changed +319 -2
conf.d/Makefile.am
+1
@@ -71,6 +71,7 @@ dist_pythonconfig_DATA = \
71 python.d/squid.conf \
72 python.d/smartd_log.conf \
73 python.d/tomcat.conf \
74 + python.d/traefik.conf \
75 python.d/varnish.conf \
76 python.d/web_log.conf \
77 $(NULL)
conf.d/python.d/traefik.conf new
+79
@@ -0,0 +1,79 @@
1 +# netdata python.d.plugin configuration for traefik health data API
2 +#
3 +# This file is in YaML format. Generally the format is:
4 +#
5 +# name: value
6 +#
7 +# There are 2 sections:
8 +# - global variables
9 +# - one or more JOBS
10 +#
11 +# JOBS allow you to collect values from multiple sources.
12 +# Each source will have its own set of charts.
13 +#
14 +# JOB parameters have to be indented (using spaces only, example below).
15 +
16 +# ----------------------------------------------------------------------
17 +# Global Variables
18 +# These variables set the defaults for all JOBs, however each JOB
19 +# may define its own, overriding the defaults.
20 +
21 +# update_every sets the default data collection frequency.
22 +# If unset, the python.d.plugin default is used.
23 +# update_every: 1
24 +
25 +# priority controls the order of charts at the netdata dashboard.
26 +# Lower numbers move the charts towards the top of the page.
27 +# If unset, the default for python.d.plugin is used.
28 +# priority: 60000
29 +
30 +# retries sets the number of retries to be made in case of failures.
31 +# If unset, the default for python.d.plugin is used.
32 +# Attempts to restore the service are made once every update_every
33 +# and only if the module has collected values in the past.
34 +# retries: 60
35 +
36 +# autodetection_retry sets the job re-check interval in seconds.
37 +# The job is not deleted if check fails.
38 +# Attempts to start the job are made once every autodetection_retry.
39 +# This feature is disabled by default.
40 +# autodetection_retry: 0
41 +
42 +# ----------------------------------------------------------------------
43 +# JOBS (data collection sources)
44 +#
45 +# The default JOBS share the same *name*. JOBS with the same name
46 +# are mutually exclusive. Only one of them will be allowed running at
47 +# any time. This allows autodetection to try several alternatives and
48 +# pick the one that works.
49 +#
50 +# Any number of jobs is supported.
51 +#
52 +# All python.d.plugin JOBS (for all its modules) support a set of
53 +# predefined parameters. These are:
54 +#
55 +# job_name:
56 +# name: myname # the JOB's name as it will appear at the
57 +# # dashboard (by default is the job_name)
58 +# # JOBs sharing a name are mutually exclusive
59 +# update_every: 1 # the JOB's data collection frequency
60 +# priority: 60000 # the JOB's order on the dashboard
61 +# retries: 10 # the JOB's number of restoration attempts
62 +# autodetection_retry: 0 # the JOB's re-check interval in seconds
63 +#
64 +# Additionally to the above, traefik plugin also supports the following:
65 +#
66 +# url: '<scheme>://<host>:<port>/<health_page_api>'
67 +# # http://localhost:8080/health
68 +#
69 +# if the URL is password protected, the following are supported:
70 +#
71 +# user: 'username'
72 +# pass: 'password'
73 +#
74 +# ----------------------------------------------------------------------
75 +# AUTO-DETECTION JOBS
76 +# only one of them will run (they have the same name)
77 +#
78 +local:
79 + url: 'http://localhost:8080/health'
python.d/Makefile.am
+1
@@ -59,6 +59,7 @@ dist_python_DATA = \
59 squid.chart.py \
60 smartd_log.chart.py \
61 tomcat.chart.py \
62 + traefik.chart.py \
63 varnish.chart.py \
64 web_log.chart.py \
65 $(NULL)
python.d/README.md
+57 -2
@@ -897,10 +897,10 @@ server:
897
898 # icecast
899
900 -This module will monitor number of listeners for active sources.
900 +This module will monitor number of listeners for active sources.
901
902 **Requirements:**
903 - * icecast version >= 2.4.0
903 + * icecast version >= 2.4.0
904
905 It produces the following charts:
906
@@ -2172,6 +2172,61 @@ So it will probably fail.
2172
2173 ---
2174
2175 +# Traefik
2176 +
2177 +Module uses the `health` API to provide statistics.
2178 +
2179 +It produces:
2180 +
2181 +1. **Responses** by statuses
2182 + * success (1xx, 2xx, 304)
2183 + * error (5xx)
2184 + * redirect (3xx except 304)
2185 + * bad (4xx)
2186 + * other (all other responses)
2187 +
2188 +2. **Responses** by codes
2189 + * 2xx (successful)
2190 + * 5xx (internal server errors)
2191 + * 3xx (redirect)
2192 + * 4xx (bad)
2193 + * 1xx (informational)
2194 + * other (non-standart responses)
2195 +
2196 +3. **Detailed Response Codes** requests/s (number of responses for each response code family individually)
2197 +
2198 +4. **Requests**/s
2199 + * request statistics
2200 +
2201 +5. **Total response time**
2202 + * sum of all response time
2203 +
2204 +6. **Average response time**
2205 +
2206 +7. **Average response time per iteration**
2207 +
2208 +8. **Uptime**
2209 + * Traefik server uptime
2210 +
2211 +### configuration
2212 +
2213 +Needs only `url` to server's `health`
2214 +
2215 +Here is an example for local server:
2216 +
2217 +```yaml
2218 +update_every : 1
2219 +priority : 60000
2220 +
2221 +local:
2222 + url : 'http://localhost:8080/health'
2223 + retries : 10
2224 +```
2225 +
2226 +Without configuration, module attempts to connect to `http://localhost:8080/health`.
2227 +
2228 +---
2229 +
2230 # varnish cache
2231
2232 Module uses the `varnishstat` command to provide varnish cache statistics.
python.d/traefik.chart.py new
+181
@@ -0,0 +1,181 @@
1 +# -*- coding: utf-8 -*-
2 +# Description: traefik netdata python.d module
3 +# Author: Alexandre Menezes (@ale_menezes)
4 +
5 +from json import loads
6 +from collections import defaultdict
7 +from bases.FrameworkServices.UrlService import UrlService
8 +
9 +# default module values (can be overridden per job in `config`)
10 +update_every = 1
11 +priority = 60000
12 +retries = 10
13 +
14 +# charts order (can be overridden if you want less charts, or different order)
15 +ORDER = [
16 + 'response_statuses',
17 + 'response_codes',
18 + 'detailed_response_codes',
19 + 'requests',
20 + 'total_response_time',
21 + 'average_response_time',
22 + 'average_response_time_per_iteration',
23 + 'uptime'
24 +]
25 +
26 +CHARTS = {
27 + 'response_statuses': {
28 + 'options': [None, 'Response statuses', 'requests/s', 'responses', 'traefik.response_statuses', 'stacked'],
29 + 'lines': [
30 + ['successful_requests', 'success', 'incremental'],
31 + ['server_errors', 'error', 'incremental'],
32 + ['redirects', 'redirect', 'incremental'],
33 + ['bad_requests', 'bad', 'incremental'],
34 + ['other_requests', 'other', 'incremental']
35 + ]},
36 + 'response_codes': {
37 + 'options': [None, 'Responses by codes', 'requests/s', 'responses', 'traefik.response_codes', 'stacked'],
38 + 'lines': [
39 + ['2xx', None, 'incremental'],
40 + ['5xx', None, 'incremental'],
41 + ['3xx', None, 'incremental'],
42 + ['4xx', None, 'incremental'],
43 + ['1xx', None, 'incremental'],
44 + ['other', None, 'incremental']
45 + ]},
46 + 'detailed_response_codes': {
47 + 'options': [None, 'Detailed response codes', 'requests/s', 'responses', 'traefik.detailed_response_codes', 'stacked'],
48 + 'lines': [
49 + ]},
50 + 'requests': {
51 + 'options': [None, 'Requests', 'requests/s', 'requests', 'traefik.requests', 'line'],
52 + 'lines': [
53 + ['total_count', 'requests', 'incremental']
54 + ]},
55 + 'total_response_time': {
56 + 'options': [None, 'Total response time', 'seconds', 'timings', 'traefik.total_response_time', 'line'],
57 + 'lines': [
58 + ['total_response_time_sec', 'response', 'absolute', 1, 10000]
59 + ]},
60 + 'average_response_time': {
61 + 'options': [None, 'Average response time', 'milliseconds', 'timings', 'traefik.average_response_time', 'line'],
62 + 'lines': [
63 + ['average_response_time_sec', 'response', 'absolute', 1, 1000]
64 + ]},
65 + 'average_response_time_per_iteration': {
66 + 'options': [None, 'Average response time per iteration', 'milliseconds', 'timings', 'traefik.average_response_time_per_iteration', 'line'],
67 + 'lines': [
68 + ['average_response_time_per_iteration_sec', 'response', 'incremental', 1, 10000]
69 + ]},
70 + 'uptime': {
71 + 'options': [None, 'Uptime', 'seconds', 'uptime', 'traefik.uptime', 'line'],
72 + 'lines': [
73 + ['uptime_sec', 'uptime', 'absolute']
74 + ]}
75 + }
76 +
77 +HEALTH_STATS = [
78 + 'uptime_sec',
79 + 'average_response_time_sec',
80 + 'total_response_time_sec',
81 + 'total_count',
82 + 'total_status_code_count'
83 +]
84 +
85 +class Service(UrlService):
86 + def __init__(self, configuration=None, name=None):
87 + UrlService.__init__(self, configuration=configuration, name=name)
88 + self.url = self.configuration.get('url', 'http://localhost:8080/health')
89 + self.order = ORDER
90 + self.definitions = CHARTS
91 + self.data = {
92 + 'successful_requests': 0, 'redirects': 0, 'bad_requests': 0,
93 + 'server_errors': 0, 'other_requests': 0, '1xx': 0, '2xx': 0,
94 + '3xx': 0, '4xx': 0, '5xx': 0, 'other': 0,
95 + 'average_response_time_per_iteration_sec': 0
96 + }
97 + self.last_total_response_time = 0
98 + self.last_total_count = 0
99 +
100 + def _get_data(self):
101 + data = self._get_raw_data()
102 +
103 + if not data:
104 + return None
105 +
106 + data = loads(data)
107 +
108 + self.get_data_per_code_status(raw_data=data)
109 +
110 + self.get_data_per_code_family(raw_data=data)
111 +
112 + self.get_data_per_code(raw_data=data)
113 +
114 + self.data.update(fetch_data_(raw_data=data, metrics=HEALTH_STATS))
115 +
116 + self.data['average_response_time_sec'] *= 1000000
117 + self.data['total_response_time_sec'] *= 10000
118 + if data['total_count'] != self.last_total_count:
119 + self.data['average_response_time_per_iteration_sec'] = (data['total_response_time_sec'] - self.last_total_response_time) * 1000000 / (data['total_count'] - self.last_total_count)
120 + else:
121 + self.data['average_response_time_per_iteration_sec'] = 0
122 + self.last_total_response_time = data['total_response_time_sec']
123 + self.last_total_count = data['total_count']
124 +
125 + return self.data or None
126 +
127 + def get_data_per_code_status(self, raw_data):
128 + data = defaultdict(int)
129 + for code, value in raw_data['total_status_code_count'].items():
130 + code_prefix = code[0]
131 + if code_prefix == '1' or code_prefix == '2' or code == '304':
132 + data['successful_requests'] += value
133 + elif code_prefix == '3':
134 + data['redirects'] += value
135 + elif code_prefix == '4':
136 + data['bad_requests'] += value
137 + elif code_prefix == '5':
138 + data['server_errors'] += value
139 + else:
140 + data['other_requests'] += value
141 + self.data.update(data)
142 +
143 + def get_data_per_code_family(self, raw_data):
144 + data = defaultdict(int)
145 + for code, value in raw_data['total_status_code_count'].items():
146 + code_prefix = code[0]
147 + if code_prefix == '1':
148 + data['1xx'] += value
149 + elif code_prefix == '2':
150 + data['2xx'] += value
151 + elif code_prefix == '3':
152 + data['3xx'] += value
153 + elif code_prefix == '4':
154 + data['4xx'] += value
155 + elif code_prefix == '5':
156 + data['5xx'] += value
157 + else:
158 + data['other'] += value
159 + self.data.update(data)
160 +
161 + def get_data_per_code(self, raw_data):
162 + for code, value in raw_data['total_status_code_count'].items():
163 + if self.charts:
164 + if code not in self.data:
165 + self.charts['detailed_response_codes'].add_dimension([code, code, 'incremental'])
166 + self.data[code] = value
167 +
168 +def fetch_data_(raw_data, metrics):
169 + data = dict()
170 +
171 + for metric in metrics:
172 + value = raw_data
173 + metrics_list = metric.split('.')
174 + try:
175 + for m in metrics_list:
176 + value = value[m]
177 + except KeyError:
178 + continue
179 + data['_'.join(metrics_list)] = value
180 +
181 + return data