@cryptotaxi247 / netdata-1 / commits / 9cb561b8e

elasticsearch: bugfix and conf file update

Ilya committed Jun 16, 2017 at 22:14 UTC 9cb561b8e9af94409ec06daf7116926cb5e47a30
2 files changed +106 -105
conf.d/python.d/elasticsearch.conf
+7 -10
@@ -61,19 +61,16 @@
61 # cluster_health: False/True # Calls to cluster health elasticsearch API. Enabled by default.
62 # cluster_stats: False/True # Calls to cluster stats elasticsearch API. Enabled by default.
63 #
64 -# ----------------------------------------------------------------------
65 -# IMPORTANT Information
66 -#
67 -# Module uses python `requests` package
64 #
69 -# You need to install it manually. (python-requests or python3-requests depending on the version of python).
65 +# if the URL is password protected, the following are supported:
66 #
67 +# user: 'username'
68 +# pass: 'password'
69 #
70 +# ----------------------------------------------------------------------
71 # AUTO-DETECTION JOBS
72 # only one of them will run (they have the same name)
73 #
75 -#local:
76 -# host: '127.0.0.1'
77 -# port: '9200'
78 -# cluster_health: True
79 -# cluster_stats: True
74 +local:
75 + host: '127.0.0.1'
76 + port: '9200'
python.d/elasticsearch.chart.py
+99 -95
@@ -85,6 +85,21 @@ HEALTH_STATS = [
85 ('active_shards_percent_as_number', 'health_active_shards_percent_as_number', None)
86 ]
87
88 +LATENCY = {
89 + 'query_latency':
90 + {'total': 'query_total',
91 + 'spent_time': 'query_time_in_millis'},
92 + 'fetch_latency':
93 + {'total': 'fetch_total',
94 + 'spent_time': 'fetch_time_in_millis'},
95 + 'indexing_latency':
96 + {'total': 'indexing_index_total',
97 + 'spent_time': 'indexing_index_time_in_millis'},
98 + 'flushing_latency':
99 + {'total': 'flush_total',
100 + 'spent_time': 'flush_total_time_in_millis'}
101 +}
102 +
103 # charts order (can be overridden if you want less charts, or different order)
104 ORDER = ['search_perf_total', 'search_perf_current', 'search_perf_time', 'search_latency', 'index_perf_total',
105 'index_perf_current', 'index_perf_time', 'index_latency', 'jvm_mem_heap', 'jvm_gc_count',
@@ -95,34 +110,34 @@ ORDER = ['search_perf_total', 'search_perf_current', 'search_perf_time', 'search
110
111 CHARTS = {
112 'search_perf_total': {
98 - 'options': [None, 'Total number of queries, fetches', 'number of', 'search performance',
113 + 'options': [None, 'Queries And Fetches', 'number of', 'search performance',
114 'es.search_query_total', 'stacked'],
115 'lines': [
116 ['query_total', 'queries', 'incremental'],
117 ['fetch_total', 'fetches', 'incremental']
118 ]},
119 'search_perf_current': {
105 - 'options': [None, 'Number of queries, fetches in progress', 'number of', 'search performance',
120 + 'options': [None, 'Queries and Fetches In Progress', 'number of', 'search performance',
121 'es.search_query_current', 'stacked'],
122 'lines': [
123 ['query_current', 'queries', 'absolute'],
124 ['fetch_current', 'fetches', 'absolute']
125 ]},
126 'search_perf_time': {
112 - 'options': [None, 'Time spent on queries, fetches', 'seconds', 'search performance',
127 + 'options': [None, 'Time Spent On Queries And Fetches', 'seconds', 'search performance',
128 'es.search_time', 'stacked'],
129 'lines': [
130 ['query_time_in_millis', 'query', 'incremental', 1, 1000],
131 ['fetch_time_in_millis', 'fetch', 'incremental', 1, 1000]
132 ]},
133 'search_latency': {
119 - 'options': [None, 'Query and fetch latency', 'ms', 'search performance', 'es.search_latency', 'stacked'],
134 + 'options': [None, 'Query And Fetch Latency', 'ms', 'search performance', 'es.search_latency', 'stacked'],
135 'lines': [
136 ['query_latency', 'query', 'absolute', 1, 1000],
137 ['fetch_latency', 'fetch', 'absolute', 1, 1000]
138 ]},
139 'index_perf_total': {
125 - 'options': [None, 'Total number of documents indexed, index refreshes, index flushes to disk', 'number of',
140 + 'options': [None, 'Indexed Documents, Index Refreshes, Index Flushes To Disk', 'number of',
141 'indexing performance', 'es.index_performance_total', 'stacked'],
142 'lines': [
143 ['indexing_index_total', 'indexed', 'incremental'],
@@ -130,13 +145,13 @@ CHARTS = {
145 ['flush_total', 'flushes', 'incremental']
146 ]},
147 'index_perf_current': {
133 - 'options': [None, 'Number of documents currently being indexed', 'currently indexed',
148 + 'options': [None, 'Number Of Documents Currently Being Indexed', 'currently indexed',
149 'indexing performance', 'es.index_performance_current', 'stacked'],
150 'lines': [
151 ['indexing_index_current', 'documents', 'absolute']
152 ]},
153 'index_perf_time': {
139 - 'options': [None, 'Time spent on indexing, refreshing, flushing', 'seconds', 'indexing performance',
154 + 'options': [None, 'Time Spent On Indexing, Refreshing, Flushing', 'seconds', 'indexing performance',
155 'es.search_time', 'stacked'],
156 'lines': [
157 ['indexing_index_time_in_millis', 'indexing', 'incremental', 1, 1000],
@@ -144,33 +159,33 @@ CHARTS = {
159 ['flush_total_time_in_millis', 'flushing', 'incremental', 1, 1000]
160 ]},
161 'index_latency': {
147 - 'options': [None, 'Indexing and flushing latency', 'ms', 'indexing performance',
162 + 'options': [None, 'Indexing And Flushing Latency', 'ms', 'indexing performance',
163 'es.index_latency', 'stacked'],
164 'lines': [
165 ['indexing_latency', 'indexing', 'absolute', 1, 1000],
166 ['flushing_latency', 'flushing', 'absolute', 1, 1000]
167 ]},
168 'jvm_mem_heap': {
154 - 'options': [None, 'JVM heap currently in use/committed', 'percent/MB', 'memory usage and gc',
169 + 'options': [None, 'JVM Heap Currently in Use/Committed', 'percent/MB', 'memory usage and gc',
170 'es.jvm_heap', 'area'],
171 'lines': [
172 ['jvm_heap_percent', 'inuse', 'absolute'],
173 ['jvm_heap_commit', 'commit', 'absolute', -1, 1048576]
174 ]},
175 'jvm_gc_count': {
161 - 'options': [None, 'Count of garbage collections', 'counts', 'memory usage and gc', 'es.gc_count', 'stacked'],
176 + 'options': [None, 'Garbage Collections', 'counts', 'memory usage and gc', 'es.gc_count', 'stacked'],
177 'lines': [
178 ['young_collection_count', 'young', 'incremental'],
179 ['old_collection_count', 'old', 'incremental']
180 ]},
181 'jvm_gc_time': {
167 - 'options': [None, 'Time spent on garbage collections', 'ms', 'memory usage and gc', 'es.gc_time', 'stacked'],
182 + 'options': [None, 'Time Spent On Garbage Collections', 'ms', 'memory usage and gc', 'es.gc_time', 'stacked'],
183 'lines': [
184 ['young_collection_time_in_millis', 'young', 'incremental'],
185 ['old_collection_time_in_millis', 'old', 'incremental']
186 ]},
187 'thread_pool_qr_q': {
173 - 'options': [None, 'Number of queued threads in thread pool', 'queued threads', 'queues and rejections',
188 + 'options': [None, 'Number Of Queued Threads In Thread Pool', 'queued threads', 'queues and rejections',
189 'es.thread_pool_queued', 'stacked'],
190 'lines': [
191 ['bulk_queue', 'bulk', 'absolute'],
@@ -179,7 +194,7 @@ CHARTS = {
194 ['merge_queue', 'merge', 'absolute']
195 ]},
196 'thread_pool_qr_r': {
182 - 'options': [None, 'Number of rejected threads in thread pool', 'rejected threads', 'queues and rejections',
197 + 'options': [None, 'Rejected Threads In Thread Pool', 'rejected threads', 'queues and rejections',
198 'es.thread_pool_rejected', 'stacked'],
199 'lines': [
200 ['bulk_rejected', 'bulk', 'absolute'],
@@ -188,19 +203,19 @@ CHARTS = {
203 ['merge_rejected', 'merge', 'absolute']
204 ]},
205 'fdata_cache': {
191 - 'options': [None, 'Fielddata cache size', 'MB', 'fielddata cache', 'es.fdata_cache', 'line'],
206 + 'options': [None, 'Fielddata Cache', 'MB', 'fielddata cache', 'es.fdata_cache', 'line'],
207 'lines': [
208 ['index_fdata_memory', 'cache', 'absolute', 1, 1048576]
209 ]},
210 'fdata_ev_tr': {
196 - 'options': [None, 'Fielddata evictions and circuit breaker tripped count', 'number of events',
211 + 'options': [None, 'Fielddata Evictions And Circuit Breaker Tripped Count', 'number of events',
212 'fielddata cache', 'es.evictions_tripped', 'line'],
213 'lines': [
214 ['evictions', None, 'incremental'],
215 ['tripped', None, 'incremental']
216 ]},
217 'cluster_health_nodes': {
203 - 'options': [None, 'Nodes and tasks statistics', 'units', 'cluster health API',
218 + 'options': [None, 'Nodes And Tasks Statistics', 'units', 'cluster health API',
219 'es.cluster_health_nodes', 'stacked'],
220 'lines': [
221 ['health_number_of_nodes', 'nodes', 'absolute'],
@@ -209,7 +224,7 @@ CHARTS = {
224 ['health_number_of_in_flight_fetch', 'in_flight_fetch', 'absolute']
225 ]},
226 'cluster_health_status': {
212 - 'options': [None, 'Cluster status', 'status', 'cluster health API',
227 + 'options': [None, 'Cluster Status', 'status', 'cluster health API',
228 'es.cluster_health_status', 'area'],
229 'lines': [
230 ['status_green', 'green', 'absolute'],
@@ -220,7 +235,7 @@ CHARTS = {
235 ['status_yellow', 'yellow', 'absolute']
236 ]},
237 'cluster_health_shards': {
223 - 'options': [None, 'Shards statistics', 'shards', 'cluster health API',
238 + 'options': [None, 'Shards Statistics', 'shards', 'cluster health API',
239 'es.cluster_health_shards', 'stacked'],
240 'lines': [
241 ['health_active_shards', 'active_shards', 'absolute'],
@@ -231,7 +246,7 @@ CHARTS = {
246 ['health_active_shards_percent_as_number', 'active_percent', 'absolute']
247 ]},
248 'cluster_stats_nodes': {
234 - 'options': [None, 'Nodes statistics', 'nodes', 'cluster stats API',
249 + 'options': [None, 'Nodes Statistics', 'nodes', 'cluster stats API',
250 'es.cluster_nodes', 'stacked'],
251 'lines': [
252 ['count_data_only', 'data_only', 'absolute'],
@@ -241,46 +256,46 @@ CHARTS = {
256 ['count_client', 'client', 'absolute']
257 ]},
258 'cluster_stats_query_cache': {
244 - 'options': [None, 'Query cache statistics', 'queries', 'cluster stats API',
259 + 'options': [None, 'Query Cache Statistics', 'queries', 'cluster stats API',
260 'es.cluster_query_cache', 'stacked'],
261 'lines': [
262 ['query_cache_hit_count', 'hit', 'incremental'],
263 ['query_cache_miss_count', 'miss', 'incremental']
264 ]},
265 'cluster_stats_docs': {
251 - 'options': [None, 'Docs statistics', 'count', 'cluster stats API',
266 + 'options': [None, 'Docs Statistics', 'count', 'cluster stats API',
267 'es.cluster_docs', 'line'],
268 'lines': [
269 ['docs_count', 'docs', 'absolute']
270 ]},
271 'cluster_stats_store': {
257 - 'options': [None, 'Store statistics', 'MB', 'cluster stats API',
272 + 'options': [None, 'Store Statistics', 'MB', 'cluster stats API',
273 'es.cluster_store', 'line'],
274 'lines': [
275 ['store_size_in_bytes', 'size', 'absolute', 1, 1048567]
276 ]},
277 'cluster_stats_indices_shards': {
263 - 'options': [None, 'Indices and shards statistics', 'count', 'cluster stats API',
278 + 'options': [None, 'Indices And Shards Statistics', 'count', 'cluster stats API',
279 'es.cluster_indices_shards', 'stacked'],
280 'lines': [
281 ['indices_count', 'indices', 'absolute'],
282 ['shards_total', 'shards', 'absolute']
283 ]},
284 'host_metrics_transport': {
270 - 'options': [None, 'Cluster communication transport metrics', 'kbit/s', 'host metrics',
285 + 'options': [None, 'Cluster Communication Transport Metrics', 'kilobit/s', 'host metrics',
286 'es.host_transport', 'area'],
287 'lines': [
288 ['transport_rx_size_in_bytes', 'in', 'incremental', 8, 1000],
289 ['transport_tx_size_in_bytes', 'out', 'incremental', -8, 1000]
290 ]},
291 'host_metrics_file_descriptors': {
277 - 'options': [None, 'Available file descriptors in percent', 'percent', 'host metrics',
292 + 'options': [None, 'Available File Descriptors In Percent', 'percent', 'host metrics',
293 'es.host_descriptors', 'area'],
294 'lines': [
295 ['file_descriptors_used', 'used', 'absolute', 1, 10]
296 ]},
297 'host_metrics_http': {
283 - 'options': [None, 'Opened HTTP connections', 'connections', 'host metrics',
298 + 'options': [None, 'Opened HTTP Connections', 'connections', 'host metrics',
299 'es.host_http_connections', 'line'],
300 'lines': [
301 ['http_current_open', 'opened', 'absolute', 1, 1]
@@ -300,12 +315,13 @@ class Service(UrlService):
315 self.methods = list()
316
317 def check(self):
303 - # We can't start if <host> AND <port> not specified
304 - if not all([self.host, self.port, isinstance(self.host, str), isinstance(self.port, (str, int))]):
318 + if not all([self.host,
319 + self.port,
320 + isinstance(self.host, str),
321 + isinstance(self.port, (str, int))]):
322 self.error('Host is not defined in the module configuration file')
323 return False
324
308 - # It as a bad idea to use hostname.
325 # Hostname -> ip address
326 try:
327 self.host = gethostbyname(self.host)
@@ -322,36 +338,24 @@ class Service(UrlService):
338 url_cluster_health = '%s://%s:%s/_cluster/health' % (scheme, self.host, self.port)
339 url_cluster_stats = '%s://%s:%s/_cluster/stats' % (scheme, self.host, self.port)
340
325 - # Create list of enabled API calls
341 user_choice = [bool(self.configuration.get('node_stats', True)),
342 bool(self.configuration.get('cluster_health', True)),
343 bool(self.configuration.get('cluster_stats', True))]
344
330 - avail_methods = [METHODS(get_data_function=self._get_node_stats_, url=url_node_stats),
331 - METHODS(get_data_function=self._get_cluster_health_, url=url_cluster_health),
332 - METHODS(get_data_function=self._get_cluster_stats_, url=url_cluster_stats)]
345 + avail_methods = [METHODS(get_data_function=self._get_node_stats_,
346 + url=url_node_stats),
347 + METHODS(get_data_function=self._get_cluster_health_,
348 + url=url_cluster_health),
349 + METHODS(get_data_function=self._get_cluster_stats_,
350 + url=url_cluster_stats)]
351
352 # Remove disabled API calls from 'avail methods'
353 self.methods = [avail_methods[e[0]] for e in enumerate(avail_methods) if user_choice[e[0]]]
336 -
337 - # Run _get_data for ALL active API calls.
338 - api_check_result = dict()
339 - data_from_check = dict()
340 - for method in self.methods:
341 - try:
342 - api_check_result[method.url] = method.get_data_function(None, method.url)
343 - data_from_check.update(api_check_result[method.url] or dict())
344 - except KeyError as error:
345 - self.error('Failed to parse %s. Error: %s' % (method.url, str(error)))
346 - return False
347 -
348 - # We can start ONLY if all active API calls returned NOT None
349 - if not all(api_check_result.values()):
350 - self.error('Plugin could not get data from all APIs')
354 + data = self._get_data()
355 + if not data:
356 return False
352 - else:
353 - self._data_from_check = data_from_check
354 - return True
357 + self._data_from_check = data
358 + return True
359
360 def _get_data(self):
361 threads = list()
@@ -359,7 +363,8 @@ class Service(UrlService):
363 result = dict()
364
365 for method in self.methods:
362 - th = Thread(target=method.get_data_function, args=(queue, method.url))
366 + th = Thread(target=method.get_data_function,
367 + args=(queue, method.url))
368 th.start()
369 threads.append(th)
370
@@ -378,18 +383,18 @@ class Service(UrlService):
383 raw_data = self._get_raw_data(url)
384
385 if not raw_data:
381 - return queue.put(dict()) if queue else None
382 - else:
383 - data = loads(raw_data)
386 + return queue.put(dict())
387
385 - to_netdata = fetch_data_(raw_data=data, metrics_list=HEALTH_STATS)
388 + data = loads(raw_data)
389 + to_netdata = fetch_data_(raw_data=data,
390 + metrics_list=HEALTH_STATS)
391
387 - to_netdata.update({'status_green': 0, 'status_red': 0, 'status_yellow': 0,
388 - 'status_foo1': 0, 'status_foo2': 0, 'status_foo3': 0})
389 - current_status = 'status_' + data['status']
390 - to_netdata[current_status] = 1
392 + to_netdata.update({'status_green': 0, 'status_red': 0, 'status_yellow': 0,
393 + 'status_foo1': 0, 'status_foo2': 0, 'status_foo3': 0})
394 + current_status = 'status_' + data['status']
395 + to_netdata[current_status] = 1
396
392 - return queue.put(to_netdata) if queue else to_netdata
397 + return queue.put(to_netdata)
398
399 def _get_cluster_stats_(self, queue, url):
400 """
@@ -400,13 +405,13 @@ class Service(UrlService):
405 raw_data = self._get_raw_data(url)
406
407 if not raw_data:
403 - return queue.put(dict()) if queue else None
404 - else:
405 - data = loads(raw_data)
408 + return queue.put(dict())
409
407 - to_netdata = fetch_data_(raw_data=data, metrics_list=CLUSTER_STATS)
410 + data = loads(raw_data)
411 + to_netdata = fetch_data_(raw_data=data,
412 + metrics_list=CLUSTER_STATS)
413
409 - return queue.put(to_netdata) if queue else to_netdata
414 + return queue.put(to_netdata)
415
416 def _get_node_stats_(self, queue, url):
417 """
@@ -417,47 +422,46 @@ class Service(UrlService):
422 raw_data = self._get_raw_data(url)
423
424 if not raw_data:
420 - return queue.put(dict()) if queue else None
421 - else:
422 - data = loads(raw_data)
423 -
424 - node = list(data['nodes'].keys())[0]
425 - to_netdata = fetch_data_(raw_data=data['nodes'][node], metrics_list=NODE_STATS)
425 + return queue.put(dict())
426
427 - # Search performance latency
428 - to_netdata['query_latency'] = self.find_avg_(to_netdata['query_total'],
429 - to_netdata['query_time_in_millis'], 'query_latency')
430 - to_netdata['fetch_latency'] = self.find_avg_(to_netdata['fetch_total'],
431 - to_netdata['fetch_time_in_millis'], 'fetch_latency')
427 + data = loads(raw_data)
428
433 - # Indexing performance latency
434 - to_netdata['indexing_latency'] = self.find_avg_(to_netdata['indexing_index_total'],
435 - to_netdata['indexing_index_time_in_millis'], 'index_latency')
436 - to_netdata['flushing_latency'] = self.find_avg_(to_netdata['flush_total'],
437 - to_netdata['flush_total_time_in_millis'], 'flush_latency')
429 + node = list(data['nodes'].keys())[0]
430 + to_netdata = fetch_data_(raw_data=data['nodes'][node],
431 + metrics_list=NODE_STATS)
432
433 + # Search, index, flush, fetch performance latency
434 + for key in LATENCY:
435 + try:
436 + to_netdata[key] = self.find_avg_(total=to_netdata[LATENCY[key]['total']],
437 + spent_time=to_netdata[LATENCY[key]['spent_time']],
438 + key=key)
439 + except KeyError:
440 + continue
441 + if 'open_file_descriptors' in to_netdata:
442 to_netdata['file_descriptors_used'] = round(float(to_netdata['open_file_descriptors'])
443 / to_netdata['max_file_descriptors'] * 1000)
444
442 - return queue.put(to_netdata) if queue else to_netdata
445 + return queue.put(to_netdata)
446
444 - def find_avg_(self, value1, value2, key):
447 + def find_avg_(self, total, spent_time, key):
448 if key not in self.latency:
446 - self.latency.update({key: [value1, value2]})
449 + self.latency[key] = dict(total=total,
450 + spent_time=spent_time)
451 return 0
448 - else:
449 - if not self.latency[key][0] == value1:
450 - latency = round(float(value2 - self.latency[key][1]) / float(value1 - self.latency[key][0]) * 1000)
451 - self.latency.update({key: [value1, value2]})
452 - return latency
453 - else:
454 - self.latency.update({key: [value1, value2]})
455 - return 0
452 + if self.latency[key]['total'] != total:
453 + latency = float(spent_time - self.latency[key]['spent_time'])\
454 + / float(total - self.latency[key]['total']) * 1000
455 + self.latency[key]['total'] = total
456 + self.latency[key]['spent_time'] = spent_time
457 + return latency
458 + self.latency[key]['spent_time'] = spent_time
459 + return 0
460
461
462 def fetch_data_(raw_data, metrics_list):
463 to_netdata = dict()
460 - for metric, new_name, function in metrics_list:
464 + for metric, new_name, func in metrics_list:
465 value = raw_data
466 for key in metric.split('.'):
467 try:
@@ -465,7 +469,7 @@ def fetch_data_(raw_data, metrics_list):
469 except KeyError:
470 break
471 if not isinstance(value, dict) and key:
468 - to_netdata[new_name or key] = value if not function else function(value)
472 + to_netdata[new_name or key] = value if not func else func(value)
473
474 return to_netdata
475