elasticsearch plugin: file descriptors; http connections; transport metrics charts added
Ilya committed
Jan 21, 2017 at 15:58 UTC
ef8013b508ded43ab7b9920e5073722d286271c9
1 file changed
+71
-45
python.d/elasticsearch.chart.py
+71
-45
@@ -19,8 +19,9 @@ retries = 60
19
20
# charts order (can be overridden if you want less charts, or different order)
21
ORDER = ['search_perf_total', 'search_perf_time', 'search_latency', 'index_perf_total', 'index_perf_time',
22
- 'index_latency', 'jvm_mem_heap', 'jvm_gc_count', 'jvm_gc_time', 'thread_pool_qr', 'fdata_cache',
23
- 'fdata_ev_tr', 'cluster_health_status', 'cluster_health_nodes', 'cluster_health_shards', 'cluster_stats_nodes',
22
+ 'index_latency', 'jvm_mem_heap', 'jvm_gc_count', 'jvm_gc_time', 'host_metrics_file_descriptors',
23
+ 'host_metrics_http', 'host_metrics_transport', 'thread_pool_qr', 'fdata_cache', 'fdata_ev_tr',
24
+ 'cluster_health_status', 'cluster_health_nodes', 'cluster_health_shards', 'cluster_stats_nodes',
25
'cluster_stats_query_cache', 'cluster_stats_docs', 'cluster_stats_store', 'cluster_stats_indices_shards']
26
27
CHARTS = {
@@ -28,30 +29,30 @@ CHARTS = {
29
'options': [None, 'Number of queries, fetches', 'queries', 'Search performance', 'es.search_query', 'stacked'],
30
'lines': [
31
['query_total', 'search_total', 'incremental'],
31
- ["fetch_total", 'fetch_total', 'incremental'],
32
- ["query_current", 'search_current', 'absolute'],
33
- ["fetch_current", 'fetch_current', 'absolute']
32
+ ['fetch_total', 'fetch_total', 'incremental'],
33
+ ['query_current', 'search_current', 'absolute'],
34
+ ['fetch_current', 'fetch_current', 'absolute']
35
]},
36
'search_perf_time': {
37
'options': [None, 'Time spent on queries, fetches', 'seconds', 'Search performance', 'es.search_time', 'stacked'],
38
'lines': [
38
- ["query_time_in_millis", 'query', 'incremental', 1, 1000],
39
- ["fetch_time_in_millis", 'fetch', 'incremental', 1, 1000]
39
+ ['query_time_in_millis', 'query', 'incremental', 1, 1000],
40
+ ['fetch_time_in_millis', 'fetch', 'incremental', 1, 1000]
41
]},
42
'search_latency': {
43
'options': [None, 'Query and fetch latency', 'ms', 'Search performance', 'es.search_latency', 'stacked'],
44
'lines': [
44
- ["query_latency", 'query', 'absolute', 1, 1000],
45
- ["fetch_latency", 'fetch', 'absolute', 1, 1000]
45
+ ['query_latency', 'query', 'absolute', 1, 1000],
46
+ ['fetch_latency', 'fetch', 'absolute', 1, 1000]
47
]},
48
'index_perf_total': {
49
'options': [None, 'Number of documents indexed, index refreshes, flushes', 'documents/indexes',
50
'Indexing performance', 'es.index_doc', 'stacked'],
51
'lines': [
52
['indexing_index_total', 'indexed', 'incremental'],
52
- ["refresh_total", 'refreshes', 'incremental'],
53
- ["flush_total", 'flushes', 'incremental'],
54
- ["indexing_index_current", 'indexed_current', 'absolute'],
53
+ ['refresh_total', 'refreshes', 'incremental'],
54
+ ['flush_total', 'flushes', 'incremental'],
55
+ ['indexing_index_current', 'indexed_current', 'absolute'],
56
]},
57
'index_perf_time': {
58
'options': [None, 'Time spent on indexing, refreshing, flushing', 'seconds', 'Indexing performance',
@@ -72,76 +73,76 @@ CHARTS = {
73
'options': [None, 'JVM heap currently in use/committed', 'percent/MB', 'Memory usage and gc',
74
'es.jvm_heap', 'area'],
75
'lines': [
75
- ["jvm_heap_percent", 'inuse', 'absolute'],
76
- ["jvm_heap_commit", 'commit', 'absolute', -1, 1048576]
76
+ ['jvm_heap_percent', 'inuse', 'absolute'],
77
+ ['jvm_heap_commit', 'commit', 'absolute', -1, 1048576]
78
]},
79
'jvm_gc_count': {
80
'options': [None, 'Count of garbage collections', 'counts', 'Memory usage and gc', 'es.gc_count', 'stacked'],
81
'lines': [
81
- ["young_collection_count", 'young', 'incremental'],
82
- ["old_collection_count", 'old', 'incremental']
82
+ ['young_collection_count', 'young', 'incremental'],
83
+ ['old_collection_count', 'old', 'incremental']
84
]},
85
'jvm_gc_time': {
86
'options': [None, 'Time spent on garbage collections', 'ms', 'Memory usage and gc', 'es.gc_time', 'stacked'],
87
'lines': [
87
- ["young_collection_time_in_millis", 'young', 'incremental'],
88
- ["old_collection_time_in_millis", 'old', 'incremental']
88
+ ['young_collection_time_in_millis', 'young', 'incremental'],
89
+ ['old_collection_time_in_millis', 'old', 'incremental']
90
]},
91
'thread_pool_qr': {
92
'options': [None, 'Number of queued/rejected threads in thread pool', 'threads', 'Queues and rejections',
93
'es.qr', 'stacked'],
94
'lines': [
94
- ["bulk_queue", 'bulk_queue', 'absolute'],
95
- ["index_queue", 'index_queue', 'absolute'],
96
- ["search_queue", 'search_queue', 'absolute'],
97
- ["merge_queue", 'merge_queue', 'absolute'],
98
- ["bulk_rejected", 'bulk_rej', 'absolute'],
99
- ["index_rejected", 'index_rej', 'absolute'],
100
- ["search_rejected", 'search_rej', 'absolute'],
101
- ["merge_rejected", 'merge_rej', 'absolute']
95
+ ['bulk_queue', 'bulk_queue', 'absolute'],
96
+ ['index_queue', 'index_queue', 'absolute'],
97
+ ['search_queue', 'search_queue', 'absolute'],
98
+ ['merge_queue', 'merge_queue', 'absolute'],
99
+ ['bulk_rejected', 'bulk_rej', 'absolute'],
100
+ ['index_rejected', 'index_rej', 'absolute'],
101
+ ['search_rejected', 'search_rej', 'absolute'],
102
+ ['merge_rejected', 'merge_rej', 'absolute']
103
]},
104
'fdata_cache': {
105
'options': [None, 'Fielddata cache size', 'MB', 'Fielddata cache', 'es.fdata_cache', 'line'],
106
'lines': [
106
- ["index_fdata_mem", 'mem_size', 'absolute', 1, 1048576]
107
+ ['index_fdata_mem', 'mem_size', 'absolute', 1, 1048576]
108
]},
109
'fdata_ev_tr': {
110
'options': [None, 'Fielddata evictions and circuit breaker tripped count', 'number of events',
111
'Fielddata cache', 'es.fdata_ev_tr', 'line'],
112
'lines': [
112
- ["index_fdata_evic", 'evictions', 'incremental'],
113
- ["breakers_fdata_trip", 'tripped', 'incremental']
113
+ ['index_fdata_evic', 'evictions', 'incremental'],
114
+ ['breakers_fdata_trip', 'tripped', 'incremental']
115
]},
116
'cluster_health_nodes': {
117
'options': [None, 'Nodes and tasks statistics', 'units', 'Cluster health API',
118
'es.cluster_health', 'stacked'],
119
'lines': [
119
- ["health_number_of_nodes", 'nodes', 'absolute'],
120
- ["health_number_of_data_nodes", 'data_nodes', 'absolute'],
121
- ["health_number_of_pending_tasks", 'pending_tasks', 'absolute'],
122
- ["health_number_of_in_flight_fetch", 'inflight_fetch', 'absolute']
120
+ ['health_number_of_nodes', 'nodes', 'absolute'],
121
+ ['health_number_of_data_nodes', 'data_nodes', 'absolute'],
122
+ ['health_number_of_pending_tasks', 'pending_tasks', 'absolute'],
123
+ ['health_number_of_in_flight_fetch', 'inflight_fetch', 'absolute']
124
]},
125
'cluster_health_status': {
126
'options': [None, 'Cluster status', 'status', 'Cluster health API',
127
'es.cluster_health_status', 'area'],
128
'lines': [
128
- ["status_green", 'green', 'absolute'],
129
- ["status_red", 'red', 'absolute'],
130
- ["status_foo1", None, 'absolute'],
131
- ["status_foo2", None, 'absolute'],
132
- ["status_foo3", None, 'absolute'],
133
- ["status_yellow", 'yellow', 'absolute']
129
+ ['status_green', 'green', 'absolute'],
130
+ ['status_red', 'red', 'absolute'],
131
+ ['status_foo1', None, 'absolute'],
132
+ ['status_foo2', None, 'absolute'],
133
+ ['status_foo3', None, 'absolute'],
134
+ ['status_yellow', 'yellow', 'absolute']
135
]},
136
'cluster_health_shards': {
137
'options': [None, 'Shards statistics', 'shards', 'Cluster health API',
138
'es.cluster_health_sharts', 'stacked'],
139
'lines': [
139
- ["health_active_shards", 'active_shards', 'absolute'],
140
- ["health_relocating_shards", 'relocating_shards', 'absolute'],
141
- ["health_unassigned_shards", 'unassigned', 'absolute'],
142
- ["health_delayed_unassigned_shards", 'delayed_unassigned', 'absolute'],
143
- ["health_initializing_shards", 'initializing', 'absolute'],
144
- ["health_active_shards_percent_as_number", 'active_percent', 'absolute']
140
+ ['health_active_shards', 'active_shards', 'absolute'],
141
+ ['health_relocating_shards', 'relocating_shards', 'absolute'],
142
+ ['health_unassigned_shards', 'unassigned', 'absolute'],
143
+ ['health_delayed_unassigned_shards', 'delayed_unassigned', 'absolute'],
144
+ ['health_initializing_shards', 'initializing', 'absolute'],
145
+ ['health_active_shards_percent_as_number', 'active_percent', 'absolute']
146
]},
147
'cluster_stats_nodes': {
148
'options': [None, 'Nodes statistics', 'nodes', 'Cluster stats API',
@@ -178,6 +179,25 @@ CHARTS = {
179
'lines': [
180
['indices_count', 'indices', 'absolute'],
181
['shards_total', 'shards', 'absolute']
182
+ ]},
183
+ 'host_metrics_transport': {
184
+ 'options': [None, 'Cluster communication transport metrics', 'kbit/s', 'Host metrics',
185
+ 'es.host_metrics_transport', 'area'],
186
+ 'lines': [
187
+ ['transport_rx_size_in_bytes', 'in', 'incremental', 8, 1000],
188
+ ['transport_tx_size_in_bytes', 'out', 'incremental', -8, 1000]
189
+ ]},
190
+ 'host_metrics_file_descriptors': {
191
+ 'options': [None, 'Available file descriptors in percent', 'percent', 'Host metrics',
192
+ 'es.host_metrics_descriptors', 'area'],
193
+ 'lines': [
194
+ ['file_descriptors_used', 'used', 'absolute', 1, 10]
195
+ ]},
196
+ 'host_metrics_http': {
197
+ 'options': [None, 'Opened HTTP connections', 'connections', 'Host metrics',
198
+ 'es.host_metrics_http', 'line'],
199
+ 'lines': [
200
+ ['http_current_open', 'opened', 'absolute', 1, 1]
201
]}
202
}
203
@@ -356,6 +376,12 @@ class Service(UrlService):
376
to_netdata['index_fdata_evic'] = data['nodes'][node]['indices']['fielddata']['evictions']
377
to_netdata['breakers_fdata_trip'] = data['nodes'][node]['breakers']['fielddata']['tripped']
378
379
+ # Host metrics
380
+ to_netdata.update(update_key('http', data['nodes'][node]['http']))
381
+ to_netdata.update(update_key('transport', data['nodes'][node]['transport']))
382
+ to_netdata['file_descriptors_used'] = round(float(data['nodes'][node]['process']['open_file_descriptors'])
383
+ / data['nodes'][node]['process']['max_file_descriptors'] * 1000)
384
+
385
queue.put(to_netdata)
386
387
def find_avg(self, value1, value2, key):