Add a riak plugin (#6286)
* Add a Riak plugin.
Johannes Christ committed
Jun 18, 2019 at 20:22 UTC
c5ab82558d5669ffec8991e338ebf548a2abcf97
9 files changed
+595
collectors/python.d.plugin/Makefile.am
+1
@@ -87,6 +87,7 @@ include rabbitmq/Makefile.inc
87
include redis/Makefile.inc
88
include rethinkdbs/Makefile.inc
89
include retroshare/Makefile.inc
90
+include riakkv/Makefile.inc
91
include samba/Makefile.inc
92
include sensors/Makefile.inc
93
include smartd_log/Makefile.inc
collectors/python.d.plugin/python.d.conf
+1
@@ -88,6 +88,7 @@ nginx_log: no
88
# redis: yes
89
# rethinkdbs: yes
90
# retroshare: yes
91
+# riakkv: yes
92
# samba: yes
93
# sensors: yes
94
# smartd_log: yes
collectors/python.d.plugin/riakkv/Makefile.inc
new
+13
@@ -0,0 +1,13 @@
1
+# SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+# THIS IS NOT A COMPLETE Makefile
4
+# IT IS INCLUDED BY ITS PARENT'S Makefile.am
5
+# IT IS REQUIRED TO REFERENCE ALL FILES RELATIVE TO THE PARENT
6
+
7
+# install these files
8
+dist_python_DATA += riakkv/riakkv.chart.py
9
+dist_pythonconfig_DATA += riakkv/riakkv.conf
10
+
11
+# do not install these files, but include them in the distribution
12
+dist_noinst_DATA += riakkv/README.md riakkv/Makefile.inc
13
+
collectors/python.d.plugin/riakkv/README.md
new
+110
@@ -0,0 +1,110 @@
1
+# riakkv
2
+
3
+Monitors one or more Riak KV servers.
4
+
5
+**Requirements:**
6
+
7
+* An accessible `/stats` endpoint. See [the Riak KV configuration reference]
8
+ documentation](https://docs.riak.com/riak/kv/2.2.3/configuring/reference/#client-interfaces)
9
+ for how to enable this.
10
+
11
+The following charts are included, which are mostly derived from the metrics
12
+listed
13
+[here](https://docs.riak.com/riak/kv/latest/using/reference/statistics-monitoring/index.html#riak-metrics-to-graph).
14
+
15
+1. **Throughput** in operations/s
16
+ * **KV operations**
17
+ * gets
18
+ * puts
19
+
20
+ * **Data type updates**
21
+ * counters
22
+ * sets
23
+ * maps
24
+
25
+ * **Search queries**
26
+ * queries
27
+
28
+ * **Search documents**
29
+ * indexed
30
+
31
+ * **Strong consistency operations**
32
+ * gets
33
+ * puts
34
+
35
+2. **Latency** in milliseconds
36
+ * **KV latency** of the past minute
37
+ * get (mean, median, 95th / 99th / 100th percentile)
38
+ * put (mean, median, 95th / 99th / 100th percentile)
39
+
40
+ * **Data type latency** of the past minute
41
+ * counter_merge (mean, median, 95th / 99th / 100th percentile)
42
+ * set_merge (mean, median, 95th / 99th / 100th percentile)
43
+ * map_merge (mean, median, 95th / 99th / 100th percentile)
44
+
45
+ * **Search latency** of the past minute
46
+ * query (median, min, max, 95th / 99th percentile)
47
+ * index (median, min, max, 95th / 99th percentile)
48
+
49
+ * **Strong consistency latency** of the past minute
50
+ * get (mean, median, 95th / 99th / 100th percentile)
51
+ * put (mean, median, 95th / 99th / 100th percentile)
52
+
53
+3. **Erlang VM metrics**
54
+ * **System counters**
55
+ * processes
56
+
57
+ * **Memory allocation** in MB
58
+ * processes.allocated
59
+ * processes.used
60
+
61
+4. **General load / health metrics**
62
+ * **Siblings encountered in KV operations** during the past minute
63
+ * get (mean, median, 95th / 99th / 100th percentile)
64
+
65
+ * **Object size in KV operations** during the past minute in KB
66
+ * get (mean, median, 95th / 99th / 100th percentile)
67
+
68
+ * **Message queue length** in unprocessed messages
69
+ * vnodeq_size (mean, median, 95th / 99th / 100th percentile)
70
+
71
+ * **Index operations** encountered by Search
72
+ * errors
73
+
74
+ * **Protocol buffer connections**
75
+ * active
76
+
77
+ * **Repair operations coordinated by this node**
78
+ * read
79
+
80
+ * **Active finite state machines by kind**
81
+ * get
82
+ * put
83
+ * secondary_index
84
+ * list_keys
85
+
86
+ * **Rejected finite state machines**
87
+ * get
88
+ * put
89
+
90
+ * **Number of writes to Search failed due to bad data format by reason**
91
+ * bad_entry
92
+ * extract_fail
93
+
94
+
95
+### configuration
96
+
97
+The module needs to be passed the full URL to Riak's stats endpoint.
98
+For example:
99
+
100
+```yaml
101
+myriak:
102
+ url: http://myriak.example.com:8098/stats
103
+```
104
+
105
+With no explicit configuration given, the module will attempt to connect to
106
+`http://localhost:8098/stats`.
107
+
108
+The default update frequency for the plugin is set to 2 seconds as Riak
109
+internally updates the metrics every second. If we were to update the metrics
110
+every second, the resulting graph would contain odd jitter.
collectors/python.d.plugin/riakkv/riakkv.chart.py
new
+315
@@ -0,0 +1,315 @@
1
+# -*- coding: utf-8 -*-
2
+# Description: riak netdata python.d module
3
+#
4
+# See also:
5
+# https://docs.riak.com/riak/kv/latest/using/reference/statistics-monitoring/index.html
6
+
7
+from json import loads
8
+
9
+from bases.FrameworkServices.UrlService import UrlService
10
+
11
+# Riak updates the metrics at the /stats endpoint every 1 second.
12
+# If we use `update_every = 1` here, that means we might get weird jitter in the graph,
13
+# so the default is set to 2 seconds to prevent it.
14
+update_every = 2
15
+
16
+# charts order (can be overridden if you want less charts, or different order)
17
+ORDER = [
18
+ # Throughput metrics
19
+ # https://docs.riak.com/riak/kv/latest/using/reference/statistics-monitoring/index.html#throughput-metrics
20
+ # Collected in totals.
21
+ "kv.node_operations", # K/V node operations.
22
+ "dt.vnode_updates", # Data type vnode updates.
23
+ "search.queries", # Search queries on the node.
24
+ "search.documents", # Documents indexed by Search.
25
+ "consistent.operations", # Consistent node operations.
26
+
27
+ # Latency metrics
28
+ # https://docs.riak.com/riak/kv/latest/using/reference/statistics-monitoring/index.html#throughput-metrics
29
+ # Collected for the past minute in milliseconds,
30
+ # returned from riak in microseconds.
31
+ "kv.latency.get", # K/V GET FSM traversal latency.
32
+ "kv.latency.put", # K/V PUT FSM traversal latency.
33
+ "dt.latency.counter", # Update Counter Data type latency.
34
+ "dt.latency.set", # Update Set Data type latency.
35
+ "dt.latency.map", # Update Map Data type latency.
36
+ "search.latency.query", # Search query latency.
37
+ "search.latency.index", # Time it takes for search to index a new document.
38
+ "consistent.latency.get", # Strong consistent read latency.
39
+ "consistent.latency.put", # Strong consistent write latency.
40
+
41
+ # Erlang resource usage metrics
42
+ # https://docs.riak.com/riak/kv/latest/using/reference/statistics-monitoring/index.html#erlang-resource-usage-metrics
43
+ # Processes collected as a gauge,
44
+ # memory collected as Megabytes, returned as bytes from Riak.
45
+ "vm.processes", # Number of processes currently running in the Erlang VM.
46
+ "vm.memory.processes", # Total amount of memory allocated & used for Erlang processes.
47
+
48
+ # General Riak Load / Health metrics
49
+ # https://docs.riak.com/riak/kv/latest/using/reference/statistics-monitoring/index.html#general-riak-load-health-metrics
50
+ # The following are collected by Riak over the past minute:
51
+ "kv.siblings_encountered.get", # Siblings encountered during GET operations by this node.
52
+ "kv.objsize.get", # Object size encountered by this node.
53
+ "search.vnodeq_size", # Number of unprocessed messages in the vnode message queues (Search).
54
+ # The following are calculated in total, or as gauges:
55
+ "search.index_errors", # Errors of the search subsystem while indexing documents.
56
+ "core.pbc", # Number of currently active protocol buffer connections.
57
+ "core.repairs", # Total read repair operations coordinated by this node.
58
+ "core.fsm_active", # Active finite state machines by kind.
59
+ "core.fsm_rejected", # Rejected finite state machines by kind.
60
+
61
+ # General Riak Search Load / Health metrics
62
+ # https://docs.riak.com/riak/kv/latest/using/reference/statistics-monitoring/index.html#general-riak-search-load-health-metrics
63
+ # Reported as counters.
64
+ "search.errors", # Write and read errors of the Search subsystem.
65
+]
66
+
67
+CHARTS = {
68
+ # Throughput metrics
69
+ "kv.node_operations": {
70
+ "options": [None, "Reads & writes coordinated by this node", "operations/s", "throughput", "riak.kv.throughput", "line"],
71
+ "lines": [
72
+ ["node_gets_total", "gets", "incremental"],
73
+ ["node_puts_total", "puts", "incremental"]
74
+ ]
75
+ },
76
+ "dt.vnode_updates": {
77
+ "options": [None, "Update operations coordinated by local vnodes by data type", "operations/s", "throughput", "riak.dt.vnode_updates", "line"],
78
+ "lines": [
79
+ ["vnode_counter_update_total", "counters", "incremental"],
80
+ ["vnode_set_update_total", "sets", "incremental"],
81
+ ["vnode_map_update_total", "maps", "incremental"],
82
+ ]
83
+ },
84
+ "search.queries": {
85
+ "options": [None, "Search queries on the node", "queries/s", "throughput", "riak.search", "line"],
86
+ "lines": [
87
+ ["search_query_throughput_count", "queries", "incremental"]
88
+ ]
89
+ },
90
+ "search.documents": {
91
+ "options": [None, "Documents indexed by search", "documents/s", "throughput", "riak.search.documents", "line"],
92
+ "lines": [
93
+ ["search_index_throughput_count", "indexed", "incremental"]
94
+ ]
95
+ },
96
+ "consistent.operations": {
97
+ "options": [None, "Consistent node operations", "operations/s", "throughput", "riak.consistent.operations", "line"],
98
+ "lines": [
99
+ ["consistent_gets_total", "gets", "incremental"],
100
+ ["consistent_puts_total", "puts", "incremental"],
101
+ ]
102
+ },
103
+
104
+ # Latency metrics
105
+ "kv.latency.get": {
106
+ "options": [None, "Time between reception of a client GET request and subsequent response to client", "ms", "latency", "riak.kv.latency.get", "line"],
107
+ "lines": [
108
+ ["node_get_fsm_time_mean", "mean", "absolute", 1, 1000],
109
+ ["node_get_fsm_time_median", "median", "absolute", 1, 1000],
110
+ ["node_get_fsm_time_95", "95", "absolute", 1, 1000],
111
+ ["node_get_fsm_time_99", "99", "absolute", 1, 1000],
112
+ ["node_get_fsm_time_100", "100", "absolute", 1, 1000],
113
+ ]
114
+ },
115
+ "kv.latency.put": {
116
+ "options": [None, "Time between reception of a client PUT request and subsequent response to client", "ms", "latency", "riak.kv.latency.put", "line"],
117
+ "lines": [
118
+ ["node_put_fsm_time_mean", "mean", "absolute", 1, 1000],
119
+ ["node_put_fsm_time_median", "median", "absolute", 1, 1000],
120
+ ["node_put_fsm_time_95", "95", "absolute", 1, 1000],
121
+ ["node_put_fsm_time_99", "99", "absolute", 1, 1000],
122
+ ["node_put_fsm_time_100", "100", "absolute", 1, 1000],
123
+ ]
124
+ },
125
+ "dt.latency.counter": {
126
+ "options": [None, "Time it takes to perform an Update Counter operation", "ms", "latency", "riak.dt.latency.counter_merge", "line"],
127
+ "lines": [
128
+ ["object_counter_merge_time_mean", "mean", "absolute", 1, 1000],
129
+ ["object_counter_merge_time_median", "median", "absolute", 1, 1000],
130
+ ["object_counter_merge_time_95", "95", "absolute", 1, 1000],
131
+ ["object_counter_merge_time_99", "99", "absolute", 1, 1000],
132
+ ["object_counter_merge_time_100", "100", "absolute", 1, 1000],
133
+ ]
134
+ },
135
+ "dt.latency.set": {
136
+ "options": [None, "Time it takes to perform an Update Set operation", "ms", "latency", "riak.dt.latency.set_merge", "line"],
137
+ "lines": [
138
+ ["object_set_merge_time_mean", "mean", "absolute", 1, 1000],
139
+ ["object_set_merge_time_median", "median", "absolute", 1, 1000],
140
+ ["object_set_merge_time_95", "95", "absolute", 1, 1000],
141
+ ["object_set_merge_time_99", "99", "absolute", 1, 1000],
142
+ ["object_set_merge_time_100", "100", "absolute", 1, 1000],
143
+ ]
144
+ },
145
+ "dt.latency.map": {
146
+ "options": [None, "Time it takes to perform an Update Map operation", "ms", "latency", "riak.dt.latency.map_merge", "line"],
147
+ "lines": [
148
+ ["object_map_merge_time_mean", "mean", "absolute", 1, 1000],
149
+ ["object_map_merge_time_median", "median", "absolute", 1, 1000],
150
+ ["object_map_merge_time_95", "95", "absolute", 1, 1000],
151
+ ["object_map_merge_time_99", "99", "absolute", 1, 1000],
152
+ ["object_map_merge_time_100", "100", "absolute", 1, 1000],
153
+ ]
154
+ },
155
+ "search.latency.query": {
156
+ "options": [None, "Search query latency", "ms", "latency", "riak.search.latency.query", "line"],
157
+ "lines": [
158
+ ["search_query_latency_median", "median", "absolute", 1, 1000],
159
+ ["search_query_latency_min", "min", "absolute", 1, 1000],
160
+ ["search_query_latency_95", "95", "absolute", 1, 1000],
161
+ ["search_query_latency_99", "99", "absolute", 1, 1000],
162
+ ["search_query_latency_999", "999", "absolute", 1, 1000],
163
+ ["search_query_latency_max", "max", "absolute", 1, 1000],
164
+ ]
165
+ },
166
+ "search.latency.index": {
167
+ "options": [None, "Time it takes Search to index a new document", "ms", "latency", "riak.search.latency.index", "line"],
168
+ "lines": [
169
+ ["search_index_latency_median", "median", "absolute", 1, 1000],
170
+ ["search_index_latency_min", "min", "absolute", 1, 1000],
171
+ ["search_index_latency_95", "95", "absolute", 1, 1000],
172
+ ["search_index_latency_99", "99", "absolute", 1, 1000],
173
+ ["search_index_latency_999", "999", "absolute", 1, 1000],
174
+ ["search_index_latency_max", "max", "absolute", 1, 1000],
175
+ ]
176
+ },
177
+
178
+ # Riak Strong Consistency metrics
179
+ "consistent.latency.get": {
180
+ "options": [None, "Strongly consistent read latency", "ms", "latency", "riak.consistent.latency.get", "line"],
181
+ "lines": [
182
+ ["consistent_get_time_mean", "mean", "absolute", 1, 1000],
183
+ ["consistent_get_time_median", "median", "absolute", 1, 1000],
184
+ ["consistent_get_time_95", "95", "absolute", 1, 1000],
185
+ ["consistent_get_time_99", "99", "absolute", 1, 1000],
186
+ ["consistent_get_time_100", "100", "absolute", 1, 1000],
187
+ ]
188
+ },
189
+ "consistent.latency.put": {
190
+ "options": [None, "Strongly consistent write latency", "ms", "latency", "riak.consistent.latency.put", "line"],
191
+ "lines": [
192
+ ["consistent_put_time_mean", "mean", "absolute", 1, 1000],
193
+ ["consistent_put_time_median", "median", "absolute", 1, 1000],
194
+ ["consistent_put_time_95", "95", "absolute", 1, 1000],
195
+ ["consistent_put_time_99", "99", "absolute", 1, 1000],
196
+ ["consistent_put_time_100", "100", "absolute", 1, 1000],
197
+ ]
198
+ },
199
+
200
+ # BEAM metrics
201
+ "vm.processes": {
202
+ "options": [None, "Total processes running in the Erlang VM", "total", "vm", "riak.vm", "line"],
203
+ "lines": [
204
+ ["sys_process_count", "processes", "absolute"],
205
+ ]
206
+ },
207
+ "vm.memory.processes": {
208
+ "options": [None, "Memory allocated & used by Erlang processes", "MB", "vm", "riak.vm.memory.processes", "line"],
209
+ "lines": [
210
+ ["memory_processes", "allocated", "absolute", 1, 1024 * 1024],
211
+ ["memory_processes_used", "used", "absolute", 1, 1024 * 1024]
212
+ ]
213
+ },
214
+
215
+ # General Riak Load/Health metrics
216
+ "kv.siblings_encountered.get": {
217
+ "options": [None, "Number of siblings encountered during GET operations by this node during the past minute", "siblings", "load", "riak.kv.siblings_encountered.get", "line"],
218
+ "lines": [
219
+ ["node_get_fsm_siblings_mean", "mean", "absolute"],
220
+ ["node_get_fsm_siblings_median", "median", "absolute"],
221
+ ["node_get_fsm_siblings_95", "95", "absolute"],
222
+ ["node_get_fsm_siblings_99", "99", "absolute"],
223
+ ["node_get_fsm_siblings_100", "100", "absolute"],
224
+ ]
225
+ },
226
+ "kv.objsize.get": {
227
+ "options": [None, "Object size encountered by this node during the past minute", "KB", "load", "riak.kv.objsize.get", "line"],
228
+ "lines": [
229
+ ["node_get_fsm_objsize_mean", "mean", "absolute", 1, 1024],
230
+ ["node_get_fsm_objsize_median", "median", "absolute", 1, 1024],
231
+ ["node_get_fsm_objsize_95", "95", "absolute", 1, 1024],
232
+ ["node_get_fsm_objsize_99", "99", "absolute", 1, 1024],
233
+ ["node_get_fsm_objsize_100", "100", "absolute", 1, 1024],
234
+ ]
235
+ },
236
+ "search.vnodeq_size": {
237
+ "options": [None, "Number of unprocessed messages in the vnode message queues of Search on this node in the past minute", "messages", "load", "riak.search.vnodeq_size", "line"],
238
+ "lines": [
239
+ ["riak_search_vnodeq_mean", "mean", "absolute"],
240
+ ["riak_search_vnodeq_median", "median", "absolute"],
241
+ ["riak_search_vnodeq_95", "95", "absolute"],
242
+ ["riak_search_vnodeq_99", "99", "absolute"],
243
+ ["riak_search_vnodeq_100", "100", "absolute"],
244
+ ]
245
+ },
246
+ "search.index_errors": {
247
+ "options": [None, "Number of document index errors encountered by Search", "errors", "load", "riak.search.index", "line"],
248
+ "lines": [
249
+ ["search_index_fail_count", "errors", "absolute"]
250
+ ]
251
+ },
252
+ "core.pbc": {
253
+ "options": [None, "Protocol buffer connections by status", "connections", "load", "riak.core.protobuf_connections", "line"],
254
+ "lines": [
255
+ ["pbc_active", "active", "absolute"],
256
+ # ["pbc_connects", "established_pastmin", "absolute"]
257
+ ]
258
+ },
259
+ "core.repairs": {
260
+ "options": [None, "Number of repair operations this node has coordinated", "repairs", "load", "riak.core.repairs", "line"],
261
+ "lines": [
262
+ ["read_repairs", "read", "absolute"]
263
+ ]
264
+ },
265
+ "core.fsm_active": {
266
+ "options": [None, "Active finite state machines by kind", "fsms", "load", "riak.core.fsm_active", "line"],
267
+ "lines": [
268
+ ["node_get_fsm_active", "get", "absolute"],
269
+ ["node_put_fsm_active", "put", "absolute"],
270
+ ["index_fsm_active", "secondary index", "absolute"],
271
+ ["list_fsm_active", "list keys", "absolute"]
272
+ ]
273
+ },
274
+ "core.fsm_rejected": {
275
+ # Writing "Sidejob's" here seems to cause some weird issues: it results in this chart being rendered in
276
+ # its own context and additionally, moves the entire Riak graph all the way up to the top of the Netdata
277
+ # dashboard for some reason.
278
+ "options": [None, "Finite state machines being rejected by Sidejobs overload protection", "fsms", "load", "riak.core.fsm_rejected", "line"],
279
+ "lines": [
280
+ ["node_get_fsm_rejected", "get", "absolute"],
281
+ ["node_put_fsm_rejected", "put", "absolute"]
282
+ ]
283
+ },
284
+
285
+ # General Riak Search Load / Health metrics
286
+ "search.errors": {
287
+ "options": [None, "Number of writes to Search failed due to bad data format by reason", "writes", "load", "riak.search.index", "line"],
288
+ "lines": [
289
+ ["search_index_bad_entry_count", "bad_entry", "absolute"],
290
+ ["search_index_extract_fail_count", "extract_fail", "absolute"],
291
+ ]
292
+ }
293
+}
294
+
295
+
296
+class Service(UrlService):
297
+ def __init__(self, configuration=None, name=None):
298
+ UrlService.__init__(self, configuration=configuration, name=name)
299
+ self.order = ORDER
300
+ self.definitions = CHARTS
301
+
302
+ def _get_data(self):
303
+ """
304
+ Format data received from http request
305
+ :return: dict
306
+ """
307
+ raw = self._get_raw_data()
308
+ if not raw:
309
+ return None
310
+
311
+ try:
312
+ return loads(raw)
313
+ except (TypeError, ValueError) as err:
314
+ self.error(err)
315
+ return None
collectors/python.d.plugin/riakkv/riakkv.conf
new
+68
@@ -0,0 +1,68 @@
1
+# netdata python.d.plugin configuration for riak
2
+#
3
+# This file is in YaML format. Generally the format is:
4
+#
5
+# name: value
6
+#
7
+# There are 2 sections:
8
+# - global variables
9
+# - one or more JOBS
10
+#
11
+# JOBS allow you to collect values from multiple sources.
12
+# Each source will have its own set of charts.
13
+#
14
+# JOB parameters have to be indented (using spaces only, example below).
15
+
16
+# ----------------------------------------------------------------------
17
+# Global Variables
18
+# These variables set the defaults for all JOBs, however each JOB
19
+# may define its own, overriding the defaults.
20
+
21
+# update_every sets the default data collection frequency.
22
+# If unset, the python.d.plugin default is used.
23
+# update_every: 1
24
+
25
+# priority controls the order of charts at the netdata dashboard.
26
+# Lower numbers move the charts towards the top of the page.
27
+# If unset, the default for python.d.plugin is used.
28
+# priority: 60000
29
+
30
+# penalty indicates whether to apply penalty to update_every in case of failures.
31
+# Penalty will increase every 5 failed updates in a row. Maximum penalty is 10 minutes.
32
+# penalty: yes
33
+
34
+# autodetection_retry sets the job re-check interval in seconds.
35
+# The job is not deleted if check fails.
36
+# Attempts to start the job are made once every autodetection_retry.
37
+# This feature is disabled by default.
38
+# autodetection_retry: 0
39
+
40
+# ----------------------------------------------------------------------
41
+# JOBS (data collection sources)
42
+#
43
+# The default JOBS share the same *name*. JOBS with the same name
44
+# are mutually exclusive. Only one of them will be allowed running at
45
+# any time. This allows autodetection to try several alternatives and
46
+# pick the one that works.
47
+#
48
+# Any number of jobs is supported.
49
+#
50
+# All python.d.plugin JOBS (for all its modules) support a set of
51
+# predefined parameters. These are:
52
+#
53
+# job_name:
54
+# name: myname # the JOB's name as it will appear at the
55
+# # dashboard (by default is the job_name)
56
+# # JOBs sharing a name are mutually exclusive
57
+# update_every: 1 # the JOB's data collection frequency
58
+# priority: 60000 # the JOB's order on the dashboard
59
+# penalty: yes # the JOB's penalty
60
+# autodetection_retry: 0 # the JOB's re-check interval in seconds
61
+#
62
+#
63
+# ----------------------------------------------------------------------
64
+# AUTO-DETECTION JOBS
65
+# only one of them will run (they have the same name)
66
+
67
+local:
68
+ url : 'http://localhost:8098/stats'
health/Makefile.am
+1
@@ -70,6 +70,7 @@ dist_healthconfig_DATA = \
70
health.d/ram.conf \
71
health.d/redis.conf \
72
health.d/retroshare.conf \
73
+ health.d/riakkv.conf \
74
health.d/softnet.conf \
75
health.d/squid.conf \
76
health.d/stiebeleltron.conf \
health/health.d/riakkv.conf
new
+80
@@ -0,0 +1,80 @@
1
+# Ensure that Riak is running. template: riak_last_collected_secs
2
+template: riak_last_collected_secs
3
+ on: riak.kv.throughput
4
+ calc: $now - $last_collected_t
5
+ units: seconds ago
6
+ every: 10s
7
+ warn: $this > (($status >= $WARNING) ? ($update_every) : ( 5 * $update_every))
8
+ crit: $this > (($status == $CRITICAL) ? ($update_every) : (60 * $update_every))
9
+ delay: down 5m multiplier 1.5 max 1h
10
+ info: number of seconds since the last successful data collection
11
+ to: dba
12
+
13
+# Warn if a list keys operation is running.
14
+template: riak_list_keys_active
15
+ on: riak.core.fsm_active
16
+ calc: $list_fsm_active
17
+ units: state machines
18
+ every: 10s
19
+ warn: $list_fsm_active > 0
20
+ info: number of currently running list keys finite state machines
21
+ to: dba
22
+
23
+
24
+## Timing healthchecks
25
+# KV GET
26
+template: 1h_kv_get_mean_latency
27
+ on: riak.kv.latency.get
28
+ calc: $node_get_fsm_time_mean
29
+ lookup: average -1h unaligned of time
30
+ every: 30s
31
+ units: ms
32
+ info: mean average KV GET latency over the last hour
33
+
34
+template: riak_kv_get_slow
35
+ on: riak.kv.latency.get
36
+ calc: $mean
37
+ lookup: average -3m unaligned of time
38
+ units: ms
39
+ every: 10s
40
+ warn: ($this > ($1h_kv_get_mean_latency * 2) )
41
+ crit: ($this > ($1h_kv_get_mean_latency * 3) )
42
+ info: average KV GET time over the last 3 minutes, compared to the average over the last hour
43
+ delay: down 5m multiplier 1.5 max 1h
44
+ to: dba
45
+
46
+# KV PUT
47
+template: 1h_kv_put_mean_latency
48
+ on: riak.kv.latency.put
49
+ calc: $node_put_fsm_time_mean
50
+ lookup: average -1h unaligned of time
51
+ every: 30s
52
+ units: ms
53
+ info: mean average KV PUT latency over the last hour
54
+
55
+template: riak_kv_put_slow
56
+ on: riak.kv.latency.put
57
+ calc: $mean
58
+ lookup: average -3m unaligned of time
59
+ units: ms
60
+ every: 10s
61
+ warn: ($this > ($1h_kv_put_mean_latency * 2) )
62
+ crit: ($this > ($1h_kv_put_mean_latency * 3) )
63
+ info: average KV PUT time over the last 3 minutes, compared to the average over the last hour
64
+ delay: down 5m multiplier 1.5 max 1h
65
+ to: dba
66
+
67
+
68
+## VM healthchecks
69
+
70
+# Default Erlang VM process limit: 262144
71
+# On systems observed, this is < 2000, but may grow depending on load.
72
+template: riak_vm_high_process_count
73
+ on: riak.vm
74
+ calc: $sys_process_count
75
+ units: processes
76
+ every: 10s
77
+ warn: $this > 10000
78
+ crit: $this > 100000
79
+ info: number of processes running in the Erlang VM (the default limit on ERTS 10.2.4 is 262144)
80
+ to: dba
web/gui/dashboard_info.js
+6
@@ -267,6 +267,12 @@ netdataDashboard.menu = {
267
info: 'Performance metrics for <b>RetroShare</b>. RetroShare is open source software for encrypted filesharing, serverless email, instant messaging, online chat, and BBS, based on a friend-to-friend network built on GNU Privacy Guard (GPG).'
268
},
269
270
+ 'riakkv': {
271
+ title: 'Riak KV',
272
+ icon: '<i class="fas fa-database"></i>',
273
+ info: 'Metrics for <b>Riak KV</b>, the distributed key-value store.'
274
+ },
275
+
276
'ipfs': {
277
title: 'IPFS',
278
icon: '<i class="fas fa-folder-open"></i>',