Add message queue statistics (#5115)
* Add IPC message queue charts * Add obsolete flag for dimensions * Delete obsolete dimensions from memory * Remove files for obsolete dimensions, filter requests * Make empty charts obsolete * Minimize obsolete dimension checks * Limit the number of dimensions in memory * Remove obsolete dimensions on netdata exit * Update documentation * Move flag to the end * Fix typo * Fix typo
Vladimir Kobal committed
Feb 11, 2019 at 13:24 UTC
2f6f8155dba6951256f5f8e080aafca6e6836dfc
14 files changed
+426
-162
backends/prometheus/backend_prometheus.c
+1
-1
@@ -291,7 +291,7 @@ static void rrd_stats_api_v1_charts_allmetrics_prometheus(RRDHOST *host, BUFFER
291
// for each dimension
292
RRDDIM *rd;
293
rrddim_foreach_read(rd, st) {
294
- if(rd->collections_counter) {
294
+ if(rd->collections_counter && !rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE)) {
295
char dimension[PROMETHEUS_ELEMENT_MAX + 1];
296
char *suffix = "";
297
collectors/all.h
+2
-2
@@ -55,8 +55,8 @@
55
#define NETDATA_CHART_PRIO_SYSTEM_ENTROPY 1000
56
#define NETDATA_CHART_PRIO_SYSTEM_UPTIME 1000
57
#define NETDATA_CHART_PRIO_SYSTEM_IPC_MSQ_QUEUES 990 // freebsd only
58
-#define NETDATA_CHART_PRIO_SYSTEM_IPC_MSQ_MESSAGES 1000 // freebsd only
59
-#define NETDATA_CHART_PRIO_SYSTEM_IPC_MSQ_SIZE 1100 // freebsd only
58
+#define NETDATA_CHART_PRIO_SYSTEM_IPC_MSQ_MESSAGES 1000
59
+#define NETDATA_CHART_PRIO_SYSTEM_IPC_MSQ_SIZE 1100
60
#define NETDATA_CHART_PRIO_SYSTEM_IPC_SEMAPHORES 1000
61
#define NETDATA_CHART_PRIO_SYSTEM_IPC_SEM_ARRAYS 1000
62
#define NETDATA_CHART_PRIO_SYSTEM_IPC_SHARED_MEM_SEGS 1000 // freebsd only
collectors/plugins.d/README.md
+13
-14
@@ -15,7 +15,7 @@ plugin|language|O/S|description
15
[node.d.plugin](../node.d.plugin/)|`node.js`|all|a **plugin orchestrator** for data collection modules written in `node.js`.
16
[python.d.plugin](../python.d.plugin/)|`python`|all|a **plugin orchestrator** for data collection modules written in `python` v2 or v3 (both are supported).
17
18
-Plugin orchestrators may also be described as **modular plugins**. They are modular since they accept custom made modules to be included. Writing modules for these plugins is easier than accessing the native netdata API directly. You will find modules already available for each orchestrator under the directory of the particular modular plugin (e.g. under python.d.plugin for the python orchestrator).
18
+Plugin orchestrators may also be described as **modular plugins**. They are modular since they accept custom made modules to be included. Writing modules for these plugins is easier than accessing the native netdata API directly. You will find modules already available for each orchestrator under the directory of the particular modular plugin (e.g. under python.d.plugin for the python orchestrator).
19
Each of these modular plugins has each own methods for defining modules. Please check the examples and their documentation.
20
21
## Motivation
@@ -49,9 +49,9 @@ Plugins can create any number of charts with any number of dimensions each. Each
49
50
## Configuration
51
52
-Netdata will supply the environment variables `NETDATA_USER_CONFIG_DIR` (for user supplied) and `NETDATA_STOCK_CONFIG_DIR` (for netdata supplied) configuration files to identify the directory where configuration files are stored. It is up to the plugin to read the configuration it needs.
52
+Netdata will supply the environment variables `NETDATA_USER_CONFIG_DIR` (for user supplied) and `NETDATA_STOCK_CONFIG_DIR` (for netdata supplied) configuration files to identify the directory where configuration files are stored. It is up to the plugin to read the configuration it needs.
53
54
-The `netdata.conf` section [plugins] section contains a list of all the plugins found at the system where netdata runs, with a boolean setting to enable them or not.
54
+The `netdata.conf` section [plugins] section contains a list of all the plugins found at the system where netdata runs, with a boolean setting to enable them or not.
55
56
Example:
57
@@ -59,7 +59,7 @@ Example:
59
[plugins]
60
# enable running new plugins = yes
61
# check for new plugins every = 60
62
-
62
+
63
# charts.d = yes
64
# fping = yes
65
# node.d = yes
@@ -70,7 +70,7 @@ The setting `enable running new plugins` changes the default behavior for all ex
70
So if set to `no`, only the plugins that are explicitly set to `yes` will be run.
71
72
The setting `check for new plugins every` controls the time the directory `/usr/libexec/netdata/plugins.d`
73
-will be rescanned for new plugins. So, new plugins can give added anytime.
73
+will be rescanned for new plugins. So, new plugins can give added anytime.
74
75
For each of the external plugins enabled, another `netdata.conf` section
76
is created, in the form of `[plugin:NAME]`, where `NAME` is the name of the external plugin.
@@ -82,7 +82,7 @@ For example, for `apps.plugin` the following section is available:
82
```
83
[plugin:apps]
84
# update every = 1
85
- # command options =
85
+ # command options =
86
```
87
88
- `update every` controls the granularity of the external plugin.
@@ -193,7 +193,7 @@ the template is:
193
is used to group charts together
194
(for example all eth0 charts should say: eth0),
195
if empty or missing, the `id` part of `type.id` will be used
196
-
196
+
197
this controls the sub-menu on the dashboard
198
199
- `context`
@@ -233,7 +233,7 @@ the template is:
233
234
the template is:
235
236
-> DIMENSION id [name [algorithm [multiplier [divisor [hidden]]]]]
236
+> DIMENSION id [name [algorithm [multiplier [divisor [options]]]]]
237
238
where:
239
@@ -283,10 +283,9 @@ the template is:
283
an integer value to divide the collected value,
284
if empty or missing, `1` is used
285
286
- - `hidden`
286
+ - `options`
287
288
- giving the keyword `hidden` will make this dimension hidden,
289
- it will take part in the calculations but will not be presented in the chart
288
+ a space separated list of options, enclosed in quotes. Options supported: `obsolete` to mark a dimension as obsolete (netdata will delete it after some time) and `hidden` to make this dimension hidden, it will take part in the calculations but will not be presented in the chart.
289
290
291
#### VARIABLE
@@ -390,7 +389,7 @@ or do not output the line at all.
389
4. **C**
390
391
Of course, C is the most efficient way of collecting data. This is why netdata itself is written in C.
393
-
392
+
393
## Writing Plugins Properly
394
395
There are a few rules for writing plugins properly:
@@ -411,7 +410,7 @@ There are a few rules for writing plugins properly:
410
var update_every = argv[1] * 1000; /* seconds * 1000 = milliseconds */
411
412
readConfiguration();
414
-
413
+
414
if(!verifyWeCanCollectValues()) {
415
print "DISABLE";
416
exit(1);
@@ -445,7 +444,7 @@ There are a few rules for writing plugins properly:
444
sleepMilliseconds(next_run - now);
445
now = currentTimeStampInMilliseconds();
446
}
448
-
447
+
448
/* calculate the time passed since the last run */
449
if ( loops > 0 )
450
dt_since_last_run = (now - last_run) * 1000; /* in microseconds */
collectors/plugins.d/plugins_d.c
+7
@@ -394,10 +394,17 @@ inline size_t pluginsd_process(RRDHOST *host, struct plugind *cd, FILE *fp, int
394
rrddim_flag_clear(rd, RRDDIM_FLAG_HIDDEN);
395
rrddim_flag_clear(rd, RRDDIM_FLAG_DONT_DETECT_RESETS_OR_OVERFLOWS);
396
if(options && *options) {
397
+ if(strstr(options, "obsolete") != NULL)
398
+ rrddim_is_obsolete(st, rd);
399
+ else
400
+ rrddim_isnot_obsolete(st, rd);
401
if(strstr(options, "hidden") != NULL) rrddim_flag_set(rd, RRDDIM_FLAG_HIDDEN);
402
if(strstr(options, "noreset") != NULL) rrddim_flag_set(rd, RRDDIM_FLAG_DONT_DETECT_RESETS_OR_OVERFLOWS);
403
if(strstr(options, "nooverflow") != NULL) rrddim_flag_set(rd, RRDDIM_FLAG_DONT_DETECT_RESETS_OR_OVERFLOWS);
404
}
405
+ else {
406
+ rrddim_isnot_obsolete(st, rd);
407
+ }
408
}
409
else if(likely(hash == VARIABLE_HASH && !strcmp(s, PLUGINSD_KEYWORD_VARIABLE))) {
410
char *name = words[1];
collectors/proc.plugin/README.md
+15
@@ -20,6 +20,7 @@
20
- `/proc/loadavg` (system load and total processes running)
21
- `/proc/sys/kernel/random/entropy_avail` (random numbers pool availability - used in cryptography)
22
- `/sys/class/power_supply` (power supply properties)
23
+ - `ipc` (IPC semaphores and message queues)
24
- `ksm` Kernel Same-Page Merging performance (several files under `/sys/kernel/mm/ksm`).
25
- `netdata` (internal netdata resources utilization)
26
@@ -343,4 +344,18 @@ corresponding `min` or `empty` attribute, then Netdata will still provide
344
the corresponding `min` or `empty`, which will then always read as zero.
345
This way, alerts which match on these will still work.
346
347
+## IPC
348
+
349
+This module monitors the number of semaphores, semaphore arrays, number of messages in message queues, and amount of memory used by message queues. As far as the message queue charts are dynamic, sane limits are applied for the number of dimensions per chart (the limit is configurable).
350
+
351
+#### configuration
352
+
353
+```
354
+[plugin:proc:ipc]
355
+ # semaphore totals = yes
356
+ # message queues = yes
357
+ # msg filename to monitor = /proc/sysvipc/msg
358
+ # max dimensions in memory allowed = 50
359
+```
360
+
361
[]()
collectors/proc.plugin/ipc.c
+268
-67
@@ -53,6 +53,18 @@ union semun {
53
};
54
#endif
55
56
+struct message_queue {
57
+ unsigned long long id;
58
+ int found;
59
+
60
+ RRDDIM *rd_messages;
61
+ RRDDIM *rd_bytes;
62
+ unsigned long long messages;
63
+ unsigned long long bytes;
64
+
65
+ struct message_queue * next;
66
+};
67
+
68
static inline int ipc_sem_get_limits(struct ipc_limits *lim) {
69
static procfile *ff = NULL;
70
static int error_shown = 0;
@@ -162,102 +174,291 @@ static inline int ipc_sem_get_status(struct ipc_status *st) {
174
return 0;
175
}
176
177
+int ipc_msq_get_info(char *msg_filename, struct message_queue **message_queue_root) {
178
+ static procfile *ff;
179
+ struct message_queue *msq;
180
+
181
+ if(unlikely(!ff)) {
182
+ ff = procfile_open(config_get("plugin:proc:ipc", "msg filename to monitor", msg_filename), " \t:", PROCFILE_FLAG_DEFAULT);
183
+ if(unlikely(!ff)) return 1;
184
+ }
185
+
186
+ ff = procfile_readall(ff);
187
+ if(unlikely(!ff)) return 1;
188
+
189
+ size_t lines = procfile_lines(ff);
190
+ size_t words = 0;
191
+
192
+ if(unlikely(lines < 2)) {
193
+ error("Cannot read %s. Expected 2 or more lines, read %zu.", ff->filename, lines);
194
+ return 1;
195
+ }
196
+
197
+ // loop through all lines except the first and the last ones
198
+ size_t l;
199
+ for(l = 1; l < lines - 1; l++) {
200
+ words = procfile_linewords(ff, l);
201
+ if(unlikely(words < 2)) continue;
202
+ if(unlikely(words < 14)) {
203
+ error("Cannot read %s line. Expected 14 params, read %zu.", ff->filename, words);
204
+ continue;
205
+ }
206
+
207
+ // find the id in the linked list or create a new stucture
208
+ int found = 0;
209
+
210
+ unsigned long long id = str2ull(procfile_lineword(ff, l, 1));
211
+ for(msq = *message_queue_root; msq ; msq = msq->next) {
212
+ if(unlikely(id == msq->id)) {
213
+ found = 1;
214
+ break;
215
+ }
216
+ }
217
+
218
+ if(unlikely(!found)) {
219
+ msq = callocz(1, sizeof(struct message_queue));
220
+ msq->next = *message_queue_root;
221
+ *message_queue_root = msq;
222
+ msq->id = id;
223
+ }
224
+
225
+ msq->messages = str2ull(procfile_lineword(ff, l, 4));
226
+ msq->bytes = str2ull(procfile_lineword(ff, l, 3));
227
+ msq->found = 1;
228
+ }
229
+
230
+ return 0;
231
+}
232
+
233
int do_ipc(int update_every, usec_t dt) {
234
(void)dt;
235
168
- static int initialized = 0, read_limits_next = -1;
236
+ static int do_sem = -1, do_msg = -1;
237
+ static int read_limits_next = -1;
238
static struct ipc_limits limits;
239
static struct ipc_status status;
240
static RRDVAR *arrays_max = NULL, *semaphores_max = NULL;
241
static RRDSET *st_semaphores = NULL, *st_arrays = NULL;
242
static RRDDIM *rd_semaphores = NULL, *rd_arrays = NULL;
243
+ static char *msg_filename = NULL;
244
+ static struct message_queue *message_queue_root = NULL;
245
+ static long long dimensions_limit;
246
+
247
+ if(unlikely(do_sem == -1)) {
248
+ do_sem = config_get_boolean("plugin:proc:ipc", "semaphore totals", CONFIG_BOOLEAN_YES);
249
+ do_msg = config_get_boolean("plugin:proc:ipc", "message queues", CONFIG_BOOLEAN_YES);
250
+
251
+ char filename[FILENAME_MAX + 1];
252
+ snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/proc/sysvipc/msg");
253
+ msg_filename = config_get("plugin:proc:ipc", "msg filename to monitor", filename);
254
+
255
+ dimensions_limit = config_get_number("plugin:proc:ipc", "max dimensions in memory allowed", 50);
256
175
- if(unlikely(!initialized)) {
176
- initialized = 1;
177
-
257
// make sure it works
258
if(ipc_sem_get_limits(&limits) == -1) {
259
error("unable to fetch semaphore limits");
181
- return 1;
260
+ do_sem = CONFIG_BOOLEAN_NO;
261
}
183
-
184
- // make sure it works
185
- if(ipc_sem_get_status(&status) == -1) {
262
+ else if(ipc_sem_get_status(&status) == -1) {
263
error("unable to fetch semaphore statistics");
187
- return 1;
264
+ do_sem = CONFIG_BOOLEAN_NO;
265
}
266
+ else {
267
+ // create the charts
268
+ if(unlikely(!st_semaphores)) {
269
+ st_semaphores = rrdset_create_localhost(
270
+ "system"
271
+ , "ipc_semaphores"
272
+ , NULL
273
+ , "ipc semaphores"
274
+ , NULL
275
+ , "IPC Semaphores"
276
+ , "semaphores"
277
+ , PLUGIN_PROC_NAME
278
+ , "ipc"
279
+ , NETDATA_CHART_PRIO_SYSTEM_IPC_SEMAPHORES
280
+ , localhost->rrd_update_every
281
+ , RRDSET_TYPE_AREA
282
+ );
283
+ rd_semaphores = rrddim_add(st_semaphores, "semaphores", NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
284
+ }
285
+
286
+ if(unlikely(!st_arrays)) {
287
+ st_arrays = rrdset_create_localhost(
288
+ "system"
289
+ , "ipc_semaphore_arrays"
290
+ , NULL
291
+ , "ipc semaphores"
292
+ , NULL
293
+ , "IPC Semaphore Arrays"
294
+ , "arrays"
295
+ , PLUGIN_PROC_NAME
296
+ , "ipc"
297
+ , NETDATA_CHART_PRIO_SYSTEM_IPC_SEM_ARRAYS
298
+ , localhost->rrd_update_every
299
+ , RRDSET_TYPE_AREA
300
+ );
301
+ rd_arrays = rrddim_add(st_arrays, "arrays", NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
302
+ }
303
190
- // create the charts
191
- if(unlikely(!st_semaphores)) {
192
- st_semaphores = rrdset_create_localhost(
193
- "system"
194
- , "ipc_semaphores"
195
- , NULL
196
- , "ipc semaphores"
197
- , NULL
198
- , "IPC Semaphores"
199
- , "semaphores"
200
- , PLUGIN_PROC_NAME
201
- , "ipc"
202
- , NETDATA_CHART_PRIO_SYSTEM_IPC_SEMAPHORES
203
- , localhost->rrd_update_every
204
- , RRDSET_TYPE_AREA
205
- );
206
- rd_semaphores = rrddim_add(st_semaphores, "semaphores", NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
304
+ // variables
305
+ semaphores_max = rrdvar_custom_host_variable_create(localhost, "ipc_semaphores_max");
306
+ arrays_max = rrdvar_custom_host_variable_create(localhost, "ipc_semaphores_arrays_max");
307
}
308
209
- if(unlikely(!st_arrays)) {
210
- st_arrays = rrdset_create_localhost(
211
- "system"
212
- , "ipc_semaphore_arrays"
213
- , NULL
214
- , "ipc semaphores"
215
- , NULL
216
- , "IPC Semaphore Arrays"
217
- , "arrays"
218
- , PLUGIN_PROC_NAME
219
- , "ipc"
220
- , NETDATA_CHART_PRIO_SYSTEM_IPC_SEM_ARRAYS
221
- , localhost->rrd_update_every
222
- , RRDSET_TYPE_AREA
223
- );
224
- rd_arrays = rrddim_add(st_arrays, "arrays", NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
309
+ struct stat stbuf;
310
+ if (stat(msg_filename, &stbuf)) {
311
+ do_msg = CONFIG_BOOLEAN_NO;
312
}
313
227
- // variables
228
- semaphores_max = rrdvar_custom_host_variable_create(localhost, "ipc_semaphores_max");
229
- arrays_max = rrdvar_custom_host_variable_create(localhost, "ipc_semaphores_arrays_max");
314
+ if(unlikely(do_sem == CONFIG_BOOLEAN_NO && do_msg == CONFIG_BOOLEAN_NO)) {
315
+ error("ipc module disabled");
316
+ return 1;
317
+ }
318
}
319
232
- if(unlikely(read_limits_next < 0)) {
233
- if(unlikely(ipc_sem_get_limits(&limits) == -1)) {
234
- error("Unable to fetch semaphore limits.");
235
- }
236
- else {
237
- if(semaphores_max) rrdvar_custom_host_variable_set(localhost, semaphores_max, limits.semmns);
238
- if(arrays_max) rrdvar_custom_host_variable_set(localhost, arrays_max, limits.semmni);
320
+ if(likely(do_sem != CONFIG_BOOLEAN_NO)) {
321
+ if(unlikely(read_limits_next < 0)) {
322
+ if(unlikely(ipc_sem_get_limits(&limits) == -1)) {
323
+ error("Unable to fetch semaphore limits.");
324
+ }
325
+ else {
326
+ if(semaphores_max) rrdvar_custom_host_variable_set(localhost, semaphores_max, limits.semmns);
327
+ if(arrays_max) rrdvar_custom_host_variable_set(localhost, arrays_max, limits.semmni);
328
240
- st_arrays->red = limits.semmni;
241
- st_semaphores->red = limits.semmns;
329
+ st_arrays->red = limits.semmni;
330
+ st_semaphores->red = limits.semmns;
331
243
- read_limits_next = 60 / update_every;
332
+ read_limits_next = 60 / update_every;
333
+ }
334
}
245
- }
246
- else
247
- read_limits_next--;
335
+ else
336
+ read_limits_next--;
337
249
- if(unlikely(ipc_sem_get_status(&status) == -1)) {
250
- error("Unable to get semaphore statistics");
251
- return 0;
338
+ if(unlikely(ipc_sem_get_status(&status) == -1)) {
339
+ error("Unable to get semaphore statistics");
340
+ return 0;
341
+ }
342
+
343
+ if(st_semaphores->counter_done) rrdset_next(st_semaphores);
344
+ rrddim_set_by_pointer(st_semaphores, rd_semaphores, status.semaem);
345
+ rrdset_done(st_semaphores);
346
+
347
+ if(st_arrays->counter_done) rrdset_next(st_arrays);
348
+ rrddim_set_by_pointer(st_arrays, rd_arrays, status.semusz);
349
+ rrdset_done(st_arrays);
350
}
351
254
- if(st_semaphores->counter_done) rrdset_next(st_semaphores);
255
- rrddim_set_by_pointer(st_semaphores, rd_semaphores, status.semaem);
256
- rrdset_done(st_semaphores);
352
+ // --------------------------------------------------------------------
353
+
354
+ if(likely(do_msg != CONFIG_BOOLEAN_NO)) {
355
+ static RRDSET *st_msq_messages = NULL, *st_msq_bytes = NULL;
356
+
357
+ int ret = ipc_msq_get_info(msg_filename, &message_queue_root);
358
+
359
+ if(!ret && message_queue_root) {
360
+ if(unlikely(!st_msq_messages))
361
+ st_msq_messages = rrdset_create_localhost(
362
+ "system"
363
+ , "message_queue_messages"
364
+ , NULL
365
+ , "ipc message queues"
366
+ , NULL
367
+ , "IPC Message Queue Number of Messages"
368
+ , "messages"
369
+ , PLUGIN_PROC_NAME
370
+ , "ipc"
371
+ , NETDATA_CHART_PRIO_SYSTEM_IPC_MSQ_MESSAGES
372
+ , update_every
373
+ , RRDSET_TYPE_STACKED
374
+ );
375
+ else
376
+ rrdset_next(st_msq_messages);
377
+
378
+ if(unlikely(!st_msq_bytes))
379
+ st_msq_bytes = rrdset_create_localhost(
380
+ "system"
381
+ , "message_queue_bytes"
382
+ , NULL
383
+ , "ipc message queues"
384
+ , NULL
385
+ , "IPC Message Queue Used Bytes"
386
+ , "bytes"
387
+ , PLUGIN_PROC_NAME
388
+ , "ipc"
389
+ , NETDATA_CHART_PRIO_SYSTEM_IPC_MSQ_SIZE
390
+ , update_every
391
+ , RRDSET_TYPE_STACKED
392
+ );
393
+ else
394
+ rrdset_next(st_msq_bytes);
395
+
396
+ struct message_queue *msq = message_queue_root, *msq_prev = NULL;
397
+ while(likely(msq)){
398
+ if(likely(msq->found)) {
399
+ if(unlikely(!msq->rd_messages || !msq->rd_bytes)) {
400
+ char id[RRD_ID_LENGTH_MAX + 1];
401
+ snprintfz(id, RRD_ID_LENGTH_MAX, "%llu", msq->id);
402
+ if(likely(!msq->rd_messages)) msq->rd_messages = rrddim_add(st_msq_messages, id, NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
403
+ if(likely(!msq->rd_bytes)) msq->rd_bytes = rrddim_add(st_msq_bytes, id, NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
404
+ }
405
+
406
+ rrddim_set_by_pointer(st_msq_messages, msq->rd_messages, msq->messages);
407
+ rrddim_set_by_pointer(st_msq_bytes, msq->rd_bytes, msq->bytes);
408
+
409
+ msq->found = 0;
410
+ }
411
+ else {
412
+ rrddim_is_obsolete(st_msq_messages, msq->rd_messages);
413
+ rrddim_is_obsolete(st_msq_bytes, msq->rd_bytes);
414
+
415
+ // remove message queue from the linked list
416
+ if(!msq_prev)
417
+ message_queue_root = msq->next;
418
+ else
419
+ msq_prev->next = msq->next;
420
+ freez(msq);
421
+ msq = NULL;
422
+ }
423
+ if(likely(msq)) {
424
+ msq_prev = msq;
425
+ msq = msq->next;
426
+ }
427
+ else if(!msq_prev)
428
+ msq = message_queue_root;
429
+ else
430
+ msq = msq_prev->next;
431
+ }
432
258
- if(st_arrays->counter_done) rrdset_next(st_arrays);
259
- rrddim_set_by_pointer(st_arrays, rd_arrays, status.semusz);
260
- rrdset_done(st_arrays);
433
+ rrdset_done(st_msq_messages);
434
+ rrdset_done(st_msq_bytes);
435
+
436
+ long long dimensions_num = 0;
437
+ RRDDIM *rd;
438
+ rrdset_rdlock(st_msq_messages);
439
+ rrddim_foreach_read(rd, st_msq_messages) dimensions_num++;
440
+ rrdset_unlock(st_msq_messages);
441
+
442
+ if(unlikely(dimensions_num > dimensions_limit)) {
443
+ info("Message queue statistics has been disabled");
444
+ info("There are %lld dimensions in memory but limit was set to %lld", dimensions_num, dimensions_limit);
445
+ rrdset_is_obsolete(st_msq_messages);
446
+ rrdset_is_obsolete(st_msq_bytes);
447
+ st_msq_messages = NULL;
448
+ st_msq_bytes = NULL;
449
+ do_msg = CONFIG_BOOLEAN_NO;
450
+ }
451
+ else if(unlikely(!message_queue_root)) {
452
+ info("Making chart %s (%s) obsolete since it does not have any dimensions", st_msq_messages->name, st_msq_messages->id);
453
+ rrdset_is_obsolete(st_msq_messages);
454
+ st_msq_messages = NULL;
455
+
456
+ info("Making chart %s (%s) obsolete since it does not have any dimensions", st_msq_bytes->name, st_msq_bytes->id);
457
+ rrdset_is_obsolete(st_msq_bytes);
458
+ st_msq_bytes = NULL;
459
+ }
460
+ }
461
+ }
462
463
return 0;
464
}
daemon/config/README.md
+35
-35
@@ -12,9 +12,9 @@ This config file **is not needed by default**. Netdata works fine out of the box
12
2. `[web]` to [configure the web server](../../web/server).
13
3. `[plugins]` to [configure](#plugins-section-options) which [collectors](../../collectors) to use and PATH settings.
14
4. `[health]` to [configure](#health-section-options) general settings for [health monitoring](../../health)
15
-5. `[registry]` for the [netdata registry](../../registry).
15
+5. `[registry]` for the [netdata registry](../../registry).
16
6. `[backend]` to set up [streaming and replication](../../streaming) options.
17
-7. `[statsd]` for the general settings of the [stats.d.plugin](../../collectors/statsd.plugin).
17
+7. `[statsd]` for the general settings of the [stats.d.plugin](../../collectors/statsd.plugin).
18
8. `[plugin:NAME]` sections for each collector plugin, under the comment [Per plugin configuration](#per-plugin-configuration).
19
9. `[CHART_NAME]` sections for each chart defined, under the comment [Per chart configuration](#per-chart-configuration).
20
@@ -46,33 +46,33 @@ process scheduling policy | `keep` | See [netdata process scheduling policy](..
46
OOM score | `1000` | See [OOM score](../#oom-score)
47
glibc malloc arena max for plugins | `1` | See [Virtual memory](../#virtual-memory).
48
glibc malloc arena max for netdata | `1` | See [Virtual memory](../#virtual-memory).
49
-hostname | auto-detected | The hostname of the computer running netdata.
50
-history | `3996` | The number of entries the netdata daemon will by default keep in memory for each chart dimension. This setting can also be configured per chart. Check [Memory Requirements](../../database/#database) for more information.
51
-update every | `1` | The frequency in seconds, for data collection. For more information see [Performance](../../docs/Performance.md#performance).
52
-config directory | `/etc/netdata` | The directory configuration files are kept.
53
-stock config directory | `/usr/lib/netdata/conf.d` |
54
-log directory | `/var/log/netdata` | The directory in which the [log files](../#log-files) are kept.
55
-web files directory | `/usr/share/netdata/web` | The directory the web static files are kept.
56
-cache directory | `/var/cache/netdata` | The directory the memory database will be stored if and when netdata exits. Netdata will re-read the database when it will start again, to continue from the same point.
49
+hostname | auto-detected | The hostname of the computer running netdata.
50
+history | `3996` | The number of entries the netdata daemon will by default keep in memory for each chart dimension. This setting can also be configured per chart. Check [Memory Requirements](../../database/#database) for more information.
51
+update every | `1` | The frequency in seconds, for data collection. For more information see [Performance](../../docs/Performance.md#performance).
52
+config directory | `/etc/netdata` | The directory configuration files are kept.
53
+stock config directory | `/usr/lib/netdata/conf.d` |
54
+log directory | `/var/log/netdata` | The directory in which the [log files](../#log-files) are kept.
55
+web files directory | `/usr/share/netdata/web` | The directory the web static files are kept.
56
+cache directory | `/var/cache/netdata` | The directory the memory database will be stored if and when netdata exits. Netdata will re-read the database when it will start again, to continue from the same point.
57
lib directory | `/var/lib/netdata` | Contains the alarm log and the netdata instance guid.
58
home directory | `/var/cache/netdata` | Contains the db files for the collected metrics
59
-plugins directory | `"/usr/libexec/netdata/plugins.d" "/etc/netdata/custom-plugins.d"` | The directory plugin programs are kept. This setting supports multiple directories, space separated. If any directory path contains spaces, enclose it in single or double quotes.
60
-memory mode | `save` | When set to `save` netdata will save its round robin database on exit and load it on startup. When set to `map` the cache files will be updated in real time (check `man mmap` - do not set this on systems with heavy load or slow disks - the disks will continuously sync the in-memory database of netdata). When set to `ram` the round robin database will be temporary and it will be lost when netdata exits. `none` disables the database at this host. This also disables health monitoring (there cannot be health monitoring without a database). host access prefix | | This is used in docker environments where /proc, /sys, etc have to be accessed via another path. You may also have to set SYS_PTRACE capability on the docker for this work. Check [issue 43](https://github.com/netdata/netdata/issues/43).
61
-memory deduplication (ksm) | `yes` | When set to `yes`, netdata will offer its in-memory round robin database to kernel same page merging (KSM) for deduplication. For more information check [Memory Deduplication - Kernel Same Page Merging - KSM](../../database/#ksm)
62
-TZ environment variable | `:/etc/localtime` | Where to find the timezone
63
-timezone | auto-detected | The timezone retrieved from the environment variable
64
-debug flags | `0x0000000000000000` | Bitmap of debug options to enable. For more information check [Tracing Options](../#debugging).
65
-debug log | `/var/log/netdata/debug.log` | The filename to save debug information. This file will not be created is debugging is not enabled. You can also set it to `syslog` to send the debug messages to syslog, or `none` to disable this log. For more information check [Tracing Options](../#debugging).
66
-error log | `/var/log/netdata/error.log` | The filename to save error messages for netdata daemon and all plugins (`stderr` is sent here for all netdata programs, including the plugins). You can also set it to `syslog` to send the errors to syslog, or `none` to disable this log.
67
-access log | `/var/log/netdata/access.log` | The filename to save the log of web clients accessing netdata charts. You can also set it to `syslog` to send the access log to syslog, or `none` to disable this log.
59
+plugins directory | `"/usr/libexec/netdata/plugins.d" "/etc/netdata/custom-plugins.d"` | The directory plugin programs are kept. This setting supports multiple directories, space separated. If any directory path contains spaces, enclose it in single or double quotes.
60
+memory mode | `save` | When set to `save` netdata will save its round robin database on exit and load it on startup. When set to `map` the cache files will be updated in real time (check `man mmap` - do not set this on systems with heavy load or slow disks - the disks will continuously sync the in-memory database of netdata). When set to `ram` the round robin database will be temporary and it will be lost when netdata exits. `none` disables the database at this host. This also disables health monitoring (there cannot be health monitoring without a database). host access prefix | | This is used in docker environments where /proc, /sys, etc have to be accessed via another path. You may also have to set SYS_PTRACE capability on the docker for this work. Check [issue 43](https://github.com/netdata/netdata/issues/43).
61
+memory deduplication (ksm) | `yes` | When set to `yes`, netdata will offer its in-memory round robin database to kernel same page merging (KSM) for deduplication. For more information check [Memory Deduplication - Kernel Same Page Merging - KSM](../../database/#ksm)
62
+TZ environment variable | `:/etc/localtime` | Where to find the timezone
63
+timezone | auto-detected | The timezone retrieved from the environment variable
64
+debug flags | `0x0000000000000000` | Bitmap of debug options to enable. For more information check [Tracing Options](../#debugging).
65
+debug log | `/var/log/netdata/debug.log` | The filename to save debug information. This file will not be created is debugging is not enabled. You can also set it to `syslog` to send the debug messages to syslog, or `none` to disable this log. For more information check [Tracing Options](../#debugging).
66
+error log | `/var/log/netdata/error.log` | The filename to save error messages for netdata daemon and all plugins (`stderr` is sent here for all netdata programs, including the plugins). You can also set it to `syslog` to send the errors to syslog, or `none` to disable this log.
67
+access log | `/var/log/netdata/access.log` | The filename to save the log of web clients accessing netdata charts. You can also set it to `syslog` to send the access log to syslog, or `none` to disable this log.
68
errors flood protection period | `1200` | UNUSED - Length of period (in sec) during which the number of errors should not exceed the `errors to trigger flood protection`.
69
errors to trigger flood protection | `200` | UNUSED - Number of errors written to the log in `errors flood protection period` sec before flood protection is activated.
70
-run as user | `netdata` | The user netdata will run as.
71
-pthread stack size | auto-detected |
72
-cleanup obsolete charts after seconds | `3600` | See [monitoring ephemeral containers](../../collectors/cgroups.plugin/#monitoring-ephemeral-containers)
73
-gap when lost iterations above | `1` |
70
+run as user | `netdata` | The user netdata will run as.
71
+pthread stack size | auto-detected |
72
+cleanup obsolete charts after seconds | `3600` | See [monitoring ephemeral containers](../../collectors/cgroups.plugin/#monitoring-ephemeral-containers), also sets the timeout for cleaning up obsolete dimensions
73
+gap when lost iterations above | `1` |
74
cleanup orphan hosts after seconds | `3600` | How long to wait until automatically removing from the DB a remote netdata host (slave) that is no longer sending data.
75
-delete obsolete charts files | `yes` | See [monitoring ephemeral containers](../../collectors/cgroups.plugin/#monitoring-ephemeral-containers)
75
+delete obsolete charts files | `yes` | See [monitoring ephemeral containers](../../collectors/cgroups.plugin/#monitoring-ephemeral-containers), also affects the deletion of files for obsolete dimensions
76
delete orphan hosts files | `yes` | Set to `no` to disable non-responsive host removal.
77
78
### [web] section options
@@ -81,40 +81,40 @@ Refer to the [web server documentation](../../web/server)
81
82
### [plugins] section options
83
84
-In this section you will see be a boolean (`yes`/`no`) option for each plugin (e.g. tc, cgroups, apps, proc etc.). Note that the configuration options in this section for the orchestrator plugins `python.d`, `charts.d` and `node.d` control **all the modules** written for that orchestrator. For instance, setting `python.d = no` means that all Python modules under `collectors/python.d.plugin` will be disabled.
84
+In this section you will see be a boolean (`yes`/`no`) option for each plugin (e.g. tc, cgroups, apps, proc etc.). Note that the configuration options in this section for the orchestrator plugins `python.d`, `charts.d` and `node.d` control **all the modules** written for that orchestrator. For instance, setting `python.d = no` means that all Python modules under `collectors/python.d.plugin` will be disabled.
85
86
Additionally, there will be the following options:
87
88
setting | default | info
89
:------:|:-------:|:----
90
-PATH environment variable | `auto-detected` |
90
+PATH environment variable | `auto-detected` |
91
PYTHONPATH environment variable | | Used to set a custom python path
92
-enable running new plugins | `yes` | When set to `yes`, netdata will enable detected plugins, even if they are not configured explicitly. Setting this to `no` will only enable plugins explicitly configirued in this file with a `yes`
92
+enable running new plugins | `yes` | When set to `yes`, netdata will enable detected plugins, even if they are not configured explicitly. Setting this to `no` will only enable plugins explicitly configirued in this file with a `yes`
93
check for new plugins every | 60 | The time in seconds to check for new plugins in the plugins directory. This allows having other applications dynamically creating plugins for netdata.
94
-checks | `no` | This is a debugging plugin for the internal latency
94
+checks | `no` | This is a debugging plugin for the internal latency
95
96
### [health] section options
97
98
-This section controls the general behavior of the health monitoring capabilities of Netdata.
98
+This section controls the general behavior of the health monitoring capabilities of Netdata.
99
100
-Specific alarms are configured in per-collector config files under the `health.d` directory. For more info, see [health monitoring](../../health/#health-monitoring).
100
+Specific alarms are configured in per-collector config files under the `health.d` directory. For more info, see [health monitoring](../../health/#health-monitoring).
101
102
-[Alarm notifications](../../health/notifications/#netdata-alarm-notifications) are configured in `health_alarm_notify.conf`.
102
+[Alarm notifications](../../health/notifications/#netdata-alarm-notifications) are configured in `health_alarm_notify.conf`.
103
104
setting | default | info
105
:------:|:-------:|:----
106
enabled | `yes` | Set to `no` to disable all alarms and notifications
107
in memory max health log entries | 1000 | Size of the alarm history held in RAM
108
-script to execute on alarm | `/usr/libexec/netdata/plugins.d/alarm-notify.sh` | The script that sends alarm notifications.
108
+script to execute on alarm | `/usr/libexec/netdata/plugins.d/alarm-notify.sh` | The script that sends alarm notifications.
109
stock health configuration directory | `/usr/lib/netdata/conf.d/health.d` | Contains the stock alarm configuration files for each collector
110
health configuration directory | `/etc/netdata/health.d` | The directory containing the user alarm configuration files, to override the stock configurations
111
-run at least every seconds | `10` | Controls how often all alarm conditions should be evaluated.
111
+run at least every seconds | `10` | Controls how often all alarm conditions should be evaluated.
112
postpone alarms during hibernation for seconds | `60` | Prevents false alarms. May need to be increased if you get alarms during hibernation.
113
rotate log every lines | 2000 | Controls the number of alarm log entries stored in `<lib directory>/health-log.db`, where <lib directory> is the one configured in the [[global] section](#global-section-options)
114
115
### [registry] section options
116
117
-To understand what this section is and how it should be configured, please refer to the [registry documentation](../../registry).
117
+To understand what this section is and how it should be configured, please refer to the [registry documentation](../../registry).
118
119
### [backend]
120
@@ -135,7 +135,7 @@ External plugins will have only 2 options at `netdata.conf`:
135
setting | default | info
136
:------:|:-------:|:----
137
update every|the value of `[global].update every` setting|The frequency in seconds the plugin should collect values. For more information check [Performance](../../docs/Performance.md#performance).
138
-command options|*empty*|Additional command line options to pass to the plugin.
138
+command options|*empty*|Additional command line options to pass to the plugin.
139
140
External plugins that need additional configuration may support a dedicated file in `/etc/netdata`. Check their documentation.
141
database/rrd.h
+23
-16
@@ -31,6 +31,7 @@ typedef struct alarm_entry ALARM_ENTRY;
31
extern int default_rrd_update_every;
32
extern int default_rrd_history_entries;
33
extern int gap_when_lost_iterations_above;
34
+extern time_t rrdset_free_obsolete_time;
35
36
#define RRD_ID_LENGTH_MAX 200
37
@@ -123,7 +124,8 @@ typedef struct rrdfamily RRDFAMILY;
124
typedef enum rrddim_flags {
125
RRDDIM_FLAG_NONE = 0,
126
RRDDIM_FLAG_HIDDEN = (1 << 0), // this dimension will not be offered to callers
126
- RRDDIM_FLAG_DONT_DETECT_RESETS_OR_OVERFLOWS = (1 << 1) // do not offer RESET or OVERFLOW info to callers
127
+ RRDDIM_FLAG_DONT_DETECT_RESETS_OR_OVERFLOWS = (1 << 1), // do not offer RESET or OVERFLOW info to callers
128
+ RRDDIM_FLAG_OBSOLETE = (1 << 2) // this is marked by the collector/module as obsolete
129
} RRDDIM_FLAGS;
130
131
#ifdef HAVE_C___ATOMIC
@@ -242,21 +244,22 @@ struct rrddim {
244
// and may lead to missing information.
245
246
typedef enum rrdset_flags {
245
- RRDSET_FLAG_ENABLED = 1 << 0, // enables or disables a chart
246
- RRDSET_FLAG_DETAIL = 1 << 1, // if set, the data set should be considered as a detail of another
247
- // (the master data set should be the one that has the same family and is not detail)
248
- RRDSET_FLAG_DEBUG = 1 << 2, // enables or disables debugging for a chart
249
- RRDSET_FLAG_OBSOLETE = 1 << 3, // this is marked by the collector/module as obsolete
250
- RRDSET_FLAG_BACKEND_SEND = 1 << 4, // if set, this chart should be sent to backends
251
- RRDSET_FLAG_BACKEND_IGNORE = 1 << 5, // if set, this chart should not be sent to backends
252
- RRDSET_FLAG_UPSTREAM_SEND = 1 << 6, // if set, this chart should be sent upstream (streaming)
253
- RRDSET_FLAG_UPSTREAM_IGNORE = 1 << 7, // if set, this chart should not be sent upstream (streaming)
254
- RRDSET_FLAG_UPSTREAM_EXPOSED = 1 << 8, // if set, we have sent this chart definition to netdata master (streaming)
255
- RRDSET_FLAG_STORE_FIRST = 1 << 9, // if set, do not eliminate the first collection during interpolation
256
- RRDSET_FLAG_HETEROGENEOUS = 1 << 10, // if set, the chart is not homogeneous (dimensions in it have multiple algorithms, multipliers or dividers)
257
- RRDSET_FLAG_HOMEGENEOUS_CHECK = 1 << 11, // if set, the chart should be checked to determine if the dimensions as homogeneous
258
- RRDSET_FLAG_HIDDEN = 1 << 12, // if set, do not show this chart on the dashboard, but use it for backends
259
- RRDSET_FLAG_SYNC_CLOCK = 1 << 13, // if set, microseconds on next data collection will be ignored (the chart will be synced to now)
247
+ RRDSET_FLAG_ENABLED = 1 << 0, // enables or disables a chart
248
+ RRDSET_FLAG_DETAIL = 1 << 1, // if set, the data set should be considered as a detail of another
249
+ // (the master data set should be the one that has the same family and is not detail)
250
+ RRDSET_FLAG_DEBUG = 1 << 2, // enables or disables debugging for a chart
251
+ RRDSET_FLAG_OBSOLETE = 1 << 3, // this is marked by the collector/module as obsolete
252
+ RRDSET_FLAG_BACKEND_SEND = 1 << 4, // if set, this chart should be sent to backends
253
+ RRDSET_FLAG_BACKEND_IGNORE = 1 << 5, // if set, this chart should not be sent to backends
254
+ RRDSET_FLAG_UPSTREAM_SEND = 1 << 6, // if set, this chart should be sent upstream (streaming)
255
+ RRDSET_FLAG_UPSTREAM_IGNORE = 1 << 7, // if set, this chart should not be sent upstream (streaming)
256
+ RRDSET_FLAG_UPSTREAM_EXPOSED = 1 << 8, // if set, we have sent this chart definition to netdata master (streaming)
257
+ RRDSET_FLAG_STORE_FIRST = 1 << 9, // if set, do not eliminate the first collection during interpolation
258
+ RRDSET_FLAG_HETEROGENEOUS = 1 << 10, // if set, the chart is not homogeneous (dimensions in it have multiple algorithms, multipliers or dividers)
259
+ RRDSET_FLAG_HOMEGENEOUS_CHECK = 1 << 11, // if set, the chart should be checked to determine if the dimensions as homogeneous
260
+ RRDSET_FLAG_HIDDEN = 1 << 12, // if set, do not show this chart on the dashboard, but use it for backends
261
+ RRDSET_FLAG_SYNC_CLOCK = 1 << 13, // if set, microseconds on next data collection will be ignored (the chart will be synced to now)
262
+ RRDSET_FLAG_OBSOLETE_DIMENSIONS = 1 << 14 // this is marked by the collector/module when a chart has obsolete dimensions
263
} RRDSET_FLAGS;
264
265
#ifdef HAVE_C___ATOMIC
@@ -846,6 +849,9 @@ extern RRDDIM *rrddim_find(RRDSET *st, const char *id);
849
extern int rrddim_hide(RRDSET *st, const char *id);
850
extern int rrddim_unhide(RRDSET *st, const char *id);
851
852
+extern void rrddim_is_obsolete(RRDSET *st, RRDDIM *rd);
853
+extern void rrddim_isnot_obsolete(RRDSET *st, RRDDIM *rd);
854
+
855
extern collected_number rrddim_set_by_pointer(RRDSET *st, RRDDIM *rd, collected_number value);
856
extern collected_number rrddim_set(RRDSET *st, const char *id, collected_number value);
857
@@ -879,6 +885,7 @@ extern void rrdset_free(RRDSET *st);
885
extern void rrdset_reset(RRDSET *st);
886
extern void rrdset_save(RRDSET *st);
887
extern void rrdset_delete(RRDSET *st);
888
+extern void rrdset_delete_obsolete_dimensions(RRDSET *st);
889
890
extern void rrdhost_cleanup_obsolete_charts(RRDHOST *host);
891
database/rrddim.c
+12
@@ -368,6 +368,18 @@ int rrddim_unhide(RRDSET *st, const char *id) {
368
return 0;
369
}
370
371
+inline void rrddim_is_obsolete(RRDSET *st, RRDDIM *rd) {
372
+ debug(D_RRD_CALLS, "rrddim_is_obsolete() for chart %s, dimension %s", st->name, rd->name);
373
+
374
+ rrddim_flag_set(rd, RRDDIM_FLAG_OBSOLETE);
375
+ rrdset_flag_set(st, RRDSET_FLAG_OBSOLETE_DIMENSIONS);
376
+}
377
+
378
+inline void rrddim_isnot_obsolete(RRDSET *st, RRDDIM *rd) {
379
+ debug(D_RRD_CALLS, "rrddim_isnot_obsolete() for chart %s, dimension %s", st->name, rd->name);
380
+
381
+ rrddim_flag_clear(rd, RRDDIM_FLAG_OBSOLETE);
382
+}
383
384
// ----------------------------------------------------------------------------
385
// RRDDIM - collect values for a dimension
database/rrdhost.c
+2
@@ -665,6 +665,8 @@ void rrdhost_cleanup_charts(RRDHOST *host) {
665
666
if(rrdhost_delete_obsolete_charts && rrdset_flag_check(st, RRDSET_FLAG_OBSOLETE))
667
rrdset_delete(st);
668
+ else if(rrdhost_delete_obsolete_charts && rrdset_flag_check(st, RRDSET_FLAG_OBSOLETE_DIMENSIONS))
669
+ rrdset_delete_obsolete_dimensions(st);
670
else
671
rrdset_save(st);
672
database/rrdset.c
+39
-19
@@ -417,6 +417,24 @@ void rrdset_delete(RRDSET *st) {
417
recursively_delete_dir(st->cache_dir, "left-over chart");
418
}
419
420
+void rrdset_delete_obsolete_dimensions(RRDSET *st) {
421
+ RRDDIM *rd;
422
+
423
+ rrdset_check_rdlock(st);
424
+
425
+ info("Deleting dimensions of chart '%s' ('%s') from disk...", st->id, st->name);
426
+
427
+ rrddim_foreach_read(rd, st) {
428
+ if(rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE)) {
429
+ if(likely(rd->rrd_memory_mode == RRD_MEMORY_MODE_SAVE || rd->rrd_memory_mode == RRD_MEMORY_MODE_MAP)) {
430
+ info("Deleting dimension file '%s'.", rd->cache_filename);
431
+ if(unlikely(unlink(rd->cache_filename) == -1))
432
+ error("Cannot delete dimension file '%s'", rd->cache_filename);
433
+ }
434
+ }
435
+ }
436
+}
437
+
438
// ----------------------------------------------------------------------------
439
// RRDSET - create a chart
440
@@ -1303,6 +1321,11 @@ void rrdset_done(RRDSET *st) {
1321
continue;
1322
}
1323
1324
+ if(unlikely(rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE))) {
1325
+ error("Dimension %s in chart '%s' has the OBSOLETE flag set, but it is collected.", rd->name, st->id);
1326
+ rrddim_isnot_obsolete(st, rd);
1327
+ }
1328
+
1329
#ifdef NETDATA_INTERNAL_CHECKS
1330
rrdset_debug(st, "%s: START "
1331
" last_collected_value = " COLLECTED_NUMBER_FORMAT
@@ -1582,37 +1605,37 @@ void rrdset_done(RRDSET *st) {
1605
// ALL DONE ABOUT THE DATA UPDATE
1606
// --------------------------------------------------------------------
1607
1585
-/*
1586
- // find if there are any obsolete dimensions (not updated recently)
1587
- if(unlikely(rrd_delete_unupdated_dimensions)) {
1608
+ // find if there are any obsolete dimensions
1609
+ time_t now = now_realtime_sec();
1610
1589
- for( rd = st->dimensions; likely(rd) ; rd = rd->next )
1590
- if((rd->last_collected_time.tv_sec + (rrd_delete_unupdated_dimensions * st->update_every)) < st->last_collected_time.tv_sec)
1611
+ if(unlikely(rrddim_flag_check(st, RRDSET_FLAG_OBSOLETE_DIMENSIONS))) {
1612
+ rrddim_foreach_read(rd, st)
1613
+ if(unlikely(rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE)))
1614
break;
1615
1616
if(unlikely(rd)) {
1617
RRDDIM *last;
1595
- // there is dimension to free
1618
+ // there is a dimension to free
1619
// upgrade our read lock to a write lock
1620
rrdset_unlock(st);
1621
rrdset_wrlock(st);
1622
1623
for( rd = st->dimensions, last = NULL ; likely(rd) ; ) {
1601
- // remove it only it is not updated in rrd_delete_unupdated_dimensions seconds
1602
-
1603
- if(unlikely((rd->last_collected_time.tv_sec + (rrd_delete_unupdated_dimensions * st->update_every)) < st->last_collected_time.tv_sec)) {
1624
+ if(unlikely(rd->last_collected_time.tv_sec + rrdset_free_obsolete_time < now)) {
1625
info("Removing obsolete dimension '%s' (%s) of '%s' (%s).", rd->name, rd->id, st->name, st->id);
1626
1627
+ if(likely(rd->rrd_memory_mode == RRD_MEMORY_MODE_SAVE || rd->rrd_memory_mode == RRD_MEMORY_MODE_MAP)) {
1628
+ info("Deleting dimension file '%s'.", rd->cache_filename);
1629
+ if(unlikely(unlink(rd->cache_filename) == -1))
1630
+ error("Cannot delete dimension file '%s'", rd->cache_filename);
1631
+ }
1632
+
1633
if(unlikely(!last)) {
1607
- st->dimensions = rd->next;
1608
- rd->next = NULL;
1634
rrddim_free(st, rd);
1635
rd = st->dimensions;
1636
continue;
1637
}
1638
else {
1614
- last->next = rd->next;
1615
- rd->next = NULL;
1639
rrddim_free(st, rd);
1640
rd = last->next;
1641
continue;
@@ -1622,14 +1645,11 @@ void rrdset_done(RRDSET *st) {
1645
last = rd;
1646
rd = rd->next;
1647
}
1625
-
1626
- if(unlikely(!st->dimensions)) {
1627
- info("Disabling chart %s (%s) since it does not have any dimensions", st->name, st->id);
1628
- st->enabled = 0;
1629
- }
1648
+ }
1649
+ else {
1650
+ rrdset_flag_clear(st, RRDSET_FLAG_OBSOLETE_DIMENSIONS);
1651
}
1652
}
1632
-*/
1653
1654
rrdset_unlock(st);
1655
streaming/rrdpush.c
+6
-5
@@ -188,12 +188,13 @@ static inline void rrdpush_send_chart_definition_nolock(RRDSET *st) {
188
rrddim_foreach_read(rd, st) {
189
buffer_sprintf(
190
host->rrdpush_sender_buffer
191
- , "DIMENSION \"%s\" \"%s\" \"%s\" " COLLECTED_NUMBER_FORMAT " " COLLECTED_NUMBER_FORMAT " \"%s %s\"\n"
191
+ , "DIMENSION \"%s\" \"%s\" \"%s\" " COLLECTED_NUMBER_FORMAT " " COLLECTED_NUMBER_FORMAT " \"%s %s %s\"\n"
192
, rd->id
193
, rd->name
194
, rrd_algorithm_name(rd->algorithm)
195
, rd->multiplier
196
, rd->divisor
197
+ , rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE)?"obsolete":""
198
, rrddim_flag_check(rd, RRDDIM_FLAG_HIDDEN)?"hidden":""
199
, rrddim_flag_check(rd, RRDDIM_FLAG_DONT_DETECT_RESETS_OR_OVERFLOWS)?"noreset":""
200
);
@@ -737,16 +738,16 @@ void *rrdpush_sender_thread(void *ptr) {
738
739
if(host->rrdpush_sender_socket != -1) {
740
char *error = NULL;
740
-
741
+
742
if (unlikely(ofd->revents & POLLERR))
743
error = "socket reports errors (POLLERR)";
743
-
744
+
745
else if (unlikely(ofd->revents & POLLHUP))
746
error = "connection closed by remote end (POLLHUP)";
746
-
747
+
748
else if (unlikely(ofd->revents & POLLNVAL))
749
error = "connection is invalid (POLLNVAL)";
749
-
750
+
751
if(unlikely(error)) {
752
debug(D_STREAM, "STREAM: %s - closing socket...", error);
753
error("STREAM %s [send to %s]: %s - reopening socket - we have sent %zu bytes on this connection.", host->hostname, connected_to, error, sent_bytes_on_this_connection);
web/api/exporters/shell/allmetrics_shell.c
+2
-2
@@ -39,7 +39,7 @@ void rrd_stats_api_v1_charts_allmetrics_shell(RRDHOST *host, BUFFER *wb) {
39
// for each dimension
40
RRDDIM *rd;
41
rrddim_foreach_read(rd, st) {
42
- if(rd->collections_counter) {
42
+ if(rd->collections_counter && !rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE)) {
43
char dimension[SHELL_ELEMENT_MAX + 1];
44
shell_name_copy(dimension, rd->name?rd->name:rd->id, SHELL_ELEMENT_MAX);
45
@@ -126,7 +126,7 @@ void rrd_stats_api_v1_charts_allmetrics_json(RRDHOST *host, BUFFER *wb) {
126
// for each dimension
127
RRDDIM *rd;
128
rrddim_foreach_read(rd, st) {
129
- if(rd->collections_counter) {
129
+ if(rd->collections_counter && !rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE)) {
130
131
buffer_sprintf(wb, "%s\n"
132
"\t\t\t\"%s\": {\n"
web/api/formatters/rrdset2json.c
+1
-1
@@ -51,7 +51,7 @@ void rrdset2json(RRDSET *st, BUFFER *wb, size_t *dimensions_count, size_t *memor
51
size_t dimensions = 0;
52
RRDDIM *rd;
53
rrddim_foreach_read(rd, st) {
54
- if(rrddim_flag_check(rd, RRDDIM_FLAG_HIDDEN)) continue;
54
+ if(rrddim_flag_check(rd, RRDDIM_FLAG_HIDDEN) || rrddim_flag_check(rd, RRDDIM_FLAG_OBSOLETE)) continue;
55
56
memory += rd->memsize;
57