Display uptime for processes (#6654)
* Get process uptime * Calculate target uptime * Update charts * Show collected data * Fix chart names * Update the documentation * Fix a flag value * Add an explanation note for the 'carried over uptime' chart * Move the functions for getting uptime to libnetdata * Rename the function for geting uptime * Remove redundant code * Fix starttime calculation * More accurate definition for the carried over uptime * fix group starttime calculation * Fix typo
Vladimir Kobal committed
Aug 29, 2019 at 20:35 UTC
c79112e85317d80f46acb26296e47a7da8d9d8b0
6 files changed
+195
-68
collectors/apps.plugin/README.md
+5
@@ -47,6 +47,11 @@ Each of these sections provides the same number of charts:
47
- Threads Running
48
- Processes Running
49
- Pipes Open
50
+ - Carried Over Uptime (since the Netdata restart)
51
+ - Minimum Uptime
52
+ - Average Uptime
53
+ - Maximum Uptime
54
+
55
- Swap Memory
56
- Swap Memory Used
57
- Major Page Faults (i.e. swap activity)
collectors/apps.plugin/apps_plugin.c
+104
-2
@@ -260,6 +260,12 @@ struct target {
260
kernel_uint_t openeventpolls;
261
kernel_uint_t openother;
262
263
+ kernel_uint_t starttime;
264
+ kernel_uint_t collected_starttime;
265
+ kernel_uint_t uptime_min;
266
+ kernel_uint_t uptime_sum;
267
+ kernel_uint_t uptime_max;
268
+
269
unsigned int processes; // how many processes have been merged to this
270
int exposed; // if set, we have sent this to netdata
271
int hidden; // if set, we set the hidden flag on the dimension
@@ -345,7 +351,7 @@ struct pid_stat {
351
// int64_t nice;
352
int32_t num_threads;
353
// int64_t itrealvalue;
348
- // kernel_uint_t starttime;
354
+ kernel_uint_t collected_starttime;
355
// kernel_uint_t vsize;
356
// kernel_uint_t rss;
357
// kernel_uint_t rsslim;
@@ -419,6 +425,8 @@ struct pid_stat {
425
usec_t io_collected_usec;
426
usec_t last_io_collected_usec;
427
428
+ kernel_uint_t uptime;
429
+
430
char *fds_dirname; // the full directory name in /proc/PID/fd
431
432
char *stat_filename;
@@ -433,6 +441,8 @@ struct pid_stat {
441
442
size_t pagesize;
443
444
+kernel_uint_t global_uptime;
445
+
446
// log each problem once per process
447
// log flood protection flags (log_thrown)
448
#define PID_LOG_IO 0x00000001
@@ -1421,7 +1431,8 @@ static inline int read_proc_pid_stat(struct pid_stat *p, void *ptr) {
1431
// p->nice = str2kernel_uint_t(procfile_lineword(ff, 0, 18));
1432
p->num_threads = (int32_t)str2uint32_t(procfile_lineword(ff, 0, 19));
1433
// p->itrealvalue = str2kernel_uint_t(procfile_lineword(ff, 0, 20));
1424
- // p->starttime = str2kernel_uint_t(procfile_lineword(ff, 0, 21));
1434
+ p->collected_starttime = str2kernel_uint_t(procfile_lineword(ff, 0, 21)) / system_hz;
1435
+ p->uptime = (global_uptime > p->collected_starttime)?(global_uptime - p->collected_starttime):0;
1436
// p->vsize = str2kernel_uint_t(procfile_lineword(ff, 0, 22));
1437
// p->rss = str2kernel_uint_t(procfile_lineword(ff, 0, 23));
1438
// p->rsslim = str2kernel_uint_t(procfile_lineword(ff, 0, 24));
@@ -1490,6 +1501,8 @@ cleanup:
1501
return 0;
1502
}
1503
1504
+// ----------------------------------------------------------------------------
1505
+
1506
static inline int read_proc_pid_io(struct pid_stat *p, void *ptr) {
1507
(void)ptr;
1508
#ifdef __FreeBSD__
@@ -2634,6 +2647,12 @@ static int collect_data_for_all_processes(void) {
2647
collect_data_for_pid(pid, &procbase[i]);
2648
}
2649
#else
2650
+ static char uptime_filename[FILENAME_MAX + 1] = "";
2651
+ if(*uptime_filename == '\0')
2652
+ snprintfz(uptime_filename, FILENAME_MAX, "%s/proc/uptime", netdata_configured_host_prefix);
2653
+
2654
+ global_uptime = (kernel_uint_t)(uptime_msec(uptime_filename) / MSEC_PER_SEC);
2655
+
2656
char dirname[FILENAME_MAX + 1];
2657
2658
snprintfz(dirname, FILENAME_MAX, "%s/proc", netdata_configured_host_prefix);
@@ -2879,6 +2898,11 @@ static size_t zero_all_targets(struct target *root) {
2898
w->openother = 0;
2899
}
2900
2901
+ w->collected_starttime = 0;
2902
+ w->uptime_min = 0;
2903
+ w->uptime_sum = 0;
2904
+ w->uptime_max = 0;
2905
+
2906
if(unlikely(w->root_pid)) {
2907
struct pid_on_target *pid_on_target_to_free, *pid_on_target = w->root_pid;
2908
@@ -3032,6 +3056,11 @@ static inline void aggregate_pid_on_target(struct target *w, struct pid_stat *p,
3056
w->processes++;
3057
w->num_threads += p->num_threads;
3058
3059
+ if(!w->collected_starttime || p->collected_starttime < w->collected_starttime) w->collected_starttime = p->collected_starttime;
3060
+ if(!w->uptime_min || p->uptime < w->uptime_min) w->uptime_min = p->uptime;
3061
+ w->uptime_sum += p->uptime;
3062
+ if(!w->uptime_max || w->uptime_max < p->uptime) w->uptime_max = p->uptime;
3063
+
3064
if(unlikely(debug_enabled || w->debug_enabled)) {
3065
debug_log_int("aggregating '%s' pid %d on target '%s' utime=" KERNEL_UINT_FORMAT ", stime=" KERNEL_UINT_FORMAT ", gtime=" KERNEL_UINT_FORMAT ", cutime=" KERNEL_UINT_FORMAT ", cstime=" KERNEL_UINT_FORMAT ", cgtime=" KERNEL_UINT_FORMAT ", minflt=" KERNEL_UINT_FORMAT ", majflt=" KERNEL_UINT_FORMAT ", cminflt=" KERNEL_UINT_FORMAT ", cmajflt=" KERNEL_UINT_FORMAT "", p->comm, p->pid, w->name, p->utime, p->stime, p->gtime, p->cutime, p->cstime, p->cgtime, p->minflt, p->majflt, p->cminflt, p->cmajflt);
3066
@@ -3042,6 +3071,19 @@ static inline void aggregate_pid_on_target(struct target *w, struct pid_stat *p,
3071
}
3072
}
3073
3074
+static inline void post_aggregate_targets(struct target *root) {
3075
+ struct target *w;
3076
+ for (w = root; w ; w = w->next) {
3077
+ if(w->collected_starttime) {
3078
+ if (!w->starttime || w->collected_starttime < w->starttime) {
3079
+ w->starttime = w->collected_starttime;
3080
+ }
3081
+ } else {
3082
+ w->starttime = 0;
3083
+ }
3084
+ }
3085
+}
3086
+
3087
static void calculate_netdata_statistics(void) {
3088
3089
apply_apps_groups_targets_inheritance();
@@ -3102,6 +3144,10 @@ static void calculate_netdata_statistics(void) {
3144
aggregate_pid_fds_on_targets(p);
3145
}
3146
3147
+ post_aggregate_targets(apps_groups_root_target);
3148
+ post_aggregate_targets(users_root_target);
3149
+ post_aggregate_targets(groups_root_target);
3150
+
3151
cleanup_exited_pids();
3152
}
3153
@@ -3457,6 +3503,36 @@ static void send_collected_data_to_netdata(struct target *root, const char *type
3503
}
3504
send_END();
3505
3506
+#ifndef __FreeBSD__
3507
+ send_BEGIN(type, "uptime", dt);
3508
+ for (w = root; w ; w = w->next) {
3509
+ if(unlikely(w->exposed && w->processes))
3510
+ send_SET(w->name, (global_uptime > w->starttime)?(global_uptime - w->starttime):0);
3511
+ }
3512
+ send_END();
3513
+
3514
+ send_BEGIN(type, "uptime_min", dt);
3515
+ for (w = root; w ; w = w->next) {
3516
+ if(unlikely(w->exposed && w->processes))
3517
+ send_SET(w->name, w->uptime_min);
3518
+ }
3519
+ send_END();
3520
+
3521
+ send_BEGIN(type, "uptime_avg", dt);
3522
+ for (w = root; w ; w = w->next) {
3523
+ if(unlikely(w->exposed && w->processes))
3524
+ send_SET(w->name, w->processes?(w->uptime_sum / w->processes):0);
3525
+ }
3526
+ send_END();
3527
+
3528
+ send_BEGIN(type, "uptime_max", dt);
3529
+ for (w = root; w ; w = w->next) {
3530
+ if(unlikely(w->exposed && w->processes))
3531
+ send_SET(w->name, w->uptime_max);
3532
+ }
3533
+ send_END();
3534
+#endif
3535
+
3536
send_BEGIN(type, "mem", dt);
3537
for (w = root; w ; w = w->next) {
3538
if(unlikely(w->exposed && w->processes))
@@ -3615,6 +3691,32 @@ static void send_charts_updates_to_netdata(struct target *root, const char *type
3691
fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3692
}
3693
3694
+#ifndef __FreeBSD__
3695
+ fprintf(stdout, "CHART %s.uptime '' '%s Carried Over Uptime' 'seconds' processes %s.uptime line 20008 %d\n", type, title, type, update_every);
3696
+ for (w = root; w ; w = w->next) {
3697
+ if(unlikely(w->exposed))
3698
+ fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3699
+ }
3700
+
3701
+ fprintf(stdout, "CHART %s.uptime_min '' '%s Minimum Uptime' 'seconds' processes %s.uptime_min line 20009 %d\n", type, title, type, update_every);
3702
+ for (w = root; w ; w = w->next) {
3703
+ if(unlikely(w->exposed))
3704
+ fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3705
+ }
3706
+
3707
+ fprintf(stdout, "CHART %s.uptime_avg '' '%s Average Uptime' 'seconds' processes %s.uptime_avg line 20010 %d\n", type, title, type, update_every);
3708
+ for (w = root; w ; w = w->next) {
3709
+ if(unlikely(w->exposed))
3710
+ fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3711
+ }
3712
+
3713
+ fprintf(stdout, "CHART %s.uptime_max '' '%s Maximum Uptime' 'seconds' processes %s.uptime_max line 20011 %d\n", type, title, type, update_every);
3714
+ for (w = root; w ; w = w->next) {
3715
+ if(unlikely(w->exposed))
3716
+ fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3717
+ }
3718
+#endif
3719
+
3720
fprintf(stdout, "CHART %s.cpu_user '' '%s CPU User Time (%d%% = %d core%s)' 'percentage' cpu %s.cpu_user stacked 20020 %d\n", type, title, (processors * 100), processors, (processors>1)?"s":"", type, update_every);
3721
for (w = root; w ; w = w->next) {
3722
if(unlikely(w->exposed))
collectors/proc.plugin/proc_uptime.c
+6
-65
@@ -2,76 +2,17 @@
2
3
#include "plugin_proc.h"
4
5
-static inline collected_number uptime_from_boottime(void) {
6
-#ifdef CLOCK_BOOTTIME_IS_AVAILABLE
7
- return now_boottime_usec() / 1000;
8
-#else
9
- error("uptime cannot be read from CLOCK_BOOTTIME on this system.");
10
- return 0;
11
-#endif
12
-}
13
-
14
-static procfile *read_proc_uptime_ff = NULL;
15
-static inline collected_number read_proc_uptime(void) {
16
- if(unlikely(!read_proc_uptime_ff)) {
17
- char filename[FILENAME_MAX + 1];
18
- snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/proc/uptime");
19
-
20
- read_proc_uptime_ff = procfile_open(config_get("plugin:proc:/proc/uptime", "filename to monitor", filename), " \t", PROCFILE_FLAG_DEFAULT);
21
- if(unlikely(!read_proc_uptime_ff)) return 0;
22
- }
23
-
24
- read_proc_uptime_ff = procfile_readall(read_proc_uptime_ff);
25
- if(unlikely(!read_proc_uptime_ff)) return 0;
26
-
27
- if(unlikely(procfile_lines(read_proc_uptime_ff) < 1)) {
28
- error("/proc/uptime has no lines.");
29
- return 0;
30
- }
31
- if(unlikely(procfile_linewords(read_proc_uptime_ff, 0) < 1)) {
32
- error("/proc/uptime has less than 1 word in it.");
33
- return 0;
34
- }
35
-
36
- return (collected_number)(strtold(procfile_lineword(read_proc_uptime_ff, 0, 0), NULL) * 1000.0);
37
-}
38
-
5
int do_proc_uptime(int update_every, usec_t dt) {
6
(void)dt;
7
42
- static int use_boottime = -1;
43
-
44
- if(unlikely(use_boottime == -1)) {
45
- collected_number uptime_boottime = uptime_from_boottime();
46
- collected_number uptime_proc = read_proc_uptime();
47
-
48
- long long delta = (long long)uptime_boottime - (long long)uptime_proc;
49
- if(delta < 0) delta = -delta;
8
+ static char *uptime_filename = NULL;
9
+ if(!uptime_filename) {
10
+ char filename[FILENAME_MAX + 1];
11
+ snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/proc/uptime");
12
51
- if(delta <= 1000 && uptime_boottime != 0) {
52
- procfile_close(read_proc_uptime_ff);
53
- info("Using now_boottime_usec() for uptime (dt is %lld ms)", delta);
54
- use_boottime = 1;
55
- }
56
- else if(uptime_proc != 0) {
57
- info("Using /proc/uptime for uptime (dt is %lld ms)", delta);
58
- use_boottime = 0;
59
- }
60
- else {
61
- error("Cannot find any way to read uptime on this system.");
62
- return 1;
63
- }
13
+ uptime_filename = config_get("plugin:proc:/proc/uptime", "filename to monitor", filename);
14
}
15
66
- collected_number uptime;
67
- if(use_boottime)
68
- uptime = uptime_from_boottime();
69
- else
70
- uptime = read_proc_uptime();
71
-
72
-
73
- // --------------------------------------------------------------------
74
-
16
static RRDSET *st = NULL;
17
static RRDDIM *rd = NULL;
18
@@ -97,7 +38,7 @@ int do_proc_uptime(int update_every, usec_t dt) {
38
else
39
rrdset_next(st);
40
100
- rrddim_set_by_pointer(st, rd, uptime);
41
+ rrddim_set_by_pointer(st, rd, uptime_msec(uptime_filename));
42
43
rrdset_done(st);
44
libnetdata/clocks/clocks.c
+65
@@ -210,3 +210,68 @@ int sleep_usec(usec_t usec) {
210
return ret;
211
#endif
212
}
213
+
214
+static inline collected_number uptime_from_boottime(void) {
215
+#ifdef CLOCK_BOOTTIME_IS_AVAILABLE
216
+ return now_boottime_usec() / 1000;
217
+#else
218
+ error("uptime cannot be read from CLOCK_BOOTTIME on this system.");
219
+ return 0;
220
+#endif
221
+}
222
+
223
+static procfile *read_proc_uptime_ff = NULL;
224
+static inline collected_number read_proc_uptime(char *filename) {
225
+ if(unlikely(!read_proc_uptime_ff)) {
226
+ read_proc_uptime_ff = procfile_open(filename, " \t", PROCFILE_FLAG_DEFAULT);
227
+ if(unlikely(!read_proc_uptime_ff)) return 0;
228
+ }
229
+
230
+ read_proc_uptime_ff = procfile_readall(read_proc_uptime_ff);
231
+ if(unlikely(!read_proc_uptime_ff)) return 0;
232
+
233
+ if(unlikely(procfile_lines(read_proc_uptime_ff) < 1)) {
234
+ error("/proc/uptime has no lines.");
235
+ return 0;
236
+ }
237
+ if(unlikely(procfile_linewords(read_proc_uptime_ff, 0) < 1)) {
238
+ error("/proc/uptime has less than 1 word in it.");
239
+ return 0;
240
+ }
241
+
242
+ return (collected_number)(strtold(procfile_lineword(read_proc_uptime_ff, 0, 0), NULL) * 1000.0);
243
+}
244
+
245
+inline collected_number uptime_msec(char *filename){
246
+ static int use_boottime = -1;
247
+
248
+ if(unlikely(use_boottime == -1)) {
249
+ collected_number uptime_boottime = uptime_from_boottime();
250
+ collected_number uptime_proc = read_proc_uptime(filename);
251
+
252
+ long long delta = (long long)uptime_boottime - (long long)uptime_proc;
253
+ if(delta < 0) delta = -delta;
254
+
255
+ if(delta <= 1000 && uptime_boottime != 0) {
256
+ procfile_close(read_proc_uptime_ff);
257
+ info("Using now_boottime_usec() for uptime (dt is %lld ms)", delta);
258
+ use_boottime = 1;
259
+ }
260
+ else if(uptime_proc != 0) {
261
+ info("Using /proc/uptime for uptime (dt is %lld ms)", delta);
262
+ use_boottime = 0;
263
+ }
264
+ else {
265
+ error("Cannot find any way to read uptime on this system.");
266
+ return 1;
267
+ }
268
+ }
269
+
270
+ collected_number uptime;
271
+ if(use_boottime)
272
+ uptime = uptime_from_boottime();
273
+ else
274
+ uptime = read_proc_uptime(filename);
275
+
276
+ return uptime;
277
+}
libnetdata/clocks/clocks.h
+2
@@ -136,4 +136,6 @@ extern int sleep_usec(usec_t usec);
136
*/
137
void test_clock_boottime(void);
138
139
+extern collected_number uptime_msec(char *filename);
140
+
141
#endif /* NETDATA_CLOCKS_H */
web/gui/dashboard_info.js
+13
-1
@@ -985,6 +985,10 @@ netdataDashboard.context = {
985
height: 2.0
986
},
987
988
+ 'apps.uptime': {
989
+ info: 'Carried over process group uptime since the Netdata restart. The period of time within which at least one process in the group was running.'
990
+ },
991
+
992
// ------------------------------------------------------------------------
993
// USERS
994
@@ -1008,6 +1012,10 @@ netdataDashboard.context = {
1012
height: 2.0
1013
},
1014
1015
+ 'users.uptime': {
1016
+ info: 'Carried over process group uptime since the Netdata restart. The period of time within which at least one process in the group was running.'
1017
+ },
1018
+
1019
// ------------------------------------------------------------------------
1020
// GROUPS
1021
@@ -1020,7 +1028,7 @@ netdataDashboard.context = {
1028
},
1029
1030
'groups.vmem': {
1023
- info: 'Virtual memory allocated per user group. Please check <a href="https://github.com/netdata/netdata/tree/master/daemon#virtual-memory" target="_blank">this article</a> for more information.'
1031
+ info: 'Virtual memory allocated per user group since the Netdata restart. Please check <a href="https://github.com/netdata/netdata/tree/master/daemon#virtual-memory" target="_blank">this article</a> for more information.'
1032
},
1033
1034
'groups.preads': {
@@ -1031,6 +1039,10 @@ netdataDashboard.context = {
1039
height: 2.0
1040
},
1041
1042
+ 'groups.uptime': {
1043
+ info: 'Carried over process group uptime. The period of time within which at least one process in the group was running.'
1044
+ },
1045
+
1046
// ------------------------------------------------------------------------
1047
// NETWORK QoS
1048