@cryptotaxi247 / netdata-1 / commits / c79112e85

Display uptime for processes (#6654)

* Get process uptime * Calculate target uptime * Update charts * Show collected data * Fix chart names * Update the documentation * Fix a flag value * Add an explanation note for the 'carried over uptime' chart * Move the functions for getting uptime to libnetdata * Rename the function for geting uptime * Remove redundant code * Fix starttime calculation * More accurate definition for the carried over uptime * fix group starttime calculation * Fix typo

Vladimir Kobal committed Aug 29, 2019 at 20:35 UTC c79112e85317d80f46acb26296e47a7da8d9d8b0
6 files changed +195 -68
collectors/apps.plugin/README.md
+5
@@ -47,6 +47,11 @@ Each of these sections provides the same number of charts:
47 - Threads Running
48 - Processes Running
49 - Pipes Open
50 + - Carried Over Uptime (since the Netdata restart)
51 + - Minimum Uptime
52 + - Average Uptime
53 + - Maximum Uptime
54 +
55 - Swap Memory
56 - Swap Memory Used
57 - Major Page Faults (i.e. swap activity)
collectors/apps.plugin/apps_plugin.c
+104 -2
@@ -260,6 +260,12 @@ struct target {
260 kernel_uint_t openeventpolls;
261 kernel_uint_t openother;
262
263 + kernel_uint_t starttime;
264 + kernel_uint_t collected_starttime;
265 + kernel_uint_t uptime_min;
266 + kernel_uint_t uptime_sum;
267 + kernel_uint_t uptime_max;
268 +
269 unsigned int processes; // how many processes have been merged to this
270 int exposed; // if set, we have sent this to netdata
271 int hidden; // if set, we set the hidden flag on the dimension
@@ -345,7 +351,7 @@ struct pid_stat {
351 // int64_t nice;
352 int32_t num_threads;
353 // int64_t itrealvalue;
348 - // kernel_uint_t starttime;
354 + kernel_uint_t collected_starttime;
355 // kernel_uint_t vsize;
356 // kernel_uint_t rss;
357 // kernel_uint_t rsslim;
@@ -419,6 +425,8 @@ struct pid_stat {
425 usec_t io_collected_usec;
426 usec_t last_io_collected_usec;
427
428 + kernel_uint_t uptime;
429 +
430 char *fds_dirname; // the full directory name in /proc/PID/fd
431
432 char *stat_filename;
@@ -433,6 +441,8 @@ struct pid_stat {
441
442 size_t pagesize;
443
444 +kernel_uint_t global_uptime;
445 +
446 // log each problem once per process
447 // log flood protection flags (log_thrown)
448 #define PID_LOG_IO 0x00000001
@@ -1421,7 +1431,8 @@ static inline int read_proc_pid_stat(struct pid_stat *p, void *ptr) {
1431 // p->nice = str2kernel_uint_t(procfile_lineword(ff, 0, 18));
1432 p->num_threads = (int32_t)str2uint32_t(procfile_lineword(ff, 0, 19));
1433 // p->itrealvalue = str2kernel_uint_t(procfile_lineword(ff, 0, 20));
1424 - // p->starttime = str2kernel_uint_t(procfile_lineword(ff, 0, 21));
1434 + p->collected_starttime = str2kernel_uint_t(procfile_lineword(ff, 0, 21)) / system_hz;
1435 + p->uptime = (global_uptime > p->collected_starttime)?(global_uptime - p->collected_starttime):0;
1436 // p->vsize = str2kernel_uint_t(procfile_lineword(ff, 0, 22));
1437 // p->rss = str2kernel_uint_t(procfile_lineword(ff, 0, 23));
1438 // p->rsslim = str2kernel_uint_t(procfile_lineword(ff, 0, 24));
@@ -1490,6 +1501,8 @@ cleanup:
1501 return 0;
1502 }
1503
1504 +// ----------------------------------------------------------------------------
1505 +
1506 static inline int read_proc_pid_io(struct pid_stat *p, void *ptr) {
1507 (void)ptr;
1508 #ifdef __FreeBSD__
@@ -2634,6 +2647,12 @@ static int collect_data_for_all_processes(void) {
2647 collect_data_for_pid(pid, &procbase[i]);
2648 }
2649 #else
2650 + static char uptime_filename[FILENAME_MAX + 1] = "";
2651 + if(*uptime_filename == '\0')
2652 + snprintfz(uptime_filename, FILENAME_MAX, "%s/proc/uptime", netdata_configured_host_prefix);
2653 +
2654 + global_uptime = (kernel_uint_t)(uptime_msec(uptime_filename) / MSEC_PER_SEC);
2655 +
2656 char dirname[FILENAME_MAX + 1];
2657
2658 snprintfz(dirname, FILENAME_MAX, "%s/proc", netdata_configured_host_prefix);
@@ -2879,6 +2898,11 @@ static size_t zero_all_targets(struct target *root) {
2898 w->openother = 0;
2899 }
2900
2901 + w->collected_starttime = 0;
2902 + w->uptime_min = 0;
2903 + w->uptime_sum = 0;
2904 + w->uptime_max = 0;
2905 +
2906 if(unlikely(w->root_pid)) {
2907 struct pid_on_target *pid_on_target_to_free, *pid_on_target = w->root_pid;
2908
@@ -3032,6 +3056,11 @@ static inline void aggregate_pid_on_target(struct target *w, struct pid_stat *p,
3056 w->processes++;
3057 w->num_threads += p->num_threads;
3058
3059 + if(!w->collected_starttime || p->collected_starttime < w->collected_starttime) w->collected_starttime = p->collected_starttime;
3060 + if(!w->uptime_min || p->uptime < w->uptime_min) w->uptime_min = p->uptime;
3061 + w->uptime_sum += p->uptime;
3062 + if(!w->uptime_max || w->uptime_max < p->uptime) w->uptime_max = p->uptime;
3063 +
3064 if(unlikely(debug_enabled || w->debug_enabled)) {
3065 debug_log_int("aggregating '%s' pid %d on target '%s' utime=" KERNEL_UINT_FORMAT ", stime=" KERNEL_UINT_FORMAT ", gtime=" KERNEL_UINT_FORMAT ", cutime=" KERNEL_UINT_FORMAT ", cstime=" KERNEL_UINT_FORMAT ", cgtime=" KERNEL_UINT_FORMAT ", minflt=" KERNEL_UINT_FORMAT ", majflt=" KERNEL_UINT_FORMAT ", cminflt=" KERNEL_UINT_FORMAT ", cmajflt=" KERNEL_UINT_FORMAT "", p->comm, p->pid, w->name, p->utime, p->stime, p->gtime, p->cutime, p->cstime, p->cgtime, p->minflt, p->majflt, p->cminflt, p->cmajflt);
3066
@@ -3042,6 +3071,19 @@ static inline void aggregate_pid_on_target(struct target *w, struct pid_stat *p,
3071 }
3072 }
3073
3074 +static inline void post_aggregate_targets(struct target *root) {
3075 + struct target *w;
3076 + for (w = root; w ; w = w->next) {
3077 + if(w->collected_starttime) {
3078 + if (!w->starttime || w->collected_starttime < w->starttime) {
3079 + w->starttime = w->collected_starttime;
3080 + }
3081 + } else {
3082 + w->starttime = 0;
3083 + }
3084 + }
3085 +}
3086 +
3087 static void calculate_netdata_statistics(void) {
3088
3089 apply_apps_groups_targets_inheritance();
@@ -3102,6 +3144,10 @@ static void calculate_netdata_statistics(void) {
3144 aggregate_pid_fds_on_targets(p);
3145 }
3146
3147 + post_aggregate_targets(apps_groups_root_target);
3148 + post_aggregate_targets(users_root_target);
3149 + post_aggregate_targets(groups_root_target);
3150 +
3151 cleanup_exited_pids();
3152 }
3153
@@ -3457,6 +3503,36 @@ static void send_collected_data_to_netdata(struct target *root, const char *type
3503 }
3504 send_END();
3505
3506 +#ifndef __FreeBSD__
3507 + send_BEGIN(type, "uptime", dt);
3508 + for (w = root; w ; w = w->next) {
3509 + if(unlikely(w->exposed && w->processes))
3510 + send_SET(w->name, (global_uptime > w->starttime)?(global_uptime - w->starttime):0);
3511 + }
3512 + send_END();
3513 +
3514 + send_BEGIN(type, "uptime_min", dt);
3515 + for (w = root; w ; w = w->next) {
3516 + if(unlikely(w->exposed && w->processes))
3517 + send_SET(w->name, w->uptime_min);
3518 + }
3519 + send_END();
3520 +
3521 + send_BEGIN(type, "uptime_avg", dt);
3522 + for (w = root; w ; w = w->next) {
3523 + if(unlikely(w->exposed && w->processes))
3524 + send_SET(w->name, w->processes?(w->uptime_sum / w->processes):0);
3525 + }
3526 + send_END();
3527 +
3528 + send_BEGIN(type, "uptime_max", dt);
3529 + for (w = root; w ; w = w->next) {
3530 + if(unlikely(w->exposed && w->processes))
3531 + send_SET(w->name, w->uptime_max);
3532 + }
3533 + send_END();
3534 +#endif
3535 +
3536 send_BEGIN(type, "mem", dt);
3537 for (w = root; w ; w = w->next) {
3538 if(unlikely(w->exposed && w->processes))
@@ -3615,6 +3691,32 @@ static void send_charts_updates_to_netdata(struct target *root, const char *type
3691 fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3692 }
3693
3694 +#ifndef __FreeBSD__
3695 + fprintf(stdout, "CHART %s.uptime '' '%s Carried Over Uptime' 'seconds' processes %s.uptime line 20008 %d\n", type, title, type, update_every);
3696 + for (w = root; w ; w = w->next) {
3697 + if(unlikely(w->exposed))
3698 + fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3699 + }
3700 +
3701 + fprintf(stdout, "CHART %s.uptime_min '' '%s Minimum Uptime' 'seconds' processes %s.uptime_min line 20009 %d\n", type, title, type, update_every);
3702 + for (w = root; w ; w = w->next) {
3703 + if(unlikely(w->exposed))
3704 + fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3705 + }
3706 +
3707 + fprintf(stdout, "CHART %s.uptime_avg '' '%s Average Uptime' 'seconds' processes %s.uptime_avg line 20010 %d\n", type, title, type, update_every);
3708 + for (w = root; w ; w = w->next) {
3709 + if(unlikely(w->exposed))
3710 + fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3711 + }
3712 +
3713 + fprintf(stdout, "CHART %s.uptime_max '' '%s Maximum Uptime' 'seconds' processes %s.uptime_max line 20011 %d\n", type, title, type, update_every);
3714 + for (w = root; w ; w = w->next) {
3715 + if(unlikely(w->exposed))
3716 + fprintf(stdout, "DIMENSION %s '' absolute 1 1\n", w->name);
3717 + }
3718 +#endif
3719 +
3720 fprintf(stdout, "CHART %s.cpu_user '' '%s CPU User Time (%d%% = %d core%s)' 'percentage' cpu %s.cpu_user stacked 20020 %d\n", type, title, (processors * 100), processors, (processors>1)?"s":"", type, update_every);
3721 for (w = root; w ; w = w->next) {
3722 if(unlikely(w->exposed))
collectors/proc.plugin/proc_uptime.c
+6 -65
@@ -2,76 +2,17 @@
2
3 #include "plugin_proc.h"
4
5 -static inline collected_number uptime_from_boottime(void) {
6 -#ifdef CLOCK_BOOTTIME_IS_AVAILABLE
7 - return now_boottime_usec() / 1000;
8 -#else
9 - error("uptime cannot be read from CLOCK_BOOTTIME on this system.");
10 - return 0;
11 -#endif
12 -}
13 -
14 -static procfile *read_proc_uptime_ff = NULL;
15 -static inline collected_number read_proc_uptime(void) {
16 - if(unlikely(!read_proc_uptime_ff)) {
17 - char filename[FILENAME_MAX + 1];
18 - snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/proc/uptime");
19 -
20 - read_proc_uptime_ff = procfile_open(config_get("plugin:proc:/proc/uptime", "filename to monitor", filename), " \t", PROCFILE_FLAG_DEFAULT);
21 - if(unlikely(!read_proc_uptime_ff)) return 0;
22 - }
23 -
24 - read_proc_uptime_ff = procfile_readall(read_proc_uptime_ff);
25 - if(unlikely(!read_proc_uptime_ff)) return 0;
26 -
27 - if(unlikely(procfile_lines(read_proc_uptime_ff) < 1)) {
28 - error("/proc/uptime has no lines.");
29 - return 0;
30 - }
31 - if(unlikely(procfile_linewords(read_proc_uptime_ff, 0) < 1)) {
32 - error("/proc/uptime has less than 1 word in it.");
33 - return 0;
34 - }
35 -
36 - return (collected_number)(strtold(procfile_lineword(read_proc_uptime_ff, 0, 0), NULL) * 1000.0);
37 -}
38 -
5 int do_proc_uptime(int update_every, usec_t dt) {
6 (void)dt;
7
42 - static int use_boottime = -1;
43 -
44 - if(unlikely(use_boottime == -1)) {
45 - collected_number uptime_boottime = uptime_from_boottime();
46 - collected_number uptime_proc = read_proc_uptime();
47 -
48 - long long delta = (long long)uptime_boottime - (long long)uptime_proc;
49 - if(delta < 0) delta = -delta;
8 + static char *uptime_filename = NULL;
9 + if(!uptime_filename) {
10 + char filename[FILENAME_MAX + 1];
11 + snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/proc/uptime");
12
51 - if(delta <= 1000 && uptime_boottime != 0) {
52 - procfile_close(read_proc_uptime_ff);
53 - info("Using now_boottime_usec() for uptime (dt is %lld ms)", delta);
54 - use_boottime = 1;
55 - }
56 - else if(uptime_proc != 0) {
57 - info("Using /proc/uptime for uptime (dt is %lld ms)", delta);
58 - use_boottime = 0;
59 - }
60 - else {
61 - error("Cannot find any way to read uptime on this system.");
62 - return 1;
63 - }
13 + uptime_filename = config_get("plugin:proc:/proc/uptime", "filename to monitor", filename);
14 }
15
66 - collected_number uptime;
67 - if(use_boottime)
68 - uptime = uptime_from_boottime();
69 - else
70 - uptime = read_proc_uptime();
71 -
72 -
73 - // --------------------------------------------------------------------
74 -
16 static RRDSET *st = NULL;
17 static RRDDIM *rd = NULL;
18
@@ -97,7 +38,7 @@ int do_proc_uptime(int update_every, usec_t dt) {
38 else
39 rrdset_next(st);
40
100 - rrddim_set_by_pointer(st, rd, uptime);
41 + rrddim_set_by_pointer(st, rd, uptime_msec(uptime_filename));
42
43 rrdset_done(st);
44
libnetdata/clocks/clocks.c
+65
@@ -210,3 +210,68 @@ int sleep_usec(usec_t usec) {
210 return ret;
211 #endif
212 }
213 +
214 +static inline collected_number uptime_from_boottime(void) {
215 +#ifdef CLOCK_BOOTTIME_IS_AVAILABLE
216 + return now_boottime_usec() / 1000;
217 +#else
218 + error("uptime cannot be read from CLOCK_BOOTTIME on this system.");
219 + return 0;
220 +#endif
221 +}
222 +
223 +static procfile *read_proc_uptime_ff = NULL;
224 +static inline collected_number read_proc_uptime(char *filename) {
225 + if(unlikely(!read_proc_uptime_ff)) {
226 + read_proc_uptime_ff = procfile_open(filename, " \t", PROCFILE_FLAG_DEFAULT);
227 + if(unlikely(!read_proc_uptime_ff)) return 0;
228 + }
229 +
230 + read_proc_uptime_ff = procfile_readall(read_proc_uptime_ff);
231 + if(unlikely(!read_proc_uptime_ff)) return 0;
232 +
233 + if(unlikely(procfile_lines(read_proc_uptime_ff) < 1)) {
234 + error("/proc/uptime has no lines.");
235 + return 0;
236 + }
237 + if(unlikely(procfile_linewords(read_proc_uptime_ff, 0) < 1)) {
238 + error("/proc/uptime has less than 1 word in it.");
239 + return 0;
240 + }
241 +
242 + return (collected_number)(strtold(procfile_lineword(read_proc_uptime_ff, 0, 0), NULL) * 1000.0);
243 +}
244 +
245 +inline collected_number uptime_msec(char *filename){
246 + static int use_boottime = -1;
247 +
248 + if(unlikely(use_boottime == -1)) {
249 + collected_number uptime_boottime = uptime_from_boottime();
250 + collected_number uptime_proc = read_proc_uptime(filename);
251 +
252 + long long delta = (long long)uptime_boottime - (long long)uptime_proc;
253 + if(delta < 0) delta = -delta;
254 +
255 + if(delta <= 1000 && uptime_boottime != 0) {
256 + procfile_close(read_proc_uptime_ff);
257 + info("Using now_boottime_usec() for uptime (dt is %lld ms)", delta);
258 + use_boottime = 1;
259 + }
260 + else if(uptime_proc != 0) {
261 + info("Using /proc/uptime for uptime (dt is %lld ms)", delta);
262 + use_boottime = 0;
263 + }
264 + else {
265 + error("Cannot find any way to read uptime on this system.");
266 + return 1;
267 + }
268 + }
269 +
270 + collected_number uptime;
271 + if(use_boottime)
272 + uptime = uptime_from_boottime();
273 + else
274 + uptime = read_proc_uptime(filename);
275 +
276 + return uptime;
277 +}
libnetdata/clocks/clocks.h
+2
@@ -136,4 +136,6 @@ extern int sleep_usec(usec_t usec);
136 */
137 void test_clock_boottime(void);
138
139 +extern collected_number uptime_msec(char *filename);
140 +
141 #endif /* NETDATA_CLOCKS_H */
web/gui/dashboard_info.js
+13 -1
@@ -985,6 +985,10 @@ netdataDashboard.context = {
985 height: 2.0
986 },
987
988 + 'apps.uptime': {
989 + info: 'Carried over process group uptime since the Netdata restart. The period of time within which at least one process in the group was running.'
990 + },
991 +
992 // ------------------------------------------------------------------------
993 // USERS
994
@@ -1008,6 +1012,10 @@ netdataDashboard.context = {
1012 height: 2.0
1013 },
1014
1015 + 'users.uptime': {
1016 + info: 'Carried over process group uptime since the Netdata restart. The period of time within which at least one process in the group was running.'
1017 + },
1018 +
1019 // ------------------------------------------------------------------------
1020 // GROUPS
1021
@@ -1020,7 +1028,7 @@ netdataDashboard.context = {
1028 },
1029
1030 'groups.vmem': {
1023 - info: 'Virtual memory allocated per user group. Please check <a href="https://github.com/netdata/netdata/tree/master/daemon#virtual-memory" target="_blank">this article</a> for more information.'
1031 + info: 'Virtual memory allocated per user group since the Netdata restart. Please check <a href="https://github.com/netdata/netdata/tree/master/daemon#virtual-memory" target="_blank">this article</a> for more information.'
1032 },
1033
1034 'groups.preads': {
@@ -1031,6 +1039,10 @@ netdataDashboard.context = {
1039 height: 2.0
1040 },
1041
1042 + 'groups.uptime': {
1043 + info: 'Carried over process group uptime. The period of time within which at least one process in the group was running.'
1044 + },
1045 +
1046 // ------------------------------------------------------------------------
1047 // NETWORK QoS
1048