@cryptotaxi247 / netdata-1 / commits / 25dc3d83e

Move cpuidle python module to proc plugin (#4635)

* Add cpuidle charts * Add cpuidle states * Minor cleanup * Add wake cpu threads * Minor fixes, disable python plugin * Fix memory leaks * Separate config option for keeping cpuidle files open * Close cpuidle name files after rescan

Vladimir Kobal committed Dec 4, 2018 at 00:21 UTC 25dc3d83e9ccfc2dd000e2a08a84be94d684a498
3 files changed +323 -6
collectors/all.h
+1
@@ -69,6 +69,7 @@
69 #define NETDATA_CHART_PRIO_CPU_PER_CORE 1000 // +1 per core
70 #define NETDATA_CHART_PRIO_CPU_TEMPERATURE 1050 // freebsd only
71 #define NETDATA_CHART_PRIO_CPUFREQ_SCALING_CUR_FREQ 5003 // freebsd only
72 +#define NETDATA_CHART_PRIO_CPUIDLE 6000
73
74 #define NETDATA_CHART_PRIO_CORE_THROTTLING 5001
75 #define NETDATA_CHART_PRIO_PACKAGE_THROTTLING 5002
collectors/proc.plugin/proc_stat.c
+321 -5
@@ -52,6 +52,7 @@ struct cpu_chart {
52 };
53
54 static int keep_per_core_fds_open = CONFIG_BOOLEAN_YES;
55 +static int keep_cpuidle_fds_open = CONFIG_BOOLEAN_YES;
56
57 static int read_per_core_files(struct cpu_chart *all_cpu_charts, size_t len, size_t index) {
58 char buf[50 + 1];
@@ -161,7 +162,7 @@ static int read_per_core_time_in_state_files(struct cpu_chart *all_cpu_charts, s
162 // the whole period under schedutil governor?
163 // freez(tsf->last_ticks);
164 // tsf->last_ticks = NULL;
164 - // tsf->last_ticks_len = 0;
165 + // tsf->last_ticks_len = 0;
166 continue;
167 }
168
@@ -237,15 +238,236 @@ static void chart_per_core_files(struct cpu_chart *all_cpu_charts, size_t len, s
238 }
239 }
240
241 +struct cpuidle_state {
242 + char *name;
243 +
244 + char *time_filename;
245 + int time_fd;
246 +
247 + collected_number value;
248 +
249 + RRDDIM *rd;
250 +};
251 +
252 +struct per_core_cpuidle_chart {
253 + RRDSET *st;
254 +
255 + RRDDIM *active_time_rd;
256 + collected_number active_time;
257 + collected_number last_active_time;
258 +
259 + struct cpuidle_state *cpuidle_state;
260 + size_t cpuidle_state_len;
261 + int rescan_cpu_states;
262 +};
263 +
264 +static void* wake_cpu_thread(void* core) {
265 + pthread_t thread;
266 + cpu_set_t cpu_set;
267 + static size_t cpu_wakeups = 0;
268 +
269 + CPU_ZERO(&cpu_set);
270 + CPU_SET(*(int*)core, &cpu_set);
271 +
272 + thread = pthread_self();
273 + if(unlikely(pthread_setaffinity_np(thread, sizeof(cpu_set_t), &cpu_set)))
274 + error("Cannot set CPU affinity");
275 +
276 + // Make the CPU core do something
277 + cpu_wakeups++;
278 +
279 + return 0;
280 +}
281 +
282 +static int read_schedstat(char* schedstat_filename, struct per_core_cpuidle_chart **cpuidle_charts_address, size_t cores_found) {
283 + static size_t cpuidle_charts_len = 0;
284 + static procfile *ff = NULL;
285 + struct per_core_cpuidle_chart *cpuidle_charts = *cpuidle_charts_address;
286 +
287 + if(unlikely(!ff)) {
288 + ff = procfile_open(schedstat_filename, " \t:", PROCFILE_FLAG_DEFAULT);
289 + if(unlikely(!ff)) return 1;
290 + }
291 +
292 + ff = procfile_readall(ff);
293 + if(unlikely(!ff)) return 1;
294 +
295 + size_t lines = procfile_lines(ff), l;
296 + size_t words;
297 +
298 + if(unlikely(cpuidle_charts_len < cores_found)) {
299 + cpuidle_charts = reallocz(cpuidle_charts, sizeof(struct per_core_cpuidle_chart) * cores_found);
300 + *cpuidle_charts_address = cpuidle_charts;
301 + memset(cpuidle_charts + cpuidle_charts_len, 0, sizeof(struct per_core_cpuidle_chart) * (cores_found - cpuidle_charts_len));
302 + cpuidle_charts_len = cores_found;
303 + }
304 +
305 + for(l = 0; l < lines ;l++) {
306 + char *row_key = procfile_lineword(ff, l, 0);
307 +
308 + // faster strncmp(row_key, "cpu", 3) == 0
309 + if(likely(row_key[0] == 'c' && row_key[1] == 'p' && row_key[2] == 'u')) {
310 + words = procfile_linewords(ff, l);
311 + if(unlikely(words < 10)) {
312 + error("Cannot read /proc/schedstat cpu line. Expected 9 params, read %zu.", words);
313 + return 1;
314 + }
315 +
316 + size_t core = str2ul(&row_key[3]);
317 + if(unlikely(core >= cores_found)) {
318 + error("Core %zu found but no more than %zu cores were expected.", core, cores_found);
319 + return 1;
320 + }
321 + cpuidle_charts[core].active_time = str2ull(procfile_lineword(ff, l, 7)) / 1000;
322 + }
323 + }
324 +
325 + return 0;
326 +}
327 +
328 +static int read_one_state(char *buf, const char *filename, int *fd) {
329 + ssize_t ret = read(*fd, buf, 50);
330 +
331 + if(unlikely(ret <= 0)) {
332 + // cannot read that file
333 + error("Cannot read file '%s'", filename);
334 + close(*fd);
335 + *fd = -1;
336 + return 0;
337 + }
338 + else {
339 + // successful read
340 +
341 + // terminate the buffer
342 + buf[ret - 1] = '\0';
343 +
344 + if(unlikely(keep_cpuidle_fds_open != CONFIG_BOOLEAN_YES)) {
345 + close(*fd);
346 + *fd = -1;
347 + }
348 + else if(lseek(*fd, 0, SEEK_SET) == -1) {
349 + error("Cannot seek in file '%s'", filename);
350 + close(*fd);
351 + *fd = -1;
352 + }
353 + }
354 +
355 + return 1;
356 +}
357 +
358 +static int read_cpuidle_states(char *cpuidle_name_filename , char *cpuidle_time_filename, struct per_core_cpuidle_chart *cpuidle_charts, size_t core) {
359 + char filename[FILENAME_MAX + 1];
360 + static char next_state_filename[FILENAME_MAX + 1];
361 + struct stat stbuf;
362 + struct per_core_cpuidle_chart *cc = &cpuidle_charts[core];
363 + size_t state;
364 +
365 + if(unlikely(!cc->cpuidle_state_len || cc->rescan_cpu_states)) {
366 + int state_file_found = 1; // check at least one state
367 +
368 + if(cc->cpuidle_state_len) {
369 + for(state = 0; state < cc->cpuidle_state_len; state++) {
370 + freez(cc->cpuidle_state[state].name);
371 +
372 + freez(cc->cpuidle_state[state].time_filename);
373 + close(cc->cpuidle_state[state].time_fd);
374 + cc->cpuidle_state[state].time_fd = -1;
375 + }
376 +
377 + freez(cc->cpuidle_state);
378 + cc->cpuidle_state = NULL;
379 + cc->cpuidle_state_len = 0;
380 +
381 + cc->active_time_rd = NULL;
382 + cc->st = NULL;
383 + }
384 +
385 + while(likely(state_file_found)) {
386 + snprintfz(filename, FILENAME_MAX, cpuidle_name_filename, core, cc->cpuidle_state_len);
387 + if (stat(filename, &stbuf) == 0)
388 + cc->cpuidle_state_len++;
389 + else
390 + state_file_found = 0;
391 + }
392 + snprintfz(next_state_filename, FILENAME_MAX, cpuidle_name_filename, core, cc->cpuidle_state_len);
393 +
394 + cc->cpuidle_state = callocz(cc->cpuidle_state_len, sizeof(struct cpuidle_state));
395 + memset(cc->cpuidle_state, 0, sizeof(struct cpuidle_state) * cc->cpuidle_state_len);
396 +
397 + for(state = 0; state < cc->cpuidle_state_len; state++) {
398 + char name_buf[50 + 1];
399 + snprintfz(filename, FILENAME_MAX, cpuidle_name_filename, core, state);
400 +
401 + int fd = open(filename, O_RDONLY, 0666);
402 + if(unlikely(fd == -1)) {
403 + error("Cannot open file '%s'", filename);
404 + cc->rescan_cpu_states = 1;
405 + return 1;
406 + }
407 +
408 + ssize_t r = read(fd, name_buf, 50);
409 + if(unlikely(r < 1)) {
410 + error("Cannot read file '%s'", filename);
411 + close(fd);
412 + cc->rescan_cpu_states = 1;
413 + return 1;
414 + }
415 +
416 + name_buf[r - 1] = '\0'; // erase extra character
417 + cc->cpuidle_state[state].name = strdupz(name_buf);
418 + close(fd);
419 +
420 + snprintfz(filename, FILENAME_MAX, cpuidle_time_filename, core, state);
421 + cc->cpuidle_state[state].time_filename = strdupz(filename);
422 + cc->cpuidle_state[state].time_fd = -1;
423 + }
424 +
425 + cc->rescan_cpu_states = 0;
426 + }
427 +
428 + for(state = 0; state < cc->cpuidle_state_len; state++) {
429 +
430 + struct cpuidle_state *cs = &cc->cpuidle_state[state];
431 +
432 + if(unlikely(cs->time_fd == -1)) {
433 + cs->time_fd = open(cs->time_filename, O_RDONLY);
434 + if (unlikely(cs->time_fd == -1)) {
435 + error("Cannot open file '%s'", cs->time_filename);
436 + cc->rescan_cpu_states = 1;
437 + return 1;
438 + }
439 + }
440 +
441 + char time_buf[50 + 1];
442 + if(likely(read_one_state(time_buf, cs->time_filename, &cs->time_fd))) {
443 + cs->value = str2ll(time_buf, NULL);
444 + }
445 + else {
446 + cc->rescan_cpu_states = 1;
447 + return 1;
448 + }
449 + }
450 +
451 + // check if the number of states was increased
452 + if(unlikely(stat(next_state_filename, &stbuf) == 0)) {
453 + cc->rescan_cpu_states = 1;
454 + return 1;
455 + }
456 +
457 + return 0;
458 +}
459 +
460 int do_proc_stat(int update_every, usec_t dt) {
461 (void)dt;
462
463 static struct cpu_chart *all_cpu_charts = NULL;
464 static size_t all_cpu_charts_size = 0;
465 static procfile *ff = NULL;
246 - static int do_cpu = -1, do_cpu_cores = -1, do_interrupts = -1, do_context = -1, do_forks = -1, do_processes = -1, do_core_throttle_count = -1, do_package_throttle_count = -1, do_cpu_freq = -1;
466 + static int do_cpu = -1, do_cpu_cores = -1, do_interrupts = -1, do_context = -1, do_forks = -1, do_processes = -1,
467 + do_core_throttle_count = -1, do_package_throttle_count = -1, do_cpu_freq = -1, do_cpuidle = -1;
468 static uint32_t hash_intr, hash_ctxt, hash_processes, hash_procs_running, hash_procs_blocked;
248 - static char *core_throttle_count_filename = NULL, *package_throttle_count_filename = NULL, *scaling_cur_freq_filename = NULL, *time_in_state_filename = NULL;
469 + static char *core_throttle_count_filename = NULL, *package_throttle_count_filename = NULL, *scaling_cur_freq_filename = NULL,
470 + *time_in_state_filename = NULL, *schedstat_filename = NULL, *cpuidle_name_filename = NULL, *cpuidle_time_filename = NULL;
471 static RRDVAR *cpus_var = NULL;
472 static int accurate_freq_avail = 0, accurate_freq_is_used = 0;
473 size_t cores_found = (size_t)processors;
@@ -265,6 +487,7 @@ int do_proc_stat(int update_every, usec_t dt) {
487 do_core_throttle_count = CONFIG_BOOLEAN_NO;
488 do_package_throttle_count = CONFIG_BOOLEAN_NO;
489 do_cpu_freq = CONFIG_BOOLEAN_NO;
490 + do_cpuidle = CONFIG_BOOLEAN_NO;
491 }
492 else {
493 // the system has a reasonable number of processors
@@ -272,12 +495,23 @@ int do_proc_stat(int update_every, usec_t dt) {
495 do_core_throttle_count = CONFIG_BOOLEAN_AUTO;
496 do_package_throttle_count = CONFIG_BOOLEAN_NO;
497 do_cpu_freq = CONFIG_BOOLEAN_YES;
498 + do_cpuidle = CONFIG_BOOLEAN_YES;
499 + }
500 + if(unlikely(processors > 24)) {
501 + // the system has too many processors
502 + keep_cpuidle_fds_open = CONFIG_BOOLEAN_NO;
503 + }
504 + else {
505 + // the system has a reasonable number of processors
506 + keep_cpuidle_fds_open = CONFIG_BOOLEAN_YES;
507 }
508
509 keep_per_core_fds_open = config_get_boolean("plugin:proc:/proc/stat", "keep per core files open", keep_per_core_fds_open);
510 + keep_cpuidle_fds_open = config_get_boolean("plugin:proc:/proc/stat", "keep cpuidle files open", keep_cpuidle_fds_open);
511 do_core_throttle_count = config_get_boolean_ondemand("plugin:proc:/proc/stat", "core_throttle_count", do_core_throttle_count);
512 do_package_throttle_count = config_get_boolean_ondemand("plugin:proc:/proc/stat", "package_throttle_count", do_package_throttle_count);
513 do_cpu_freq = config_get_boolean_ondemand("plugin:proc:/proc/stat", "cpu frequency", do_cpu_freq);
514 + do_cpuidle = config_get_boolean_ondemand("plugin:proc:/proc/stat", "cpu idle states", do_cpuidle);
515
516 hash_intr = simple_hash("intr");
517 hash_ctxt = simple_hash("ctxt");
@@ -297,6 +531,15 @@ int do_proc_stat(int update_every, usec_t dt) {
531
532 snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/sys/devices/system/cpu/%s/cpufreq/stats/time_in_state");
533 time_in_state_filename = config_get("plugin:proc:/proc/stat", "time_in_state filename to monitor", filename);
534 +
535 + snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/proc/schedstat");
536 + schedstat_filename = config_get("plugin:proc:/proc/stat", "schedstat filename to monitor", filename);
537 +
538 + snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/sys/devices/system/cpu/cpu%zu/cpuidle/state%zu/name");
539 + cpuidle_name_filename = config_get("plugin:proc:/proc/stat", "cpuidle name filename to monitor", filename);
540 +
541 + snprintfz(filename, FILENAME_MAX, "%s%s", netdata_configured_host_prefix, "/sys/devices/system/cpu/cpu%zu/cpuidle/state%zu/time");
542 + cpuidle_time_filename = config_get("plugin:proc:/proc/stat", "cpuidle time filename to monitor", filename);
543 }
544
545 if(unlikely(!ff)) {
@@ -407,7 +650,7 @@ int do_proc_stat(int update_every, usec_t dt) {
650 cpu_chart->files[CPU_FREQ_INDEX].fd = -1;
651 do_cpu_freq = CONFIG_BOOLEAN_YES;
652 }
410 -
653 +
654 snprintfz(filename, FILENAME_MAX, time_in_state_filename, id);
655
656 if (stat(filename, &stbuf) == 0) {
@@ -702,7 +945,7 @@ int do_proc_stat(int update_every, usec_t dt) {
945 , "MHz"
946 , PLUGIN_PROC_NAME
947 , PLUGIN_PROC_MODULE_STAT_NAME
705 - , 5003
948 + , NETDATA_CHART_PRIO_CPUFREQ_SCALING_CUR_FREQ
949 , update_every
950 , RRDSET_TYPE_LINE
951 );
@@ -715,6 +958,79 @@ int do_proc_stat(int update_every, usec_t dt) {
958 }
959 }
960
961 + // --------------------------------------------------------------------
962 +
963 + static struct per_core_cpuidle_chart *cpuidle_charts = NULL;
964 +
965 + if(likely(do_cpuidle != CONFIG_BOOLEAN_NO && !read_schedstat(schedstat_filename, &cpuidle_charts, cores_found))) {
966 + int cpu_states_updated = 0;
967 + size_t core, state;
968 +
969 +
970 + // proc.plugin runs on Linux systems only. Multi-platform compatibility is not needed here,
971 + // so bare pthread functions are used to avoid unneeded overheads.
972 + for(core = 0; core < cores_found; core++) {
973 + if(unlikely(!(cpuidle_charts[core].active_time - cpuidle_charts[core].last_active_time))) {
974 + pthread_t thread;
975 +
976 + if(unlikely(pthread_create(&thread, NULL, wake_cpu_thread, (void *)&core)))
977 + error("Cannot create wake_cpu_thread");
978 + else if(unlikely(pthread_join(thread, NULL)))
979 + error("Cannot join wake_cpu_thread");
980 + cpu_states_updated = 1;
981 + }
982 + }
983 +
984 + if(unlikely(!cpu_states_updated || !read_schedstat(schedstat_filename, &cpuidle_charts, cores_found))) {
985 + for(core = 0; core < cores_found; core++) {
986 + cpuidle_charts[core].last_active_time = cpuidle_charts[core].active_time;
987 +
988 + int r = read_cpuidle_states(cpuidle_name_filename, cpuidle_time_filename, cpuidle_charts, core);
989 + if(likely(r != -1 && (do_cpuidle == CONFIG_BOOLEAN_YES || r > 0))) {
990 + do_cpuidle = CONFIG_BOOLEAN_YES;
991 +
992 + char cpuidle_chart_id[RRD_ID_LENGTH_MAX + 1];
993 + snprintfz(cpuidle_chart_id, RRD_ID_LENGTH_MAX, "cpu%zu_cpuidle", core);
994 +
995 + if(unlikely(!cpuidle_charts[core].st)) {
996 + cpuidle_charts[core].st = rrdset_create_localhost(
997 + "cpu"
998 + , cpuidle_chart_id
999 + , NULL
1000 + , "cpuidle"
1001 + , "cpuidle.cpuidle"
1002 + , "C-state residency"
1003 + , "time%"
1004 + , PLUGIN_PROC_NAME
1005 + , PLUGIN_PROC_MODULE_STAT_NAME
1006 + , NETDATA_CHART_PRIO_CPUIDLE + core
1007 + , update_every
1008 + , RRDSET_TYPE_STACKED
1009 + );
1010 +
1011 + char cpuidle_dim_id[RRD_ID_LENGTH_MAX + 1];
1012 + snprintfz(cpuidle_dim_id, RRD_ID_LENGTH_MAX, "cpu%zu_active_time", core);
1013 + cpuidle_charts[core].active_time_rd = rrddim_add(cpuidle_charts[core].st, cpuidle_dim_id, "C0 (active)", 1, 1, RRD_ALGORITHM_PCENT_OVER_DIFF_TOTAL);
1014 + for(state = 0; state < cpuidle_charts[core].cpuidle_state_len; state++) {
1015 + snprintfz(cpuidle_dim_id, RRD_ID_LENGTH_MAX, "cpu%zu_cpuidle_state%zu_time", core, state);
1016 + cpuidle_charts[core].cpuidle_state[state].rd = rrddim_add(cpuidle_charts[core].st, cpuidle_dim_id,
1017 + cpuidle_charts[core].cpuidle_state[state].name,
1018 + 1, 1, RRD_ALGORITHM_PCENT_OVER_DIFF_TOTAL);
1019 + }
1020 + }
1021 + else
1022 + rrdset_next(cpuidle_charts[core].st);
1023 +
1024 + rrddim_set_by_pointer(cpuidle_charts[core].st, cpuidle_charts[core].active_time_rd, cpuidle_charts[core].active_time);
1025 + for(state = 0; state < cpuidle_charts[core].cpuidle_state_len; state++) {
1026 + rrddim_set_by_pointer(cpuidle_charts[core].st, cpuidle_charts[core].cpuidle_state[state].rd, cpuidle_charts[core].cpuidle_state[state].value);
1027 + }
1028 + rrdset_done(cpuidle_charts[core].st);
1029 + }
1030 + }
1031 + }
1032 + }
1033 +
1034 if(cpus_var)
1035 rrdvar_custom_host_variable_set(localhost, cpus_var, cores_found);
1036
collectors/python.d.plugin/python.d.plugin.in
+1 -1
@@ -56,7 +56,7 @@ BASE_CONFIG = {'update_every': os.getenv('NETDATA_UPDATE_EVERY', 1),
56
57
58 MODULE_EXTENSION = '.chart.py'
59 -OBSOLETE_MODULES = ['apache_cache', 'gunicorn_log', 'nginx_log', 'cpufreq']
59 +OBSOLETE_MODULES = ['apache_cache', 'gunicorn_log', 'nginx_log', 'cpufreq', 'cpuidle']
60
61
62 def module_ok(m):