@cryptotaxi247 / netdata-1 / commits / 8a70725c1

proc.plugin: add pressure stall information (#7209)

* proc.plugin: add pressure stall information * dashboard_info: add "Pressure" section * proc.plugin: mention PSI collector in doc * dashboard_info: fix grammar in PSI section * proc_pressure: fix wrong line name for "full" metrics * proc_pressure: fix copypasta * proc_pressure: refactor to prepare for cgroup changes * cgroups.plugin: add pressure monitoring * add proc_pressure.h to targets * Makefile.am: fix indentation * cgroups.plugin: remove a useless comment * cgroups.plugin: fix pressure config name * proc.plugin: arrange pressure charts under corresponding sections * dashboard_info: rearrange pressure chart descriptions * dashboard_info: reword PSI descriptions

Haochen Tong committed Dec 2, 2019 at 22:04 UTC 8a70725c132deb6fc313edcdccaea756a685fc49
9 files changed +512
CMakeLists.txt
+2
@@ -438,6 +438,8 @@ set(PROC_PLUGIN_FILES
438 collectors/proc.plugin/proc_sys_kernel_random_entropy_avail.c
439 collectors/proc.plugin/proc_vmstat.c
440 collectors/proc.plugin/proc_uptime.c
441 + collectors/proc.plugin/proc_pressure.c
442 + collectors/proc.plugin/proc_pressure.h
443 collectors/proc.plugin/sys_kernel_mm_ksm.c
444 collectors/proc.plugin/sys_block_zram.c
445 collectors/proc.plugin/sys_devices_system_edac_mc.c
Makefile.am
+2
@@ -263,6 +263,8 @@ PROC_PLUGIN_FILES = \
263 collectors/proc.plugin/proc_loadavg.c \
264 collectors/proc.plugin/proc_meminfo.c \
265 collectors/proc.plugin/proc_pagetypeinfo.c \
266 + collectors/proc.plugin/proc_pressure.c \
267 + collectors/proc.plugin/proc_pressure.h \
268 collectors/proc.plugin/proc_net_dev.c \
269 collectors/proc.plugin/proc_net_ip_vs_stats.c \
270 collectors/proc.plugin/proc_net_netstat.c \
collectors/cgroups.plugin/sys_fs_cgroup.c
+269
@@ -23,6 +23,11 @@ static int cgroup_enable_blkio_throttle_io = CONFIG_BOOLEAN_AUTO;
23 static int cgroup_enable_blkio_throttle_ops = CONFIG_BOOLEAN_AUTO;
24 static int cgroup_enable_blkio_merged_ops = CONFIG_BOOLEAN_AUTO;
25 static int cgroup_enable_blkio_queued_ops = CONFIG_BOOLEAN_AUTO;
26 +static int cgroup_enable_pressure_cpu = CONFIG_BOOLEAN_AUTO;
27 +static int cgroup_enable_pressure_io_some = CONFIG_BOOLEAN_AUTO;
28 +static int cgroup_enable_pressure_io_full = CONFIG_BOOLEAN_AUTO;
29 +static int cgroup_enable_pressure_memory_some = CONFIG_BOOLEAN_AUTO;
30 +static int cgroup_enable_pressure_memory_full = CONFIG_BOOLEAN_AUTO;
31
32 static int cgroup_enable_systemd_services = CONFIG_BOOLEAN_YES;
33 static int cgroup_enable_systemd_services_detailed_memory = CONFIG_BOOLEAN_NO;
@@ -105,6 +110,12 @@ void read_cgroup_plugin_configuration() {
110 cgroup_enable_blkio_queued_ops = config_get_boolean_ondemand("plugin:cgroups", "enable blkio queued operations", cgroup_enable_blkio_queued_ops);
111 cgroup_enable_blkio_merged_ops = config_get_boolean_ondemand("plugin:cgroups", "enable blkio merged operations", cgroup_enable_blkio_merged_ops);
112
113 + cgroup_enable_pressure_cpu = config_get_boolean_ondemand("plugin:cgroups", "enable cpu pressure", cgroup_enable_pressure_cpu);
114 + cgroup_enable_pressure_io_some = config_get_boolean_ondemand("plugin:cgroups", "enable io some pressure", cgroup_enable_pressure_io_some);
115 + cgroup_enable_pressure_io_full = config_get_boolean_ondemand("plugin:cgroups", "enable io full pressure", cgroup_enable_pressure_io_full);
116 + cgroup_enable_pressure_memory_some = config_get_boolean_ondemand("plugin:cgroups", "enable memory some pressure", cgroup_enable_pressure_memory_some);
117 + cgroup_enable_pressure_memory_full = config_get_boolean_ondemand("plugin:cgroups", "enable memory full pressure", cgroup_enable_pressure_memory_full);
118 +
119 cgroup_recheck_zero_blkio_every_iterations = (int)config_get_number("plugin:cgroups", "recheck zero blkio every iterations", cgroup_recheck_zero_blkio_every_iterations);
120 cgroup_recheck_zero_mem_failcnt_every_iterations = (int)config_get_number("plugin:cgroups", "recheck zero memory failcnt every iterations", cgroup_recheck_zero_mem_failcnt_every_iterations);
121 cgroup_recheck_zero_mem_detailed_every_iterations = (int)config_get_number("plugin:cgroups", "recheck zero detailed memory every iterations", cgroup_recheck_zero_mem_detailed_every_iterations);
@@ -116,6 +127,13 @@ void read_cgroup_plugin_configuration() {
127 char filename[FILENAME_MAX + 1], *s;
128 struct mountinfo *mi, *root = mountinfo_read(0);
129 if(!cgroup_use_unified_cgroups) {
130 + // cgroup v1 does not have pressure metrics
131 + cgroup_enable_pressure_cpu =
132 + cgroup_enable_pressure_io_some =
133 + cgroup_enable_pressure_io_full =
134 + cgroup_enable_pressure_memory_some =
135 + cgroup_enable_pressure_memory_full = CONFIG_BOOLEAN_NO;
136 +
137 mi = mountinfo_find_by_filesystem_super_option(root, "cgroup", "cpuacct");
138 if(!mi) mi = mountinfo_find_by_filesystem_mount_source(root, "cgroup", "cpuacct");
139 if(!mi) {
@@ -461,6 +479,10 @@ struct cgroup {
479
480 struct cgroup_network_interface *interfaces;
481
482 + struct pressure cpu_pressure;
483 + struct pressure io_pressure;
484 + struct pressure memory_pressure;
485 +
486 // per cgroup charts
487 RRDSET *st_cpu;
488 RRDSET *st_cpu_limit;
@@ -798,6 +820,54 @@ static inline void cgroup2_read_blkio(struct blkio *io, unsigned int word_offset
820 }
821 }
822
823 +static inline void cgroup2_read_pressure(struct pressure *res) {
824 + static procfile *ff = NULL;
825 +
826 + if (likely(res->filename)) {
827 + ff = procfile_reopen(ff, res->filename, " =", PROCFILE_FLAG_DEFAULT);
828 + if (unlikely(!ff)) {
829 + res->updated = 0;
830 + cgroups_check = 1;
831 + return;
832 + }
833 +
834 + ff = procfile_readall(ff);
835 + if (unlikely(!ff)) {
836 + res->updated = 0;
837 + cgroups_check = 1;
838 + return;
839 + }
840 +
841 + size_t lines = procfile_lines(ff);
842 + if (lines < 1) {
843 + error("CGROUP: file '%s' should have 1+ lines.", res->filename);
844 + res->updated = 0;
845 + return;
846 + }
847 +
848 + res->some.value10 = strtod(procfile_lineword(ff, 0, 2), NULL);
849 + res->some.value60 = strtod(procfile_lineword(ff, 0, 4), NULL);
850 + res->some.value300 = strtod(procfile_lineword(ff, 0, 6), NULL);
851 +
852 + if (lines > 2) {
853 + res->full.value10 = strtod(procfile_lineword(ff, 1, 2), NULL);
854 + res->full.value60 = strtod(procfile_lineword(ff, 1, 4), NULL);
855 + res->full.value300 = strtod(procfile_lineword(ff, 1, 6), NULL);
856 + }
857 +
858 + res->updated = 1;
859 +
860 + if (unlikely(res->some.enabled == CONFIG_BOOLEAN_AUTO)) {
861 + res->some.enabled = CONFIG_BOOLEAN_YES;
862 + if (lines > 2) {
863 + res->full.enabled = CONFIG_BOOLEAN_YES;
864 + } else {
865 + res->full.enabled = CONFIG_BOOLEAN_NO;
866 + }
867 + }
868 + }
869 +}
870 +
871 static inline void cgroup_read_memory(struct memory *mem, char parent_cg_is_unified) {
872 static procfile *ff = NULL;
873
@@ -946,6 +1016,9 @@ static inline void cgroup_read(struct cgroup *cg) {
1016 cgroup2_read_blkio(&cg->io_service_bytes, 0);
1017 cgroup2_read_blkio(&cg->io_serviced, 4);
1018 cgroup2_read_cpuacct_stat(&cg->cpuacct_stat);
1019 + cgroup2_read_pressure(&cg->cpu_pressure);
1020 + cgroup2_read_pressure(&cg->io_pressure);
1021 + cgroup2_read_pressure(&cg->memory_pressure);
1022 cgroup_read_memory(&cg->memory, 1);
1023 }
1024 }
@@ -1236,6 +1309,12 @@ static inline struct cgroup *cgroup_add(const char *id) {
1309 return cg;
1310 }
1311
1312 +static inline void free_pressure(struct pressure *res) {
1313 + if (res->some.st) rrdset_is_obsolete(res->some.st);
1314 + if (res->full.st) rrdset_is_obsolete(res->full.st);
1315 + freez(res->filename);
1316 +}
1317 +
1318 static inline void cgroup_free(struct cgroup *cg) {
1319 debug(D_CGROUP, "Removing cgroup '%s' with chart id '%s' (was %s and %s)", cg->id, cg->chart_id, (cg->enabled)?"enabled":"disabled", (cg->available)?"available":"not available");
1320
@@ -1284,6 +1363,10 @@ static inline void cgroup_free(struct cgroup *cg) {
1363 freez(cg->io_merged.filename);
1364 freez(cg->io_queued.filename);
1365
1366 + free_pressure(&cg->cpu_pressure);
1367 + free_pressure(&cg->io_pressure);
1368 + free_pressure(&cg->memory_pressure);
1369 +
1370 freez(cg->id);
1371 freez(cg->chart_id);
1372 freez(cg->chart_title);
@@ -1748,6 +1831,42 @@ static inline void find_all_cgroups() {
1831 else
1832 debug(D_CGROUP, "memory.swap file for cgroup '%s': '%s' does not exist.", cg->id, filename);
1833 }
1834 +
1835 + if (unlikely(cgroup_enable_pressure_cpu && !cg->cpu_pressure.filename)) {
1836 + snprintfz(filename, FILENAME_MAX, "%s%s/cpu.pressure", cgroup_unified_base, cg->id);
1837 + if (likely(stat(filename, &buf) != -1)) {
1838 + cg->cpu_pressure.filename = strdupz(filename);
1839 + cg->cpu_pressure.some.enabled = cgroup_enable_pressure_cpu;
1840 + cg->cpu_pressure.full.enabled = CONFIG_BOOLEAN_NO;
1841 + debug(D_CGROUP, "cpu.pressure filename for cgroup '%s': '%s'", cg->id, cg->cpu_pressure.filename);
1842 + } else {
1843 + debug(D_CGROUP, "cpu.pressure file for cgroup '%s': '%s' does not exist", cg->id, filename);
1844 + }
1845 + }
1846 +
1847 + if (unlikely((cgroup_enable_pressure_io_some || cgroup_enable_pressure_io_full) && !cg->io_pressure.filename)) {
1848 + snprintfz(filename, FILENAME_MAX, "%s%s/io.pressure", cgroup_unified_base, cg->id);
1849 + if (likely(stat(filename, &buf) != -1)) {
1850 + cg->io_pressure.filename = strdupz(filename);
1851 + cg->io_pressure.some.enabled = cgroup_enable_pressure_io_some;
1852 + cg->io_pressure.full.enabled = cgroup_enable_pressure_io_full;
1853 + debug(D_CGROUP, "io.pressure filename for cgroup '%s': '%s'", cg->id, cg->io_pressure.filename);
1854 + } else {
1855 + debug(D_CGROUP, "io.pressure file for cgroup '%s': '%s' does not exist", cg->id, filename);
1856 + }
1857 + }
1858 +
1859 + if (unlikely((cgroup_enable_pressure_memory_some || cgroup_enable_pressure_memory_full) && !cg->memory_pressure.filename)) {
1860 + snprintfz(filename, FILENAME_MAX, "%s%s/memory.pressure", cgroup_unified_base, cg->id);
1861 + if (likely(stat(filename, &buf) != -1)) {
1862 + cg->memory_pressure.filename = strdupz(filename);
1863 + cg->memory_pressure.some.enabled = cgroup_enable_pressure_memory_some;
1864 + cg->memory_pressure.full.enabled = cgroup_enable_pressure_memory_full;
1865 + debug(D_CGROUP, "memory.pressure filename for cgroup '%s': '%s'", cg->id, cg->memory_pressure.filename);
1866 + } else {
1867 + debug(D_CGROUP, "memory.pressure file for cgroup '%s': '%s' does not exist", cg->id, filename);
1868 + }
1869 + }
1870 }
1871 }
1872
@@ -3364,6 +3483,156 @@ void update_cgroup_charts(int update_every) {
3483 rrddim_set(cg->st_merged_ops, "write", cg->io_merged.Write);
3484 rrdset_done(cg->st_merged_ops);
3485 }
3486 +
3487 + if (cg->options & CGROUP_OPTIONS_IS_UNIFIED) {
3488 + struct pressure *res = &cg->cpu_pressure;
3489 + if (likely(res->updated && res->some.enabled)) {
3490 + if (unlikely(!res->some.st)) {
3491 + RRDSET *chart;
3492 + snprintfz(title, CHART_TITLE_MAX, "CPU pressure for cgroup %s", cg->chart_title);
3493 +
3494 + chart = res->some.st = rrdset_create_localhost(
3495 + cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3496 + , "cpu_pressure"
3497 + , NULL
3498 + , "cpu"
3499 + , "cgroup.cpu_pressure"
3500 + , title
3501 + , "percentage"
3502 + , PLUGIN_CGROUPS_NAME
3503 + , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3504 + , cgroup_containers_chart_priority + 2200
3505 + , update_every,
3506 + RRDSET_TYPE_LINE);
3507 +
3508 + res->some.rd10 = rrddim_add(chart, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3509 + res->some.rd60 = rrddim_add(chart, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3510 + res->some.rd300 = rrddim_add(chart, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3511 + } else {
3512 + rrdset_next(res->some.st);
3513 + }
3514 +
3515 + update_pressure_chart(&res->some);
3516 + }
3517 +
3518 + res = &cg->memory_pressure;
3519 + if (likely(res->updated && res->some.enabled)) {
3520 + if (unlikely(!res->some.st)) {
3521 + RRDSET *chart;
3522 + snprintfz(title, CHART_TITLE_MAX, "Memory pressure for cgroup %s", cg->chart_title);
3523 +
3524 + chart = res->some.st = rrdset_create_localhost(
3525 + cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3526 + , "mem_pressure"
3527 + , NULL
3528 + , "mem"
3529 + , "cgroup.memory_pressure"
3530 + , title
3531 + , "percentage"
3532 + , PLUGIN_CGROUPS_NAME
3533 + , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3534 + , cgroup_containers_chart_priority + 2300
3535 + , update_every,
3536 + RRDSET_TYPE_LINE);
3537 +
3538 + res->some.rd10 = rrddim_add(chart, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3539 + res->some.rd60 = rrddim_add(chart, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3540 + res->some.rd300 = rrddim_add(chart, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3541 + } else {
3542 + rrdset_next(res->some.st);
3543 + }
3544 +
3545 + update_pressure_chart(&res->some);
3546 + }
3547 +
3548 + if (likely(res->updated && res->full.enabled)) {
3549 + if (unlikely(!res->full.st)) {
3550 + RRDSET *chart;
3551 + snprintfz(title, CHART_TITLE_MAX, "Memory full pressure for cgroup %s", cg->chart_title);
3552 +
3553 + chart = res->full.st = rrdset_create_localhost(
3554 + cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3555 + , "mem_full_pressure"
3556 + , NULL
3557 + , "mem"
3558 + , "cgroup.memory_full_pressure"
3559 + , title
3560 + , "percentage"
3561 + , PLUGIN_CGROUPS_NAME
3562 + , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3563 + , cgroup_containers_chart_priority + 2350
3564 + , update_every,
3565 + RRDSET_TYPE_LINE);
3566 +
3567 + res->full.rd10 = rrddim_add(chart, "full 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3568 + res->full.rd60 = rrddim_add(chart, "full 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3569 + res->full.rd300 = rrddim_add(chart, "full 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3570 + } else {
3571 + rrdset_next(res->full.st);
3572 + }
3573 +
3574 + update_pressure_chart(&res->full);
3575 + }
3576 +
3577 + res = &cg->io_pressure;
3578 + if (likely(res->updated && res->some.enabled)) {
3579 + if (unlikely(!res->some.st)) {
3580 + RRDSET *chart;
3581 + snprintfz(title, CHART_TITLE_MAX, "I/O pressure for cgroup %s", cg->chart_title);
3582 +
3583 + chart = res->some.st = rrdset_create_localhost(
3584 + cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3585 + , "io_pressure"
3586 + , NULL
3587 + , "disk"
3588 + , "cgroup.io_pressure"
3589 + , title
3590 + , "percentage"
3591 + , PLUGIN_CGROUPS_NAME
3592 + , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3593 + , cgroup_containers_chart_priority + 2400
3594 + , update_every,
3595 + RRDSET_TYPE_LINE);
3596 +
3597 + res->some.rd10 = rrddim_add(chart, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3598 + res->some.rd60 = rrddim_add(chart, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3599 + res->some.rd300 = rrddim_add(chart, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3600 + } else {
3601 + rrdset_next(res->some.st);
3602 + }
3603 +
3604 + update_pressure_chart(&res->some);
3605 + }
3606 +
3607 + if (likely(res->updated && res->full.enabled)) {
3608 + if (unlikely(!res->full.st)) {
3609 + RRDSET *chart;
3610 + snprintfz(title, CHART_TITLE_MAX, "I/O full pressure for cgroup %s", cg->chart_title);
3611 +
3612 + chart = res->full.st = rrdset_create_localhost(
3613 + cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3614 + , "io_full_pressure"
3615 + , NULL
3616 + , "disk"
3617 + , "cgroup.io_full_pressure"
3618 + , title
3619 + , "percentage"
3620 + , PLUGIN_CGROUPS_NAME
3621 + , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3622 + , cgroup_containers_chart_priority + 2450
3623 + , update_every,
3624 + RRDSET_TYPE_LINE);
3625 +
3626 + res->full.rd10 = rrddim_add(chart, "full 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3627 + res->full.rd60 = rrddim_add(chart, "full 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3628 + res->full.rd300 = rrddim_add(chart, "full 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3629 + } else {
3630 + rrdset_next(res->full.st);
3631 + }
3632 +
3633 + update_pressure_chart(&res->full);
3634 + }
3635 + }
3636 }
3637
3638 if(likely(cgroup_enable_systemd_services))
collectors/proc.plugin/README.md
+1
@@ -18,6 +18,7 @@
18 - `/proc/interrupts` (total and per core hardware interrupts)
19 - `/proc/softirqs` (total and per core software interrupts)
20 - `/proc/loadavg` (system load and total processes running)
21 +- `/proc/pressure/{cpu,memory,io}` (pressure stall information)
22 - `/proc/sys/kernel/random/entropy_avail` (random numbers pool availability - used in cryptography)
23 - `/sys/class/power_supply` (power supply properties)
24 - `ipc` (IPC semaphores and message queues)
collectors/proc.plugin/plugin_proc.c
+3
@@ -21,6 +21,9 @@ static struct proc_module {
21 { .name = "/proc/loadavg", .dim = "loadavg", .func = do_proc_loadavg },
22 { .name = "/proc/sys/kernel/random/entropy_avail", .dim = "entropy", .func = do_proc_sys_kernel_random_entropy_avail },
23
24 + // pressure metrics
25 + { .name = "/proc/pressure", .dim = "pressure", .func = do_proc_pressure },
26 +
27 // CPU metrics
28 { .name = "/proc/interrupts", .dim = "interrupts", .func = do_proc_interrupts },
29 { .name = "/proc/softirqs", .dim = "softirqs", .func = do_proc_softirqs },
collectors/proc.plugin/plugin_proc.h
+2
@@ -40,6 +40,7 @@ extern int do_proc_net_rpc_nfsd(int update_every, usec_t dt);
40 extern int do_proc_sys_kernel_random_entropy_avail(int update_every, usec_t dt);
41 extern int do_proc_interrupts(int update_every, usec_t dt);
42 extern int do_proc_softirqs(int update_every, usec_t dt);
43 +extern int do_proc_pressure(int update_every, usec_t dt);
44 extern int do_sys_kernel_mm_ksm(int update_every, usec_t dt);
45 extern int do_sys_block_zram(int update_every, usec_t dt);
46 extern int do_proc_loadavg(int update_every, usec_t dt);
@@ -66,6 +67,7 @@ extern void netdev_rename_device_add(const char *host_device, const char *contai
67 extern void netdev_rename_device_del(const char *host_device);
68
69 #include "proc_self_mountinfo.h"
70 +#include "proc_pressure.h"
71 #include "zfs_common.h"
72
73 #else // (TARGET_OS == OS_LINUX)
collectors/proc.plugin/proc_pressure.c new
+177
@@ -0,0 +1,177 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#include "plugin_proc.h"
4 +
5 +#define PLUGIN_PROC_MODULE_PRESSURE_NAME "/proc/pressure"
6 +#define CONFIG_SECTION_PLUGIN_PROC_PRESSURE "plugin:" PLUGIN_PROC_CONFIG_NAME ":" PLUGIN_PROC_MODULE_PRESSURE_NAME
7 +
8 +// linux calculates this every 2 seconds, see kernel/sched/psi.c PSI_FREQ
9 +#define MIN_PRESSURE_UPDATE_EVERY 2
10 +
11 +
12 +static struct pressure resources[PRESSURE_NUM_RESOURCES] = {
13 + {
14 + .some = { .id = "cpu_pressure", .title = "CPU Pressure" },
15 + },
16 + {
17 + .some = { .id = "memory_some_pressure", .title = "Memory Pressure" },
18 + .full = { .id = "memory_full_pressure", .title = "Memory Full Pressure" },
19 + },
20 + {
21 + .some = { .id = "io_some_pressure", .title = "I/O Pressure" },
22 + .full = { .id = "io_full_pressure", .title = "I/O Full Pressure" },
23 + },
24 +};
25 +
26 +static struct {
27 + procfile *pf;
28 + const char *name; // metric file name
29 + const char *family; // webui section name
30 + int section_priority;
31 +} resource_info[PRESSURE_NUM_RESOURCES] = {
32 + { .name = "cpu", .family = "cpu", .section_priority = NETDATA_CHART_PRIO_SYSTEM_CPU },
33 + { .name = "memory", .family = "ram", .section_priority = NETDATA_CHART_PRIO_SYSTEM_RAM },
34 + { .name = "io", .family = "disk", .section_priority = NETDATA_CHART_PRIO_SYSTEM_IO },
35 +};
36 +
37 +void update_pressure_chart(struct pressure_chart *chart) {
38 + rrddim_set_by_pointer(chart->st, chart->rd10, (collected_number)(chart->value10 * 100));
39 + rrddim_set_by_pointer(chart->st, chart->rd60, (collected_number) (chart->value60 * 100));
40 + rrddim_set_by_pointer(chart->st, chart->rd300, (collected_number) (chart->value300 * 100));
41 +
42 + rrdset_done(chart->st);
43 +}
44 +
45 +int do_proc_pressure(int update_every, usec_t dt) {
46 + int fail_count = 0;
47 + int i;
48 +
49 + static usec_t next_pressure_dt = 0;
50 + static char *base_path = NULL;
51 +
52 + update_every = (update_every < MIN_PRESSURE_UPDATE_EVERY) ? MIN_PRESSURE_UPDATE_EVERY : update_every;
53 +
54 + if (next_pressure_dt <= dt) {
55 + next_pressure_dt = update_every * USEC_PER_SEC;
56 + } else {
57 + next_pressure_dt -= dt;
58 + return 0;
59 + }
60 +
61 + if (unlikely(!base_path)) {
62 + base_path = config_get(CONFIG_SECTION_PLUGIN_PROC_PRESSURE, "base path of pressure metrics", "/proc/pressure");
63 + }
64 +
65 + for (i = 0; i < PRESSURE_NUM_RESOURCES; i++) {
66 + procfile *ff = resource_info[i].pf;
67 + int do_some = resources[i].some.enabled, do_full = resources[i].full.enabled;
68 +
69 + if (unlikely(!ff)) {
70 + char filename[FILENAME_MAX + 1];
71 + char config_key[CONFIG_MAX_NAME + 1];
72 +
73 + snprintfz(filename
74 + , FILENAME_MAX
75 + , "%s%s/%s"
76 + , netdata_configured_host_prefix
77 + , base_path
78 + , resource_info[i].name);
79 +
80 + snprintfz(config_key, CONFIG_MAX_NAME, "enable %s some pressure", resource_info[i].name);
81 + do_some = config_get_boolean(CONFIG_SECTION_PLUGIN_PROC_PRESSURE, config_key, CONFIG_BOOLEAN_YES);
82 + resources[i].some.enabled = do_some;
83 + if (resources[i].full.id) {
84 + snprintfz(config_key, CONFIG_MAX_NAME, "enable %s full pressure", resource_info[i].name);
85 + do_full = config_get_boolean(CONFIG_SECTION_PLUGIN_PROC_PRESSURE, config_key, CONFIG_BOOLEAN_YES);
86 + resources[i].full.enabled = do_full;
87 + }
88 +
89 + ff = procfile_open(filename, " =", PROCFILE_FLAG_DEFAULT);
90 + if (unlikely(!ff)) {
91 + error("Cannot read pressure information from %s.", filename);
92 + fail_count++;
93 + continue;
94 + }
95 + }
96 +
97 + ff = procfile_readall(ff);
98 + resource_info[i].pf = ff;
99 + if (unlikely(!ff)) {
100 + continue;
101 + }
102 +
103 + size_t lines = procfile_lines(ff);
104 + if (unlikely(lines < 1)) {
105 + error("%s has no lines.", procfile_filename(ff));
106 + fail_count++;
107 + continue;
108 + }
109 +
110 + struct pressure_chart *chart;
111 + if (do_some) {
112 + chart = &resources[i].some;
113 + if (unlikely(!chart->st)) {
114 + chart->st = rrdset_create_localhost(
115 + "system"
116 + , chart->id
117 + , NULL
118 + , resource_info[i].family
119 + , NULL
120 + , chart->title
121 + , "percentage"
122 + , PLUGIN_PROC_NAME
123 + , PLUGIN_PROC_MODULE_PRESSURE_NAME
124 + , resource_info[i].section_priority + 40
125 + , update_every
126 + , RRDSET_TYPE_LINE
127 + );
128 + chart->rd10 = rrddim_add(chart->st, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
129 + chart->rd60 = rrddim_add(chart->st, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
130 + chart->rd300 = rrddim_add(chart->st, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
131 + } else {
132 + rrdset_next(chart->st);
133 + }
134 +
135 + chart->value10 = strtod(procfile_lineword(ff, 0, 2), NULL);
136 + chart->value60 = strtod(procfile_lineword(ff, 0, 4), NULL);
137 + chart->value300 = strtod(procfile_lineword(ff, 0, 6), NULL);
138 + update_pressure_chart(chart);
139 + }
140 +
141 + if (do_full && lines > 2) {
142 + chart = &resources[i].full;
143 + if (unlikely(!chart->st)) {
144 + chart->st = rrdset_create_localhost(
145 + "system"
146 + , chart->id
147 + , NULL
148 + , resource_info[i].family
149 + , NULL
150 + , chart->title
151 + , "percentage"
152 + , PLUGIN_PROC_NAME
153 + , PLUGIN_PROC_MODULE_PRESSURE_NAME
154 + , resource_info[i].section_priority + 45
155 + , update_every
156 + , RRDSET_TYPE_LINE
157 + );
158 + chart->rd10 = rrddim_add(chart->st, "full 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
159 + chart->rd60 = rrddim_add(chart->st, "full 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
160 + chart->rd300 = rrddim_add(chart->st, "full 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
161 + } else {
162 + rrdset_next(chart->st);
163 + }
164 +
165 + chart->value10 = strtod(procfile_lineword(ff, 1, 2), NULL);
166 + chart->value60 = strtod(procfile_lineword(ff, 1, 4), NULL);
167 + chart->value300 = strtod(procfile_lineword(ff, 1, 6), NULL);
168 + update_pressure_chart(chart);
169 + }
170 + }
171 +
172 + if (PRESSURE_NUM_RESOURCES == fail_count) {
173 + return 1;
174 + }
175 +
176 + return 0;
177 +}
collectors/proc.plugin/proc_pressure.h new
+31
@@ -0,0 +1,31 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_PROC_PRESSURE_H
4 +#define NETDATA_PROC_PRESSURE_H
5 +
6 +#define PRESSURE_NUM_RESOURCES 3
7 +
8 +struct pressure {
9 + int updated;
10 + char *filename;
11 +
12 + struct pressure_chart {
13 + int enabled;
14 +
15 + const char *id;
16 + const char *title;
17 +
18 + double value10;
19 + double value60;
20 + double value300;
21 +
22 + RRDSET *st;
23 + RRDDIM *rd10;
24 + RRDDIM *rd60;
25 + RRDDIM *rd300;
26 + } some, full;
27 +};
28 +
29 +extern void update_pressure_chart(struct pressure_chart *chart);
30 +
31 +#endif //NETDATA_PROC_PRESSURE_H
web/gui/dashboard_info.js
+25
@@ -718,6 +718,31 @@ netdataDashboard.context = {
718 height: 0.7
719 },
720
721 + 'system.cpu_pressure': {
722 + info: '<a href="https://www.kernel.org/doc/html/latest/accounting/psi.html">Pressure Stall Information</a> ' +
723 + 'identifies and quantifies the disruptions caused by resource contentions. ' +
724 + 'The "some" line indicates the share of time in which at least <b>some</b> tasks are stalled on CPU. ' +
725 + 'The ratios (in %) are tracked as recent trends over 10-, 60-, and 300-second windows.'
726 + },
727 +
728 + 'system.memory_some_pressure': {
729 + info: '<a href="https://www.kernel.org/doc/html/latest/accounting/psi.html">Pressure Stall Information</a> ' +
730 + 'identifies and quantifies the disruptions caused by resource contentions. ' +
731 + 'The "some" line indicates the share of time in which at least <b>some</b> tasks are stalled on memory. ' +
732 + 'The "full" line indicates the share of time in which <b>all non-idle</b> tasks are stalled on memory simultaneously. ' +
733 + 'In this state actual CPU cycles are going to waste, and a workload that spends extended time in this state is considered to be thrashing. ' +
734 + 'The ratios (in %) are tracked as recent trends over 10-, 60-, and 300-second windows.'
735 + },
736 +
737 + 'system.io_some_pressure': {
738 + info: '<a href="https://www.kernel.org/doc/html/latest/accounting/psi.html">Pressure Stall Information</a> ' +
739 + 'identifies and quantifies the disruptions caused by resource contentions. ' +
740 + 'The "some" line indicates the share of time in which at least <b>some</b> tasks are stalled on I/O. ' +
741 + 'The "full" line indicates the share of time in which <b>all non-idle</b> tasks are stalled on I/O simultaneously. ' +
742 + 'In this state actual CPU cycles are going to waste, and a workload that spends extended time in this state is considered to be thrashing. ' +
743 + 'The ratios (in %) are tracked as recent trends over 10-, 60-, and 300-second windows.'
744 + },
745 +
746 'system.io': {
747 info: function (os) {
748 var s = 'Total Disk I/O, for all physical disks. You can get detailed information about each disk at the <a href="#menu_disk">Disks</a> section and per application Disk usage at the <a href="#menu_apps">Applications Monitoring</a> section.';