proc.plugin: add pressure stall information (#7209)
* proc.plugin: add pressure stall information * dashboard_info: add "Pressure" section * proc.plugin: mention PSI collector in doc * dashboard_info: fix grammar in PSI section * proc_pressure: fix wrong line name for "full" metrics * proc_pressure: fix copypasta * proc_pressure: refactor to prepare for cgroup changes * cgroups.plugin: add pressure monitoring * add proc_pressure.h to targets * Makefile.am: fix indentation * cgroups.plugin: remove a useless comment * cgroups.plugin: fix pressure config name * proc.plugin: arrange pressure charts under corresponding sections * dashboard_info: rearrange pressure chart descriptions * dashboard_info: reword PSI descriptions
Haochen Tong committed
Dec 2, 2019 at 22:04 UTC
8a70725c132deb6fc313edcdccaea756a685fc49
9 files changed
+512
CMakeLists.txt
+2
@@ -438,6 +438,8 @@ set(PROC_PLUGIN_FILES
438
collectors/proc.plugin/proc_sys_kernel_random_entropy_avail.c
439
collectors/proc.plugin/proc_vmstat.c
440
collectors/proc.plugin/proc_uptime.c
441
+ collectors/proc.plugin/proc_pressure.c
442
+ collectors/proc.plugin/proc_pressure.h
443
collectors/proc.plugin/sys_kernel_mm_ksm.c
444
collectors/proc.plugin/sys_block_zram.c
445
collectors/proc.plugin/sys_devices_system_edac_mc.c
Makefile.am
+2
@@ -263,6 +263,8 @@ PROC_PLUGIN_FILES = \
263
collectors/proc.plugin/proc_loadavg.c \
264
collectors/proc.plugin/proc_meminfo.c \
265
collectors/proc.plugin/proc_pagetypeinfo.c \
266
+ collectors/proc.plugin/proc_pressure.c \
267
+ collectors/proc.plugin/proc_pressure.h \
268
collectors/proc.plugin/proc_net_dev.c \
269
collectors/proc.plugin/proc_net_ip_vs_stats.c \
270
collectors/proc.plugin/proc_net_netstat.c \
collectors/cgroups.plugin/sys_fs_cgroup.c
+269
@@ -23,6 +23,11 @@ static int cgroup_enable_blkio_throttle_io = CONFIG_BOOLEAN_AUTO;
23
static int cgroup_enable_blkio_throttle_ops = CONFIG_BOOLEAN_AUTO;
24
static int cgroup_enable_blkio_merged_ops = CONFIG_BOOLEAN_AUTO;
25
static int cgroup_enable_blkio_queued_ops = CONFIG_BOOLEAN_AUTO;
26
+static int cgroup_enable_pressure_cpu = CONFIG_BOOLEAN_AUTO;
27
+static int cgroup_enable_pressure_io_some = CONFIG_BOOLEAN_AUTO;
28
+static int cgroup_enable_pressure_io_full = CONFIG_BOOLEAN_AUTO;
29
+static int cgroup_enable_pressure_memory_some = CONFIG_BOOLEAN_AUTO;
30
+static int cgroup_enable_pressure_memory_full = CONFIG_BOOLEAN_AUTO;
31
32
static int cgroup_enable_systemd_services = CONFIG_BOOLEAN_YES;
33
static int cgroup_enable_systemd_services_detailed_memory = CONFIG_BOOLEAN_NO;
@@ -105,6 +110,12 @@ void read_cgroup_plugin_configuration() {
110
cgroup_enable_blkio_queued_ops = config_get_boolean_ondemand("plugin:cgroups", "enable blkio queued operations", cgroup_enable_blkio_queued_ops);
111
cgroup_enable_blkio_merged_ops = config_get_boolean_ondemand("plugin:cgroups", "enable blkio merged operations", cgroup_enable_blkio_merged_ops);
112
113
+ cgroup_enable_pressure_cpu = config_get_boolean_ondemand("plugin:cgroups", "enable cpu pressure", cgroup_enable_pressure_cpu);
114
+ cgroup_enable_pressure_io_some = config_get_boolean_ondemand("plugin:cgroups", "enable io some pressure", cgroup_enable_pressure_io_some);
115
+ cgroup_enable_pressure_io_full = config_get_boolean_ondemand("plugin:cgroups", "enable io full pressure", cgroup_enable_pressure_io_full);
116
+ cgroup_enable_pressure_memory_some = config_get_boolean_ondemand("plugin:cgroups", "enable memory some pressure", cgroup_enable_pressure_memory_some);
117
+ cgroup_enable_pressure_memory_full = config_get_boolean_ondemand("plugin:cgroups", "enable memory full pressure", cgroup_enable_pressure_memory_full);
118
+
119
cgroup_recheck_zero_blkio_every_iterations = (int)config_get_number("plugin:cgroups", "recheck zero blkio every iterations", cgroup_recheck_zero_blkio_every_iterations);
120
cgroup_recheck_zero_mem_failcnt_every_iterations = (int)config_get_number("plugin:cgroups", "recheck zero memory failcnt every iterations", cgroup_recheck_zero_mem_failcnt_every_iterations);
121
cgroup_recheck_zero_mem_detailed_every_iterations = (int)config_get_number("plugin:cgroups", "recheck zero detailed memory every iterations", cgroup_recheck_zero_mem_detailed_every_iterations);
@@ -116,6 +127,13 @@ void read_cgroup_plugin_configuration() {
127
char filename[FILENAME_MAX + 1], *s;
128
struct mountinfo *mi, *root = mountinfo_read(0);
129
if(!cgroup_use_unified_cgroups) {
130
+ // cgroup v1 does not have pressure metrics
131
+ cgroup_enable_pressure_cpu =
132
+ cgroup_enable_pressure_io_some =
133
+ cgroup_enable_pressure_io_full =
134
+ cgroup_enable_pressure_memory_some =
135
+ cgroup_enable_pressure_memory_full = CONFIG_BOOLEAN_NO;
136
+
137
mi = mountinfo_find_by_filesystem_super_option(root, "cgroup", "cpuacct");
138
if(!mi) mi = mountinfo_find_by_filesystem_mount_source(root, "cgroup", "cpuacct");
139
if(!mi) {
@@ -461,6 +479,10 @@ struct cgroup {
479
480
struct cgroup_network_interface *interfaces;
481
482
+ struct pressure cpu_pressure;
483
+ struct pressure io_pressure;
484
+ struct pressure memory_pressure;
485
+
486
// per cgroup charts
487
RRDSET *st_cpu;
488
RRDSET *st_cpu_limit;
@@ -798,6 +820,54 @@ static inline void cgroup2_read_blkio(struct blkio *io, unsigned int word_offset
820
}
821
}
822
823
+static inline void cgroup2_read_pressure(struct pressure *res) {
824
+ static procfile *ff = NULL;
825
+
826
+ if (likely(res->filename)) {
827
+ ff = procfile_reopen(ff, res->filename, " =", PROCFILE_FLAG_DEFAULT);
828
+ if (unlikely(!ff)) {
829
+ res->updated = 0;
830
+ cgroups_check = 1;
831
+ return;
832
+ }
833
+
834
+ ff = procfile_readall(ff);
835
+ if (unlikely(!ff)) {
836
+ res->updated = 0;
837
+ cgroups_check = 1;
838
+ return;
839
+ }
840
+
841
+ size_t lines = procfile_lines(ff);
842
+ if (lines < 1) {
843
+ error("CGROUP: file '%s' should have 1+ lines.", res->filename);
844
+ res->updated = 0;
845
+ return;
846
+ }
847
+
848
+ res->some.value10 = strtod(procfile_lineword(ff, 0, 2), NULL);
849
+ res->some.value60 = strtod(procfile_lineword(ff, 0, 4), NULL);
850
+ res->some.value300 = strtod(procfile_lineword(ff, 0, 6), NULL);
851
+
852
+ if (lines > 2) {
853
+ res->full.value10 = strtod(procfile_lineword(ff, 1, 2), NULL);
854
+ res->full.value60 = strtod(procfile_lineword(ff, 1, 4), NULL);
855
+ res->full.value300 = strtod(procfile_lineword(ff, 1, 6), NULL);
856
+ }
857
+
858
+ res->updated = 1;
859
+
860
+ if (unlikely(res->some.enabled == CONFIG_BOOLEAN_AUTO)) {
861
+ res->some.enabled = CONFIG_BOOLEAN_YES;
862
+ if (lines > 2) {
863
+ res->full.enabled = CONFIG_BOOLEAN_YES;
864
+ } else {
865
+ res->full.enabled = CONFIG_BOOLEAN_NO;
866
+ }
867
+ }
868
+ }
869
+}
870
+
871
static inline void cgroup_read_memory(struct memory *mem, char parent_cg_is_unified) {
872
static procfile *ff = NULL;
873
@@ -946,6 +1016,9 @@ static inline void cgroup_read(struct cgroup *cg) {
1016
cgroup2_read_blkio(&cg->io_service_bytes, 0);
1017
cgroup2_read_blkio(&cg->io_serviced, 4);
1018
cgroup2_read_cpuacct_stat(&cg->cpuacct_stat);
1019
+ cgroup2_read_pressure(&cg->cpu_pressure);
1020
+ cgroup2_read_pressure(&cg->io_pressure);
1021
+ cgroup2_read_pressure(&cg->memory_pressure);
1022
cgroup_read_memory(&cg->memory, 1);
1023
}
1024
}
@@ -1236,6 +1309,12 @@ static inline struct cgroup *cgroup_add(const char *id) {
1309
return cg;
1310
}
1311
1312
+static inline void free_pressure(struct pressure *res) {
1313
+ if (res->some.st) rrdset_is_obsolete(res->some.st);
1314
+ if (res->full.st) rrdset_is_obsolete(res->full.st);
1315
+ freez(res->filename);
1316
+}
1317
+
1318
static inline void cgroup_free(struct cgroup *cg) {
1319
debug(D_CGROUP, "Removing cgroup '%s' with chart id '%s' (was %s and %s)", cg->id, cg->chart_id, (cg->enabled)?"enabled":"disabled", (cg->available)?"available":"not available");
1320
@@ -1284,6 +1363,10 @@ static inline void cgroup_free(struct cgroup *cg) {
1363
freez(cg->io_merged.filename);
1364
freez(cg->io_queued.filename);
1365
1366
+ free_pressure(&cg->cpu_pressure);
1367
+ free_pressure(&cg->io_pressure);
1368
+ free_pressure(&cg->memory_pressure);
1369
+
1370
freez(cg->id);
1371
freez(cg->chart_id);
1372
freez(cg->chart_title);
@@ -1748,6 +1831,42 @@ static inline void find_all_cgroups() {
1831
else
1832
debug(D_CGROUP, "memory.swap file for cgroup '%s': '%s' does not exist.", cg->id, filename);
1833
}
1834
+
1835
+ if (unlikely(cgroup_enable_pressure_cpu && !cg->cpu_pressure.filename)) {
1836
+ snprintfz(filename, FILENAME_MAX, "%s%s/cpu.pressure", cgroup_unified_base, cg->id);
1837
+ if (likely(stat(filename, &buf) != -1)) {
1838
+ cg->cpu_pressure.filename = strdupz(filename);
1839
+ cg->cpu_pressure.some.enabled = cgroup_enable_pressure_cpu;
1840
+ cg->cpu_pressure.full.enabled = CONFIG_BOOLEAN_NO;
1841
+ debug(D_CGROUP, "cpu.pressure filename for cgroup '%s': '%s'", cg->id, cg->cpu_pressure.filename);
1842
+ } else {
1843
+ debug(D_CGROUP, "cpu.pressure file for cgroup '%s': '%s' does not exist", cg->id, filename);
1844
+ }
1845
+ }
1846
+
1847
+ if (unlikely((cgroup_enable_pressure_io_some || cgroup_enable_pressure_io_full) && !cg->io_pressure.filename)) {
1848
+ snprintfz(filename, FILENAME_MAX, "%s%s/io.pressure", cgroup_unified_base, cg->id);
1849
+ if (likely(stat(filename, &buf) != -1)) {
1850
+ cg->io_pressure.filename = strdupz(filename);
1851
+ cg->io_pressure.some.enabled = cgroup_enable_pressure_io_some;
1852
+ cg->io_pressure.full.enabled = cgroup_enable_pressure_io_full;
1853
+ debug(D_CGROUP, "io.pressure filename for cgroup '%s': '%s'", cg->id, cg->io_pressure.filename);
1854
+ } else {
1855
+ debug(D_CGROUP, "io.pressure file for cgroup '%s': '%s' does not exist", cg->id, filename);
1856
+ }
1857
+ }
1858
+
1859
+ if (unlikely((cgroup_enable_pressure_memory_some || cgroup_enable_pressure_memory_full) && !cg->memory_pressure.filename)) {
1860
+ snprintfz(filename, FILENAME_MAX, "%s%s/memory.pressure", cgroup_unified_base, cg->id);
1861
+ if (likely(stat(filename, &buf) != -1)) {
1862
+ cg->memory_pressure.filename = strdupz(filename);
1863
+ cg->memory_pressure.some.enabled = cgroup_enable_pressure_memory_some;
1864
+ cg->memory_pressure.full.enabled = cgroup_enable_pressure_memory_full;
1865
+ debug(D_CGROUP, "memory.pressure filename for cgroup '%s': '%s'", cg->id, cg->memory_pressure.filename);
1866
+ } else {
1867
+ debug(D_CGROUP, "memory.pressure file for cgroup '%s': '%s' does not exist", cg->id, filename);
1868
+ }
1869
+ }
1870
}
1871
}
1872
@@ -3364,6 +3483,156 @@ void update_cgroup_charts(int update_every) {
3483
rrddim_set(cg->st_merged_ops, "write", cg->io_merged.Write);
3484
rrdset_done(cg->st_merged_ops);
3485
}
3486
+
3487
+ if (cg->options & CGROUP_OPTIONS_IS_UNIFIED) {
3488
+ struct pressure *res = &cg->cpu_pressure;
3489
+ if (likely(res->updated && res->some.enabled)) {
3490
+ if (unlikely(!res->some.st)) {
3491
+ RRDSET *chart;
3492
+ snprintfz(title, CHART_TITLE_MAX, "CPU pressure for cgroup %s", cg->chart_title);
3493
+
3494
+ chart = res->some.st = rrdset_create_localhost(
3495
+ cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3496
+ , "cpu_pressure"
3497
+ , NULL
3498
+ , "cpu"
3499
+ , "cgroup.cpu_pressure"
3500
+ , title
3501
+ , "percentage"
3502
+ , PLUGIN_CGROUPS_NAME
3503
+ , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3504
+ , cgroup_containers_chart_priority + 2200
3505
+ , update_every,
3506
+ RRDSET_TYPE_LINE);
3507
+
3508
+ res->some.rd10 = rrddim_add(chart, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3509
+ res->some.rd60 = rrddim_add(chart, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3510
+ res->some.rd300 = rrddim_add(chart, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3511
+ } else {
3512
+ rrdset_next(res->some.st);
3513
+ }
3514
+
3515
+ update_pressure_chart(&res->some);
3516
+ }
3517
+
3518
+ res = &cg->memory_pressure;
3519
+ if (likely(res->updated && res->some.enabled)) {
3520
+ if (unlikely(!res->some.st)) {
3521
+ RRDSET *chart;
3522
+ snprintfz(title, CHART_TITLE_MAX, "Memory pressure for cgroup %s", cg->chart_title);
3523
+
3524
+ chart = res->some.st = rrdset_create_localhost(
3525
+ cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3526
+ , "mem_pressure"
3527
+ , NULL
3528
+ , "mem"
3529
+ , "cgroup.memory_pressure"
3530
+ , title
3531
+ , "percentage"
3532
+ , PLUGIN_CGROUPS_NAME
3533
+ , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3534
+ , cgroup_containers_chart_priority + 2300
3535
+ , update_every,
3536
+ RRDSET_TYPE_LINE);
3537
+
3538
+ res->some.rd10 = rrddim_add(chart, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3539
+ res->some.rd60 = rrddim_add(chart, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3540
+ res->some.rd300 = rrddim_add(chart, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3541
+ } else {
3542
+ rrdset_next(res->some.st);
3543
+ }
3544
+
3545
+ update_pressure_chart(&res->some);
3546
+ }
3547
+
3548
+ if (likely(res->updated && res->full.enabled)) {
3549
+ if (unlikely(!res->full.st)) {
3550
+ RRDSET *chart;
3551
+ snprintfz(title, CHART_TITLE_MAX, "Memory full pressure for cgroup %s", cg->chart_title);
3552
+
3553
+ chart = res->full.st = rrdset_create_localhost(
3554
+ cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3555
+ , "mem_full_pressure"
3556
+ , NULL
3557
+ , "mem"
3558
+ , "cgroup.memory_full_pressure"
3559
+ , title
3560
+ , "percentage"
3561
+ , PLUGIN_CGROUPS_NAME
3562
+ , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3563
+ , cgroup_containers_chart_priority + 2350
3564
+ , update_every,
3565
+ RRDSET_TYPE_LINE);
3566
+
3567
+ res->full.rd10 = rrddim_add(chart, "full 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3568
+ res->full.rd60 = rrddim_add(chart, "full 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3569
+ res->full.rd300 = rrddim_add(chart, "full 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3570
+ } else {
3571
+ rrdset_next(res->full.st);
3572
+ }
3573
+
3574
+ update_pressure_chart(&res->full);
3575
+ }
3576
+
3577
+ res = &cg->io_pressure;
3578
+ if (likely(res->updated && res->some.enabled)) {
3579
+ if (unlikely(!res->some.st)) {
3580
+ RRDSET *chart;
3581
+ snprintfz(title, CHART_TITLE_MAX, "I/O pressure for cgroup %s", cg->chart_title);
3582
+
3583
+ chart = res->some.st = rrdset_create_localhost(
3584
+ cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3585
+ , "io_pressure"
3586
+ , NULL
3587
+ , "disk"
3588
+ , "cgroup.io_pressure"
3589
+ , title
3590
+ , "percentage"
3591
+ , PLUGIN_CGROUPS_NAME
3592
+ , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3593
+ , cgroup_containers_chart_priority + 2400
3594
+ , update_every,
3595
+ RRDSET_TYPE_LINE);
3596
+
3597
+ res->some.rd10 = rrddim_add(chart, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3598
+ res->some.rd60 = rrddim_add(chart, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3599
+ res->some.rd300 = rrddim_add(chart, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3600
+ } else {
3601
+ rrdset_next(res->some.st);
3602
+ }
3603
+
3604
+ update_pressure_chart(&res->some);
3605
+ }
3606
+
3607
+ if (likely(res->updated && res->full.enabled)) {
3608
+ if (unlikely(!res->full.st)) {
3609
+ RRDSET *chart;
3610
+ snprintfz(title, CHART_TITLE_MAX, "I/O full pressure for cgroup %s", cg->chart_title);
3611
+
3612
+ chart = res->full.st = rrdset_create_localhost(
3613
+ cgroup_chart_type(type, cg->chart_id, RRD_ID_LENGTH_MAX)
3614
+ , "io_full_pressure"
3615
+ , NULL
3616
+ , "disk"
3617
+ , "cgroup.io_full_pressure"
3618
+ , title
3619
+ , "percentage"
3620
+ , PLUGIN_CGROUPS_NAME
3621
+ , PLUGIN_CGROUPS_MODULE_CGROUPS_NAME
3622
+ , cgroup_containers_chart_priority + 2450
3623
+ , update_every,
3624
+ RRDSET_TYPE_LINE);
3625
+
3626
+ res->full.rd10 = rrddim_add(chart, "full 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3627
+ res->full.rd60 = rrddim_add(chart, "full 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3628
+ res->full.rd300 = rrddim_add(chart, "full 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
3629
+ } else {
3630
+ rrdset_next(res->full.st);
3631
+ }
3632
+
3633
+ update_pressure_chart(&res->full);
3634
+ }
3635
+ }
3636
}
3637
3638
if(likely(cgroup_enable_systemd_services))
collectors/proc.plugin/README.md
+1
@@ -18,6 +18,7 @@
18
- `/proc/interrupts` (total and per core hardware interrupts)
19
- `/proc/softirqs` (total and per core software interrupts)
20
- `/proc/loadavg` (system load and total processes running)
21
+- `/proc/pressure/{cpu,memory,io}` (pressure stall information)
22
- `/proc/sys/kernel/random/entropy_avail` (random numbers pool availability - used in cryptography)
23
- `/sys/class/power_supply` (power supply properties)
24
- `ipc` (IPC semaphores and message queues)
collectors/proc.plugin/plugin_proc.c
+3
@@ -21,6 +21,9 @@ static struct proc_module {
21
{ .name = "/proc/loadavg", .dim = "loadavg", .func = do_proc_loadavg },
22
{ .name = "/proc/sys/kernel/random/entropy_avail", .dim = "entropy", .func = do_proc_sys_kernel_random_entropy_avail },
23
24
+ // pressure metrics
25
+ { .name = "/proc/pressure", .dim = "pressure", .func = do_proc_pressure },
26
+
27
// CPU metrics
28
{ .name = "/proc/interrupts", .dim = "interrupts", .func = do_proc_interrupts },
29
{ .name = "/proc/softirqs", .dim = "softirqs", .func = do_proc_softirqs },
collectors/proc.plugin/plugin_proc.h
+2
@@ -40,6 +40,7 @@ extern int do_proc_net_rpc_nfsd(int update_every, usec_t dt);
40
extern int do_proc_sys_kernel_random_entropy_avail(int update_every, usec_t dt);
41
extern int do_proc_interrupts(int update_every, usec_t dt);
42
extern int do_proc_softirqs(int update_every, usec_t dt);
43
+extern int do_proc_pressure(int update_every, usec_t dt);
44
extern int do_sys_kernel_mm_ksm(int update_every, usec_t dt);
45
extern int do_sys_block_zram(int update_every, usec_t dt);
46
extern int do_proc_loadavg(int update_every, usec_t dt);
@@ -66,6 +67,7 @@ extern void netdev_rename_device_add(const char *host_device, const char *contai
67
extern void netdev_rename_device_del(const char *host_device);
68
69
#include "proc_self_mountinfo.h"
70
+#include "proc_pressure.h"
71
#include "zfs_common.h"
72
73
#else // (TARGET_OS == OS_LINUX)
collectors/proc.plugin/proc_pressure.c
new
+177
@@ -0,0 +1,177 @@
1
+// SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+#include "plugin_proc.h"
4
+
5
+#define PLUGIN_PROC_MODULE_PRESSURE_NAME "/proc/pressure"
6
+#define CONFIG_SECTION_PLUGIN_PROC_PRESSURE "plugin:" PLUGIN_PROC_CONFIG_NAME ":" PLUGIN_PROC_MODULE_PRESSURE_NAME
7
+
8
+// linux calculates this every 2 seconds, see kernel/sched/psi.c PSI_FREQ
9
+#define MIN_PRESSURE_UPDATE_EVERY 2
10
+
11
+
12
+static struct pressure resources[PRESSURE_NUM_RESOURCES] = {
13
+ {
14
+ .some = { .id = "cpu_pressure", .title = "CPU Pressure" },
15
+ },
16
+ {
17
+ .some = { .id = "memory_some_pressure", .title = "Memory Pressure" },
18
+ .full = { .id = "memory_full_pressure", .title = "Memory Full Pressure" },
19
+ },
20
+ {
21
+ .some = { .id = "io_some_pressure", .title = "I/O Pressure" },
22
+ .full = { .id = "io_full_pressure", .title = "I/O Full Pressure" },
23
+ },
24
+};
25
+
26
+static struct {
27
+ procfile *pf;
28
+ const char *name; // metric file name
29
+ const char *family; // webui section name
30
+ int section_priority;
31
+} resource_info[PRESSURE_NUM_RESOURCES] = {
32
+ { .name = "cpu", .family = "cpu", .section_priority = NETDATA_CHART_PRIO_SYSTEM_CPU },
33
+ { .name = "memory", .family = "ram", .section_priority = NETDATA_CHART_PRIO_SYSTEM_RAM },
34
+ { .name = "io", .family = "disk", .section_priority = NETDATA_CHART_PRIO_SYSTEM_IO },
35
+};
36
+
37
+void update_pressure_chart(struct pressure_chart *chart) {
38
+ rrddim_set_by_pointer(chart->st, chart->rd10, (collected_number)(chart->value10 * 100));
39
+ rrddim_set_by_pointer(chart->st, chart->rd60, (collected_number) (chart->value60 * 100));
40
+ rrddim_set_by_pointer(chart->st, chart->rd300, (collected_number) (chart->value300 * 100));
41
+
42
+ rrdset_done(chart->st);
43
+}
44
+
45
+int do_proc_pressure(int update_every, usec_t dt) {
46
+ int fail_count = 0;
47
+ int i;
48
+
49
+ static usec_t next_pressure_dt = 0;
50
+ static char *base_path = NULL;
51
+
52
+ update_every = (update_every < MIN_PRESSURE_UPDATE_EVERY) ? MIN_PRESSURE_UPDATE_EVERY : update_every;
53
+
54
+ if (next_pressure_dt <= dt) {
55
+ next_pressure_dt = update_every * USEC_PER_SEC;
56
+ } else {
57
+ next_pressure_dt -= dt;
58
+ return 0;
59
+ }
60
+
61
+ if (unlikely(!base_path)) {
62
+ base_path = config_get(CONFIG_SECTION_PLUGIN_PROC_PRESSURE, "base path of pressure metrics", "/proc/pressure");
63
+ }
64
+
65
+ for (i = 0; i < PRESSURE_NUM_RESOURCES; i++) {
66
+ procfile *ff = resource_info[i].pf;
67
+ int do_some = resources[i].some.enabled, do_full = resources[i].full.enabled;
68
+
69
+ if (unlikely(!ff)) {
70
+ char filename[FILENAME_MAX + 1];
71
+ char config_key[CONFIG_MAX_NAME + 1];
72
+
73
+ snprintfz(filename
74
+ , FILENAME_MAX
75
+ , "%s%s/%s"
76
+ , netdata_configured_host_prefix
77
+ , base_path
78
+ , resource_info[i].name);
79
+
80
+ snprintfz(config_key, CONFIG_MAX_NAME, "enable %s some pressure", resource_info[i].name);
81
+ do_some = config_get_boolean(CONFIG_SECTION_PLUGIN_PROC_PRESSURE, config_key, CONFIG_BOOLEAN_YES);
82
+ resources[i].some.enabled = do_some;
83
+ if (resources[i].full.id) {
84
+ snprintfz(config_key, CONFIG_MAX_NAME, "enable %s full pressure", resource_info[i].name);
85
+ do_full = config_get_boolean(CONFIG_SECTION_PLUGIN_PROC_PRESSURE, config_key, CONFIG_BOOLEAN_YES);
86
+ resources[i].full.enabled = do_full;
87
+ }
88
+
89
+ ff = procfile_open(filename, " =", PROCFILE_FLAG_DEFAULT);
90
+ if (unlikely(!ff)) {
91
+ error("Cannot read pressure information from %s.", filename);
92
+ fail_count++;
93
+ continue;
94
+ }
95
+ }
96
+
97
+ ff = procfile_readall(ff);
98
+ resource_info[i].pf = ff;
99
+ if (unlikely(!ff)) {
100
+ continue;
101
+ }
102
+
103
+ size_t lines = procfile_lines(ff);
104
+ if (unlikely(lines < 1)) {
105
+ error("%s has no lines.", procfile_filename(ff));
106
+ fail_count++;
107
+ continue;
108
+ }
109
+
110
+ struct pressure_chart *chart;
111
+ if (do_some) {
112
+ chart = &resources[i].some;
113
+ if (unlikely(!chart->st)) {
114
+ chart->st = rrdset_create_localhost(
115
+ "system"
116
+ , chart->id
117
+ , NULL
118
+ , resource_info[i].family
119
+ , NULL
120
+ , chart->title
121
+ , "percentage"
122
+ , PLUGIN_PROC_NAME
123
+ , PLUGIN_PROC_MODULE_PRESSURE_NAME
124
+ , resource_info[i].section_priority + 40
125
+ , update_every
126
+ , RRDSET_TYPE_LINE
127
+ );
128
+ chart->rd10 = rrddim_add(chart->st, "some 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
129
+ chart->rd60 = rrddim_add(chart->st, "some 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
130
+ chart->rd300 = rrddim_add(chart->st, "some 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
131
+ } else {
132
+ rrdset_next(chart->st);
133
+ }
134
+
135
+ chart->value10 = strtod(procfile_lineword(ff, 0, 2), NULL);
136
+ chart->value60 = strtod(procfile_lineword(ff, 0, 4), NULL);
137
+ chart->value300 = strtod(procfile_lineword(ff, 0, 6), NULL);
138
+ update_pressure_chart(chart);
139
+ }
140
+
141
+ if (do_full && lines > 2) {
142
+ chart = &resources[i].full;
143
+ if (unlikely(!chart->st)) {
144
+ chart->st = rrdset_create_localhost(
145
+ "system"
146
+ , chart->id
147
+ , NULL
148
+ , resource_info[i].family
149
+ , NULL
150
+ , chart->title
151
+ , "percentage"
152
+ , PLUGIN_PROC_NAME
153
+ , PLUGIN_PROC_MODULE_PRESSURE_NAME
154
+ , resource_info[i].section_priority + 45
155
+ , update_every
156
+ , RRDSET_TYPE_LINE
157
+ );
158
+ chart->rd10 = rrddim_add(chart->st, "full 10", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
159
+ chart->rd60 = rrddim_add(chart->st, "full 60", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
160
+ chart->rd300 = rrddim_add(chart->st, "full 300", NULL, 1, 100, RRD_ALGORITHM_ABSOLUTE);
161
+ } else {
162
+ rrdset_next(chart->st);
163
+ }
164
+
165
+ chart->value10 = strtod(procfile_lineword(ff, 1, 2), NULL);
166
+ chart->value60 = strtod(procfile_lineword(ff, 1, 4), NULL);
167
+ chart->value300 = strtod(procfile_lineword(ff, 1, 6), NULL);
168
+ update_pressure_chart(chart);
169
+ }
170
+ }
171
+
172
+ if (PRESSURE_NUM_RESOURCES == fail_count) {
173
+ return 1;
174
+ }
175
+
176
+ return 0;
177
+}
collectors/proc.plugin/proc_pressure.h
new
+31
@@ -0,0 +1,31 @@
1
+// SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+#ifndef NETDATA_PROC_PRESSURE_H
4
+#define NETDATA_PROC_PRESSURE_H
5
+
6
+#define PRESSURE_NUM_RESOURCES 3
7
+
8
+struct pressure {
9
+ int updated;
10
+ char *filename;
11
+
12
+ struct pressure_chart {
13
+ int enabled;
14
+
15
+ const char *id;
16
+ const char *title;
17
+
18
+ double value10;
19
+ double value60;
20
+ double value300;
21
+
22
+ RRDSET *st;
23
+ RRDDIM *rd10;
24
+ RRDDIM *rd60;
25
+ RRDDIM *rd300;
26
+ } some, full;
27
+};
28
+
29
+extern void update_pressure_chart(struct pressure_chart *chart);
30
+
31
+#endif //NETDATA_PROC_PRESSURE_H
web/gui/dashboard_info.js
+25
@@ -718,6 +718,31 @@ netdataDashboard.context = {
718
height: 0.7
719
},
720
721
+ 'system.cpu_pressure': {
722
+ info: '<a href="https://www.kernel.org/doc/html/latest/accounting/psi.html">Pressure Stall Information</a> ' +
723
+ 'identifies and quantifies the disruptions caused by resource contentions. ' +
724
+ 'The "some" line indicates the share of time in which at least <b>some</b> tasks are stalled on CPU. ' +
725
+ 'The ratios (in %) are tracked as recent trends over 10-, 60-, and 300-second windows.'
726
+ },
727
+
728
+ 'system.memory_some_pressure': {
729
+ info: '<a href="https://www.kernel.org/doc/html/latest/accounting/psi.html">Pressure Stall Information</a> ' +
730
+ 'identifies and quantifies the disruptions caused by resource contentions. ' +
731
+ 'The "some" line indicates the share of time in which at least <b>some</b> tasks are stalled on memory. ' +
732
+ 'The "full" line indicates the share of time in which <b>all non-idle</b> tasks are stalled on memory simultaneously. ' +
733
+ 'In this state actual CPU cycles are going to waste, and a workload that spends extended time in this state is considered to be thrashing. ' +
734
+ 'The ratios (in %) are tracked as recent trends over 10-, 60-, and 300-second windows.'
735
+ },
736
+
737
+ 'system.io_some_pressure': {
738
+ info: '<a href="https://www.kernel.org/doc/html/latest/accounting/psi.html">Pressure Stall Information</a> ' +
739
+ 'identifies and quantifies the disruptions caused by resource contentions. ' +
740
+ 'The "some" line indicates the share of time in which at least <b>some</b> tasks are stalled on I/O. ' +
741
+ 'The "full" line indicates the share of time in which <b>all non-idle</b> tasks are stalled on I/O simultaneously. ' +
742
+ 'In this state actual CPU cycles are going to waste, and a workload that spends extended time in this state is considered to be thrashing. ' +
743
+ 'The ratios (in %) are tracked as recent trends over 10-, 60-, and 300-second windows.'
744
+ },
745
+
746
'system.io': {
747
info: function (os) {
748
var s = 'Total Disk I/O, for all physical disks. You can get detailed information about each disk at the <a href="#menu_disk">Disks</a> section and per application Disk usage at the <a href="#menu_apps">Applications Monitoring</a> section.';