Fix cpuidle statistics in containers (#5065)
Vladimir Kobal committed
Dec 28, 2018 at 15:44 UTC
90d6eab6cf0de3815115c90e978d20244b057fc6
1 file changed
+30
-17
collectors/proc.plugin/proc_stat.c
+30
-17
@@ -265,24 +265,34 @@ static void* wake_cpu_thread(void* core) {
265
pthread_t thread;
266
cpu_set_t cpu_set;
267
static size_t cpu_wakeups = 0;
268
+ static int errors = 0;
269
270
CPU_ZERO(&cpu_set);
271
CPU_SET(*(int*)core, &cpu_set);
272
273
thread = pthread_self();
273
- if(unlikely(pthread_setaffinity_np(thread, sizeof(cpu_set_t), &cpu_set)))
274
- error("Cannot set CPU affinity");
274
+ if(unlikely(pthread_setaffinity_np(thread, sizeof(cpu_set_t), &cpu_set))) {
275
+ if(unlikely(errors < 8)) {
276
+ error("Cannot set CPU affinity for core %d", *(int*)core);
277
+ errors++;
278
+ }
279
+ else if(unlikely(errors < 9)) {
280
+ error("CPU affinity errors are disabled");
281
+ errors++;
282
+ }
283
+ }
284
276
- // Make the CPU core do something
285
+ // Make the CPU core do something to force it to update its idle counters
286
cpu_wakeups++;
287
288
return 0;
289
}
290
282
-static int read_schedstat(char* schedstat_filename, struct per_core_cpuidle_chart **cpuidle_charts_address, size_t cores_found) {
291
+static int read_schedstat(char *schedstat_filename, struct per_core_cpuidle_chart **cpuidle_charts_address, size_t *schedstat_cores_found) {
292
static size_t cpuidle_charts_len = 0;
293
static procfile *ff = NULL;
294
struct per_core_cpuidle_chart *cpuidle_charts = *cpuidle_charts_address;
295
+ size_t cores_found = 0;
296
297
if(unlikely(!ff)) {
298
ff = procfile_open(schedstat_filename, " \t:", PROCFILE_FLAG_DEFAULT);
@@ -295,13 +305,6 @@ static int read_schedstat(char* schedstat_filename, struct per_core_cpuidle_char
305
size_t lines = procfile_lines(ff), l;
306
size_t words;
307
298
- if(unlikely(cpuidle_charts_len < cores_found)) {
299
- cpuidle_charts = reallocz(cpuidle_charts, sizeof(struct per_core_cpuidle_chart) * cores_found);
300
- *cpuidle_charts_address = cpuidle_charts;
301
- memset(cpuidle_charts + cpuidle_charts_len, 0, sizeof(struct per_core_cpuidle_chart) * (cores_found - cpuidle_charts_len));
302
- cpuidle_charts_len = cores_found;
303
- }
304
-
308
for(l = 0; l < lines ;l++) {
309
char *row_key = procfile_lineword(ff, l, 0);
310
@@ -312,17 +315,26 @@ static int read_schedstat(char* schedstat_filename, struct per_core_cpuidle_char
315
error("Cannot read /proc/schedstat cpu line. Expected 9 params, read %zu.", words);
316
return 1;
317
}
318
+ cores_found++;
319
320
size_t core = str2ul(&row_key[3]);
321
if(unlikely(core >= cores_found)) {
318
- // Temporary workaround for issue 3945
319
- // error("Core %zu found but no more than %zu cores were expected.", core, cores_found);
322
+ error("Core %zu found but no more than %zu cores were expected.", core, cores_found);
323
return 1;
324
}
325
+
326
+ if(unlikely(cpuidle_charts_len < cores_found)) {
327
+ cpuidle_charts = reallocz(cpuidle_charts, sizeof(struct per_core_cpuidle_chart) * cores_found);
328
+ *cpuidle_charts_address = cpuidle_charts;
329
+ memset(cpuidle_charts + cpuidle_charts_len, 0, sizeof(struct per_core_cpuidle_chart) * (cores_found - cpuidle_charts_len));
330
+ cpuidle_charts_len = cores_found;
331
+ }
332
+
333
cpuidle_charts[core].active_time = str2ull(procfile_lineword(ff, l, 7)) / 1000;
334
}
335
}
336
337
+ *schedstat_cores_found = cores_found;
338
return 0;
339
}
340
@@ -969,15 +981,16 @@ int do_proc_stat(int update_every, usec_t dt) {
981
// --------------------------------------------------------------------
982
983
static struct per_core_cpuidle_chart *cpuidle_charts = NULL;
984
+ size_t schedstat_cores_found = 0;
985
973
- if(likely(do_cpuidle != CONFIG_BOOLEAN_NO && !read_schedstat(schedstat_filename, &cpuidle_charts, cores_found))) {
986
+ if(likely(do_cpuidle != CONFIG_BOOLEAN_NO && !read_schedstat(schedstat_filename, &cpuidle_charts, &schedstat_cores_found))) {
987
int cpu_states_updated = 0;
988
size_t core, state;
989
990
991
// proc.plugin runs on Linux systems only. Multi-platform compatibility is not needed here,
992
// so bare pthread functions are used to avoid unneeded overheads.
980
- for(core = 0; core < cores_found; core++) {
993
+ for(core = 0; core < schedstat_cores_found; core++) {
994
if(unlikely(!(cpuidle_charts[core].active_time - cpuidle_charts[core].last_active_time))) {
995
pthread_t thread;
996
@@ -989,8 +1002,8 @@ int do_proc_stat(int update_every, usec_t dt) {
1002
}
1003
}
1004
992
- if(unlikely(!cpu_states_updated || !read_schedstat(schedstat_filename, &cpuidle_charts, cores_found))) {
993
- for(core = 0; core < cores_found; core++) {
1005
+ if(unlikely(!cpu_states_updated || !read_schedstat(schedstat_filename, &cpuidle_charts, &schedstat_cores_found))) {
1006
+ for(core = 0; core < schedstat_cores_found; core++) {
1007
cpuidle_charts[core].last_active_time = cpuidle_charts[core].active_time;
1008
1009
int r = read_cpuidle_states(cpuidle_name_filename, cpuidle_time_filename, cpuidle_charts, core);