detect if the disk cannot keep up with data collection (#7139)
* Adjust dbengine flushing speed more dynamically * Added error tracking statistics for failure to flush events * Added alarm for dbengine flushing errors * Improved dbengine accounting for commited to be written pages
Markos Fountoulakis committed
Oct 24, 2019 at 19:43 UTC
88f966593abc5c7888e7c0be83780a97d4326ac2
10 files changed
+122
-51
configs.signatures
+1
-1
@@ -381,7 +381,7 @@ declare -A configs_signatures=(
381
['7deb236ec68a512b9bdd18e6a51d76f7']='python.d/mysql.conf'
382
['7e5fc1644aa7a54f9dbb1bd102521b09']='health.d/memcached.conf'
383
['7f13631183fbdf79c21c8e5a171e9b34']='health.d/zfs.conf'
384
- ['e48b89d4a97b96acf9a88970ab858c3b']='health.d/dbengine.conf'
384
+ ['8edc8c73a8f3ca40b32e27fe452c70f3']='health.d/dbengine.conf'
385
['7fb8184d56a27040e73261ed9c6fc76f']='health_alarm_notify.conf'
386
['80266bddd3df374923c750a6de91d120']='health.d/apache.conf'
387
['803a7f9dcb942eeac0fd764b9e3e38ca']='fping.conf'
daemon/global_statistics.c
+4
-1
@@ -544,7 +544,7 @@ void global_statistics_charts(void) {
544
if (host->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
545
++hosts_with_dbengine;
546
/* get localhost's DB engine's statistics */
547
- rrdeng_get_33_statistics(host->rrdeng_ctx, local_stats_array);
547
+ rrdeng_get_35_statistics(host->rrdeng_ctx, local_stats_array);
548
for (i = 0 ; i < RRDENG_NR_STATS ; ++i) {
549
/* aggregate statistics across hosts */
550
stats_array[i] += local_stats_array[i];
@@ -775,6 +775,7 @@ void global_statistics_charts(void) {
775
static RRDSET *st_errors = NULL;
776
static RRDDIM *rd_fs_errors = NULL;
777
static RRDDIM *rd_io_errors = NULL;
778
+ static RRDDIM *rd_flushing_errors = NULL;
779
780
if (unlikely(!st_errors)) {
781
st_errors = rrdset_create_localhost(
@@ -794,12 +795,14 @@ void global_statistics_charts(void) {
795
796
rd_io_errors = rrddim_add(st_errors, "I/O errors", NULL, 1, 1, RRD_ALGORITHM_INCREMENTAL);
797
rd_fs_errors = rrddim_add(st_errors, "FS errors", NULL, 1, 1, RRD_ALGORITHM_INCREMENTAL);
798
+ rd_flushing_errors = rrddim_add(st_errors, "flushing errors", NULL, 1, 1, RRD_ALGORITHM_INCREMENTAL);
799
}
800
else
801
rrdset_next(st_errors);
802
803
rrddim_set_by_pointer(st_errors, rd_io_errors, (collected_number)stats_array[30]);
804
rrddim_set_by_pointer(st_errors, rd_fs_errors, (collected_number)stats_array[31]);
805
+ rrddim_set_by_pointer(st_errors, rd_flushing_errors, (collected_number)stats_array[34]);
806
rrdset_done(st_errors);
807
}
808
database/engine/pagecache.c
+11
-11
@@ -214,7 +214,7 @@ static void pg_cache_release_pages(struct rrdengine_instance *ctx, unsigned numb
214
* This function returns the maximum number of pages allowed in the page cache.
215
* The caller must hold the page cache lock.
216
*/
217
-static inline unsigned long pg_cache_hard_limit(struct rrdengine_instance *ctx)
217
+unsigned long pg_cache_hard_limit(struct rrdengine_instance *ctx)
218
{
219
/* it's twice the number of producers since we pin 2 pages per producer */
220
return ctx->max_cache_pages + 2 * (unsigned long)ctx->stats.metric_API_producers;
@@ -225,7 +225,7 @@ static inline unsigned long pg_cache_hard_limit(struct rrdengine_instance *ctx)
225
* number of pages below that number.
226
* The caller must hold the page cache lock.
227
*/
228
-static inline unsigned long pg_cache_soft_limit(struct rrdengine_instance *ctx)
228
+unsigned long pg_cache_soft_limit(struct rrdengine_instance *ctx)
229
{
230
/* it's twice the number of producers since we pin 2 pages per producer */
231
return ctx->cache_pages_low_watermark + 2 * (unsigned long)ctx->stats.metric_API_producers;
@@ -1029,14 +1029,14 @@ static void init_replaceQ(struct rrdengine_instance *ctx)
1029
assert(0 == uv_rwlock_init(&pg_cache->replaceQ.lock));
1030
}
1031
1032
-static void init_commited_page_index(struct rrdengine_instance *ctx)
1032
+static void init_committed_page_index(struct rrdengine_instance *ctx)
1033
{
1034
struct page_cache *pg_cache = &ctx->pg_cache;
1035
1036
- pg_cache->commited_page_index.JudyL_array = (Pvoid_t) NULL;
1037
- assert(0 == uv_rwlock_init(&pg_cache->commited_page_index.lock));
1038
- pg_cache->commited_page_index.latest_corr_id = 0;
1039
- pg_cache->commited_page_index.nr_commited_pages = 0;
1036
+ pg_cache->committed_page_index.JudyL_array = (Pvoid_t) NULL;
1037
+ assert(0 == uv_rwlock_init(&pg_cache->committed_page_index.lock));
1038
+ pg_cache->committed_page_index.latest_corr_id = 0;
1039
+ pg_cache->committed_page_index.nr_committed_pages = 0;
1040
}
1041
1042
void init_page_cache(struct rrdengine_instance *ctx)
@@ -1049,7 +1049,7 @@ void init_page_cache(struct rrdengine_instance *ctx)
1049
1050
init_metrics_index(ctx);
1051
init_replaceQ(ctx);
1052
- init_commited_page_index(ctx);
1052
+ init_committed_page_index(ctx);
1053
}
1054
1055
void free_page_cache(struct rrdengine_instance *ctx)
@@ -1062,9 +1062,9 @@ void free_page_cache(struct rrdengine_instance *ctx)
1062
struct rrdeng_page_descr *descr;
1063
struct page_cache_descr *pg_cache_descr;
1064
1065
- /* Free commited page index */
1066
- ret_Judy = JudyLFreeArray(&pg_cache->commited_page_index.JudyL_array, PJE0);
1067
- assert(NULL == pg_cache->commited_page_index.JudyL_array);
1065
+ /* Free committed page index */
1066
+ ret_Judy = JudyLFreeArray(&pg_cache->committed_page_index.JudyL_array, PJE0);
1067
+ assert(NULL == pg_cache->committed_page_index.JudyL_array);
1068
bytes_freed += ret_Judy;
1069
1070
for (page_index = pg_cache->metrics_index.last_page_index ;
database/engine/pagecache.h
+5
-3
@@ -110,7 +110,7 @@ struct pg_cache_metrics_index {
110
};
111
112
/* gathers dirty pages to be written on disk */
113
-struct pg_cache_commited_page_index {
113
+struct pg_cache_committed_page_index {
114
uv_rwlock_t lock;
115
116
Pvoid_t JudyL_array;
@@ -122,7 +122,7 @@ struct pg_cache_commited_page_index {
122
*/
123
Word_t latest_corr_id;
124
125
- unsigned nr_commited_pages;
125
+ unsigned nr_committed_pages;
126
};
127
128
/*
@@ -140,7 +140,7 @@ struct page_cache { /* TODO: add statistics */
140
uv_rwlock_t pg_cache_rwlock; /* page cache lock */
141
142
struct pg_cache_metrics_index metrics_index;
143
- struct pg_cache_commited_page_index commited_page_index;
143
+ struct pg_cache_committed_page_index committed_page_index;
144
struct pg_cache_replaceQ replaceQ;
145
146
unsigned page_descriptors;
@@ -182,6 +182,8 @@ extern void init_page_cache(struct rrdengine_instance *ctx);
182
extern void free_page_cache(struct rrdengine_instance *ctx);
183
extern void pg_cache_add_new_metric_time(struct pg_cache_page_index *page_index, struct rrdeng_page_descr *descr);
184
extern void pg_cache_update_metric_times(struct pg_cache_page_index *page_index);
185
+extern unsigned long pg_cache_hard_limit(struct rrdengine_instance *ctx);
186
+extern unsigned long pg_cache_soft_limit(struct rrdengine_instance *ctx);
187
188
static inline void
189
pg_cache_atomic_get_pg_info(struct rrdeng_page_descr *descr, usec_t *end_timep, uint32_t *page_lengthp)
database/engine/rrdengine.c
+39
-17
@@ -5,9 +5,8 @@
5
6
rrdeng_stats_t global_io_errors = 0;
7
rrdeng_stats_t global_fs_errors = 0;
8
-rrdeng_stats_t global_pg_cache_warnings = 0;
9
-rrdeng_stats_t global_pg_cache_errors = 0;
8
rrdeng_stats_t rrdeng_reserved_file_descriptors = 0;
9
+rrdeng_stats_t global_flushing_errors = 0;
10
11
void sanity_check(void)
12
{
@@ -253,6 +252,7 @@ void flush_pages_cb(uv_fs_t* req)
252
{
253
struct rrdengine_worker_config* wc = req->loop->data;
254
struct rrdengine_instance *ctx = wc->ctx;
255
+ struct page_cache *pg_cache = &ctx->pg_cache;
256
struct extent_io_descriptor *xt_io_descr;
257
struct rrdeng_page_descr *descr;
258
struct page_cache_descr *pg_cache_descr;
@@ -290,6 +290,10 @@ void flush_pages_cb(uv_fs_t* req)
290
uv_fs_req_cleanup(req);
291
free(xt_io_descr->buf);
292
freez(xt_io_descr);
293
+
294
+ uv_rwlock_wrlock(&pg_cache->committed_page_index.lock);
295
+ pg_cache->committed_page_index.nr_committed_pages -= count;
296
+ uv_rwlock_wrunlock(&pg_cache->committed_page_index.lock);
297
}
298
299
/*
@@ -323,14 +327,14 @@ static int do_flush_pages(struct rrdengine_worker_config* wc, int force, struct
327
if (force) {
328
debug(D_RRDENGINE, "Asynchronous flushing of extent has been forced by page pressure.");
329
}
326
- uv_rwlock_wrlock(&pg_cache->commited_page_index.lock);
330
+ uv_rwlock_wrlock(&pg_cache->committed_page_index.lock);
331
for (Index = 0, count = 0, uncompressed_payload_length = 0,
328
- PValue = JudyLFirst(pg_cache->commited_page_index.JudyL_array, &Index, PJE0),
332
+ PValue = JudyLFirst(pg_cache->committed_page_index.JudyL_array, &Index, PJE0),
333
descr = unlikely(NULL == PValue) ? NULL : *PValue ;
334
335
descr != NULL && count != MAX_PAGES_PER_EXTENT ;
336
333
- PValue = JudyLNext(pg_cache->commited_page_index.JudyL_array, &Index, PJE0),
337
+ PValue = JudyLNext(pg_cache->committed_page_index.JudyL_array, &Index, PJE0),
338
descr = unlikely(NULL == PValue) ? NULL : *PValue) {
339
uint8_t page_write_pending;
340
@@ -350,12 +354,11 @@ static int do_flush_pages(struct rrdengine_worker_config* wc, int force, struct
354
rrdeng_page_descr_mutex_unlock(ctx, descr);
355
356
if (page_write_pending) {
353
- ret = JudyLDel(&pg_cache->commited_page_index.JudyL_array, Index, PJE0);
357
+ ret = JudyLDel(&pg_cache->committed_page_index.JudyL_array, Index, PJE0);
358
assert(1 == ret);
355
- --pg_cache->commited_page_index.nr_commited_pages;
359
}
360
}
358
- uv_rwlock_wrunlock(&pg_cache->commited_page_index.lock);
361
+ uv_rwlock_wrunlock(&pg_cache->committed_page_index.lock);
362
363
if (!count) {
364
debug(D_RRDENGINE, "%s: no pages eligible for flushing.", __func__);
@@ -648,6 +651,9 @@ void async_cb(uv_async_t *handle)
651
debug(D_RRDENGINE, "%s called, active=%d.", __func__, uv_is_active((uv_handle_t *)handle));
652
}
653
654
+/* Flushes dirty pages when timer expires */
655
+#define TIMER_PERIOD_MS (1000)
656
+
657
void timer_cb(uv_timer_t* handle)
658
{
659
struct rrdengine_worker_config* wc = handle->data;
@@ -657,12 +663,31 @@ void timer_cb(uv_timer_t* handle)
663
rrdeng_test_quota(wc);
664
debug(D_RRDENGINE, "%s: timeout reached.", __func__);
665
if (likely(!wc->now_deleting.data)) {
660
- unsigned total_bytes, bytes_written;
661
-
666
/* There is free space so we can write to disk */
667
+ struct rrdengine_instance *ctx = wc->ctx;
668
+ struct page_cache *pg_cache = &ctx->pg_cache;
669
+ unsigned long total_bytes, bytes_written, nr_committed_pages, bytes_to_write = 0, producers, low_watermark,
670
+ high_watermark;
671
+
672
+ uv_rwlock_wrlock(&pg_cache->committed_page_index.lock);
673
+ nr_committed_pages = pg_cache->committed_page_index.nr_committed_pages;
674
+ uv_rwlock_wrunlock(&pg_cache->committed_page_index.lock);
675
+ producers = ctx->stats.metric_API_producers;
676
+ /* are flushable pages more than 25% of the maximum page cache size */
677
+ high_watermark = (ctx->max_cache_pages * 25LLU) / 100;
678
+ low_watermark = (ctx->max_cache_pages * 5LLU) / 100; /* 5%, must be smaller than high_watermark */
679
+
680
+ if (nr_committed_pages > producers &&
681
+ /* committed to be written pages are more than the produced number */
682
+ nr_committed_pages - producers > high_watermark) {
683
+ /* Flushing speed must increase to stop page cache from filling with dirty pages */
684
+ bytes_to_write = (nr_committed_pages - producers - low_watermark) * RRDENG_BLOCK_SIZE;
685
+ }
686
+ bytes_to_write = MAX(DATAFILE_IDEAL_IO_SIZE, bytes_to_write);
687
+
688
debug(D_RRDENGINE, "Flushing pages to disk.");
689
for (total_bytes = bytes_written = do_flush_pages(wc, 0, NULL) ;
665
- bytes_written && (total_bytes < DATAFILE_IDEAL_IO_SIZE) ;
690
+ bytes_written && (total_bytes < bytes_to_write) ;
691
total_bytes += bytes_written) {
692
bytes_written = do_flush_pages(wc, 0, NULL);
693
}
@@ -675,9 +700,6 @@ void timer_cb(uv_timer_t* handle)
700
#endif
701
}
702
678
-/* Flushes dirty pages when timer expires */
679
-#define TIMER_PERIOD_MS (1000)
680
-
703
#define MAX_CMD_BATCH_SIZE (256)
704
705
void rrdeng_worker(void* arg)
@@ -771,8 +793,8 @@ void rrdeng_worker(void* arg)
793
/* First I/O should be enough to call completion */
794
bytes_written = do_flush_pages(wc, 1, cmd.completion);
795
if (bytes_written) {
774
- while (do_flush_pages(wc, 1, NULL)) {
775
- ; /* Force flushing of all commited pages. */
796
+ while (do_flush_pages(wc, 1, NULL) && likely(!wc->now_deleting.data)) {
797
+ ; /* Force flushing of all committed pages if there is free space. */
798
}
799
}
800
break;
@@ -789,7 +811,7 @@ void rrdeng_worker(void* arg)
811
}
812
info("Shutting down RRD engine event loop.");
813
while (do_flush_pages(wc, 1, NULL)) {
792
- ; /* Force flushing of all commited pages. */
814
+ ; /* Force flushing of all committed pages. */
815
}
816
wal_flush_transaction_buffer(wc);
817
uv_run(loop, UV_RUN_DEFAULT);
database/engine/rrdengine.h
+3
@@ -148,6 +148,7 @@ struct rrdengine_statistics {
148
rrdeng_stats_t page_cache_descriptors;
149
rrdeng_stats_t io_errors;
150
rrdeng_stats_t fs_errors;
151
+ rrdeng_stats_t flushing_errors;
152
};
153
154
/* I/O errors global counter */
@@ -156,6 +157,8 @@ extern rrdeng_stats_t global_io_errors;
157
extern rrdeng_stats_t global_fs_errors;
158
/* number of File-Descriptors that have been reserved by dbengine */
159
extern rrdeng_stats_t rrdeng_reserved_file_descriptors;
160
+/* inability to flush global counter */
161
+extern rrdeng_stats_t global_flushing_errors;
162
163
struct rrdengine_instance {
164
struct rrdengine_worker_config worker_config;
database/engine/rrdengineapi.c
+26
-11
@@ -173,9 +173,9 @@ void rrdeng_store_metric_next(RRDDIM *rd, usec_t point_in_time, storage_number n
173
174
handle->descr = descr;
175
176
- uv_rwlock_wrlock(&pg_cache->commited_page_index.lock);
177
- handle->page_correlation_id = pg_cache->commited_page_index.latest_corr_id++;
178
- uv_rwlock_wrunlock(&pg_cache->commited_page_index.lock);
176
+ uv_rwlock_wrlock(&pg_cache->committed_page_index.lock);
177
+ handle->page_correlation_id = pg_cache->committed_page_index.latest_corr_id++;
178
+ uv_rwlock_wrunlock(&pg_cache->committed_page_index.lock);
179
180
if (0 == rd->rrdset->rrddim_page_alignment) {
181
/* this is the leading dimension that defines chart alignment */
@@ -614,18 +614,31 @@ void rrdeng_commit_page(struct rrdengine_instance *ctx, struct rrdeng_page_descr
614
{
615
struct page_cache *pg_cache = &ctx->pg_cache;
616
Pvoid_t *PValue;
617
+ unsigned nr_committed_pages;
618
619
if (unlikely(NULL == descr)) {
619
- debug(D_RRDENGINE, "%s: page descriptor is NULL, page has already been force-commited.", __func__);
620
+ debug(D_RRDENGINE, "%s: page descriptor is NULL, page has already been force-committed.", __func__);
621
return;
622
}
623
assert(descr->page_length);
624
624
- uv_rwlock_wrlock(&pg_cache->commited_page_index.lock);
625
- PValue = JudyLIns(&pg_cache->commited_page_index.JudyL_array, page_correlation_id, PJE0);
625
+ uv_rwlock_wrlock(&pg_cache->committed_page_index.lock);
626
+ PValue = JudyLIns(&pg_cache->committed_page_index.JudyL_array, page_correlation_id, PJE0);
627
*PValue = descr;
627
- ++pg_cache->commited_page_index.nr_commited_pages;
628
- uv_rwlock_wrunlock(&pg_cache->commited_page_index.lock);
628
+ nr_committed_pages = ++pg_cache->committed_page_index.nr_committed_pages;
629
+ uv_rwlock_wrunlock(&pg_cache->committed_page_index.lock);
630
+
631
+ if (nr_committed_pages >= (pg_cache_hard_limit(ctx) - (unsigned long)ctx->stats.metric_API_producers) / 2) {
632
+ /* 50% of pages have not been committed yet */
633
+ if (0 == (unsigned long)ctx->stats.flushing_errors) {
634
+ /* only print the first time */
635
+ error("Failed to flush quickly enough in dbengine instance \"%s\""
636
+ ". Metric data will not be stored in the database"
637
+ ", please reduce disk load or use a faster disk.", ctx->dbfiles_path);
638
+ }
639
+ rrd_stat_atomic_add(&ctx->stats.flushing_errors, 1);
640
+ rrd_stat_atomic_add(&global_flushing_errors, 1);
641
+ }
642
643
pg_cache_put(ctx, descr);
644
}
@@ -674,7 +687,7 @@ void *rrdeng_get_page(struct rrdengine_instance *ctx, uuid_t *id, usec_t point_i
687
* You must not change the indices of the statistics or user code will break.
688
* You must not exceed RRDENG_NR_STATS or it will crash.
689
*/
677
-void rrdeng_get_33_statistics(struct rrdengine_instance *ctx, unsigned long long *array)
690
+void rrdeng_get_35_statistics(struct rrdengine_instance *ctx, unsigned long long *array)
691
{
692
struct page_cache *pg_cache = &ctx->pg_cache;
693
@@ -682,7 +695,7 @@ void rrdeng_get_33_statistics(struct rrdengine_instance *ctx, unsigned long long
695
array[1] = (uint64_t)ctx->stats.metric_API_consumers;
696
array[2] = (uint64_t)pg_cache->page_descriptors;
697
array[3] = (uint64_t)pg_cache->populated_pages;
685
- array[4] = (uint64_t)pg_cache->commited_page_index.nr_commited_pages;
698
+ array[4] = (uint64_t)pg_cache->committed_page_index.nr_committed_pages;
699
array[5] = (uint64_t)ctx->stats.pg_cache_insertions;
700
array[6] = (uint64_t)ctx->stats.pg_cache_deletions;
701
array[7] = (uint64_t)ctx->stats.pg_cache_hits;
@@ -711,7 +724,9 @@ void rrdeng_get_33_statistics(struct rrdengine_instance *ctx, unsigned long long
724
array[30] = (uint64_t)global_io_errors;
725
array[31] = (uint64_t)global_fs_errors;
726
array[32] = (uint64_t)rrdeng_reserved_file_descriptors;
714
- assert(RRDENG_NR_STATS == 33);
727
+ array[33] = (uint64_t)ctx->stats.flushing_errors;
728
+ array[34] = (uint64_t)global_flushing_errors;
729
+ assert(RRDENG_NR_STATS == 35);
730
}
731
732
/* Releases reference to page */
database/engine/rrdengineapi.h
+2
-2
@@ -8,7 +8,7 @@
8
#define RRDENG_MIN_PAGE_CACHE_SIZE_MB (8)
9
#define RRDENG_MIN_DISK_SPACE_MB (256)
10
11
-#define RRDENG_NR_STATS (33)
11
+#define RRDENG_NR_STATS (35)
12
13
#define RRDENG_FD_BUDGET_PER_INSTANCE (50)
14
@@ -41,7 +41,7 @@ extern int rrdeng_load_metric_is_finished(struct rrddim_query_handle *rrdimm_han
41
extern void rrdeng_load_metric_finalize(struct rrddim_query_handle *rrdimm_handle);
42
extern time_t rrdeng_metric_latest_time(RRDDIM *rd);
43
extern time_t rrdeng_metric_oldest_time(RRDDIM *rd);
44
-extern void rrdeng_get_33_statistics(struct rrdengine_instance *ctx, unsigned long long *array);
44
+extern void rrdeng_get_35_statistics(struct rrdengine_instance *ctx, unsigned long long *array);
45
46
/* must call once before using anything */
47
extern int rrdeng_init(struct rrdengine_instance **ctxp, char *dbfiles_path, unsigned page_cache_mb,
database/engine/rrdenginelib.c
+18
-4
@@ -131,7 +131,7 @@ char *get_rrdeng_statistics(struct rrdengine_instance *ctx, char *str, size_t si
131
"page_cache_total_pages: %ld\n"
132
"page_cache_descriptors: %ld\n"
133
"page_cache_populated_pages: %ld\n"
134
- "page_cache_commited_pages: %ld\n"
134
+ "page_cache_committed_pages: %ld\n"
135
"page_cache_insertions: %ld\n"
136
"page_cache_deletions: %ld\n"
137
"page_cache_hits: %ld\n"
@@ -153,13 +153,20 @@ char *get_rrdeng_statistics(struct rrdengine_instance *ctx, char *str, size_t si
153
"datafile_creations: %ld\n"
154
"datafile_deletions: %ld\n"
155
"journalfile_creations: %ld\n"
156
- "journalfile_deletions: %ld\n",
156
+ "journalfile_deletions: %ld\n"
157
+ "io_errors: %ld\n"
158
+ "fs_errors: %ld\n"
159
+ "global_io_errors: %ld\n"
160
+ "global_fs_errors: %ld\n"
161
+ "rrdeng_reserved_file_descriptors: %ld\n"
162
+ "flushing_errors: %ld\n"
163
+ "global_flushing_errors: %ld\n",
164
(long)ctx->stats.metric_API_producers,
165
(long)ctx->stats.metric_API_consumers,
166
(long)pg_cache->page_descriptors,
167
(long)ctx->stats.page_cache_descriptors,
168
(long)pg_cache->populated_pages,
162
- (long)pg_cache->commited_page_index.nr_commited_pages,
169
+ (long)pg_cache->committed_page_index.nr_committed_pages,
170
(long)ctx->stats.pg_cache_insertions,
171
(long)ctx->stats.pg_cache_deletions,
172
(long)ctx->stats.pg_cache_hits,
@@ -181,7 +188,14 @@ char *get_rrdeng_statistics(struct rrdengine_instance *ctx, char *str, size_t si
188
(long)ctx->stats.datafile_creations,
189
(long)ctx->stats.datafile_deletions,
190
(long)ctx->stats.journalfile_creations,
184
- (long)ctx->stats.journalfile_deletions
191
+ (long)ctx->stats.journalfile_deletions,
192
+ (long)ctx->stats.io_errors,
193
+ (long)ctx->stats.fs_errors,
194
+ (long)global_io_errors,
195
+ (long)global_fs_errors,
196
+ (long)rrdeng_reserved_file_descriptors,
197
+ (long)ctx->stats.flushing_errors,
198
+ (long)global_flushing_errors
199
);
200
return str;
201
}
health/health.d/dbengine.conf
+13
-1
@@ -23,4 +23,16 @@ lookup: sum -10m unaligned of I/O errors
23
crit: $this > 0
24
delay: down 1h multiplier 1.5 max 3h
25
info: number of IO errors dbengine came across the last 10 minutes (CRC errors, out of space, bad disk etc)
26
- to: sysadmin
\ No newline at end of file
26
+ to: sysadmin
27
+
28
+ alarm: 10min_dbengine_global_flushing_errors
29
+ on: netdata.dbengine_global_errors
30
+ os: linux freebsd macos
31
+ hosts: *
32
+lookup: sum -10m unaligned of flushing errors
33
+ units: errors
34
+ every: 3s
35
+ crit: $this > 0
36
+ delay: down 1h multiplier 1.5 max 3h
37
+ info: number of times in the last 10 minutes that the dbengine failed to completely flush data to disk, metric data will not be stored in the database, please reduce disk load or use a faster disk
38
+ to: sysadmin