@cryptotaxi247 / netdata-1 / commits / 192a868b0

Fix CRC and I/O error handling in dbengine so that netdata is not halted and relevant error messages are printed and alarms are raised (#6452)

Markos Fountoulakis committed Jul 15, 2019 at 10:07 UTC 192a868b03289a583ac1cd3f648ef185c6ac9209
4 files changed +39 -21
configs.signatures
+1 -1
@@ -381,7 +381,7 @@ declare -A configs_signatures=(
381 ['7deb236ec68a512b9bdd18e6a51d76f7']='python.d/mysql.conf'
382 ['7e5fc1644aa7a54f9dbb1bd102521b09']='health.d/memcached.conf'
383 ['7f13631183fbdf79c21c8e5a171e9b34']='health.d/zfs.conf'
384 - ['ce285c90747428ee5da4efb547418dda']='health.d/dbengine.conf'
384 + ['93674f3206872ae9c43ecbc54988413b']='health.d/dbengine.conf'
385 ['7fb8184d56a27040e73261ed9c6fc76f']='health_alarm_notify.conf'
386 ['80266bddd3df374923c750a6de91d120']='health.d/apache.conf'
387 ['803a7f9dcb942eeac0fd764b9e3e38ca']='fping.conf'
database/engine/journalfile.c
+9 -3
@@ -3,13 +3,18 @@
3
4 static void flush_transaction_buffer_cb(uv_fs_t* req)
5 {
6 - struct generic_io_descriptor *io_descr;
6 + struct generic_io_descriptor *io_descr = req->data;
7 + struct rrdengine_worker_config* wc = req->loop->data;
8 + struct rrdengine_instance *ctx = wc->ctx;
9
10 debug(D_RRDENGINE, "%s: Journal block was written to disk.", __func__);
11 if (req->result < 0) {
10 - fatal("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
12 + ++ctx->stats.io_errors;
13 + rrd_stat_atomic_add(&global_io_errors, 1);
14 + error("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
15 + } else {
16 + debug(D_RRDENGINE, "%s: Journal block was written to disk.", __func__);
17 }
12 - io_descr = req->data;
18
19 uv_fs_req_cleanup(req);
20 free(io_descr->buf);
@@ -348,6 +353,7 @@ static unsigned replay_transaction(struct rrdengine_instance *ctx, struct rrdeng
353 ret = crc32cmp(jf_trailer->checksum, crc);
354 debug(D_RRDENGINE, "Transaction %"PRIu64" was read from disk. CRC32 check: %s", *id, ret ? "FAILED" : "SUCCEEDED");
355 if (unlikely(ret)) {
356 + error("Transaction %"PRIu64" was read from disk. CRC32 check: FAILED", *id);
357 return size_bytes;
358 }
359 switch (jf_header->type) {
database/engine/rrdengine.c
+28 -16
@@ -37,24 +37,29 @@ void read_extent_cb(uv_fs_t* req)
37 unsigned i, j, count;
38 void *page, *uncompressed_buf = NULL;
39 uint32_t payload_length, payload_offset, page_offset, uncompressed_payload_length;
40 + uint8_t have_read_error = 0;
41 /* persistent structures */
42 struct rrdeng_df_extent_header *header;
43 struct rrdeng_df_extent_trailer *trailer;
44 uLong crc;
45
46 xt_io_descr = req->data;
46 - if (req->result < 0) {
47 - error("%s: uv_fs_read: %s", __func__, uv_strerror((int)req->result));
48 - goto cleanup;
49 - }
50 -
47 header = xt_io_descr->buf;
48 payload_length = header->payload_length;
49 count = header->number_of_pages;
54 -
50 payload_offset = sizeof(*header) + sizeof(header->descr[0]) * count;
56 -
51 trailer = xt_io_descr->buf + xt_io_descr->bytes - sizeof(*trailer);
52 +
53 + if (req->result < 0) {
54 + struct rrdengine_datafile *datafile = xt_io_descr->descr_array[0]->extent->datafile;
55 +
56 + ++ctx->stats.io_errors;
57 + rrd_stat_atomic_add(&global_io_errors, 1);
58 + have_read_error = 1;
59 + error("%s: uv_fs_read - %s - extent at offset %"PRIu64"(%u) in datafile %u-%u.", __func__,
60 + uv_strerror((int)req->result), xt_io_descr->pos, xt_io_descr->bytes, datafile->tier, datafile->fileno);
61 + goto after_crc_check;
62 + }
63 crc = crc32(0L, Z_NULL, 0);
64 crc = crc32(crc, xt_io_descr->buf, xt_io_descr->bytes - sizeof(*trailer));
65 ret = crc32cmp(trailer->checksum, crc);
@@ -66,12 +71,17 @@ void read_extent_cb(uv_fs_t* req)
71 }
72 #endif
73 if (unlikely(ret)) {
69 - /* TODO: handle errors */
70 - exit(UV_EIO);
71 - goto cleanup;
74 + struct rrdengine_datafile *datafile = xt_io_descr->descr_array[0]->extent->datafile;
75 +
76 + ++ctx->stats.io_errors;
77 + rrd_stat_atomic_add(&global_io_errors, 1);
78 + have_read_error = 1;
79 + error("%s: Extent at offset %"PRIu64"(%u) was read from datafile %u-%u. CRC32 check: FAILED", __func__,
80 + xt_io_descr->pos, xt_io_descr->bytes, datafile->tier, datafile->fileno);
81 }
82
74 - if (RRD_NO_COMPRESSION != header->compression_algorithm) {
83 +after_crc_check:
84 + if (!have_read_error && RRD_NO_COMPRESSION != header->compression_algorithm) {
85 uncompressed_payload_length = 0;
86 for (i = 0 ; i < count ; ++i) {
87 uncompressed_payload_length += header->descr[i].page_length;
@@ -99,7 +109,10 @@ void read_extent_cb(uv_fs_t* req)
109 page_offset += header->descr[j].page_length;
110 }
111 /* care, we don't hold the descriptor mutex */
102 - if (RRD_NO_COMPRESSION == header->compression_algorithm) {
112 + if (have_read_error) {
113 + /* Applications should make sure NULL values match 0 as does SN_EMPTY_SLOT */
114 + memset(page, 0, descr->page_length);
115 + } else if (RRD_NO_COMPRESSION == header->compression_algorithm) {
116 (void) memcpy(page, xt_io_descr->buf + payload_offset + page_offset, descr->page_length);
117 } else {
118 (void) memcpy(page, uncompressed_buf + page_offset, descr->page_length);
@@ -118,12 +131,11 @@ void read_extent_cb(uv_fs_t* req)
131 }
132 rrdeng_page_descr_mutex_unlock(ctx, descr);
133 }
121 - if (RRD_NO_COMPRESSION != header->compression_algorithm) {
134 + if (!have_read_error && RRD_NO_COMPRESSION != header->compression_algorithm) {
135 freez(uncompressed_buf);
136 }
137 if (xt_io_descr->completion)
138 complete(xt_io_descr->completion);
126 -cleanup:
139 uv_fs_req_cleanup(req);
140 free(xt_io_descr->buf);
141 freez(xt_io_descr);
@@ -246,8 +258,9 @@ void flush_pages_cb(uv_fs_t* req)
258
259 xt_io_descr = req->data;
260 if (req->result < 0) {
261 + ++ctx->stats.io_errors;
262 + rrd_stat_atomic_add(&global_io_errors, 1);
263 error("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
250 - goto cleanup;
264 }
265 #ifdef NETDATA_INTERNAL_CHECKS
266 {
@@ -279,7 +292,6 @@ void flush_pages_cb(uv_fs_t* req)
292 }
293 if (xt_io_descr->completion)
294 complete(xt_io_descr->completion);
282 -cleanup:
295 uv_fs_req_cleanup(req);
296 free(xt_io_descr->buf);
297 freez(xt_io_descr);
health/health.d/dbengine.conf
+1 -1
@@ -22,5 +22,5 @@
22 every: 10s
23 crit: $this > 0
24 delay: down 1h multiplier 1.5 max 3h
25 - info: number of IO errors dbengine came across the last 10 minutes (out of space, bad disk etc)
25 + info: number of IO errors dbengine came across the last 10 minutes (CRC errors, out of space, bad disk etc)
26 to: sysadmin