Fix CRC and I/O error handling in dbengine so that netdata is not halted and relevant error messages are printed and alarms are raised (#6452)
Markos Fountoulakis committed
Jul 15, 2019 at 10:07 UTC
192a868b03289a583ac1cd3f648ef185c6ac9209
4 files changed
+39
-21
configs.signatures
+1
-1
@@ -381,7 +381,7 @@ declare -A configs_signatures=(
381
['7deb236ec68a512b9bdd18e6a51d76f7']='python.d/mysql.conf'
382
['7e5fc1644aa7a54f9dbb1bd102521b09']='health.d/memcached.conf'
383
['7f13631183fbdf79c21c8e5a171e9b34']='health.d/zfs.conf'
384
- ['ce285c90747428ee5da4efb547418dda']='health.d/dbengine.conf'
384
+ ['93674f3206872ae9c43ecbc54988413b']='health.d/dbengine.conf'
385
['7fb8184d56a27040e73261ed9c6fc76f']='health_alarm_notify.conf'
386
['80266bddd3df374923c750a6de91d120']='health.d/apache.conf'
387
['803a7f9dcb942eeac0fd764b9e3e38ca']='fping.conf'
database/engine/journalfile.c
+9
-3
@@ -3,13 +3,18 @@
3
4
static void flush_transaction_buffer_cb(uv_fs_t* req)
5
{
6
- struct generic_io_descriptor *io_descr;
6
+ struct generic_io_descriptor *io_descr = req->data;
7
+ struct rrdengine_worker_config* wc = req->loop->data;
8
+ struct rrdengine_instance *ctx = wc->ctx;
9
10
debug(D_RRDENGINE, "%s: Journal block was written to disk.", __func__);
11
if (req->result < 0) {
10
- fatal("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
12
+ ++ctx->stats.io_errors;
13
+ rrd_stat_atomic_add(&global_io_errors, 1);
14
+ error("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
15
+ } else {
16
+ debug(D_RRDENGINE, "%s: Journal block was written to disk.", __func__);
17
}
12
- io_descr = req->data;
18
19
uv_fs_req_cleanup(req);
20
free(io_descr->buf);
@@ -348,6 +353,7 @@ static unsigned replay_transaction(struct rrdengine_instance *ctx, struct rrdeng
353
ret = crc32cmp(jf_trailer->checksum, crc);
354
debug(D_RRDENGINE, "Transaction %"PRIu64" was read from disk. CRC32 check: %s", *id, ret ? "FAILED" : "SUCCEEDED");
355
if (unlikely(ret)) {
356
+ error("Transaction %"PRIu64" was read from disk. CRC32 check: FAILED", *id);
357
return size_bytes;
358
}
359
switch (jf_header->type) {
database/engine/rrdengine.c
+28
-16
@@ -37,24 +37,29 @@ void read_extent_cb(uv_fs_t* req)
37
unsigned i, j, count;
38
void *page, *uncompressed_buf = NULL;
39
uint32_t payload_length, payload_offset, page_offset, uncompressed_payload_length;
40
+ uint8_t have_read_error = 0;
41
/* persistent structures */
42
struct rrdeng_df_extent_header *header;
43
struct rrdeng_df_extent_trailer *trailer;
44
uLong crc;
45
46
xt_io_descr = req->data;
46
- if (req->result < 0) {
47
- error("%s: uv_fs_read: %s", __func__, uv_strerror((int)req->result));
48
- goto cleanup;
49
- }
50
-
47
header = xt_io_descr->buf;
48
payload_length = header->payload_length;
49
count = header->number_of_pages;
54
-
50
payload_offset = sizeof(*header) + sizeof(header->descr[0]) * count;
56
-
51
trailer = xt_io_descr->buf + xt_io_descr->bytes - sizeof(*trailer);
52
+
53
+ if (req->result < 0) {
54
+ struct rrdengine_datafile *datafile = xt_io_descr->descr_array[0]->extent->datafile;
55
+
56
+ ++ctx->stats.io_errors;
57
+ rrd_stat_atomic_add(&global_io_errors, 1);
58
+ have_read_error = 1;
59
+ error("%s: uv_fs_read - %s - extent at offset %"PRIu64"(%u) in datafile %u-%u.", __func__,
60
+ uv_strerror((int)req->result), xt_io_descr->pos, xt_io_descr->bytes, datafile->tier, datafile->fileno);
61
+ goto after_crc_check;
62
+ }
63
crc = crc32(0L, Z_NULL, 0);
64
crc = crc32(crc, xt_io_descr->buf, xt_io_descr->bytes - sizeof(*trailer));
65
ret = crc32cmp(trailer->checksum, crc);
@@ -66,12 +71,17 @@ void read_extent_cb(uv_fs_t* req)
71
}
72
#endif
73
if (unlikely(ret)) {
69
- /* TODO: handle errors */
70
- exit(UV_EIO);
71
- goto cleanup;
74
+ struct rrdengine_datafile *datafile = xt_io_descr->descr_array[0]->extent->datafile;
75
+
76
+ ++ctx->stats.io_errors;
77
+ rrd_stat_atomic_add(&global_io_errors, 1);
78
+ have_read_error = 1;
79
+ error("%s: Extent at offset %"PRIu64"(%u) was read from datafile %u-%u. CRC32 check: FAILED", __func__,
80
+ xt_io_descr->pos, xt_io_descr->bytes, datafile->tier, datafile->fileno);
81
}
82
74
- if (RRD_NO_COMPRESSION != header->compression_algorithm) {
83
+after_crc_check:
84
+ if (!have_read_error && RRD_NO_COMPRESSION != header->compression_algorithm) {
85
uncompressed_payload_length = 0;
86
for (i = 0 ; i < count ; ++i) {
87
uncompressed_payload_length += header->descr[i].page_length;
@@ -99,7 +109,10 @@ void read_extent_cb(uv_fs_t* req)
109
page_offset += header->descr[j].page_length;
110
}
111
/* care, we don't hold the descriptor mutex */
102
- if (RRD_NO_COMPRESSION == header->compression_algorithm) {
112
+ if (have_read_error) {
113
+ /* Applications should make sure NULL values match 0 as does SN_EMPTY_SLOT */
114
+ memset(page, 0, descr->page_length);
115
+ } else if (RRD_NO_COMPRESSION == header->compression_algorithm) {
116
(void) memcpy(page, xt_io_descr->buf + payload_offset + page_offset, descr->page_length);
117
} else {
118
(void) memcpy(page, uncompressed_buf + page_offset, descr->page_length);
@@ -118,12 +131,11 @@ void read_extent_cb(uv_fs_t* req)
131
}
132
rrdeng_page_descr_mutex_unlock(ctx, descr);
133
}
121
- if (RRD_NO_COMPRESSION != header->compression_algorithm) {
134
+ if (!have_read_error && RRD_NO_COMPRESSION != header->compression_algorithm) {
135
freez(uncompressed_buf);
136
}
137
if (xt_io_descr->completion)
138
complete(xt_io_descr->completion);
126
-cleanup:
139
uv_fs_req_cleanup(req);
140
free(xt_io_descr->buf);
141
freez(xt_io_descr);
@@ -246,8 +258,9 @@ void flush_pages_cb(uv_fs_t* req)
258
259
xt_io_descr = req->data;
260
if (req->result < 0) {
261
+ ++ctx->stats.io_errors;
262
+ rrd_stat_atomic_add(&global_io_errors, 1);
263
error("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
250
- goto cleanup;
264
}
265
#ifdef NETDATA_INTERNAL_CHECKS
266
{
@@ -279,7 +292,6 @@ void flush_pages_cb(uv_fs_t* req)
292
}
293
if (xt_io_descr->completion)
294
complete(xt_io_descr->completion);
282
-cleanup:
295
uv_fs_req_cleanup(req);
296
free(xt_io_descr->buf);
297
freez(xt_io_descr);
health/health.d/dbengine.conf
+1
-1
@@ -22,5 +22,5 @@
22
every: 10s
23
crit: $this > 0
24
delay: down 1h multiplier 1.5 max 3h
25
- info: number of IO errors dbengine came across the last 10 minutes (out of space, bad disk etc)
25
+ info: number of IO errors dbengine came across the last 10 minutes (CRC errors, out of space, bad disk etc)
26
to: sysadmin