@cryptotaxi247 / netdata-1 / commits / 6ca6d840d

Database engine (#5282)

* Database engine prototype version 0 * Database engine initial integration with netdata POC * Scalable database engine with file and memory management. * Database engine integration with netdata * Added MIN MAX definitions to fix alpine build of travis CI * Bugfix for backends and new DB engine, remove useless rrdset_time2slot() calls and erroneous checks * DB engine disk protocol correction * Moved DB engine storage file location to /var/cache/netdata/{host}/dbengine * Fix configure to require openSSL for DB engine * Fix netdata daemon health not holding read lock when iterating chart dimensions * Optimized query API for new DB engine and old netdata DB fallback code-path * netdata database internal query API improvements and cleanup * Bugfix for DB engine queries returning empty values * Added netdata internal check for data queries for old and new DB * Added statistics to DB engine and fixed memory corruption bug * Added preliminary charts for DB engine statistics * Changed DB engine ratio statistics to incremental * Added netdata statistics charts for DB engine internal statistics * Fix for netdata not compiling successfully when missing dbengine dependencies * Added DB engine functional test to netdata unittest command parameter * Implemented DB engine dataset generator based on example.random chart * Fix build error in CI * Support older versions of libuv1 * Fixes segmentation fault when using multiple DB engine instances concurrently * Fix memory corruption bug * Fixed createdataset advanced option not exiting * Fix for DB engine not working on FreeBSD * Support FreeBSD library paths of new dependencies * Workaround for unsupported O_DIRECT in OS X * Fix unittest crashing during cleanup * Disable DB engine FS caching in Apple OS X since O_DIRECT is not available * Fix segfault when unittest and DB engine dataset generator don't have permissions to create temporary host * Modified DB engine dataset generator to create multiple files * Toned down overzealous page cache prefetcher * Reduce internal memory fragmentation for page-cache data pages * Added documentation describing the DB engine * Documentation bugfixes * Fixed unit tests compilation errors since last rebase * Added note to back-up the DB engine files in documentation * Added codacy fix. * Support old gcc versions for atomic counters in DB engine

Markos Fountoulakis committed May 15, 2019 at 08:28 UTC 6ca6d840dd19d5d7e9bacf93e011803ea5861496
40 files changed +4823 -70
CMakeLists.txt
+53
@@ -89,6 +89,46 @@ set(NETDATA_COMMON_CFLAGS ${NETDATA_COMMON_CFLAGS} ${ZLIB_CFLAGS_OTHER})
89 set(NETDATA_COMMON_LIBRARIES ${NETDATA_COMMON_LIBRARIES} ${ZLIB_LIBRARIES})
90 set(NETDATA_COMMON_INCLUDE_DIRS ${NETDATA_COMMON_INCLUDE_DIRS} ${ZLIB_INCLUDE_DIRS})
91
92 +# -----------------------------------------------------------------------------
93 +# libuv multi-platform support library with a focus on asynchronous I/O
94 +
95 +pkg_check_modules(LIBUV REQUIRED libuv)
96 +set(NETDATA_COMMON_CFLAGS ${NETDATA_COMMON_CFLAGS} ${LIBUV_CFLAGS_OTHER})
97 +set(NETDATA_COMMON_LIBRARIES ${NETDATA_COMMON_LIBRARIES} ${LIBUV_LIBRARIES})
98 +set(NETDATA_COMMON_INCLUDE_DIRS ${NETDATA_COMMON_INCLUDE_DIRS} ${LIBUV_INCLUDE_DIRS})
99 +
100 +# -----------------------------------------------------------------------------
101 +# lz4 Extremely Fast Compression algorithm
102 +
103 +pkg_check_modules(LIBLZ4 REQUIRED liblz4)
104 +set(NETDATA_COMMON_CFLAGS ${NETDATA_COMMON_CFLAGS} ${LIBLZ4_CFLAGS_OTHER})
105 +set(NETDATA_COMMON_LIBRARIES ${NETDATA_COMMON_LIBRARIES} ${LIBLZ4_LIBRARIES})
106 +set(NETDATA_COMMON_INCLUDE_DIRS ${NETDATA_COMMON_INCLUDE_DIRS} ${LIBLZ4_INCLUDE_DIRS})
107 +
108 +# -----------------------------------------------------------------------------
109 +# Judy General purpose dynamic array
110 +
111 +# pkgconfig not working in Ubuntu, why? upstream package broken?
112 +#pkg_check_modules(JUDY REQUIRED Judy)
113 +#set(NETDATA_COMMON_CFLAGS ${NETDATA_COMMON_CFLAGS} ${JUDY_CFLAGS_OTHER})
114 +#set(NETDATA_COMMON_LIBRARIES ${NETDATA_COMMON_LIBRARIES} ${JUDY_LIBRARIES})
115 +#set(NETDATA_COMMON_INCLUDE_DIRS ${NETDATA_COMMON_INCLUDE_DIRS} ${JUDY_INCLUDE_DIRS})
116 +set(NETDATA_COMMON_LIBRARIES ${NETDATA_COMMON_LIBRARIES} "-lJudy")
117 +set(CMAKE_REQUIRED_LIBRARIES "Judy")
118 +check_symbol_exists("JudyLLast" "Judy.h" HAVE_JUDY)
119 +IF(HAVE_JUDY)
120 + message(STATUS "Judy library found")
121 +ELSE()
122 + message( FATAL_ERROR "libJudy required but not found. Try installing 'libjudy-dev' or 'Judy-devel'." )
123 +ENDIF()
124 +
125 +# -----------------------------------------------------------------------------
126 +# OpenSSL Cryptography and SSL/TLS Toolkit
127 +
128 +pkg_check_modules(OPENSSL REQUIRED openssl)
129 +set(NETDATA_COMMON_CFLAGS ${NETDATA_COMMON_CFLAGS} ${OPENSSL_CFLAGS_OTHER})
130 +set(NETDATA_COMMON_LIBRARIES ${NETDATA_COMMON_LIBRARIES} ${OPENSSL_LIBRARIES})
131 +set(NETDATA_COMMON_INCLUDE_DIRS ${NETDATA_COMMON_INCLUDE_DIRS} ${OPENSSL_INCLUDE_DIRS})
132
133 # -----------------------------------------------------------------------------
134 # Detect libcap
@@ -403,6 +443,19 @@ set(RRD_PLUGIN_FILES
443 database/rrdsetvar.h
444 database/rrdvar.c
445 database/rrdvar.h
446 + database/engine/rrdengine.c
447 + database/engine/rrdengine.h
448 + database/engine/rrddiskprotocol.h
449 + database/engine/datafile.c
450 + database/engine/datafile.h
451 + database/engine/journalfile.c
452 + database/engine/journalfile.h
453 + database/engine/rrdenginelib.c
454 + database/engine/rrdenginelib.h
455 + database/engine/rrdengineapi.c
456 + database/engine/rrdengineapi.h
457 + database/engine/pagecache.c
458 + database/engine/pagecache.h
459 )
460
461 set(WEB_PLUGIN_FILES
Makefile.am
+22
@@ -311,6 +311,24 @@ RRD_PLUGIN_FILES = \
311 database/rrdvar.h \
312 $(NULL)
313
314 +if ENABLE_DBENGINE
315 + RRD_PLUGIN_FILES += \
316 + database/engine/rrdengine.c \
317 + database/engine/rrdengine.h \
318 + database/engine/rrddiskprotocol.h \
319 + database/engine/datafile.c \
320 + database/engine/datafile.h \
321 + database/engine/journalfile.c \
322 + database/engine/journalfile.h \
323 + database/engine/rrdenginelib.c \
324 + database/engine/rrdenginelib.h \
325 + database/engine/rrdengineapi.c \
326 + database/engine/rrdengineapi.h \
327 + database/engine/pagecache.c \
328 + database/engine/pagecache.h \
329 + $(NULL)
330 +endif
331 +
332 API_PLUGIN_FILES = \
333 web/api/badges/web_buffer_svg.c \
334 web/api/badges/web_buffer_svg.h \
@@ -477,6 +495,10 @@ NETDATA_COMMON_LIBS = \
495 $(OPTIONAL_MATH_LIBS) \
496 $(OPTIONAL_ZLIB_LIBS) \
497 $(OPTIONAL_UUID_LIBS) \
498 + $(OPTIONAL_UV_LIBS) \
499 + $(OPTIONAL_LZ4_LIBS) \
500 + $(OPTIONAL_JUDY_LIBS) \
501 + $(OPTIONAL_SSL_LIBS) \
502 $(NULL)
503 # TODO: Find more graceful way to add libs for AWS Kinesis
504
backends/backends.c
+19 -2
@@ -62,9 +62,11 @@ calculated_number backend_calculate_value_from_stored_data(
62 (void)host;
63
64 // find the edges of the rrd database for this chart
65 - time_t first_t = rrdset_first_entry_t(st);
66 - time_t last_t = rrdset_last_entry_t(st);
65 + time_t first_t = rd->state->query_ops.oldest_time(rd);
66 + time_t last_t = rd->state->query_ops.latest_time(rd);
67 time_t update_every = st->update_every;
68 + struct rrddim_query_handle handle;
69 + storage_number n;
70
71 // step back a little, to make sure we have complete data collection
72 // for all metrics
@@ -105,6 +107,7 @@ calculated_number backend_calculate_value_from_stored_data(
107 size_t counter = 0;
108 calculated_number sum = 0;
109
110 +/*
111 long start_at_slot = rrdset_time2slot(st, before),
112 stop_at_slot = rrdset_time2slot(st, after),
113 slot, stop_now = 0;
@@ -126,7 +129,21 @@ calculated_number backend_calculate_value_from_stored_data(
129
130 counter++;
131 }
132 +*/
133 + for(rd->state->query_ops.init(rd, &handle, before, after) ; !rd->state->query_ops.is_finished(&handle) ; ) {
134 + n = rd->state->query_ops.next_metric(&handle);
135
136 + if(unlikely(!does_storage_number_exist(n))) {
137 + // not collected
138 + continue;
139 + }
140 +
141 + calculated_number value = unpack_storage_number(n);
142 + sum += value;
143 +
144 + counter++;
145 + }
146 + rd->state->query_ops.finalize(&handle);
147 if(unlikely(!counter)) {
148 debug(D_BACKEND, "BACKEND: %s.%s.%s: no values stored in database for range %lu to %lu",
149 host->hostname, st->id, rd->id,
configure.ac
+91 -2
@@ -131,6 +131,12 @@ AC_ARG_ENABLE(
131 ,
132 [enable_lto="detect"]
133 )
134 +AC_ARG_ENABLE(
135 + [dbengine],
136 + [AS_HELP_STRING([--disable-dbengine], [disable netdata dbengine @<:@default autodetect@:>@])],
137 + ,
138 + [enable_dbengine="detect"]
139 +)
140
141
142 # -----------------------------------------------------------------------------
@@ -188,7 +194,7 @@ case "$host_os" in
194 freebsd*)
195 build_target=freebsd
196 build_target_id=2
191 - CFLAGS="${CFLAGS} -I/usr/local/include"
197 + CFLAGS="${CFLAGS} -I/usr/local/include -L/usr/local/lib"
198 ;;
199 darwin*)
200 build_target=macos
@@ -242,6 +248,46 @@ fi
248 AC_MSG_RESULT([${with_math}])
249
250
251 +# -----------------------------------------------------------------------------
252 +# libuv multi-platform support library with a focus on asynchronous I/O
253 +# TODO: check version, uv_fs_scandir_next only available in version >= 1.0
254 +
255 +AC_CHECK_LIB(
256 + [uv],
257 + [uv_fs_scandir_next],
258 + [UV_LIBS="-luv"]
259 +)
260 +
261 +OPTIONAL_UV_CLFAGS="${UV_CFLAGS}"
262 +OPTIONAL_UV_LIBS="${UV_LIBS}"
263 +
264 +
265 +# -----------------------------------------------------------------------------
266 +# lz4 Extremely Fast Compression algorithm
267 +
268 +AC_CHECK_LIB(
269 + [lz4],
270 + [LZ4_decompress_safe],
271 + [LZ4_LIBS="-llz4"]
272 +)
273 +
274 +OPTIONAL_LZ4_CLFAGS="${LZ4_CFLAGS}"
275 +OPTIONAL_LZ4_LIBS="${LZ4_LIBS}"
276 +
277 +
278 +# -----------------------------------------------------------------------------
279 +# Judy General purpose dynamic array
280 +
281 +AC_CHECK_LIB(
282 + [Judy],
283 + [JudyLIns],
284 + [JUDY_LIBS="-lJudy"]
285 +)
286 +
287 +OPTIONAL_JUDY_CLFAGS="${JUDY_CFLAGS}"
288 +OPTIONAL_JUDY_LIBS="${JUDY_LIBS}"
289 +
290 +
291 # -----------------------------------------------------------------------------
292 # zlib
293
@@ -279,6 +325,43 @@ OPTIONAL_UUID_CFLAGS="${UUID_CFLAGS}"
325 OPTIONAL_UUID_LIBS="${UUID_LIBS}"
326
327
328 +# -----------------------------------------------------------------------------
329 +# OpenSSL Cryptography and SSL/TLS Toolkit
330 +
331 +AC_CHECK_LIB(
332 + [crypto],
333 + [SHA256_Init],
334 + [SSL_LIBS="-lcrypto -lssl"]
335 +)
336 +
337 +OPTIONAL_SSL_CLFAGS="${SSL_CFLAGS}"
338 +OPTIONAL_SSL_LIBS="${SSL_LIBS}"
339 +
340 +# -----------------------------------------------------------------------------
341 +# DB engine
342 +test "${enable_dbengine}" = "yes" -a -z "${UV_LIBS}" && \
343 + AC_MSG_ERROR([libuv required but not found. Try installing 'libuv1-dev' or 'libuv-devel'.])
344 +
345 +test "${enable_dbengine}" = "yes" -a -z "${LZ4_LIBS}" && \
346 + AC_MSG_ERROR([liblz4 required but not found. Try installing 'liblz4-dev' or 'lz4-devel'.])
347 +
348 +test "${enable_dbengine}" = "yes" -a -z "${JUDY_LIBS}" && \
349 + AC_MSG_ERROR([libJudy required but not found. Try installing 'libjudy-dev' or 'Judy-devel'.])
350 +
351 +test "${enable_dbengine}" = "yes" -a -z "${SSL_LIBS}" && \
352 + AC_MSG_ERROR([OpenSSL required but not found. Try installing 'libssl-dev' or 'openssl-devel'.])
353 +
354 +AC_MSG_CHECKING([if netdata dbengine should be used])
355 +if test "${enable_dbengine}" != "no" -a "${UV_LIBS}" -a "${LZ4_LIBS}" -a "${JUDY_LIBS}" -a "${SSL_LIBS}"; then
356 + enable_dbengine="yes"
357 + AC_DEFINE([ENABLE_DBENGINE], [1], [netdata dbengine usability])
358 +else
359 + enable_dbengine="no"
360 +fi
361 +AC_MSG_RESULT([${enable_dbengine}])
362 +AM_CONDITIONAL([ENABLE_DBENGINE], [test "${enable_dbengine}" = "yes"])
363 +
364 +
365 # -----------------------------------------------------------------------------
366 # compiler options
367
@@ -781,7 +864,12 @@ CPPFLAGS="\
864
865 AC_SUBST([OPTIONAL_MATH_CFLAGS])
866 AC_SUBST([OPTIONAL_MATH_LIBS])
784 -AC_SUBST([OPTIONAL_NFACCT_CFLAGS])
867 +AC_SUBST([OPTIONAL_RT_CLFAGS])
868 +AC_SUBST([OPTIONAL_UV_LIBS])
869 +AC_SUBST([OPTIONAL_LZ4_LIBS])
870 +AC_SUBST([OPTIONAL_JUDY_LIBS])
871 +AC_SUBST([OPTIONAL_SSL_LIBS])
872 +AC_SUBST([OPTIONAL_NFACCT_CLFAGS])
873 AC_SUBST([OPTIONAL_NFACCT_LIBS])
874 AC_SUBST([OPTIONAL_ZLIB_CFLAGS])
875 AC_SUBST([OPTIONAL_ZLIB_LIBS])
@@ -831,6 +919,7 @@ AC_CONFIG_FILES([
919 collectors/xenstat.plugin/Makefile
920 daemon/Makefile
921 database/Makefile
922 + database/engine/Makefile
923 diagrams/Makefile
924 health/Makefile
925 health/notifications/Makefile
daemon/README.md
+2
@@ -164,6 +164,8 @@ The command line options of the netdata 1.10.0 version are the following:
164
165 -W unittest Run internal unittests and exit.
166
167 + -W createdataset=N Create a DB engine dataset of N seconds and exit.
168 +
169 -W set section option value
170 set netdata.conf option from the command line.
171
daemon/config/README.md
+1 -1
@@ -57,7 +57,7 @@ cache directory | `/var/cache/netdata` | The directory the memory database will
57 lib directory | `/var/lib/netdata` | Contains the alarm log and the netdata instance guid.
58 home directory | `/var/cache/netdata` | Contains the db files for the collected metrics
59 plugins directory | `"/usr/libexec/netdata/plugins.d" "/etc/netdata/custom-plugins.d"` | The directory plugin programs are kept. This setting supports multiple directories, space separated. If any directory path contains spaces, enclose it in single or double quotes.
60 -memory mode | `save` | When set to `save` netdata will save its round robin database on exit and load it on startup. When set to `map` the cache files will be updated in real time (check `man mmap` - do not set this on systems with heavy load or slow disks - the disks will continuously sync the in-memory database of netdata). When set to `ram` the round robin database will be temporary and it will be lost when netdata exits. `none` disables the database at this host. This also disables health monitoring (there cannot be health monitoring without a database). host access prefix | | This is used in docker environments where /proc, /sys, etc have to be accessed via another path. You may also have to set SYS_PTRACE capability on the docker for this work. Check [issue 43](https://github.com/netdata/netdata/issues/43).
60 +memory mode | `save` | When set to `save` netdata will save its round robin database on exit and load it on startup. When set to `map` the cache files will be updated in real time (check `man mmap` - do not set this on systems with heavy load or slow disks - the disks will continuously sync the in-memory database of netdata). When set to `dbengine` it behaves similarly to `map` but with much better disk and memory efficiency, however, with higher overhead. When set to `ram` the round robin database will be temporary and it will be lost when netdata exits. `none` disables the database at this host. This also disables health monitoring (there cannot be health monitoring without a database). host access prefix | | This is used in docker environments where /proc, /sys, etc have to be accessed via another path. You may also have to set SYS_PTRACE capability on the docker for this work. Check [issue 43](https://github.com/netdata/netdata/issues/43).
61 memory deduplication (ksm) | `yes` | When set to `yes`, netdata will offer its in-memory round robin database to kernel same page merging (KSM) for deduplication. For more information check [Memory Deduplication - Kernel Same Page Merging - KSM](../../database/#ksm)
62 TZ environment variable | `:/etc/localtime` | Where to find the timezone
63 timezone | auto-detected | The timezone retrieved from the environment variable
daemon/global_statistics.c
+219
@@ -530,4 +530,223 @@ void global_statistics_charts(void) {
530
531 rrdset_done(st_rrdr_points);
532 }
533 +
534 + // ----------------------------------------------------------------
535 +
536 +#ifdef ENABLE_DBENGINE
537 + if (localhost->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
538 + unsigned long long stats_array[27];
539 +
540 + /* get localhost's DB engine's statistics */
541 + rrdeng_get_27_statistics(localhost->rrdeng_ctx, stats_array);
542 +
543 + // ----------------------------------------------------------------
544 +
545 + {
546 + static RRDSET *st_compression = NULL;
547 + static RRDDIM *rd_savings = NULL;
548 +
549 + if (unlikely(!st_compression)) {
550 + st_compression = rrdset_create_localhost(
551 + "netdata"
552 + , "dbengine_compression_ratio"
553 + , NULL
554 + , "dbengine"
555 + , NULL
556 + , "NetData DB engine data extents' compression savings ratio"
557 + , "percentage"
558 + , "netdata"
559 + , "stats"
560 + , 130502
561 + , localhost->rrd_update_every
562 + , RRDSET_TYPE_LINE
563 + );
564 +
565 + rd_savings = rrddim_add(st_compression, "savings", NULL, 1, 1000, RRD_ALGORITHM_ABSOLUTE);
566 + }
567 + else
568 + rrdset_next(st_compression);
569 +
570 + unsigned long long ratio;
571 + unsigned long long compressed_content_size = stats_array[12];
572 + unsigned long long content_size = stats_array[11];
573 +
574 + if (content_size) {
575 + // allow negative savings
576 + ratio = ((content_size - compressed_content_size) * 100 * 1000) / content_size;
577 + } else {
578 + ratio = 0;
579 + }
580 + rrddim_set_by_pointer(st_compression, rd_savings, ratio);
581 +
582 + rrdset_done(st_compression);
583 + }
584 +
585 + // ----------------------------------------------------------------
586 +
587 + {
588 + static RRDSET *st_pg_cache_hit_ratio = NULL;
589 + static RRDDIM *rd_hit_ratio = NULL;
590 +
591 + if (unlikely(!st_pg_cache_hit_ratio)) {
592 + st_pg_cache_hit_ratio = rrdset_create_localhost(
593 + "netdata"
594 + , "page_cache_hit_ratio"
595 + , NULL
596 + , "dbengine"
597 + , NULL
598 + , "NetData DB engine page cache hit ratio"
599 + , "percentage"
600 + , "netdata"
601 + , "stats"
602 + , 130503
603 + , localhost->rrd_update_every
604 + , RRDSET_TYPE_LINE
605 + );
606 +
607 + rd_hit_ratio = rrddim_add(st_pg_cache_hit_ratio, "ratio", NULL, 1, 1000, RRD_ALGORITHM_ABSOLUTE);
608 + }
609 + else
610 + rrdset_next(st_pg_cache_hit_ratio);
611 +
612 + static unsigned long long old_hits = 0;
613 + static unsigned long long old_misses = 0;
614 + unsigned long long hits = stats_array[7];
615 + unsigned long long misses = stats_array[8];
616 + unsigned long long hits_delta;
617 + unsigned long long misses_delta;
618 + unsigned long long ratio;
619 +
620 + hits_delta = hits - old_hits;
621 + misses_delta = misses - old_misses;
622 + old_hits = hits;
623 + old_misses = misses;
624 +
625 + if (hits_delta + misses_delta) {
626 + // allow negative savings
627 + ratio = (hits_delta * 100 * 1000) / (hits_delta + misses_delta);
628 + } else {
629 + ratio = 0;
630 + }
631 + rrddim_set_by_pointer(st_pg_cache_hit_ratio, rd_hit_ratio, ratio);
632 +
633 + rrdset_done(st_pg_cache_hit_ratio);
634 + }
635 +
636 + // ----------------------------------------------------------------
637 +
638 + {
639 + static RRDSET *st_pg_cache_pages = NULL;
640 + static RRDDIM *rd_populated = NULL;
641 + static RRDDIM *rd_commited = NULL;
642 + static RRDDIM *rd_insertions = NULL;
643 + static RRDDIM *rd_deletions = NULL;
644 + static RRDDIM *rd_backfills = NULL;
645 + static RRDDIM *rd_evictions = NULL;
646 +
647 + if (unlikely(!st_pg_cache_pages)) {
648 + st_pg_cache_pages = rrdset_create_localhost(
649 + "netdata"
650 + , "page_cache_stats"
651 + , NULL
652 + , "dbengine"
653 + , NULL
654 + , "NetData DB engine page statistics"
655 + , "pages"
656 + , "netdata"
657 + , "stats"
658 + , 130504
659 + , localhost->rrd_update_every
660 + , RRDSET_TYPE_LINE
661 + );
662 +
663 + rd_populated = rrddim_add(st_pg_cache_pages, "populated", NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
664 + rd_commited = rrddim_add(st_pg_cache_pages, "commited", NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
665 + rd_insertions = rrddim_add(st_pg_cache_pages, "insertions", NULL, 1, 1, RRD_ALGORITHM_INCREMENTAL);
666 + rd_deletions = rrddim_add(st_pg_cache_pages, "deletions", NULL, -1, 1, RRD_ALGORITHM_INCREMENTAL);
667 + rd_backfills = rrddim_add(st_pg_cache_pages, "backfills", NULL, 1, 1, RRD_ALGORITHM_INCREMENTAL);
668 + rd_evictions = rrddim_add(st_pg_cache_pages, "evictions", NULL, -1, 1, RRD_ALGORITHM_INCREMENTAL);
669 + }
670 + else
671 + rrdset_next(st_pg_cache_pages);
672 +
673 + rrddim_set_by_pointer(st_pg_cache_pages, rd_populated, (collected_number)stats_array[3]);
674 + rrddim_set_by_pointer(st_pg_cache_pages, rd_commited, (collected_number)stats_array[4]);
675 + rrddim_set_by_pointer(st_pg_cache_pages, rd_insertions, (collected_number)stats_array[5]);
676 + rrddim_set_by_pointer(st_pg_cache_pages, rd_deletions, (collected_number)stats_array[6]);
677 + rrddim_set_by_pointer(st_pg_cache_pages, rd_backfills, (collected_number)stats_array[9]);
678 + rrddim_set_by_pointer(st_pg_cache_pages, rd_evictions, (collected_number)stats_array[10]);
679 + rrdset_done(st_pg_cache_pages);
680 + }
681 +
682 + // ----------------------------------------------------------------
683 +
684 + {
685 + static RRDSET *st_io_stats = NULL;
686 + static RRDDIM *rd_reads = NULL;
687 + static RRDDIM *rd_writes = NULL;
688 +
689 + if (unlikely(!st_io_stats)) {
690 + st_io_stats = rrdset_create_localhost(
691 + "netdata"
692 + , "dbengine_io_throughput"
693 + , NULL
694 + , "dbengine"
695 + , NULL
696 + , "NetData DB engine I/O throughput"
697 + , "MiB/s"
698 + , "netdata"
699 + , "stats"
700 + , 130505
701 + , localhost->rrd_update_every
702 + , RRDSET_TYPE_LINE
703 + );
704 +
705 + rd_reads = rrddim_add(st_io_stats, "reads", NULL, 1, 1024 * 1024, RRD_ALGORITHM_INCREMENTAL);
706 + rd_writes = rrddim_add(st_io_stats, "writes", NULL, -1, 1024 * 1024, RRD_ALGORITHM_INCREMENTAL);
707 + }
708 + else
709 + rrdset_next(st_io_stats);
710 +
711 + rrddim_set_by_pointer(st_io_stats, rd_reads, (collected_number)stats_array[17]);
712 + rrddim_set_by_pointer(st_io_stats, rd_writes, (collected_number)stats_array[15]);
713 + rrdset_done(st_io_stats);
714 + }
715 +
716 + // ----------------------------------------------------------------
717 +
718 + {
719 + static RRDSET *st_io_stats = NULL;
720 + static RRDDIM *rd_reads = NULL;
721 + static RRDDIM *rd_writes = NULL;
722 +
723 + if (unlikely(!st_io_stats)) {
724 + st_io_stats = rrdset_create_localhost(
725 + "netdata"
726 + , "dbengine_io_operations"
727 + , NULL
728 + , "dbengine"
729 + , NULL
730 + , "NetData DB engine I/O operations"
731 + , "operations/s"
732 + , "netdata"
733 + , "stats"
734 + , 130506
735 + , localhost->rrd_update_every
736 + , RRDSET_TYPE_LINE
737 + );
738 +
739 + rd_reads = rrddim_add(st_io_stats, "reads", NULL, 1, 1, RRD_ALGORITHM_INCREMENTAL);
740 + rd_writes = rrddim_add(st_io_stats, "writes", NULL, -1, 1, RRD_ALGORITHM_INCREMENTAL);
741 + }
742 + else
743 + rrdset_next(st_io_stats);
744 +
745 + rrddim_set_by_pointer(st_io_stats, rd_reads, (collected_number)stats_array[18]);
746 + rrddim_set_by_pointer(st_io_stats, rd_writes, (collected_number)stats_array[16]);
747 + rrdset_done(st_io_stats);
748 + }
749 + }
750 +#endif
751 +
752 }
daemon/main.c
+35 -1
@@ -301,6 +301,7 @@ int help(int exitcode) {
301 " -W stacksize=N Set the stacksize (in bytes).\n\n"
302 " -W debug_flags=N Set runtime tracing to debug.log.\n\n"
303 " -W unittest Run internal unittests and exit.\n\n"
304 + " -W createdataset=N Create a DB engine dataset of N seconds and exit.\n\n"
305 " -W set section option value\n"
306 " set netdata.conf option from the command line.\n\n"
307 " -W simple-pattern pattern string\n"
@@ -471,6 +472,25 @@ static void get_netdata_configured_variables() {
472
473 default_rrd_memory_mode = rrd_memory_mode_id(config_get(CONFIG_SECTION_GLOBAL, "memory mode", rrd_memory_mode_name(default_rrd_memory_mode)));
474
475 +#ifdef ENABLE_DBENGINE
476 + // ------------------------------------------------------------------------
477 + // get default Database Engine page cache size in MiB
478 +
479 + default_rrdeng_page_cache_mb = (int) config_get_number(CONFIG_SECTION_GLOBAL, "page cache size", default_rrdeng_page_cache_mb);
480 + if(default_rrdeng_page_cache_mb < RRDENG_MIN_PAGE_CACHE_SIZE_MB) {
481 + error("Invalid page cache size %d given. Defaulting to %d.", default_rrdeng_page_cache_mb, RRDENG_MIN_PAGE_CACHE_SIZE_MB);
482 + default_rrdeng_page_cache_mb = RRDENG_MIN_PAGE_CACHE_SIZE_MB;
483 + }
484 +
485 + // ------------------------------------------------------------------------
486 + // get default Database Engine disk space quota in MiB
487 +
488 + default_rrdeng_disk_quota_mb = (int) config_get_number(CONFIG_SECTION_GLOBAL, "dbengine disk space", default_rrdeng_disk_quota_mb);
489 + if(default_rrdeng_disk_quota_mb < RRDENG_MIN_DISK_SPACE_MB) {
490 + error("Invalid dbengine disk space %d given. Defaulting to %d.", default_rrdeng_disk_quota_mb, RRDENG_MIN_DISK_SPACE_MB);
491 + default_rrdeng_disk_quota_mb = RRDENG_MIN_DISK_SPACE_MB;
492 + }
493 +#endif
494 // ------------------------------------------------------------------------
495
496 netdata_configured_host_prefix = config_get(CONFIG_SECTION_GLOBAL, "host access prefix", "");
@@ -841,6 +861,7 @@ int main(int argc, char **argv) {
861 {
862 char* stacksize_string = "stacksize=";
863 char* debug_flags_string = "debug_flags=";
864 + char* createdataset_string = "createdataset=";
865
866 if(strcmp(optarg, "unittest") == 0) {
867 if(unit_test_buffer()) return 1;
@@ -853,9 +874,23 @@ int main(int argc, char **argv) {
874 default_rrdpush_enabled = 0;
875 if(run_all_mockup_tests()) return 1;
876 if(unit_test_storage()) return 1;
877 +#ifdef ENABLE_DBENGINE
878 + if(test_dbengine()) return 1;
879 +#endif
880 fprintf(stderr, "\n\nALL TESTS PASSED\n\n");
881 return 0;
882 }
883 + else if(strncmp(optarg, createdataset_string, strlen(createdataset_string)) == 0) {
884 + unsigned history_seconds;
885 +
886 + optarg += strlen(createdataset_string);
887 + history_seconds = (unsigned )strtoull(optarg, NULL, 0);
888 +
889 +#ifdef ENABLE_DBENGINE
890 + generate_dbengine_dataset(history_seconds);
891 +#endif
892 + return 0;
893 + }
894 else if(strcmp(optarg, "simple-pattern") == 0) {
895 if(optind + 2 > argc) {
896 fprintf(stderr, "%s", "\nUSAGE: -W simple-pattern 'pattern' 'string'\n\n"
@@ -1138,7 +1173,6 @@ int main(int argc, char **argv) {
1173
1174 rrd_init(netdata_configured_hostname, system_info);
1175 rrdhost_system_info_free(system_info);
1141 -
1176 // ------------------------------------------------------------------------
1177 // enable log flood protection
1178
daemon/unit_test.c
+212
@@ -1566,3 +1566,215 @@ int unit_test(long delay, long shift)
1566
1567 return ret;
1568 }
1569 +
1570 +#ifdef ENABLE_DBENGINE
1571 +static inline void rrddim_set_by_pointer_fake_time(RRDDIM *rd, collected_number value, time_t now)
1572 +{
1573 + rd->last_collected_time.tv_sec = now;
1574 + rd->last_collected_time.tv_usec = 0;
1575 + rd->collected_value = value;
1576 + rd->updated = 1;
1577 +
1578 + rd->collections_counter++;
1579 +
1580 + collected_number v = (value >= 0) ? value : -value;
1581 + if(unlikely(v > rd->collected_value_max)) rd->collected_value_max = v;
1582 +}
1583 +
1584 +int test_dbengine(void)
1585 +{
1586 + const int CHARTS = 128;
1587 + const int DIMS = 16; /* That gives us 2048 metrics */
1588 + const int POINTS = 16384; /* This produces 128MiB of metric data */
1589 + const int QUERY_BATCH = 4096;
1590 + uint8_t same;
1591 + int i, j, k, c, errors;
1592 + RRDHOST *host = NULL;
1593 + RRDSET *st[CHARTS];
1594 + RRDDIM *rd[CHARTS][DIMS];
1595 + char name[101];
1596 + time_t time_now;
1597 + collected_number last;
1598 + struct rrddim_query_handle handle;
1599 + calculated_number value, expected;
1600 + storage_number n;
1601 +
1602 + error_log_limit_unlimited();
1603 + fprintf(stderr, "\nRunning DB-engine test\n");
1604 +
1605 + default_rrd_memory_mode = RRD_MEMORY_MODE_DBENGINE;
1606 +
1607 + debug(D_RRDHOST, "Initializing localhost with hostname 'unittest-dbengine'");
1608 + host = rrdhost_find_or_create(
1609 + "unittest-dbengine"
1610 + , "unittest-dbengine"
1611 + , "unittest-dbengine"
1612 + , os_type
1613 + , netdata_configured_timezone
1614 + , config_get(CONFIG_SECTION_BACKEND, "host tags", "")
1615 + , program_name
1616 + , program_version
1617 + , default_rrd_update_every
1618 + , default_rrd_history_entries
1619 + , RRD_MEMORY_MODE_DBENGINE
1620 + , default_health_enabled
1621 + , default_rrdpush_enabled
1622 + , default_rrdpush_destination
1623 + , default_rrdpush_api_key
1624 + , default_rrdpush_send_charts_matching
1625 + , NULL
1626 + );
1627 + if (NULL == host)
1628 + return 1;
1629 +
1630 + for (i = 0 ; i < CHARTS ; ++i) {
1631 + snprintfz(name, 100, "dbengine-chart-%d", i);
1632 +
1633 + // create the chart
1634 + st[i] = rrdset_create(host, "netdata", name, name, "netdata", NULL, "Unit Testing", "a value", "unittest",
1635 + NULL, 1, 1, RRDSET_TYPE_LINE);
1636 + rrdset_flag_set(st[i], RRDSET_FLAG_DEBUG);
1637 + rrdset_flag_set(st[i], RRDSET_FLAG_STORE_FIRST);
1638 + for (j = 0 ; j < DIMS ; ++j) {
1639 + snprintfz(name, 100, "dim-%d", j);
1640 +
1641 + rd[i][j] = rrddim_add(st[i], name, NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
1642 + }
1643 + }
1644 +
1645 + // feed it with the test data
1646 + time_now = 1;
1647 + last = 0;
1648 + for (i = 0 ; i < CHARTS ; ++i) {
1649 + for (j = 0 ; j < DIMS ; ++j) {
1650 + rd[i][j]->last_collected_time.tv_sec =
1651 + st[i]->last_collected_time.tv_sec = st[i]->last_updated.tv_sec = time_now;
1652 + rd[i][j]->last_collected_time.tv_usec =
1653 + st[i]->last_collected_time.tv_usec = st[i]->last_updated.tv_usec = 0;
1654 + }
1655 + }
1656 + for(c = 0; c < POINTS ; ++c) {
1657 + ++time_now; // time_now = c + 2
1658 + for (i = 0 ; i < CHARTS ; ++i) {
1659 + st[i]->usec_since_last_update = USEC_PER_SEC;
1660 +
1661 + for (j = 0; j < DIMS; ++j) {
1662 + last = i * DIMS * POINTS + j * POINTS + c;
1663 + rrddim_set_by_pointer_fake_time(rd[i][j], last, time_now);
1664 + }
1665 + rrdset_done(st[i]);
1666 + }
1667 + }
1668 +
1669 + // check the result
1670 + errors = 0;
1671 +
1672 + for(c = 0; c < POINTS ; c += QUERY_BATCH) {
1673 + time_now = c + 2;
1674 + for (i = 0 ; i < CHARTS ; ++i) {
1675 + for (j = 0; j < DIMS; ++j) {
1676 + rd[i][j]->state->query_ops.init(rd[i][j], &handle, time_now, time_now + QUERY_BATCH);
1677 + for (k = 0; k < QUERY_BATCH; ++k) {
1678 + last = i * DIMS * POINTS + j * POINTS + c + k;
1679 + expected = unpack_storage_number(pack_storage_number((calculated_number)last, SN_EXISTS));
1680 +
1681 + n = rd[i][j]->state->query_ops.next_metric(&handle);
1682 + value = unpack_storage_number(n);
1683 +
1684 + same = (calculated_number_round(value * 10000000.0) == calculated_number_round(expected * 10000000.0)) ? 1 : 0;
1685 + if(!same) {
1686 + fprintf(stderr, " DB-engine unittest %s/%s: at %lu secs, expecting value "
1687 + CALCULATED_NUMBER_FORMAT ", found " CALCULATED_NUMBER_FORMAT ", ### E R R O R ###\n",
1688 + st[i]->name, rd[i][j]->name, (unsigned long)time_now + k, expected, value);
1689 + errors++;
1690 + }
1691 + }
1692 + rd[i][j]->state->query_ops.finalize(&handle);
1693 + }
1694 + }
1695 + }
1696 +
1697 + rrdeng_exit(host->rrdeng_ctx);
1698 + rrd_wrlock();
1699 + rrdhost_delete_charts(host);
1700 + rrd_unlock();
1701 +
1702 + return errors;
1703 +}
1704 +
1705 +void generate_dbengine_dataset(unsigned history_seconds)
1706 +{
1707 + const int DIMS = 128;
1708 + const uint64_t EXPECTED_COMPRESSION_RATIO = 94;
1709 + int j;
1710 + RRDHOST *host = NULL;
1711 + RRDSET *st;
1712 + RRDDIM *rd[DIMS];
1713 + char name[101];
1714 + time_t time_current, time_present;
1715 +
1716 + default_rrd_memory_mode = RRD_MEMORY_MODE_DBENGINE;
1717 + default_rrdeng_page_cache_mb = 128;
1718 + /* Worst case for uncompressible data */
1719 + default_rrdeng_disk_quota_mb = (((uint64_t)DIMS) * sizeof(storage_number) * history_seconds) / (1024 * 1024);
1720 + default_rrdeng_disk_quota_mb -= default_rrdeng_disk_quota_mb * EXPECTED_COMPRESSION_RATIO / 100;
1721 +
1722 + error_log_limit_unlimited();
1723 + debug(D_RRDHOST, "Initializing localhost with hostname 'dbengine-dataset'");
1724 +
1725 + host = rrdhost_find_or_create(
1726 + "dbengine-dataset"
1727 + , "dbengine-dataset"
1728 + , "dbengine-dataset"
1729 + , os_type
1730 + , netdata_configured_timezone
1731 + , config_get(CONFIG_SECTION_BACKEND, "host tags", "")
1732 + , program_name
1733 + , program_version
1734 + , default_rrd_update_every
1735 + , default_rrd_history_entries
1736 + , RRD_MEMORY_MODE_DBENGINE
1737 + , default_health_enabled
1738 + , default_rrdpush_enabled
1739 + , default_rrdpush_destination
1740 + , default_rrdpush_api_key
1741 + , default_rrdpush_send_charts_matching
1742 + , NULL
1743 + );
1744 + if (NULL == host)
1745 + return;
1746 +
1747 + fprintf(stderr, "\nRunning DB-engine workload generator\n");
1748 +
1749 + // create the chart
1750 + st = rrdset_create(host, "example", "random", "random", "example", NULL, "random", "random", "random",
1751 + NULL, 1, 1, RRDSET_TYPE_LINE);
1752 + for (j = 0 ; j < DIMS ; ++j) {
1753 + snprintfz(name, 100, "random%d", j);
1754 +
1755 + rd[j] = rrddim_add(st, name, NULL, 1, 1, RRD_ALGORITHM_ABSOLUTE);
1756 + }
1757 +
1758 + time_present = now_realtime_sec();
1759 + // feed it with the test data
1760 + time_current = time_present - history_seconds;
1761 + for (j = 0 ; j < DIMS ; ++j) {
1762 + rd[j]->last_collected_time.tv_sec =
1763 + st->last_collected_time.tv_sec = st->last_updated.tv_sec = time_current;
1764 + rd[j]->last_collected_time.tv_usec =
1765 + st->last_collected_time.tv_usec = st->last_updated.tv_usec = 0;
1766 + }
1767 + for( ; time_current < time_present; ++time_current) {
1768 + st->usec_since_last_update = USEC_PER_SEC;
1769 +
1770 + for (j = 0; j < DIMS; ++j) {
1771 + rrddim_set_by_pointer_fake_time(rd[j], (time_current + j) % 128, time_current);
1772 + }
1773 + rrdset_done(st);
1774 + }
1775 + rrd_wrlock();
1776 + rrdhost_free(host);
1777 + rrd_unlock();
1778 +
1779 +}
1780 +#endif
\ No newline at end of file
daemon/unit_test.h
+4
@@ -8,5 +8,9 @@ extern int unit_test(long delay, long shift);
8 extern int run_all_mockup_tests(void);
9 extern int unit_test_str2ld(void);
10 extern int unit_test_buffer(void);
11 +#ifdef ENABLE_DBENGINE
12 +extern int test_dbengine(void);
13 +extern void generate_dbengine_dataset(unsigned history_seconds);
14 +#endif
15
16 #endif /* NETDATA_UNIT_TEST_H */
database/Makefile.am
+4
@@ -3,6 +3,10 @@
3 AUTOMAKE_OPTIONS = subdir-objects
4 MAINTAINERCLEANFILES = $(srcdir)/Makefile.in
5
6 +SUBDIRS = \
7 + engine \
8 + $(NULL)
9 +
10 dist_noinst_DATA = \
11 README.md \
12 $(NULL)
database/README.md
+28 -6
@@ -17,12 +17,13 @@ to 1 second. You will have just one hour of data.
17 For a day of data and 1.000 dimensions, you will need: 86.400 seconds * 4 bytes * 1.000
18 dimensions = 345MB of RAM.
19
20 -Currently the only option you have to lower this number is to use
21 -**[Memory Deduplication - Kernel Same Page Merging - KSM](#ksm)**.
20 +One option you have to lower this number is to use
21 +**[Memory Deduplication - Kernel Same Page Merging - KSM](#ksm)**. Another possibility is to
22 +use the **[Database Engine](engine/)**.
23
24 ## Memory modes
25
25 -Currently netdata supports 5 memory modes:
26 +Currently netdata supports 6 memory modes:
27
28 1. `ram`, data are purely in memory. Data are never saved on disk. This mode uses `mmap()` and
29 supports [KSM](#ksm).
@@ -42,6 +43,12 @@ Currently netdata supports 5 memory modes:
43 5. `alloc`, like `ram` but it uses `calloc()` and does not support [KSM](#ksm). This mode is the
44 fallback for all others except `none`.
45
46 +6. `dbengine`, data are in database files. The [Database Engine](engine/) works like a traditional
47 + database. There is some amount of RAM dedicated to data caching and indexing and the rest of
48 + the data reside compressed on disk. The number of history entries is not fixed in this case,
49 + but depends on the configured disk space and the effective compression ratio of the data stored.
50 + For more details see [here](engine/).
51 +
52 You can select the memory mode by editing netdata.conf and setting:
53
54 ```
@@ -80,7 +87,7 @@ server that will maintain the entire database for all nodes, and will also run h
87 for all nodes.
88
89 For this central netdata, memory size can be a problem. Fortunately, netdata supports several
83 -memory modes. What is interesting for this setup is `memory mode = map`.
90 +memory modes. One interesting option for this setup is `memory mode = map`.
91
92 In this mode, the database of netdata is stored in memory mapped files. netdata continues to read
93 and write the database in memory, but the kernel automatically loads and saves memory pages from/to
@@ -88,7 +95,7 @@ disk.
95
96 **We suggest _not_ to use this mode on nodes that run other applications.** There will always be
97 dirty memory to be synced and this syncing process may influence the way other applications work.
91 -This mode however is ideal when we need a central netdata server that would normally need huge
98 +This mode however is useful when we need a central netdata server that would normally need huge
99 amounts of memory. Using memory mode `map` we can overcome all memory restrictions.
100
101 There are a few kernel options that provide finer control on the way this syncing works. But before
@@ -155,9 +162,24 @@ vm.dirty_ratio = 90
162 vm.dirty_writeback_centisecs = 0
163 ```
164
165 +There is another memory mode to help overcome the memory size problem. What is most interesting
166 +for this setup is `memory mode = dbengine`.
167 +
168 +In this mode, the database of netdata is stored in database files. The [Database Engine](engine/)
169 +works like a traditional database. There is some amount of RAM dedicated to data caching and
170 +indexing and the rest of the data reside compressed on disk. The number of history entries is not
171 +fixed in this case, but depends on the configured disk space and the effective compression ratio
172 +of the data stored.
173 +
174 +We suggest to use **this** mode on nodes that also run other applications. The Database Engine uses
175 +direct I/O to avoid polluting the OS filesystem caches and does not generate excessive I/O traffic
176 +so as to create the minimum possible interference with other applications. Using memory mode
177 +`dbengine` we can overcome most memory restrictions. For more details see [here](engine/).
178 +
179 ## KSM
180
160 -Netdata offers all its round robin database to kernel for deduplication.
181 +Netdata offers all its round robin database to kernel for deduplication
182 +(except for `memory mode = dbengine`).
183
184 In the past KSM has been criticized for consuming a lot of CPU resources.
185 Although this is true when KSM is used for deduplicating certain applications, it is not true with
database/engine/Makefile.am new
+8
@@ -0,0 +1,8 @@
1 +# SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +AUTOMAKE_OPTIONS = subdir-objects
4 +MAINTAINERCLEANFILES = $(srcdir)/Makefile.in
5 +
6 +dist_noinst_DATA = \
7 + README.md \
8 + $(NULL)
database/engine/README.md new
+109
@@ -0,0 +1,109 @@
1 +# Database engine
2 +
3 +The Database Engine works like a traditional
4 +database. There is some amount of RAM dedicated to data caching and indexing and the rest of
5 +the data reside compressed on disk. The number of history entries is not fixed in this case,
6 +but depends on the configured disk space and the effective compression ratio of the data stored.
7 +
8 +## Files
9 +
10 +With the DB engine memory mode the metric data are stored in database files. These files are
11 +organized in pairs, the datafiles and their corresponding journalfiles, e.g.:
12 +
13 +```
14 +datafile-1-0000000001.ndf
15 +journalfile-1-0000000001.njf
16 +datafile-1-0000000002.ndf
17 +journalfile-1-0000000002.njf
18 +datafile-1-0000000003.ndf
19 +journalfile-1-0000000003.njf
20 +...
21 +```
22 +
23 +They are located under their host's cache directory in the directory `./dbengine`
24 +(e.g. for localhost the default location is `/var/cache/netdata/dbengine/*`). The higher
25 +numbered filenames contain more recent metric data. The user can safely delete some pairs
26 +of files when netdata is stopped to manually free up some space.
27 +
28 +*Users should* **back up** *their `./dbengine` folders if they consider this data to be important.*
29 +
30 +## Configuration
31 +
32 +There is one DB engine instance per netdata host/node. That is, there is one `./dbengine` folder
33 +per node, and all charts of `dbengine` memory mode in such a host share the same storage space
34 +and DB engine instance memory state. You can select the memory mode for localhost by editing
35 +netdata.conf and setting:
36 +
37 +```
38 +[global]
39 + memory mode = dbengine
40 +```
41 +
42 +For setting the memory mode for the rest of the nodes you should look at
43 +[streaming](../../streaming/).
44 +
45 +The `history` configuration option is meaningless for `memory mode = dbengine` and is ignored
46 +for any metrics being stored in the DB engine.
47 +
48 +All DB engine instances, for localhost and all other streaming recipient nodes inherit their
49 +configuration from `netdata.conf`:
50 +
51 +```
52 +[global]
53 + page cache size = 32
54 + dbengine disk space = 256
55 +```
56 +
57 +The above values are the default and minimum values for Page Cache size and DB engine disk space
58 +quota. Both numbers are in **MiB**. All DB engine instances will allocate the configured resources
59 +separately.
60 +
61 +The `page cache size` option determines the amount of RAM in **MiB** that is dedicated to caching
62 +netdata metric values themselves.
63 +
64 +The `dbengine disk space` option determines the amount of disk space in **MiB** that is dedicated
65 +to storing netdata metric values and all related metadata describing them.
66 +
67 +## Operation
68 +
69 +The DB engine stores chart metric values in 4096-byte pages in memory. Each chart dimension gets
70 +its own page to store consecutive values generated from the data collectors. Those pages comprise
71 +the **Page Cache**.
72 +
73 +When those pages fill up they are slowly compressed and flushed to disk.
74 +It can take `4096 / 4 = 1024 seconds = 17 minutes`, for a chart dimension that is being collected
75 +every 1 second, to fill a page. Pages can be cut short when we stop netdata or the DB engine
76 +instance so as to not lose the data. When we query the DB engine for data we trigger disk read
77 +I/O requests that fill the Page Cache with the requested pages and potentially evict cold
78 +(not recently used) pages.
79 +
80 +When the disk quota is exceeded the oldest values are removed from the DB engine at real time, by
81 +automatically deleting the oldest datafile and journalfile pair. Any corresponding pages residing
82 +in the Page Cache will also be invalidated and removed. The DB engine logic will try to maintain
83 +between 10 and 20 file pairs at any point in time.
84 +
85 +The Database Engine uses direct I/O to avoid polluting the OS filesystem caches and does not
86 +generate excessive I/O traffic so as to create the minimum possible interference with other
87 +applications.
88 +
89 +## Memory requirements
90 +
91 +Using memory mode `dbengine` we can overcome most memory restrictions and store a dataset that
92 +is much larger than the available memory.
93 +
94 +There are explicit memory requirements **per** DB engine **instance**, meaning **per** netdata
95 +**node** (e.g. localhost and streaming recipient nodes):
96 +
97 +- `page cache size` must be at least `#dimensions-being-collected x 4096 x 2` bytes.
98 +
99 +- an additional `#pages-on-disk x 4096 x 0.06` bytes of RAM are allocated for metadata.
100 +
101 + - roughly speaking this is 6% of the uncompressed disk space taken by the DB files.
102 +
103 + - for very highly compressible data (compression ratio > 90%) this RAM overhead
104 + is comparable to the disk space footprint.
105 +
106 +An important observation is that RAM usage depends on both the `page cache size` and the
107 +`dbengine disk space` options.
108 +
109 +[![analytics](https://www.google-analytics.com/collect?v=1&aip=1&t=pageview&_s=1&ds=github&dr=https%3A%2F%2Fgithub.com%2Fnetdata%2Fnetdata&dl=https%3A%2F%2Fmy-netdata.io%2Fgithub%2Fdatabase%2Fengine%2FREADME&_u=MAC~&cid=5792dfd7-8dc4-476b-af31-da2fdb9f93d2&tid=UA-64295674-3)]()
database/engine/datafile.c new
+335
@@ -0,0 +1,335 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +#include "rrdengine.h"
3 +
4 +void df_extent_insert(struct extent_info *extent)
5 +{
6 + struct rrdengine_datafile *datafile = extent->datafile;
7 +
8 + if (likely(NULL != datafile->extents.last)) {
9 + datafile->extents.last->next = extent;
10 + }
11 + if (unlikely(NULL == datafile->extents.first)) {
12 + datafile->extents.first = extent;
13 + }
14 + datafile->extents.last = extent;
15 +}
16 +
17 +void datafile_list_insert(struct rrdengine_instance *ctx, struct rrdengine_datafile *datafile)
18 +{
19 + if (likely(NULL != ctx->datafiles.last)) {
20 + ctx->datafiles.last->next = datafile;
21 + }
22 + if (unlikely(NULL == ctx->datafiles.first)) {
23 + ctx->datafiles.first = datafile;
24 + }
25 + ctx->datafiles.last = datafile;
26 +}
27 +
28 +void datafile_list_delete(struct rrdengine_instance *ctx, struct rrdengine_datafile *datafile)
29 +{
30 + struct rrdengine_datafile *next;
31 +
32 + next = datafile->next;
33 + assert((NULL != next) && (ctx->datafiles.first == datafile) && (ctx->datafiles.last != datafile));
34 + ctx->datafiles.first = next;
35 +}
36 +
37 +
38 +static void datafile_init(struct rrdengine_datafile *datafile, struct rrdengine_instance *ctx,
39 + unsigned tier, unsigned fileno)
40 +{
41 + assert(tier == 1);
42 + datafile->tier = tier;
43 + datafile->fileno = fileno;
44 + datafile->file = (uv_file)0;
45 + datafile->pos = 0;
46 + datafile->extents.first = datafile->extents.last = NULL; /* will be populated by journalfile */
47 + datafile->journalfile = NULL;
48 + datafile->next = NULL;
49 + datafile->ctx = ctx;
50 +}
51 +
52 +static void generate_datafilepath(struct rrdengine_datafile *datafile, char *str, size_t maxlen)
53 +{
54 + (void) snprintf(str, maxlen, "%s/" DATAFILE_PREFIX RRDENG_FILE_NUMBER_PRINT_TMPL DATAFILE_EXTENSION,
55 + datafile->ctx->dbfiles_path, datafile->tier, datafile->fileno);
56 +}
57 +
58 +int destroy_data_file(struct rrdengine_datafile *datafile)
59 +{
60 + struct rrdengine_instance *ctx = datafile->ctx;
61 + uv_fs_t req;
62 + int ret, fd;
63 + char path[1024];
64 +
65 + ret = uv_fs_ftruncate(NULL, &req, datafile->file, 0, NULL);
66 + if (ret < 0) {
67 + fatal("uv_fs_ftruncate: %s", uv_strerror(ret));
68 + }
69 + assert(0 == req.result);
70 + uv_fs_req_cleanup(&req);
71 +
72 + ret = uv_fs_close(NULL, &req, datafile->file, NULL);
73 + if (ret < 0) {
74 + fatal("uv_fs_close: %s", uv_strerror(ret));
75 + }
76 + assert(0 == req.result);
77 + uv_fs_req_cleanup(&req);
78 +
79 + generate_datafilepath(datafile, path, sizeof(path));
80 + fd = uv_fs_unlink(NULL, &req, path, NULL);
81 + if (fd < 0) {
82 + fatal("uv_fs_fsunlink: %s", uv_strerror(fd));
83 + }
84 + assert(0 == req.result);
85 + uv_fs_req_cleanup(&req);
86 +
87 + ++ctx->stats.datafile_deletions;
88 +
89 + return 0;
90 +}
91 +
92 +int create_data_file(struct rrdengine_datafile *datafile)
93 +{
94 + struct rrdengine_instance *ctx = datafile->ctx;
95 + uv_fs_t req;
96 + uv_file file;
97 + int ret, fd;
98 + struct rrdeng_df_sb *superblock;
99 + uv_buf_t iov;
100 + char path[1024];
101 +
102 + generate_datafilepath(datafile, path, sizeof(path));
103 + fd = uv_fs_open(NULL, &req, path, O_DIRECT | O_CREAT | O_RDWR | O_TRUNC,
104 + S_IRUSR | S_IWUSR, NULL);
105 + if (fd < 0) {
106 + fatal("uv_fs_fsopen: %s", uv_strerror(fd));
107 + }
108 + assert(req.result >= 0);
109 + file = req.result;
110 + uv_fs_req_cleanup(&req);
111 +#ifdef __APPLE__
112 + info("Disabling OS X caching for file \"%s\".", path);
113 + fcntl(fd, F_NOCACHE, 1);
114 +#endif
115 +
116 + ret = posix_memalign((void *)&superblock, RRDFILE_ALIGNMENT, sizeof(*superblock));
117 + if (unlikely(ret)) {
118 + fatal("posix_memalign:%s", strerror(ret));
119 + }
120 + (void) strncpy(superblock->magic_number, RRDENG_DF_MAGIC, RRDENG_MAGIC_SZ);
121 + (void) strncpy(superblock->version, RRDENG_DF_VER, RRDENG_VER_SZ);
122 + superblock->tier = 1;
123 +
124 + iov = uv_buf_init((void *)superblock, sizeof(*superblock));
125 +
126 + ret = uv_fs_write(NULL, &req, file, &iov, 1, 0, NULL);
127 + if (ret < 0) {
128 + fatal("uv_fs_write: %s", uv_strerror(ret));
129 + }
130 + if (req.result < 0) {
131 + fatal("uv_fs_write: %s", uv_strerror((int)req.result));
132 + }
133 + uv_fs_req_cleanup(&req);
134 + free(superblock);
135 +
136 + datafile->file = file;
137 + datafile->pos = sizeof(*superblock);
138 + ctx->stats.io_write_bytes += sizeof(*superblock);
139 + ++ctx->stats.io_write_requests;
140 + ++ctx->stats.datafile_creations;
141 +
142 + return 0;
143 +}
144 +
145 +static int check_data_file_superblock(uv_file file)
146 +{
147 + int ret;
148 + struct rrdeng_df_sb *superblock;
149 + uv_buf_t iov;
150 + uv_fs_t req;
151 +
152 + ret = posix_memalign((void *)&superblock, RRDFILE_ALIGNMENT, sizeof(*superblock));
153 + if (unlikely(ret)) {
154 + fatal("posix_memalign:%s", strerror(ret));
155 + }
156 + iov = uv_buf_init((void *)superblock, sizeof(*superblock));
157 +
158 + ret = uv_fs_read(NULL, &req, file, &iov, 1, 0, NULL);
159 + if (ret < 0) {
160 + error("uv_fs_read: %s", uv_strerror(ret));
161 + uv_fs_req_cleanup(&req);
162 + goto error;
163 + }
164 + assert(req.result >= 0);
165 + uv_fs_req_cleanup(&req);
166 +
167 + if (strncmp(superblock->magic_number, RRDENG_DF_MAGIC, RRDENG_MAGIC_SZ) ||
168 + strncmp(superblock->version, RRDENG_DF_VER, RRDENG_VER_SZ) ||
169 + superblock->tier != 1) {
170 + error("File has invalid superblock.");
171 + ret = UV_EINVAL;
172 + } else {
173 + ret = 0;
174 + }
175 + error:
176 + free(superblock);
177 + return ret;
178 +}
179 +
180 +static int load_data_file(struct rrdengine_datafile *datafile)
181 +{
182 + struct rrdengine_instance *ctx = datafile->ctx;
183 + uv_fs_t req;
184 + uv_file file;
185 + int ret, fd;
186 + uint64_t file_size;
187 + char path[1024];
188 +
189 + generate_datafilepath(datafile, path, sizeof(path));
190 + fd = uv_fs_open(NULL, &req, path, O_DIRECT | O_RDWR, S_IRUSR | S_IWUSR, NULL);
191 + if (fd < 0) {
192 + /* if (UV_ENOENT != fd) */
193 + error("uv_fs_fsopen: %s", uv_strerror(fd));
194 + uv_fs_req_cleanup(&req);
195 + return fd;
196 + }
197 + assert(req.result >= 0);
198 + file = req.result;
199 + uv_fs_req_cleanup(&req);
200 +#ifdef __APPLE__
201 + info("Disabling OS X caching for file \"%s\".", path);
202 + fcntl(fd, F_NOCACHE, 1);
203 +#endif
204 + info("Initializing data file \"%s\".", path);
205 +
206 + ret = check_file_properties(file, &file_size, sizeof(struct rrdeng_df_sb));
207 + if (ret)
208 + goto error;
209 + file_size = ALIGN_BYTES_CEILING(file_size);
210 +
211 + ret = check_data_file_superblock(file);
212 + if (ret)
213 + goto error;
214 + ctx->stats.io_read_bytes += sizeof(struct rrdeng_df_sb);
215 + ++ctx->stats.io_read_requests;
216 +
217 + datafile->file = file;
218 + datafile->pos = file_size;
219 +
220 + info("Data file \"%s\" initialized (size:%"PRIu64").", path, file_size);
221 + return 0;
222 +
223 + error:
224 + (void) uv_fs_close(NULL, &req, file, NULL);
225 + uv_fs_req_cleanup(&req);
226 + return ret;
227 +}
228 +
229 +static int scan_data_files_cmp(const void *a, const void *b)
230 +{
231 + struct rrdengine_datafile *file1, *file2;
232 + char path1[1024], path2[1024];
233 +
234 + file1 = *(struct rrdengine_datafile **)a;
235 + file2 = *(struct rrdengine_datafile **)b;
236 + generate_datafilepath(file1, path1, sizeof(path1));
237 + generate_datafilepath(file2, path2, sizeof(path2));
238 + return strcmp(path1, path2);
239 +}
240 +
241 +/* Returns number of datafiles that were loaded */
242 +static int scan_data_files(struct rrdengine_instance *ctx)
243 +{
244 + int ret;
245 + unsigned tier, no, matched_files, i,failed_to_load;
246 + static uv_fs_t req;
247 + uv_dirent_t dent;
248 + struct rrdengine_datafile **datafiles, *datafile;
249 + struct rrdengine_journalfile *journalfile;
250 +
251 + ret = uv_fs_scandir(NULL, &req, ctx->dbfiles_path, 0, NULL);
252 + assert(ret >= 0);
253 + assert(req.result >= 0);
254 + info("Found %d files in path %s", ret, ctx->dbfiles_path);
255 +
256 + datafiles = callocz(MIN(ret, MAX_DATAFILES), sizeof(*datafiles));
257 + for (matched_files = 0 ; UV_EOF != uv_fs_scandir_next(&req, &dent) && matched_files < MAX_DATAFILES ; ) {
258 + info("Scanning file \"%s\"", dent.name);
259 + ret = sscanf(dent.name, DATAFILE_PREFIX RRDENG_FILE_NUMBER_SCAN_TMPL DATAFILE_EXTENSION, &tier, &no);
260 + if (2 == ret) {
261 + info("Matched file \"%s\"", dent.name);
262 + datafile = mallocz(sizeof(*datafile));
263 + datafile_init(datafile, ctx, tier, no);
264 + datafiles[matched_files++] = datafile;
265 + }
266 + }
267 + uv_fs_req_cleanup(&req);
268 +
269 + if (matched_files == MAX_DATAFILES) {
270 + error("Warning: hit maximum database engine file limit of %d files", MAX_DATAFILES);
271 + }
272 + qsort(datafiles, matched_files, sizeof(*datafiles), scan_data_files_cmp);
273 + for (failed_to_load = 0, i = 0 ; i < matched_files ; ++i) {
274 + datafile = datafiles[i];
275 + ret = load_data_file(datafile);
276 + if (0 != ret) {
277 + free(datafile);
278 + ++failed_to_load;
279 + continue;
280 + }
281 + journalfile = mallocz(sizeof(*journalfile));
282 + datafile->journalfile = journalfile;
283 + journalfile_init(journalfile, datafile);
284 + ret = load_journal_file(ctx, journalfile, datafile);
285 + if (0 != ret) {
286 + free(datafile);
287 + free(journalfile);
288 + ++failed_to_load;
289 + continue;
290 + }
291 + datafile_list_insert(ctx, datafile);
292 + ctx->disk_space += datafile->pos + journalfile->pos;
293 + }
294 + if (failed_to_load) {
295 + error("%u files failed to load.", failed_to_load);
296 + }
297 + free(datafiles);
298 +
299 + return matched_files - failed_to_load;
300 +}
301 +
302 +/* Creates a datafile and a journalfile pair */
303 +void create_new_datafile_pair(struct rrdengine_instance *ctx, unsigned tier, unsigned fileno)
304 +{
305 + struct rrdengine_datafile *datafile;
306 + struct rrdengine_journalfile *journalfile;
307 + int ret;
308 +
309 + info("Creating new data and journal files.");
310 + datafile = mallocz(sizeof(*datafile));
311 + datafile_init(datafile, ctx, tier, fileno);
312 + ret = create_data_file(datafile);
313 + assert(!ret);
314 +
315 + journalfile = mallocz(sizeof(*journalfile));
316 + datafile->journalfile = journalfile;
317 + journalfile_init(journalfile, datafile);
318 + ret = create_journal_file(journalfile, datafile);
319 + assert(!ret);
320 + datafile_list_insert(ctx, datafile);
321 + ctx->disk_space += datafile->pos + journalfile->pos;
322 +}
323 +
324 +/* Page cache must already be initialized. */
325 +int init_data_files(struct rrdengine_instance *ctx)
326 +{
327 + int ret;
328 +
329 + ret = scan_data_files(ctx);
330 + if (0 == ret) {
331 + info("Data files not found, creating.");
332 + create_new_datafile_pair(ctx, 1, 1);
333 + }
334 + return 0;
335 +}
\ No newline at end of file
database/engine/datafile.h new
+63
@@ -0,0 +1,63 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_DATAFILE_H
4 +#define NETDATA_DATAFILE_H
5 +
6 +#include "rrdengine.h"
7 +
8 +/* Forward declarations */
9 +struct rrdengine_datafile;
10 +struct rrdengine_journalfile;
11 +struct rrdengine_instance;
12 +
13 +#define DATAFILE_PREFIX "datafile-"
14 +#define DATAFILE_EXTENSION ".ndf"
15 +
16 +#define MAX_DATAFILE_SIZE (1073741824LU)
17 +#define MIN_DATAFILE_SIZE (16777216LU)
18 +#define MAX_DATAFILES (65536) /* Supports up to 64TiB for now */
19 +#define TARGET_DATAFILES (20)
20 +
21 +#define DATAFILE_IDEAL_IO_SIZE (1048576U)
22 +
23 +struct extent_info {
24 + uint64_t offset;
25 + uint32_t size;
26 + uint8_t number_of_pages;
27 + struct rrdengine_datafile *datafile;
28 + struct extent_info *next;
29 + struct rrdeng_page_cache_descr *pages[];
30 +};
31 +
32 +struct rrdengine_df_extents {
33 + /* the extent list is sorted based on disk offset */
34 + struct extent_info *first;
35 + struct extent_info *last;
36 +};
37 +
38 +/* only one event loop is supported for now */
39 +struct rrdengine_datafile {
40 + unsigned tier;
41 + unsigned fileno;
42 + uv_file file;
43 + uint64_t pos;
44 + struct rrdengine_instance *ctx;
45 + struct rrdengine_df_extents extents;
46 + struct rrdengine_journalfile *journalfile;
47 + struct rrdengine_datafile *next;
48 +};
49 +
50 +struct rrdengine_datafile_list {
51 + struct rrdengine_datafile *first; /* oldest */
52 + struct rrdengine_datafile *last; /* newest */
53 +};
54 +
55 +extern void df_extent_insert(struct extent_info *extent);
56 +extern void datafile_list_insert(struct rrdengine_instance *ctx, struct rrdengine_datafile *datafile);
57 +extern void datafile_list_delete(struct rrdengine_instance *ctx, struct rrdengine_datafile *datafile);
58 +extern int destroy_data_file(struct rrdengine_datafile *datafile);
59 +extern int create_data_file(struct rrdengine_datafile *datafile);
60 +extern void create_new_datafile_pair(struct rrdengine_instance *ctx, unsigned tier, unsigned fileno);
61 +extern int init_data_files(struct rrdengine_instance *ctx);
62 +
63 +#endif /* NETDATA_DATAFILE_H */
\ No newline at end of file
database/engine/journalfile.c new
+462
@@ -0,0 +1,462 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +#include "rrdengine.h"
3 +
4 +static void flush_transaction_buffer_cb(uv_fs_t* req)
5 +{
6 + struct generic_io_descriptor *io_descr;
7 +
8 + debug(D_RRDENGINE, "%s: Journal block was written to disk.", __func__);
9 + if (req->result < 0) {
10 + fatal("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
11 + }
12 + io_descr = req->data;
13 +
14 + uv_fs_req_cleanup(req);
15 + free(io_descr->buf);
16 + free(io_descr);
17 +}
18 +
19 +/* Careful to always call this before creating a new journal file */
20 +void wal_flush_transaction_buffer(struct rrdengine_worker_config* wc)
21 +{
22 + struct rrdengine_instance *ctx = wc->ctx;
23 + int ret;
24 + struct generic_io_descriptor *io_descr;
25 + unsigned pos, size;
26 + struct rrdengine_journalfile *journalfile;
27 +
28 + if (unlikely(NULL == ctx->commit_log.buf || 0 == ctx->commit_log.buf_pos)) {
29 + return;
30 + }
31 + /* care with outstanding transactions when switching journal files */
32 + journalfile = ctx->datafiles.last->journalfile;
33 +
34 + io_descr = mallocz(sizeof(*io_descr));
35 + pos = ctx->commit_log.buf_pos;
36 + size = ctx->commit_log.buf_size;
37 + if (pos < size) {
38 + /* simulate an empty transaction to skip the rest of the block */
39 + *(uint8_t *) (ctx->commit_log.buf + pos) = STORE_PADDING;
40 + }
41 + io_descr->buf = ctx->commit_log.buf;
42 + io_descr->bytes = size;
43 + io_descr->pos = journalfile->pos;
44 + io_descr->req.data = io_descr;
45 + io_descr->completion = NULL;
46 +
47 + io_descr->iov = uv_buf_init((void *)io_descr->buf, size);
48 + ret = uv_fs_write(wc->loop, &io_descr->req, journalfile->file, &io_descr->iov, 1,
49 + journalfile->pos, flush_transaction_buffer_cb);
50 + assert (-1 != ret);
51 + journalfile->pos += RRDENG_BLOCK_SIZE;
52 + ctx->disk_space += RRDENG_BLOCK_SIZE;
53 + ctx->commit_log.buf = NULL;
54 + ctx->stats.io_write_bytes += RRDENG_BLOCK_SIZE;
55 + ++ctx->stats.io_write_requests;
56 +}
57 +
58 +void * wal_get_transaction_buffer(struct rrdengine_worker_config* wc, unsigned size)
59 +{
60 + struct rrdengine_instance *ctx = wc->ctx;
61 + int ret;
62 + unsigned buf_pos, buf_size;
63 +
64 + assert(size);
65 + if (ctx->commit_log.buf) {
66 + unsigned remaining;
67 +
68 + buf_pos = ctx->commit_log.buf_pos;
69 + buf_size = ctx->commit_log.buf_size;
70 + remaining = buf_size - buf_pos;
71 + if (size > remaining) {
72 + /* we need a new buffer */
73 + wal_flush_transaction_buffer(wc);
74 + }
75 + }
76 + if (NULL == ctx->commit_log.buf) {
77 + buf_size = ALIGN_BYTES_CEILING(size);
78 + ret = posix_memalign((void *)&ctx->commit_log.buf, RRDFILE_ALIGNMENT, buf_size);
79 + if (unlikely(ret)) {
80 + fatal("posix_memalign:%s", strerror(ret));
81 + }
82 + buf_pos = ctx->commit_log.buf_pos = 0;
83 + ctx->commit_log.buf_size = buf_size;
84 + }
85 + ctx->commit_log.buf_pos += size;
86 +
87 + return ctx->commit_log.buf + buf_pos;
88 +}
89 +
90 +static void generate_journalfilepath(struct rrdengine_datafile *datafile, char *str, size_t maxlen)
91 +{
92 + (void) snprintf(str, maxlen, "%s/" WALFILE_PREFIX RRDENG_FILE_NUMBER_PRINT_TMPL WALFILE_EXTENSION,
93 + datafile->ctx->dbfiles_path, datafile->tier, datafile->fileno);
94 +}
95 +
96 +void journalfile_init(struct rrdengine_journalfile *journalfile, struct rrdengine_datafile *datafile)
97 +{
98 + journalfile->file = (uv_file)0;
99 + journalfile->pos = 0;
100 + journalfile->datafile = datafile;
101 +}
102 +
103 +int destroy_journal_file(struct rrdengine_journalfile *journalfile, struct rrdengine_datafile *datafile)
104 +{
105 + struct rrdengine_instance *ctx = datafile->ctx;
106 + uv_fs_t req;
107 + int ret, fd;
108 + char path[1024];
109 +
110 + ret = uv_fs_ftruncate(NULL, &req, journalfile->file, 0, NULL);
111 + if (ret < 0) {
112 + fatal("uv_fs_ftruncate: %s", uv_strerror(ret));
113 + }
114 + assert(0 == req.result);
115 + uv_fs_req_cleanup(&req);
116 +
117 + ret = uv_fs_close(NULL, &req, journalfile->file, NULL);
118 + if (ret < 0) {
119 + fatal("uv_fs_close: %s", uv_strerror(ret));
120 + exit(ret);
121 + }
122 + assert(0 == req.result);
123 + uv_fs_req_cleanup(&req);
124 +
125 + generate_journalfilepath(datafile, path, sizeof(path));
126 + fd = uv_fs_unlink(NULL, &req, path, NULL);
127 + if (fd < 0) {
128 + fatal("uv_fs_fsunlink: %s", uv_strerror(fd));
129 + }
130 + assert(0 == req.result);
131 + uv_fs_req_cleanup(&req);
132 +
133 + ++ctx->stats.journalfile_deletions;
134 +
135 + return 0;
136 +}
137 +
138 +int create_journal_file(struct rrdengine_journalfile *journalfile, struct rrdengine_datafile *datafile)
139 +{
140 + struct rrdengine_instance *ctx = datafile->ctx;
141 + uv_fs_t req;
142 + uv_file file;
143 + int ret, fd;
144 + struct rrdeng_jf_sb *superblock;
145 + uv_buf_t iov;
146 + char path[1024];
147 +
148 + generate_journalfilepath(datafile, path, sizeof(path));
149 + fd = uv_fs_open(NULL, &req, path, O_DIRECT | O_CREAT | O_RDWR | O_TRUNC,
150 + S_IRUSR | S_IWUSR, NULL);
151 + if (fd < 0) {
152 + fatal("uv_fs_fsopen: %s", uv_strerror(fd));
153 + }
154 + assert(req.result >= 0);
155 + file = req.result;
156 + uv_fs_req_cleanup(&req);
157 +#ifdef __APPLE__
158 + info("Disabling OS X caching for file \"%s\".", path);
159 + fcntl(fd, F_NOCACHE, 1);
160 +#endif
161 +
162 + ret = posix_memalign((void *)&superblock, RRDFILE_ALIGNMENT, sizeof(*superblock));
163 + if (unlikely(ret)) {
164 + fatal("posix_memalign:%s", strerror(ret));
165 + }
166 + (void) strncpy(superblock->magic_number, RRDENG_JF_MAGIC, RRDENG_MAGIC_SZ);
167 + (void) strncpy(superblock->version, RRDENG_JF_VER, RRDENG_VER_SZ);
168 +
169 + iov = uv_buf_init((void *)superblock, sizeof(*superblock));
170 +
171 + ret = uv_fs_write(NULL, &req, file, &iov, 1, 0, NULL);
172 + if (ret < 0) {
173 + fatal("uv_fs_write: %s", uv_strerror(ret));
174 + }
175 + if (req.result < 0) {
176 + fatal("uv_fs_write: %s", uv_strerror((int)req.result));
177 + }
178 + uv_fs_req_cleanup(&req);
179 + free(superblock);
180 +
181 + journalfile->file = file;
182 + journalfile->pos = sizeof(*superblock);
183 + ctx->stats.io_write_bytes += sizeof(*superblock);
184 + ++ctx->stats.io_write_requests;
185 + ++ctx->stats.journalfile_creations;
186 +
187 + return 0;
188 +}
189 +
190 +static int check_journal_file_superblock(uv_file file)
191 +{
192 + int ret;
193 + struct rrdeng_jf_sb *superblock;
194 + uv_buf_t iov;
195 + uv_fs_t req;
196 +
197 + ret = posix_memalign((void *)&superblock, RRDFILE_ALIGNMENT, sizeof(*superblock));
198 + if (unlikely(ret)) {
199 + fatal("posix_memalign:%s", strerror(ret));
200 + }
201 + iov = uv_buf_init((void *)superblock, sizeof(*superblock));
202 +
203 + ret = uv_fs_read(NULL, &req, file, &iov, 1, 0, NULL);
204 + if (ret < 0) {
205 + error("uv_fs_read: %s", uv_strerror(ret));
206 + uv_fs_req_cleanup(&req);
207 + goto error;
208 + }
209 + assert(req.result >= 0);
210 + uv_fs_req_cleanup(&req);
211 +
212 + if (strncmp(superblock->magic_number, RRDENG_JF_MAGIC, RRDENG_MAGIC_SZ) ||
213 + strncmp(superblock->version, RRDENG_JF_VER, RRDENG_VER_SZ)) {
214 + error("File has invalid superblock.");
215 + ret = UV_EINVAL;
216 + } else {
217 + ret = 0;
218 + }
219 + error:
220 + free(superblock);
221 + return ret;
222 +}
223 +
224 +static void restore_extent_metadata(struct rrdengine_instance *ctx, struct rrdengine_journalfile *journalfile,
225 + void *buf, unsigned max_size)
226 +{
227 + struct page_cache *pg_cache = &ctx->pg_cache;
228 + unsigned i, count, payload_length, descr_size, valid_pages;
229 + struct rrdeng_page_cache_descr *descr;
230 + struct extent_info *extent;
231 + /* persistent structures */
232 + struct rrdeng_jf_store_data *jf_metric_data;
233 +
234 + jf_metric_data = buf;
235 + count = jf_metric_data->number_of_pages;
236 + descr_size = sizeof(*jf_metric_data->descr) * count;
237 + payload_length = sizeof(*jf_metric_data) + descr_size;
238 + if (payload_length > max_size) {
239 + error("Corrupted transaction payload.");
240 + return;
241 + }
242 +
243 + extent = mallocz(sizeof(*extent) + count * sizeof(extent->pages[0]));
244 + extent->offset = jf_metric_data->extent_offset;
245 + extent->size = jf_metric_data->extent_size;
246 + extent->number_of_pages = count;
247 + extent->datafile = journalfile->datafile;
248 + extent->next = NULL;
249 +
250 + for (i = 0, valid_pages = 0 ; i < count ; ++i) {
251 + uuid_t *temp_id;
252 + Pvoid_t *PValue;
253 + struct pg_cache_page_index *page_index;
254 +
255 + if (PAGE_METRICS != jf_metric_data->descr[i].type) {
256 + error("Unknown page type encountered.");
257 + continue;
258 + }
259 + ++valid_pages;
260 + temp_id = (uuid_t *)jf_metric_data->descr[i].uuid;
261 +
262 + uv_rwlock_rdlock(&pg_cache->metrics_index.lock);
263 + PValue = JudyHSGet(pg_cache->metrics_index.JudyHS_array, temp_id, sizeof(uuid_t));
264 + if (likely(NULL != PValue)) {
265 + page_index = *PValue;
266 + }
267 + uv_rwlock_rdunlock(&pg_cache->metrics_index.lock);
268 + if (NULL == PValue) {
269 + /* First time we see the UUID */
270 + uv_rwlock_wrlock(&pg_cache->metrics_index.lock);
271 + PValue = JudyHSIns(&pg_cache->metrics_index.JudyHS_array, temp_id, sizeof(uuid_t), PJE0);
272 + assert(NULL == *PValue); /* TODO: figure out concurrency model */
273 + *PValue = page_index = create_page_index(temp_id);
274 + uv_rwlock_wrunlock(&pg_cache->metrics_index.lock);
275 + }
276 +
277 + descr = pg_cache_create_descr();
278 + descr->page_length = jf_metric_data->descr[i].page_length;
279 + descr->start_time = jf_metric_data->descr[i].start_time;
280 + descr->end_time = jf_metric_data->descr[i].end_time;
281 + descr->id = &page_index->id;
282 + descr->extent = extent;
283 + extent->pages[i] = descr;
284 + pg_cache_insert(ctx, page_index, descr);
285 + }
286 + if (likely(valid_pages))
287 + df_extent_insert(extent);
288 +}
289 +
290 +/*
291 + * Replays transaction by interpreting up to max_size bytes from buf.
292 + * Sets id to the current transaction id or to 0 if unknown.
293 + * Returns size of transaction record or 0 for unknown size.
294 + */
295 +static unsigned replay_transaction(struct rrdengine_instance *ctx, struct rrdengine_journalfile *journalfile,
296 + void *buf, uint64_t *id, unsigned max_size)
297 +{
298 + unsigned payload_length, size_bytes;
299 + int ret;
300 + /* persistent structures */
301 + struct rrdeng_jf_transaction_header *jf_header;
302 + struct rrdeng_jf_transaction_trailer *jf_trailer;
303 + uLong crc;
304 +
305 + *id = 0;
306 + jf_header = buf;
307 + if (STORE_PADDING == jf_header->type) {
308 + debug(D_RRDENGINE, "Skipping padding.");
309 + return 0;
310 + }
311 + if (sizeof(*jf_header) > max_size) {
312 + error("Corrupted transaction record, skipping.");
313 + return 0;
314 + }
315 + *id = jf_header->id;
316 + payload_length = jf_header->payload_length;
317 + size_bytes = sizeof(*jf_header) + payload_length + sizeof(*jf_trailer);
318 + if (size_bytes > max_size) {
319 + error("Corrupted transaction record, skipping.");
320 + return 0;
321 + }
322 + jf_trailer = buf + sizeof(*jf_header) + payload_length;
323 + crc = crc32(0L, Z_NULL, 0);
324 + crc = crc32(crc, buf, sizeof(*jf_header) + payload_length);
325 + ret = crc32cmp(jf_trailer->checksum, crc);
326 + debug(D_RRDENGINE, "Transaction %"PRIu64" was read from disk. CRC32 check: %s", *id, ret ? "FAILED" : "SUCCEEDED");
327 + if (unlikely(ret)) {
328 + return size_bytes;
329 + }
330 + switch (jf_header->type) {
331 + case STORE_DATA:
332 + debug(D_RRDENGINE, "Replaying transaction %"PRIu64"", jf_header->id);
333 + restore_extent_metadata(ctx, journalfile, buf + sizeof(*jf_header), payload_length);
334 + break;
335 + default:
336 + error("Unknown transaction type. Skipping record.");
337 + break;
338 + }
339 +
340 + return size_bytes;
341 +}
342 +
343 +
344 +#define READAHEAD_BYTES (RRDENG_BLOCK_SIZE * 256)
345 +/*
346 + * Iterates journal file transactions and populates the page cache.
347 + * Page cache must already be initialized.
348 + * Returns the maximum transaction id it discovered.
349 + */
350 +static uint64_t iterate_transactions(struct rrdengine_instance *ctx, struct rrdengine_journalfile *journalfile)
351 +{
352 + uv_file file;
353 + uint64_t file_size;//, data_file_size;
354 + int ret;
355 + uint64_t pos, pos_i, max_id, id;
356 + unsigned size_bytes;
357 + void *buf;
358 + uv_buf_t iov;
359 + uv_fs_t req;
360 +
361 + file = journalfile->file;
362 + file_size = journalfile->pos;
363 + //data_file_size = journalfile->datafile->pos; TODO: utilize this?
364 +
365 + max_id = 1;
366 + ret = posix_memalign((void *)&buf, RRDFILE_ALIGNMENT, READAHEAD_BYTES);
367 + if (unlikely(ret)) {
368 + fatal("posix_memalign:%s", strerror(ret));
369 + }
370 +
371 + for (pos = sizeof(struct rrdeng_jf_sb) ; pos < file_size ; pos += READAHEAD_BYTES) {
372 + size_bytes = MIN(READAHEAD_BYTES, file_size - pos);
373 + iov = uv_buf_init(buf, size_bytes);
374 + ret = uv_fs_read(NULL, &req, file, &iov, 1, pos, NULL);
375 + if (ret < 0) {
376 + fatal("uv_fs_read: %s", uv_strerror(ret));
377 + /*uv_fs_req_cleanup(&req);*/
378 + }
379 + assert(req.result >= 0);
380 + uv_fs_req_cleanup(&req);
381 + ctx->stats.io_read_bytes += size_bytes;
382 + ++ctx->stats.io_read_requests;
383 +
384 + //pos_i = pos;
385 + //while (pos_i < pos + size_bytes) {
386 + for (pos_i = 0 ; pos_i < size_bytes ; ) {
387 + unsigned max_size;
388 +
389 + max_size = pos + size_bytes - pos_i;
390 + ret = replay_transaction(ctx, journalfile, buf + pos_i, &id, max_size);
391 + if (!ret) /* TODO: support transactions bigger than 4K */
392 + /* unknown transaction size, move on to the next block */
393 + pos_i = ALIGN_BYTES_FLOOR(pos_i + RRDENG_BLOCK_SIZE);
394 + else
395 + pos_i += ret;
396 + max_id = MAX(max_id, id);
397 + }
398 + }
399 +
400 + free(buf);
401 + return max_id;
402 +}
403 +
404 +int load_journal_file(struct rrdengine_instance *ctx, struct rrdengine_journalfile *journalfile,
405 + struct rrdengine_datafile *datafile)
406 +{
407 + uv_fs_t req;
408 + uv_file file;
409 + int ret, fd;
410 + uint64_t file_size, max_id;
411 + char path[1024];
412 +
413 + generate_journalfilepath(datafile, path, sizeof(path));
414 + fd = uv_fs_open(NULL, &req, path, O_DIRECT | O_RDWR, S_IRUSR | S_IWUSR, NULL);
415 + if (fd < 0) {
416 + /* if (UV_ENOENT != fd) */
417 + error("uv_fs_fsopen: %s", uv_strerror(fd));
418 + uv_fs_req_cleanup(&req);
419 + return fd;
420 + }
421 + assert(req.result >= 0);
422 + file = req.result;
423 + uv_fs_req_cleanup(&req);
424 +#ifdef __APPLE__
425 + info("Disabling OS X caching for file \"%s\".", path);
426 + fcntl(fd, F_NOCACHE, 1);
427 +#endif
428 + info("Loading journal file \"%s\".", path);
429 +
430 + ret = check_file_properties(file, &file_size, sizeof(struct rrdeng_df_sb));
431 + if (ret)
432 + goto error;
433 + file_size = ALIGN_BYTES_FLOOR(file_size);
434 +
435 + ret = check_journal_file_superblock(file);
436 + if (ret)
437 + goto error;
438 + ctx->stats.io_read_bytes += sizeof(struct rrdeng_jf_sb);
439 + ++ctx->stats.io_read_requests;
440 +
441 + journalfile->file = file;
442 + journalfile->pos = file_size;
443 +
444 + max_id = iterate_transactions(ctx, journalfile);
445 +
446 + ctx->commit_log.transaction_id = MAX(ctx->commit_log.transaction_id, max_id + 1);
447 +
448 + info("Journal file \"%s\" loaded (size:%"PRIu64").", path, file_size);
449 + return 0;
450 +
451 + error:
452 + (void) uv_fs_close(NULL, &req, file, NULL);
453 + uv_fs_req_cleanup(&req);
454 + return ret;
455 +}
456 +
457 +void init_commit_log(struct rrdengine_instance *ctx)
458 +{
459 + ctx->commit_log.buf = NULL;
460 + ctx->commit_log.buf_pos = 0;
461 + ctx->commit_log.transaction_id = 1;
462 +}
\ No newline at end of file
database/engine/journalfile.h new
+46
@@ -0,0 +1,46 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_JOURNALFILE_H
4 +#define NETDATA_JOURNALFILE_H
5 +
6 +#include "rrdengine.h"
7 +
8 +/* Forward declarations */
9 +struct rrdengine_instance;
10 +struct rrdengine_worker_config;
11 +struct rrdengine_datafile;
12 +struct rrdengine_journalfile;
13 +
14 +#define WALFILE_PREFIX "journalfile-"
15 +#define WALFILE_EXTENSION ".njf"
16 +
17 +
18 +/* only one event loop is supported for now */
19 +struct rrdengine_journalfile {
20 + uv_file file;
21 + uint64_t pos;
22 +
23 + struct rrdengine_datafile *datafile;
24 +};
25 +
26 +/* only one event loop is supported for now */
27 +struct transaction_commit_log {
28 + uint64_t transaction_id;
29 +
30 + /* outstanding transaction buffer */
31 + void *buf;
32 + unsigned buf_pos;
33 + unsigned buf_size;
34 +};
35 +
36 +extern void journalfile_init(struct rrdengine_journalfile *journalfile, struct rrdengine_datafile *datafile);
37 +extern void *wal_get_transaction_buffer(struct rrdengine_worker_config* wc, unsigned size);
38 +extern void wal_flush_transaction_buffer(struct rrdengine_worker_config* wc);
39 +extern int destroy_journal_file(struct rrdengine_journalfile *journalfile, struct rrdengine_datafile *datafile);
40 +extern int create_journal_file(struct rrdengine_journalfile *journalfile, struct rrdengine_datafile *datafile);
41 +extern int load_journal_file(struct rrdengine_instance *ctx, struct rrdengine_journalfile *journalfile,
42 + struct rrdengine_datafile *datafile);
43 +extern void init_commit_log(struct rrdengine_instance *ctx);
44 +
45 +
46 +#endif /* NETDATA_JOURNALFILE_H */
\ No newline at end of file
database/engine/pagecache.c new
+785
@@ -0,0 +1,785 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +#define NETDATA_RRD_INTERNALS
3 +
4 +#include "rrdengine.h"
5 +
6 +/* Forward declerations */
7 +static int pg_cache_try_evict_one_page_unsafe(struct rrdengine_instance *ctx);
8 +
9 +/* always inserts into tail */
10 +static inline void pg_cache_replaceQ_insert_unsafe(struct rrdengine_instance *ctx,
11 + struct rrdeng_page_cache_descr *descr)
12 +{
13 + struct page_cache *pg_cache = &ctx->pg_cache;
14 +
15 + if (likely(NULL != pg_cache->replaceQ.tail)) {
16 + descr->prev = pg_cache->replaceQ.tail;
17 + pg_cache->replaceQ.tail->next = descr;
18 + }
19 + if (unlikely(NULL == pg_cache->replaceQ.head)) {
20 + pg_cache->replaceQ.head = descr;
21 + }
22 + pg_cache->replaceQ.tail = descr;
23 +}
24 +
25 +static inline void pg_cache_replaceQ_delete_unsafe(struct rrdengine_instance *ctx,
26 + struct rrdeng_page_cache_descr *descr)
27 +{
28 + struct page_cache *pg_cache = &ctx->pg_cache;
29 + struct rrdeng_page_cache_descr *prev, *next;
30 +
31 + prev = descr->prev;
32 + next = descr->next;
33 +
34 + if (likely(NULL != prev)) {
35 + prev->next = next;
36 + }
37 + if (likely(NULL != next)) {
38 + next->prev = prev;
39 + }
40 + if (unlikely(descr == pg_cache->replaceQ.head)) {
41 + pg_cache->replaceQ.head = next;
42 + }
43 + if (unlikely(descr == pg_cache->replaceQ.tail)) {
44 + pg_cache->replaceQ.tail = prev;
45 + }
46 + descr->prev = descr->next = NULL;
47 +}
48 +
49 +void pg_cache_replaceQ_insert(struct rrdengine_instance *ctx,
50 + struct rrdeng_page_cache_descr *descr)
51 +{
52 + struct page_cache *pg_cache = &ctx->pg_cache;
53 +
54 + uv_rwlock_wrlock(&pg_cache->replaceQ.lock);
55 + pg_cache_replaceQ_insert_unsafe(ctx, descr);
56 + uv_rwlock_wrunlock(&pg_cache->replaceQ.lock);
57 +}
58 +
59 +void pg_cache_replaceQ_delete(struct rrdengine_instance *ctx,
60 + struct rrdeng_page_cache_descr *descr)
61 +{
62 + struct page_cache *pg_cache = &ctx->pg_cache;
63 +
64 + uv_rwlock_wrlock(&pg_cache->replaceQ.lock);
65 + pg_cache_replaceQ_delete_unsafe(ctx, descr);
66 + uv_rwlock_wrunlock(&pg_cache->replaceQ.lock);
67 +}
68 +void pg_cache_replaceQ_set_hot(struct rrdengine_instance *ctx,
69 + struct rrdeng_page_cache_descr *descr)
70 +{
71 + struct page_cache *pg_cache = &ctx->pg_cache;
72 +
73 + uv_rwlock_wrlock(&pg_cache->replaceQ.lock);
74 + pg_cache_replaceQ_delete_unsafe(ctx, descr);
75 + pg_cache_replaceQ_insert_unsafe(ctx, descr);
76 + uv_rwlock_wrunlock(&pg_cache->replaceQ.lock);
77 +}
78 +
79 +struct rrdeng_page_cache_descr *pg_cache_create_descr(void)
80 +{
81 + struct rrdeng_page_cache_descr *descr;
82 +
83 + descr = mallocz(sizeof(*descr));
84 + descr->page = NULL;
85 + descr->page_length = 0;
86 + descr->start_time = INVALID_TIME;
87 + descr->end_time = INVALID_TIME;
88 + descr->id = NULL;
89 + descr->extent = NULL;
90 + descr->flags = 0;
91 + descr->prev = descr->next = descr->private = NULL;
92 + descr->refcnt = 0;
93 + descr->waiters = 0;
94 + descr->handle = NULL;
95 + assert(0 == uv_cond_init(&descr->cond));
96 + assert(0 == uv_mutex_init(&descr->mutex));
97 +
98 + return descr;
99 +}
100 +
101 +void pg_cache_destroy_descr(struct rrdeng_page_cache_descr *descr)
102 +{
103 + uv_cond_destroy(&descr->cond);
104 + uv_mutex_destroy(&descr->mutex);
105 + free(descr);
106 +}
107 +
108 +/* The caller must hold page descriptor lock. */
109 +void pg_cache_wake_up_waiters_unsafe(struct rrdeng_page_cache_descr *descr)
110 +{
111 + if (descr->waiters)
112 + uv_cond_broadcast(&descr->cond);
113 +}
114 +
115 +/*
116 + * The caller must hold page descriptor lock.
117 + * The lock will be released and re-acquired. The descriptor is not guaranteed
118 + * to exist after this function returns.
119 + */
120 +void pg_cache_wait_event_unsafe(struct rrdeng_page_cache_descr *descr)
121 +{
122 + ++descr->waiters;
123 + uv_cond_wait(&descr->cond, &descr->mutex);
124 + --descr->waiters;
125 +}
126 +
127 +/*
128 + * Returns page flags.
129 + * The lock will be released and re-acquired. The descriptor is not guaranteed
130 + * to exist after this function returns.
131 + */
132 +unsigned long pg_cache_wait_event(struct rrdeng_page_cache_descr *descr)
133 +{
134 + unsigned long flags;
135 +
136 + uv_mutex_lock(&descr->mutex);
137 + pg_cache_wait_event_unsafe(descr);
138 + flags = descr->flags;
139 + uv_mutex_unlock(&descr->mutex);
140 +
141 + return flags;
142 +}
143 +
144 +/*
145 + * The caller must hold page descriptor lock.
146 + * Gets a reference to the page descriptor.
147 + * Returns 1 on success and 0 on failure.
148 + */
149 +int pg_cache_try_get_unsafe(struct rrdeng_page_cache_descr *descr, int exclusive_access)
150 +{
151 + if ((descr->flags & (RRD_PAGE_LOCKED | RRD_PAGE_READ_PENDING)) ||
152 + (exclusive_access && descr->refcnt)) {
153 + return 0;
154 + }
155 + if (exclusive_access)
156 + descr->flags |= RRD_PAGE_LOCKED;
157 + ++descr->refcnt;
158 +
159 + return 1;
160 +}
161 +
162 +/*
163 + * The caller must hold page descriptor lock.
164 + * Same return values as pg_cache_try_get_unsafe() without doing anything.
165 + */
166 +int pg_cache_can_get_unsafe(struct rrdeng_page_cache_descr *descr, int exclusive_access)
167 +{
168 + if ((descr->flags & (RRD_PAGE_LOCKED | RRD_PAGE_READ_PENDING)) ||
169 + (exclusive_access && descr->refcnt)) {
170 + return 0;
171 + }
172 +
173 + return 1;
174 +}
175 +
176 +/*
177 + * The caller must hold the page descriptor lock.
178 + * This function may block doing cleanup.
179 + */
180 +void pg_cache_put_unsafe(struct rrdeng_page_cache_descr *descr)
181 +{
182 + descr->flags &= ~RRD_PAGE_LOCKED;
183 + if (0 == --descr->refcnt) {
184 + pg_cache_wake_up_waiters_unsafe(descr);
185 + }
186 + /* TODO: perform cleanup */
187 +}
188 +
189 +/*
190 + * This function may block doing cleanup.
191 + */
192 +void pg_cache_put(struct rrdeng_page_cache_descr *descr)
193 +{
194 + uv_mutex_lock(&descr->mutex);
195 + pg_cache_put_unsafe(descr);
196 + uv_mutex_unlock(&descr->mutex);
197 +}
198 +
199 +/* The caller must hold the page cache lock */
200 +static void pg_cache_release_pages_unsafe(struct rrdengine_instance *ctx, unsigned number)
201 +{
202 + struct page_cache *pg_cache = &ctx->pg_cache;
203 +
204 + pg_cache->populated_pages -= number;
205 +}
206 +
207 +static void pg_cache_release_pages(struct rrdengine_instance *ctx, unsigned number)
208 +{
209 + struct page_cache *pg_cache = &ctx->pg_cache;
210 +
211 + uv_rwlock_wrlock(&pg_cache->pg_cache_rwlock);
212 + pg_cache_release_pages_unsafe(ctx, number);
213 + uv_rwlock_wrunlock(&pg_cache->pg_cache_rwlock);
214 +}
215 +/*
216 + * This function will block until it reserves #number populated pages.
217 + * It will trigger evictions or dirty page flushing if the ctx->max_cache_pages limit is hit.
218 + */
219 +static void pg_cache_reserve_pages(struct rrdengine_instance *ctx, unsigned number)
220 +{
221 + struct page_cache *pg_cache = &ctx->pg_cache;
222 +
223 + assert(number < ctx->max_cache_pages);
224 +
225 + uv_rwlock_wrlock(&pg_cache->pg_cache_rwlock);
226 + if (pg_cache->populated_pages + number >= ctx->max_cache_pages + 1)
227 + debug(D_RRDENGINE, "=================================\nPage cache full. Reserving %u pages.\n=================================",
228 + number);
229 + while (pg_cache->populated_pages + number >= ctx->max_cache_pages + 1) {
230 + if (!pg_cache_try_evict_one_page_unsafe(ctx)) {
231 + /* failed to evict */
232 + struct completion compl;
233 + struct rrdeng_cmd cmd;
234 +
235 + uv_rwlock_wrunlock(&pg_cache->pg_cache_rwlock);
236 +
237 + init_completion(&compl);
238 + cmd.opcode = RRDENG_FLUSH_PAGES;
239 + cmd.completion = &compl;
240 + rrdeng_enq_cmd(&ctx->worker_config, &cmd);
241 + /* wait for some pages to be flushed */
242 + debug(D_RRDENGINE, "%s: waiting for pages to be written to disk before evicting.", __func__);
243 + wait_for_completion(&compl);
244 + destroy_completion(&compl);
245 +
246 + uv_rwlock_wrlock(&pg_cache->pg_cache_rwlock);
247 + }
248 + }
249 + pg_cache->populated_pages += number;
250 + uv_rwlock_wrunlock(&pg_cache->pg_cache_rwlock);
251 +}
252 +
253 +/*
254 + * This function will attempt to reserve #number populated pages.
255 + * It may trigger evictions if the ctx->cache_pages_low_watermark limit is hit.
256 + * Returns 0 on failure and 1 on success.
257 + */
258 +static int pg_cache_try_reserve_pages(struct rrdengine_instance *ctx, unsigned number)
259 +{
260 + struct page_cache *pg_cache = &ctx->pg_cache;
261 + unsigned count = 0;
262 + int ret = 0;
263 +
264 + assert(number < ctx->max_cache_pages);
265 +
266 + uv_rwlock_wrlock(&pg_cache->pg_cache_rwlock);
267 + if (pg_cache->populated_pages + number >= ctx->cache_pages_low_watermark + 1) {
268 + debug(D_RRDENGINE,
269 + "=================================\nPage cache full. Trying to reserve %u pages.\n=================================",
270 + number);
271 + do {
272 + if (!pg_cache_try_evict_one_page_unsafe(ctx))
273 + break;
274 + ++count;
275 + } while (pg_cache->populated_pages + number >= ctx->cache_pages_low_watermark + 1);
276 + debug(D_RRDENGINE, "Evicted %u pages.", count);
277 + }
278 +
279 + if (pg_cache->populated_pages + number < ctx->max_cache_pages + 1) {
280 + pg_cache->populated_pages += number;
281 + ret = 1; /* success */
282 + }
283 + uv_rwlock_wrunlock(&pg_cache->pg_cache_rwlock);
284 +
285 + return ret;
286 +}
287 +
288 +/* The caller must hold the page cache and the page descriptor locks in that order */
289 +static void pg_cache_evict_unsafe(struct rrdengine_instance *ctx, struct rrdeng_page_cache_descr *descr)
290 +{
291 + free(descr->page);
292 + descr->page = NULL;
293 + descr->flags &= ~RRD_PAGE_POPULATED;
294 + pg_cache_release_pages_unsafe(ctx, 1);
295 + ++ctx->stats.pg_cache_evictions;
296 +}
297 +
298 +/*
299 + * The caller must hold the page cache lock.
300 + * Lock order: page cache -> replaceQ -> descriptor
301 + * This function iterates all pages and tries to evict one.
302 + * If it fails it sets in_flight_descr to the oldest descriptor that has write-back in progress,
303 + * or it sets it to NULL if no write-back is in progress.
304 + *
305 + * Returns 1 on success and 0 on failure.
306 + */
307 +static int pg_cache_try_evict_one_page_unsafe(struct rrdengine_instance *ctx)
308 +{
309 + struct page_cache *pg_cache = &ctx->pg_cache;
310 + unsigned long old_flags;
311 + struct rrdeng_page_cache_descr *descr;
312 +
313 + uv_rwlock_wrlock(&pg_cache->replaceQ.lock);
314 + for (descr = pg_cache->replaceQ.head ; NULL != descr ; descr = descr->next) {
315 + uv_mutex_lock(&descr->mutex);
316 + old_flags = descr->flags;
317 + if ((old_flags & RRD_PAGE_POPULATED) && !(old_flags & RRD_PAGE_DIRTY) && pg_cache_try_get_unsafe(descr, 1)) {
318 + /* must evict */
319 + pg_cache_evict_unsafe(ctx, descr);
320 + pg_cache_put_unsafe(descr);
321 + uv_mutex_unlock(&descr->mutex);
322 + pg_cache_replaceQ_delete_unsafe(ctx, descr);
323 + uv_rwlock_wrunlock(&pg_cache->replaceQ.lock);
324 +
325 + return 1;
326 + }
327 + uv_mutex_unlock(&descr->mutex);
328 + };
329 + uv_rwlock_wrunlock(&pg_cache->replaceQ.lock);
330 +
331 + /* failed to evict */
332 + return 0;
333 +}
334 +
335 +/*
336 + * TODO: last waiter frees descriptor ?
337 + */
338 +void pg_cache_punch_hole(struct rrdengine_instance *ctx, struct rrdeng_page_cache_descr *descr)
339 +{
340 + struct page_cache *pg_cache = &ctx->pg_cache;
341 + Pvoid_t *PValue;
342 + struct pg_cache_page_index *page_index;
343 + int ret;
344 +
345 + uv_rwlock_rdlock(&pg_cache->metrics_index.lock);
346 + PValue = JudyHSGet(pg_cache->metrics_index.JudyHS_array, descr->id, sizeof(uuid_t));
347 + assert(NULL != PValue);
348 + page_index = *PValue;
349 + uv_rwlock_rdunlock(&pg_cache->metrics_index.lock);
350 +
351 + uv_rwlock_wrlock(&page_index->lock);
352 + ret = JudyLDel(&page_index->JudyL_array, (Word_t)(descr->start_time / USEC_PER_SEC), PJE0);
353 + assert(1 == ret);
354 + uv_rwlock_wrunlock(&page_index->lock);
355 +
356 + uv_rwlock_wrlock(&pg_cache->pg_cache_rwlock);
357 + ++ctx->stats.pg_cache_deletions;
358 + --pg_cache->page_descriptors;
359 + uv_rwlock_wrunlock(&pg_cache->pg_cache_rwlock);
360 +
361 + uv_mutex_lock(&descr->mutex);
362 + while (!pg_cache_try_get_unsafe(descr, 1)) {
363 + debug(D_RRDENGINE, "%s: Waiting for locked page:", __func__);
364 + if(unlikely(debug_flags & D_RRDENGINE))
365 + print_page_cache_descr(descr);
366 + pg_cache_wait_event_unsafe(descr);
367 + }
368 + /* even a locked page could be dirty */
369 + while (unlikely(descr->flags & RRD_PAGE_DIRTY)) {
370 + debug(D_RRDENGINE, "%s: Found dirty page, waiting for it to be flushed:", __func__);
371 + if(unlikely(debug_flags & D_RRDENGINE))
372 + print_page_cache_descr(descr);
373 + pg_cache_wait_event_unsafe(descr);
374 + }
375 + uv_mutex_unlock(&descr->mutex);
376 +
377 + if (descr->flags & RRD_PAGE_POPULATED) {
378 + /* only after locking can it be safely deleted from LRU */
379 + pg_cache_replaceQ_delete(ctx, descr);
380 +
381 + uv_rwlock_wrlock(&pg_cache->pg_cache_rwlock);
382 + pg_cache_evict_unsafe(ctx, descr);
383 + uv_rwlock_wrunlock(&pg_cache->pg_cache_rwlock);
384 + }
385 + pg_cache_put(descr);
386 + pg_cache_destroy_descr(descr);
387 + pg_cache_update_metric_times(page_index);
388 +}
389 +
390 +static inline int is_page_in_time_range(struct rrdeng_page_cache_descr *descr, usec_t start_time, usec_t end_time)
391 +{
392 + usec_t pg_start, pg_end;
393 +
394 + pg_start = descr->start_time;
395 + pg_end = descr->end_time;
396 +
397 + return (pg_start < start_time && pg_end >= start_time) ||
398 + (pg_start >= start_time && pg_start <= end_time);
399 +}
400 +
401 +static inline int is_point_in_time_in_page(struct rrdeng_page_cache_descr *descr, usec_t point_in_time)
402 +{
403 + return (point_in_time >= descr->start_time && point_in_time <= descr->end_time);
404 +}
405 +
406 +/* Update metric oldest and latest timestamps efficiently when adding new values */
407 +void pg_cache_add_new_metric_time(struct pg_cache_page_index *page_index, struct rrdeng_page_cache_descr *descr)
408 +{
409 + usec_t oldest_time = page_index->oldest_time;
410 + usec_t latest_time = page_index->latest_time;
411 +
412 + if (unlikely(oldest_time == INVALID_TIME || descr->start_time < oldest_time)) {
413 + page_index->oldest_time = descr->start_time;
414 + }
415 + if (likely(descr->end_time > latest_time || latest_time == INVALID_TIME)) {
416 + page_index->latest_time = descr->end_time;
417 + }
418 +}
419 +
420 +/* Update metric oldest and latest timestamps when removing old values */
421 +void pg_cache_update_metric_times(struct pg_cache_page_index *page_index)
422 +{
423 + Pvoid_t *firstPValue, *lastPValue;
424 + Word_t firstIndex, lastIndex;
425 + struct rrdeng_page_cache_descr *descr;
426 + usec_t oldest_time = INVALID_TIME;
427 + usec_t latest_time = INVALID_TIME;
428 +
429 + uv_rwlock_rdlock(&page_index->lock);
430 + /* Find first page in range */
431 + firstIndex = (Word_t)0;
432 + firstPValue = JudyLFirst(page_index->JudyL_array, &firstIndex, PJE0);
433 + if (likely(NULL != firstPValue)) {
434 + descr = *firstPValue;
435 + oldest_time = descr->start_time;
436 + }
437 + lastIndex = (Word_t)-1;
438 + lastPValue = JudyLLast(page_index->JudyL_array, &lastIndex, PJE0);
439 + if (likely(NULL != lastPValue)) {
440 + descr = *lastPValue;
441 + latest_time = descr->end_time;
442 + }
443 + uv_rwlock_rdunlock(&page_index->lock);
444 +
445 + if (unlikely(NULL == firstPValue)) {
446 + assert(NULL == lastPValue);
447 + page_index->oldest_time = page_index->latest_time = INVALID_TIME;
448 + return;
449 + }
450 + page_index->oldest_time = oldest_time;
451 + page_index->latest_time = latest_time;
452 +}
453 +
454 +/* If index is NULL lookup by UUID (descr->id) */
455 +void pg_cache_insert(struct rrdengine_instance *ctx, struct pg_cache_page_index *index,
456 + struct rrdeng_page_cache_descr *descr)
457 +{
458 + struct page_cache *pg_cache = &ctx->pg_cache;
459 + Pvoid_t *PValue;
460 + struct pg_cache_page_index *page_index;
461 +
462 + if (descr->flags & RRD_PAGE_POPULATED) {
463 + pg_cache_reserve_pages(ctx, 1);
464 + if (!(descr->flags & RRD_PAGE_DIRTY))
465 + pg_cache_replaceQ_insert(ctx, descr);
466 + }
467 +
468 + if (unlikely(NULL == index)) {
469 + uv_rwlock_rdlock(&pg_cache->metrics_index.lock);
470 + PValue = JudyHSGet(pg_cache->metrics_index.JudyHS_array, descr->id, sizeof(uuid_t));
471 + assert(NULL != PValue);
472 + page_index = *PValue;
473 + uv_rwlock_rdunlock(&pg_cache->metrics_index.lock);
474 + } else {
475 + page_index = index;
476 + }
477 +
478 + uv_rwlock_wrlock(&page_index->lock);
479 + PValue = JudyLIns(&page_index->JudyL_array, (Word_t)(descr->start_time / USEC_PER_SEC), PJE0);
480 + *PValue = descr;
481 + pg_cache_add_new_metric_time(page_index, descr);
482 + uv_rwlock_wrunlock(&page_index->lock);
483 +
484 + uv_rwlock_wrlock(&pg_cache->pg_cache_rwlock);
485 + ++ctx->stats.pg_cache_insertions;
486 + ++pg_cache->page_descriptors;
487 + uv_rwlock_wrunlock(&pg_cache->pg_cache_rwlock);
488 +}
489 +
490 +/*
491 + * Searches for a page and triggers disk I/O if necessary and possible.
492 + * Does not get a reference.
493 + * Returns page index pointer for given metric UUID.
494 + */
495 +struct pg_cache_page_index *
496 + pg_cache_preload(struct rrdengine_instance *ctx, uuid_t *id, usec_t start_time, usec_t end_time)
497 +{
498 + struct page_cache *pg_cache = &ctx->pg_cache;
499 + struct rrdeng_page_cache_descr *descr = NULL, *preload_array[PAGE_CACHE_MAX_PRELOAD_PAGES];
500 + int i, j, k, count, found;
501 + unsigned long flags;
502 + Pvoid_t *PValue;
503 + struct pg_cache_page_index *page_index;
504 + Word_t Index;
505 + uint8_t failed_to_reserve;
506 +
507 + uv_rwlock_rdlock(&pg_cache->metrics_index.lock);
508 + PValue = JudyHSGet(pg_cache->metrics_index.JudyHS_array, id, sizeof(uuid_t));
509 + if (likely(NULL != PValue)) {
510 + page_index = *PValue;
511 + }
512 + uv_rwlock_rdunlock(&pg_cache->metrics_index.lock);
513 + if (NULL == PValue) {
514 + debug(D_RRDENGINE, "%s: No page was found to attempt preload.", __func__);
515 + return NULL;
516 + }
517 +
518 + uv_rwlock_rdlock(&page_index->lock);
519 + /* Find first page in range */
520 + found = 0;
521 + Index = (Word_t)(start_time / USEC_PER_SEC);
522 + PValue = JudyLLast(page_index->JudyL_array, &Index, PJE0);
523 + if (likely(NULL != PValue)) {
524 + descr = *PValue;
525 + if (is_page_in_time_range(descr, start_time, end_time)) {
526 + found = 1;
527 + }
528 + }
529 + if (!found) {
530 + Index = (Word_t)(start_time / USEC_PER_SEC);
531 + PValue = JudyLFirst(page_index->JudyL_array, &Index, PJE0);
532 + if (likely(NULL != PValue)) {
533 + descr = *PValue;
534 + if (is_page_in_time_range(descr, start_time, end_time)) {
535 + found = 1;
536 + }
537 + }
538 + }
539 + if (!found) {
540 + uv_rwlock_rdunlock(&page_index->lock);
541 + debug(D_RRDENGINE, "%s: No page was found to attempt preload.", __func__);
542 + return page_index;
543 + }
544 +
545 + for (count = 0 ;
546 + descr != NULL && is_page_in_time_range(descr, start_time, end_time);
547 + PValue = JudyLNext(page_index->JudyL_array, &Index, PJE0),
548 + descr = unlikely(NULL == PValue) ? NULL : *PValue) {
549 + /* Iterate all pages in range */
550 +
551 + if (unlikely(0 == descr->page_length))
552 + continue;
553 + uv_mutex_lock(&descr->mutex);
554 + flags = descr->flags;
555 + if (pg_cache_can_get_unsafe(descr, 0)) {
556 + if (flags & RRD_PAGE_POPULATED) {
557 + /* success */
558 + uv_mutex_unlock(&descr->mutex);
559 + debug(D_RRDENGINE, "%s: Page was found in memory.", __func__);
560 + continue;
561 + }
562 + }
563 + if (!(flags & RRD_PAGE_POPULATED) && pg_cache_try_get_unsafe(descr, 1)) {
564 + preload_array[count++] = descr;
565 + if (PAGE_CACHE_MAX_PRELOAD_PAGES == count) {
566 + uv_mutex_unlock(&descr->mutex);
567 + break;
568 + }
569 + }
570 + uv_mutex_unlock(&descr->mutex);
571 +
572 + };
573 + uv_rwlock_rdunlock(&page_index->lock);
574 +
575 + failed_to_reserve = 0;
576 + for (i = 0 ; i < count && !failed_to_reserve ; ++i) {
577 + struct rrdeng_cmd cmd;
578 + struct rrdeng_page_cache_descr *next;
579 +
580 + descr = preload_array[i];
581 + if (NULL == descr) {
582 + continue;
583 + }
584 + if (!pg_cache_try_reserve_pages(ctx, 1)) {
585 + failed_to_reserve = 1;
586 + break;
587 + }
588 + cmd.opcode = RRDENG_READ_EXTENT;
589 + cmd.read_extent.page_cache_descr[0] = descr;
590 + /* don't use this page again */
591 + preload_array[i] = NULL;
592 + for (j = 0, k = 1 ; j < count ; ++j) {
593 + next = preload_array[j];
594 + if (NULL == next) {
595 + continue;
596 + }
597 + if (descr->extent == next->extent) {
598 + /* same extent, consolidate */
599 + if (!pg_cache_try_reserve_pages(ctx, 1)) {
600 + failed_to_reserve = 1;
601 + break;
602 + }
603 + cmd.read_extent.page_cache_descr[k++] = next;
604 + /* don't use this page again */
605 + preload_array[j] = NULL;
606 + }
607 + }
608 + cmd.read_extent.page_count = k;
609 + rrdeng_enq_cmd(&ctx->worker_config, &cmd);
610 + }
611 + if (failed_to_reserve) {
612 + debug(D_RRDENGINE, "%s: Failed to reserve enough memory, canceling I/O.", __func__);
613 + for (i = 0 ; i < count ; ++i) {
614 + descr = preload_array[i];
615 + if (NULL == descr) {
616 + continue;
617 + }
618 + pg_cache_put(descr);
619 + }
620 + }
621 + if (!count) {
622 + /* no such page */
623 + debug(D_RRDENGINE, "%s: No page was eligible to attempt preload.", __func__);
624 + }
625 + return page_index;
626 +}
627 +
628 +/*
629 + * Searches for a page and gets a reference.
630 + * When point_in_time is INVALID_TIME get any page.
631 + * If index is NULL lookup by UUID (id).
632 + */
633 +struct rrdeng_page_cache_descr *
634 + pg_cache_lookup(struct rrdengine_instance *ctx, struct pg_cache_page_index *index, uuid_t *id,
635 + usec_t point_in_time)
636 +{
637 + struct page_cache *pg_cache = &ctx->pg_cache;
638 + struct rrdeng_page_cache_descr *descr = NULL;
639 + unsigned long flags;
640 + Pvoid_t *PValue;
641 + struct pg_cache_page_index *page_index;
642 + Word_t Index;
643 + uint8_t page_not_in_cache;
644 +
645 + if (unlikely(NULL == index)) {
646 + uv_rwlock_rdlock(&pg_cache->metrics_index.lock);
647 + PValue = JudyHSGet(pg_cache->metrics_index.JudyHS_array, id, sizeof(uuid_t));
648 + if (likely(NULL != PValue)) {
649 + page_index = *PValue;
650 + }
651 + uv_rwlock_rdunlock(&pg_cache->metrics_index.lock);
652 + if (NULL == PValue) {
653 + return NULL;
654 + }
655 + } else {
656 + page_index = index;
657 + }
658 + pg_cache_reserve_pages(ctx, 1);
659 +
660 + page_not_in_cache = 0;
661 + uv_rwlock_rdlock(&page_index->lock);
662 + while (1) {
663 + Index = (Word_t)(point_in_time / USEC_PER_SEC);
664 + PValue = JudyLLast(page_index->JudyL_array, &Index, PJE0);
665 + if (likely(NULL != PValue)) {
666 + descr = *PValue;
667 + }
668 + if (NULL == PValue ||
669 + 0 == descr->page_length ||
670 + (INVALID_TIME != point_in_time &&
671 + !is_point_in_time_in_page(descr, point_in_time))) {
672 + /* non-empty page not found */
673 + uv_rwlock_rdunlock(&page_index->lock);
674 +
675 + pg_cache_release_pages(ctx, 1);
676 + return NULL;
677 + }
678 + uv_mutex_lock(&descr->mutex);
679 + flags = descr->flags;
680 + if ((flags & RRD_PAGE_POPULATED) && pg_cache_try_get_unsafe(descr, 0)) {
681 + /* success */
682 + uv_mutex_unlock(&descr->mutex);
683 + debug(D_RRDENGINE, "%s: Page was found in memory.", __func__);
684 + break;
685 + }
686 + if (!(flags & RRD_PAGE_POPULATED) && pg_cache_try_get_unsafe(descr, 1)) {
687 + struct rrdeng_cmd cmd;
688 +
689 + uv_rwlock_rdunlock(&page_index->lock);
690 +
691 + cmd.opcode = RRDENG_READ_PAGE;
692 + cmd.read_page.page_cache_descr = descr;
693 + rrdeng_enq_cmd(&ctx->worker_config, &cmd);
694 +
695 + debug(D_RRDENGINE, "%s: Waiting for page to be asynchronously read from disk:", __func__);
696 + if(unlikely(debug_flags & D_RRDENGINE))
697 + print_page_cache_descr(descr);
698 + while (!(descr->flags & RRD_PAGE_POPULATED)) {
699 + pg_cache_wait_event_unsafe(descr);
700 + }
701 + /* success */
702 + /* Downgrade exclusive reference to allow other readers */
703 + descr->flags &= ~RRD_PAGE_LOCKED;
704 + pg_cache_wake_up_waiters_unsafe(descr);
705 + uv_mutex_unlock(&descr->mutex);
706 + rrd_stat_atomic_add(&ctx->stats.pg_cache_misses, 1);
707 + return descr;
708 + }
709 + uv_rwlock_rdunlock(&page_index->lock);
710 + debug(D_RRDENGINE, "%s: Waiting for page to be unlocked:", __func__);
711 + if(unlikely(debug_flags & D_RRDENGINE))
712 + print_page_cache_descr(descr);
713 + if (!(flags & RRD_PAGE_POPULATED))
714 + page_not_in_cache = 1;
715 + pg_cache_wait_event_unsafe(descr);
716 + uv_mutex_unlock(&descr->mutex);
717 +
718 + /* reset scan to find again */
719 + uv_rwlock_rdlock(&page_index->lock);
720 + }
721 + uv_rwlock_rdunlock(&page_index->lock);
722 +
723 + if (!(flags & RRD_PAGE_DIRTY))
724 + pg_cache_replaceQ_set_hot(ctx, descr);
725 + pg_cache_release_pages(ctx, 1);
726 + if (page_not_in_cache)
727 + rrd_stat_atomic_add(&ctx->stats.pg_cache_misses, 1);
728 + else
729 + rrd_stat_atomic_add(&ctx->stats.pg_cache_hits, 1);
730 + return descr;
731 +}
732 +
733 +struct pg_cache_page_index *create_page_index(uuid_t *id)
734 +{
735 + struct pg_cache_page_index *page_index;
736 +
737 + page_index = mallocz(sizeof(*page_index));
738 + page_index->JudyL_array = (Pvoid_t) NULL;
739 + uuid_copy(page_index->id, *id);
740 + assert(0 == uv_rwlock_init(&page_index->lock));
741 + page_index->oldest_time = INVALID_TIME;
742 + page_index->latest_time = INVALID_TIME;
743 +
744 + return page_index;
745 +}
746 +
747 +static void init_metrics_index(struct rrdengine_instance *ctx)
748 +{
749 + struct page_cache *pg_cache = &ctx->pg_cache;
750 +
751 + pg_cache->metrics_index.JudyHS_array = (Pvoid_t) NULL;
752 + assert(0 == uv_rwlock_init(&pg_cache->metrics_index.lock));
753 +}
754 +
755 +static void init_replaceQ(struct rrdengine_instance *ctx)
756 +{
757 + struct page_cache *pg_cache = &ctx->pg_cache;
758 +
759 + pg_cache->replaceQ.head = NULL;
760 + pg_cache->replaceQ.tail = NULL;
761 + assert(0 == uv_rwlock_init(&pg_cache->replaceQ.lock));
762 +}
763 +
764 +static void init_commited_page_index(struct rrdengine_instance *ctx)
765 +{
766 + struct page_cache *pg_cache = &ctx->pg_cache;
767 +
768 + pg_cache->commited_page_index.JudyL_array = (Pvoid_t) NULL;
769 + assert(0 == uv_rwlock_init(&pg_cache->commited_page_index.lock));
770 + pg_cache->commited_page_index.latest_corr_id = 0;
771 + pg_cache->commited_page_index.nr_commited_pages = 0;
772 +}
773 +
774 +void init_page_cache(struct rrdengine_instance *ctx)
775 +{
776 + struct page_cache *pg_cache = &ctx->pg_cache;
777 +
778 + pg_cache->page_descriptors = 0;
779 + pg_cache->populated_pages = 0;
780 + assert(0 == uv_rwlock_init(&pg_cache->pg_cache_rwlock));
781 +
782 + init_metrics_index(ctx);
783 + init_replaceQ(ctx);
784 + init_commited_page_index(ctx);
785 +}
\ No newline at end of file
database/engine/pagecache.h new
+132
@@ -0,0 +1,132 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_PAGECACHE_H
4 +#define NETDATA_PAGECACHE_H
5 +
6 +#include "rrdengine.h"
7 +
8 +/* Forward declerations */
9 +struct rrdengine_instance;
10 +struct extent_info;
11 +
12 +#define INVALID_TIME (0)
13 +
14 +/* Page flags */
15 +#define RRD_PAGE_DIRTY (1LU << 0)
16 +#define RRD_PAGE_LOCKED (1LU << 1)
17 +#define RRD_PAGE_READ_PENDING (1LU << 2)
18 +#define RRD_PAGE_WRITE_PENDING (1LU << 3)
19 +#define RRD_PAGE_POPULATED (1LU << 4)
20 +
21 +struct rrdeng_page_cache_descr {
22 + void *page;
23 + uint32_t page_length;
24 + usec_t start_time;
25 + usec_t end_time;
26 + uuid_t *id; /* never changes */
27 + struct extent_info *extent;
28 + unsigned long flags;
29 + void *private;
30 + struct rrdeng_page_cache_descr *prev;
31 + struct rrdeng_page_cache_descr *next;
32 +
33 + /* TODO: move waiter logic to concurrency table */
34 + unsigned refcnt;
35 + uv_mutex_t mutex; /* always take it after the page cache lock or after the commit lock */
36 + uv_cond_t cond;
37 + unsigned waiters;
38 + struct rrdeng_collect_handle *handle; /* API user */
39 +};
40 +
41 +#define PAGE_CACHE_MAX_PRELOAD_PAGES (256)
42 +
43 +/* maps time ranges to pages */
44 +struct pg_cache_page_index {
45 + uuid_t id;
46 + /*
47 + * care: JudyL_array indices are converted from useconds to seconds to fit in one word in 32-bit architectures
48 + * TODO: examine if we want to support better granularity than seconds
49 + */
50 + Pvoid_t JudyL_array;
51 + uv_rwlock_t lock;
52 +
53 + /*
54 + * Only one effective writer, data deletion workqueue.
55 + * It's also written during the DB loading phase.
56 + */
57 + usec_t oldest_time;
58 +
59 + /*
60 + * Only one effective writer, data collection thread.
61 + * It's also written by the data deletion workqueue when data collection is disabled for this metric.
62 + */
63 + usec_t latest_time;
64 +};
65 +
66 +/* maps UUIDs to page indices */
67 +struct pg_cache_metrics_index {
68 + uv_rwlock_t lock;
69 + Pvoid_t JudyHS_array;
70 +};
71 +
72 +/* gathers dirty pages to be written on disk */
73 +struct pg_cache_commited_page_index {
74 + uv_rwlock_t lock;
75 +
76 + Pvoid_t JudyL_array;
77 +
78 + /*
79 + * Dirty page correlation ID is a hint. Dirty pages that are correlated should have
80 + * a small correlation ID difference. Dirty pages in memory should never have the
81 + * same ID at the same time for correctness.
82 + */
83 + Word_t latest_corr_id;
84 +
85 + unsigned nr_commited_pages;
86 +};
87 +
88 +/* gathers populated pages to be evicted */
89 +struct pg_cache_replaceQ {
90 + uv_rwlock_t lock; /* LRU lock */
91 +
92 + struct rrdeng_page_cache_descr *head; /* LRU */
93 + struct rrdeng_page_cache_descr *tail; /* MRU */
94 +};
95 +
96 +struct page_cache { /* TODO: add statistics */
97 + uv_rwlock_t pg_cache_rwlock; /* page cache lock */
98 +
99 + struct pg_cache_metrics_index metrics_index;
100 + struct pg_cache_commited_page_index commited_page_index;
101 + struct pg_cache_replaceQ replaceQ;
102 +
103 + unsigned page_descriptors;
104 + unsigned populated_pages;
105 +};
106 +
107 +extern void pg_cache_wake_up_waiters_unsafe(struct rrdeng_page_cache_descr *descr);
108 +extern void pg_cache_wait_event_unsafe(struct rrdeng_page_cache_descr *descr);
109 +extern unsigned long pg_cache_wait_event(struct rrdeng_page_cache_descr *descr);
110 +extern void pg_cache_replaceQ_insert(struct rrdengine_instance *ctx,
111 + struct rrdeng_page_cache_descr *descr);
112 +extern void pg_cache_replaceQ_delete(struct rrdengine_instance *ctx,
113 + struct rrdeng_page_cache_descr *descr);
114 +extern void pg_cache_replaceQ_set_hot(struct rrdengine_instance *ctx,
115 + struct rrdeng_page_cache_descr *descr);
116 +extern struct rrdeng_page_cache_descr *pg_cache_create_descr(void);
117 +extern void pg_cache_put_unsafe(struct rrdeng_page_cache_descr *descr);
118 +extern void pg_cache_put(struct rrdeng_page_cache_descr *descr);
119 +extern void pg_cache_insert(struct rrdengine_instance *ctx, struct pg_cache_page_index *index,
120 + struct rrdeng_page_cache_descr *descr);
121 +extern void pg_cache_punch_hole(struct rrdengine_instance *ctx, struct rrdeng_page_cache_descr *descr);
122 +extern struct pg_cache_page_index *
123 + pg_cache_preload(struct rrdengine_instance *ctx, uuid_t *id, usec_t start_time, usec_t end_time);
124 +extern struct rrdeng_page_cache_descr *
125 + pg_cache_lookup(struct rrdengine_instance *ctx, struct pg_cache_page_index *index, uuid_t *id,
126 + usec_t point_in_time);
127 +extern struct pg_cache_page_index *create_page_index(uuid_t *id);
128 +extern void init_page_cache(struct rrdengine_instance *ctx);
129 +extern void pg_cache_add_new_metric_time(struct pg_cache_page_index *page_index, struct rrdeng_page_cache_descr *descr);
130 +extern void pg_cache_update_metric_times(struct pg_cache_page_index *page_index);
131 +
132 +#endif /* NETDATA_PAGECACHE_H */
\ No newline at end of file
database/engine/rrddiskprotocol.h new
+119
@@ -0,0 +1,119 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_RRDDISKPROTOCOL_H
4 +#define NETDATA_RRDDISKPROTOCOL_H
5 +
6 +#define RRDENG_BLOCK_SIZE (4096)
7 +#define RRDFILE_ALIGNMENT RRDENG_BLOCK_SIZE
8 +
9 +#define RRDENG_MAGIC_SZ (32)
10 +#define RRDENG_DF_MAGIC "netdata-data-file"
11 +#define RRDENG_JF_MAGIC "netdata-journal-file"
12 +
13 +#define RRDENG_VER_SZ (16)
14 +#define RRDENG_DF_VER "1.0"
15 +#define RRDENG_JF_VER "1.0"
16 +
17 +#define UUID_SZ (16)
18 +#define CHECKSUM_SZ (4) /* CRC32 */
19 +
20 +#define RRD_NO_COMPRESSION (0)
21 +#define RRD_LZ4 (1)
22 +
23 +#define RRDENG_DF_SB_PADDING_SZ (RRDENG_BLOCK_SIZE - (RRDENG_MAGIC_SZ + RRDENG_VER_SZ + sizeof(uint8_t)))
24 +/*
25 + * Data file persistent super-block
26 + */
27 +struct rrdeng_df_sb {
28 + char magic_number[RRDENG_MAGIC_SZ];
29 + char version[RRDENG_VER_SZ];
30 + uint8_t tier;
31 + uint8_t padding[RRDENG_DF_SB_PADDING_SZ];
32 +} __attribute__ ((packed));
33 +
34 +/*
35 + * Page types
36 + */
37 +#define PAGE_METRICS (0)
38 +#define PAGE_LOGS (1) /* reserved */
39 +
40 +/*
41 + * Data file page descriptor
42 + */
43 +struct rrdeng_extent_page_descr {
44 + uint8_t type;
45 +
46 + uint8_t uuid[UUID_SZ];
47 + uint32_t page_length;
48 + uint64_t start_time;
49 + uint64_t end_time;
50 +} __attribute__ ((packed));
51 +
52 +/*
53 + * Data file extent header
54 + */
55 +struct rrdeng_df_extent_header {
56 + uint32_t payload_length;
57 + uint8_t compression_algorithm;
58 + uint8_t number_of_pages;
59 + /* #number_of_pages page descriptors follow */
60 + struct rrdeng_extent_page_descr descr[];
61 +} __attribute__ ((packed));
62 +
63 +/*
64 + * Data file extent trailer
65 + */
66 +struct rrdeng_df_extent_trailer {
67 + uint8_t checksum[CHECKSUM_SZ]; /* CRC32 */
68 +} __attribute__ ((packed));
69 +
70 +#define RRDENG_JF_SB_PADDING_SZ (RRDENG_BLOCK_SIZE - (RRDENG_MAGIC_SZ + RRDENG_VER_SZ))
71 +/*
72 + * Journal file super-block
73 + */
74 +struct rrdeng_jf_sb {
75 + char magic_number[RRDENG_MAGIC_SZ];
76 + char version[RRDENG_VER_SZ];
77 + uint8_t padding[RRDENG_JF_SB_PADDING_SZ];
78 +} __attribute__ ((packed));
79 +
80 +/*
81 + * Transaction record types
82 + */
83 +#define STORE_PADDING (0)
84 +#define STORE_DATA (1)
85 +#define STORE_LOGS (2) /* reserved */
86 +
87 +/*
88 + * Journal file transaction record header
89 + */
90 +struct rrdeng_jf_transaction_header {
91 + /* when set to STORE_PADDING jump to start of next block */
92 + uint8_t type;
93 +
94 + uint32_t reserved; /* reserved for future use */
95 + uint64_t id;
96 + uint16_t payload_length;
97 +} __attribute__ ((packed));
98 +
99 +/*
100 + * Journal file transaction record trailer
101 + */
102 +struct rrdeng_jf_transaction_trailer {
103 + uint8_t checksum[CHECKSUM_SZ]; /* CRC32 */
104 +} __attribute__ ((packed));
105 +
106 +/*
107 + * Journal file STORE_DATA action
108 + */
109 +struct rrdeng_jf_store_data {
110 + /* data file extent information */
111 + uint64_t extent_offset;
112 + uint32_t extent_size;
113 +
114 + uint8_t number_of_pages;
115 + /* #number_of_pages page descriptors follow */
116 + struct rrdeng_extent_page_descr descr[];
117 +} __attribute__ ((packed));
118 +
119 +#endif /* NETDATA_RRDDISKPROTOCOL_H */
\ No newline at end of file
database/engine/rrdengine.c new
+780
@@ -0,0 +1,780 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +#define NETDATA_RRD_INTERNALS
3 +
4 +#include "rrdengine.h"
5 +
6 +void sanity_check(void)
7 +{
8 + /* Magic numbers must fit in the super-blocks */
9 + BUILD_BUG_ON(strlen(RRDENG_DF_MAGIC) > RRDENG_MAGIC_SZ);
10 + BUILD_BUG_ON(strlen(RRDENG_JF_MAGIC) > RRDENG_MAGIC_SZ);
11 +
12 + /* Version strings must fit in the super-blocks */
13 + BUILD_BUG_ON(strlen(RRDENG_DF_VER) > RRDENG_VER_SZ);
14 + BUILD_BUG_ON(strlen(RRDENG_JF_VER) > RRDENG_VER_SZ);
15 +
16 + /* Data file super-block cannot be larger than RRDENG_BLOCK_SIZE */
17 + BUILD_BUG_ON(RRDENG_DF_SB_PADDING_SZ < 0);
18 +
19 + BUILD_BUG_ON(sizeof(uuid_t) != UUID_SZ); /* check UUID size */
20 +
21 + /* page count must fit in 8 bits */
22 + BUILD_BUG_ON(MAX_PAGES_PER_EXTENT > 255);
23 +}
24 +
25 +void read_extent_cb(uv_fs_t* req)
26 +{
27 + struct rrdengine_worker_config* wc = req->loop->data;
28 + struct rrdengine_instance *ctx = wc->ctx;
29 + struct extent_io_descriptor *xt_io_descr;
30 + struct rrdeng_page_cache_descr *descr;
31 + int ret;
32 + unsigned i, j, count;
33 + void *page, *uncompressed_buf = NULL;
34 + uint32_t payload_length, payload_offset, page_offset, uncompressed_payload_length;
35 + struct rrdengine_datafile *datafile;
36 + /* persistent structures */
37 + struct rrdeng_df_extent_header *header;
38 + struct rrdeng_df_extent_trailer *trailer;
39 + uLong crc;
40 +
41 + xt_io_descr = req->data;
42 + if (req->result < 0) {
43 + error("%s: uv_fs_read: %s", __func__, uv_strerror((int)req->result));
44 + goto cleanup;
45 + }
46 +
47 + header = xt_io_descr->buf;
48 + payload_length = header->payload_length;
49 + count = header->number_of_pages;
50 +
51 + payload_offset = sizeof(*header) + sizeof(header->descr[0]) * count;
52 +
53 + trailer = xt_io_descr->buf + xt_io_descr->bytes - sizeof(*trailer);
54 + crc = crc32(0L, Z_NULL, 0);
55 + crc = crc32(crc, xt_io_descr->buf, xt_io_descr->bytes - sizeof(*trailer));
56 + ret = crc32cmp(trailer->checksum, crc);
57 + datafile = xt_io_descr->descr_array[0]->extent->datafile;
58 + debug(D_RRDENGINE, "%s: Extent at offset %"PRIu64"(%u) was read from datafile %u-%u. CRC32 check: %s", __func__,
59 + xt_io_descr->pos, xt_io_descr->bytes, datafile->tier, datafile->fileno, ret ? "FAILED" : "SUCCEEDED");
60 + if (unlikely(ret)) {
61 + /* TODO: handle errors */
62 + exit(UV_EIO);
63 + goto cleanup;
64 + }
65 +
66 + if (RRD_NO_COMPRESSION != header->compression_algorithm) {
67 + uncompressed_payload_length = 0;
68 + for (i = 0 ; i < count ; ++i) {
69 + uncompressed_payload_length += header->descr[i].page_length;
70 + }
71 + uncompressed_buf = mallocz(uncompressed_payload_length);
72 + ret = LZ4_decompress_safe(xt_io_descr->buf + payload_offset, uncompressed_buf,
73 + payload_length, uncompressed_payload_length);
74 + ctx->stats.before_decompress_bytes += payload_length;
75 + ctx->stats.after_decompress_bytes += ret;
76 + debug(D_RRDENGINE, "LZ4 decompressed %u bytes to %d bytes.", payload_length, ret);
77 + /* care, we don't hold the descriptor mutex */
78 + }
79 +
80 + for (i = 0 ; i < xt_io_descr->descr_count; ++i) {
81 + page = mallocz(RRDENG_BLOCK_SIZE);
82 + descr = xt_io_descr->descr_array[i];
83 + for (j = 0, page_offset = 0; j < count; ++j) {
84 + /* care, we don't hold the descriptor mutex */
85 + if (!uuid_compare(*(uuid_t *) header->descr[j].uuid, *descr->id) &&
86 + header->descr[j].page_length == descr->page_length &&
87 + header->descr[j].start_time == descr->start_time &&
88 + header->descr[j].end_time == descr->end_time) {
89 + break;
90 + }
91 + page_offset += header->descr[j].page_length;
92 + }
93 + /* care, we don't hold the descriptor mutex */
94 + if (RRD_NO_COMPRESSION == header->compression_algorithm) {
95 + (void) memcpy(page, xt_io_descr->buf + payload_offset + page_offset, descr->page_length);
96 + } else {
97 + (void) memcpy(page, uncompressed_buf + page_offset, descr->page_length);
98 + }
99 + pg_cache_replaceQ_insert(ctx, descr);
100 + uv_mutex_lock(&descr->mutex);
101 + descr->page = page;
102 + descr->flags |= RRD_PAGE_POPULATED;
103 + descr->flags &= ~RRD_PAGE_READ_PENDING;
104 + debug(D_RRDENGINE, "%s: Waking up waiters.", __func__);
105 + if (xt_io_descr->release_descr) {
106 + pg_cache_put_unsafe(descr);
107 + } else {
108 + pg_cache_wake_up_waiters_unsafe(descr);
109 + }
110 + uv_mutex_unlock(&descr->mutex);
111 + }
112 + if (RRD_NO_COMPRESSION != header->compression_algorithm) {
113 + free(uncompressed_buf);
114 + }
115 + if (xt_io_descr->completion)
116 + complete(xt_io_descr->completion);
117 +cleanup:
118 + uv_fs_req_cleanup(req);
119 + free(xt_io_descr->buf);
120 + free(xt_io_descr);
121 +}
122 +
123 +
124 +static void do_read_extent(struct rrdengine_worker_config* wc,
125 + struct rrdeng_page_cache_descr **descr,
126 + unsigned count,
127 + uint8_t release_descr)
128 +{
129 + struct rrdengine_instance *ctx = wc->ctx;
130 + int ret;
131 + unsigned i, size_bytes, pos, real_io_size;
132 +// uint32_t payload_length;
133 + struct extent_io_descriptor *xt_io_descr;
134 + struct rrdengine_datafile *datafile;
135 +
136 + datafile = descr[0]->extent->datafile;
137 + pos = descr[0]->extent->offset;
138 + size_bytes = descr[0]->extent->size;
139 +
140 + xt_io_descr = mallocz(sizeof(*xt_io_descr));
141 + ret = posix_memalign((void *)&xt_io_descr->buf, RRDFILE_ALIGNMENT, ALIGN_BYTES_CEILING(size_bytes));
142 + if (unlikely(ret)) {
143 + fatal("posix_memalign:%s", strerror(ret));
144 + /* free(xt_io_descr);
145 + return;*/
146 + }
147 + for (i = 0 ; i < count; ++i) {
148 + uv_mutex_lock(&descr[i]->mutex);
149 + descr[i]->flags |= RRD_PAGE_READ_PENDING;
150 +// payload_length = descr[i]->page_length;
151 + uv_mutex_unlock(&descr[i]->mutex);
152 +
153 + xt_io_descr->descr_array[i] = descr[i];
154 + }
155 + xt_io_descr->descr_count = count;
156 + xt_io_descr->bytes = size_bytes;
157 + xt_io_descr->pos = pos;
158 + xt_io_descr->req.data = xt_io_descr;
159 + xt_io_descr->completion = NULL;
160 + /* xt_io_descr->descr_commit_idx_array[0] */
161 + xt_io_descr->release_descr = release_descr;
162 +
163 + real_io_size = ALIGN_BYTES_CEILING(size_bytes);
164 + xt_io_descr->iov = uv_buf_init((void *)xt_io_descr->buf, real_io_size);
165 + ret = uv_fs_read(wc->loop, &xt_io_descr->req, datafile->file, &xt_io_descr->iov, 1, pos, read_extent_cb);
166 + assert (-1 != ret);
167 + ctx->stats.io_read_bytes += real_io_size;
168 + ++ctx->stats.io_read_requests;
169 + ctx->stats.io_read_extent_bytes += real_io_size;
170 + ++ctx->stats.io_read_extents;
171 + ctx->stats.pg_cache_backfills += count;
172 +}
173 +
174 +static void commit_data_extent(struct rrdengine_worker_config* wc, struct extent_io_descriptor *xt_io_descr)
175 +{
176 + struct rrdengine_instance *ctx = wc->ctx;
177 + unsigned count, payload_length, descr_size, size_bytes;
178 + void *buf;
179 + /* persistent structures */
180 + struct rrdeng_df_extent_header *df_header;
181 + struct rrdeng_jf_transaction_header *jf_header;
182 + struct rrdeng_jf_store_data *jf_metric_data;
183 + struct rrdeng_jf_transaction_trailer *jf_trailer;
184 + uLong crc;
185 +
186 + df_header = xt_io_descr->buf;
187 + count = df_header->number_of_pages;
188 + descr_size = sizeof(*jf_metric_data->descr) * count;
189 + payload_length = sizeof(*jf_metric_data) + descr_size;
190 + size_bytes = sizeof(*jf_header) + payload_length + sizeof(*jf_trailer);
191 +
192 + buf = wal_get_transaction_buffer(wc, size_bytes);
193 +
194 + jf_header = buf;
195 + jf_header->type = STORE_DATA;
196 + jf_header->reserved = 0;
197 + jf_header->id = ctx->commit_log.transaction_id++;
198 + jf_header->payload_length = payload_length;
199 +
200 + jf_metric_data = buf + sizeof(*jf_header);
201 + jf_metric_data->extent_offset = xt_io_descr->pos;
202 + jf_metric_data->extent_size = xt_io_descr->bytes;
203 + jf_metric_data->number_of_pages = count;
204 + memcpy(jf_metric_data->descr, df_header->descr, descr_size);
205 +
206 + jf_trailer = buf + sizeof(*jf_header) + payload_length;
207 + crc = crc32(0L, Z_NULL, 0);
208 + crc = crc32(crc, buf, sizeof(*jf_header) + payload_length);
209 + crc32set(jf_trailer->checksum, crc);
210 +}
211 +
212 +static void do_commit_transaction(struct rrdengine_worker_config* wc, uint8_t type, void *data)
213 +{
214 + switch (type) {
215 + case STORE_DATA:
216 + commit_data_extent(wc, (struct extent_io_descriptor *)data);
217 + break;
218 + default:
219 + assert(type == STORE_DATA);
220 + break;
221 + }
222 +}
223 +
224 +void flush_pages_cb(uv_fs_t* req)
225 +{
226 + struct rrdengine_worker_config* wc = req->loop->data;
227 + struct rrdengine_instance *ctx = wc->ctx;
228 + struct page_cache *pg_cache = &ctx->pg_cache;
229 + struct extent_io_descriptor *xt_io_descr;
230 + struct rrdeng_page_cache_descr *descr;
231 + struct rrdengine_datafile *datafile;
232 + int ret;
233 + unsigned i, count;
234 + Word_t commit_id;
235 +
236 + xt_io_descr = req->data;
237 + if (req->result < 0) {
238 + error("%s: uv_fs_write: %s", __func__, uv_strerror((int)req->result));
239 + goto cleanup;
240 + }
241 + datafile = xt_io_descr->descr_array[0]->extent->datafile;
242 + debug(D_RRDENGINE, "%s: Extent at offset %"PRIu64"(%u) was written to datafile %u-%u. Waking up waiters.",
243 + __func__, xt_io_descr->pos, xt_io_descr->bytes, datafile->tier, datafile->fileno);
244 +
245 + count = xt_io_descr->descr_count;
246 + for (i = 0 ; i < count ; ++i) {
247 + /* care, we don't hold the descriptor mutex */
248 + descr = xt_io_descr->descr_array[i];
249 +
250 + uv_rwlock_wrlock(&pg_cache->commited_page_index.lock);
251 + commit_id = xt_io_descr->descr_commit_idx_array[i];
252 + ret = JudyLDel(&pg_cache->commited_page_index.JudyL_array, commit_id, PJE0);
253 + assert(1 == ret);
254 + --pg_cache->commited_page_index.nr_commited_pages;
255 + uv_rwlock_wrunlock(&pg_cache->commited_page_index.lock);
256 +
257 + pg_cache_replaceQ_insert(ctx, descr);
258 +
259 + uv_mutex_lock(&descr->mutex);
260 + descr->flags &= ~(RRD_PAGE_DIRTY | RRD_PAGE_WRITE_PENDING);
261 + /* wake up waiters, care no reference being held */
262 + pg_cache_wake_up_waiters_unsafe(descr);
263 + uv_mutex_unlock(&descr->mutex);
264 + }
265 + if (xt_io_descr->completion)
266 + complete(xt_io_descr->completion);
267 +cleanup:
268 + uv_fs_req_cleanup(req);
269 + free(xt_io_descr->buf);
270 + free(xt_io_descr);
271 +}
272 +
273 +/*
274 + * completion must be NULL or valid.
275 + * Returns 0 when no flushing can take place.
276 + * Returns datafile bytes to be written on successful flushing initiation.
277 + */
278 +static int do_flush_pages(struct rrdengine_worker_config* wc, int force, struct completion *completion)
279 +{
280 + struct rrdengine_instance *ctx = wc->ctx;
281 + struct page_cache *pg_cache = &ctx->pg_cache;
282 + int ret;
283 + int compressed_size, max_compressed_size = 0;
284 + unsigned i, count, size_bytes, pos, real_io_size;
285 + uint32_t uncompressed_payload_length, payload_offset;
286 + struct rrdeng_page_cache_descr *descr, *eligible_pages[MAX_PAGES_PER_EXTENT];
287 + struct extent_io_descriptor *xt_io_descr;
288 + void *compressed_buf = NULL;
289 + Word_t descr_commit_idx_array[MAX_PAGES_PER_EXTENT];
290 + Pvoid_t *PValue;
291 + Word_t Index;
292 + uint8_t compression_algorithm = ctx->global_compress_alg;
293 + struct extent_info *extent;
294 + struct rrdengine_datafile *datafile;
295 + /* persistent structures */
296 + struct rrdeng_df_extent_header *header;
297 + struct rrdeng_df_extent_trailer *trailer;
298 + uLong crc;
299 +
300 + if (force) {
301 + debug(D_RRDENGINE, "Asynchronous flushing of extent has been forced by page pressure.");
302 + }
303 + uv_rwlock_rdlock(&pg_cache->commited_page_index.lock);
304 + for (Index = 0, count = 0, uncompressed_payload_length = 0,
305 + PValue = JudyLFirst(pg_cache->commited_page_index.JudyL_array, &Index, PJE0),
306 + descr = unlikely(NULL == PValue) ? NULL : *PValue ;
307 +
308 + descr != NULL && count != MAX_PAGES_PER_EXTENT ;
309 +
310 + PValue = JudyLNext(pg_cache->commited_page_index.JudyL_array, &Index, PJE0),
311 + descr = unlikely(NULL == PValue) ? NULL : *PValue) {
312 + assert(0 != descr->page_length);
313 +
314 + uv_mutex_lock(&descr->mutex);
315 + if (!(descr->flags & RRD_PAGE_WRITE_PENDING)) {
316 + /* care, no reference being held */
317 + descr->flags |= RRD_PAGE_WRITE_PENDING;
318 + uncompressed_payload_length += descr->page_length;
319 + descr_commit_idx_array[count] = Index;
320 + eligible_pages[count++] = descr;
321 + }
322 + uv_mutex_unlock(&descr->mutex);
323 + }
324 + uv_rwlock_rdunlock(&pg_cache->commited_page_index.lock);
325 +
326 + if (!count) {
327 + debug(D_RRDENGINE, "%s: no pages eligible for flushing.", __func__);
328 + if (completion)
329 + complete(completion);
330 + return 0;
331 + }
332 + xt_io_descr = mallocz(sizeof(*xt_io_descr));
333 + payload_offset = sizeof(*header) + count * sizeof(header->descr[0]);
334 + switch (compression_algorithm) {
335 + case RRD_NO_COMPRESSION:
336 + size_bytes = payload_offset + uncompressed_payload_length + sizeof(*trailer);
337 + break;
338 + default: /* Compress */
339 + assert(uncompressed_payload_length < LZ4_MAX_INPUT_SIZE);
340 + max_compressed_size = LZ4_compressBound(uncompressed_payload_length);
341 + compressed_buf = mallocz(max_compressed_size);
342 + size_bytes = payload_offset + MAX(uncompressed_payload_length, (unsigned)max_compressed_size) + sizeof(*trailer);
343 + break;
344 + }
345 + ret = posix_memalign((void *)&xt_io_descr->buf, RRDFILE_ALIGNMENT, ALIGN_BYTES_CEILING(size_bytes));
346 + if (unlikely(ret)) {
347 + fatal("posix_memalign:%s", strerror(ret));
348 + /* free(xt_io_descr);*/
349 + }
350 + (void) memcpy(xt_io_descr->descr_array, eligible_pages, sizeof(struct rrdeng_page_cache_descr *) * count);
351 + xt_io_descr->descr_count = count;
352 +
353 + pos = 0;
354 + header = xt_io_descr->buf;
355 + header->compression_algorithm = compression_algorithm;
356 + header->number_of_pages = count;
357 + pos += sizeof(*header);
358 +
359 + extent = mallocz(sizeof(*extent) + count * sizeof(extent->pages[0]));
360 + datafile = ctx->datafiles.last; /* TODO: check for exceeded size quota */
361 + extent->offset = datafile->pos;
362 + extent->number_of_pages = count;
363 + extent->datafile = datafile;
364 + extent->next = NULL;
365 +
366 + for (i = 0 ; i < count ; ++i) {
367 + /* This is here for performance reasons */
368 + xt_io_descr->descr_commit_idx_array[i] = descr_commit_idx_array[i];
369 +
370 + descr = xt_io_descr->descr_array[i];
371 + header->descr[i].type = PAGE_METRICS;
372 + uuid_copy(*(uuid_t *)header->descr[i].uuid, *descr->id);
373 + header->descr[i].page_length = descr->page_length;
374 + header->descr[i].start_time = descr->start_time;
375 + header->descr[i].end_time = descr->end_time;
376 + pos += sizeof(header->descr[i]);
377 + }
378 + for (i = 0 ; i < count ; ++i) {
379 + descr = xt_io_descr->descr_array[i];
380 + /* care, we don't hold the descriptor mutex */
381 + (void) memcpy(xt_io_descr->buf + pos, descr->page, descr->page_length);
382 + descr->extent = extent;
383 + extent->pages[i] = descr;
384 +
385 + pos += descr->page_length;
386 + }
387 + df_extent_insert(extent);
388 +
389 + switch (compression_algorithm) {
390 + case RRD_NO_COMPRESSION:
391 + header->payload_length = uncompressed_payload_length;
392 + break;
393 + default: /* Compress */
394 + compressed_size = LZ4_compress_default(xt_io_descr->buf + payload_offset, compressed_buf,
395 + uncompressed_payload_length, max_compressed_size);
396 + ctx->stats.before_compress_bytes += uncompressed_payload_length;
397 + ctx->stats.after_compress_bytes += compressed_size;
398 + debug(D_RRDENGINE, "LZ4 compressed %"PRIu32" bytes to %d bytes.", uncompressed_payload_length, compressed_size);
399 + (void) memcpy(xt_io_descr->buf + payload_offset, compressed_buf, compressed_size);
400 + free(compressed_buf);
401 + size_bytes = payload_offset + compressed_size + sizeof(*trailer);
402 + header->payload_length = compressed_size;
403 + break;
404 + }
405 + extent->size = size_bytes;
406 + xt_io_descr->bytes = size_bytes;
407 + xt_io_descr->pos = datafile->pos;
408 + xt_io_descr->req.data = xt_io_descr;
409 + xt_io_descr->completion = completion;
410 +
411 + trailer = xt_io_descr->buf + size_bytes - sizeof(*trailer);
412 + crc = crc32(0L, Z_NULL, 0);
413 + crc = crc32(crc, xt_io_descr->buf, size_bytes - sizeof(*trailer));
414 + crc32set(trailer->checksum, crc);
415 +
416 + real_io_size = ALIGN_BYTES_CEILING(size_bytes);
417 + xt_io_descr->iov = uv_buf_init((void *)xt_io_descr->buf, real_io_size);
418 + ret = uv_fs_write(wc->loop, &xt_io_descr->req, datafile->file, &xt_io_descr->iov, 1, datafile->pos, flush_pages_cb);
419 + assert (-1 != ret);
420 + ctx->stats.io_write_bytes += real_io_size;
421 + ++ctx->stats.io_write_requests;
422 + ctx->stats.io_write_extent_bytes += real_io_size;
423 + ++ctx->stats.io_write_extents;
424 + do_commit_transaction(wc, STORE_DATA, xt_io_descr);
425 + datafile->pos += ALIGN_BYTES_CEILING(size_bytes);
426 + ctx->disk_space += ALIGN_BYTES_CEILING(size_bytes);
427 + rrdeng_test_quota(wc);
428 +
429 + return ALIGN_BYTES_CEILING(size_bytes);
430 +}
431 +
432 +static void after_delete_old_data(uv_work_t *req, int status)
433 +{
434 + struct rrdengine_instance *ctx = req->data;
435 + struct rrdengine_worker_config* wc = &ctx->worker_config;
436 + struct rrdengine_datafile *datafile;
437 + struct rrdengine_journalfile *journalfile;
438 + unsigned bytes;
439 +
440 + (void)status;
441 + datafile = ctx->datafiles.first;
442 + journalfile = datafile->journalfile;
443 + bytes = datafile->pos + journalfile->pos;
444 +
445 + datafile_list_delete(ctx, datafile);
446 + destroy_journal_file(journalfile, datafile);
447 + destroy_data_file(datafile);
448 + info("Deleted data file \""DATAFILE_PREFIX RRDENG_FILE_NUMBER_PRINT_TMPL DATAFILE_EXTENSION"\".",
449 + datafile->tier, datafile->fileno);
450 + free(journalfile);
451 + free(datafile);
452 +
453 + ctx->disk_space -= bytes;
454 + info("Reclaimed %u bytes of disk space.", bytes);
455 +
456 + /* unfreeze command processing */
457 + wc->now_deleting.data = NULL;
458 + /* wake up event loop */
459 + assert(0 == uv_async_send(&wc->async));
460 +}
461 +
462 +static void delete_old_data(uv_work_t *req)
463 +{
464 + struct rrdengine_instance *ctx = req->data;
465 + struct rrdengine_datafile *datafile;
466 + struct extent_info *extent, *next;
467 + struct rrdeng_page_cache_descr *descr;
468 + unsigned count, i;
469 +
470 + /* Safe to use since it will be deleted after we are done */
471 + datafile = ctx->datafiles.first;
472 +
473 + for (extent = datafile->extents.first ; extent != NULL ; extent = next) {
474 + count = extent->number_of_pages;
475 + for (i = 0 ; i < count ; ++i) {
476 + descr = extent->pages[i];
477 + pg_cache_punch_hole(ctx, descr);
478 + }
479 + next = extent->next;
480 + free(extent);
481 + }
482 +}
483 +
484 +void rrdeng_test_quota(struct rrdengine_worker_config* wc)
485 +{
486 + struct rrdengine_instance *ctx = wc->ctx;
487 + struct rrdengine_datafile *datafile;
488 + unsigned current_size, target_size;
489 + uint8_t out_of_space, only_one_datafile;
490 +
491 + out_of_space = 0;
492 + if (unlikely(ctx->disk_space > ctx->max_disk_space)) {
493 + out_of_space = 1;
494 + }
495 + datafile = ctx->datafiles.last;
496 + current_size = datafile->pos;
497 + target_size = ctx->max_disk_space / TARGET_DATAFILES;
498 + target_size = MIN(target_size, MAX_DATAFILE_SIZE);
499 + target_size = MAX(target_size, MIN_DATAFILE_SIZE);
500 + only_one_datafile = (datafile == ctx->datafiles.first) ? 1 : 0;
501 + if (unlikely(current_size >= target_size || (out_of_space && only_one_datafile))) {
502 + /* Finalize data and journal file and create a new pair */
503 + wal_flush_transaction_buffer(wc);
504 + create_new_datafile_pair(ctx, 1, datafile->fileno + 1);
505 + }
506 + if (unlikely(out_of_space)) {
507 + /* delete old data */
508 + if (wc->now_deleting.data) {
509 + /* already deleting data */
510 + return;
511 + }
512 + info("Deleting data file \""DATAFILE_PREFIX RRDENG_FILE_NUMBER_PRINT_TMPL DATAFILE_EXTENSION"\".",
513 + ctx->datafiles.first->tier, ctx->datafiles.first->fileno);
514 + wc->now_deleting.data = ctx;
515 + uv_queue_work(wc->loop, &wc->now_deleting, delete_old_data, after_delete_old_data);
516 + }
517 +}
518 +
519 +int init_rrd_files(struct rrdengine_instance *ctx)
520 +{
521 + return init_data_files(ctx);
522 +}
523 +
524 +void rrdeng_init_cmd_queue(struct rrdengine_worker_config* wc)
525 +{
526 + wc->cmd_queue.head = wc->cmd_queue.tail = 0;
527 + wc->queue_size = 0;
528 + assert(0 == uv_cond_init(&wc->cmd_cond));
529 + assert(0 == uv_mutex_init(&wc->cmd_mutex));
530 +}
531 +
532 +void rrdeng_enq_cmd(struct rrdengine_worker_config* wc, struct rrdeng_cmd *cmd)
533 +{
534 + unsigned queue_size;
535 +
536 + /* wait for free space in queue */
537 + uv_mutex_lock(&wc->cmd_mutex);
538 + while ((queue_size = wc->queue_size) == RRDENG_CMD_Q_MAX_SIZE) {
539 + uv_cond_wait(&wc->cmd_cond, &wc->cmd_mutex);
540 + }
541 + assert(queue_size < RRDENG_CMD_Q_MAX_SIZE);
542 + /* enqueue command */
543 + wc->cmd_queue.cmd_array[wc->cmd_queue.tail] = *cmd;
544 + wc->cmd_queue.tail = wc->cmd_queue.tail != RRDENG_CMD_Q_MAX_SIZE - 1 ?
545 + wc->cmd_queue.tail + 1 : 0;
546 + wc->queue_size = queue_size + 1;
547 + uv_mutex_unlock(&wc->cmd_mutex);
548 +
549 + /* wake up event loop */
550 + assert(0 == uv_async_send(&wc->async));
551 +}
552 +
553 +struct rrdeng_cmd rrdeng_deq_cmd(struct rrdengine_worker_config* wc)
554 +{
555 + struct rrdeng_cmd ret;
556 + unsigned queue_size;
557 +
558 + uv_mutex_lock(&wc->cmd_mutex);
559 + queue_size = wc->queue_size;
560 + if (queue_size == 0) {
561 + ret.opcode = RRDENG_NOOP;
562 + } else {
563 + /* dequeue command */
564 + ret = wc->cmd_queue.cmd_array[wc->cmd_queue.head];
565 + if (queue_size == 1) {
566 + wc->cmd_queue.head = wc->cmd_queue.tail = 0;
567 + } else {
568 + wc->cmd_queue.head = wc->cmd_queue.head != RRDENG_CMD_Q_MAX_SIZE - 1 ?
569 + wc->cmd_queue.head + 1 : 0;
570 + }
571 + wc->queue_size = queue_size - 1;
572 +
573 + /* wake up producers */
574 + uv_cond_signal(&wc->cmd_cond);
575 + }
576 + uv_mutex_unlock(&wc->cmd_mutex);
577 +
578 + return ret;
579 +}
580 +
581 +void async_cb(uv_async_t *handle)
582 +{
583 + uv_stop(handle->loop);
584 + uv_update_time(handle->loop);
585 + debug(D_RRDENGINE, "%s called, active=%d.", __func__, uv_is_active((uv_handle_t *)handle));
586 +}
587 +
588 +void timer_cb(uv_timer_t* handle)
589 +{
590 + struct rrdengine_worker_config* wc = handle->data;
591 + struct rrdengine_instance *ctx = wc->ctx;
592 +
593 + uv_stop(handle->loop);
594 + uv_update_time(handle->loop);
595 + rrdeng_test_quota(wc);
596 + debug(D_RRDENGINE, "%s: timeout reached.", __func__);
597 + if (likely(!wc->now_deleting.data)) {
598 + unsigned total_bytes, bytes_written;
599 +
600 + /* There is free space so we can write to disk */
601 + debug(D_RRDENGINE, "Flushing pages to disk.");
602 + for (total_bytes = bytes_written = do_flush_pages(wc, 0, NULL) ;
603 + bytes_written && (total_bytes < DATAFILE_IDEAL_IO_SIZE) ;
604 + total_bytes += bytes_written) {
605 + bytes_written = do_flush_pages(wc, 0, NULL);
606 + }
607 + }
608 +#ifdef NETDATA_INTERNAL_CHECKS
609 + {
610 + char buf[4096];
611 + debug(D_RRDENGINE, "%s", get_rrdeng_statistics(ctx, buf, sizeof(buf)));
612 + }
613 +#endif
614 +}
615 +
616 +/* Flushes dirty pages when timer expires */
617 +#define TIMER_PERIOD_MS (1000)
618 +
619 +#define CMD_BATCH_SIZE (256)
620 +
621 +void rrdeng_worker(void* arg)
622 +{
623 + struct rrdengine_worker_config* wc = arg;
624 + struct rrdengine_instance *ctx = wc->ctx;
625 + uv_loop_t* loop;
626 + int shutdown;
627 + enum rrdeng_opcode opcode;
628 + uv_timer_t timer_req;
629 + struct rrdeng_cmd cmd;
630 +
631 + rrdeng_init_cmd_queue(wc);
632 +
633 + loop = wc->loop = mallocz(sizeof(uv_loop_t));
634 + uv_loop_init(loop);
635 + loop->data = wc;
636 +
637 + uv_async_init(wc->loop, &wc->async, async_cb);
638 + wc->async.data = wc;
639 +
640 + wc->now_deleting.data = NULL;
641 +
642 + /* dirty page flushing timer */
643 + uv_timer_init(loop, &timer_req);
644 + timer_req.data = wc;
645 +
646 + /* wake up initialization thread */
647 + complete(&ctx->rrdengine_completion);
648 +
649 + uv_timer_start(&timer_req, timer_cb, TIMER_PERIOD_MS, TIMER_PERIOD_MS);
650 + shutdown = 0;
651 + while (shutdown == 0 || uv_loop_alive(loop)) {
652 + uv_run(loop, UV_RUN_DEFAULT);
653 + /* wait for commands */
654 + do {
655 + cmd = rrdeng_deq_cmd(wc);
656 + opcode = cmd.opcode;
657 +
658 + switch (opcode) {
659 + case RRDENG_NOOP:
660 + /* the command queue was empty, do nothing */
661 + break;
662 + case RRDENG_SHUTDOWN:
663 + shutdown = 1;
664 + if (unlikely(wc->now_deleting.data)) {
665 + /* postpone shutdown until after deletion */
666 + info("Postponing shutting RRD engine event loop down until after datafile deletion is finished.");
667 + rrdeng_enq_cmd(wc, &cmd);
668 + break;
669 + }
670 + /*
671 + * uv_async_send after uv_close does not seem to crash in linux at the moment,
672 + * it is however undocumented behaviour and we need to be aware if this becomes
673 + * an issue in the future.
674 + */
675 + uv_close((uv_handle_t *)&wc->async, NULL);
676 + assert(0 == uv_timer_stop(&timer_req));
677 + uv_close((uv_handle_t *)&timer_req, NULL);
678 + info("Shutting down RRD engine event loop.");
679 + while (do_flush_pages(wc, 1, NULL)) {
680 + ; /* Force flushing of all commited pages. */
681 + }
682 + break;
683 + case RRDENG_READ_PAGE:
684 + do_read_extent(wc, &cmd.read_page.page_cache_descr, 1, 0);
685 + break;
686 + case RRDENG_READ_EXTENT:
687 + do_read_extent(wc, cmd.read_extent.page_cache_descr, cmd.read_extent.page_count, 1);
688 + break;
689 + case RRDENG_COMMIT_PAGE:
690 + do_commit_transaction(wc, STORE_DATA, NULL);
691 + break;
692 + case RRDENG_FLUSH_PAGES: {
693 + unsigned total_bytes, bytes_written;
694 +
695 + /* First I/O should be enough to call completion */
696 + bytes_written = do_flush_pages(wc, 1, cmd.completion);
697 + for (total_bytes = bytes_written ;
698 + bytes_written && (total_bytes < DATAFILE_IDEAL_IO_SIZE) ;
699 + total_bytes += bytes_written) {
700 + bytes_written = do_flush_pages(wc, 1, NULL);
701 + }
702 + break;
703 + }
704 + default:
705 + debug(D_RRDENGINE, "%s: default.", __func__);
706 + break;
707 + }
708 + } while (opcode != RRDENG_NOOP);
709 + }
710 + /* cleanup operations of the event loop */
711 + wal_flush_transaction_buffer(wc);
712 + uv_run(loop, UV_RUN_DEFAULT);
713 +
714 + info("Shutting down RRD engine event loop complete.");
715 + /* TODO: don't let the API block by waiting to enqueue commands */
716 + uv_cond_destroy(&wc->cmd_cond);
717 +/* uv_mutex_destroy(&wc->cmd_mutex); */
718 + assert(0 == uv_loop_close(loop));
719 + free(loop);
720 +}
721 +
722 +
723 +#define NR_PAGES (256)
724 +static void basic_functional_test(struct rrdengine_instance *ctx)
725 +{
726 + int i, j, failed_validations;
727 + uuid_t uuid[NR_PAGES];
728 + void *buf;
729 + struct rrdeng_page_cache_descr *handle[NR_PAGES];
730 + char uuid_str[37];
731 + char backup[NR_PAGES][37 * 100]; /* backup storage for page data verification */
732 +
733 + for (i = 0 ; i < NR_PAGES ; ++i) {
734 + uuid_generate(uuid[i]);
735 + uuid_unparse_lower(uuid[i], uuid_str);
736 +// fprintf(stderr, "Generated uuid[%d]=%s\n", i, uuid_str);
737 + buf = rrdeng_create_page(&uuid[i], &handle[i]);
738 + /* Each page contains 10 times its own UUID stringified */
739 + for (j = 0 ; j < 100 ; ++j) {
740 + strcpy(buf + 37 * j, uuid_str);
741 + strcpy(backup[i] + 37 * j, uuid_str);
742 + }
743 + rrdeng_commit_page(ctx, handle[i], (Word_t)i);
744 + }
745 + fprintf(stderr, "\n********** CREATED %d METRIC PAGES ***********\n\n", NR_PAGES);
746 + failed_validations = 0;
747 + for (i = 0 ; i < NR_PAGES ; ++i) {
748 + buf = rrdeng_get_latest_page(ctx, &uuid[i], (void **)&handle[i]);
749 + if (NULL == buf) {
750 + ++failed_validations;
751 + fprintf(stderr, "Page %d was LOST.\n", i);
752 + }
753 + if (memcmp(backup[i], buf, 37 * 100)) {
754 + ++failed_validations;
755 + fprintf(stderr, "Page %d data comparison with backup FAILED validation.\n", i);
756 + }
757 + rrdeng_put_page(ctx, handle[i]);
758 + }
759 + fprintf(stderr, "\n********** CORRECTLY VALIDATED %d/%d METRIC PAGES ***********\n\n",
760 + NR_PAGES - failed_validations, NR_PAGES);
761 +
762 +}
763 +/* C entry point for development purposes
764 + * make "LDFLAGS=-errdengine_main"
765 + */
766 +void rrdengine_main(void)
767 +{
768 + int ret;
769 + struct rrdengine_instance *ctx;
770 +
771 + ret = rrdeng_init(&ctx, "/tmp", RRDENG_MIN_PAGE_CACHE_SIZE_MB, RRDENG_MIN_DISK_SPACE_MB);
772 + if (ret) {
773 + exit(ret);
774 + }
775 + basic_functional_test(ctx);
776 +
777 + rrdeng_exit(ctx);
778 + fprintf(stderr, "Hello world!");
779 + exit(0);
780 +}
\ No newline at end of file
database/engine/rrdengine.h new
+171
@@ -0,0 +1,171 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_RRDENGINE_H
4 +#define NETDATA_RRDENGINE_H
5 +
6 +#ifndef _GNU_SOURCE
7 +#define _GNU_SOURCE
8 +#endif
9 +#include <fcntl.h>
10 +#include <aio.h>
11 +#include <uv.h>
12 +#include <assert.h>
13 +#include <lz4.h>
14 +#include <Judy.h>
15 +#include <openssl/sha.h>
16 +#include <openssl/evp.h>
17 +#include <stdint.h>
18 +#include "../rrd.h"
19 +#include "rrddiskprotocol.h"
20 +#include "rrdenginelib.h"
21 +#include "datafile.h"
22 +#include "journalfile.h"
23 +#include "rrdengineapi.h"
24 +#include "pagecache.h"
25 +
26 +#ifdef NETDATA_RRD_INTERNALS
27 +
28 +#endif /* NETDATA_RRD_INTERNALS */
29 +
30 +/* Forward declerations */
31 +struct rrdengine_instance;
32 +
33 +#define MAX_PAGES_PER_EXTENT (64) /* TODO: can go higher only when journal supports bigger than 4KiB transactions */
34 +
35 +#define RRDENG_FILE_NUMBER_SCAN_TMPL "%1u-%10u"
36 +#define RRDENG_FILE_NUMBER_PRINT_TMPL "%1.1u-%10.10u"
37 +
38 +
39 +typedef enum {
40 + RRDENGINE_STATUS_UNINITIALIZED = 0,
41 + RRDENGINE_STATUS_INITIALIZING,
42 + RRDENGINE_STATUS_INITIALIZED
43 +} rrdengine_state_t;
44 +
45 +enum rrdeng_opcode {
46 + /* can be used to return empty status or flush the command queue */
47 + RRDENG_NOOP = 0,
48 +
49 + RRDENG_READ_PAGE,
50 + RRDENG_READ_EXTENT,
51 + RRDENG_COMMIT_PAGE,
52 + RRDENG_FLUSH_PAGES,
53 + RRDENG_SHUTDOWN,
54 +
55 + RRDENG_MAX_OPCODE
56 +};
57 +
58 +struct rrdeng_cmd {
59 + enum rrdeng_opcode opcode;
60 + union {
61 + struct rrdeng_read_page {
62 + struct rrdeng_page_cache_descr *page_cache_descr;
63 + } read_page;
64 + struct rrdeng_read_extent {
65 + struct rrdeng_page_cache_descr *page_cache_descr[MAX_PAGES_PER_EXTENT];
66 + int page_count;
67 + } read_extent;
68 + struct completion *completion;
69 + };
70 +};
71 +
72 +#define RRDENG_CMD_Q_MAX_SIZE (2048)
73 +
74 +struct rrdeng_cmdqueue {
75 + unsigned head, tail;
76 + struct rrdeng_cmd cmd_array[RRDENG_CMD_Q_MAX_SIZE];
77 +};
78 +
79 +struct extent_io_descriptor {
80 + uv_fs_t req;
81 + uv_buf_t iov;
82 + void *buf;
83 + uint64_t pos;
84 + unsigned bytes;
85 + struct completion *completion;
86 + unsigned descr_count;
87 + int release_descr;
88 + struct rrdeng_page_cache_descr *descr_array[MAX_PAGES_PER_EXTENT];
89 + Word_t descr_commit_idx_array[MAX_PAGES_PER_EXTENT];
90 +};
91 +
92 +struct generic_io_descriptor {
93 + uv_fs_t req;
94 + uv_buf_t iov;
95 + void *buf;
96 + uint64_t pos;
97 + unsigned bytes;
98 + struct completion *completion;
99 +};
100 +
101 +struct rrdengine_worker_config {
102 + struct rrdengine_instance *ctx;
103 +
104 + uv_thread_t thread;
105 + uv_loop_t* loop;
106 + uv_async_t async;
107 + uv_work_t now_deleting;
108 +
109 + /* FIFO command queue */
110 + uv_mutex_t cmd_mutex;
111 + uv_cond_t cmd_cond;
112 + volatile unsigned queue_size;
113 + struct rrdeng_cmdqueue cmd_queue;
114 +};
115 +
116 +/*
117 + * Debug statistics not used by code logic.
118 + * They only describe operations since DB engine instance load time.
119 + */
120 +struct rrdengine_statistics {
121 + rrdeng_stats_t metric_API_producers;
122 + rrdeng_stats_t metric_API_consumers;
123 + rrdeng_stats_t pg_cache_insertions;
124 + rrdeng_stats_t pg_cache_deletions;
125 + rrdeng_stats_t pg_cache_hits;
126 + rrdeng_stats_t pg_cache_misses;
127 + rrdeng_stats_t pg_cache_backfills;
128 + rrdeng_stats_t pg_cache_evictions;
129 + rrdeng_stats_t before_decompress_bytes;
130 + rrdeng_stats_t after_decompress_bytes;
131 + rrdeng_stats_t before_compress_bytes;
132 + rrdeng_stats_t after_compress_bytes;
133 + rrdeng_stats_t io_write_bytes;
134 + rrdeng_stats_t io_write_requests;
135 + rrdeng_stats_t io_read_bytes;
136 + rrdeng_stats_t io_read_requests;
137 + rrdeng_stats_t io_write_extent_bytes;
138 + rrdeng_stats_t io_write_extents;
139 + rrdeng_stats_t io_read_extent_bytes;
140 + rrdeng_stats_t io_read_extents;
141 + rrdeng_stats_t datafile_creations;
142 + rrdeng_stats_t datafile_deletions;
143 + rrdeng_stats_t journalfile_creations;
144 + rrdeng_stats_t journalfile_deletions;
145 +};
146 +
147 +struct rrdengine_instance {
148 + rrdengine_state_t rrdengine_state;
149 + struct rrdengine_worker_config worker_config;
150 + struct completion rrdengine_completion;
151 + struct page_cache pg_cache;
152 + uint8_t global_compress_alg;
153 + struct transaction_commit_log commit_log;
154 + struct rrdengine_datafile_list datafiles;
155 + char dbfiles_path[FILENAME_MAX+1];
156 + uint64_t disk_space;
157 + uint64_t max_disk_space;
158 + unsigned long max_cache_pages;
159 + unsigned long cache_pages_low_watermark;
160 +
161 + struct rrdengine_statistics stats;
162 +};
163 +
164 +extern void sanity_check(void);
165 +extern int init_rrd_files(struct rrdengine_instance *ctx);
166 +extern void rrdeng_test_quota(struct rrdengine_worker_config* wc);
167 +extern void rrdeng_worker(void* arg);
168 +extern void rrdeng_enq_cmd(struct rrdengine_worker_config* wc, struct rrdeng_cmd *cmd);
169 +extern struct rrdeng_cmd rrdeng_deq_cmd(struct rrdengine_worker_config* wc);
170 +
171 +#endif /* NETDATA_RRDENGINE_H */
\ No newline at end of file
database/engine/rrdengineapi.c new
+484
@@ -0,0 +1,484 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +#include "rrdengine.h"
3 +
4 +/* Default global database instance */
5 +static struct rrdengine_instance default_global_ctx;
6 +
7 +int default_rrdeng_page_cache_mb = RRDENG_MIN_PAGE_CACHE_SIZE_MB;
8 +int default_rrdeng_disk_quota_mb = RRDENG_MIN_DISK_SPACE_MB;
9 +
10 +/*
11 + * Gets a handle for storing metrics to the database.
12 + * The handle must be released with rrdeng_store_metric_final().
13 + */
14 +void rrdeng_store_metric_init(RRDDIM *rd)
15 +{
16 + struct rrdeng_collect_handle *handle;
17 + struct page_cache *pg_cache;
18 + struct rrdengine_instance *ctx;
19 + uuid_t temp_id;
20 + Pvoid_t *PValue;
21 + struct pg_cache_page_index *page_index;
22 + EVP_MD_CTX *evpctx;
23 + unsigned char hash_value[EVP_MAX_MD_SIZE];
24 + unsigned int hash_len;
25 +
26 + //&default_global_ctx; TODO: test this use case or remove it?
27 +
28 + ctx = rd->rrdset->rrdhost->rrdeng_ctx;
29 + pg_cache = &ctx->pg_cache;
30 + handle = &rd->state->handle.rrdeng;
31 + handle->ctx = ctx;
32 +
33 + evpctx = EVP_MD_CTX_create();
34 + EVP_DigestInit_ex(evpctx, EVP_sha256(), NULL);
35 + EVP_DigestUpdate(evpctx, rd->id, strlen(rd->id));
36 + EVP_DigestUpdate(evpctx, rd->rrdset->id, strlen(rd->rrdset->id));
37 + EVP_DigestFinal_ex(evpctx, hash_value, &hash_len);
38 + EVP_MD_CTX_destroy(evpctx);
39 + assert(hash_len > sizeof(temp_id));
40 + memcpy(&temp_id, hash_value, sizeof(temp_id));
41 +
42 + handle->descr = NULL;
43 + handle->prev_descr = NULL;
44 +
45 + uv_rwlock_rdlock(&pg_cache->metrics_index.lock);
46 + PValue = JudyHSGet(pg_cache->metrics_index.JudyHS_array, &temp_id, sizeof(uuid_t));
47 + if (likely(NULL != PValue)) {
48 + page_index = *PValue;
49 + }
50 + uv_rwlock_rdunlock(&pg_cache->metrics_index.lock);
51 + if (NULL == PValue) {
52 + /* First time we see the UUID */
53 + uv_rwlock_wrlock(&pg_cache->metrics_index.lock);
54 + PValue = JudyHSIns(&pg_cache->metrics_index.JudyHS_array, &temp_id, sizeof(uuid_t), PJE0);
55 + assert(NULL == *PValue); /* TODO: figure out concurrency model */
56 + *PValue = page_index = create_page_index(&temp_id);
57 + uv_rwlock_wrunlock(&pg_cache->metrics_index.lock);
58 + }
59 + rd->state->rrdeng_uuid = &page_index->id;
60 + handle->page_index = page_index;
61 +}
62 +
63 +void rrdeng_store_metric_next(RRDDIM *rd, usec_t point_in_time, storage_number number)
64 +{
65 + struct rrdeng_collect_handle *handle;
66 + struct rrdengine_instance *ctx;
67 + struct page_cache *pg_cache;
68 + struct rrdeng_page_cache_descr *descr;
69 + storage_number *page;
70 +
71 + handle = &rd->state->handle.rrdeng;
72 + ctx = handle->ctx;
73 + pg_cache = &ctx->pg_cache;
74 + descr = handle->descr;
75 + if (unlikely(NULL == descr || descr->page_length + sizeof(number) > RRDENG_BLOCK_SIZE)) {
76 + if (descr) {
77 + descr->handle = NULL;
78 + if (descr->page_length) {
79 +#ifdef NETDATA_INTERNAL_CHECKS
80 + rrd_stat_atomic_add(&ctx->stats.metric_API_producers, -1);
81 +#endif
82 + /* added 1 extra reference to keep 2 dirty pages pinned per metric, expected refcnt = 2 */
83 + ++descr->refcnt;
84 + rrdeng_commit_page(ctx, descr, handle->page_correlation_id);
85 + if (handle->prev_descr) {
86 + /* unpin old second page */
87 + pg_cache_put(handle->prev_descr);
88 + }
89 + handle->prev_descr = descr;
90 + } else {
91 + free(descr->page);
92 + free(descr);
93 + handle->descr = NULL;
94 + }
95 + }
96 + page = rrdeng_create_page(&handle->page_index->id, &descr);
97 + assert(page);
98 + handle->prev_descr = handle->descr;
99 + handle->descr = descr;
100 + descr->handle = handle;
101 + uv_rwlock_wrlock(&pg_cache->commited_page_index.lock);
102 + handle->page_correlation_id = pg_cache->commited_page_index.latest_corr_id++;
103 + uv_rwlock_wrunlock(&pg_cache->commited_page_index.lock);
104 + }
105 + page = descr->page;
106 +
107 + page[descr->page_length / sizeof(number)] = number;
108 + descr->end_time = point_in_time;
109 + descr->page_length += sizeof(number);
110 + if (unlikely(INVALID_TIME == descr->start_time)) {
111 + descr->start_time = point_in_time;
112 +
113 +#ifdef NETDATA_INTERNAL_CHECKS
114 + rrd_stat_atomic_add(&ctx->stats.metric_API_producers, 1);
115 +#endif
116 + pg_cache_insert(ctx, handle->page_index, descr);
117 + } else {
118 + pg_cache_add_new_metric_time(handle->page_index, descr);
119 + }
120 +}
121 +
122 +/*
123 + * Releases the database reference from the handle for storing metrics.
124 + */
125 +void rrdeng_store_metric_finalize(RRDDIM *rd)
126 +{
127 + struct rrdeng_collect_handle *handle;
128 + struct rrdengine_instance *ctx;
129 + struct rrdeng_page_cache_descr *descr;
130 +
131 + handle = &rd->state->handle.rrdeng;
132 + ctx = handle->ctx;
133 + descr = handle->descr;
134 + if (descr) {
135 + descr->handle = NULL;
136 + if (descr->page_length) {
137 +#ifdef NETDATA_INTERNAL_CHECKS
138 + rrd_stat_atomic_add(&ctx->stats.metric_API_producers, -1);
139 +#endif
140 + rrdeng_commit_page(ctx, descr, handle->page_correlation_id);
141 + if (handle->prev_descr) {
142 + /* unpin old second page */
143 + pg_cache_put(handle->prev_descr);
144 + }
145 + } else {
146 + free(descr->page);
147 + free(descr);
148 + }
149 + }
150 +}
151 +
152 +/*
153 + * Gets a handle for loading metrics from the database.
154 + * The handle must be released with rrdeng_load_metric_final().
155 + */
156 +void rrdeng_load_metric_init(RRDDIM *rd, struct rrddim_query_handle *rrdimm_handle, time_t start_time, time_t end_time)
157 +{
158 + struct rrdeng_query_handle *handle;
159 + struct rrdengine_instance *ctx;
160 +
161 + ctx = rd->rrdset->rrdhost->rrdeng_ctx;
162 + rrdimm_handle->start_time = start_time;
163 + rrdimm_handle->end_time = end_time;
164 + handle = &rrdimm_handle->rrdeng;
165 + handle->now = start_time;
166 + handle->dt = rd->rrdset->update_every;
167 + handle->ctx = ctx;
168 + handle->descr = NULL;
169 + handle->page_index = pg_cache_preload(ctx, rd->state->rrdeng_uuid,
170 + start_time * USEC_PER_SEC, end_time * USEC_PER_SEC);
171 +}
172 +
173 +storage_number rrdeng_load_metric_next(struct rrddim_query_handle *rrdimm_handle)
174 +{
175 + struct rrdeng_query_handle *handle;
176 + struct rrdengine_instance *ctx;
177 + struct rrdeng_page_cache_descr *descr;
178 + storage_number *page, ret;
179 + unsigned position;
180 + usec_t point_in_time;
181 +
182 + handle = &rrdimm_handle->rrdeng;
183 + if (unlikely(INVALID_TIME == handle->now)) {
184 + return SN_EMPTY_SLOT;
185 + }
186 + ctx = handle->ctx;
187 + point_in_time = handle->now * USEC_PER_SEC;
188 + descr = handle->descr;
189 +
190 + if (unlikely(NULL == handle->page_index)) {
191 + ret = SN_EMPTY_SLOT;
192 + goto out;
193 + }
194 + if (unlikely(NULL == descr ||
195 + point_in_time < descr->start_time ||
196 + point_in_time > descr->end_time)) {
197 + if (descr) {
198 +#ifdef NETDATA_INTERNAL_CHECKS
199 + rrd_stat_atomic_add(&ctx->stats.metric_API_consumers, -1);
200 +#endif
201 + pg_cache_put(descr);
202 + handle->descr = NULL;
203 + }
204 + descr = pg_cache_lookup(ctx, handle->page_index, &handle->page_index->id, point_in_time);
205 + if (NULL == descr) {
206 + ret = SN_EMPTY_SLOT;
207 + goto out;
208 + }
209 +#ifdef NETDATA_INTERNAL_CHECKS
210 + rrd_stat_atomic_add(&ctx->stats.metric_API_consumers, 1);
211 +#endif
212 + handle->descr = descr;
213 + }
214 + if (unlikely(INVALID_TIME == descr->start_time ||
215 + INVALID_TIME == descr->end_time)) {
216 + ret = SN_EMPTY_SLOT;
217 + goto out;
218 + }
219 + page = descr->page;
220 + if (unlikely(descr->start_time == descr->end_time)) {
221 + ret = page[0];
222 + goto out;
223 + }
224 + position = ((uint64_t)(point_in_time - descr->start_time)) * (descr->page_length / sizeof(storage_number)) /
225 + (descr->end_time - descr->start_time + 1);
226 + ret = page[position];
227 +
228 +out:
229 + handle->now += handle->dt;
230 + if (unlikely(handle->now > rrdimm_handle->end_time)) {
231 + handle->now = INVALID_TIME;
232 + }
233 + return ret;
234 +}
235 +
236 +int rrdeng_load_metric_is_finished(struct rrddim_query_handle *rrdimm_handle)
237 +{
238 + struct rrdeng_query_handle *handle;
239 +
240 + handle = &rrdimm_handle->rrdeng;
241 + return (INVALID_TIME == handle->now);
242 +}
243 +
244 +/*
245 + * Releases the database reference from the handle for loading metrics.
246 + */
247 +void rrdeng_load_metric_finalize(struct rrddim_query_handle *rrdimm_handle)
248 +{
249 + struct rrdeng_query_handle *handle;
250 + struct rrdengine_instance *ctx;
251 + struct rrdeng_page_cache_descr *descr;
252 +
253 + handle = &rrdimm_handle->rrdeng;
254 + ctx = handle->ctx;
255 + descr = handle->descr;
256 + if (descr) {
257 +#ifdef NETDATA_INTERNAL_CHECKS
258 + rrd_stat_atomic_add(&ctx->stats.metric_API_consumers, -1);
259 +#endif
260 + pg_cache_put(descr);
261 + }
262 +}
263 +
264 +time_t rrdeng_metric_latest_time(RRDDIM *rd)
265 +{
266 + struct rrdeng_collect_handle *handle;
267 + struct pg_cache_page_index *page_index;
268 +
269 + handle = &rd->state->handle.rrdeng;
270 + page_index = handle->page_index;
271 +
272 + return page_index->latest_time / USEC_PER_SEC;
273 +}
274 +time_t rrdeng_metric_oldest_time(RRDDIM *rd)
275 +{
276 + struct rrdeng_collect_handle *handle;
277 + struct pg_cache_page_index *page_index;
278 +
279 + handle = &rd->state->handle.rrdeng;
280 + page_index = handle->page_index;
281 +
282 + return page_index->oldest_time / USEC_PER_SEC;
283 +}
284 +
285 +/* Also gets a reference for the page */
286 +void *rrdeng_create_page(uuid_t *id, struct rrdeng_page_cache_descr **ret_descr)
287 +{
288 + struct rrdeng_page_cache_descr *descr;
289 + void *page;
290 + int ret;
291 +
292 + /* TODO: check maximum number of pages in page cache limit */
293 +
294 + page = mallocz(RRDENG_BLOCK_SIZE); /*TODO: add page size */
295 + descr = pg_cache_create_descr();
296 + descr->page = page;
297 + descr->id = id; /* TODO: add page type: metric, log, something? */
298 + descr->flags = RRD_PAGE_DIRTY /*| RRD_PAGE_LOCKED */ | RRD_PAGE_POPULATED /* | BEING_COLLECTED */;
299 + descr->refcnt = 1;
300 +
301 + debug(D_RRDENGINE, "-----------------\nCreated new page:\n-----------------");
302 + if(unlikely(debug_flags & D_RRDENGINE))
303 + print_page_cache_descr(descr);
304 + *ret_descr = descr;
305 + return page;
306 +}
307 +
308 +/* The page must not be empty */
309 +void rrdeng_commit_page(struct rrdengine_instance *ctx, struct rrdeng_page_cache_descr *descr,
310 + Word_t page_correlation_id)
311 +{
312 + struct page_cache *pg_cache = &ctx->pg_cache;
313 + Pvoid_t *PValue;
314 +
315 + if (unlikely(NULL == descr)) {
316 + debug(D_RRDENGINE, "%s: page descriptor is NULL, page has already been force-commited.", __func__);
317 + return;
318 + }
319 + assert(descr->page_length);
320 +
321 + uv_rwlock_wrlock(&pg_cache->commited_page_index.lock);
322 + PValue = JudyLIns(&pg_cache->commited_page_index.JudyL_array, page_correlation_id, PJE0);
323 + *PValue = descr;
324 + ++pg_cache->commited_page_index.nr_commited_pages;
325 + uv_rwlock_wrunlock(&pg_cache->commited_page_index.lock);
326 +
327 + pg_cache_put(descr);
328 +}
329 +
330 +/* Gets a reference for the page */
331 +void *rrdeng_get_latest_page(struct rrdengine_instance *ctx, uuid_t *id, void **handle)
332 +{
333 + struct rrdeng_page_cache_descr *descr;
334 +
335 + debug(D_RRDENGINE, "----------------------\nReading existing page:\n----------------------");
336 + descr = pg_cache_lookup(ctx, NULL, id, INVALID_TIME);
337 + if (NULL == descr) {
338 + *handle = NULL;
339 +
340 + return NULL;
341 + }
342 + *handle = descr;
343 +
344 + return descr->page;
345 +}
346 +
347 +/* Gets a reference for the page */
348 +void *rrdeng_get_page(struct rrdengine_instance *ctx, uuid_t *id, usec_t point_in_time, void **handle)
349 +{
350 + struct rrdeng_page_cache_descr *descr;
351 +
352 + debug(D_RRDENGINE, "----------------------\nReading existing page:\n----------------------");
353 + descr = pg_cache_lookup(ctx, NULL, id, point_in_time);
354 + if (NULL == descr) {
355 + *handle = NULL;
356 +
357 + return NULL;
358 + }
359 + *handle = descr;
360 +
361 + return descr->page;
362 +}
363 +
364 +void rrdeng_get_27_statistics(struct rrdengine_instance *ctx, unsigned long long *array)
365 +{
366 + struct page_cache *pg_cache = &ctx->pg_cache;
367 +
368 + array[0] = (uint64_t)ctx->stats.metric_API_producers;
369 + array[1] = (uint64_t)ctx->stats.metric_API_consumers;
370 + array[2] = (uint64_t)pg_cache->page_descriptors;
371 + array[3] = (uint64_t)pg_cache->populated_pages;
372 + array[4] = (uint64_t)pg_cache->commited_page_index.nr_commited_pages;
373 + array[5] = (uint64_t)ctx->stats.pg_cache_insertions;
374 + array[6] = (uint64_t)ctx->stats.pg_cache_deletions;
375 + array[7] = (uint64_t)ctx->stats.pg_cache_hits;
376 + array[8] = (uint64_t)ctx->stats.pg_cache_misses;
377 + array[9] = (uint64_t)ctx->stats.pg_cache_backfills;
378 + array[10] = (uint64_t)ctx->stats.pg_cache_evictions;
379 + array[11] = (uint64_t)ctx->stats.before_compress_bytes;
380 + array[12] = (uint64_t)ctx->stats.after_compress_bytes;
381 + array[13] = (uint64_t)ctx->stats.before_decompress_bytes;
382 + array[14] = (uint64_t)ctx->stats.after_decompress_bytes;
383 + array[15] = (uint64_t)ctx->stats.io_write_bytes;
384 + array[16] = (uint64_t)ctx->stats.io_write_requests;
385 + array[17] = (uint64_t)ctx->stats.io_read_bytes;
386 + array[18] = (uint64_t)ctx->stats.io_read_requests;
387 + array[19] = (uint64_t)ctx->stats.io_write_extent_bytes;
388 + array[20] = (uint64_t)ctx->stats.io_write_extents;
389 + array[21] = (uint64_t)ctx->stats.io_read_extent_bytes;
390 + array[22] = (uint64_t)ctx->stats.io_read_extents;
391 + array[23] = (uint64_t)ctx->stats.datafile_creations;
392 + array[24] = (uint64_t)ctx->stats.datafile_deletions;
393 + array[25] = (uint64_t)ctx->stats.journalfile_creations;
394 + array[26] = (uint64_t)ctx->stats.journalfile_deletions;
395 +}
396 +
397 +/* Releases reference to page */
398 +void rrdeng_put_page(struct rrdengine_instance *ctx, void *handle)
399 +{
400 + (void)ctx;
401 + pg_cache_put((struct rrdeng_page_cache_descr *)handle);
402 +}
403 +
404 +/*
405 + * Returns 0 on success, 1 on error
406 + */
407 +int rrdeng_init(struct rrdengine_instance **ctxp, char *dbfiles_path, unsigned page_cache_mb, unsigned disk_space_mb)
408 +{
409 + struct rrdengine_instance *ctx;
410 + int error;
411 +
412 + sanity_check();
413 + if (NULL == ctxp) {
414 + /* for testing */
415 + ctx = &default_global_ctx;
416 + memset(ctx, 0, sizeof(*ctx));
417 + } else {
418 + *ctxp = ctx = callocz(1, sizeof(*ctx));
419 + }
420 + if (ctx->rrdengine_state != RRDENGINE_STATUS_UNINITIALIZED) {
421 + return 1;
422 + }
423 + ctx->rrdengine_state = RRDENGINE_STATUS_INITIALIZING;
424 + ctx->global_compress_alg = RRD_LZ4;
425 + if (page_cache_mb < RRDENG_MIN_PAGE_CACHE_SIZE_MB)
426 + page_cache_mb = RRDENG_MIN_PAGE_CACHE_SIZE_MB;
427 + ctx->max_cache_pages = page_cache_mb * (1048576LU / RRDENG_BLOCK_SIZE);
428 + /* try to keep 5% of the page cache free */
429 + ctx->cache_pages_low_watermark = (ctx->max_cache_pages * 95LLU) / 100;
430 + if (disk_space_mb < RRDENG_MIN_DISK_SPACE_MB)
431 + disk_space_mb = RRDENG_MIN_DISK_SPACE_MB;
432 + ctx->max_disk_space = disk_space_mb * 1048576LLU;
433 + strncpyz(ctx->dbfiles_path, dbfiles_path, sizeof(ctx->dbfiles_path) - 1);
434 + ctx->dbfiles_path[sizeof(ctx->dbfiles_path) - 1] = '\0';
435 +
436 + memset(&ctx->worker_config, 0, sizeof(ctx->worker_config));
437 + ctx->worker_config.ctx = ctx;
438 + init_page_cache(ctx);
439 + init_commit_log(ctx);
440 + error = init_rrd_files(ctx);
441 + if (error) {
442 + ctx->rrdengine_state = RRDENGINE_STATUS_UNINITIALIZED;
443 + if (ctx != &default_global_ctx) {
444 + freez(ctx);
445 + }
446 + return 1;
447 + }
448 +
449 + init_completion(&ctx->rrdengine_completion);
450 + assert(0 == uv_thread_create(&ctx->worker_config.thread, rrdeng_worker, &ctx->worker_config));
451 + /* wait for worker thread to initialize */
452 + wait_for_completion(&ctx->rrdengine_completion);
453 + destroy_completion(&ctx->rrdengine_completion);
454 +
455 + ctx->rrdengine_state = RRDENGINE_STATUS_INITIALIZED;
456 + return 0;
457 +}
458 +
459 +/*
460 + * Returns 0 on success, 1 on error
461 + */
462 +int rrdeng_exit(struct rrdengine_instance *ctx)
463 +{
464 + struct rrdeng_cmd cmd;
465 +
466 + if (NULL == ctx) {
467 + /* TODO: move to per host basis */
468 + ctx = &default_global_ctx;
469 + }
470 + if (ctx->rrdengine_state != RRDENGINE_STATUS_INITIALIZED) {
471 + return 1;
472 + }
473 +
474 + /* TODO: add page to page cache */
475 + cmd.opcode = RRDENG_SHUTDOWN;
476 + rrdeng_enq_cmd(&ctx->worker_config, &cmd);
477 +
478 + assert(0 == uv_thread_join(&ctx->worker_config.thread));
479 +
480 + if (ctx != &default_global_ctx) {
481 + freez(ctx);
482 + }
483 + return 0;
484 +}
\ No newline at end of file
database/engine/rrdengineapi.h new
+37
@@ -0,0 +1,37 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_RRDENGINEAPI_H
4 +#define NETDATA_RRDENGINEAPI_H
5 +
6 +#include "rrdengine.h"
7 +
8 +#define RRDENG_MIN_PAGE_CACHE_SIZE_MB (32)
9 +#define RRDENG_MIN_DISK_SPACE_MB (256)
10 +extern int default_rrdeng_page_cache_mb;
11 +extern int default_rrdeng_disk_quota_mb;
12 +
13 +extern void *rrdeng_create_page(uuid_t *id, struct rrdeng_page_cache_descr **ret_descr);
14 +extern void rrdeng_commit_page(struct rrdengine_instance *ctx, struct rrdeng_page_cache_descr *descr,
15 + Word_t page_correlation_id);
16 +extern void *rrdeng_get_latest_page(struct rrdengine_instance *ctx, uuid_t *id, void **handle);
17 +extern void *rrdeng_get_page(struct rrdengine_instance *ctx, uuid_t *id, usec_t point_in_time, void **handle);
18 +extern void rrdeng_put_page(struct rrdengine_instance *ctx, void *handle);
19 +extern void rrdeng_store_metric_init(RRDDIM *rd);
20 +extern void rrdeng_store_metric_next(RRDDIM *rd, usec_t point_in_time, storage_number number);
21 +extern void rrdeng_store_metric_finalize(RRDDIM *rd);
22 +extern void rrdeng_load_metric_init(RRDDIM *rd, struct rrddim_query_handle *rrdimm_handle,
23 + time_t start_time, time_t end_time);
24 +extern storage_number rrdeng_load_metric_next(struct rrddim_query_handle *rrdimm_handle);
25 +extern int rrdeng_load_metric_is_finished(struct rrddim_query_handle *rrdimm_handle);
26 +extern void rrdeng_load_metric_finalize(struct rrddim_query_handle *rrdimm_handle);
27 +extern time_t rrdeng_metric_latest_time(RRDDIM *rd);
28 +extern time_t rrdeng_metric_oldest_time(RRDDIM *rd);
29 +extern void rrdeng_get_27_statistics(struct rrdengine_instance *ctx, unsigned long long *array);
30 +
31 +/* must call once before using anything */
32 +extern int rrdeng_init(struct rrdengine_instance **ctxp, char *dbfiles_path, unsigned page_cache_mb,
33 + unsigned disk_space_mb);
34 +
35 +extern int rrdeng_exit(struct rrdengine_instance *ctx);
36 +
37 +#endif /* NETDATA_RRDENGINEAPI_H */
\ No newline at end of file
database/engine/rrdenginelib.c new
+116
@@ -0,0 +1,116 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +#include "rrdengine.h"
3 +
4 +void print_page_cache_descr(struct rrdeng_page_cache_descr *page_cache_descr)
5 +{
6 + char uuid_str[37];
7 + char str[512];
8 + int pos = 0;
9 +
10 + uuid_unparse_lower(*page_cache_descr->id, uuid_str);
11 + pos += snprintfz(str, 512 - pos, "page(%p) id=%s\n"
12 + "--->len:%"PRIu32" time:%"PRIu64"->%"PRIu64" xt_offset:",
13 + page_cache_descr->page, uuid_str,
14 + page_cache_descr->page_length,
15 + (uint64_t)page_cache_descr->start_time,
16 + (uint64_t)page_cache_descr->end_time);
17 + if (!page_cache_descr->extent) {
18 + pos += snprintfz(str + pos, 512 - pos, "N/A");
19 + } else {
20 + pos += snprintfz(str + pos, 512 - pos, "%"PRIu64, page_cache_descr->extent->offset);
21 + }
22 + snprintfz(str + pos, 512 - pos, " flags:0x%2.2lX refcnt:%u\n\n", page_cache_descr->flags, page_cache_descr->refcnt);
23 + fputs(str, stderr);
24 +}
25 +
26 +int check_file_properties(uv_file file, uint64_t *file_size, size_t min_size)
27 +{
28 + int ret;
29 + uv_fs_t req;
30 + uv_stat_t* s;
31 +
32 + ret = uv_fs_fstat(NULL, &req, file, NULL);
33 + if (ret < 0) {
34 + fatal("uv_fs_fstat: %s\n", uv_strerror(ret));
35 + }
36 + assert(req.result == 0);
37 + s = req.ptr;
38 + if (!(s->st_mode & S_IFREG)) {
39 + error("Not a regular file.\n");
40 + uv_fs_req_cleanup(&req);
41 + return UV_EINVAL;
42 + }
43 + if (s->st_size < min_size) {
44 + error("File length is too short.\n");
45 + uv_fs_req_cleanup(&req);
46 + return UV_EINVAL;
47 + }
48 + *file_size = s->st_size;
49 + uv_fs_req_cleanup(&req);
50 +
51 + return 0;
52 +}
53 +
54 +char *get_rrdeng_statistics(struct rrdengine_instance *ctx, char *str, size_t size)
55 +{
56 + struct page_cache *pg_cache;
57 +
58 + pg_cache = &ctx->pg_cache;
59 + snprintfz(str, size,
60 + "metric_API_producers: %ld\n"
61 + "metric_API_consumers: %ld\n"
62 + "page_cache_total_pages: %ld\n"
63 + "page_cache_populated_pages: %ld\n"
64 + "page_cache_commited_pages: %ld\n"
65 + "page_cache_insertions: %ld\n"
66 + "page_cache_deletions: %ld\n"
67 + "page_cache_hits: %ld\n"
68 + "page_cache_misses: %ld\n"
69 + "page_cache_backfills: %ld\n"
70 + "page_cache_evictions: %ld\n"
71 + "compress_before_bytes: %ld\n"
72 + "compress_after_bytes: %ld\n"
73 + "decompress_before_bytes: %ld\n"
74 + "decompress_after_bytes: %ld\n"
75 + "io_write_bytes: %ld\n"
76 + "io_write_requests: %ld\n"
77 + "io_read_bytes: %ld\n"
78 + "io_read_requests: %ld\n"
79 + "io_write_extent_bytes: %ld\n"
80 + "io_write_extents: %ld\n"
81 + "io_read_extent_bytes: %ld\n"
82 + "io_read_extents: %ld\n"
83 + "datafile_creations: %ld\n"
84 + "datafile_deletions: %ld\n"
85 + "journalfile_creations: %ld\n"
86 + "journalfile_deletions: %ld\n",
87 + (long)ctx->stats.metric_API_producers,
88 + (long)ctx->stats.metric_API_consumers,
89 + (long)pg_cache->page_descriptors,
90 + (long)pg_cache->populated_pages,
91 + (long)pg_cache->commited_page_index.nr_commited_pages,
92 + (long)ctx->stats.pg_cache_insertions,
93 + (long)ctx->stats.pg_cache_deletions,
94 + (long)ctx->stats.pg_cache_hits,
95 + (long)ctx->stats.pg_cache_misses,
96 + (long)ctx->stats.pg_cache_backfills,
97 + (long)ctx->stats.pg_cache_evictions,
98 + (long)ctx->stats.before_compress_bytes,
99 + (long)ctx->stats.after_compress_bytes,
100 + (long)ctx->stats.before_decompress_bytes,
101 + (long)ctx->stats.after_decompress_bytes,
102 + (long)ctx->stats.io_write_bytes,
103 + (long)ctx->stats.io_write_requests,
104 + (long)ctx->stats.io_read_bytes,
105 + (long)ctx->stats.io_read_requests,
106 + (long)ctx->stats.io_write_extent_bytes,
107 + (long)ctx->stats.io_write_extents,
108 + (long)ctx->stats.io_read_extent_bytes,
109 + (long)ctx->stats.io_read_extents,
110 + (long)ctx->stats.datafile_creations,
111 + (long)ctx->stats.datafile_deletions,
112 + (long)ctx->stats.journalfile_creations,
113 + (long)ctx->stats.journalfile_deletions
114 + );
115 + return str;
116 +}
database/engine/rrdenginelib.h new
+84
@@ -0,0 +1,84 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#ifndef NETDATA_RRDENGINELIB_H
4 +#define NETDATA_RRDENGINELIB_H
5 +
6 +#include "rrdengine.h"
7 +
8 +/* Forward declarations */
9 +struct rrdeng_page_cache_descr;
10 +
11 +#define STR_HELPER(x) #x
12 +#define STR(x) STR_HELPER(x)
13 +
14 +/* Taken from linux kernel */
15 +#define BUILD_BUG_ON(condition) ((void)sizeof(char[1 - 2*!!(condition)]))
16 +
17 +#define ALIGN_BYTES_FLOOR(x) (((x) / RRDENG_BLOCK_SIZE) * RRDENG_BLOCK_SIZE)
18 +#define ALIGN_BYTES_CEILING(x) ((((x) + RRDENG_BLOCK_SIZE - 1) / RRDENG_BLOCK_SIZE) * RRDENG_BLOCK_SIZE)
19 +
20 +typedef uintptr_t rrdeng_stats_t;
21 +
22 +#ifdef __ATOMIC_RELAXED
23 +#define rrd_stat_atomic_add(p, n) do {(void) __atomic_fetch_add(p, n, __ATOMIC_RELAXED);} while(0)
24 +#else
25 +#define rrd_stat_atomic_add(p, n) do {(void) __sync_fetch_and_add(p, n);} while(0)
26 +#endif
27 +
28 +#ifndef O_DIRECT
29 +/* Workaround for OS X */
30 +#define O_DIRECT (0)
31 +#endif
32 +
33 +struct completion {
34 + uv_mutex_t mutex;
35 + uv_cond_t cond;
36 + volatile unsigned completed;
37 +};
38 +
39 +static inline void init_completion(struct completion *p)
40 +{
41 + p->completed = 0;
42 + assert(0 == uv_cond_init(&p->cond));
43 + assert(0 == uv_mutex_init(&p->mutex));
44 +}
45 +
46 +static inline void destroy_completion(struct completion *p)
47 +{
48 + uv_cond_destroy(&p->cond);
49 + uv_mutex_destroy(&p->mutex);
50 +}
51 +
52 +static inline void wait_for_completion(struct completion *p)
53 +{
54 + uv_mutex_lock(&p->mutex);
55 + while (0 == p->completed) {
56 + uv_cond_wait(&p->cond, &p->mutex);
57 + }
58 + assert(1 == p->completed);
59 + uv_mutex_unlock(&p->mutex);
60 +}
61 +
62 +static inline void complete(struct completion *p)
63 +{
64 + uv_mutex_lock(&p->mutex);
65 + p->completed = 1;
66 + uv_mutex_unlock(&p->mutex);
67 + uv_cond_broadcast(&p->cond);
68 +}
69 +
70 +static inline int crc32cmp(void *crcp, uLong crc)
71 +{
72 + return (*(uint32_t *)crcp != crc);
73 +}
74 +
75 +static inline void crc32set(void *crcp, uLong crc)
76 +{
77 + *(uint32_t *)crcp = crc;
78 +}
79 +
80 +extern void print_page_cache_descr(struct rrdeng_page_cache_descr *page_cache_descr);
81 +extern int check_file_properties(uv_file file, uint64_t *file_size, size_t min_size);
82 +extern char *get_rrdeng_statistics(struct rrdengine_instance *ctx, char *str, size_t size);
83 +
84 +#endif /* NETDATA_RRDENGINELIB_H */
\ No newline at end of file
database/rrd.c
+8 -1
@@ -38,6 +38,9 @@ inline const char *rrd_memory_mode_name(RRD_MEMORY_MODE id) {
38
39 case RRD_MEMORY_MODE_ALLOC:
40 return RRD_MEMORY_MODE_ALLOC_NAME;
41 +
42 + case RRD_MEMORY_MODE_DBENGINE:
43 + return RRD_MEMORY_MODE_DBENGINE_NAME;
44 }
45
46 return RRD_MEMORY_MODE_SAVE_NAME;
@@ -56,6 +59,9 @@ RRD_MEMORY_MODE rrd_memory_mode_id(const char *name) {
59 else if(unlikely(!strcmp(name, RRD_MEMORY_MODE_ALLOC_NAME)))
60 return RRD_MEMORY_MODE_ALLOC;
61
62 + else if(unlikely(!strcmp(name, RRD_MEMORY_MODE_DBENGINE_NAME)))
63 + return RRD_MEMORY_MODE_DBENGINE;
64 +
65 return RRD_MEMORY_MODE_SAVE;
66 }
67
@@ -140,7 +146,8 @@ char *rrdset_cache_dir(RRDHOST *host, const char *id, const char *config_section
146 snprintfz(n, FILENAME_MAX, "%s/%s", host->cache_dir, b);
147 ret = config_get(config_section, "cache directory", n);
148
143 - if(host->rrd_memory_mode == RRD_MEMORY_MODE_MAP || host->rrd_memory_mode == RRD_MEMORY_MODE_SAVE) {
149 + if(host->rrd_memory_mode == RRD_MEMORY_MODE_MAP || host->rrd_memory_mode == RRD_MEMORY_MODE_SAVE ||
150 + host->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
151 int r = mkdir(ret, 0775);
152 if(r != 0 && errno != EEXIST)
153 error("Cannot create directory '%s'", ret);
database/rrd.h
+144 -4
@@ -14,6 +14,14 @@ typedef struct rrdcalc RRDCALC;
14 typedef struct rrdcalctemplate RRDCALCTEMPLATE;
15 typedef struct alarm_entry ALARM_ENTRY;
16
17 +// forward declarations
18 +struct rrddim_volatile;
19 +#ifdef ENABLE_DBENGINE
20 +struct rrdeng_page_cache_descr;
21 +struct rrdengine_instance;
22 +struct pg_cache_page_index;
23 +#endif
24 +
25 #include "../daemon/common.h"
26 #include "web/api/queries/query.h"
27 #include "rrdvar.h"
@@ -66,7 +74,8 @@ typedef enum rrd_memory_mode {
74 RRD_MEMORY_MODE_RAM = 1,
75 RRD_MEMORY_MODE_MAP = 2,
76 RRD_MEMORY_MODE_SAVE = 3,
69 - RRD_MEMORY_MODE_ALLOC = 4
77 + RRD_MEMORY_MODE_ALLOC = 4,
78 + RRD_MEMORY_MODE_DBENGINE = 5
79 } RRD_MEMORY_MODE;
80
81 #define RRD_MEMORY_MODE_NONE_NAME "none"
@@ -74,6 +83,7 @@ typedef enum rrd_memory_mode {
83 #define RRD_MEMORY_MODE_MAP_NAME "map"
84 #define RRD_MEMORY_MODE_SAVE_NAME "save"
85 #define RRD_MEMORY_MODE_ALLOC_NAME "alloc"
86 +#define RRD_MEMORY_MODE_DBENGINE_NAME "dbengine"
87
88 extern RRD_MEMORY_MODE default_rrd_memory_mode;
89
@@ -178,7 +188,8 @@ struct rrddim {
188 char *cache_filename; // the filename we load/save from/to this set
189
190 size_t collections_counter; // the number of times we added values to this rrdim
181 - size_t unused[9];
191 + struct rrddim_volatile *state; // volatile state that is not persistently stored
192 + size_t unused[8];
193
194 collected_number collected_value_max; // the absolute maximum of the collected value
195
@@ -226,6 +237,90 @@ struct rrddim {
237 storage_number values[]; // the array of values - THIS HAS TO BE THE LAST MEMBER
238 };
239
240 +// ----------------------------------------------------------------------------
241 +// iterator state for RRD dimension data collection
242 +union rrddim_collect_handle {
243 + struct {
244 + long slot;
245 + long entries;
246 + } slotted; // state the legacy code uses
247 +#ifdef ENABLE_DBENGINE
248 + struct rrdeng_collect_handle {
249 + struct rrdeng_page_cache_descr *descr, *prev_descr;
250 + unsigned long page_correlation_id;
251 + struct rrdengine_instance *ctx;
252 + struct pg_cache_page_index *page_index;
253 + } rrdeng; // state the database engine uses
254 +#endif
255 +};
256 +
257 +// ----------------------------------------------------------------------------
258 +// iterator state for RRD dimension data queries
259 +struct rrddim_query_handle {
260 + RRDDIM *rd;
261 + time_t start_time;
262 + time_t end_time;
263 + union {
264 + struct {
265 + long slot;
266 + long last_slot;
267 + uint8_t finished;
268 + } slotted; // state the legacy code uses
269 +#ifdef ENABLE_DBENGINE
270 + struct rrdeng_query_handle {
271 + struct rrdeng_page_cache_descr *descr;
272 + struct rrdengine_instance *ctx;
273 + struct pg_cache_page_index *page_index;
274 + time_t now; //TODO: remove now to implement next point iteration
275 + time_t dt; //TODO: remove dt to implement next point iteration
276 + } rrdeng; // state the database engine uses
277 +#endif
278 + };
279 +};
280 +
281 +
282 +// ----------------------------------------------------------------------------
283 +// volatile state per RRD dimension
284 +struct rrddim_volatile {
285 +#ifdef ENABLE_DBENGINE
286 + uuid_t *rrdeng_uuid; // database engine metric UUID
287 +#endif
288 + union rrddim_collect_handle handle;
289 + // ------------------------------------------------------------------------
290 + // function pointers that handle data collection
291 + struct rrddim_collect_ops {
292 + // an initialization function to run before starting collection
293 + void (*init)(RRDDIM *rd);
294 +
295 + // run this to store each metric into the database
296 + void (*store_metric)(RRDDIM *rd, usec_t point_in_time, storage_number number);
297 +
298 + // an finalization function to run after collection is over
299 + void (*finalize)(RRDDIM *rd);
300 + } collect_ops;
301 +
302 + // function pointers that handle database queries
303 + struct rrddim_query_ops {
304 + // run this before starting a series of next_metric() database queries
305 + void (*init)(RRDDIM *rd, struct rrddim_query_handle *handle, time_t start_time, time_t end_time);
306 +
307 + // run this to load each metric number from the database
308 + storage_number (*next_metric)(struct rrddim_query_handle *handle);
309 +
310 + // run this to test if the series of next_metric() database queries is finished
311 + int (*is_finished)(struct rrddim_query_handle *handle);
312 +
313 + // run this after finishing a series of load_metric() database queries
314 + void (*finalize)(struct rrddim_query_handle *handle);
315 +
316 + // get the timestamp of the last entry of this metric
317 + time_t (*latest_time)(RRDDIM *rd);
318 +
319 + // get the timestamp of the first entry of this metric
320 + time_t (*oldest_time)(RRDDIM *rd);
321 + } query_ops;
322 +};
323 +
324 // ----------------------------------------------------------------------------
325 // these loop macros make sure the linked list is accessed with the right lock
326
@@ -528,6 +623,10 @@ struct rrdhost {
623
624 int rrd_update_every; // the update frequency of the host
625 long rrd_history_entries; // the number of history entries for the host's charts
626 +#ifdef ENABLE_DBENGINE
627 + unsigned page_cache_mb; // Database Engine page cache size in MiB
628 + unsigned disk_space_mb; // Database Engine disk space quota in MiB
629 +#endif
630 RRD_MEMORY_MODE rrd_memory_mode; // the memory more for the charts of this host
631
632 char *cache_dir; // the directory to save RRD cache files
@@ -620,6 +719,10 @@ struct rrdhost {
719 avl_tree_lock rrdfamily_root_index; // the host's chart families index
720 avl_tree_lock rrdvar_root_index; // the host's chart variables index
721
722 +#ifdef ENABLE_DBENGINE
723 + struct rrdengine_instance *rrdeng_ctx; // DB engine instance for this host
724 +#endif
725 +
726 struct rrdhost *next;
727 };
728 extern RRDHOST *localhost;
@@ -771,10 +874,41 @@ extern void rrdset_isnot_obsolete(RRDSET *st);
874 #define rrdset_duration(st) ((time_t)( (((st)->counter >= ((unsigned long)(st)->entries))?(unsigned long)(st)->entries:(st)->counter) * (st)->update_every ))
875
876 // get the timestamp of the last entry in the round robin database
774 -#define rrdset_last_entry_t(st) ((time_t)(((st)->last_updated.tv_sec)))
877 +static inline time_t rrdset_last_entry_t(RRDSET *st) {
878 + if (st->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
879 + RRDDIM *rd;
880 + time_t last_entry_t = 0;
881 +
882 + int ret = netdata_rwlock_tryrdlock(&st->rrdset_rwlock);
883 + rrddim_foreach_read(rd, st) {
884 + last_entry_t = MAX(last_entry_t, rd->state->query_ops.latest_time(rd));
885 + }
886 + if(0 == ret) netdata_rwlock_unlock(&st->rrdset_rwlock);
887 +
888 + return last_entry_t;
889 + } else {
890 + return (time_t)st->last_updated.tv_sec;
891 + }
892 +}
893
894 // get the timestamp of first entry in the round robin database
777 -#define rrdset_first_entry_t(st) ((time_t)(rrdset_last_entry_t(st) - rrdset_duration(st)))
895 +static inline time_t rrdset_first_entry_t(RRDSET *st) {
896 + if (st->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
897 + RRDDIM *rd;
898 + time_t first_entry_t = LONG_MAX;
899 +
900 + int ret = netdata_rwlock_tryrdlock(&st->rrdset_rwlock);
901 + rrddim_foreach_read(rd, st) {
902 + first_entry_t = MIN(first_entry_t, rd->state->query_ops.oldest_time(rd));
903 + }
904 + if(0 == ret) netdata_rwlock_unlock(&st->rrdset_rwlock);
905 +
906 + if (unlikely(LONG_MAX == first_entry_t)) return 0;
907 + return first_entry_t;
908 + } else {
909 + return (time_t)(rrdset_last_entry_t(st) - rrdset_duration(st));
910 + }
911 +}
912
913 // get the last slot updated in the round robin database
914 #define rrdset_last_slot(st) ((size_t)(((st)->current_entry == 0) ? (st)->entries - 1 : (st)->current_entry - 1))
@@ -914,5 +1048,11 @@ extern void rrdhost_cleanup_obsolete_charts(RRDHOST *host);
1048
1049 #endif /* NETDATA_RRD_INTERNALS */
1050
1051 +// ----------------------------------------------------------------------------
1052 +// RRD DB engine declarations
1053 +
1054 +#ifdef ENABLE_DBENGINE
1055 +#include "database/engine/rrdengineapi.h"
1056 +#endif
1057
1058 #endif /* NETDATA_RRD_H */
database/rrddim.c
+96 -5
@@ -89,6 +89,69 @@ inline int rrddim_set_divisor(RRDSET *st, RRDDIM *rd, collected_number divisor)
89 return 1;
90 }
91
92 +// ----------------------------------------------------------------------------
93 +// RRDDIM legacy data collection functions
94 +
95 +static void rrddim_collect_init(RRDDIM *rd) {
96 + rd->values[rd->rrdset->current_entry] = SN_EMPTY_SLOT; // pack_storage_number(0, SN_NOT_EXISTS);
97 +}
98 +static void rrddim_collect_store_metric(RRDDIM *rd, usec_t point_in_time, storage_number number) {
99 + (void)point_in_time;
100 +
101 + rd->values[rd->rrdset->current_entry] = number;
102 +}
103 +static void rrddim_collect_finalize(RRDDIM *rd) {
104 + (void)rd;
105 +
106 + return;
107 +}
108 +
109 +// ----------------------------------------------------------------------------
110 +// RRDDIM legacy database query functions
111 +
112 +static void rrddim_query_init(RRDDIM *rd, struct rrddim_query_handle *handle, time_t start_time, time_t end_time) {
113 + handle->rd = rd;
114 + handle->start_time = start_time;
115 + handle->end_time = end_time;
116 + handle->slotted.slot = rrdset_time2slot(rd->rrdset, start_time);
117 + handle->slotted.last_slot = rrdset_time2slot(rd->rrdset, end_time);
118 + handle->slotted.finished = 0;
119 +}
120 +
121 +static storage_number rrddim_query_next_metric(struct rrddim_query_handle *handle) {
122 + RRDDIM *rd = handle->rd;
123 + long entries = rd->rrdset->entries;
124 + long slot = handle->slotted.slot;
125 +
126 + if (unlikely(handle->slotted.slot == handle->slotted.last_slot))
127 + handle->slotted.finished = 1;
128 + storage_number n = rd->values[slot++];
129 +
130 + if(unlikely(slot >= entries)) slot = 0;
131 + handle->slotted.slot = slot;
132 +
133 + return n;
134 +}
135 +
136 +static int rrddim_query_is_finished(struct rrddim_query_handle *handle) {
137 + return handle->slotted.finished;
138 +}
139 +
140 +static void rrddim_query_finalize(struct rrddim_query_handle *handle) {
141 + (void)handle;
142 +
143 + return;
144 +}
145 +
146 +static time_t rrddim_query_latest_time(RRDDIM *rd) {
147 + return rrdset_last_entry_t(rd->rrdset);
148 +}
149 +
150 +static time_t rrddim_query_oldest_time(RRDDIM *rd) {
151 + return rrdset_first_entry_t(rd->rrdset);
152 +}
153 +
154 +
155 // ----------------------------------------------------------------------------
156 // RRDDIM create a dimension
157
@@ -123,9 +186,10 @@ RRDDIM *rrddim_add_custom(RRDSET *st, const char *id, const char *name, collecte
186 rrdset_strncpyz_name(filename, id, FILENAME_MAX);
187 snprintfz(fullfilename, FILENAME_MAX, "%s/%s.db", st->cache_dir, filename);
188
126 - if(memory_mode == RRD_MEMORY_MODE_SAVE || memory_mode == RRD_MEMORY_MODE_MAP || memory_mode == RRD_MEMORY_MODE_RAM) {
189 + if(memory_mode == RRD_MEMORY_MODE_SAVE || memory_mode == RRD_MEMORY_MODE_MAP ||
190 + memory_mode == RRD_MEMORY_MODE_RAM || memory_mode == RRD_MEMORY_MODE_DBENGINE) {
191 rd = (RRDDIM *)mymmap(
128 - (memory_mode == RRD_MEMORY_MODE_RAM)?NULL:fullfilename
192 + (memory_mode == RRD_MEMORY_MODE_RAM || memory_mode == RRD_MEMORY_MODE_DBENGINE)?NULL:fullfilename
193 , size
194 , ((memory_mode == RRD_MEMORY_MODE_MAP) ? MAP_SHARED : MAP_PRIVATE)
195 , 1
@@ -146,7 +210,7 @@ RRDDIM *rrddim_add_custom(RRDSET *st, const char *id, const char *name, collecte
210 struct timeval now;
211 now_realtime_timeval(&now);
212
149 - if(memory_mode == RRD_MEMORY_MODE_RAM) {
213 + if(memory_mode == RRD_MEMORY_MODE_RAM || memory_mode == RRD_MEMORY_MODE_DBENGINE) {
214 memset(rd, 0, size);
215 }
216 else {
@@ -243,11 +307,34 @@ RRDDIM *rrddim_add_custom(RRDSET *st, const char *id, const char *name, collecte
307 rd->collected_volume = 0;
308 rd->stored_volume = 0;
309 rd->last_stored_value = 0;
246 - rd->values[st->current_entry] = SN_EMPTY_SLOT; // pack_storage_number(0, SN_NOT_EXISTS);
310 rd->last_collected_time.tv_sec = 0;
311 rd->last_collected_time.tv_usec = 0;
312 rd->rrdset = st;
250 -
313 + rd->state = mallocz(sizeof(*rd->state));
314 + if(memory_mode == RRD_MEMORY_MODE_DBENGINE) {
315 +#ifdef ENABLE_DBENGINE
316 + rd->state->collect_ops.init = rrdeng_store_metric_init;
317 + rd->state->collect_ops.store_metric = rrdeng_store_metric_next;
318 + rd->state->collect_ops.finalize = rrdeng_store_metric_finalize;
319 + rd->state->query_ops.init = rrdeng_load_metric_init;
320 + rd->state->query_ops.next_metric = rrdeng_load_metric_next;
321 + rd->state->query_ops.is_finished = rrdeng_load_metric_is_finished;
322 + rd->state->query_ops.finalize = rrdeng_load_metric_finalize;
323 + rd->state->query_ops.latest_time = rrdeng_metric_latest_time;
324 + rd->state->query_ops.oldest_time = rrdeng_metric_oldest_time;
325 +#endif
326 + } else {
327 + rd->state->collect_ops.init = rrddim_collect_init;
328 + rd->state->collect_ops.store_metric = rrddim_collect_store_metric;
329 + rd->state->collect_ops.finalize = rrddim_collect_finalize;
330 + rd->state->query_ops.init = rrddim_query_init;
331 + rd->state->query_ops.next_metric = rrddim_query_next_metric;
332 + rd->state->query_ops.is_finished = rrddim_query_is_finished;
333 + rd->state->query_ops.finalize = rrddim_query_finalize;
334 + rd->state->query_ops.latest_time = rrddim_query_latest_time;
335 + rd->state->query_ops.oldest_time = rrddim_query_oldest_time;
336 + }
337 + rd->state->collect_ops.init(rd);
338 // append this dimension
339 if(!st->dimensions)
340 st->dimensions = rd;
@@ -294,6 +381,9 @@ void rrddim_free(RRDSET *st, RRDDIM *rd)
381 {
382 debug(D_RRD_CALLS, "rrddim_free() %s.%s", st->name, rd->name);
383
384 + rd->state->collect_ops.finalize(rd);
385 + freez(rd->state);
386 +
387 if(rd == st->dimensions)
388 st->dimensions = rd->next;
389 else {
@@ -319,6 +409,7 @@ void rrddim_free(RRDSET *st, RRDDIM *rd)
409 case RRD_MEMORY_MODE_SAVE:
410 case RRD_MEMORY_MODE_MAP:
411 case RRD_MEMORY_MODE_RAM:
412 + case RRD_MEMORY_MODE_DBENGINE:
413 debug(D_RRD_CALLS, "Unmapping dimension '%s'.", rd->name);
414 freez((void *)rd->id);
415 freez(rd->cache_filename);
database/rrdhost.c
+36 -1
@@ -134,6 +134,10 @@ RRDHOST *rrdhost_create(const char *hostname,
134 host->rrd_update_every = (update_every > 0)?update_every:1;
135 host->rrd_history_entries = align_entries_to_pagesize(memory_mode, entries);
136 host->rrd_memory_mode = memory_mode;
137 +#ifdef ENABLE_DBENGINE
138 + host->page_cache_mb = default_rrdeng_page_cache_mb;
139 + host->disk_space_mb = default_rrdeng_disk_quota_mb;
140 +#endif
141 host->health_enabled = (memory_mode == RRD_MEMORY_MODE_NONE)? 0 : health_enabled;
142 host->rrdpush_send_enabled = (rrdpush_enabled && rrdpush_destination && *rrdpush_destination && rrdpush_api_key && *rrdpush_api_key) ? 1 : 0;
143 host->rrdpush_send_destination = (host->rrdpush_send_enabled)?strdupz(rrdpush_destination):NULL;
@@ -205,7 +209,8 @@ RRDHOST *rrdhost_create(const char *hostname,
209 snprintfz(filename, FILENAME_MAX, "%s/%s", netdata_configured_cache_dir, host->machine_guid);
210 host->cache_dir = strdupz(filename);
211
208 - if(host->rrd_memory_mode == RRD_MEMORY_MODE_MAP || host->rrd_memory_mode == RRD_MEMORY_MODE_SAVE) {
212 + if(host->rrd_memory_mode == RRD_MEMORY_MODE_MAP || host->rrd_memory_mode == RRD_MEMORY_MODE_SAVE ||
213 + host->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
214 int r = mkdir(host->cache_dir, 0775);
215 if(r != 0 && errno != EEXIST)
216 error("Host '%s': cannot create directory '%s'", host->hostname, host->cache_dir);
@@ -221,6 +226,30 @@ RRDHOST *rrdhost_create(const char *hostname,
226 }
227
228 }
229 + if (host->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
230 +#ifdef ENABLE_DBENGINE
231 + char dbenginepath[FILENAME_MAX + 1];
232 + int ret;
233 +
234 + snprintfz(dbenginepath, FILENAME_MAX, "%s/dbengine", host->cache_dir);
235 + ret = mkdir(dbenginepath, 0775);
236 + if(ret != 0 && errno != EEXIST)
237 + error("Host '%s': cannot create directory '%s'", host->hostname, dbenginepath);
238 + else
239 + ret = rrdeng_init(&host->rrdeng_ctx, dbenginepath, host->page_cache_mb, host->disk_space_mb);
240 + if(ret) {
241 + error("Host '%s': cannot initialize host with machine guid '%s'. Failed to initialize DB engine at '%s'.",
242 + host->hostname, host->machine_guid, host->cache_dir);
243 + rrdhost_free(host);
244 + host = NULL;
245 + //rrd_hosts_available++; //TODO: maybe we want this?
246 +
247 + return host;
248 + }
249 +#else
250 + fatal("RRD_MEMORY_MODE_DBENGINE is not supported in this platform.");
251 +#endif
252 + }
253
254 if(host->health_enabled) {
255 snprintfz(filename, FILENAME_MAX, "%s/health", host->varlib_dir);
@@ -569,6 +598,12 @@ void rrdhost_free(RRDHOST *host) {
598
599 health_alarm_log_free(host);
600
601 + if (host->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
602 +#ifdef ENABLE_DBENGINE
603 + rrdeng_exit(host->rrdeng_ctx);
604 +#endif
605 + }
606 +
607 // ------------------------------------------------------------------------
608 // remove it from the indexes
609
database/rrdset.c
+37 -12
@@ -363,6 +363,7 @@ void rrdset_free(RRDSET *st) {
363 case RRD_MEMORY_MODE_SAVE:
364 case RRD_MEMORY_MODE_MAP:
365 case RRD_MEMORY_MODE_RAM:
366 + case RRD_MEMORY_MODE_DBENGINE:
367 debug(D_RRD_CALLS, "Unmapping stats '%s'.", st->name);
368 munmap(st, st->memsize);
369 break;
@@ -541,6 +542,9 @@ RRDSET *rrdset_create_custom(
542 int enabled = config_get_boolean(config_section, "enabled", 1);
543 if(!enabled) entries = 5;
544
545 + if(memory_mode == RRD_MEMORY_MODE_DBENGINE)
546 + entries = config_set_number(config_section, "history", 5);
547 +
548 unsigned long size = sizeof(RRDSET);
549 char *cache_dir = rrdset_cache_dir(host, fullid, config_section);
550
@@ -552,9 +556,10 @@ RRDSET *rrdset_create_custom(
556 debug(D_RRD_CALLS, "Creating RRD_STATS for '%s.%s'.", type, id);
557
558 snprintfz(fullfilename, FILENAME_MAX, "%s/main.db", cache_dir);
555 - if(memory_mode == RRD_MEMORY_MODE_SAVE || memory_mode == RRD_MEMORY_MODE_MAP || memory_mode == RRD_MEMORY_MODE_RAM) {
559 + if(memory_mode == RRD_MEMORY_MODE_SAVE || memory_mode == RRD_MEMORY_MODE_MAP ||
560 + memory_mode == RRD_MEMORY_MODE_RAM || memory_mode == RRD_MEMORY_MODE_DBENGINE) {
561 st = (RRDSET *) mymmap(
557 - (memory_mode == RRD_MEMORY_MODE_RAM)?NULL:fullfilename
562 + (memory_mode == RRD_MEMORY_MODE_RAM || memory_mode == RRD_MEMORY_MODE_DBENGINE)?NULL:fullfilename
563 , size
564 , ((memory_mode == RRD_MEMORY_MODE_MAP) ? MAP_SHARED : MAP_PRIVATE)
565 , 0
@@ -585,7 +590,7 @@ RRDSET *rrdset_create_custom(
590 st->alarms = NULL;
591 st->flags = 0x00000000;
592
588 - if(memory_mode == RRD_MEMORY_MODE_RAM) {
593 + if(memory_mode == RRD_MEMORY_MODE_RAM || memory_mode == RRD_MEMORY_MODE_DBENGINE) {
594 memset(st, 0, size);
595 }
596 else {
@@ -631,7 +636,10 @@ RRDSET *rrdset_create_custom(
636
637 if(unlikely(!st)) {
638 st = callocz(1, size);
634 - st->rrd_memory_mode = (memory_mode == RRD_MEMORY_MODE_NONE) ? RRD_MEMORY_MODE_NONE : RRD_MEMORY_MODE_ALLOC;
639 + if (memory_mode == RRD_MEMORY_MODE_DBENGINE)
640 + st->rrd_memory_mode = RRD_MEMORY_MODE_DBENGINE;
641 + else
642 + st->rrd_memory_mode = (memory_mode == RRD_MEMORY_MODE_NONE) ? RRD_MEMORY_MODE_NONE : RRD_MEMORY_MODE_ALLOC;
643 }
644
645 st->plugin_name = plugin?strdupz(plugin):NULL;
@@ -1052,12 +1060,14 @@ static inline size_t rrdset_done_interpolate(
1060 }
1061
1062 if(unlikely(!store_this_entry)) {
1055 - rd->values[current_entry] = SN_EMPTY_SLOT; //pack_storage_number(0, SN_NOT_EXISTS);
1063 + rd->state->collect_ops.store_metric(rd, next_store_ut, SN_EMPTY_SLOT); //pack_storage_number(0, SN_NOT_EXISTS)
1064 +// rd->values[current_entry] = SN_EMPTY_SLOT; //pack_storage_number(0, SN_NOT_EXISTS);
1065 continue;
1066 }
1067
1068 if(likely(rd->updated && rd->collections_counter > 1 && iterations < st->gap_when_lost_iterations_above)) {
1060 - rd->values[current_entry] = pack_storage_number(new_value, storage_flags );
1069 + rd->state->collect_ops.store_metric(rd, next_store_ut, pack_storage_number(new_value, storage_flags));
1070 +// rd->values[current_entry] = pack_storage_number(new_value, storage_flags );
1071 rd->last_stored_value = new_value;
1072
1073 #ifdef NETDATA_INTERNAL_CHECKS
@@ -1079,7 +1089,8 @@ static inline size_t rrdset_done_interpolate(
1089 );
1090 #endif
1091
1082 - rd->values[current_entry] = SN_EMPTY_SLOT; // pack_storage_number(0, SN_NOT_EXISTS);
1092 +// rd->values[current_entry] = SN_EMPTY_SLOT; // pack_storage_number(0, SN_NOT_EXISTS);
1093 + rd->state->collect_ops.store_metric(rd, next_store_ut, SN_EMPTY_SLOT); //pack_storage_number(0, SN_NOT_EXISTS)
1094 rd->last_stored_value = NAN;
1095 }
1096
@@ -1119,11 +1130,16 @@ static inline size_t rrdset_done_interpolate(
1130 // reset the storage flags for the next point, if any;
1131 storage_flags = SN_EXISTS;
1132
1122 - counter++;
1123 - current_entry = ((current_entry + 1) >= st->entries) ? 0 : current_entry + 1;
1133 + st->counter = ++counter;
1134 + st->current_entry = current_entry = ((current_entry + 1) >= st->entries) ? 0 : current_entry + 1;
1135 +
1136 + st->last_updated.tv_sec = (time_t) (last_ut / USEC_PER_SEC);
1137 + st->last_updated.tv_usec = 0;
1138 +
1139 last_stored_ut = next_store_ut;
1140 }
1141
1142 +/*
1143 st->counter = counter;
1144 st->current_entry = current_entry;
1145
@@ -1131,6 +1147,7 @@ static inline size_t rrdset_done_interpolate(
1147 st->last_updated.tv_sec = (time_t) (last_ut / USEC_PER_SEC);
1148 st->last_updated.tv_usec = 0;
1149 }
1150 +*/
1151
1152 return stored_entries;
1153 }
@@ -1201,7 +1218,8 @@ void rrdset_done(RRDSET *st) {
1218 }
1219
1220 // check if the chart has a long time to be updated
1204 - if(unlikely(st->usec_since_last_update > st->entries * update_every_ut)) {
1221 + if(unlikely(st->usec_since_last_update > st->entries * update_every_ut &&
1222 + st->rrd_memory_mode != RRD_MEMORY_MODE_DBENGINE)) {
1223 info("host '%s', chart %s: took too long to be updated (counter #%zu, update #%zu, %0.3" LONG_DOUBLE_MODIFIER " secs). Resetting it.", st->rrdhost->hostname, st->name, st->counter, st->counter_done, (LONG_DOUBLE)st->usec_since_last_update / USEC_PER_SEC);
1224 rrdset_reset(st);
1225 st->usec_since_last_update = update_every_ut;
@@ -1242,7 +1260,8 @@ void rrdset_done(RRDSET *st) {
1260 }
1261
1262 // check if we will re-write the entire data set
1245 - if(unlikely(dt_usec(&st->last_collected_time, &st->last_updated) > st->entries * update_every_ut)) {
1263 + if(unlikely(dt_usec(&st->last_collected_time, &st->last_updated) > st->entries * update_every_ut &&
1264 + st->rrd_memory_mode != RRD_MEMORY_MODE_DBENGINE)) {
1265 info("%s: too old data (last updated at %ld.%ld, last collected at %ld.%ld). Resetting it. Will not store the next entry.", st->name, st->last_updated.tv_sec, st->last_updated.tv_usec, st->last_collected_time.tv_sec, st->last_collected_time.tv_usec);
1266 rrdset_reset(st);
1267 rrdset_init_last_updated_time(st);
@@ -1266,11 +1285,17 @@ void rrdset_done(RRDSET *st) {
1285 // if we have not collected metrics this session (st->counter_done == 0)
1286 // and we have collected metrics for this chart in the past (st->counter != 0)
1287 // fill the gap (the chart has been just loaded from disk)
1269 - if(unlikely(st->counter)) {
1288 + if(unlikely(st->counter) && st->rrd_memory_mode != RRD_MEMORY_MODE_DBENGINE) {
1289 rrdset_done_fill_the_gap(st);
1290 last_stored_ut = st->last_updated.tv_sec * USEC_PER_SEC + st->last_updated.tv_usec;
1291 next_store_ut = (st->last_updated.tv_sec + st->update_every) * USEC_PER_SEC;
1292 }
1293 + if (st->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
1294 + // set a fake last_updated to jump to current time
1295 + rrdset_init_last_updated_time(st);
1296 + last_stored_ut = st->last_updated.tv_sec * USEC_PER_SEC + st->last_updated.tv_usec;
1297 + next_store_ut = (st->last_updated.tv_sec + st->update_every) * USEC_PER_SEC;
1298 + }
1299
1300 if(unlikely(rrdset_flag_check(st, RRDSET_FLAG_STORE_FIRST))) {
1301 store_this_entry = 1;
libnetdata/libnetdata.h
+3
@@ -202,6 +202,9 @@
202 #endif
203 #define abs(x) (((x) < 0)? (-(x)) : (x))
204
205 +#define MIN(a,b) (((a)<(b))?(a):(b))
206 +#define MAX(a,b) (((a)>(b))?(a):(b))
207 +
208 #define GUID_LEN 36
209
210 extern void netdata_fix_chart_id(char *s);
libnetdata/log/log.h
+1
@@ -36,6 +36,7 @@
36 #define D_STATSD 0x0000000010000000
37 #define D_POLLFD 0x0000000020000000
38 #define D_STREAM 0x0000000040000000
39 +#define D_RRDENGINE 0x0000000100000000
40 #define D_SYSTEM 0x8000000000000000
41
42 //#define DEBUG (D_WEB_CLIENT_ACCESS|D_LISTENER|D_RRD_STATS)
packaging/installer/README.md
+14 -3
@@ -191,13 +191,13 @@ This is how to do it by hand:
191
192 ```sh
193 # Debian / Ubuntu
194 -apt-get install zlib1g-dev uuid-dev libmnl-dev gcc make git autoconf autoconf-archive autogen automake pkg-config curl
194 +apt-get install zlib1g-dev uuid-dev libuv1-dev liblz4-dev libjudy-dev libssl-dev libmnl-dev gcc make git autoconf autoconf-archive autogen automake pkg-config curl
195
196 # Fedora
197 -dnf install zlib-devel libuuid-devel libmnl-devel gcc make git autoconf autoconf-archive autogen automake pkgconfig curl findutils
197 +dnf install zlib-devel libuuid-devel libuv-devel lz4-devel Judy-devel openssl-devel libmnl-devel gcc make git autoconf autoconf-archive autogen automake pkgconfig curl findutils
198
199 # CentOS / Red Hat Enterprise Linux
200 -yum install autoconf automake curl gcc git libmnl-devel libuuid-devel lm_sensors make MySQL-python nc pkgconfig python python-psycopg2 PyYAML zlib-devel
200 +yum install autoconf automake curl gcc git libmnl-devel libuuid-devel openssl-devel libuv-devel lz4-devel Judy-devel lm_sensors make MySQL-python nc pkgconfig python python-psycopg2 PyYAML zlib-devel
201
202 ```
203
@@ -234,6 +234,17 @@ package|description
234
235 *Netdata will greatly benefit if you have the above packages installed, but it will still work without them.*
236
237 +Netdata DB engine can be enabled when these are installed (they are optional):
238 +
239 +|package|description|
240 +|:-----:|-----------|
241 +|`libuv`|multi-platform support library with a focus on asynchronous I/O|
242 +|`liblz4`|Extremely Fast Compression algorithm|
243 +|`Judy`|General purpose dynamic array|
244 +|`openssl`|Cryptography and SSL/TLS Toolkit|
245 +
246 +*Netdata will greatly benefit if you have the above packages installed, but it will still work without them.*
247 +
248 ---
249
250 ### Install Netdata
streaming/README.md
+5 -1
@@ -73,7 +73,7 @@ These are options that affect the operation of netdata in this area:
73
74 ```
75 [global]
76 - memory mode = none | ram | save | map
76 + memory mode = none | ram | save | map | dbengine
77 ```
78
79 `[global].memory mode = none` disables the database at this host. This also disables health
@@ -170,6 +170,10 @@ the unique id the netdata generating the metrics (i.e. the netdata that original
170 them `/var/lib/netdata/registry/netdata.unique.id`). So, metrics for netdata `A` that pass through
171 any number of other netdata, will have the same `MACHINE_GUID`.
172
173 +You can also use `default memory mode = dbengine` for an API key or `memory mode = dbengine` for
174 + a single host. The additional `page cache size` and `dbengine disk space` configuration options
175 + are inherited from the global netdata configuration.
176 +
177 ##### allow from
178
179 `allow from` settings are [netdata simple patterns](../libnetdata/simple_pattern): string matches
streaming/stream.conf
+6 -5
@@ -103,10 +103,11 @@
103 # You can also set it per host below.
104 # If you don't set it here, the memory mode of netdata.conf will be used.
105 # Valid modes:
106 - # save save on exit, load on start
107 - # map like swap (continuously syncing to disks - you need SSD)
108 - # ram keep it in RAM, don't touch the disk
109 - # none no database at all (use this on headless proxies)
106 + # save save on exit, load on start
107 + # map like swap (continuously syncing to disks - you need SSD)
108 + # ram keep it in RAM, don't touch the disk
109 + # none no database at all (use this on headless proxies)
110 + # dbengine like a traditional database
111 default memory mode = ram
112
113 # Shall we enable health monitoring for the hosts using this API key?
@@ -167,7 +168,7 @@
168 # The number of entries in the database
169 history = 3600
170
170 - # The memory mode of the database: save | map | ram | none
171 + # The memory mode of the database: save | map | ram | none | dbengine
172 memory mode = save
173
174 # Health / alarms control: yes | no | auto
web/api/formatters/json_wrapper.c
+7
@@ -96,12 +96,19 @@ void rrdr_json_wrapper_begin(RRDR *r, BUFFER *wb, uint32_t format, RRDR_OPTIONS
96 if(i) buffer_strcat(wb, ", ");
97 i++;
98
99 + calculated_number value = rd->last_stored_value;
100 + if (NAN == value)
101 + buffer_strcat(wb, "null");
102 + else
103 + buffer_rrd_value(wb, value);
104 + /*
105 storage_number n = rd->values[rrdset_last_slot(r->st)];
106
107 if(!does_storage_number_exist(n))
108 buffer_strcat(wb, "null");
109 else
110 buffer_rrd_value(wb, unpack_storage_number(n));
111 + */
112 }
113 if(!i) {
114 rows = 0;
web/api/formatters/rrdset2json.c
+6 -3
@@ -7,6 +7,9 @@
7 void rrdset2json(RRDSET *st, BUFFER *wb, size_t *dimensions_count, size_t *memory_used) {
8 rrdset_rdlock(st);
9
10 + time_t first_entry_t = rrdset_first_entry_t(st);
11 + time_t last_entry_t = rrdset_last_entry_t(st);
12 +
13 buffer_sprintf(wb,
14 "\t\t{\n"
15 "\t\t\t\"id\": \"%s\",\n"
@@ -40,9 +43,9 @@ void rrdset2json(RRDSET *st, BUFFER *wb, size_t *dimensions_count, size_t *memor
43 , st->units
44 , st->name
45 , rrdset_type_name(st->chart_type)
43 - , st->entries * st->update_every
44 - , rrdset_first_entry_t(st)
45 - , rrdset_last_entry_t(st)
46 + , last_entry_t - first_entry_t + st->update_every//st->entries * st->update_every
47 + , first_entry_t//rrdset_first_entry_t(st)
48 + , last_entry_t//rrdset_last_entry_t(st)
49 , st->update_every
50 );
51
web/api/queries/query.c
+39 -23
@@ -381,13 +381,9 @@ static inline void do_dimension(
381 , long points_wanted
382 , RRDDIM *rd
383 , long dim_id_in_rrdr
384 - , long after_slot
385 - , long before_slot
384 , time_t after_wanted
385 , time_t before_wanted
386 ){
389 - (void) before_slot;
390 -
387 RRDSET *st = r->st;
388
389 time_t
@@ -397,21 +393,22 @@ static inline void do_dimension(
393 min_date = 0;
394
395 long
400 - slot = after_slot,
396 group_size = r->group,
397 points_added = 0,
398 values_in_group = 0,
399 values_in_group_non_zero = 0,
405 - rrdr_line = -1,
406 - entries = st->entries;
400 + rrdr_line = -1;
401
402 RRDR_VALUE_FLAGS
403 group_value_flags = RRDR_VALUE_NOTHING;
404
405 + struct rrddim_query_handle handle;
406 + uint8_t initialized_query;
407 +
408 calculated_number min = r->min, max = r->max;
409 size_t db_points_read = 0;
413 - for( ; points_added < points_wanted ; now += dt, slot++ ) {
414 - if(unlikely(slot >= entries)) slot = 0;
410 +
411 + for(initialized_query = 0 ; points_added < points_wanted ; now += dt) {
412
413 // make sure we return data in the proper time range
414 if(unlikely(now > before_wanted)) {
@@ -427,8 +424,23 @@ static inline void do_dimension(
424 continue;
425 }
426
427 + if (unlikely(!initialized_query)) {
428 + rd->state->query_ops.init(rd, &handle, now, before_wanted);
429 + initialized_query = 1;
430 + }
431 // read the value from the database
431 - storage_number n = rd->values[slot];
432 + //storage_number n = rd->values[slot];
433 +#ifdef NETDATA_INTERNAL_CHECKS
434 + if (rd->rrd_memory_mode == RRD_MEMORY_MODE_DBENGINE) {
435 +#ifdef ENABLE_DBENGINE
436 + if (now != handle.rrdeng.now)
437 + error("INTERNAL CHECK: Unaligned query for %s, database time: %ld, expected time: %ld", rd->id, (long)handle.rrdeng.now, (long)now);
438 +#endif
439 + } else if (rrdset_time2slot(st, now) != (long unsigned)handle.slotted.slot) {
440 + error("INTERNAL CHECK: Unaligned query for %s, database slot: %lu, expected slot: %lu", rd->id, (long unsigned)handle.slotted.slot, rrdset_time2slot(st, now));
441 + }
442 +#endif
443 + storage_number n = rd->state->query_ops.next_metric(&handle);
444 calculated_number value = NAN;
445 if(likely(does_storage_number_exist(n))) {
446
@@ -485,6 +497,8 @@ static inline void do_dimension(
497 values_in_group_non_zero = 0;
498 }
499 }
500 + if (likely(initialized_query))
501 + rd->state->query_ops.finalize(&handle);
502
503 r->internal.db_points_read += db_points_read;
504 r->internal.result_points_generated += points_added;
@@ -517,15 +531,15 @@ static void rrd2rrdr_log_request_response_metdata(RRDR *r
531 , time_t before_requested
532 , long points_requested
533 , long points_wanted
520 - , size_t after_slot
521 - , size_t before_slot
534 + //, size_t after_slot
535 + //, size_t before_slot
536 , const char *msg
537 ) {
538 info("INTERNAL ERROR: rrd2rrdr() on %s update every %d with %s grouping %s (group: %ld, resampling_time: %ld, resampling_group: %ld), "
539 "after (got: %zu, want: %zu, req: %zu, db: %zu), "
540 "before (got: %zu, want: %zu, req: %zu, db: %zu), "
541 "duration (got: %zu, want: %zu, req: %zu, db: %zu), "
528 - "slot (after: %zu, before: %zu, delta: %zu), "
542 + //"slot (after: %zu, before: %zu, delta: %zu), "
543 "points (got: %ld, want: %ld, req: %ld, db: %ld), "
544 "%s"
545 , r->st->name
@@ -557,9 +571,11 @@ static void rrd2rrdr_log_request_response_metdata(RRDR *r
571 , (size_t)((rrdset_last_entry_t(r->st) - rrdset_first_entry_t(r->st)) + r->st->update_every)
572
573 // slot
574 + /*
575 , after_slot
576 , before_slot
577 , (after_slot > before_slot) ? (r->st->entries - after_slot + before_slot) : (before_slot - after_slot)
578 + */
579
580 // points
581 , r->rows
@@ -721,7 +737,7 @@ RRDR *rrd2rrdr(
737
738 before_wanted = last_entry_t - (last_entry_t % ( ((aligned)?group:1) * st->update_every ));
739 }
724 - size_t before_slot = rrdset_time2slot(st, before_wanted);
740 + //size_t before_slot = rrdset_time2slot(st, before_wanted);
741
742 // we need to estimate the number of points, for having
743 // an integer number of values per point
@@ -743,7 +759,7 @@ RRDR *rrd2rrdr(
759 after_wanted = first_entry_t - (first_entry_t % ( ((aligned)?group:1) * st->update_every )) + ( ((aligned)?group:1) * st->update_every );
760 }
761 }
746 - size_t after_slot = rrdset_time2slot(st, after_wanted);
762 + //size_t after_slot = rrdset_time2slot(st, after_wanted);
763
764 // check if they are reversed
765 if(unlikely(after_wanted > before_wanted)) {
@@ -779,11 +795,13 @@ RRDR *rrd2rrdr(
795 if(before_wanted > last_entry_t)
796 error("INTERNAL CHECK: before_wanted %u is too big, maximum %u", (uint32_t)before_wanted, (uint32_t)last_entry_t);
797
798 +/*
799 if(before_slot >= (size_t)st->entries)
800 error("INTERNAL CHECK: before_slot is invalid %zu, expected 0 to %ld", before_slot, st->entries - 1);
801
802 if(after_slot >= (size_t)st->entries)
803 error("INTERNAL CHECK: after_slot is invalid %zu, expected 0 to %ld", after_slot, st->entries - 1);
804 +*/
805
806 if(points_wanted > (before_wanted - after_wanted) / group / st->update_every + 1)
807 error("INTERNAL CHECK: points_wanted %ld is more than points %ld", points_wanted, (before_wanted - after_wanted) / group / st->update_every + 1);
@@ -900,8 +918,6 @@ RRDR *rrd2rrdr(
918 , points_wanted
919 , rd
920 , c
903 - , after_slot
904 - , before_slot
921 , after_wanted
922 , before_wanted
923 );
@@ -947,27 +963,27 @@ RRDR *rrd2rrdr(
963 #ifdef NETDATA_INTERNAL_CHECKS
964
965 if(r->internal.log)
950 - rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, after_slot, before_slot, r->internal.log);
966 + rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, /*after_slot, before_slot,*/ r->internal.log);
967
968 if(r->rows != points_wanted)
953 - rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, after_slot, before_slot, "got 'points' is not wanted 'points'");
969 + rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, /*after_slot, before_slot,*/ "got 'points' is not wanted 'points'");
970
971 if(aligned && (r->before % group) != 0)
956 - rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, after_slot, before_slot, "'before' is not aligned but alignment is required");
972 + rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, /*after_slot, before_slot,*/ "'before' is not aligned but alignment is required");
973
974 // 'after' should not be aligned, since we start inside the first group
975 //if(aligned && (r->after % group) != 0)
976 // rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, after_slot, before_slot, "'after' is not aligned but alignment is required");
977
978 if(r->before != before_requested)
963 - rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, after_slot, before_slot, "chart is not aligned to requested 'before'");
979 + rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, /*after_slot, before_slot,*/ "chart is not aligned to requested 'before'");
980
981 if(r->before != before_wanted)
966 - rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, after_slot, before_slot, "got 'before' is not wanted 'before'");
982 + rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, /*after_slot, before_slot,*/ "got 'before' is not wanted 'before'");
983
984 // reported 'after' varies, depending on group
985 if(r->after != after_wanted)
970 - rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, after_slot, before_slot, "got 'after' is not wanted 'after'");
986 + rrd2rrdr_log_request_response_metdata(r, group_method, aligned, group, resampling_time_requested, resampling_group, after_wanted, after_requested, before_wanted, before_requested, points_requested, points_wanted, /*after_slot, before_slot,*/ "got 'after' is not wanted 'after'");
987
988 #endif
989