@cryptotaxi247 / netdata-1 / commits / d2ddb5470

Preserve UTF-8 characters in RRD string fields (#21694)

* Preserve UTF-8 characters in RRD string fields Replace json_fix_string() with text_sanitize() for RRD string fields (units, title, family, context, plugin, module) to preserve valid UTF-8 characters like ms² instead of converting them to ms__. Changes: - Add rrd_string_allowed_chars lookup table for permissive sanitization - Use text_sanitize() in rrd_string_strdupz(), query_target, and host_os_name instead of json_fix_string() - Remove now-unused json_fix_string() function - Fix buffer overflow in text_sanitize() hex encoding path - Fix memory read OOB for truncated UTF-8 sequences - Add comprehensive unit tests (734 tests) The new sanitization preserves valid UTF-8 while still transforming: - Control characters to space (deduplicated) - Double quotes to single quotes (for JSON/protocol safety) - Backslash to forward slash (to avoid escape issues) * Fix VLA stack overflow and test registration issues - Replace variable-length array with heap allocation in rrd_string_strdupz() to prevent stack overflow on large strings - Move utf8sanitizertest outside ENABLE_DBENGINE guard so it's available on all builds * Fix query_target sanitization efficiency - Use strcpy instead of memcpy to avoid copying uninitialized bytes - Fix misleading comment about output size * Increase buffer sizes to accommodate sanitized strings and prevent potential overflows during UTF-8 sanitization --------- Co-authored-by: Stelios Fragkakis <52996999+stelfrag@users.noreply.github.com>

Costa Tsaousis committed Mar 1, 2026 at 22:52 UTC d2ddb54705cac8087d358f2805b27533cccdb0f5
10 files changed +2120 -27
CMakeLists.txt
+1
@@ -1050,6 +1050,7 @@ set(LIBNETDATA_FILES
1050 src/libnetdata/sanitizers/chart_id_and_name.h
1051 src/libnetdata/sanitizers/utf8-sanitizer.c
1052 src/libnetdata/sanitizers/utf8-sanitizer.h
1053 + src/libnetdata/sanitizers/utf8-sanitizer-unittest.c
1054 src/libnetdata/sanitizers/sanitizers.h
1055 src/libnetdata/sanitizers/sanitizers-labels.c
1056 src/libnetdata/sanitizers/sanitizers-labels.h
src/daemon/main.c
+6
@@ -220,6 +220,7 @@ int dyncfg_unittest(void);
220 int eval_unittest(void);
221 int duration_unittest(void);
222 int health_config_unittest(void);
223 +int utf8_sanitizer_unittest(void);
224 bool netdata_random_session_id_generate(void);
225
226 #ifdef OS_WINDOWS
@@ -409,6 +410,7 @@ int netdata_main(int argc, char **argv) {
410 if (dyncfg_unittest()) return 1;
411 if (eval_unittest()) return 1;
412 if (duration_unittest()) return 1;
413 + if (utf8_sanitizer_unittest()) return 1;
414 if (health_config_unittest()) return 1;
415 if (unittest_waiting_queue()) return 1;
416 if (uuidmap_unittest()) return 1;
@@ -489,6 +491,10 @@ int netdata_main(int argc, char **argv) {
491 return perflibnamestest_main();
492 }
493 #endif
494 + else if(strcmp(optarg, "utf8sanitizertest") == 0) {
495 + unittest_running = true;
496 + return utf8_sanitizer_unittest();
497 + }
498 #ifdef ENABLE_DBENGINE
499 else if(strcmp(optarg, "mctest") == 0) {
500 unittest_running = true;
src/database/contexts/query_target.c
+5 -1
@@ -1151,7 +1151,11 @@ void query_target_generate_name(QUERY_TARGET *qt) {
1151 , tier_buffer
1152 );
1153
1154 - json_fix_string(qt->id);
1154 + // Sanitize the query ID - safe because qt->id is ASCII-only (from snprintfz)
1155 + char buf[MAX_QUERY_TARGET_ID_LENGTH + 1];
1156 + text_sanitize((unsigned char *)buf, (const unsigned char *)qt->id, sizeof(buf),
1157 + rrd_string_allowed_chars, true, "", NULL);
1158 + strcpy(qt->id, buf);
1159 }
1160
1161 QUERY_TARGET *query_target_create(QUERY_TARGET_REQUEST *qtr) {
src/database/rrd.c
+11 -5
@@ -27,11 +27,17 @@ int gap_when_lost_iterations_above = 1;
27 STRING *rrd_string_strdupz(const char *s) {
28 if(unlikely(!s || !*s)) return string_strdupz(s);
29
30 - char *tmp = strdupz(s);
31 - json_fix_string(tmp);
32 - STRING *ret = string_strdupz(tmp);
33 - freez(tmp);
34 - return ret;
30 + size_t len = strlen(s);
31 + size_t dst_size = (len * 2) + 1;
32 + char *buf = mallocz(dst_size);
33 +
34 + // Sanitize the string, preserving valid UTF-8
35 + text_sanitize((unsigned char *)buf, (const unsigned char *)s, dst_size,
36 + rrd_string_allowed_chars, true, "", NULL);
37 +
38 + STRING *result = string_strdupz(buf);
39 + freez(buf);
40 + return result;
41 }
42
43 // --------------------------------------------------------------------------------------------------------------------
src/database/rrdhost-system-info.c
+6 -3
@@ -60,8 +60,11 @@ int rrdhost_system_info_set_by_name(struct rrdhost_system_info *system_info, cha
60 }
61 else if(!strcmp(name, "NETDATA_HOST_OS_NAME")){
62 freez(system_info->host_os_name);
63 - system_info->host_os_name = strdupz(value);
64 - json_fix_string(system_info->host_os_name);
63 + size_t len = strlen(value);
64 + size_t dst_size = (len * 2) + 1;
65 + system_info->host_os_name = mallocz(dst_size);
66 + text_sanitize((unsigned char *)system_info->host_os_name, (const unsigned char *)value,
67 + dst_size, rrd_string_allowed_chars, true, "", NULL);
68 }
69 else if(!strcmp(name, "NETDATA_HOST_OS_ID")){
70 freez(system_info->host_os_id);
@@ -707,4 +710,4 @@ bool localhost_is_docker() {
710 return (localhost->system_info->container && strcmp(localhost->system_info->container, "docker") == 0);
711 }
712 return false;
710 -};
\ No newline at end of file
713 +};
src/libnetdata/libnetdata.c
-16
@@ -25,22 +25,6 @@ void json_escape_string(char *dst, const char *src, size_t size) {
25 *d = '\0';
26 }
27
28 -void json_fix_string(char *s) {
29 - unsigned char c;
30 - while((c = (unsigned char)*s)) {
31 - if(unlikely(c == '\\'))
32 - *s++ = '/';
33 - else if(unlikely(c == '"'))
34 - *s++ = '\'';
35 - else if(unlikely(isspace(c) || iscntrl(c)))
36 - *s++ = ' ';
37 - else if(unlikely(!isprint(c) || c > 127))
38 - *s++ = '_';
39 - else
40 - s++;
41 - }
42 -}
43 -
28 char *fgets_trim_len(char *buf, size_t buf_size, FILE *fp, size_t *len) {
29 char *s = fgets(buf, (int)buf_size, fp);
30 if (!s) return NULL;
src/libnetdata/libnetdata.h
-2
@@ -32,8 +32,6 @@ int vsnprintfz(char *dst, size_t n, const char *fmt, va_list args);
32 int snprintfz(char *dst, size_t n, const char *fmt, ...) PRINTFLIKE(3, 4);
33
34 void json_escape_string(char *dst, const char *src, size_t size);
35 -void json_fix_string(char *s);
36 -
35
36 extern struct rlimit rlimit_nofile;
37
src/libnetdata/sanitizers/chart_id_and_name.c
+60
@@ -2,6 +2,66 @@
2
3 #include "../libnetdata.h"
4
5 +// --------------------------------------------------------------------------------------------------------------------
6 +// RRD string sanitization (for units, title, family, context, plugin, module)
7 +//
8 +// This is more permissive than chart names - it preserves most printable ASCII.
9 +// The only transformations are:
10 +// - control characters → space (deduplicated by text_sanitize)
11 +// - backslash → forward slash (to avoid escape sequence issues in protocols)
12 +// - double quote → single quote (to avoid issues with quoted string parsing)
13 +//
14 +// UTF-8 is preserved by text_sanitize() when called with utf=true.
15 +
16 +unsigned char rrd_string_allowed_chars[256] = {
17 + // Control characters (0-31) → space
18 + [0] = '\0', // NUL stays NUL (string terminator)
19 + [1] = ' ', [2] = ' ', [3] = ' ', [4] = ' ', [5] = ' ', [6] = ' ', [7] = ' ', [8] = ' ',
20 + ['\t'] = ' ', ['\n'] = ' ', ['\v'] = ' ', ['\f'] = ' ', ['\r'] = ' ',
21 + [14] = ' ', [15] = ' ', [16] = ' ', [17] = ' ', [18] = ' ', [19] = ' ', [20] = ' ', [21] = ' ',
22 + [22] = ' ', [23] = ' ', [24] = ' ', [25] = ' ', [26] = ' ', [27] = ' ', [28] = ' ', [29] = ' ',
23 + [30] = ' ', [31] = ' ',
24 +
25 + // Printable ASCII (32-126) - mostly pass through
26 + [' '] = ' ',
27 + ['!'] = '!', ['"'] = '\'', ['#'] = '#', ['$'] = '$', ['%'] = '%', ['&'] = '&', ['\''] = '\'',
28 + ['('] = '(', [')'] = ')', ['*'] = '*', ['+'] = '+', [','] = ',', ['-'] = '-', ['.'] = '.', ['/'] = '/',
29 + ['0'] = '0', ['1'] = '1', ['2'] = '2', ['3'] = '3', ['4'] = '4', ['5'] = '5', ['6'] = '6', ['7'] = '7',
30 + ['8'] = '8', ['9'] = '9',
31 + [':'] = ':', [';'] = ';', ['<'] = '<', ['='] = '=', ['>'] = '>', ['?'] = '?', ['@'] = '@',
32 + ['A'] = 'A', ['B'] = 'B', ['C'] = 'C', ['D'] = 'D', ['E'] = 'E', ['F'] = 'F', ['G'] = 'G', ['H'] = 'H',
33 + ['I'] = 'I', ['J'] = 'J', ['K'] = 'K', ['L'] = 'L', ['M'] = 'M', ['N'] = 'N', ['O'] = 'O', ['P'] = 'P',
34 + ['Q'] = 'Q', ['R'] = 'R', ['S'] = 'S', ['T'] = 'T', ['U'] = 'U', ['V'] = 'V', ['W'] = 'W', ['X'] = 'X',
35 + ['Y'] = 'Y', ['Z'] = 'Z',
36 + ['['] = '[', ['\\'] = '/', [']'] = ']', ['^'] = '^', ['_'] = '_', ['`'] = '`',
37 + ['a'] = 'a', ['b'] = 'b', ['c'] = 'c', ['d'] = 'd', ['e'] = 'e', ['f'] = 'f', ['g'] = 'g', ['h'] = 'h',
38 + ['i'] = 'i', ['j'] = 'j', ['k'] = 'k', ['l'] = 'l', ['m'] = 'm', ['n'] = 'n', ['o'] = 'o', ['p'] = 'p',
39 + ['q'] = 'q', ['r'] = 'r', ['s'] = 's', ['t'] = 't', ['u'] = 'u', ['v'] = 'v', ['w'] = 'w', ['x'] = 'x',
40 + ['y'] = 'y', ['z'] = 'z',
41 + ['{'] = '{', ['|'] = '|', ['}'] = '}', ['~'] = '~',
42 +
43 + // DEL and high bytes (127-255) - text_sanitize handles UTF-8, these are fallbacks for orphan bytes
44 + [127] = ' ',
45 + [128] = ' ', [129] = ' ', [130] = ' ', [131] = ' ', [132] = ' ', [133] = ' ', [134] = ' ', [135] = ' ',
46 + [136] = ' ', [137] = ' ', [138] = ' ', [139] = ' ', [140] = ' ', [141] = ' ', [142] = ' ', [143] = ' ',
47 + [144] = ' ', [145] = ' ', [146] = ' ', [147] = ' ', [148] = ' ', [149] = ' ', [150] = ' ', [151] = ' ',
48 + [152] = ' ', [153] = ' ', [154] = ' ', [155] = ' ', [156] = ' ', [157] = ' ', [158] = ' ', [159] = ' ',
49 + [160] = ' ', [161] = ' ', [162] = ' ', [163] = ' ', [164] = ' ', [165] = ' ', [166] = ' ', [167] = ' ',
50 + [168] = ' ', [169] = ' ', [170] = ' ', [171] = ' ', [172] = ' ', [173] = ' ', [174] = ' ', [175] = ' ',
51 + [176] = ' ', [177] = ' ', [178] = ' ', [179] = ' ', [180] = ' ', [181] = ' ', [182] = ' ', [183] = ' ',
52 + [184] = ' ', [185] = ' ', [186] = ' ', [187] = ' ', [188] = ' ', [189] = ' ', [190] = ' ', [191] = ' ',
53 + [192] = ' ', [193] = ' ', [194] = ' ', [195] = ' ', [196] = ' ', [197] = ' ', [198] = ' ', [199] = ' ',
54 + [200] = ' ', [201] = ' ', [202] = ' ', [203] = ' ', [204] = ' ', [205] = ' ', [206] = ' ', [207] = ' ',
55 + [208] = ' ', [209] = ' ', [210] = ' ', [211] = ' ', [212] = ' ', [213] = ' ', [214] = ' ', [215] = ' ',
56 + [216] = ' ', [217] = ' ', [218] = ' ', [219] = ' ', [220] = ' ', [221] = ' ', [222] = ' ', [223] = ' ',
57 + [224] = ' ', [225] = ' ', [226] = ' ', [227] = ' ', [228] = ' ', [229] = ' ', [230] = ' ', [231] = ' ',
58 + [232] = ' ', [233] = ' ', [234] = ' ', [235] = ' ', [236] = ' ', [237] = ' ', [238] = ' ', [239] = ' ',
59 + [240] = ' ', [241] = ' ', [242] = ' ', [243] = ' ', [244] = ' ', [245] = ' ', [246] = ' ', [247] = ' ',
60 + [248] = ' ', [249] = ' ', [250] = ' ', [251] = ' ', [252] = ' ', [253] = ' ', [254] = ' ', [255] = ' '
61 +};
62 +
63 +// --------------------------------------------------------------------------------------------------------------------
64 +
65 /*
66 * control characters become space, which are deduplicated.
67 *
src/libnetdata/sanitizers/chart_id_and_name.h
+2
@@ -11,6 +11,8 @@ char *rrdset_strncpyz_name(char *dst, const char *src, size_t dst_size_minus_1);
11 bool rrdvar_fix_name(char *variable);
12
13 extern unsigned char chart_names_allowed_chars[256];
14 +extern unsigned char rrd_string_allowed_chars[256];
15 +
16 static inline bool is_netdata_api_valid_character(char c) {
17 if(IS_UTF8_BYTE(c)) return true;
18 unsigned char t = chart_names_allowed_chars[(unsigned char)c];
src/libnetdata/sanitizers/utf8-sanitizer-unittest.c new
+2029
@@ -0,0 +1,2029 @@
1 +// SPDX-License-Identifier: GPL-3.0-or-later
2 +
3 +#include "../libnetdata.h"
4 +
5 +// ============================================================================
6 +// TEST INFRASTRUCTURE
7 +// ============================================================================
8 +
9 +static int tests_run = 0;
10 +static int tests_passed = 0;
11 +static int tests_failed = 0;
12 +
13 +// Identity char_map - everything passes through unchanged
14 +static unsigned char identity_char_map[256];
15 +
16 +// Test char_map similar to rrd_string_allowed_chars
17 +static unsigned char test_rrd_char_map[256];
18 +
19 +static void init_char_maps(void) {
20 + // Identity map
21 + for (int i = 0; i < 256; i++)
22 + identity_char_map[i] = (unsigned char)i;
23 + identity_char_map[0] = '\0';
24 +
25 + // RRD-like map
26 + for (int i = 0; i < 256; i++)
27 + test_rrd_char_map[i] = (unsigned char)i;
28 +
29 + // Control characters (0-31, 127) → space
30 + for (int i = 1; i < 32; i++)
31 + test_rrd_char_map[i] = ' ';
32 + test_rrd_char_map[127] = ' ';
33 +
34 + // High bytes (128-255) → space (fallback for orphan UTF-8 bytes)
35 + for (int i = 128; i < 256; i++)
36 + test_rrd_char_map[i] = ' ';
37 +
38 + test_rrd_char_map[0] = '\0';
39 + test_rrd_char_map['"'] = '\''; // double quote → single quote
40 + test_rrd_char_map['\\'] = '/'; // backslash → forward slash
41 +}
42 +
43 +// Test result macro
44 +#define TEST_ASSERT(name, condition, ...) do { \
45 + tests_run++; \
46 + if (condition) { \
47 + tests_passed++; \
48 + } else { \
49 + tests_failed++; \
50 + fprintf(stderr, "FAILED [%s]: ", name); \
51 + fprintf(stderr, __VA_ARGS__); \
52 + fprintf(stderr, "\n"); \
53 + } \
54 +} while(0)
55 +
56 +// Helper to run a single sanitize test with overflow detection
57 +typedef struct {
58 + const char *name;
59 + const unsigned char *input;
60 + size_t dst_size;
61 + const unsigned char *char_map;
62 + bool utf;
63 + const char *empty;
64 + const char *expected_output;
65 + size_t expected_len;
66 + size_t expected_mblen;
67 +} sanitize_test_t;
68 +
69 +static void run_sanitize_test(const sanitize_test_t *t) {
70 + // Allocate with guard bytes
71 + size_t guard = 16;
72 + unsigned char *buffer = callocz(1, t->dst_size + guard * 2);
73 + unsigned char *dst = buffer + guard;
74 +
75 + // Fill guards
76 + memset(buffer, 0xAA, guard);
77 + memset(dst + t->dst_size, 0xBB, guard);
78 + memset(dst, 0xCC, t->dst_size);
79 +
80 + size_t mblen = 0;
81 + size_t len = text_sanitize(dst, t->input, t->dst_size, t->char_map, t->utf, t->empty, &mblen);
82 +
83 + // Check overflow
84 + bool overflow_before = false, overflow_after = false;
85 + for (size_t i = 0; i < guard; i++) {
86 + if (buffer[i] != 0xAA) overflow_before = true;
87 + if (dst[t->dst_size + i] != 0xBB) overflow_after = true;
88 + }
89 +
90 + TEST_ASSERT(t->name, !overflow_before && !overflow_after,
91 + "Buffer overflow! before=%d after=%d", overflow_before, overflow_after);
92 +
93 + TEST_ASSERT(t->name, len == t->expected_len,
94 + "Length mismatch: expected %zu, got %zu", t->expected_len, len);
95 +
96 + TEST_ASSERT(t->name, strcmp((char *)dst, t->expected_output) == 0,
97 + "Content mismatch: expected '%s', got '%s'", t->expected_output, dst);
98 +
99 + if (t->expected_mblen > 0) {
100 + TEST_ASSERT(t->name, mblen == t->expected_mblen,
101 + "Multibyte length mismatch: expected %zu, got %zu", t->expected_mblen, mblen);
102 + }
103 +
104 + // Verify null termination
105 + TEST_ASSERT(t->name, dst[len] == '\0',
106 + "Missing null terminator at position %zu", len);
107 +
108 + freez(buffer);
109 +}
110 +
111 +// ============================================================================
112 +// TEST: VALID UTF-8 SEQUENCES
113 +// ============================================================================
114 +
115 +static void test_valid_utf8_sequences(void) {
116 + fprintf(stderr, "\n=== Valid UTF-8 Sequences ===\n");
117 +
118 + // 2-byte: Latin characters with diacritics
119 + {
120 + sanitize_test_t t = {
121 + .name = "utf8_2byte_e_acute",
122 + .input = (unsigned char *)"caf\xC3\xA9", // café
123 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
124 + .expected_output = "caf\xC3\xA9", .expected_len = 5, .expected_mblen = 4
125 + };
126 + run_sanitize_test(&t);
127 + }
128 +
129 + // 2-byte: Superscript ² (U+00B2)
130 + {
131 + sanitize_test_t t = {
132 + .name = "utf8_2byte_superscript2",
133 + .input = (unsigned char *)"m/s\xC2\xB2", // m/s²
134 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
135 + .expected_output = "m/s\xC2\xB2", .expected_len = 5, .expected_mblen = 4
136 + };
137 + run_sanitize_test(&t);
138 + }
139 +
140 + // 2-byte: Degree symbol ° (U+00B0)
141 + {
142 + sanitize_test_t t = {
143 + .name = "utf8_2byte_degree",
144 + .input = (unsigned char *)"25\xC2\xB0""C", // 25°C
145 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
146 + .expected_output = "25\xC2\xB0""C", .expected_len = 5, .expected_mblen = 4
147 + };
148 + run_sanitize_test(&t);
149 + }
150 +
151 + // 2-byte: Micro sign µ (U+00B5)
152 + {
153 + sanitize_test_t t = {
154 + .name = "utf8_2byte_micro",
155 + .input = (unsigned char *)"\xC2\xB5s", // µs
156 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
157 + .expected_output = "\xC2\xB5s", .expected_len = 3, .expected_mblen = 2
158 + };
159 + run_sanitize_test(&t);
160 + }
161 +
162 + // 3-byte: Euro sign € (U+20AC)
163 + {
164 + sanitize_test_t t = {
165 + .name = "utf8_3byte_euro",
166 + .input = (unsigned char *)"100\xE2\x82\xAC", // 100€
167 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
168 + .expected_output = "100\xE2\x82\xAC", .expected_len = 6, .expected_mblen = 4
169 + };
170 + run_sanitize_test(&t);
171 + }
172 +
173 + // 3-byte: Japanese hiragana あ (U+3042)
174 + {
175 + sanitize_test_t t = {
176 + .name = "utf8_3byte_hiragana",
177 + .input = (unsigned char *)"\xE3\x81\x82", // あ
178 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
179 + .expected_output = "\xE3\x81\x82", .expected_len = 3, .expected_mblen = 1
180 + };
181 + run_sanitize_test(&t);
182 + }
183 +
184 + // 3-byte: Chinese character 中 (U+4E2D)
185 + {
186 + sanitize_test_t t = {
187 + .name = "utf8_3byte_chinese",
188 + .input = (unsigned char *)"\xE4\xB8\xAD\xE6\x96\x87", // 中文
189 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
190 + .expected_output = "\xE4\xB8\xAD\xE6\x96\x87", .expected_len = 6, .expected_mblen = 2
191 + };
192 + run_sanitize_test(&t);
193 + }
194 +
195 + // 4-byte: Emoji 😀 (U+1F600)
196 + {
197 + sanitize_test_t t = {
198 + .name = "utf8_4byte_emoji",
199 + .input = (unsigned char *)"hi\xF0\x9F\x98\x80", // hi😀
200 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
201 + .expected_output = "hi\xF0\x9F\x98\x80", .expected_len = 6, .expected_mblen = 3
202 + };
203 + run_sanitize_test(&t);
204 + }
205 +
206 + // 4-byte: Mathematical bold A 𝐀 (U+1D400)
207 + {
208 + sanitize_test_t t = {
209 + .name = "utf8_4byte_math",
210 + .input = (unsigned char *)"\xF0\x9D\x90\x80", // 𝐀
211 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
212 + .expected_output = "\xF0\x9D\x90\x80", .expected_len = 4, .expected_mblen = 1
213 + };
214 + run_sanitize_test(&t);
215 + }
216 +
217 + // Mixed: ASCII + 2-byte + 3-byte + 4-byte
218 + {
219 + sanitize_test_t t = {
220 + .name = "utf8_mixed_all_types",
221 + .input = (unsigned char *)"A\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80", // A°€😀
222 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
223 + .expected_output = "A\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80", .expected_len = 10, .expected_mblen = 4
224 + };
225 + run_sanitize_test(&t);
226 + }
227 +
228 + // Multiple same-type UTF-8 characters
229 + {
230 + sanitize_test_t t = {
231 + .name = "utf8_multiple_2byte",
232 + .input = (unsigned char *)"\xC3\xA9\xC3\xA8\xC3\xA0", // éèà
233 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
234 + .expected_output = "\xC3\xA9\xC3\xA8\xC3\xA0", .expected_len = 6, .expected_mblen = 3
235 + };
236 + run_sanitize_test(&t);
237 + }
238 +
239 + // UTF-8 at beginning of string
240 + {
241 + sanitize_test_t t = {
242 + .name = "utf8_at_beginning",
243 + .input = (unsigned char *)"\xC2\xB5sec",
244 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
245 + .expected_output = "\xC2\xB5sec", .expected_len = 5, .expected_mblen = 4
246 + };
247 + run_sanitize_test(&t);
248 + }
249 +
250 + // UTF-8 in middle of string
251 + {
252 + sanitize_test_t t = {
253 + .name = "utf8_in_middle",
254 + .input = (unsigned char *)"pre\xC2\xB0post",
255 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
256 + .expected_output = "pre\xC2\xB0post", .expected_len = 9, .expected_mblen = 8
257 + };
258 + run_sanitize_test(&t);
259 + }
260 +
261 + // UTF-8 at end of string
262 + {
263 + sanitize_test_t t = {
264 + .name = "utf8_at_end",
265 + .input = (unsigned char *)"temp\xC2\xB0",
266 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
267 + .expected_output = "temp\xC2\xB0", .expected_len = 6, .expected_mblen = 5
268 + };
269 + run_sanitize_test(&t);
270 + }
271 +
272 + // Boundary: Minimum 2-byte (U+0080)
273 + {
274 + sanitize_test_t t = {
275 + .name = "utf8_2byte_min",
276 + .input = (unsigned char *)"\xC2\x80",
277 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
278 + .expected_output = "\xC2\x80", .expected_len = 2, .expected_mblen = 1
279 + };
280 + run_sanitize_test(&t);
281 + }
282 +
283 + // Boundary: Maximum 2-byte (U+07FF)
284 + {
285 + sanitize_test_t t = {
286 + .name = "utf8_2byte_max",
287 + .input = (unsigned char *)"\xDF\xBF",
288 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
289 + .expected_output = "\xDF\xBF", .expected_len = 2, .expected_mblen = 1
290 + };
291 + run_sanitize_test(&t);
292 + }
293 +
294 + // Boundary: Minimum 3-byte (U+0800)
295 + {
296 + sanitize_test_t t = {
297 + .name = "utf8_3byte_min",
298 + .input = (unsigned char *)"\xE0\xA0\x80",
299 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
300 + .expected_output = "\xE0\xA0\x80", .expected_len = 3, .expected_mblen = 1
301 + };
302 + run_sanitize_test(&t);
303 + }
304 +
305 + // Boundary: Maximum 3-byte (U+FFFF, excluding surrogates)
306 + {
307 + sanitize_test_t t = {
308 + .name = "utf8_3byte_max",
309 + .input = (unsigned char *)"\xEF\xBF\xBD", // U+FFFD replacement char
310 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
311 + .expected_output = "\xEF\xBF\xBD", .expected_len = 3, .expected_mblen = 1
312 + };
313 + run_sanitize_test(&t);
314 + }
315 +
316 + // Boundary: Minimum 4-byte (U+10000)
317 + {
318 + sanitize_test_t t = {
319 + .name = "utf8_4byte_min",
320 + .input = (unsigned char *)"\xF0\x90\x80\x80",
321 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
322 + .expected_output = "\xF0\x90\x80\x80", .expected_len = 4, .expected_mblen = 1
323 + };
324 + run_sanitize_test(&t);
325 + }
326 +
327 + // Boundary: Maximum valid 4-byte (U+10FFFF)
328 + {
329 + sanitize_test_t t = {
330 + .name = "utf8_4byte_max",
331 + .input = (unsigned char *)"\xF4\x8F\xBF\xBF",
332 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
333 + .expected_output = "\xF4\x8F\xBF\xBF", .expected_len = 4, .expected_mblen = 1
334 + };
335 + run_sanitize_test(&t);
336 + }
337 +}
338 +
339 +// ============================================================================
340 +// TEST: INVALID UTF-8 SEQUENCES
341 +// ============================================================================
342 +
343 +static void test_invalid_utf8_sequences(void) {
344 + fprintf(stderr, "\n=== Invalid UTF-8 Sequences ===\n");
345 +
346 + // Orphan continuation byte (0x80-0xBF without start byte)
347 + {
348 + sanitize_test_t t = {
349 + .name = "invalid_orphan_continuation",
350 + .input = (unsigned char *)"A\x80""B",
351 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
352 + .expected_output = "A B", .expected_len = 3, .expected_mblen = 3
353 + };
354 + run_sanitize_test(&t);
355 + }
356 +
357 + // Multiple orphan continuation bytes
358 + {
359 + sanitize_test_t t = {
360 + .name = "invalid_multiple_orphan",
361 + .input = (unsigned char *)"\x80\x81\x82",
362 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
363 + .expected_output = "", .expected_len = 0, .expected_mblen = 0 // All become spaces, trimmed
364 + };
365 + run_sanitize_test(&t);
366 + }
367 +
368 + // Overlong 0xC0 (structurally valid 2-byte, semantically invalid)
369 + // NOTE: Function does structural validation only - passes through
370 + {
371 + sanitize_test_t t = {
372 + .name = "overlong_C0_structural_valid",
373 + .input = (unsigned char *)"X\xC0\x80Y",
374 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
375 + .expected_output = "X\xC0\x80Y", .expected_len = 4, .expected_mblen = 3
376 + };
377 + run_sanitize_test(&t);
378 + }
379 +
380 + // Overlong 0xC1 (structurally valid 2-byte)
381 + {
382 + sanitize_test_t t = {
383 + .name = "overlong_C1_structural_valid",
384 + .input = (unsigned char *)"\xC1\xBF",
385 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
386 + .expected_output = "\xC1\xBF", .expected_len = 2, .expected_mblen = 1
387 + };
388 + run_sanitize_test(&t);
389 + }
390 +
391 + // 0xF5 with continuation bytes (structurally valid 4-byte, but beyond Unicode)
392 + {
393 + sanitize_test_t t = {
394 + .name = "out_of_range_F5_structural_valid",
395 + .input = (unsigned char *)"\xF5\x80\x80\x80",
396 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
397 + .expected_output = "\xF5\x80\x80\x80", .expected_len = 4, .expected_mblen = 1
398 + };
399 + run_sanitize_test(&t);
400 + }
401 +
402 + // 0xFF alone - not a valid start byte pattern, gets hex encoded
403 + {
404 + sanitize_test_t t = {
405 + .name = "invalid_FF_hex_encoded",
406 + .input = (unsigned char *)"A\xFF""B",
407 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
408 + .expected_output = "AffB", .expected_len = 4, .expected_mblen = 3
409 + };
410 + run_sanitize_test(&t);
411 + }
412 +
413 + // Truncated 2-byte sequence at end - hex encoded
414 + {
415 + sanitize_test_t t = {
416 + .name = "truncated_2byte_hex",
417 + .input = (unsigned char *)"abc\xC2",
418 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
419 + .expected_output = "abcc2", .expected_len = 5, .expected_mblen = 4
420 + };
421 + run_sanitize_test(&t);
422 + }
423 +
424 + // Truncated 3-byte sequence (only 1 continuation) - hex encoded
425 + {
426 + sanitize_test_t t = {
427 + .name = "truncated_3byte_1cont_hex",
428 + .input = (unsigned char *)"X\xE2\x82",
429 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
430 + .expected_output = "Xe282", .expected_len = 5, .expected_mblen = 2
431 + };
432 + run_sanitize_test(&t);
433 + }
434 +
435 + // Truncated 3-byte sequence (no continuation) - hex encoded
436 + {
437 + sanitize_test_t t = {
438 + .name = "truncated_3byte_0cont_hex",
439 + .input = (unsigned char *)"Y\xE2",
440 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
441 + .expected_output = "Ye2", .expected_len = 3, .expected_mblen = 2
442 + };
443 + run_sanitize_test(&t);
444 + }
445 +
446 + // Truncated 4-byte sequence - hex encoded
447 + {
448 + sanitize_test_t t = {
449 + .name = "truncated_4byte_hex",
450 + .input = (unsigned char *)"\xF0\x9F\x98",
451 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
452 + .expected_output = "f09f98", .expected_len = 6, .expected_mblen = 1
453 + };
454 + run_sanitize_test(&t);
455 + }
456 +
457 + // Wrong continuation byte (ASCII instead of 0x80-0xBF) - hex encoded
458 + {
459 + sanitize_test_t t = {
460 + .name = "wrong_continuation_ascii_hex",
461 + .input = (unsigned char *)"\xC2X",
462 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
463 + .expected_output = "c2X", .expected_len = 3, .expected_mblen = 2
464 + };
465 + run_sanitize_test(&t);
466 + }
467 +
468 + // Wrong continuation byte (another start byte) - first hex encoded, second valid
469 + {
470 + sanitize_test_t t = {
471 + .name = "wrong_continuation_start_hex",
472 + .input = (unsigned char *)"\xC2\xC2\x80", // Second C2 is wrong, should be 80-BF
473 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
474 + .expected_output = "c2\xC2\x80", .expected_len = 4, .expected_mblen = 2
475 + };
476 + run_sanitize_test(&t);
477 + }
478 +
479 + // Overlong NUL (structurally valid, security concern but passed through)
480 + {
481 + sanitize_test_t t = {
482 + .name = "overlong_nul_structural_valid",
483 + .input = (unsigned char *)"\xC0\x80",
484 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
485 + .expected_output = "\xC0\x80", .expected_len = 2, .expected_mblen = 1
486 + };
487 + run_sanitize_test(&t);
488 + }
489 +
490 + // Overlong space (structurally valid 3-byte)
491 + {
492 + sanitize_test_t t = {
493 + .name = "overlong_space_structural_valid",
494 + .input = (unsigned char *)"\xE0\x80\xA0", // Overlong space
495 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
496 + .expected_output = "\xE0\x80\xA0", .expected_len = 3, .expected_mblen = 1
497 + };
498 + run_sanitize_test(&t);
499 + }
500 +
501 + // UTF-16 surrogate (invalid in UTF-8)
502 + {
503 + sanitize_test_t t = {
504 + .name = "invalid_surrogate_high",
505 + .input = (unsigned char *)"\xED\xA0\x80", // U+D800 high surrogate
506 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
507 + .expected_output = "\xED\xA0\x80", .expected_len = 3, .expected_mblen = 1
508 + // Note: Current implementation doesn't reject surrogates (structural only)
509 + };
510 + run_sanitize_test(&t);
511 + }
512 +
513 + // Out of range (beyond U+10FFFF)
514 + {
515 + sanitize_test_t t = {
516 + .name = "invalid_out_of_range",
517 + .input = (unsigned char *)"\xF4\x90\x80\x80", // U+110000 (invalid)
518 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
519 + .expected_output = "\xF4\x90\x80\x80", .expected_len = 4, .expected_mblen = 1
520 + // Note: Current implementation doesn't reject out of range (structural only)
521 + };
522 + run_sanitize_test(&t);
523 + }
524 +
525 + // Mixed valid UTF-8 and structurally valid overlong
526 + {
527 + sanitize_test_t t = {
528 + .name = "mixed_valid_and_overlong",
529 + .input = (unsigned char *)"A\xC2\xB0\xC0\x80\xE2\x82\xAC", // A° + overlong + €
530 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
531 + // All are structurally valid, so all pass through
532 + .expected_output = "A\xC2\xB0\xC0\x80\xE2\x82\xAC", .expected_len = 8, .expected_mblen = 4
533 + };
534 + run_sanitize_test(&t);
535 + }
536 +}
537 +
538 +// ============================================================================
539 +// TEST: BUFFER BOUNDARY CONDITIONS
540 +// ============================================================================
541 +
542 +static void test_buffer_boundaries(void) {
543 + fprintf(stderr, "\n=== Buffer Boundary Conditions ===\n");
544 +
545 + // dst_size = 0
546 + {
547 + unsigned char dst[16] = {0xCC, 0xCC, 0xCC, 0xCC};
548 + size_t len = text_sanitize(dst, (unsigned char *)"hello", 0, identity_char_map, true, "", NULL);
549 + TEST_ASSERT("buffer_size_0", len == 0, "Expected 0, got %zu", len);
550 + TEST_ASSERT("buffer_size_0_unchanged", dst[0] == 0xCC, "Buffer was modified");
551 + }
552 +
553 + // dst_size = 1 (only null terminator fits)
554 + {
555 + sanitize_test_t t = {
556 + .name = "buffer_size_1",
557 + .input = (unsigned char *)"hello",
558 + .dst_size = 1, .char_map = identity_char_map, .utf = true, .empty = "",
559 + .expected_output = "", .expected_len = 0, .expected_mblen = 0
560 + };
561 + run_sanitize_test(&t);
562 + }
563 +
564 + // dst_size = 2 (one char + null)
565 + {
566 + sanitize_test_t t = {
567 + .name = "buffer_size_2",
568 + .input = (unsigned char *)"hello",
569 + .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
570 + .expected_output = "h", .expected_len = 1, .expected_mblen = 1
571 + };
572 + run_sanitize_test(&t);
573 + }
574 +
575 + // Exact fit for ASCII
576 + {
577 + sanitize_test_t t = {
578 + .name = "buffer_exact_ascii",
579 + .input = (unsigned char *)"abc",
580 + .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
581 + .expected_output = "abc", .expected_len = 3, .expected_mblen = 3
582 + };
583 + run_sanitize_test(&t);
584 + }
585 +
586 + // Off-by-one for ASCII (truncation)
587 + {
588 + sanitize_test_t t = {
589 + .name = "buffer_truncate_ascii",
590 + .input = (unsigned char *)"abcd",
591 + .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
592 + .expected_output = "abc", .expected_len = 3, .expected_mblen = 3
593 + };
594 + run_sanitize_test(&t);
595 + }
596 +
597 + // Exact fit for 2-byte UTF-8
598 + {
599 + sanitize_test_t t = {
600 + .name = "buffer_exact_2byte",
601 + .input = (unsigned char *)"\xC2\xB0", // ° (2 bytes)
602 + .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
603 + .expected_output = "\xC2\xB0", .expected_len = 2, .expected_mblen = 1
604 + };
605 + run_sanitize_test(&t);
606 + }
607 +
608 + // Off-by-one for 2-byte UTF-8 (can't fit, hex encode)
609 + {
610 + sanitize_test_t t = {
611 + .name = "buffer_truncate_2byte",
612 + .input = (unsigned char *)"\xC2\xB0",
613 + .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
614 + .expected_output = "", .expected_len = 0, .expected_mblen = 0 // Can't fit hex either
615 + };
616 + run_sanitize_test(&t);
617 + }
618 +
619 + // Overlong sequence (structurally valid) with exact fit
620 + {
621 + sanitize_test_t t = {
622 + .name = "buffer_overlong_exact_fit",
623 + .input = (unsigned char *)"\xC0\x80", // Structurally valid 2-byte
624 + .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
625 + .expected_output = "\xC0\x80", .expected_len = 2, .expected_mblen = 1
626 + };
627 + run_sanitize_test(&t);
628 + }
629 +
630 + // ASCII + UTF-8 boundary
631 + {
632 + sanitize_test_t t = {
633 + .name = "buffer_ascii_utf8_boundary",
634 + .input = (unsigned char *)"X\xC2\xB0", // X° (3 bytes)
635 + .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
636 + .expected_output = "X\xC2\xB0", .expected_len = 3, .expected_mblen = 2
637 + };
638 + run_sanitize_test(&t);
639 + }
640 +
641 + // UTF-8 doesn't fit at buffer end - nothing written for UTF-8 but mblen still counts
642 + {
643 + sanitize_test_t t = {
644 + .name = "buffer_utf8_no_fit",
645 + .input = (unsigned char *)"XY\xC2\xB0", // XY° (4 bytes, but UTF-8 needs 2)
646 + .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
647 + // UTF-8 can't fit, hex can't fit either, nothing written for °
648 + // But mblen still increments (counts processed, not written)
649 + .expected_output = "XY", .expected_len = 2, .expected_mblen = 3
650 + };
651 + run_sanitize_test(&t);
652 + }
653 +
654 + // Overlong UTF-8 near buffer end (structurally valid, can't fit)
655 + {
656 + sanitize_test_t t = {
657 + .name = "buffer_overlong_no_fit",
658 + .input = (unsigned char *)"A\xC0\x80", // A + overlong (structurally valid)
659 + .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
660 + // Only A fits (1 byte), overlong needs 2 bytes but only 1 left
661 + // mblen counts 2 (A + attempted UTF-8)
662 + .expected_output = "A", .expected_len = 1, .expected_mblen = 2
663 + };
664 + run_sanitize_test(&t);
665 + }
666 +
667 + // Overlong 2-byte fits exactly, orphan continuation bytes follow
668 + {
669 + sanitize_test_t t = {
670 + .name = "buffer_overlong_with_orphans",
671 + .input = (unsigned char *)"X\xC0\x80\x80\x80", // X + overlong + orphan continuations
672 + .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
673 + // X (1) + overlong \xC0\x80 (2) = 3 bytes, fits in dst_size=4
674 + // Orphan bytes don't fit
675 + .expected_output = "X\xC0\x80", .expected_len = 3, .expected_mblen = 2
676 + };
677 + run_sanitize_test(&t);
678 + }
679 +
680 + // Very long input (256 bytes)
681 + {
682 + unsigned char long_input[257];
683 + memset(long_input, 'A', 256);
684 + long_input[256] = '\0';
685 +
686 + unsigned char expected[101];
687 + memset(expected, 'A', 100);
688 + expected[100] = '\0';
689 +
690 + sanitize_test_t t = {
691 + .name = "buffer_long_input",
692 + .input = long_input,
693 + .dst_size = 101, .char_map = identity_char_map, .utf = true, .empty = "",
694 + .expected_output = (char *)expected, .expected_len = 100, .expected_mblen = 100
695 + };
696 + run_sanitize_test(&t);
697 + }
698 +}
699 +
700 +// ============================================================================
701 +// TEST: CHARACTER MAP TRANSFORMATIONS
702 +// ============================================================================
703 +
704 +static void test_char_map_transformations(void) {
705 + fprintf(stderr, "\n=== Character Map Transformations ===\n");
706 +
707 + // Double quote → single quote
708 + {
709 + sanitize_test_t t = {
710 + .name = "charmap_quote",
711 + .input = (unsigned char *)"say \"hello\"",
712 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
713 + .expected_output = "say 'hello'", .expected_len = 11, .expected_mblen = 11
714 + };
715 + run_sanitize_test(&t);
716 + }
717 +
718 + // Backslash → forward slash
719 + {
720 + sanitize_test_t t = {
721 + .name = "charmap_backslash",
722 + .input = (unsigned char *)"C:\\path\\to\\file",
723 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
724 + .expected_output = "C:/path/to/file", .expected_len = 15, .expected_mblen = 15
725 + };
726 + run_sanitize_test(&t);
727 + }
728 +
729 + // Tab → space
730 + {
731 + sanitize_test_t t = {
732 + .name = "charmap_tab",
733 + .input = (unsigned char *)"col1\tcol2",
734 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
735 + .expected_output = "col1 col2", .expected_len = 9, .expected_mblen = 9
736 + };
737 + run_sanitize_test(&t);
738 + }
739 +
740 + // Newline → space
741 + {
742 + sanitize_test_t t = {
743 + .name = "charmap_newline",
744 + .input = (unsigned char *)"line1\nline2",
745 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
746 + .expected_output = "line1 line2", .expected_len = 11, .expected_mblen = 11
747 + };
748 + run_sanitize_test(&t);
749 + }
750 +
751 + // Carriage return → space
752 + {
753 + sanitize_test_t t = {
754 + .name = "charmap_cr",
755 + .input = (unsigned char *)"line1\rline2",
756 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
757 + .expected_output = "line1 line2", .expected_len = 11, .expected_mblen = 11
758 + };
759 + run_sanitize_test(&t);
760 + }
761 +
762 + // CRLF → space (deduplicated)
763 + {
764 + sanitize_test_t t = {
765 + .name = "charmap_crlf",
766 + .input = (unsigned char *)"line1\r\nline2",
767 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
768 + .expected_output = "line1 line2", .expected_len = 11, .expected_mblen = 11
769 + };
770 + run_sanitize_test(&t);
771 + }
772 +
773 + // Multiple control characters → single space
774 + {
775 + sanitize_test_t t = {
776 + .name = "charmap_multi_control",
777 + .input = (unsigned char *)"a\t\n\r\vb",
778 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
779 + .expected_output = "a b", .expected_len = 3, .expected_mblen = 3
780 + };
781 + run_sanitize_test(&t);
782 + }
783 +
784 + // NUL character (should terminate)
785 + {
786 + unsigned char input[] = {'a', 'b', '\0', 'c', 'd', '\0'};
787 + sanitize_test_t t = {
788 + .name = "charmap_nul",
789 + .input = input,
790 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
791 + .expected_output = "ab", .expected_len = 2, .expected_mblen = 2
792 + };
793 + run_sanitize_test(&t);
794 + }
795 +
796 + // DEL character (0x7F) → space
797 + {
798 + sanitize_test_t t = {
799 + .name = "charmap_del",
800 + .input = (unsigned char *)"a\x7F""b",
801 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
802 + .expected_output = "a b", .expected_len = 3, .expected_mblen = 3
803 + };
804 + run_sanitize_test(&t);
805 + }
806 +
807 + // All printable ASCII preserved (30 characters)
808 + {
809 + sanitize_test_t t = {
810 + .name = "charmap_printable_ascii",
811 + .input = (unsigned char *)"!#$%&'()*+,-./:;<=>?@[]^_`{|}~",
812 + .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
813 + .expected_output = "!#$%&'()*+,-./:;<=>?@[]^_`{|}~", .expected_len = 30, .expected_mblen = 30
814 + };
815 + run_sanitize_test(&t);
816 + }
817 +
818 + // High bytes (0x80-0xBF) are continuation bytes → char_map (space)
819 + // 0xFF is a start byte but invalid pattern → hex encoded
820 + {
821 + sanitize_test_t t = {
822 + .name = "charmap_high_byte_mixed",
823 + .input = (unsigned char *)"a\x80\x90\xA0\xB0\xFF""b",
824 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
825 + // \x80-\xB0 are continuation bytes (10xxxxxx) → go through char_map → space
826 + // \xFF is start byte but invalid pattern → hex encoded as "ff"
827 + .expected_output = "a ffb", .expected_len = 5, .expected_mblen = 4
828 + };
829 + run_sanitize_test(&t);
830 + }
831 +
832 + // Combined transformations
833 + {
834 + sanitize_test_t t = {
835 + .name = "charmap_combined",
836 + .input = (unsigned char *)"\"path\\to\\file\"\t(100%)",
837 + .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
838 + .expected_output = "'path/to/file' (100%)", .expected_len = 21, .expected_mblen = 21
839 + };
840 + run_sanitize_test(&t);
841 + }
842 +}
843 +
844 +// ============================================================================
845 +// TEST: SPACE HANDLING
846 +// ============================================================================
847 +
848 +static void test_space_handling(void) {
849 + fprintf(stderr, "\n=== Space Handling ===\n");
850 +
851 + // Leading spaces removed
852 + {
853 + sanitize_test_t t = {
854 + .name = "space_leading",
855 + .input = (unsigned char *)" hello",
856 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
857 + .expected_output = "hello", .expected_len = 5, .expected_mblen = 5
858 + };
859 + run_sanitize_test(&t);
860 + }
861 +
862 + // Trailing spaces removed
863 + {
864 + sanitize_test_t t = {
865 + .name = "space_trailing",
866 + .input = (unsigned char *)"hello ",
867 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
868 + .expected_output = "hello", .expected_len = 5, .expected_mblen = 5
869 + };
870 + run_sanitize_test(&t);
871 + }
872 +
873 + // Both leading and trailing
874 + {
875 + sanitize_test_t t = {
876 + .name = "space_both_ends",
877 + .input = (unsigned char *)" hello ",
878 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
879 + .expected_output = "hello", .expected_len = 5, .expected_mblen = 5
880 + };
881 + run_sanitize_test(&t);
882 + }
883 +
884 + // Multiple consecutive spaces → single space
885 + {
886 + sanitize_test_t t = {
887 + .name = "space_consecutive",
888 + .input = (unsigned char *)"hello world",
889 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
890 + .expected_output = "hello world", .expected_len = 11, .expected_mblen = 11
891 + };
892 + run_sanitize_test(&t);
893 + }
894 +
895 + // Only spaces → empty
896 + {
897 + sanitize_test_t t = {
898 + .name = "space_only",
899 + .input = (unsigned char *)" ",
900 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "default",
901 + .expected_output = "default", .expected_len = 7, .expected_mblen = 7
902 + };
903 + run_sanitize_test(&t);
904 + }
905 +
906 + // Control chars becoming spaces and deduplicating
907 + {
908 + sanitize_test_t t = {
909 + .name = "space_from_control",
910 + .input = (unsigned char *)"a\t\t\t\nb",
911 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
912 + .expected_output = "a b", .expected_len = 3, .expected_mblen = 3
913 + };
914 + run_sanitize_test(&t);
915 + }
916 +
917 + // Space before UTF-8
918 + {
919 + sanitize_test_t t = {
920 + .name = "space_before_utf8",
921 + .input = (unsigned char *)"temp \xC2\xB0""C",
922 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
923 + .expected_output = "temp \xC2\xB0""C", .expected_len = 8, .expected_mblen = 7
924 + };
925 + run_sanitize_test(&t);
926 + }
927 +
928 + // Space after UTF-8
929 + {
930 + sanitize_test_t t = {
931 + .name = "space_after_utf8",
932 + .input = (unsigned char *)"\xC2\xB0 Celsius",
933 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
934 + .expected_output = "\xC2\xB0 Celsius", .expected_len = 10, .expected_mblen = 9
935 + };
936 + run_sanitize_test(&t);
937 + }
938 +
939 + // Tab-separated values
940 + {
941 + sanitize_test_t t = {
942 + .name = "space_tsv",
943 + .input = (unsigned char *)"col1\tcol2\tcol3",
944 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
945 + .expected_output = "col1 col2 col3", .expected_len = 14, .expected_mblen = 14
946 + };
947 + run_sanitize_test(&t);
948 + }
949 +}
950 +
951 +// ============================================================================
952 +// TEST: EMPTY AND SPECIAL CASES
953 +// ============================================================================
954 +
955 +static void test_empty_and_special(void) {
956 + fprintf(stderr, "\n=== Empty and Special Cases ===\n");
957 +
958 + // NULL input
959 + {
960 + unsigned char dst[32];
961 + size_t len = text_sanitize(dst, NULL, sizeof(dst), identity_char_map, true, "null_val", NULL);
962 + TEST_ASSERT("null_input", strcmp((char *)dst, "null_val") == 0,
963 + "Expected 'null_val', got '%s'", dst);
964 + TEST_ASSERT("null_input_len", len == 8, "Expected 8, got %zu", len);
965 + }
966 +
967 + // NULL dst
968 + {
969 + size_t len = text_sanitize(NULL, (unsigned char *)"hello", 32, identity_char_map, true, "", NULL);
970 + TEST_ASSERT("null_dst", len == 0, "Expected 0, got %zu", len);
971 + }
972 +
973 + // Empty string input
974 + {
975 + sanitize_test_t t = {
976 + .name = "empty_input",
977 + .input = (unsigned char *)"",
978 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "empty_val",
979 + .expected_output = "empty_val", .expected_len = 9, .expected_mblen = 9
980 + };
981 + run_sanitize_test(&t);
982 + }
983 +
984 + // All underscores → empty (special rule)
985 + {
986 + sanitize_test_t t = {
987 + .name = "all_underscores",
988 + .input = (unsigned char *)"___",
989 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "default",
990 + .expected_output = "default", .expected_len = 7, .expected_mblen = 7
991 + };
992 + run_sanitize_test(&t);
993 + }
994 +
995 + // Underscore followed by text (not empty)
996 + {
997 + sanitize_test_t t = {
998 + .name = "underscore_prefix",
999 + .input = (unsigned char *)"___abc",
1000 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1001 + .expected_output = "___abc", .expected_len = 6, .expected_mblen = 6
1002 + };
1003 + run_sanitize_test(&t);
1004 + }
1005 +
1006 + // Only control characters → empty
1007 + {
1008 + sanitize_test_t t = {
1009 + .name = "only_control_chars",
1010 + .input = (unsigned char *)"\t\n\r\v",
1011 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "ctrl_empty",
1012 + .expected_output = "ctrl_empty", .expected_len = 10, .expected_mblen = 10
1013 + };
1014 + run_sanitize_test(&t);
1015 + }
1016 +
1017 + // Invalid UTF-8 that becomes all underscores with utf=false
1018 + {
1019 + sanitize_test_t t = {
1020 + .name = "utf8_to_underscores",
1021 + .input = (unsigned char *)"\xC2\x80\xC2\x80",
1022 + .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "utf_empty",
1023 + .expected_output = "utf_empty", .expected_len = 9, .expected_mblen = 9
1024 + };
1025 + run_sanitize_test(&t);
1026 + }
1027 +
1028 + // Empty string with empty default
1029 + {
1030 + sanitize_test_t t = {
1031 + .name = "empty_with_empty_default",
1032 + .input = (unsigned char *)"",
1033 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1034 + .expected_output = "", .expected_len = 0, .expected_mblen = 0
1035 + };
1036 + run_sanitize_test(&t);
1037 + }
1038 +
1039 + // Single character
1040 + {
1041 + sanitize_test_t t = {
1042 + .name = "single_char",
1043 + .input = (unsigned char *)"X",
1044 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1045 + .expected_output = "X", .expected_len = 1, .expected_mblen = 1
1046 + };
1047 + run_sanitize_test(&t);
1048 + }
1049 +
1050 + // Single UTF-8 character
1051 + {
1052 + sanitize_test_t t = {
1053 + .name = "single_utf8_char",
1054 + .input = (unsigned char *)"\xC2\xB0",
1055 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1056 + .expected_output = "\xC2\xB0", .expected_len = 2, .expected_mblen = 1
1057 + };
1058 + run_sanitize_test(&t);
1059 + }
1060 +}
1061 +
1062 +// ============================================================================
1063 +// TEST: UTF PARAMETER (true vs false)
1064 +// ============================================================================
1065 +
1066 +static void test_utf_parameter(void) {
1067 + fprintf(stderr, "\n=== UTF Parameter (true vs false) ===\n");
1068 +
1069 + // utf=true: valid UTF-8 preserved
1070 + {
1071 + sanitize_test_t t = {
1072 + .name = "utf_true_valid",
1073 + .input = (unsigned char *)"test\xC2\xB0""C",
1074 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1075 + .expected_output = "test\xC2\xB0""C", .expected_len = 7, .expected_mblen = 6
1076 + };
1077 + run_sanitize_test(&t);
1078 + }
1079 +
1080 + // utf=false: valid UTF-8 → underscore
1081 + {
1082 + sanitize_test_t t = {
1083 + .name = "utf_false_valid",
1084 + .input = (unsigned char *)"test\xC2\xB0""C",
1085 + .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1086 + .expected_output = "test_C", .expected_len = 6, .expected_mblen = 6
1087 + };
1088 + run_sanitize_test(&t);
1089 + }
1090 +
1091 + // utf=true: overlong (structurally valid) passes through
1092 + {
1093 + sanitize_test_t t = {
1094 + .name = "utf_true_overlong",
1095 + .input = (unsigned char *)"test\xC0\x80""X",
1096 + .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1097 + // \xC0\x80 is structurally valid (2-byte pattern), passes through
1098 + .expected_output = "test\xC0\x80X", .expected_len = 7, .expected_mblen = 6
1099 + };
1100 + run_sanitize_test(&t);
1101 + }
1102 +
1103 + // utf=false: invalid UTF-8 → underscore
1104 + {
1105 + sanitize_test_t t = {
1106 + .name = "utf_false_invalid",
1107 + .input = (unsigned char *)"test\xC0\x80""X",
1108 + .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1109 + .expected_output = "test_X", .expected_len = 6, .expected_mblen = 6
1110 + };
1111 + run_sanitize_test(&t);
1112 + }
1113 +
1114 + // utf=false: 3-byte UTF-8 → single underscore
1115 + {
1116 + sanitize_test_t t = {
1117 + .name = "utf_false_3byte",
1118 + .input = (unsigned char *)"price\xE2\x82\xAC""100", // price€100
1119 + .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1120 + .expected_output = "price_100", .expected_len = 9, .expected_mblen = 9
1121 + };
1122 + run_sanitize_test(&t);
1123 + }
1124 +
1125 + // utf=false: 4-byte UTF-8 → single underscore
1126 + {
1127 + sanitize_test_t t = {
1128 + .name = "utf_false_4byte",
1129 + .input = (unsigned char *)"hi\xF0\x9F\x98\x80""!", // hi😀!
1130 + .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1131 + .expected_output = "hi_!", .expected_len = 4, .expected_mblen = 4
1132 + };
1133 + run_sanitize_test(&t);
1134 + }
1135 +
1136 + // utf=false: multiple UTF-8 → multiple underscores (but collapse doesn't happen)
1137 + {
1138 + sanitize_test_t t = {
1139 + .name = "utf_false_multiple",
1140 + .input = (unsigned char *)"\xC2\xB0\xC2\xB5", // °µ
1141 + .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "x",
1142 + .expected_output = "x", .expected_len = 1, .expected_mblen = 1
1143 + // Two underscores collapse to empty due to all-underscore rule
1144 + };
1145 + run_sanitize_test(&t);
1146 + }
1147 +}
1148 +
1149 +// ============================================================================
1150 +// TEST: MULTIBYTE LENGTH OUTPUT
1151 +// ============================================================================
1152 +
1153 +static void test_multibyte_length(void) {
1154 + fprintf(stderr, "\n=== Multibyte Length Output ===\n");
1155 +
1156 + // ASCII only: byte length == char count
1157 + {
1158 + unsigned char dst[32];
1159 + size_t mblen = 0;
1160 + size_t len = text_sanitize(dst, (unsigned char *)"hello", sizeof(dst),
1161 + identity_char_map, true, "", &mblen);
1162 + TEST_ASSERT("mblen_ascii", len == 5 && mblen == 5,
1163 + "len=%zu mblen=%zu, expected both 5", len, mblen);
1164 + }
1165 +
1166 + // Single 2-byte UTF-8: byte length > char count
1167 + {
1168 + unsigned char dst[32];
1169 + size_t mblen = 0;
1170 + size_t len = text_sanitize(dst, (unsigned char *)"\xC2\xB0", sizeof(dst),
1171 + identity_char_map, true, "", &mblen);
1172 + TEST_ASSERT("mblen_2byte", len == 2 && mblen == 1,
1173 + "len=%zu mblen=%zu, expected len=2 mblen=1", len, mblen);
1174 + }
1175 +
1176 + // Single 3-byte UTF-8
1177 + {
1178 + unsigned char dst[32];
1179 + size_t mblen = 0;
1180 + size_t len = text_sanitize(dst, (unsigned char *)"\xE2\x82\xAC", sizeof(dst),
1181 + identity_char_map, true, "", &mblen);
1182 + TEST_ASSERT("mblen_3byte", len == 3 && mblen == 1,
1183 + "len=%zu mblen=%zu, expected len=3 mblen=1", len, mblen);
1184 + }
1185 +
1186 + // Single 4-byte UTF-8
1187 + {
1188 + unsigned char dst[32];
1189 + size_t mblen = 0;
1190 + size_t len = text_sanitize(dst, (unsigned char *)"\xF0\x9F\x98\x80", sizeof(dst),
1191 + identity_char_map, true, "", &mblen);
1192 + TEST_ASSERT("mblen_4byte", len == 4 && mblen == 1,
1193 + "len=%zu mblen=%zu, expected len=4 mblen=1", len, mblen);
1194 + }
1195 +
1196 + // Mixed: ASCII + UTF-8
1197 + {
1198 + unsigned char dst[32];
1199 + size_t mblen = 0;
1200 + // "A°€😀" = 1 + 2 + 3 + 4 = 10 bytes, 4 chars
1201 + size_t len = text_sanitize(dst, (unsigned char *)"A\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80",
1202 + sizeof(dst), identity_char_map, true, "", &mblen);
1203 + TEST_ASSERT("mblen_mixed", len == 10 && mblen == 4,
1204 + "len=%zu mblen=%zu, expected len=10 mblen=4", len, mblen);
1205 + }
1206 +
1207 + // NULL mblen pointer (shouldn't crash)
1208 + {
1209 + unsigned char dst[32];
1210 + size_t len = text_sanitize(dst, (unsigned char *)"test", sizeof(dst),
1211 + identity_char_map, true, "", NULL);
1212 + TEST_ASSERT("mblen_null_ptr", len == 4, "len=%zu, expected 4", len);
1213 + }
1214 +}
1215 +
1216 +// ============================================================================
1217 +// TEST: RRD STRING ALLOWED CHARS SPECIFIC
1218 +// ============================================================================
1219 +
1220 +static void test_rrd_string_allowed_chars(void) {
1221 + fprintf(stderr, "\n=== RRD String Allowed Chars ===\n");
1222 +
1223 + // Use actual rrd_string_allowed_chars from the codebase
1224 + extern unsigned char rrd_string_allowed_chars[256];
1225 +
1226 + // Basic ASCII passes through
1227 + {
1228 + sanitize_test_t t = {
1229 + .name = "rrd_ascii",
1230 + .input = (unsigned char *)"cpu.user",
1231 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1232 + .expected_output = "cpu.user", .expected_len = 8, .expected_mblen = 8
1233 + };
1234 + run_sanitize_test(&t);
1235 + }
1236 +
1237 + // Double quote transformed
1238 + {
1239 + sanitize_test_t t = {
1240 + .name = "rrd_double_quote",
1241 + .input = (unsigned char *)"\"value\"",
1242 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1243 + .expected_output = "'value'", .expected_len = 7, .expected_mblen = 7
1244 + };
1245 + run_sanitize_test(&t);
1246 + }
1247 +
1248 + // Backslash transformed
1249 + {
1250 + sanitize_test_t t = {
1251 + .name = "rrd_backslash",
1252 + .input = (unsigned char *)"path\\file",
1253 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1254 + .expected_output = "path/file", .expected_len = 9, .expected_mblen = 9
1255 + };
1256 + run_sanitize_test(&t);
1257 + }
1258 +
1259 + // UTF-8 units preserved
1260 + {
1261 + sanitize_test_t t = {
1262 + .name = "rrd_utf8_units",
1263 + .input = (unsigned char *)"requests/s\xC2\xB2", // requests/s²
1264 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1265 + .expected_output = "requests/s\xC2\xB2", .expected_len = 12, .expected_mblen = 11
1266 + };
1267 + run_sanitize_test(&t);
1268 + }
1269 +
1270 + // Temperature with degree symbol
1271 + {
1272 + sanitize_test_t t = {
1273 + .name = "rrd_temperature",
1274 + .input = (unsigned char *)"Temperature (\xC2\xB0""C)",
1275 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1276 + .expected_output = "Temperature (\xC2\xB0""C)", .expected_len = 17, .expected_mblen = 16
1277 + };
1278 + run_sanitize_test(&t);
1279 + }
1280 +
1281 + // Microseconds
1282 + {
1283 + sanitize_test_t t = {
1284 + .name = "rrd_microseconds",
1285 + .input = (unsigned char *)"\xC2\xB5s", // µs
1286 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1287 + .expected_output = "\xC2\xB5s", .expected_len = 3, .expected_mblen = 2
1288 + };
1289 + run_sanitize_test(&t);
1290 + }
1291 +
1292 + // Complex metric title
1293 + {
1294 + sanitize_test_t t = {
1295 + .name = "rrd_complex_title",
1296 + .input = (unsigned char *)"CPU \"usage\" on C:\\Windows (100%)",
1297 + .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1298 + .expected_output = "CPU 'usage' on C:/Windows (100%)", .expected_len = 32, .expected_mblen = 32
1299 + };
1300 + run_sanitize_test(&t);
1301 + }
1302 +
1303 + // Prometheus-style metric
1304 + {
1305 + sanitize_test_t t = {
1306 + .name = "rrd_prometheus_style",
1307 + .input = (unsigned char *)"http_requests_total{method=\"GET\"}",
1308 + .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1309 + .expected_output = "http_requests_total{method='GET'}", .expected_len = 33, .expected_mblen = 33
1310 + };
1311 + run_sanitize_test(&t);
1312 + }
1313 +}
1314 +
1315 +// ============================================================================
1316 +// TEST: SECURITY-FOCUSED CASES
1317 +// ============================================================================
1318 +
1319 +static void test_security_cases(void) {
1320 + fprintf(stderr, "\n=== Security-Focused Cases ===\n");
1321 +
1322 + // Path traversal attempt (should be handled safely)
1323 + {
1324 + sanitize_test_t t = {
1325 + .name = "security_path_traversal",
1326 + .input = (unsigned char *)"../../../etc/passwd",
1327 + .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1328 + .expected_output = "../../../etc/passwd", .expected_len = 19, .expected_mblen = 19
1329 + };
1330 + run_sanitize_test(&t);
1331 + }
1332 +
1333 + // Path traversal with backslash (Windows style, converted to /)
1334 + {
1335 + sanitize_test_t t = {
1336 + .name = "security_path_traversal_win",
1337 + .input = (unsigned char *)"..\\..\\..\\etc\\passwd",
1338 + .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1339 + .expected_output = "../../../etc/passwd", .expected_len = 19, .expected_mblen = 19
1340 + };
1341 + run_sanitize_test(&t);
1342 + }
1343 +
1344 + // Overlong NUL - structurally valid, passes through
1345 + // NOTE: This is a security concern in some systems but this function
1346 + // only does structural validation for sanitization purposes
1347 + {
1348 + sanitize_test_t t = {
1349 + .name = "security_overlong_nul_passthrough",
1350 + .input = (unsigned char *)"test\xC0\x80test", // Overlong NUL
1351 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1352 + .expected_output = "test\xC0\x80test", .expected_len = 10, .expected_mblen = 9
1353 + };
1354 + run_sanitize_test(&t);
1355 + }
1356 +
1357 + // Overlong slash - structurally valid, passes through
1358 + {
1359 + sanitize_test_t t = {
1360 + .name = "security_overlong_slash_passthrough",
1361 + .input = (unsigned char *)"\xC0\xAF", // Overlong /
1362 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1363 + .expected_output = "\xC0\xAF", .expected_len = 2, .expected_mblen = 1
1364 + };
1365 + run_sanitize_test(&t);
1366 + }
1367 +
1368 + // Overlong A (3 bytes) - structurally valid, passes through
1369 + {
1370 + sanitize_test_t t = {
1371 + .name = "security_overlong_A_passthrough",
1372 + .input = (unsigned char *)"\xE0\x81\x81", // Overlong A
1373 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1374 + .expected_output = "\xE0\x81\x81", .expected_len = 3, .expected_mblen = 1
1375 + };
1376 + run_sanitize_test(&t);
1377 + }
1378 +
1379 + // XSS attempt with angle brackets
1380 + {
1381 + sanitize_test_t t = {
1382 + .name = "security_xss_tags",
1383 + .input = (unsigned char *)"<script>alert(1)</script>",
1384 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1385 + .expected_output = "<script>alert(1)</script>", .expected_len = 25, .expected_mblen = 25
1386 + };
1387 + run_sanitize_test(&t);
1388 + }
1389 +
1390 + // SQL injection attempt (quotes transformed)
1391 + {
1392 + sanitize_test_t t = {
1393 + .name = "security_sql_injection",
1394 + .input = (unsigned char *)"test' OR '1'='1",
1395 + .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1396 + .expected_output = "test' OR '1'='1", .expected_len = 15, .expected_mblen = 15
1397 + };
1398 + run_sanitize_test(&t);
1399 + }
1400 +
1401 + // Null byte injection (string terminates at NUL)
1402 + {
1403 + unsigned char input[] = {'t', 'e', 's', 't', '\0', 'e', 'v', 'i', 'l', '\0'};
1404 + sanitize_test_t t = {
1405 + .name = "security_null_byte",
1406 + .input = input,
1407 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1408 + .expected_output = "test", .expected_len = 4, .expected_mblen = 4
1409 + };
1410 + run_sanitize_test(&t);
1411 + }
1412 +
1413 + // BOM (Byte Order Mark) at start - should be preserved as valid UTF-8
1414 + {
1415 + sanitize_test_t t = {
1416 + .name = "security_bom",
1417 + .input = (unsigned char *)"\xEF\xBB\xBFtext", // UTF-8 BOM + text
1418 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1419 + .expected_output = "\xEF\xBB\xBFtext", .expected_len = 7, .expected_mblen = 5
1420 + };
1421 + run_sanitize_test(&t);
1422 + }
1423 +
1424 + // UTF-7 encoding attempt (should just pass through as ASCII)
1425 + {
1426 + sanitize_test_t t = {
1427 + .name = "security_utf7",
1428 + .input = (unsigned char *)"+ADw-script+AD4-",
1429 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1430 + .expected_output = "+ADw-script+AD4-", .expected_len = 16, .expected_mblen = 16
1431 + };
1432 + run_sanitize_test(&t);
1433 + }
1434 +
1435 + // Private Use Area character (valid UTF-8, possibly suspicious)
1436 + {
1437 + sanitize_test_t t = {
1438 + .name = "security_private_use",
1439 + .input = (unsigned char *)"\xEE\x80\x80", // U+E000 (Private Use)
1440 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1441 + .expected_output = "\xEE\x80\x80", .expected_len = 3, .expected_mblen = 1
1442 + };
1443 + run_sanitize_test(&t);
1444 + }
1445 +}
1446 +
1447 +// ============================================================================
1448 +// TEST: REGRESSION TESTS FOR FIXED BUGS
1449 +// ============================================================================
1450 +
1451 +static void test_regression_fixed_bugs(void) {
1452 + fprintf(stderr, "\n=== Regression Tests for Fixed Bugs ===\n");
1453 +
1454 + // REGRESSION: The original buffer overflow bug was in hex encoding path.
1455 + // Test with TRULY invalid UTF-8 (truncated sequence) that triggers hex encoding
1456 + {
1457 + sanitize_test_t t = {
1458 + .name = "regression_hex_buffer_overflow",
1459 + .input = (unsigned char *)"\xC2", // Truncated 2-byte sequence
1460 + .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
1461 + // Hex needs 2 chars ("c2") + NUL = 3, exact fit
1462 + .expected_output = "c2", .expected_len = 2, .expected_mblen = 1
1463 + };
1464 + run_sanitize_test(&t);
1465 + }
1466 +
1467 + // Test truncated sequence that would overflow if not properly bounded
1468 + {
1469 + sanitize_test_t t = {
1470 + .name = "regression_hex_no_overflow",
1471 + .input = (unsigned char *)"\xC2", // Truncated
1472 + .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
1473 + // Hex needs 2 chars but only 1 space (plus NUL) - nothing written
1474 + // Note: mblen is 0 when loop doesn't process due to buffer constraints
1475 + .expected_output = "", .expected_len = 0, .expected_mblen = 0
1476 + };
1477 + run_sanitize_test(&t);
1478 + }
1479 +
1480 + // Overlong \xC0\x80 is structurally VALID - test it passes through
1481 + {
1482 + sanitize_test_t t = {
1483 + .name = "regression_overlong_passthrough",
1484 + .input = (unsigned char *)"X\xC0\x80Y",
1485 + .dst_size = 16, .char_map = identity_char_map, .utf = true, .empty = "",
1486 + .expected_output = "X\xC0\x80Y", .expected_len = 4, .expected_mblen = 3
1487 + };
1488 + run_sanitize_test(&t);
1489 + }
1490 +
1491 + // REGRESSION: Memory read OOB (Issue: Loop didn't check for NUL before accessing src[i])
1492 + {
1493 + sanitize_test_t t = {
1494 + .name = "regression_memory_oob_truncated",
1495 + .input = (unsigned char *)"test\xE2\x82", // Truncated 3-byte sequence
1496 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1497 + .expected_output = "teste282", .expected_len = 8, .expected_mblen = 5
1498 + };
1499 + run_sanitize_test(&t);
1500 + }
1501 +
1502 + // REGRESSION: Memory read OOB with 4-byte truncated at various points
1503 + {
1504 + sanitize_test_t t1 = {
1505 + .name = "regression_oob_4byte_1cont",
1506 + .input = (unsigned char *)"\xF0\x9F", // Only 2 of 4 bytes
1507 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1508 + .expected_output = "f09f", .expected_len = 4, .expected_mblen = 1
1509 + };
1510 + run_sanitize_test(&t1);
1511 +
1512 + sanitize_test_t t2 = {
1513 + .name = "regression_oob_4byte_2cont",
1514 + .input = (unsigned char *)"\xF0\x9F\x98", // Only 3 of 4 bytes
1515 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1516 + .expected_output = "f09f98", .expected_len = 6, .expected_mblen = 1
1517 + };
1518 + run_sanitize_test(&t2);
1519 + }
1520 +
1521 + // Edge case: \xF5 is treated as 4-byte start, but alone it's truncated → hex
1522 + // NOTE: \xF5 without continuation bytes triggers hex encoding
1523 + {
1524 + sanitize_test_t t = {
1525 + .name = "regression_edge_F5_truncated",
1526 + .input = (unsigned char *)"X\xF5", // X + truncated F5 (no continuation)
1527 + .dst_size = 5, .char_map = identity_char_map, .utf = true, .empty = "",
1528 + // X (1) + "f5" (2) + NUL = 4, fits in 5
1529 + .expected_output = "Xf5", .expected_len = 3, .expected_mblen = 2
1530 + };
1531 + run_sanitize_test(&t);
1532 + }
1533 +
1534 + // Edge case: Exactly 2 spaces for hex (dst_size=4, one char used)
1535 + {
1536 + sanitize_test_t t = {
1537 + .name = "regression_edge_exact_hex_fit",
1538 + .input = (unsigned char *)"A\xC0", // A + truncated (missing continuation)
1539 + .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
1540 + // A (1) + "c0" (2) + NUL = 4 (exact fit)
1541 + .expected_output = "Ac0", .expected_len = 3, .expected_mblen = 2
1542 + };
1543 + run_sanitize_test(&t);
1544 + }
1545 +
1546 + // Verify the original bug scenario from the PR: ms² being preserved
1547 + {
1548 + sanitize_test_t t = {
1549 + .name = "regression_ms_squared",
1550 + .input = (unsigned char *)"ms\xC2\xB2", // ms²
1551 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1552 + .expected_output = "ms\xC2\xB2", .expected_len = 4, .expected_mblen = 3
1553 + };
1554 + run_sanitize_test(&t);
1555 + }
1556 +}
1557 +
1558 +// ============================================================================
1559 +// TEST: ALL CONTROL CHARACTERS (0x00-0x1F, 0x7F)
1560 +// ============================================================================
1561 +
1562 +static void test_all_control_characters(void) {
1563 + fprintf(stderr, "\n=== All Control Characters ===\n");
1564 +
1565 + // Test each control character 0x01-0x1F individually
1566 + for (unsigned int ctrl = 1; ctrl < 32; ctrl++) {
1567 + unsigned char input[4] = {'A', (unsigned char)ctrl, 'B', '\0'};
1568 + char expected[8];
1569 +
1570 + // With test_rrd_char_map, all control chars become space
1571 + snprintf(expected, sizeof(expected), "A B");
1572 +
1573 + char name[32];
1574 + snprintf(name, sizeof(name), "ctrl_0x%02X", ctrl);
1575 +
1576 + size_t guard = 16;
1577 + unsigned char *buffer = callocz(1, 32 + guard * 2);
1578 + unsigned char *dst = buffer + guard;
1579 + memset(buffer, 0xAA, guard);
1580 + memset(dst + 32, 0xBB, guard);
1581 + memset(dst, 0xCC, 32);
1582 +
1583 + size_t mblen = 0;
1584 + text_sanitize(dst, input, 32, test_rrd_char_map, true, "", &mblen);
1585 +
1586 + bool overflow = false;
1587 + for (size_t i = 0; i < guard; i++) {
1588 + if (buffer[i] != 0xAA || dst[32 + i] != 0xBB) {
1589 + overflow = true;
1590 + break;
1591 + }
1592 + }
1593 +
1594 + TEST_ASSERT(name, !overflow && strcmp((char *)dst, expected) == 0,
1595 + "ctrl=0x%02X: overflow=%d, expected '%s', got '%s'", ctrl, overflow, expected, dst);
1596 +
1597 + freez(buffer);
1598 + }
1599 +
1600 + // Test DEL (0x7F)
1601 + {
1602 + unsigned char input[] = {'A', 0x7F, 'B', '\0'};
1603 + sanitize_test_t t = {
1604 + .name = "ctrl_DEL_0x7F",
1605 + .input = input,
1606 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1607 + .expected_output = "A B", .expected_len = 3, .expected_mblen = 3
1608 + };
1609 + run_sanitize_test(&t);
1610 + }
1611 +
1612 + // Multiple different control characters in sequence
1613 + {
1614 + unsigned char input[] = {'X', 0x01, 0x02, 0x03, 0x04, 0x05, 'Y', '\0'};
1615 + sanitize_test_t t = {
1616 + .name = "ctrl_multiple_sequence",
1617 + .input = input,
1618 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1619 + .expected_output = "X Y", .expected_len = 3, .expected_mblen = 3
1620 + // All control chars become spaces, then deduplicated
1621 + };
1622 + run_sanitize_test(&t);
1623 + }
1624 +
1625 + // Bell character (0x07) - common in terminal output
1626 + {
1627 + unsigned char input[] = {'b', 'e', 'l', 'l', 0x07, 't', 'e', 's', 't', '\0'};
1628 + sanitize_test_t t = {
1629 + .name = "ctrl_bell",
1630 + .input = input,
1631 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1632 + .expected_output = "bell test", .expected_len = 9, .expected_mblen = 9
1633 + };
1634 + run_sanitize_test(&t);
1635 + }
1636 +
1637 + // Escape sequence (0x1B) - ANSI escape
1638 + {
1639 + unsigned char input[] = {0x1B, '[', '3', '1', 'm', 'r', 'e', 'd', 0x1B, '[', '0', 'm', '\0'};
1640 + sanitize_test_t t = {
1641 + .name = "ctrl_ansi_escape",
1642 + .input = input,
1643 + .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1644 + .expected_output = "[31mred [0m", .expected_len = 11, .expected_mblen = 11
1645 + // 0x1B becomes space, which is leading/duplicated so gets handled
1646 + };
1647 + run_sanitize_test(&t);
1648 + }
1649 +}
1650 +
1651 +// ============================================================================
1652 +// TEST: REAL-WORLD METRIC STRINGS
1653 +// ============================================================================
1654 +
1655 +static void test_real_world_metrics(void) {
1656 + fprintf(stderr, "\n=== Real-World Metric Strings ===\n");
1657 +
1658 + // Use actual rrd_string_allowed_chars
1659 + extern unsigned char rrd_string_allowed_chars[256];
1660 +
1661 + // CPU metric title
1662 + {
1663 + sanitize_test_t t = {
1664 + .name = "metric_cpu_title",
1665 + .input = (unsigned char *)"CPU utilization (user, system, iowait, irq, softirq)",
1666 + .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1667 + .expected_output = "CPU utilization (user, system, iowait, irq, softirq)", .expected_len = 52, .expected_mblen = 52
1668 + };
1669 + run_sanitize_test(&t);
1670 + }
1671 +
1672 + // Memory with units
1673 + {
1674 + sanitize_test_t t = {
1675 + .name = "metric_memory_unit",
1676 + .input = (unsigned char *)"Memory (MiB)",
1677 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1678 + .expected_output = "Memory (MiB)", .expected_len = 12, .expected_mblen = 12
1679 + };
1680 + run_sanitize_test(&t);
1681 + }
1682 +
1683 + // Network bandwidth with special chars
1684 + {
1685 + sanitize_test_t t = {
1686 + .name = "metric_network_bandwidth",
1687 + .input = (unsigned char *)"eth0: kilobits/s",
1688 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1689 + .expected_output = "eth0: kilobits/s", .expected_len = 16, .expected_mblen = 16
1690 + };
1691 + run_sanitize_test(&t);
1692 + }
1693 +
1694 + // Disk I/O with latency units (microseconds)
1695 + {
1696 + sanitize_test_t t = {
1697 + .name = "metric_disk_latency",
1698 + .input = (unsigned char *)"Disk latency (\xC2\xB5s)", // µs
1699 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1700 + .expected_output = "Disk latency (\xC2\xB5s)", .expected_len = 18, .expected_mblen = 17
1701 + };
1702 + run_sanitize_test(&t);
1703 + }
1704 +
1705 + // Temperature sensor
1706 + {
1707 + sanitize_test_t t = {
1708 + .name = "metric_temperature",
1709 + .input = (unsigned char *)"core_temp_0: \xC2\xB0""Celsius",
1710 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1711 + // "core_temp_0: " (13) + "°" (2 bytes) + "Celsius" (7) = 22 bytes, 21 chars
1712 + .expected_output = "core_temp_0: \xC2\xB0""Celsius", .expected_len = 22, .expected_mblen = 21
1713 + };
1714 + run_sanitize_test(&t);
1715 + }
1716 +
1717 + // Docker container ID (common in Netdata)
1718 + {
1719 + sanitize_test_t t = {
1720 + .name = "metric_docker_id",
1721 + .input = (unsigned char *)"container_a1b2c3d4e5f6",
1722 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1723 + .expected_output = "container_a1b2c3d4e5f6", .expected_len = 22, .expected_mblen = 22
1724 + };
1725 + run_sanitize_test(&t);
1726 + }
1727 +
1728 + // Kubernetes pod name
1729 + {
1730 + sanitize_test_t t = {
1731 + .name = "metric_k8s_pod",
1732 + .input = (unsigned char *)"nginx-deployment-5d8b7f9-xyz12",
1733 + .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1734 + .expected_output = "nginx-deployment-5d8b7f9-xyz12", .expected_len = 30, .expected_mblen = 30
1735 + };
1736 + run_sanitize_test(&t);
1737 + }
1738 +
1739 + // Windows path (backslash conversion)
1740 + {
1741 + sanitize_test_t t = {
1742 + .name = "metric_windows_path",
1743 + .input = (unsigned char *)"C:\\Program Files\\Application\\metric.exe",
1744 + .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1745 + .expected_output = "C:/Program Files/Application/metric.exe", .expected_len = 39, .expected_mblen = 39
1746 + };
1747 + run_sanitize_test(&t);
1748 + }
1749 +
1750 + // Prometheus metric with labels (quotes converted)
1751 + {
1752 + sanitize_test_t t = {
1753 + .name = "metric_prometheus_labels",
1754 + .input = (unsigned char *)"http_requests{method=\"POST\",status=\"200\"}",
1755 + .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1756 + .expected_output = "http_requests{method='POST',status='200'}", .expected_len = 41, .expected_mblen = 41
1757 + };
1758 + run_sanitize_test(&t);
1759 + }
1760 +
1761 + // Acceleration units (m/s²)
1762 + {
1763 + sanitize_test_t t = {
1764 + .name = "metric_acceleration",
1765 + .input = (unsigned char *)"Acceleration (m/s\xC2\xB2)",
1766 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1767 + .expected_output = "Acceleration (m/s\xC2\xB2)", .expected_len = 20, .expected_mblen = 19
1768 + };
1769 + run_sanitize_test(&t);
1770 + }
1771 +
1772 + // Percentage with degree
1773 + {
1774 + sanitize_test_t t = {
1775 + .name = "metric_angle_degree",
1776 + .input = (unsigned char *)"Rotation angle: 90\xC2\xB0",
1777 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1778 + // "Rotation angle: 90" (18) + "°" (2 bytes) = 20 bytes, 19 chars
1779 + .expected_output = "Rotation angle: 90\xC2\xB0", .expected_len = 20, .expected_mblen = 19
1780 + };
1781 + run_sanitize_test(&t);
1782 + }
1783 +
1784 + // IPv6 address in metric context
1785 + {
1786 + sanitize_test_t t = {
1787 + .name = "metric_ipv6",
1788 + .input = (unsigned char *)"host:2001:db8::1",
1789 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1790 + .expected_output = "host:2001:db8::1", .expected_len = 16, .expected_mblen = 16
1791 + };
1792 + run_sanitize_test(&t);
1793 + }
1794 +
1795 + // Process name with parentheses and numbers
1796 + {
1797 + sanitize_test_t t = {
1798 + .name = "metric_process_name",
1799 + .input = (unsigned char *)"python3.11 (worker-1)",
1800 + .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1801 + .expected_output = "python3.11 (worker-1)", .expected_len = 21, .expected_mblen = 21
1802 + };
1803 + run_sanitize_test(&t);
1804 + }
1805 +}
1806 +
1807 +// ============================================================================
1808 +// TEST: HEX ENCODING EDGE CASES
1809 +// ============================================================================
1810 +
1811 +static void test_hex_encoding_edge_cases(void) {
1812 + fprintf(stderr, "\n=== Hex Encoding Edge Cases ===\n");
1813 +
1814 + // Truncated 2-byte sequence - gets hex encoded
1815 + // mblen counts the whole invalid sequence as 1 character
1816 + {
1817 + sanitize_test_t t = {
1818 + .name = "hex_truncated_2byte",
1819 + .input = (unsigned char *)"\xC2", // Truncated (needs continuation)
1820 + .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
1821 + .expected_output = "c2", .expected_len = 2, .expected_mblen = 1
1822 + };
1823 + run_sanitize_test(&t);
1824 + }
1825 +
1826 + // Not enough space for hex encoding
1827 + {
1828 + sanitize_test_t t = {
1829 + .name = "hex_no_space",
1830 + .input = (unsigned char *)"\xC2", // Truncated
1831 + .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
1832 + // Can't fit "c2" (needs 2 chars + NUL = 3) - nothing written, mblen=0
1833 + .expected_output = "", .expected_len = 0, .expected_mblen = 0
1834 + };
1835 + run_sanitize_test(&t);
1836 + }
1837 +
1838 + // Multiple truncated sequences - each counts as 1 mblen
1839 + {
1840 + sanitize_test_t t = {
1841 + .name = "hex_multiple_truncated",
1842 + .input = (unsigned char *)"\xC2\xC3", // Two truncated 2-byte starts
1843 + .dst_size = 5, .char_map = identity_char_map, .utf = true, .empty = "",
1844 + .expected_output = "c2c3", .expected_len = 4, .expected_mblen = 2
1845 + };
1846 + run_sanitize_test(&t);
1847 + }
1848 +
1849 + // ASCII + truncated hex
1850 + {
1851 + sanitize_test_t t = {
1852 + .name = "hex_ascii_plus_truncated",
1853 + .input = (unsigned char *)"AB\xC2", // AB + truncated
1854 + .dst_size = 5, .char_map = identity_char_map, .utf = true, .empty = "",
1855 + .expected_output = "ABc2", .expected_len = 4, .expected_mblen = 3
1856 + };
1857 + run_sanitize_test(&t);
1858 + }
1859 +
1860 + // 0xFE and 0xFF don't match valid UTF-8 start patterns - hex encoded
1861 + {
1862 + sanitize_test_t t = {
1863 + .name = "hex_FE_FF",
1864 + .input = (unsigned char *)"\xFE\xFF",
1865 + .dst_size = 8, .char_map = identity_char_map, .utf = true, .empty = "",
1866 + .expected_output = "feff", .expected_len = 4, .expected_mblen = 2
1867 + };
1868 + run_sanitize_test(&t);
1869 + }
1870 +
1871 + // Structurally valid overlong + orphan continuation
1872 + {
1873 + sanitize_test_t t = {
1874 + .name = "hex_valid_plus_orphan",
1875 + .input = (unsigned char *)"X\xC0\x80\x80", // X + valid 2-byte + orphan
1876 + .dst_size = 8, .char_map = identity_char_map, .utf = true, .empty = "",
1877 + // \xC0\x80 is structurally valid (passes through), \x80 is orphan (char_map)
1878 + .expected_output = "X\xC0\x80\x80", .expected_len = 4, .expected_mblen = 3
1879 + };
1880 + run_sanitize_test(&t);
1881 + }
1882 +
1883 + // Orphan continuation bytes go through char_map
1884 + {
1885 + sanitize_test_t t = {
1886 + .name = "hex_orphan_continuations",
1887 + .input = (unsigned char *)"\x80\x81\x82\x83",
1888 + .dst_size = 16, .char_map = identity_char_map, .utf = true, .empty = "",
1889 + // Orphan continuation bytes (10xxxxxx pattern) go through char_map
1890 + .expected_output = "\x80\x81\x82\x83", .expected_len = 4, .expected_mblen = 4
1891 + };
1892 + run_sanitize_test(&t);
1893 + }
1894 +}
1895 +
1896 +// ============================================================================
1897 +// TEST: STRESS AND EDGE CASES
1898 +// ============================================================================
1899 +
1900 +static void test_stress_and_edge_cases(void) {
1901 + fprintf(stderr, "\n=== Stress and Edge Cases ===\n");
1902 +
1903 + // Very long UTF-8 string
1904 + {
1905 + // Create string with 100 2-byte UTF-8 characters (200 bytes)
1906 + unsigned char input[201];
1907 + for (int i = 0; i < 100; i++) {
1908 + input[i*2] = 0xC2;
1909 + input[i*2+1] = 0xB0; // ° repeated 100 times
1910 + }
1911 + input[200] = '\0';
1912 +
1913 + unsigned char expected[201];
1914 + memcpy(expected, input, 201);
1915 +
1916 + sanitize_test_t t = {
1917 + .name = "stress_long_utf8",
1918 + .input = input,
1919 + .dst_size = 256, .char_map = identity_char_map, .utf = true, .empty = "",
1920 + .expected_output = (char *)expected, .expected_len = 200, .expected_mblen = 100
1921 + };
1922 + run_sanitize_test(&t);
1923 + }
1924 +
1925 + // Alternating valid UTF-8 and structurally valid overlong (both pass through)
1926 + {
1927 + sanitize_test_t t = {
1928 + .name = "stress_alternating",
1929 + .input = (unsigned char *)"\xC2\xB0\xC0\x80\xC2\xB0\xC0\x80",
1930 + .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1931 + // Both \xC2\xB0 and \xC0\x80 are structurally valid 2-byte sequences
1932 + .expected_output = "\xC2\xB0\xC0\x80\xC2\xB0\xC0\x80", .expected_len = 8, .expected_mblen = 4
1933 + };
1934 + run_sanitize_test(&t);
1935 + }
1936 +
1937 + // All 256 byte values (non-UTF-8 mode)
1938 + {
1939 + unsigned char input[256];
1940 + for (int i = 1; i < 256; i++) // Skip NUL
1941 + input[i-1] = (unsigned char)i;
1942 + input[255] = '\0';
1943 +
1944 + // With identity map, most pass through; control chars and high bytes
1945 + // will be handled. This just tests no crash.
1946 + unsigned char dst[512];
1947 + size_t len = text_sanitize(dst, input, sizeof(dst), identity_char_map, false, "", NULL);
1948 + TEST_ASSERT("stress_all_bytes", len > 0, "Expected non-zero length, got %zu", len);
1949 + }
1950 +
1951 + // Rapid buffer size changes (fuzz-like)
1952 + {
1953 + const unsigned char *input = (unsigned char *)"test\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80";
1954 + bool all_ok = true;
1955 +
1956 + for (size_t sz = 1; sz <= 20; sz++) {
1957 + unsigned char *buffer = callocz(1, sz + 32);
1958 + unsigned char *dst = buffer + 16;
1959 + memset(buffer, 0xAA, 16);
1960 + memset(dst + sz, 0xBB, 16);
1961 +
1962 + text_sanitize(dst, input, sz, identity_char_map, true, "", NULL);
1963 +
1964 + // Check for overflow
1965 + for (size_t i = 0; i < 16; i++) {
1966 + if (buffer[i] != 0xAA || dst[sz + i] != 0xBB) {
1967 + all_ok = false;
1968 + fprintf(stderr, " Overflow at buffer size %zu\n", sz);
1969 + break;
1970 + }
1971 + }
1972 + freez(buffer);
1973 + }
1974 + TEST_ASSERT("stress_buffer_sizes", all_ok, "Buffer overflow detected in size sweep");
1975 + }
1976 +
1977 + // Repeated sanitization (idempotent for valid input)
1978 + {
1979 + unsigned char input[] = "test\xC2\xB0""C";
1980 + unsigned char dst1[32], dst2[32];
1981 +
1982 + text_sanitize(dst1, input, sizeof(dst1), identity_char_map, true, "", NULL);
1983 + text_sanitize(dst2, dst1, sizeof(dst2), identity_char_map, true, "", NULL);
1984 +
1985 + TEST_ASSERT("stress_idempotent", strcmp((char *)dst1, (char *)dst2) == 0,
1986 + "Not idempotent: '%s' vs '%s'", dst1, dst2);
1987 + }
1988 +}
1989 +
1990 +// ============================================================================
1991 +// MAIN TEST RUNNER
1992 +// ============================================================================
1993 +
1994 +int utf8_sanitizer_unittest(void) {
1995 + fprintf(stderr, "\n");
1996 + fprintf(stderr, "================================================================\n");
1997 + fprintf(stderr, "UTF-8 Sanitizer Exhaustive Unit Tests\n");
1998 + fprintf(stderr, "================================================================\n");
1999 +
2000 + init_char_maps();
2001 +
2002 + test_valid_utf8_sequences();
2003 + test_invalid_utf8_sequences();
2004 + test_buffer_boundaries();
2005 + test_char_map_transformations();
2006 + test_space_handling();
2007 + test_empty_and_special();
2008 + test_utf_parameter();
2009 + test_multibyte_length();
2010 + test_rrd_string_allowed_chars();
2011 + test_security_cases();
2012 + test_regression_fixed_bugs();
2013 + test_all_control_characters();
2014 + test_real_world_metrics();
2015 + test_hex_encoding_edge_cases();
2016 + test_stress_and_edge_cases();
2017 +
2018 + fprintf(stderr, "\n================================================================\n");
2019 + fprintf(stderr, "Tests run: %d, Passed: %d, Failed: %d\n", tests_run, tests_passed, tests_failed);
2020 +
2021 + if (tests_failed == 0) {
2022 + fprintf(stderr, "ALL TESTS PASSED\n");
2023 + } else {
2024 + fprintf(stderr, "SOME TESTS FAILED\n");
2025 + }
2026 + fprintf(stderr, "================================================================\n\n");
2027 +
2028 + return tests_failed;
2029 +}