1
+// SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+#include "../libnetdata.h"
4
+
5
+// ============================================================================
6
+// TEST INFRASTRUCTURE
7
+// ============================================================================
8
+
9
+static int tests_run = 0;
10
+static int tests_passed = 0;
11
+static int tests_failed = 0;
12
+
13
+// Identity char_map - everything passes through unchanged
14
+static unsigned char identity_char_map[256];
15
+
16
+// Test char_map similar to rrd_string_allowed_chars
17
+static unsigned char test_rrd_char_map[256];
18
+
19
+static void init_char_maps(void) {
20
+ // Identity map
21
+ for (int i = 0; i < 256; i++)
22
+ identity_char_map[i] = (unsigned char)i;
23
+ identity_char_map[0] = '\0';
24
+
25
+ // RRD-like map
26
+ for (int i = 0; i < 256; i++)
27
+ test_rrd_char_map[i] = (unsigned char)i;
28
+
29
+ // Control characters (0-31, 127) → space
30
+ for (int i = 1; i < 32; i++)
31
+ test_rrd_char_map[i] = ' ';
32
+ test_rrd_char_map[127] = ' ';
33
+
34
+ // High bytes (128-255) → space (fallback for orphan UTF-8 bytes)
35
+ for (int i = 128; i < 256; i++)
36
+ test_rrd_char_map[i] = ' ';
37
+
38
+ test_rrd_char_map[0] = '\0';
39
+ test_rrd_char_map['"'] = '\''; // double quote → single quote
40
+ test_rrd_char_map['\\'] = '/'; // backslash → forward slash
41
+}
42
+
43
+// Test result macro
44
+#define TEST_ASSERT(name, condition, ...) do { \
45
+ tests_run++; \
46
+ if (condition) { \
47
+ tests_passed++; \
48
+ } else { \
49
+ tests_failed++; \
50
+ fprintf(stderr, "FAILED [%s]: ", name); \
51
+ fprintf(stderr, __VA_ARGS__); \
52
+ fprintf(stderr, "\n"); \
53
+ } \
54
+} while(0)
55
+
56
+// Helper to run a single sanitize test with overflow detection
57
+typedef struct {
58
+ const char *name;
59
+ const unsigned char *input;
60
+ size_t dst_size;
61
+ const unsigned char *char_map;
62
+ bool utf;
63
+ const char *empty;
64
+ const char *expected_output;
65
+ size_t expected_len;
66
+ size_t expected_mblen;
67
+} sanitize_test_t;
68
+
69
+static void run_sanitize_test(const sanitize_test_t *t) {
70
+ // Allocate with guard bytes
71
+ size_t guard = 16;
72
+ unsigned char *buffer = callocz(1, t->dst_size + guard * 2);
73
+ unsigned char *dst = buffer + guard;
74
+
75
+ // Fill guards
76
+ memset(buffer, 0xAA, guard);
77
+ memset(dst + t->dst_size, 0xBB, guard);
78
+ memset(dst, 0xCC, t->dst_size);
79
+
80
+ size_t mblen = 0;
81
+ size_t len = text_sanitize(dst, t->input, t->dst_size, t->char_map, t->utf, t->empty, &mblen);
82
+
83
+ // Check overflow
84
+ bool overflow_before = false, overflow_after = false;
85
+ for (size_t i = 0; i < guard; i++) {
86
+ if (buffer[i] != 0xAA) overflow_before = true;
87
+ if (dst[t->dst_size + i] != 0xBB) overflow_after = true;
88
+ }
89
+
90
+ TEST_ASSERT(t->name, !overflow_before && !overflow_after,
91
+ "Buffer overflow! before=%d after=%d", overflow_before, overflow_after);
92
+
93
+ TEST_ASSERT(t->name, len == t->expected_len,
94
+ "Length mismatch: expected %zu, got %zu", t->expected_len, len);
95
+
96
+ TEST_ASSERT(t->name, strcmp((char *)dst, t->expected_output) == 0,
97
+ "Content mismatch: expected '%s', got '%s'", t->expected_output, dst);
98
+
99
+ if (t->expected_mblen > 0) {
100
+ TEST_ASSERT(t->name, mblen == t->expected_mblen,
101
+ "Multibyte length mismatch: expected %zu, got %zu", t->expected_mblen, mblen);
102
+ }
103
+
104
+ // Verify null termination
105
+ TEST_ASSERT(t->name, dst[len] == '\0',
106
+ "Missing null terminator at position %zu", len);
107
+
108
+ freez(buffer);
109
+}
110
+
111
+// ============================================================================
112
+// TEST: VALID UTF-8 SEQUENCES
113
+// ============================================================================
114
+
115
+static void test_valid_utf8_sequences(void) {
116
+ fprintf(stderr, "\n=== Valid UTF-8 Sequences ===\n");
117
+
118
+ // 2-byte: Latin characters with diacritics
119
+ {
120
+ sanitize_test_t t = {
121
+ .name = "utf8_2byte_e_acute",
122
+ .input = (unsigned char *)"caf\xC3\xA9", // café
123
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
124
+ .expected_output = "caf\xC3\xA9", .expected_len = 5, .expected_mblen = 4
125
+ };
126
+ run_sanitize_test(&t);
127
+ }
128
+
129
+ // 2-byte: Superscript ² (U+00B2)
130
+ {
131
+ sanitize_test_t t = {
132
+ .name = "utf8_2byte_superscript2",
133
+ .input = (unsigned char *)"m/s\xC2\xB2", // m/s²
134
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
135
+ .expected_output = "m/s\xC2\xB2", .expected_len = 5, .expected_mblen = 4
136
+ };
137
+ run_sanitize_test(&t);
138
+ }
139
+
140
+ // 2-byte: Degree symbol ° (U+00B0)
141
+ {
142
+ sanitize_test_t t = {
143
+ .name = "utf8_2byte_degree",
144
+ .input = (unsigned char *)"25\xC2\xB0""C", // 25°C
145
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
146
+ .expected_output = "25\xC2\xB0""C", .expected_len = 5, .expected_mblen = 4
147
+ };
148
+ run_sanitize_test(&t);
149
+ }
150
+
151
+ // 2-byte: Micro sign µ (U+00B5)
152
+ {
153
+ sanitize_test_t t = {
154
+ .name = "utf8_2byte_micro",
155
+ .input = (unsigned char *)"\xC2\xB5s", // µs
156
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
157
+ .expected_output = "\xC2\xB5s", .expected_len = 3, .expected_mblen = 2
158
+ };
159
+ run_sanitize_test(&t);
160
+ }
161
+
162
+ // 3-byte: Euro sign € (U+20AC)
163
+ {
164
+ sanitize_test_t t = {
165
+ .name = "utf8_3byte_euro",
166
+ .input = (unsigned char *)"100\xE2\x82\xAC", // 100€
167
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
168
+ .expected_output = "100\xE2\x82\xAC", .expected_len = 6, .expected_mblen = 4
169
+ };
170
+ run_sanitize_test(&t);
171
+ }
172
+
173
+ // 3-byte: Japanese hiragana あ (U+3042)
174
+ {
175
+ sanitize_test_t t = {
176
+ .name = "utf8_3byte_hiragana",
177
+ .input = (unsigned char *)"\xE3\x81\x82", // あ
178
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
179
+ .expected_output = "\xE3\x81\x82", .expected_len = 3, .expected_mblen = 1
180
+ };
181
+ run_sanitize_test(&t);
182
+ }
183
+
184
+ // 3-byte: Chinese character 中 (U+4E2D)
185
+ {
186
+ sanitize_test_t t = {
187
+ .name = "utf8_3byte_chinese",
188
+ .input = (unsigned char *)"\xE4\xB8\xAD\xE6\x96\x87", // 中文
189
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
190
+ .expected_output = "\xE4\xB8\xAD\xE6\x96\x87", .expected_len = 6, .expected_mblen = 2
191
+ };
192
+ run_sanitize_test(&t);
193
+ }
194
+
195
+ // 4-byte: Emoji 😀 (U+1F600)
196
+ {
197
+ sanitize_test_t t = {
198
+ .name = "utf8_4byte_emoji",
199
+ .input = (unsigned char *)"hi\xF0\x9F\x98\x80", // hi😀
200
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
201
+ .expected_output = "hi\xF0\x9F\x98\x80", .expected_len = 6, .expected_mblen = 3
202
+ };
203
+ run_sanitize_test(&t);
204
+ }
205
+
206
+ // 4-byte: Mathematical bold A 𝐀 (U+1D400)
207
+ {
208
+ sanitize_test_t t = {
209
+ .name = "utf8_4byte_math",
210
+ .input = (unsigned char *)"\xF0\x9D\x90\x80", // 𝐀
211
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
212
+ .expected_output = "\xF0\x9D\x90\x80", .expected_len = 4, .expected_mblen = 1
213
+ };
214
+ run_sanitize_test(&t);
215
+ }
216
+
217
+ // Mixed: ASCII + 2-byte + 3-byte + 4-byte
218
+ {
219
+ sanitize_test_t t = {
220
+ .name = "utf8_mixed_all_types",
221
+ .input = (unsigned char *)"A\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80", // A°€😀
222
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
223
+ .expected_output = "A\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80", .expected_len = 10, .expected_mblen = 4
224
+ };
225
+ run_sanitize_test(&t);
226
+ }
227
+
228
+ // Multiple same-type UTF-8 characters
229
+ {
230
+ sanitize_test_t t = {
231
+ .name = "utf8_multiple_2byte",
232
+ .input = (unsigned char *)"\xC3\xA9\xC3\xA8\xC3\xA0", // éèà
233
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
234
+ .expected_output = "\xC3\xA9\xC3\xA8\xC3\xA0", .expected_len = 6, .expected_mblen = 3
235
+ };
236
+ run_sanitize_test(&t);
237
+ }
238
+
239
+ // UTF-8 at beginning of string
240
+ {
241
+ sanitize_test_t t = {
242
+ .name = "utf8_at_beginning",
243
+ .input = (unsigned char *)"\xC2\xB5sec",
244
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
245
+ .expected_output = "\xC2\xB5sec", .expected_len = 5, .expected_mblen = 4
246
+ };
247
+ run_sanitize_test(&t);
248
+ }
249
+
250
+ // UTF-8 in middle of string
251
+ {
252
+ sanitize_test_t t = {
253
+ .name = "utf8_in_middle",
254
+ .input = (unsigned char *)"pre\xC2\xB0post",
255
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
256
+ .expected_output = "pre\xC2\xB0post", .expected_len = 9, .expected_mblen = 8
257
+ };
258
+ run_sanitize_test(&t);
259
+ }
260
+
261
+ // UTF-8 at end of string
262
+ {
263
+ sanitize_test_t t = {
264
+ .name = "utf8_at_end",
265
+ .input = (unsigned char *)"temp\xC2\xB0",
266
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
267
+ .expected_output = "temp\xC2\xB0", .expected_len = 6, .expected_mblen = 5
268
+ };
269
+ run_sanitize_test(&t);
270
+ }
271
+
272
+ // Boundary: Minimum 2-byte (U+0080)
273
+ {
274
+ sanitize_test_t t = {
275
+ .name = "utf8_2byte_min",
276
+ .input = (unsigned char *)"\xC2\x80",
277
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
278
+ .expected_output = "\xC2\x80", .expected_len = 2, .expected_mblen = 1
279
+ };
280
+ run_sanitize_test(&t);
281
+ }
282
+
283
+ // Boundary: Maximum 2-byte (U+07FF)
284
+ {
285
+ sanitize_test_t t = {
286
+ .name = "utf8_2byte_max",
287
+ .input = (unsigned char *)"\xDF\xBF",
288
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
289
+ .expected_output = "\xDF\xBF", .expected_len = 2, .expected_mblen = 1
290
+ };
291
+ run_sanitize_test(&t);
292
+ }
293
+
294
+ // Boundary: Minimum 3-byte (U+0800)
295
+ {
296
+ sanitize_test_t t = {
297
+ .name = "utf8_3byte_min",
298
+ .input = (unsigned char *)"\xE0\xA0\x80",
299
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
300
+ .expected_output = "\xE0\xA0\x80", .expected_len = 3, .expected_mblen = 1
301
+ };
302
+ run_sanitize_test(&t);
303
+ }
304
+
305
+ // Boundary: Maximum 3-byte (U+FFFF, excluding surrogates)
306
+ {
307
+ sanitize_test_t t = {
308
+ .name = "utf8_3byte_max",
309
+ .input = (unsigned char *)"\xEF\xBF\xBD", // U+FFFD replacement char
310
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
311
+ .expected_output = "\xEF\xBF\xBD", .expected_len = 3, .expected_mblen = 1
312
+ };
313
+ run_sanitize_test(&t);
314
+ }
315
+
316
+ // Boundary: Minimum 4-byte (U+10000)
317
+ {
318
+ sanitize_test_t t = {
319
+ .name = "utf8_4byte_min",
320
+ .input = (unsigned char *)"\xF0\x90\x80\x80",
321
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
322
+ .expected_output = "\xF0\x90\x80\x80", .expected_len = 4, .expected_mblen = 1
323
+ };
324
+ run_sanitize_test(&t);
325
+ }
326
+
327
+ // Boundary: Maximum valid 4-byte (U+10FFFF)
328
+ {
329
+ sanitize_test_t t = {
330
+ .name = "utf8_4byte_max",
331
+ .input = (unsigned char *)"\xF4\x8F\xBF\xBF",
332
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
333
+ .expected_output = "\xF4\x8F\xBF\xBF", .expected_len = 4, .expected_mblen = 1
334
+ };
335
+ run_sanitize_test(&t);
336
+ }
337
+}
338
+
339
+// ============================================================================
340
+// TEST: INVALID UTF-8 SEQUENCES
341
+// ============================================================================
342
+
343
+static void test_invalid_utf8_sequences(void) {
344
+ fprintf(stderr, "\n=== Invalid UTF-8 Sequences ===\n");
345
+
346
+ // Orphan continuation byte (0x80-0xBF without start byte)
347
+ {
348
+ sanitize_test_t t = {
349
+ .name = "invalid_orphan_continuation",
350
+ .input = (unsigned char *)"A\x80""B",
351
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
352
+ .expected_output = "A B", .expected_len = 3, .expected_mblen = 3
353
+ };
354
+ run_sanitize_test(&t);
355
+ }
356
+
357
+ // Multiple orphan continuation bytes
358
+ {
359
+ sanitize_test_t t = {
360
+ .name = "invalid_multiple_orphan",
361
+ .input = (unsigned char *)"\x80\x81\x82",
362
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
363
+ .expected_output = "", .expected_len = 0, .expected_mblen = 0 // All become spaces, trimmed
364
+ };
365
+ run_sanitize_test(&t);
366
+ }
367
+
368
+ // Overlong 0xC0 (structurally valid 2-byte, semantically invalid)
369
+ // NOTE: Function does structural validation only - passes through
370
+ {
371
+ sanitize_test_t t = {
372
+ .name = "overlong_C0_structural_valid",
373
+ .input = (unsigned char *)"X\xC0\x80Y",
374
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
375
+ .expected_output = "X\xC0\x80Y", .expected_len = 4, .expected_mblen = 3
376
+ };
377
+ run_sanitize_test(&t);
378
+ }
379
+
380
+ // Overlong 0xC1 (structurally valid 2-byte)
381
+ {
382
+ sanitize_test_t t = {
383
+ .name = "overlong_C1_structural_valid",
384
+ .input = (unsigned char *)"\xC1\xBF",
385
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
386
+ .expected_output = "\xC1\xBF", .expected_len = 2, .expected_mblen = 1
387
+ };
388
+ run_sanitize_test(&t);
389
+ }
390
+
391
+ // 0xF5 with continuation bytes (structurally valid 4-byte, but beyond Unicode)
392
+ {
393
+ sanitize_test_t t = {
394
+ .name = "out_of_range_F5_structural_valid",
395
+ .input = (unsigned char *)"\xF5\x80\x80\x80",
396
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
397
+ .expected_output = "\xF5\x80\x80\x80", .expected_len = 4, .expected_mblen = 1
398
+ };
399
+ run_sanitize_test(&t);
400
+ }
401
+
402
+ // 0xFF alone - not a valid start byte pattern, gets hex encoded
403
+ {
404
+ sanitize_test_t t = {
405
+ .name = "invalid_FF_hex_encoded",
406
+ .input = (unsigned char *)"A\xFF""B",
407
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
408
+ .expected_output = "AffB", .expected_len = 4, .expected_mblen = 3
409
+ };
410
+ run_sanitize_test(&t);
411
+ }
412
+
413
+ // Truncated 2-byte sequence at end - hex encoded
414
+ {
415
+ sanitize_test_t t = {
416
+ .name = "truncated_2byte_hex",
417
+ .input = (unsigned char *)"abc\xC2",
418
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
419
+ .expected_output = "abcc2", .expected_len = 5, .expected_mblen = 4
420
+ };
421
+ run_sanitize_test(&t);
422
+ }
423
+
424
+ // Truncated 3-byte sequence (only 1 continuation) - hex encoded
425
+ {
426
+ sanitize_test_t t = {
427
+ .name = "truncated_3byte_1cont_hex",
428
+ .input = (unsigned char *)"X\xE2\x82",
429
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
430
+ .expected_output = "Xe282", .expected_len = 5, .expected_mblen = 2
431
+ };
432
+ run_sanitize_test(&t);
433
+ }
434
+
435
+ // Truncated 3-byte sequence (no continuation) - hex encoded
436
+ {
437
+ sanitize_test_t t = {
438
+ .name = "truncated_3byte_0cont_hex",
439
+ .input = (unsigned char *)"Y\xE2",
440
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
441
+ .expected_output = "Ye2", .expected_len = 3, .expected_mblen = 2
442
+ };
443
+ run_sanitize_test(&t);
444
+ }
445
+
446
+ // Truncated 4-byte sequence - hex encoded
447
+ {
448
+ sanitize_test_t t = {
449
+ .name = "truncated_4byte_hex",
450
+ .input = (unsigned char *)"\xF0\x9F\x98",
451
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
452
+ .expected_output = "f09f98", .expected_len = 6, .expected_mblen = 1
453
+ };
454
+ run_sanitize_test(&t);
455
+ }
456
+
457
+ // Wrong continuation byte (ASCII instead of 0x80-0xBF) - hex encoded
458
+ {
459
+ sanitize_test_t t = {
460
+ .name = "wrong_continuation_ascii_hex",
461
+ .input = (unsigned char *)"\xC2X",
462
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
463
+ .expected_output = "c2X", .expected_len = 3, .expected_mblen = 2
464
+ };
465
+ run_sanitize_test(&t);
466
+ }
467
+
468
+ // Wrong continuation byte (another start byte) - first hex encoded, second valid
469
+ {
470
+ sanitize_test_t t = {
471
+ .name = "wrong_continuation_start_hex",
472
+ .input = (unsigned char *)"\xC2\xC2\x80", // Second C2 is wrong, should be 80-BF
473
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
474
+ .expected_output = "c2\xC2\x80", .expected_len = 4, .expected_mblen = 2
475
+ };
476
+ run_sanitize_test(&t);
477
+ }
478
+
479
+ // Overlong NUL (structurally valid, security concern but passed through)
480
+ {
481
+ sanitize_test_t t = {
482
+ .name = "overlong_nul_structural_valid",
483
+ .input = (unsigned char *)"\xC0\x80",
484
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
485
+ .expected_output = "\xC0\x80", .expected_len = 2, .expected_mblen = 1
486
+ };
487
+ run_sanitize_test(&t);
488
+ }
489
+
490
+ // Overlong space (structurally valid 3-byte)
491
+ {
492
+ sanitize_test_t t = {
493
+ .name = "overlong_space_structural_valid",
494
+ .input = (unsigned char *)"\xE0\x80\xA0", // Overlong space
495
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
496
+ .expected_output = "\xE0\x80\xA0", .expected_len = 3, .expected_mblen = 1
497
+ };
498
+ run_sanitize_test(&t);
499
+ }
500
+
501
+ // UTF-16 surrogate (invalid in UTF-8)
502
+ {
503
+ sanitize_test_t t = {
504
+ .name = "invalid_surrogate_high",
505
+ .input = (unsigned char *)"\xED\xA0\x80", // U+D800 high surrogate
506
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
507
+ .expected_output = "\xED\xA0\x80", .expected_len = 3, .expected_mblen = 1
508
+ // Note: Current implementation doesn't reject surrogates (structural only)
509
+ };
510
+ run_sanitize_test(&t);
511
+ }
512
+
513
+ // Out of range (beyond U+10FFFF)
514
+ {
515
+ sanitize_test_t t = {
516
+ .name = "invalid_out_of_range",
517
+ .input = (unsigned char *)"\xF4\x90\x80\x80", // U+110000 (invalid)
518
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
519
+ .expected_output = "\xF4\x90\x80\x80", .expected_len = 4, .expected_mblen = 1
520
+ // Note: Current implementation doesn't reject out of range (structural only)
521
+ };
522
+ run_sanitize_test(&t);
523
+ }
524
+
525
+ // Mixed valid UTF-8 and structurally valid overlong
526
+ {
527
+ sanitize_test_t t = {
528
+ .name = "mixed_valid_and_overlong",
529
+ .input = (unsigned char *)"A\xC2\xB0\xC0\x80\xE2\x82\xAC", // A° + overlong + €
530
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
531
+ // All are structurally valid, so all pass through
532
+ .expected_output = "A\xC2\xB0\xC0\x80\xE2\x82\xAC", .expected_len = 8, .expected_mblen = 4
533
+ };
534
+ run_sanitize_test(&t);
535
+ }
536
+}
537
+
538
+// ============================================================================
539
+// TEST: BUFFER BOUNDARY CONDITIONS
540
+// ============================================================================
541
+
542
+static void test_buffer_boundaries(void) {
543
+ fprintf(stderr, "\n=== Buffer Boundary Conditions ===\n");
544
+
545
+ // dst_size = 0
546
+ {
547
+ unsigned char dst[16] = {0xCC, 0xCC, 0xCC, 0xCC};
548
+ size_t len = text_sanitize(dst, (unsigned char *)"hello", 0, identity_char_map, true, "", NULL);
549
+ TEST_ASSERT("buffer_size_0", len == 0, "Expected 0, got %zu", len);
550
+ TEST_ASSERT("buffer_size_0_unchanged", dst[0] == 0xCC, "Buffer was modified");
551
+ }
552
+
553
+ // dst_size = 1 (only null terminator fits)
554
+ {
555
+ sanitize_test_t t = {
556
+ .name = "buffer_size_1",
557
+ .input = (unsigned char *)"hello",
558
+ .dst_size = 1, .char_map = identity_char_map, .utf = true, .empty = "",
559
+ .expected_output = "", .expected_len = 0, .expected_mblen = 0
560
+ };
561
+ run_sanitize_test(&t);
562
+ }
563
+
564
+ // dst_size = 2 (one char + null)
565
+ {
566
+ sanitize_test_t t = {
567
+ .name = "buffer_size_2",
568
+ .input = (unsigned char *)"hello",
569
+ .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
570
+ .expected_output = "h", .expected_len = 1, .expected_mblen = 1
571
+ };
572
+ run_sanitize_test(&t);
573
+ }
574
+
575
+ // Exact fit for ASCII
576
+ {
577
+ sanitize_test_t t = {
578
+ .name = "buffer_exact_ascii",
579
+ .input = (unsigned char *)"abc",
580
+ .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
581
+ .expected_output = "abc", .expected_len = 3, .expected_mblen = 3
582
+ };
583
+ run_sanitize_test(&t);
584
+ }
585
+
586
+ // Off-by-one for ASCII (truncation)
587
+ {
588
+ sanitize_test_t t = {
589
+ .name = "buffer_truncate_ascii",
590
+ .input = (unsigned char *)"abcd",
591
+ .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
592
+ .expected_output = "abc", .expected_len = 3, .expected_mblen = 3
593
+ };
594
+ run_sanitize_test(&t);
595
+ }
596
+
597
+ // Exact fit for 2-byte UTF-8
598
+ {
599
+ sanitize_test_t t = {
600
+ .name = "buffer_exact_2byte",
601
+ .input = (unsigned char *)"\xC2\xB0", // ° (2 bytes)
602
+ .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
603
+ .expected_output = "\xC2\xB0", .expected_len = 2, .expected_mblen = 1
604
+ };
605
+ run_sanitize_test(&t);
606
+ }
607
+
608
+ // Off-by-one for 2-byte UTF-8 (can't fit, hex encode)
609
+ {
610
+ sanitize_test_t t = {
611
+ .name = "buffer_truncate_2byte",
612
+ .input = (unsigned char *)"\xC2\xB0",
613
+ .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
614
+ .expected_output = "", .expected_len = 0, .expected_mblen = 0 // Can't fit hex either
615
+ };
616
+ run_sanitize_test(&t);
617
+ }
618
+
619
+ // Overlong sequence (structurally valid) with exact fit
620
+ {
621
+ sanitize_test_t t = {
622
+ .name = "buffer_overlong_exact_fit",
623
+ .input = (unsigned char *)"\xC0\x80", // Structurally valid 2-byte
624
+ .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
625
+ .expected_output = "\xC0\x80", .expected_len = 2, .expected_mblen = 1
626
+ };
627
+ run_sanitize_test(&t);
628
+ }
629
+
630
+ // ASCII + UTF-8 boundary
631
+ {
632
+ sanitize_test_t t = {
633
+ .name = "buffer_ascii_utf8_boundary",
634
+ .input = (unsigned char *)"X\xC2\xB0", // X° (3 bytes)
635
+ .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
636
+ .expected_output = "X\xC2\xB0", .expected_len = 3, .expected_mblen = 2
637
+ };
638
+ run_sanitize_test(&t);
639
+ }
640
+
641
+ // UTF-8 doesn't fit at buffer end - nothing written for UTF-8 but mblen still counts
642
+ {
643
+ sanitize_test_t t = {
644
+ .name = "buffer_utf8_no_fit",
645
+ .input = (unsigned char *)"XY\xC2\xB0", // XY° (4 bytes, but UTF-8 needs 2)
646
+ .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
647
+ // UTF-8 can't fit, hex can't fit either, nothing written for °
648
+ // But mblen still increments (counts processed, not written)
649
+ .expected_output = "XY", .expected_len = 2, .expected_mblen = 3
650
+ };
651
+ run_sanitize_test(&t);
652
+ }
653
+
654
+ // Overlong UTF-8 near buffer end (structurally valid, can't fit)
655
+ {
656
+ sanitize_test_t t = {
657
+ .name = "buffer_overlong_no_fit",
658
+ .input = (unsigned char *)"A\xC0\x80", // A + overlong (structurally valid)
659
+ .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
660
+ // Only A fits (1 byte), overlong needs 2 bytes but only 1 left
661
+ // mblen counts 2 (A + attempted UTF-8)
662
+ .expected_output = "A", .expected_len = 1, .expected_mblen = 2
663
+ };
664
+ run_sanitize_test(&t);
665
+ }
666
+
667
+ // Overlong 2-byte fits exactly, orphan continuation bytes follow
668
+ {
669
+ sanitize_test_t t = {
670
+ .name = "buffer_overlong_with_orphans",
671
+ .input = (unsigned char *)"X\xC0\x80\x80\x80", // X + overlong + orphan continuations
672
+ .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
673
+ // X (1) + overlong \xC0\x80 (2) = 3 bytes, fits in dst_size=4
674
+ // Orphan bytes don't fit
675
+ .expected_output = "X\xC0\x80", .expected_len = 3, .expected_mblen = 2
676
+ };
677
+ run_sanitize_test(&t);
678
+ }
679
+
680
+ // Very long input (256 bytes)
681
+ {
682
+ unsigned char long_input[257];
683
+ memset(long_input, 'A', 256);
684
+ long_input[256] = '\0';
685
+
686
+ unsigned char expected[101];
687
+ memset(expected, 'A', 100);
688
+ expected[100] = '\0';
689
+
690
+ sanitize_test_t t = {
691
+ .name = "buffer_long_input",
692
+ .input = long_input,
693
+ .dst_size = 101, .char_map = identity_char_map, .utf = true, .empty = "",
694
+ .expected_output = (char *)expected, .expected_len = 100, .expected_mblen = 100
695
+ };
696
+ run_sanitize_test(&t);
697
+ }
698
+}
699
+
700
+// ============================================================================
701
+// TEST: CHARACTER MAP TRANSFORMATIONS
702
+// ============================================================================
703
+
704
+static void test_char_map_transformations(void) {
705
+ fprintf(stderr, "\n=== Character Map Transformations ===\n");
706
+
707
+ // Double quote → single quote
708
+ {
709
+ sanitize_test_t t = {
710
+ .name = "charmap_quote",
711
+ .input = (unsigned char *)"say \"hello\"",
712
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
713
+ .expected_output = "say 'hello'", .expected_len = 11, .expected_mblen = 11
714
+ };
715
+ run_sanitize_test(&t);
716
+ }
717
+
718
+ // Backslash → forward slash
719
+ {
720
+ sanitize_test_t t = {
721
+ .name = "charmap_backslash",
722
+ .input = (unsigned char *)"C:\\path\\to\\file",
723
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
724
+ .expected_output = "C:/path/to/file", .expected_len = 15, .expected_mblen = 15
725
+ };
726
+ run_sanitize_test(&t);
727
+ }
728
+
729
+ // Tab → space
730
+ {
731
+ sanitize_test_t t = {
732
+ .name = "charmap_tab",
733
+ .input = (unsigned char *)"col1\tcol2",
734
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
735
+ .expected_output = "col1 col2", .expected_len = 9, .expected_mblen = 9
736
+ };
737
+ run_sanitize_test(&t);
738
+ }
739
+
740
+ // Newline → space
741
+ {
742
+ sanitize_test_t t = {
743
+ .name = "charmap_newline",
744
+ .input = (unsigned char *)"line1\nline2",
745
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
746
+ .expected_output = "line1 line2", .expected_len = 11, .expected_mblen = 11
747
+ };
748
+ run_sanitize_test(&t);
749
+ }
750
+
751
+ // Carriage return → space
752
+ {
753
+ sanitize_test_t t = {
754
+ .name = "charmap_cr",
755
+ .input = (unsigned char *)"line1\rline2",
756
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
757
+ .expected_output = "line1 line2", .expected_len = 11, .expected_mblen = 11
758
+ };
759
+ run_sanitize_test(&t);
760
+ }
761
+
762
+ // CRLF → space (deduplicated)
763
+ {
764
+ sanitize_test_t t = {
765
+ .name = "charmap_crlf",
766
+ .input = (unsigned char *)"line1\r\nline2",
767
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
768
+ .expected_output = "line1 line2", .expected_len = 11, .expected_mblen = 11
769
+ };
770
+ run_sanitize_test(&t);
771
+ }
772
+
773
+ // Multiple control characters → single space
774
+ {
775
+ sanitize_test_t t = {
776
+ .name = "charmap_multi_control",
777
+ .input = (unsigned char *)"a\t\n\r\vb",
778
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
779
+ .expected_output = "a b", .expected_len = 3, .expected_mblen = 3
780
+ };
781
+ run_sanitize_test(&t);
782
+ }
783
+
784
+ // NUL character (should terminate)
785
+ {
786
+ unsigned char input[] = {'a', 'b', '\0', 'c', 'd', '\0'};
787
+ sanitize_test_t t = {
788
+ .name = "charmap_nul",
789
+ .input = input,
790
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
791
+ .expected_output = "ab", .expected_len = 2, .expected_mblen = 2
792
+ };
793
+ run_sanitize_test(&t);
794
+ }
795
+
796
+ // DEL character (0x7F) → space
797
+ {
798
+ sanitize_test_t t = {
799
+ .name = "charmap_del",
800
+ .input = (unsigned char *)"a\x7F""b",
801
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
802
+ .expected_output = "a b", .expected_len = 3, .expected_mblen = 3
803
+ };
804
+ run_sanitize_test(&t);
805
+ }
806
+
807
+ // All printable ASCII preserved (30 characters)
808
+ {
809
+ sanitize_test_t t = {
810
+ .name = "charmap_printable_ascii",
811
+ .input = (unsigned char *)"!#$%&'()*+,-./:;<=>?@[]^_`{|}~",
812
+ .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
813
+ .expected_output = "!#$%&'()*+,-./:;<=>?@[]^_`{|}~", .expected_len = 30, .expected_mblen = 30
814
+ };
815
+ run_sanitize_test(&t);
816
+ }
817
+
818
+ // High bytes (0x80-0xBF) are continuation bytes → char_map (space)
819
+ // 0xFF is a start byte but invalid pattern → hex encoded
820
+ {
821
+ sanitize_test_t t = {
822
+ .name = "charmap_high_byte_mixed",
823
+ .input = (unsigned char *)"a\x80\x90\xA0\xB0\xFF""b",
824
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
825
+ // \x80-\xB0 are continuation bytes (10xxxxxx) → go through char_map → space
826
+ // \xFF is start byte but invalid pattern → hex encoded as "ff"
827
+ .expected_output = "a ffb", .expected_len = 5, .expected_mblen = 4
828
+ };
829
+ run_sanitize_test(&t);
830
+ }
831
+
832
+ // Combined transformations
833
+ {
834
+ sanitize_test_t t = {
835
+ .name = "charmap_combined",
836
+ .input = (unsigned char *)"\"path\\to\\file\"\t(100%)",
837
+ .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
838
+ .expected_output = "'path/to/file' (100%)", .expected_len = 21, .expected_mblen = 21
839
+ };
840
+ run_sanitize_test(&t);
841
+ }
842
+}
843
+
844
+// ============================================================================
845
+// TEST: SPACE HANDLING
846
+// ============================================================================
847
+
848
+static void test_space_handling(void) {
849
+ fprintf(stderr, "\n=== Space Handling ===\n");
850
+
851
+ // Leading spaces removed
852
+ {
853
+ sanitize_test_t t = {
854
+ .name = "space_leading",
855
+ .input = (unsigned char *)" hello",
856
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
857
+ .expected_output = "hello", .expected_len = 5, .expected_mblen = 5
858
+ };
859
+ run_sanitize_test(&t);
860
+ }
861
+
862
+ // Trailing spaces removed
863
+ {
864
+ sanitize_test_t t = {
865
+ .name = "space_trailing",
866
+ .input = (unsigned char *)"hello ",
867
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
868
+ .expected_output = "hello", .expected_len = 5, .expected_mblen = 5
869
+ };
870
+ run_sanitize_test(&t);
871
+ }
872
+
873
+ // Both leading and trailing
874
+ {
875
+ sanitize_test_t t = {
876
+ .name = "space_both_ends",
877
+ .input = (unsigned char *)" hello ",
878
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
879
+ .expected_output = "hello", .expected_len = 5, .expected_mblen = 5
880
+ };
881
+ run_sanitize_test(&t);
882
+ }
883
+
884
+ // Multiple consecutive spaces → single space
885
+ {
886
+ sanitize_test_t t = {
887
+ .name = "space_consecutive",
888
+ .input = (unsigned char *)"hello world",
889
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
890
+ .expected_output = "hello world", .expected_len = 11, .expected_mblen = 11
891
+ };
892
+ run_sanitize_test(&t);
893
+ }
894
+
895
+ // Only spaces → empty
896
+ {
897
+ sanitize_test_t t = {
898
+ .name = "space_only",
899
+ .input = (unsigned char *)" ",
900
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "default",
901
+ .expected_output = "default", .expected_len = 7, .expected_mblen = 7
902
+ };
903
+ run_sanitize_test(&t);
904
+ }
905
+
906
+ // Control chars becoming spaces and deduplicating
907
+ {
908
+ sanitize_test_t t = {
909
+ .name = "space_from_control",
910
+ .input = (unsigned char *)"a\t\t\t\nb",
911
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
912
+ .expected_output = "a b", .expected_len = 3, .expected_mblen = 3
913
+ };
914
+ run_sanitize_test(&t);
915
+ }
916
+
917
+ // Space before UTF-8
918
+ {
919
+ sanitize_test_t t = {
920
+ .name = "space_before_utf8",
921
+ .input = (unsigned char *)"temp \xC2\xB0""C",
922
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
923
+ .expected_output = "temp \xC2\xB0""C", .expected_len = 8, .expected_mblen = 7
924
+ };
925
+ run_sanitize_test(&t);
926
+ }
927
+
928
+ // Space after UTF-8
929
+ {
930
+ sanitize_test_t t = {
931
+ .name = "space_after_utf8",
932
+ .input = (unsigned char *)"\xC2\xB0 Celsius",
933
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
934
+ .expected_output = "\xC2\xB0 Celsius", .expected_len = 10, .expected_mblen = 9
935
+ };
936
+ run_sanitize_test(&t);
937
+ }
938
+
939
+ // Tab-separated values
940
+ {
941
+ sanitize_test_t t = {
942
+ .name = "space_tsv",
943
+ .input = (unsigned char *)"col1\tcol2\tcol3",
944
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
945
+ .expected_output = "col1 col2 col3", .expected_len = 14, .expected_mblen = 14
946
+ };
947
+ run_sanitize_test(&t);
948
+ }
949
+}
950
+
951
+// ============================================================================
952
+// TEST: EMPTY AND SPECIAL CASES
953
+// ============================================================================
954
+
955
+static void test_empty_and_special(void) {
956
+ fprintf(stderr, "\n=== Empty and Special Cases ===\n");
957
+
958
+ // NULL input
959
+ {
960
+ unsigned char dst[32];
961
+ size_t len = text_sanitize(dst, NULL, sizeof(dst), identity_char_map, true, "null_val", NULL);
962
+ TEST_ASSERT("null_input", strcmp((char *)dst, "null_val") == 0,
963
+ "Expected 'null_val', got '%s'", dst);
964
+ TEST_ASSERT("null_input_len", len == 8, "Expected 8, got %zu", len);
965
+ }
966
+
967
+ // NULL dst
968
+ {
969
+ size_t len = text_sanitize(NULL, (unsigned char *)"hello", 32, identity_char_map, true, "", NULL);
970
+ TEST_ASSERT("null_dst", len == 0, "Expected 0, got %zu", len);
971
+ }
972
+
973
+ // Empty string input
974
+ {
975
+ sanitize_test_t t = {
976
+ .name = "empty_input",
977
+ .input = (unsigned char *)"",
978
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "empty_val",
979
+ .expected_output = "empty_val", .expected_len = 9, .expected_mblen = 9
980
+ };
981
+ run_sanitize_test(&t);
982
+ }
983
+
984
+ // All underscores → empty (special rule)
985
+ {
986
+ sanitize_test_t t = {
987
+ .name = "all_underscores",
988
+ .input = (unsigned char *)"___",
989
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "default",
990
+ .expected_output = "default", .expected_len = 7, .expected_mblen = 7
991
+ };
992
+ run_sanitize_test(&t);
993
+ }
994
+
995
+ // Underscore followed by text (not empty)
996
+ {
997
+ sanitize_test_t t = {
998
+ .name = "underscore_prefix",
999
+ .input = (unsigned char *)"___abc",
1000
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1001
+ .expected_output = "___abc", .expected_len = 6, .expected_mblen = 6
1002
+ };
1003
+ run_sanitize_test(&t);
1004
+ }
1005
+
1006
+ // Only control characters → empty
1007
+ {
1008
+ sanitize_test_t t = {
1009
+ .name = "only_control_chars",
1010
+ .input = (unsigned char *)"\t\n\r\v",
1011
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "ctrl_empty",
1012
+ .expected_output = "ctrl_empty", .expected_len = 10, .expected_mblen = 10
1013
+ };
1014
+ run_sanitize_test(&t);
1015
+ }
1016
+
1017
+ // Invalid UTF-8 that becomes all underscores with utf=false
1018
+ {
1019
+ sanitize_test_t t = {
1020
+ .name = "utf8_to_underscores",
1021
+ .input = (unsigned char *)"\xC2\x80\xC2\x80",
1022
+ .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "utf_empty",
1023
+ .expected_output = "utf_empty", .expected_len = 9, .expected_mblen = 9
1024
+ };
1025
+ run_sanitize_test(&t);
1026
+ }
1027
+
1028
+ // Empty string with empty default
1029
+ {
1030
+ sanitize_test_t t = {
1031
+ .name = "empty_with_empty_default",
1032
+ .input = (unsigned char *)"",
1033
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1034
+ .expected_output = "", .expected_len = 0, .expected_mblen = 0
1035
+ };
1036
+ run_sanitize_test(&t);
1037
+ }
1038
+
1039
+ // Single character
1040
+ {
1041
+ sanitize_test_t t = {
1042
+ .name = "single_char",
1043
+ .input = (unsigned char *)"X",
1044
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1045
+ .expected_output = "X", .expected_len = 1, .expected_mblen = 1
1046
+ };
1047
+ run_sanitize_test(&t);
1048
+ }
1049
+
1050
+ // Single UTF-8 character
1051
+ {
1052
+ sanitize_test_t t = {
1053
+ .name = "single_utf8_char",
1054
+ .input = (unsigned char *)"\xC2\xB0",
1055
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1056
+ .expected_output = "\xC2\xB0", .expected_len = 2, .expected_mblen = 1
1057
+ };
1058
+ run_sanitize_test(&t);
1059
+ }
1060
+}
1061
+
1062
+// ============================================================================
1063
+// TEST: UTF PARAMETER (true vs false)
1064
+// ============================================================================
1065
+
1066
+static void test_utf_parameter(void) {
1067
+ fprintf(stderr, "\n=== UTF Parameter (true vs false) ===\n");
1068
+
1069
+ // utf=true: valid UTF-8 preserved
1070
+ {
1071
+ sanitize_test_t t = {
1072
+ .name = "utf_true_valid",
1073
+ .input = (unsigned char *)"test\xC2\xB0""C",
1074
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1075
+ .expected_output = "test\xC2\xB0""C", .expected_len = 7, .expected_mblen = 6
1076
+ };
1077
+ run_sanitize_test(&t);
1078
+ }
1079
+
1080
+ // utf=false: valid UTF-8 → underscore
1081
+ {
1082
+ sanitize_test_t t = {
1083
+ .name = "utf_false_valid",
1084
+ .input = (unsigned char *)"test\xC2\xB0""C",
1085
+ .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1086
+ .expected_output = "test_C", .expected_len = 6, .expected_mblen = 6
1087
+ };
1088
+ run_sanitize_test(&t);
1089
+ }
1090
+
1091
+ // utf=true: overlong (structurally valid) passes through
1092
+ {
1093
+ sanitize_test_t t = {
1094
+ .name = "utf_true_overlong",
1095
+ .input = (unsigned char *)"test\xC0\x80""X",
1096
+ .dst_size = 32, .char_map = identity_char_map, .utf = true, .empty = "",
1097
+ // \xC0\x80 is structurally valid (2-byte pattern), passes through
1098
+ .expected_output = "test\xC0\x80X", .expected_len = 7, .expected_mblen = 6
1099
+ };
1100
+ run_sanitize_test(&t);
1101
+ }
1102
+
1103
+ // utf=false: invalid UTF-8 → underscore
1104
+ {
1105
+ sanitize_test_t t = {
1106
+ .name = "utf_false_invalid",
1107
+ .input = (unsigned char *)"test\xC0\x80""X",
1108
+ .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1109
+ .expected_output = "test_X", .expected_len = 6, .expected_mblen = 6
1110
+ };
1111
+ run_sanitize_test(&t);
1112
+ }
1113
+
1114
+ // utf=false: 3-byte UTF-8 → single underscore
1115
+ {
1116
+ sanitize_test_t t = {
1117
+ .name = "utf_false_3byte",
1118
+ .input = (unsigned char *)"price\xE2\x82\xAC""100", // price€100
1119
+ .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1120
+ .expected_output = "price_100", .expected_len = 9, .expected_mblen = 9
1121
+ };
1122
+ run_sanitize_test(&t);
1123
+ }
1124
+
1125
+ // utf=false: 4-byte UTF-8 → single underscore
1126
+ {
1127
+ sanitize_test_t t = {
1128
+ .name = "utf_false_4byte",
1129
+ .input = (unsigned char *)"hi\xF0\x9F\x98\x80""!", // hi😀!
1130
+ .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "",
1131
+ .expected_output = "hi_!", .expected_len = 4, .expected_mblen = 4
1132
+ };
1133
+ run_sanitize_test(&t);
1134
+ }
1135
+
1136
+ // utf=false: multiple UTF-8 → multiple underscores (but collapse doesn't happen)
1137
+ {
1138
+ sanitize_test_t t = {
1139
+ .name = "utf_false_multiple",
1140
+ .input = (unsigned char *)"\xC2\xB0\xC2\xB5", // °µ
1141
+ .dst_size = 32, .char_map = identity_char_map, .utf = false, .empty = "x",
1142
+ .expected_output = "x", .expected_len = 1, .expected_mblen = 1
1143
+ // Two underscores collapse to empty due to all-underscore rule
1144
+ };
1145
+ run_sanitize_test(&t);
1146
+ }
1147
+}
1148
+
1149
+// ============================================================================
1150
+// TEST: MULTIBYTE LENGTH OUTPUT
1151
+// ============================================================================
1152
+
1153
+static void test_multibyte_length(void) {
1154
+ fprintf(stderr, "\n=== Multibyte Length Output ===\n");
1155
+
1156
+ // ASCII only: byte length == char count
1157
+ {
1158
+ unsigned char dst[32];
1159
+ size_t mblen = 0;
1160
+ size_t len = text_sanitize(dst, (unsigned char *)"hello", sizeof(dst),
1161
+ identity_char_map, true, "", &mblen);
1162
+ TEST_ASSERT("mblen_ascii", len == 5 && mblen == 5,
1163
+ "len=%zu mblen=%zu, expected both 5", len, mblen);
1164
+ }
1165
+
1166
+ // Single 2-byte UTF-8: byte length > char count
1167
+ {
1168
+ unsigned char dst[32];
1169
+ size_t mblen = 0;
1170
+ size_t len = text_sanitize(dst, (unsigned char *)"\xC2\xB0", sizeof(dst),
1171
+ identity_char_map, true, "", &mblen);
1172
+ TEST_ASSERT("mblen_2byte", len == 2 && mblen == 1,
1173
+ "len=%zu mblen=%zu, expected len=2 mblen=1", len, mblen);
1174
+ }
1175
+
1176
+ // Single 3-byte UTF-8
1177
+ {
1178
+ unsigned char dst[32];
1179
+ size_t mblen = 0;
1180
+ size_t len = text_sanitize(dst, (unsigned char *)"\xE2\x82\xAC", sizeof(dst),
1181
+ identity_char_map, true, "", &mblen);
1182
+ TEST_ASSERT("mblen_3byte", len == 3 && mblen == 1,
1183
+ "len=%zu mblen=%zu, expected len=3 mblen=1", len, mblen);
1184
+ }
1185
+
1186
+ // Single 4-byte UTF-8
1187
+ {
1188
+ unsigned char dst[32];
1189
+ size_t mblen = 0;
1190
+ size_t len = text_sanitize(dst, (unsigned char *)"\xF0\x9F\x98\x80", sizeof(dst),
1191
+ identity_char_map, true, "", &mblen);
1192
+ TEST_ASSERT("mblen_4byte", len == 4 && mblen == 1,
1193
+ "len=%zu mblen=%zu, expected len=4 mblen=1", len, mblen);
1194
+ }
1195
+
1196
+ // Mixed: ASCII + UTF-8
1197
+ {
1198
+ unsigned char dst[32];
1199
+ size_t mblen = 0;
1200
+ // "A°€😀" = 1 + 2 + 3 + 4 = 10 bytes, 4 chars
1201
+ size_t len = text_sanitize(dst, (unsigned char *)"A\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80",
1202
+ sizeof(dst), identity_char_map, true, "", &mblen);
1203
+ TEST_ASSERT("mblen_mixed", len == 10 && mblen == 4,
1204
+ "len=%zu mblen=%zu, expected len=10 mblen=4", len, mblen);
1205
+ }
1206
+
1207
+ // NULL mblen pointer (shouldn't crash)
1208
+ {
1209
+ unsigned char dst[32];
1210
+ size_t len = text_sanitize(dst, (unsigned char *)"test", sizeof(dst),
1211
+ identity_char_map, true, "", NULL);
1212
+ TEST_ASSERT("mblen_null_ptr", len == 4, "len=%zu, expected 4", len);
1213
+ }
1214
+}
1215
+
1216
+// ============================================================================
1217
+// TEST: RRD STRING ALLOWED CHARS SPECIFIC
1218
+// ============================================================================
1219
+
1220
+static void test_rrd_string_allowed_chars(void) {
1221
+ fprintf(stderr, "\n=== RRD String Allowed Chars ===\n");
1222
+
1223
+ // Use actual rrd_string_allowed_chars from the codebase
1224
+ extern unsigned char rrd_string_allowed_chars[256];
1225
+
1226
+ // Basic ASCII passes through
1227
+ {
1228
+ sanitize_test_t t = {
1229
+ .name = "rrd_ascii",
1230
+ .input = (unsigned char *)"cpu.user",
1231
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1232
+ .expected_output = "cpu.user", .expected_len = 8, .expected_mblen = 8
1233
+ };
1234
+ run_sanitize_test(&t);
1235
+ }
1236
+
1237
+ // Double quote transformed
1238
+ {
1239
+ sanitize_test_t t = {
1240
+ .name = "rrd_double_quote",
1241
+ .input = (unsigned char *)"\"value\"",
1242
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1243
+ .expected_output = "'value'", .expected_len = 7, .expected_mblen = 7
1244
+ };
1245
+ run_sanitize_test(&t);
1246
+ }
1247
+
1248
+ // Backslash transformed
1249
+ {
1250
+ sanitize_test_t t = {
1251
+ .name = "rrd_backslash",
1252
+ .input = (unsigned char *)"path\\file",
1253
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1254
+ .expected_output = "path/file", .expected_len = 9, .expected_mblen = 9
1255
+ };
1256
+ run_sanitize_test(&t);
1257
+ }
1258
+
1259
+ // UTF-8 units preserved
1260
+ {
1261
+ sanitize_test_t t = {
1262
+ .name = "rrd_utf8_units",
1263
+ .input = (unsigned char *)"requests/s\xC2\xB2", // requests/s²
1264
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1265
+ .expected_output = "requests/s\xC2\xB2", .expected_len = 12, .expected_mblen = 11
1266
+ };
1267
+ run_sanitize_test(&t);
1268
+ }
1269
+
1270
+ // Temperature with degree symbol
1271
+ {
1272
+ sanitize_test_t t = {
1273
+ .name = "rrd_temperature",
1274
+ .input = (unsigned char *)"Temperature (\xC2\xB0""C)",
1275
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1276
+ .expected_output = "Temperature (\xC2\xB0""C)", .expected_len = 17, .expected_mblen = 16
1277
+ };
1278
+ run_sanitize_test(&t);
1279
+ }
1280
+
1281
+ // Microseconds
1282
+ {
1283
+ sanitize_test_t t = {
1284
+ .name = "rrd_microseconds",
1285
+ .input = (unsigned char *)"\xC2\xB5s", // µs
1286
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1287
+ .expected_output = "\xC2\xB5s", .expected_len = 3, .expected_mblen = 2
1288
+ };
1289
+ run_sanitize_test(&t);
1290
+ }
1291
+
1292
+ // Complex metric title
1293
+ {
1294
+ sanitize_test_t t = {
1295
+ .name = "rrd_complex_title",
1296
+ .input = (unsigned char *)"CPU \"usage\" on C:\\Windows (100%)",
1297
+ .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1298
+ .expected_output = "CPU 'usage' on C:/Windows (100%)", .expected_len = 32, .expected_mblen = 32
1299
+ };
1300
+ run_sanitize_test(&t);
1301
+ }
1302
+
1303
+ // Prometheus-style metric
1304
+ {
1305
+ sanitize_test_t t = {
1306
+ .name = "rrd_prometheus_style",
1307
+ .input = (unsigned char *)"http_requests_total{method=\"GET\"}",
1308
+ .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1309
+ .expected_output = "http_requests_total{method='GET'}", .expected_len = 33, .expected_mblen = 33
1310
+ };
1311
+ run_sanitize_test(&t);
1312
+ }
1313
+}
1314
+
1315
+// ============================================================================
1316
+// TEST: SECURITY-FOCUSED CASES
1317
+// ============================================================================
1318
+
1319
+static void test_security_cases(void) {
1320
+ fprintf(stderr, "\n=== Security-Focused Cases ===\n");
1321
+
1322
+ // Path traversal attempt (should be handled safely)
1323
+ {
1324
+ sanitize_test_t t = {
1325
+ .name = "security_path_traversal",
1326
+ .input = (unsigned char *)"../../../etc/passwd",
1327
+ .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1328
+ .expected_output = "../../../etc/passwd", .expected_len = 19, .expected_mblen = 19
1329
+ };
1330
+ run_sanitize_test(&t);
1331
+ }
1332
+
1333
+ // Path traversal with backslash (Windows style, converted to /)
1334
+ {
1335
+ sanitize_test_t t = {
1336
+ .name = "security_path_traversal_win",
1337
+ .input = (unsigned char *)"..\\..\\..\\etc\\passwd",
1338
+ .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1339
+ .expected_output = "../../../etc/passwd", .expected_len = 19, .expected_mblen = 19
1340
+ };
1341
+ run_sanitize_test(&t);
1342
+ }
1343
+
1344
+ // Overlong NUL - structurally valid, passes through
1345
+ // NOTE: This is a security concern in some systems but this function
1346
+ // only does structural validation for sanitization purposes
1347
+ {
1348
+ sanitize_test_t t = {
1349
+ .name = "security_overlong_nul_passthrough",
1350
+ .input = (unsigned char *)"test\xC0\x80test", // Overlong NUL
1351
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1352
+ .expected_output = "test\xC0\x80test", .expected_len = 10, .expected_mblen = 9
1353
+ };
1354
+ run_sanitize_test(&t);
1355
+ }
1356
+
1357
+ // Overlong slash - structurally valid, passes through
1358
+ {
1359
+ sanitize_test_t t = {
1360
+ .name = "security_overlong_slash_passthrough",
1361
+ .input = (unsigned char *)"\xC0\xAF", // Overlong /
1362
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1363
+ .expected_output = "\xC0\xAF", .expected_len = 2, .expected_mblen = 1
1364
+ };
1365
+ run_sanitize_test(&t);
1366
+ }
1367
+
1368
+ // Overlong A (3 bytes) - structurally valid, passes through
1369
+ {
1370
+ sanitize_test_t t = {
1371
+ .name = "security_overlong_A_passthrough",
1372
+ .input = (unsigned char *)"\xE0\x81\x81", // Overlong A
1373
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1374
+ .expected_output = "\xE0\x81\x81", .expected_len = 3, .expected_mblen = 1
1375
+ };
1376
+ run_sanitize_test(&t);
1377
+ }
1378
+
1379
+ // XSS attempt with angle brackets
1380
+ {
1381
+ sanitize_test_t t = {
1382
+ .name = "security_xss_tags",
1383
+ .input = (unsigned char *)"<script>alert(1)</script>",
1384
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1385
+ .expected_output = "<script>alert(1)</script>", .expected_len = 25, .expected_mblen = 25
1386
+ };
1387
+ run_sanitize_test(&t);
1388
+ }
1389
+
1390
+ // SQL injection attempt (quotes transformed)
1391
+ {
1392
+ sanitize_test_t t = {
1393
+ .name = "security_sql_injection",
1394
+ .input = (unsigned char *)"test' OR '1'='1",
1395
+ .dst_size = 64, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1396
+ .expected_output = "test' OR '1'='1", .expected_len = 15, .expected_mblen = 15
1397
+ };
1398
+ run_sanitize_test(&t);
1399
+ }
1400
+
1401
+ // Null byte injection (string terminates at NUL)
1402
+ {
1403
+ unsigned char input[] = {'t', 'e', 's', 't', '\0', 'e', 'v', 'i', 'l', '\0'};
1404
+ sanitize_test_t t = {
1405
+ .name = "security_null_byte",
1406
+ .input = input,
1407
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1408
+ .expected_output = "test", .expected_len = 4, .expected_mblen = 4
1409
+ };
1410
+ run_sanitize_test(&t);
1411
+ }
1412
+
1413
+ // BOM (Byte Order Mark) at start - should be preserved as valid UTF-8
1414
+ {
1415
+ sanitize_test_t t = {
1416
+ .name = "security_bom",
1417
+ .input = (unsigned char *)"\xEF\xBB\xBFtext", // UTF-8 BOM + text
1418
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1419
+ .expected_output = "\xEF\xBB\xBFtext", .expected_len = 7, .expected_mblen = 5
1420
+ };
1421
+ run_sanitize_test(&t);
1422
+ }
1423
+
1424
+ // UTF-7 encoding attempt (should just pass through as ASCII)
1425
+ {
1426
+ sanitize_test_t t = {
1427
+ .name = "security_utf7",
1428
+ .input = (unsigned char *)"+ADw-script+AD4-",
1429
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1430
+ .expected_output = "+ADw-script+AD4-", .expected_len = 16, .expected_mblen = 16
1431
+ };
1432
+ run_sanitize_test(&t);
1433
+ }
1434
+
1435
+ // Private Use Area character (valid UTF-8, possibly suspicious)
1436
+ {
1437
+ sanitize_test_t t = {
1438
+ .name = "security_private_use",
1439
+ .input = (unsigned char *)"\xEE\x80\x80", // U+E000 (Private Use)
1440
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1441
+ .expected_output = "\xEE\x80\x80", .expected_len = 3, .expected_mblen = 1
1442
+ };
1443
+ run_sanitize_test(&t);
1444
+ }
1445
+}
1446
+
1447
+// ============================================================================
1448
+// TEST: REGRESSION TESTS FOR FIXED BUGS
1449
+// ============================================================================
1450
+
1451
+static void test_regression_fixed_bugs(void) {
1452
+ fprintf(stderr, "\n=== Regression Tests for Fixed Bugs ===\n");
1453
+
1454
+ // REGRESSION: The original buffer overflow bug was in hex encoding path.
1455
+ // Test with TRULY invalid UTF-8 (truncated sequence) that triggers hex encoding
1456
+ {
1457
+ sanitize_test_t t = {
1458
+ .name = "regression_hex_buffer_overflow",
1459
+ .input = (unsigned char *)"\xC2", // Truncated 2-byte sequence
1460
+ .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
1461
+ // Hex needs 2 chars ("c2") + NUL = 3, exact fit
1462
+ .expected_output = "c2", .expected_len = 2, .expected_mblen = 1
1463
+ };
1464
+ run_sanitize_test(&t);
1465
+ }
1466
+
1467
+ // Test truncated sequence that would overflow if not properly bounded
1468
+ {
1469
+ sanitize_test_t t = {
1470
+ .name = "regression_hex_no_overflow",
1471
+ .input = (unsigned char *)"\xC2", // Truncated
1472
+ .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
1473
+ // Hex needs 2 chars but only 1 space (plus NUL) - nothing written
1474
+ // Note: mblen is 0 when loop doesn't process due to buffer constraints
1475
+ .expected_output = "", .expected_len = 0, .expected_mblen = 0
1476
+ };
1477
+ run_sanitize_test(&t);
1478
+ }
1479
+
1480
+ // Overlong \xC0\x80 is structurally VALID - test it passes through
1481
+ {
1482
+ sanitize_test_t t = {
1483
+ .name = "regression_overlong_passthrough",
1484
+ .input = (unsigned char *)"X\xC0\x80Y",
1485
+ .dst_size = 16, .char_map = identity_char_map, .utf = true, .empty = "",
1486
+ .expected_output = "X\xC0\x80Y", .expected_len = 4, .expected_mblen = 3
1487
+ };
1488
+ run_sanitize_test(&t);
1489
+ }
1490
+
1491
+ // REGRESSION: Memory read OOB (Issue: Loop didn't check for NUL before accessing src[i])
1492
+ {
1493
+ sanitize_test_t t = {
1494
+ .name = "regression_memory_oob_truncated",
1495
+ .input = (unsigned char *)"test\xE2\x82", // Truncated 3-byte sequence
1496
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1497
+ .expected_output = "teste282", .expected_len = 8, .expected_mblen = 5
1498
+ };
1499
+ run_sanitize_test(&t);
1500
+ }
1501
+
1502
+ // REGRESSION: Memory read OOB with 4-byte truncated at various points
1503
+ {
1504
+ sanitize_test_t t1 = {
1505
+ .name = "regression_oob_4byte_1cont",
1506
+ .input = (unsigned char *)"\xF0\x9F", // Only 2 of 4 bytes
1507
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1508
+ .expected_output = "f09f", .expected_len = 4, .expected_mblen = 1
1509
+ };
1510
+ run_sanitize_test(&t1);
1511
+
1512
+ sanitize_test_t t2 = {
1513
+ .name = "regression_oob_4byte_2cont",
1514
+ .input = (unsigned char *)"\xF0\x9F\x98", // Only 3 of 4 bytes
1515
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1516
+ .expected_output = "f09f98", .expected_len = 6, .expected_mblen = 1
1517
+ };
1518
+ run_sanitize_test(&t2);
1519
+ }
1520
+
1521
+ // Edge case: \xF5 is treated as 4-byte start, but alone it's truncated → hex
1522
+ // NOTE: \xF5 without continuation bytes triggers hex encoding
1523
+ {
1524
+ sanitize_test_t t = {
1525
+ .name = "regression_edge_F5_truncated",
1526
+ .input = (unsigned char *)"X\xF5", // X + truncated F5 (no continuation)
1527
+ .dst_size = 5, .char_map = identity_char_map, .utf = true, .empty = "",
1528
+ // X (1) + "f5" (2) + NUL = 4, fits in 5
1529
+ .expected_output = "Xf5", .expected_len = 3, .expected_mblen = 2
1530
+ };
1531
+ run_sanitize_test(&t);
1532
+ }
1533
+
1534
+ // Edge case: Exactly 2 spaces for hex (dst_size=4, one char used)
1535
+ {
1536
+ sanitize_test_t t = {
1537
+ .name = "regression_edge_exact_hex_fit",
1538
+ .input = (unsigned char *)"A\xC0", // A + truncated (missing continuation)
1539
+ .dst_size = 4, .char_map = identity_char_map, .utf = true, .empty = "",
1540
+ // A (1) + "c0" (2) + NUL = 4 (exact fit)
1541
+ .expected_output = "Ac0", .expected_len = 3, .expected_mblen = 2
1542
+ };
1543
+ run_sanitize_test(&t);
1544
+ }
1545
+
1546
+ // Verify the original bug scenario from the PR: ms² being preserved
1547
+ {
1548
+ sanitize_test_t t = {
1549
+ .name = "regression_ms_squared",
1550
+ .input = (unsigned char *)"ms\xC2\xB2", // ms²
1551
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1552
+ .expected_output = "ms\xC2\xB2", .expected_len = 4, .expected_mblen = 3
1553
+ };
1554
+ run_sanitize_test(&t);
1555
+ }
1556
+}
1557
+
1558
+// ============================================================================
1559
+// TEST: ALL CONTROL CHARACTERS (0x00-0x1F, 0x7F)
1560
+// ============================================================================
1561
+
1562
+static void test_all_control_characters(void) {
1563
+ fprintf(stderr, "\n=== All Control Characters ===\n");
1564
+
1565
+ // Test each control character 0x01-0x1F individually
1566
+ for (unsigned int ctrl = 1; ctrl < 32; ctrl++) {
1567
+ unsigned char input[4] = {'A', (unsigned char)ctrl, 'B', '\0'};
1568
+ char expected[8];
1569
+
1570
+ // With test_rrd_char_map, all control chars become space
1571
+ snprintf(expected, sizeof(expected), "A B");
1572
+
1573
+ char name[32];
1574
+ snprintf(name, sizeof(name), "ctrl_0x%02X", ctrl);
1575
+
1576
+ size_t guard = 16;
1577
+ unsigned char *buffer = callocz(1, 32 + guard * 2);
1578
+ unsigned char *dst = buffer + guard;
1579
+ memset(buffer, 0xAA, guard);
1580
+ memset(dst + 32, 0xBB, guard);
1581
+ memset(dst, 0xCC, 32);
1582
+
1583
+ size_t mblen = 0;
1584
+ text_sanitize(dst, input, 32, test_rrd_char_map, true, "", &mblen);
1585
+
1586
+ bool overflow = false;
1587
+ for (size_t i = 0; i < guard; i++) {
1588
+ if (buffer[i] != 0xAA || dst[32 + i] != 0xBB) {
1589
+ overflow = true;
1590
+ break;
1591
+ }
1592
+ }
1593
+
1594
+ TEST_ASSERT(name, !overflow && strcmp((char *)dst, expected) == 0,
1595
+ "ctrl=0x%02X: overflow=%d, expected '%s', got '%s'", ctrl, overflow, expected, dst);
1596
+
1597
+ freez(buffer);
1598
+ }
1599
+
1600
+ // Test DEL (0x7F)
1601
+ {
1602
+ unsigned char input[] = {'A', 0x7F, 'B', '\0'};
1603
+ sanitize_test_t t = {
1604
+ .name = "ctrl_DEL_0x7F",
1605
+ .input = input,
1606
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1607
+ .expected_output = "A B", .expected_len = 3, .expected_mblen = 3
1608
+ };
1609
+ run_sanitize_test(&t);
1610
+ }
1611
+
1612
+ // Multiple different control characters in sequence
1613
+ {
1614
+ unsigned char input[] = {'X', 0x01, 0x02, 0x03, 0x04, 0x05, 'Y', '\0'};
1615
+ sanitize_test_t t = {
1616
+ .name = "ctrl_multiple_sequence",
1617
+ .input = input,
1618
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1619
+ .expected_output = "X Y", .expected_len = 3, .expected_mblen = 3
1620
+ // All control chars become spaces, then deduplicated
1621
+ };
1622
+ run_sanitize_test(&t);
1623
+ }
1624
+
1625
+ // Bell character (0x07) - common in terminal output
1626
+ {
1627
+ unsigned char input[] = {'b', 'e', 'l', 'l', 0x07, 't', 'e', 's', 't', '\0'};
1628
+ sanitize_test_t t = {
1629
+ .name = "ctrl_bell",
1630
+ .input = input,
1631
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1632
+ .expected_output = "bell test", .expected_len = 9, .expected_mblen = 9
1633
+ };
1634
+ run_sanitize_test(&t);
1635
+ }
1636
+
1637
+ // Escape sequence (0x1B) - ANSI escape
1638
+ {
1639
+ unsigned char input[] = {0x1B, '[', '3', '1', 'm', 'r', 'e', 'd', 0x1B, '[', '0', 'm', '\0'};
1640
+ sanitize_test_t t = {
1641
+ .name = "ctrl_ansi_escape",
1642
+ .input = input,
1643
+ .dst_size = 32, .char_map = test_rrd_char_map, .utf = true, .empty = "",
1644
+ .expected_output = "[31mred [0m", .expected_len = 11, .expected_mblen = 11
1645
+ // 0x1B becomes space, which is leading/duplicated so gets handled
1646
+ };
1647
+ run_sanitize_test(&t);
1648
+ }
1649
+}
1650
+
1651
+// ============================================================================
1652
+// TEST: REAL-WORLD METRIC STRINGS
1653
+// ============================================================================
1654
+
1655
+static void test_real_world_metrics(void) {
1656
+ fprintf(stderr, "\n=== Real-World Metric Strings ===\n");
1657
+
1658
+ // Use actual rrd_string_allowed_chars
1659
+ extern unsigned char rrd_string_allowed_chars[256];
1660
+
1661
+ // CPU metric title
1662
+ {
1663
+ sanitize_test_t t = {
1664
+ .name = "metric_cpu_title",
1665
+ .input = (unsigned char *)"CPU utilization (user, system, iowait, irq, softirq)",
1666
+ .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1667
+ .expected_output = "CPU utilization (user, system, iowait, irq, softirq)", .expected_len = 52, .expected_mblen = 52
1668
+ };
1669
+ run_sanitize_test(&t);
1670
+ }
1671
+
1672
+ // Memory with units
1673
+ {
1674
+ sanitize_test_t t = {
1675
+ .name = "metric_memory_unit",
1676
+ .input = (unsigned char *)"Memory (MiB)",
1677
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1678
+ .expected_output = "Memory (MiB)", .expected_len = 12, .expected_mblen = 12
1679
+ };
1680
+ run_sanitize_test(&t);
1681
+ }
1682
+
1683
+ // Network bandwidth with special chars
1684
+ {
1685
+ sanitize_test_t t = {
1686
+ .name = "metric_network_bandwidth",
1687
+ .input = (unsigned char *)"eth0: kilobits/s",
1688
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1689
+ .expected_output = "eth0: kilobits/s", .expected_len = 16, .expected_mblen = 16
1690
+ };
1691
+ run_sanitize_test(&t);
1692
+ }
1693
+
1694
+ // Disk I/O with latency units (microseconds)
1695
+ {
1696
+ sanitize_test_t t = {
1697
+ .name = "metric_disk_latency",
1698
+ .input = (unsigned char *)"Disk latency (\xC2\xB5s)", // µs
1699
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1700
+ .expected_output = "Disk latency (\xC2\xB5s)", .expected_len = 18, .expected_mblen = 17
1701
+ };
1702
+ run_sanitize_test(&t);
1703
+ }
1704
+
1705
+ // Temperature sensor
1706
+ {
1707
+ sanitize_test_t t = {
1708
+ .name = "metric_temperature",
1709
+ .input = (unsigned char *)"core_temp_0: \xC2\xB0""Celsius",
1710
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1711
+ // "core_temp_0: " (13) + "°" (2 bytes) + "Celsius" (7) = 22 bytes, 21 chars
1712
+ .expected_output = "core_temp_0: \xC2\xB0""Celsius", .expected_len = 22, .expected_mblen = 21
1713
+ };
1714
+ run_sanitize_test(&t);
1715
+ }
1716
+
1717
+ // Docker container ID (common in Netdata)
1718
+ {
1719
+ sanitize_test_t t = {
1720
+ .name = "metric_docker_id",
1721
+ .input = (unsigned char *)"container_a1b2c3d4e5f6",
1722
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1723
+ .expected_output = "container_a1b2c3d4e5f6", .expected_len = 22, .expected_mblen = 22
1724
+ };
1725
+ run_sanitize_test(&t);
1726
+ }
1727
+
1728
+ // Kubernetes pod name
1729
+ {
1730
+ sanitize_test_t t = {
1731
+ .name = "metric_k8s_pod",
1732
+ .input = (unsigned char *)"nginx-deployment-5d8b7f9-xyz12",
1733
+ .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1734
+ .expected_output = "nginx-deployment-5d8b7f9-xyz12", .expected_len = 30, .expected_mblen = 30
1735
+ };
1736
+ run_sanitize_test(&t);
1737
+ }
1738
+
1739
+ // Windows path (backslash conversion)
1740
+ {
1741
+ sanitize_test_t t = {
1742
+ .name = "metric_windows_path",
1743
+ .input = (unsigned char *)"C:\\Program Files\\Application\\metric.exe",
1744
+ .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1745
+ .expected_output = "C:/Program Files/Application/metric.exe", .expected_len = 39, .expected_mblen = 39
1746
+ };
1747
+ run_sanitize_test(&t);
1748
+ }
1749
+
1750
+ // Prometheus metric with labels (quotes converted)
1751
+ {
1752
+ sanitize_test_t t = {
1753
+ .name = "metric_prometheus_labels",
1754
+ .input = (unsigned char *)"http_requests{method=\"POST\",status=\"200\"}",
1755
+ .dst_size = 64, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1756
+ .expected_output = "http_requests{method='POST',status='200'}", .expected_len = 41, .expected_mblen = 41
1757
+ };
1758
+ run_sanitize_test(&t);
1759
+ }
1760
+
1761
+ // Acceleration units (m/s²)
1762
+ {
1763
+ sanitize_test_t t = {
1764
+ .name = "metric_acceleration",
1765
+ .input = (unsigned char *)"Acceleration (m/s\xC2\xB2)",
1766
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1767
+ .expected_output = "Acceleration (m/s\xC2\xB2)", .expected_len = 20, .expected_mblen = 19
1768
+ };
1769
+ run_sanitize_test(&t);
1770
+ }
1771
+
1772
+ // Percentage with degree
1773
+ {
1774
+ sanitize_test_t t = {
1775
+ .name = "metric_angle_degree",
1776
+ .input = (unsigned char *)"Rotation angle: 90\xC2\xB0",
1777
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1778
+ // "Rotation angle: 90" (18) + "°" (2 bytes) = 20 bytes, 19 chars
1779
+ .expected_output = "Rotation angle: 90\xC2\xB0", .expected_len = 20, .expected_mblen = 19
1780
+ };
1781
+ run_sanitize_test(&t);
1782
+ }
1783
+
1784
+ // IPv6 address in metric context
1785
+ {
1786
+ sanitize_test_t t = {
1787
+ .name = "metric_ipv6",
1788
+ .input = (unsigned char *)"host:2001:db8::1",
1789
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1790
+ .expected_output = "host:2001:db8::1", .expected_len = 16, .expected_mblen = 16
1791
+ };
1792
+ run_sanitize_test(&t);
1793
+ }
1794
+
1795
+ // Process name with parentheses and numbers
1796
+ {
1797
+ sanitize_test_t t = {
1798
+ .name = "metric_process_name",
1799
+ .input = (unsigned char *)"python3.11 (worker-1)",
1800
+ .dst_size = 32, .char_map = rrd_string_allowed_chars, .utf = true, .empty = "",
1801
+ .expected_output = "python3.11 (worker-1)", .expected_len = 21, .expected_mblen = 21
1802
+ };
1803
+ run_sanitize_test(&t);
1804
+ }
1805
+}
1806
+
1807
+// ============================================================================
1808
+// TEST: HEX ENCODING EDGE CASES
1809
+// ============================================================================
1810
+
1811
+static void test_hex_encoding_edge_cases(void) {
1812
+ fprintf(stderr, "\n=== Hex Encoding Edge Cases ===\n");
1813
+
1814
+ // Truncated 2-byte sequence - gets hex encoded
1815
+ // mblen counts the whole invalid sequence as 1 character
1816
+ {
1817
+ sanitize_test_t t = {
1818
+ .name = "hex_truncated_2byte",
1819
+ .input = (unsigned char *)"\xC2", // Truncated (needs continuation)
1820
+ .dst_size = 3, .char_map = identity_char_map, .utf = true, .empty = "",
1821
+ .expected_output = "c2", .expected_len = 2, .expected_mblen = 1
1822
+ };
1823
+ run_sanitize_test(&t);
1824
+ }
1825
+
1826
+ // Not enough space for hex encoding
1827
+ {
1828
+ sanitize_test_t t = {
1829
+ .name = "hex_no_space",
1830
+ .input = (unsigned char *)"\xC2", // Truncated
1831
+ .dst_size = 2, .char_map = identity_char_map, .utf = true, .empty = "",
1832
+ // Can't fit "c2" (needs 2 chars + NUL = 3) - nothing written, mblen=0
1833
+ .expected_output = "", .expected_len = 0, .expected_mblen = 0
1834
+ };
1835
+ run_sanitize_test(&t);
1836
+ }
1837
+
1838
+ // Multiple truncated sequences - each counts as 1 mblen
1839
+ {
1840
+ sanitize_test_t t = {
1841
+ .name = "hex_multiple_truncated",
1842
+ .input = (unsigned char *)"\xC2\xC3", // Two truncated 2-byte starts
1843
+ .dst_size = 5, .char_map = identity_char_map, .utf = true, .empty = "",
1844
+ .expected_output = "c2c3", .expected_len = 4, .expected_mblen = 2
1845
+ };
1846
+ run_sanitize_test(&t);
1847
+ }
1848
+
1849
+ // ASCII + truncated hex
1850
+ {
1851
+ sanitize_test_t t = {
1852
+ .name = "hex_ascii_plus_truncated",
1853
+ .input = (unsigned char *)"AB\xC2", // AB + truncated
1854
+ .dst_size = 5, .char_map = identity_char_map, .utf = true, .empty = "",
1855
+ .expected_output = "ABc2", .expected_len = 4, .expected_mblen = 3
1856
+ };
1857
+ run_sanitize_test(&t);
1858
+ }
1859
+
1860
+ // 0xFE and 0xFF don't match valid UTF-8 start patterns - hex encoded
1861
+ {
1862
+ sanitize_test_t t = {
1863
+ .name = "hex_FE_FF",
1864
+ .input = (unsigned char *)"\xFE\xFF",
1865
+ .dst_size = 8, .char_map = identity_char_map, .utf = true, .empty = "",
1866
+ .expected_output = "feff", .expected_len = 4, .expected_mblen = 2
1867
+ };
1868
+ run_sanitize_test(&t);
1869
+ }
1870
+
1871
+ // Structurally valid overlong + orphan continuation
1872
+ {
1873
+ sanitize_test_t t = {
1874
+ .name = "hex_valid_plus_orphan",
1875
+ .input = (unsigned char *)"X\xC0\x80\x80", // X + valid 2-byte + orphan
1876
+ .dst_size = 8, .char_map = identity_char_map, .utf = true, .empty = "",
1877
+ // \xC0\x80 is structurally valid (passes through), \x80 is orphan (char_map)
1878
+ .expected_output = "X\xC0\x80\x80", .expected_len = 4, .expected_mblen = 3
1879
+ };
1880
+ run_sanitize_test(&t);
1881
+ }
1882
+
1883
+ // Orphan continuation bytes go through char_map
1884
+ {
1885
+ sanitize_test_t t = {
1886
+ .name = "hex_orphan_continuations",
1887
+ .input = (unsigned char *)"\x80\x81\x82\x83",
1888
+ .dst_size = 16, .char_map = identity_char_map, .utf = true, .empty = "",
1889
+ // Orphan continuation bytes (10xxxxxx pattern) go through char_map
1890
+ .expected_output = "\x80\x81\x82\x83", .expected_len = 4, .expected_mblen = 4
1891
+ };
1892
+ run_sanitize_test(&t);
1893
+ }
1894
+}
1895
+
1896
+// ============================================================================
1897
+// TEST: STRESS AND EDGE CASES
1898
+// ============================================================================
1899
+
1900
+static void test_stress_and_edge_cases(void) {
1901
+ fprintf(stderr, "\n=== Stress and Edge Cases ===\n");
1902
+
1903
+ // Very long UTF-8 string
1904
+ {
1905
+ // Create string with 100 2-byte UTF-8 characters (200 bytes)
1906
+ unsigned char input[201];
1907
+ for (int i = 0; i < 100; i++) {
1908
+ input[i*2] = 0xC2;
1909
+ input[i*2+1] = 0xB0; // ° repeated 100 times
1910
+ }
1911
+ input[200] = '\0';
1912
+
1913
+ unsigned char expected[201];
1914
+ memcpy(expected, input, 201);
1915
+
1916
+ sanitize_test_t t = {
1917
+ .name = "stress_long_utf8",
1918
+ .input = input,
1919
+ .dst_size = 256, .char_map = identity_char_map, .utf = true, .empty = "",
1920
+ .expected_output = (char *)expected, .expected_len = 200, .expected_mblen = 100
1921
+ };
1922
+ run_sanitize_test(&t);
1923
+ }
1924
+
1925
+ // Alternating valid UTF-8 and structurally valid overlong (both pass through)
1926
+ {
1927
+ sanitize_test_t t = {
1928
+ .name = "stress_alternating",
1929
+ .input = (unsigned char *)"\xC2\xB0\xC0\x80\xC2\xB0\xC0\x80",
1930
+ .dst_size = 64, .char_map = identity_char_map, .utf = true, .empty = "",
1931
+ // Both \xC2\xB0 and \xC0\x80 are structurally valid 2-byte sequences
1932
+ .expected_output = "\xC2\xB0\xC0\x80\xC2\xB0\xC0\x80", .expected_len = 8, .expected_mblen = 4
1933
+ };
1934
+ run_sanitize_test(&t);
1935
+ }
1936
+
1937
+ // All 256 byte values (non-UTF-8 mode)
1938
+ {
1939
+ unsigned char input[256];
1940
+ for (int i = 1; i < 256; i++) // Skip NUL
1941
+ input[i-1] = (unsigned char)i;
1942
+ input[255] = '\0';
1943
+
1944
+ // With identity map, most pass through; control chars and high bytes
1945
+ // will be handled. This just tests no crash.
1946
+ unsigned char dst[512];
1947
+ size_t len = text_sanitize(dst, input, sizeof(dst), identity_char_map, false, "", NULL);
1948
+ TEST_ASSERT("stress_all_bytes", len > 0, "Expected non-zero length, got %zu", len);
1949
+ }
1950
+
1951
+ // Rapid buffer size changes (fuzz-like)
1952
+ {
1953
+ const unsigned char *input = (unsigned char *)"test\xC2\xB0\xE2\x82\xAC\xF0\x9F\x98\x80";
1954
+ bool all_ok = true;
1955
+
1956
+ for (size_t sz = 1; sz <= 20; sz++) {
1957
+ unsigned char *buffer = callocz(1, sz + 32);
1958
+ unsigned char *dst = buffer + 16;
1959
+ memset(buffer, 0xAA, 16);
1960
+ memset(dst + sz, 0xBB, 16);
1961
+
1962
+ text_sanitize(dst, input, sz, identity_char_map, true, "", NULL);
1963
+
1964
+ // Check for overflow
1965
+ for (size_t i = 0; i < 16; i++) {
1966
+ if (buffer[i] != 0xAA || dst[sz + i] != 0xBB) {
1967
+ all_ok = false;
1968
+ fprintf(stderr, " Overflow at buffer size %zu\n", sz);
1969
+ break;
1970
+ }
1971
+ }
1972
+ freez(buffer);
1973
+ }
1974
+ TEST_ASSERT("stress_buffer_sizes", all_ok, "Buffer overflow detected in size sweep");
1975
+ }
1976
+
1977
+ // Repeated sanitization (idempotent for valid input)
1978
+ {
1979
+ unsigned char input[] = "test\xC2\xB0""C";
1980
+ unsigned char dst1[32], dst2[32];
1981
+
1982
+ text_sanitize(dst1, input, sizeof(dst1), identity_char_map, true, "", NULL);
1983
+ text_sanitize(dst2, dst1, sizeof(dst2), identity_char_map, true, "", NULL);
1984
+
1985
+ TEST_ASSERT("stress_idempotent", strcmp((char *)dst1, (char *)dst2) == 0,
1986
+ "Not idempotent: '%s' vs '%s'", dst1, dst2);
1987
+ }
1988
+}
1989
+
1990
+// ============================================================================
1991
+// MAIN TEST RUNNER
1992
+// ============================================================================
1993
+
1994
+int utf8_sanitizer_unittest(void) {
1995
+ fprintf(stderr, "\n");
1996
+ fprintf(stderr, "================================================================\n");
1997
+ fprintf(stderr, "UTF-8 Sanitizer Exhaustive Unit Tests\n");
1998
+ fprintf(stderr, "================================================================\n");
1999
+
2000
+ init_char_maps();
2001
+
2002
+ test_valid_utf8_sequences();
2003
+ test_invalid_utf8_sequences();
2004
+ test_buffer_boundaries();
2005
+ test_char_map_transformations();
2006
+ test_space_handling();
2007
+ test_empty_and_special();
2008
+ test_utf_parameter();
2009
+ test_multibyte_length();
2010
+ test_rrd_string_allowed_chars();
2011
+ test_security_cases();
2012
+ test_regression_fixed_bugs();
2013
+ test_all_control_characters();
2014
+ test_real_world_metrics();
2015
+ test_hex_encoding_edge_cases();
2016
+ test_stress_and_edge_cases();
2017
+
2018
+ fprintf(stderr, "\n================================================================\n");
2019
+ fprintf(stderr, "Tests run: %d, Passed: %d, Failed: %d\n", tests_run, tests_passed, tests_failed);
2020
+
2021
+ if (tests_failed == 0) {
2022
+ fprintf(stderr, "ALL TESTS PASSED\n");
2023
+ } else {
2024
+ fprintf(stderr, "SOME TESTS FAILED\n");
2025
+ }
2026
+ fprintf(stderr, "================================================================\n\n");
2027
+
2028
+ return tests_failed;
2029
+}