357
char **make_augmented_environ(const char *const *vars);
358
void free_environ(char **env);
359
360
+/**
361
+ * Converts UTF-8 encoded string to UTF-16LE.
362
+ *
363
+ * To support repositories with legacy-encoded file names, invalid UTF-8 bytes
364
+ * 0xa0 - 0xff are converted to corresponding printable Unicode chars \u00a0 -
365
+ * \u00ff, and invalid UTF-8 bytes 0x80 - 0x9f (which would make non-printable
366
+ * Unicode) are converted to hex-code.
367
+ *
368
+ * Lead-bytes not followed by an appropriate number of trail-bytes, over-long
369
+ * encodings and 4-byte encodings > \u10ffff are detected as invalid UTF-8.
370
+ *
371
+ * Maximum space requirement for the target buffer is two wide chars per UTF-8
372
+ * char (((strlen(utf) * 2) + 1) [* sizeof(wchar_t)]).
373
+ *
374
+ * The maximum space is needed only if the entire input string consists of
375
+ * invalid UTF-8 bytes in range 0x80-0x9f, as per the following table:
376
+ *
377
+ * | | UTF-8 | UTF-16 |
378
+ * Code point | UTF-8 sequence | bytes | words | ratio
379
+ * --------------+-------------------+-------+--------+-------
380
+ * 000000-00007f | 0-7f | 1 | 1 | 1
381
+ * 000080-0007ff | c2-df + 80-bf | 2 | 1 | 0.5
382
+ * 000800-00ffff | e0-ef + 2 * 80-bf | 3 | 1 | 0.33
383
+ * 010000-10ffff | f0-f4 + 3 * 80-bf | 4 | 2 (a) | 0.5
384
+ * invalid | 80-9f | 1 | 2 (b) | 2
385
+ * invalid | a0-ff | 1 | 1 | 1
386
+ *
387
+ * (a) encoded as UTF-16 surrogate pair
388
+ * (b) encoded as two hex digits
389
+ *
390
+ * Note that, while the UTF-8 encoding scheme can be extended to 5-byte, 6-byte
391
+ * or even indefinite-byte sequences, the largest valid code point \u10ffff
392
+ * encodes as only 4 UTF-8 bytes.
393
+ *
394
+ * Parameters:
395
+ * wcs: wide char target buffer
396
+ * utf: string to convert
397
+ * wcslen: size of target buffer (in wchar_t's)
398
+ * utflen: size of string to convert, or -1 if 0-terminated
399
+ *
400
+ * Returns:
401
+ * length of converted string (_wcslen(wcs)), or -1 on failure
402
+ *
403
+ * Errors:
404
+ * EINVAL: one of the input parameters is invalid (e.g. NULL)
405
+ * ERANGE: the output buffer is too small
406
+ */
407
+int xutftowcsn(wchar_t *wcs, const char *utf, size_t wcslen, int utflen);
408
+
409
+/**
410
+ * Simplified variant of xutftowcsn, assumes input string is \0-terminated.
411
+ */
412
+static inline int xutftowcs(wchar_t *wcs, const char *utf, size_t wcslen)
413
+{
414
+ return xutftowcsn(wcs, utf, wcslen, -1);
415
+}
416
+
417
+/**
418
+ * Simplified file system specific variant of xutftowcsn, assumes output
419
+ * buffer size is MAX_PATH wide chars and input string is \0-terminated,
420
+ * fails with ENAMETOOLONG if input string is too long.
421
+ */
422
+static inline int xutftowcs_path(wchar_t *wcs, const char *utf)
423
+{
424
+ int result = xutftowcsn(wcs, utf, MAX_PATH, -1);
425
+ if (result < 0 && errno == ERANGE)
426
+ errno = ENAMETOOLONG;
427
+ return result;
428
+}
429
+
430
+/**
431
+ * Converts UTF-16LE encoded string to UTF-8.
432
+ *
433
+ * Maximum space requirement for the target buffer is three UTF-8 chars per
434
+ * wide char ((_wcslen(wcs) * 3) + 1).
435
+ *
436
+ * The maximum space is needed only if the entire input string consists of
437
+ * UTF-16 words in range 0x0800-0xd7ff or 0xe000-0xffff (i.e. \u0800-\uffff
438
+ * modulo surrogate pairs), as per the following table:
439
+ *
440
+ * | | UTF-16 | UTF-8 |
441
+ * Code point | UTF-16 sequence | words | bytes | ratio
442
+ * --------------+-----------------------+--------+-------+-------
443
+ * 000000-00007f | 0000-007f | 1 | 1 | 1
444
+ * 000080-0007ff | 0080-07ff | 1 | 2 | 2
445
+ * 000800-00ffff | 0800-d7ff / e000-ffff | 1 | 3 | 3
446
+ * 010000-10ffff | d800-dbff + dc00-dfff | 2 | 4 | 2
447
+ *
448
+ * Note that invalid code points > 10ffff cannot be represented in UTF-16.
449
+ *
450
+ * Parameters:
451
+ * utf: target buffer
452
+ * wcs: wide string to convert
453
+ * utflen: size of target buffer
454
+ *
455
+ * Returns:
456
+ * length of converted string, or -1 on failure
457
+ *
458
+ * Errors:
459
+ * EINVAL: one of the input parameters is invalid (e.g. NULL)
460
+ * ERANGE: the output buffer is too small
461
+ */
462
+int xwcstoutf(char *utf, const wchar_t *wcs, size_t utflen);
463
+
464
/*
465
* A critical section used in the implementation of the spawn
466
* functions (mingw_spawnv[p]e()) and waitpid(). Intialised in