|
1 | 1 | #pragma once |
2 | 2 |
|
| 3 | +#include <cstddef> |
3 | 4 | #include <cstdlib> |
4 | 5 | #include <string> |
| 6 | +#include <string_view> |
| 7 | +#include <type_traits> |
5 | 8 |
|
6 | 9 | namespace headsetcontrol { |
7 | 10 |
|
@@ -60,4 +63,174 @@ inline std::string wstring_to_string(const wchar_t* wstr) |
60 | 63 | #endif |
61 | 64 | } |
62 | 65 |
|
| 66 | +/** |
| 67 | + * @brief Convert UTF-8 string to wide string |
| 68 | + * |
| 69 | + * Inverse of wstring_to_string() for the UTF-8 strings carried in DeviceMetadata. |
| 70 | + * The decoding is written out rather than delegated to mbstowcs() so that it does |
| 71 | + * not depend on the process locale: a device-supplied name arrives as raw bytes |
| 72 | + * off the wire and never passes through the C library's conversion on the way in. |
| 73 | + * |
| 74 | + * Widening byte by byte instead would produce one wide character per byte, so a |
| 75 | + * name like "Muller" spelled with an umlaut would come back out mangled. |
| 76 | + * |
| 77 | + * Malformed input is replaced with U+FFFD rather than rejected, so one bad byte |
| 78 | + * in a name read from a device does not discard the rest of it. |
| 79 | + * |
| 80 | + * @param str UTF-8 string |
| 81 | + * @return Wide string: UTF-16 where wchar_t is 16 bits, UTF-32 where it is wider |
| 82 | + */ |
| 83 | +inline std::wstring string_to_wstring(std::string_view str) |
| 84 | +{ |
| 85 | + constexpr char32_t REPLACEMENT = 0xFFFD; |
| 86 | + constexpr char32_t MAX_CODEPOINT = 0x10FFFF; |
| 87 | + |
| 88 | + std::wstring result; |
| 89 | + result.reserve(str.size()); |
| 90 | + |
| 91 | + auto append = [&result](char32_t codepoint) { |
| 92 | + if constexpr (sizeof(wchar_t) >= 4) { |
| 93 | + result += static_cast<wchar_t>(codepoint); |
| 94 | + } else if (codepoint <= 0xFFFF) { |
| 95 | + result += static_cast<wchar_t>(codepoint); |
| 96 | + } else { |
| 97 | + // Split into a UTF-16 surrogate pair, as on Windows where wchar_t is 16 bits |
| 98 | + const char32_t offset = codepoint - 0x10000; |
| 99 | + result += static_cast<wchar_t>(0xD800 + (offset >> 10)); |
| 100 | + result += static_cast<wchar_t>(0xDC00 + (offset & 0x3FF)); |
| 101 | + } |
| 102 | + }; |
| 103 | + |
| 104 | + for (std::size_t i = 0; i < str.size();) { |
| 105 | + const auto lead = static_cast<unsigned char>(str[i]); |
| 106 | + |
| 107 | + // Length of the sequence, and the smallest codepoint it may legally encode - |
| 108 | + // a larger sequence than a codepoint needs is an overlong encoding |
| 109 | + std::size_t continuations = 0; |
| 110 | + char32_t codepoint = 0; |
| 111 | + char32_t minimum = 0; |
| 112 | + if (lead < 0x80) { |
| 113 | + codepoint = lead; |
| 114 | + } else if ((lead & 0xE0) == 0xC0) { |
| 115 | + continuations = 1; |
| 116 | + codepoint = lead & 0x1FU; |
| 117 | + minimum = 0x80; |
| 118 | + } else if ((lead & 0xF0) == 0xE0) { |
| 119 | + continuations = 2; |
| 120 | + codepoint = lead & 0x0FU; |
| 121 | + minimum = 0x800; |
| 122 | + } else if ((lead & 0xF8) == 0xF0) { |
| 123 | + continuations = 3; |
| 124 | + codepoint = lead & 0x07U; |
| 125 | + minimum = 0x10000; |
| 126 | + } else { |
| 127 | + append(REPLACEMENT); |
| 128 | + ++i; |
| 129 | + continue; |
| 130 | + } |
| 131 | + |
| 132 | + if (i + continuations >= str.size()) { |
| 133 | + append(REPLACEMENT); |
| 134 | + ++i; |
| 135 | + continue; |
| 136 | + } |
| 137 | + |
| 138 | + bool valid = true; |
| 139 | + for (std::size_t k = 1; k <= continuations; ++k) { |
| 140 | + const auto continuation = static_cast<unsigned char>(str[i + k]); |
| 141 | + if ((continuation & 0xC0) != 0x80) { |
| 142 | + valid = false; |
| 143 | + break; |
| 144 | + } |
| 145 | + codepoint = (codepoint << 6) | (continuation & 0x3FU); |
| 146 | + } |
| 147 | + |
| 148 | + // Surrogates are not valid on their own, and are not encodable in UTF-8 |
| 149 | + const bool is_surrogate = codepoint >= 0xD800 && codepoint <= 0xDFFF; |
| 150 | + if (!valid || codepoint < minimum || codepoint > MAX_CODEPOINT || is_surrogate) { |
| 151 | + append(REPLACEMENT); |
| 152 | + ++i; |
| 153 | + continue; |
| 154 | + } |
| 155 | + |
| 156 | + append(codepoint); |
| 157 | + i += continuations + 1; |
| 158 | + } |
| 159 | + |
| 160 | + return result; |
| 161 | +} |
| 162 | + |
| 163 | +/** |
| 164 | + * @brief Convert wide string to UTF-8 string |
| 165 | + * |
| 166 | + * Counterpart to string_to_wstring(), and an exact inverse of it. Unlike |
| 167 | + * wstring_to_string() this does not consult the locale: it is for strings that are |
| 168 | + * defined to be UTF-8, such as the device names in DeviceMetadata, where going |
| 169 | + * through wcstombs() would replace anything outside the current locale's encoding |
| 170 | + * with '?' - which in the default C locale means every non-ASCII character. |
| 171 | + * |
| 172 | + * Where wchar_t is 16 bits a surrogate pair is recombined into the codepoint it |
| 173 | + * encodes; an unpaired surrogate is replaced with U+FFFD. |
| 174 | + * |
| 175 | + * @param str Wide string |
| 176 | + * @return UTF-8 string |
| 177 | + */ |
| 178 | +inline std::string wstring_to_utf8(std::wstring_view str) |
| 179 | +{ |
| 180 | + constexpr char32_t REPLACEMENT = 0xFFFD; |
| 181 | + constexpr char32_t MAX_CODEPOINT = 0x10FFFF; |
| 182 | + |
| 183 | + std::string result; |
| 184 | + result.reserve(str.size()); |
| 185 | + |
| 186 | + auto append = [&result](char32_t codepoint) { |
| 187 | + if (codepoint < 0x80) { |
| 188 | + result += static_cast<char>(codepoint); |
| 189 | + } else if (codepoint < 0x800) { |
| 190 | + result += static_cast<char>(0xC0 | (codepoint >> 6)); |
| 191 | + result += static_cast<char>(0x80 | (codepoint & 0x3F)); |
| 192 | + } else if (codepoint < 0x10000) { |
| 193 | + result += static_cast<char>(0xE0 | (codepoint >> 12)); |
| 194 | + result += static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F)); |
| 195 | + result += static_cast<char>(0x80 | (codepoint & 0x3F)); |
| 196 | + } else { |
| 197 | + result += static_cast<char>(0xF0 | (codepoint >> 18)); |
| 198 | + result += static_cast<char>(0x80 | ((codepoint >> 12) & 0x3F)); |
| 199 | + result += static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F)); |
| 200 | + result += static_cast<char>(0x80 | (codepoint & 0x3F)); |
| 201 | + } |
| 202 | + }; |
| 203 | + |
| 204 | + for (std::size_t i = 0; i < str.size(); ++i) { |
| 205 | + // Mask rather than cast: char32_t is unsigned, but wchar_t is signed on |
| 206 | + // some platforms, and a plain conversion would sign-extend |
| 207 | + auto codepoint = static_cast<char32_t>( |
| 208 | + static_cast<std::make_unsigned_t<wchar_t>>(str[i])); |
| 209 | + |
| 210 | + if (codepoint >= 0xD800 && codepoint <= 0xDBFF) { |
| 211 | + // High surrogate: needs the matching low surrogate to mean anything |
| 212 | + const bool has_low = i + 1 < str.size(); |
| 213 | + const auto low = has_low |
| 214 | + ? static_cast<char32_t>(static_cast<std::make_unsigned_t<wchar_t>>(str[i + 1])) |
| 215 | + : char32_t { 0 }; |
| 216 | + if (has_low && low >= 0xDC00 && low <= 0xDFFF) { |
| 217 | + codepoint = 0x10000 + ((codepoint - 0xD800) << 10) + (low - 0xDC00); |
| 218 | + ++i; |
| 219 | + } else { |
| 220 | + codepoint = REPLACEMENT; |
| 221 | + } |
| 222 | + } else if (codepoint >= 0xDC00 && codepoint <= 0xDFFF) { |
| 223 | + // Low surrogate without a high one before it |
| 224 | + codepoint = REPLACEMENT; |
| 225 | + } |
| 226 | + |
| 227 | + if (codepoint > MAX_CODEPOINT) { |
| 228 | + codepoint = REPLACEMENT; |
| 229 | + } |
| 230 | + append(codepoint); |
| 231 | + } |
| 232 | + |
| 233 | + return result; |
| 234 | +} |
| 235 | + |
63 | 236 | } // namespace headsetcontrol |
0 commit comments