OdbDesignLib
OdbDesign ODB++ Parsing Library
 
Loading...
Searching...
No Matches
Utf8Sanitizer.cpp
Go to the documentation of this file.
1
11#include "Utf8Sanitizer.h"
12
13#include <google/protobuf/descriptor.h>
14#include <google/protobuf/message.h>
15
16#include <cstdint>
17#include <cstdio>
18#include <cstdlib>
19#include <string>
20
21namespace Odb::Lib::Text
22{
23 namespace
24 {
25 // clang-format off
36 constexpr uint32_t kCp1252ToUnicode[256] = {
37 // 0x00-0x7F: ASCII (identity)
38 0x0000, 0x0001, 0x0002, 0x0003, 0x0004, 0x0005, 0x0006, 0x0007,
39 0x0008, 0x0009, 0x000A, 0x000B, 0x000C, 0x000D, 0x000E, 0x000F,
40 0x0010, 0x0011, 0x0012, 0x0013, 0x0014, 0x0015, 0x0016, 0x0017,
41 0x0018, 0x0019, 0x001A, 0x001B, 0x001C, 0x001D, 0x001E, 0x001F,
42 0x0020, 0x0021, 0x0022, 0x0023, 0x0024, 0x0025, 0x0026, 0x0027,
43 0x0028, 0x0029, 0x002A, 0x002B, 0x002C, 0x002D, 0x002E, 0x002F,
44 0x0030, 0x0031, 0x0032, 0x0033, 0x0034, 0x0035, 0x0036, 0x0037,
45 0x0038, 0x0039, 0x003A, 0x003B, 0x003C, 0x003D, 0x003E, 0x003F,
46 0x0040, 0x0041, 0x0042, 0x0043, 0x0044, 0x0045, 0x0046, 0x0047,
47 0x0048, 0x0049, 0x004A, 0x004B, 0x004C, 0x004D, 0x004E, 0x004F,
48 0x0050, 0x0051, 0x0052, 0x0053, 0x0054, 0x0055, 0x0056, 0x0057,
49 0x0058, 0x0059, 0x005A, 0x005B, 0x005C, 0x005D, 0x005E, 0x005F,
50 0x0060, 0x0061, 0x0062, 0x0063, 0x0064, 0x0065, 0x0066, 0x0067,
51 0x0068, 0x0069, 0x006A, 0x006B, 0x006C, 0x006D, 0x006E, 0x006F,
52 0x0070, 0x0071, 0x0072, 0x0073, 0x0074, 0x0075, 0x0076, 0x0077,
53 0x0078, 0x0079, 0x007A, 0x007B, 0x007C, 0x007D, 0x007E, 0x007F,
54
55 // 0x80-0x9F: CP1252 special characters
56 0x20AC, // 0x80: EURO SIGN
57 0xFFFD, // 0x81: UNDEFINED → REPLACEMENT CHARACTER
58 0x201A, // 0x82: SINGLE LOW-9 QUOTATION MARK
59 0x0192, // 0x83: LATIN SMALL LETTER F WITH HOOK
60 0x201E, // 0x84: DOUBLE LOW-9 QUOTATION MARK
61 0x2026, // 0x85: HORIZONTAL ELLIPSIS
62 0x2020, // 0x86: DAGGER
63 0x2021, // 0x87: DOUBLE DAGGER
64 0x02C6, // 0x88: MODIFIER LETTER CIRCUMFLEX ACCENT
65 0x2030, // 0x89: PER MILLE SIGN
66 0x0160, // 0x8A: LATIN CAPITAL LETTER S WITH CARON
67 0x2039, // 0x8B: SINGLE LEFT-POINTING ANGLE QUOTATION MARK
68 0x0152, // 0x8C: LATIN CAPITAL LIGATURE OE
69 0xFFFD, // 0x8D: UNDEFINED → REPLACEMENT CHARACTER
70 0x017D, // 0x8E: LATIN CAPITAL LETTER Z WITH CARON
71 0xFFFD, // 0x8F: UNDEFINED → REPLACEMENT CHARACTER
72 0xFFFD, // 0x90: UNDEFINED → REPLACEMENT CHARACTER
73 0x2018, // 0x91: LEFT SINGLE QUOTATION MARK
74 0x2019, // 0x92: RIGHT SINGLE QUOTATION MARK
75 0x201C, // 0x93: LEFT DOUBLE QUOTATION MARK
76 0x201D, // 0x94: RIGHT DOUBLE QUOTATION MARK
77 0x2022, // 0x95: BULLET
78 0x2013, // 0x96: EN DASH
79 0x2014, // 0x97: EM DASH
80 0x02DC, // 0x98: SMALL TILDE
81 0x2122, // 0x99: TRADE MARK SIGN
82 0x0161, // 0x9A: LATIN SMALL LETTER S WITH CARON
83 0x203A, // 0x9B: SINGLE RIGHT-POINTING ANGLE QUOTATION MARK
84 0x0153, // 0x9C: LATIN SMALL LIGATURE OE
85 0xFFFD, // 0x9D: UNDEFINED → REPLACEMENT CHARACTER
86 0x017E, // 0x9E: LATIN SMALL LETTER Z WITH CARON
87 0x0178, // 0x9F: LATIN CAPITAL LETTER Y WITH DIAERESIS
88
89 // 0xA0-0xFF: ISO-8859-1 upper half (Latin-1 Supplement)
90 0x00A0, 0x00A1, 0x00A2, 0x00A3, 0x00A4, 0x00A5, 0x00A6, 0x00A7,
91 0x00A8, 0x00A9, 0x00AA, 0x00AB, 0x00AC, 0x00AD, 0x00AE, 0x00AF,
92 0x00B0, 0x00B1, 0x00B2, 0x00B3, 0x00B4, 0x00B5, 0x00B6, 0x00B7,
93 0x00B8, 0x00B9, 0x00BA, 0x00BB, 0x00BC, 0x00BD, 0x00BE, 0x00BF,
94 0x00C0, 0x00C1, 0x00C2, 0x00C3, 0x00C4, 0x00C5, 0x00C6, 0x00C7,
95 0x00C8, 0x00C9, 0x00CA, 0x00CB, 0x00CC, 0x00CD, 0x00CE, 0x00CF,
96 0x00D0, 0x00D1, 0x00D2, 0x00D3, 0x00D4, 0x00D5, 0x00D6, 0x00D7,
97 0x00D8, 0x00D9, 0x00DA, 0x00DB, 0x00DC, 0x00DD, 0x00DE, 0x00DF,
98 0x00E0, 0x00E1, 0x00E2, 0x00E3, 0x00E4, 0x00E5, 0x00E6, 0x00E7,
99 0x00E8, 0x00E9, 0x00EA, 0x00EB, 0x00EC, 0x00ED, 0x00EE, 0x00EF,
100 0x00F0, 0x00F1, 0x00F2, 0x00F3, 0x00F4, 0x00F5, 0x00F6, 0x00F7,
101 0x00F8, 0x00F9, 0x00FA, 0x00FB, 0x00FC, 0x00FD, 0x00FE, 0x00FF
102 };
103 // clang-format on
104
111 inline int EncodeUtf8Codepoint(uint32_t codepoint, char* output) noexcept
112 {
113 if (codepoint <= 0x7F)
114 {
115 output[0] = static_cast<char>(codepoint);
116 return 1;
117 }
118 else if (codepoint <= 0x7FF)
119 {
120 output[0] = static_cast<char>(0xC0 | (codepoint >> 6));
121 output[1] = static_cast<char>(0x80 | (codepoint & 0x3F));
122 return 2;
123 }
124 else if (codepoint <= 0xFFFF)
125 {
126 output[0] = static_cast<char>(0xE0 | (codepoint >> 12));
127 output[1] = static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F));
128 output[2] = static_cast<char>(0x80 | (codepoint & 0x3F));
129 return 3;
130 }
131 else
132 {
133 output[0] = static_cast<char>(0xF0 | (codepoint >> 18));
134 output[1] = static_cast<char>(0x80 | ((codepoint >> 12) & 0x3F));
135 output[2] = static_cast<char>(0x80 | ((codepoint >> 6) & 0x3F));
136 output[3] = static_cast<char>(0x80 | (codepoint & 0x3F));
137 return 4;
138 }
139 }
140
149 std::size_t Utf8SequenceLengthAt(const uint8_t* bytes, std::size_t i, std::size_t size) noexcept
150 {
151 const uint8_t byte = bytes[i];
152
153 // Single byte: 0xxxxxxx
154 if (byte <= 0x7F)
155 {
156 return 1;
157 }
158
159 // Two bytes: 110xxxxx 10xxxxxx
160 if ((byte & 0xE0) == 0xC0)
161 {
162 if (i + 1 >= size) return 0;
163 if ((bytes[i + 1] & 0xC0) != 0x80) return 0;
164
165 // Check for overlong encoding
166 const uint32_t codepoint = ((byte & 0x1F) << 6) | (bytes[i + 1] & 0x3F);
167 if (codepoint < 0x80) return 0;
168
169 return 2;
170 }
171
172 // Three bytes: 1110xxxx 10xxxxxx 10xxxxxx
173 if ((byte & 0xF0) == 0xE0)
174 {
175 if (i + 2 >= size) return 0;
176 if ((bytes[i + 1] & 0xC0) != 0x80 || (bytes[i + 2] & 0xC0) != 0x80)
177 return 0;
178
179 const uint32_t codepoint =
180 ((byte & 0x0F) << 12) | ((bytes[i + 1] & 0x3F) << 6) | (bytes[i + 2] & 0x3F);
181
182 // Check for overlong encoding
183 if (codepoint < 0x800) return 0;
184
185 // Reject surrogate halves U+D800..U+DFFF
186 if (codepoint >= 0xD800 && codepoint <= 0xDFFF) return 0;
187
188 return 3;
189 }
190
191 // Four bytes: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx
192 if ((byte & 0xF8) == 0xF0)
193 {
194 if (i + 3 >= size) return 0;
195 if ((bytes[i + 1] & 0xC0) != 0x80 || (bytes[i + 2] & 0xC0) != 0x80 ||
196 (bytes[i + 3] & 0xC0) != 0x80)
197 return 0;
198
199 const uint32_t codepoint =
200 ((byte & 0x07) << 18) | ((bytes[i + 1] & 0x3F) << 12) |
201 ((bytes[i + 2] & 0x3F) << 6) | (bytes[i + 3] & 0x3F);
202
203 // Check for overlong encoding
204 if (codepoint < 0x10000) return 0;
205
206 // Reject code points above U+10FFFF
207 if (codepoint > 0x10FFFF) return 0;
208
209 return 4;
210 }
211
212 // Invalid leading byte (includes bare continuation bytes)
213 return 0;
214 }
215
216#ifndef NDEBUG
224 [[noreturn]] void FailOnInvalidUtf8(const std::string& fieldPath, const std::string& value)
225 {
226 std::fprintf(stderr, "Utf8Sanitizer assertion failed: invalid UTF-8 in %s: \"", fieldPath.c_str());
227 constexpr std::size_t kMaxSampleLen = 32;
228 const bool truncated = value.size() > kMaxSampleLen;
229 const std::size_t sampleLen = truncated ? kMaxSampleLen : value.size();
230 for (std::size_t i = 0; i < sampleLen; ++i)
231 {
232 const unsigned char c = static_cast<unsigned char>(value[i]);
233 if (c >= 0x20 && c < 0x7F)
234 {
235 std::fputc(c, stderr);
236 }
237 else
238 {
239 std::fprintf(stderr, "\\x%02X", c);
240 }
241 }
242 if (truncated)
243 {
244 std::fprintf(stderr, "...");
245 }
246 std::fprintf(stderr, "\"\n");
247 std::abort();
248 }
249
250 void AssertMessageStringsAreValidUtf8(const google::protobuf::Message& msg, std::string& fieldPath);
251
255 void AssertStringIsValidUtf8(const std::string& value, const std::string& fieldPath)
256 {
257 if (!value.empty() && !IsValidUtf8(value))
258 {
259 FailOnInvalidUtf8(fieldPath, value);
260 }
261 }
262
275 void AssertMessageStringsAreValidUtf8(const google::protobuf::Message& msg, std::string& fieldPath)
276 {
277 const google::protobuf::Descriptor* descriptor = msg.GetDescriptor();
278 const google::protobuf::Reflection* reflection = msg.GetReflection();
279
280 for (int i = 0; i < descriptor->field_count(); ++i)
281 {
282 const google::protobuf::FieldDescriptor* field = descriptor->field(i);
283 const std::size_t basePathLen = fieldPath.size();
284 fieldPath.append(".").append(field->name());
285
286 if (field->is_repeated())
287 {
288 const int count = reflection->FieldSize(msg, field);
289 if (field->type() == google::protobuf::FieldDescriptor::TYPE_STRING)
290 {
291 for (int j = 0; j < count; ++j)
292 {
293 AssertStringIsValidUtf8(reflection->GetRepeatedString(msg, field, j), fieldPath);
294 }
295 }
296 else if (field->cpp_type() == google::protobuf::FieldDescriptor::CPPTYPE_MESSAGE)
297 {
298 for (int j = 0; j < count; ++j)
299 {
300 const std::size_t indexPathLen = fieldPath.size();
301 fieldPath.append("[").append(std::to_string(j)).append("]");
302 AssertMessageStringsAreValidUtf8(reflection->GetRepeatedMessage(msg, field, j), fieldPath);
303 fieldPath.resize(indexPathLen);
304 }
305 }
306 }
307 else
308 {
309 // Skip unset fields when presence tracking is available
310 if (field->has_presence() && !reflection->HasField(msg, field))
311 {
312 fieldPath.resize(basePathLen);
313 continue;
314 }
315
316 if (field->type() == google::protobuf::FieldDescriptor::TYPE_STRING)
317 {
318 AssertStringIsValidUtf8(reflection->GetString(msg, field), fieldPath);
319 }
320 else if (field->cpp_type() == google::protobuf::FieldDescriptor::CPPTYPE_MESSAGE)
321 {
322 AssertMessageStringsAreValidUtf8(reflection->GetMessage(msg, field), fieldPath);
323 }
324 }
325
326 fieldPath.resize(basePathLen);
327 }
328 }
329#endif // !NDEBUG
330 } // anonymous namespace
331
332 bool IsValidUtf8(const char* data, std::size_t size) noexcept
333 {
334 if (data == nullptr && size > 0) return false;
335
336 const auto* bytes = reinterpret_cast<const uint8_t*>(data);
337 std::size_t i = 0;
338
339 while (i < size)
340 {
341 const std::size_t sequenceLength = Utf8SequenceLengthAt(bytes, i, size);
342 if (sequenceLength == 0) return false;
343 i += sequenceLength;
344 }
345
346 return true;
347 }
348
349 bool IsValidUtf8(std::string_view s) noexcept
350 {
351 return IsValidUtf8(s.data(), s.size());
352 }
353
354 std::string ToUtf8(std::string_view input)
355 {
356 // Fast path: already valid UTF-8
357 if (IsValidUtf8(input))
358 {
359 return std::string(input);
360 }
361
362 // Repair path: keep every valid UTF-8 sequence as-is and transcode each
363 // invalid byte individually as CP1252. Transcoding the whole string as
364 // CP1252 would corrupt the already-valid parts of mixed input into
365 // mojibake (e.g. a valid "Ö" would become "Ö").
366 // Worst case: each byte becomes 3 UTF-8 bytes (CP1252 0x80-0x9F map to
367 // BMP code points that encode as 3-byte UTF-8 sequences).
368 std::string result;
369 result.reserve(input.size() * 3);
370
371 const auto* bytes = reinterpret_cast<const uint8_t*>(input.data());
372 char utf8Buf[4];
373 std::size_t i = 0;
374
375 while (i < input.size())
376 {
377 const std::size_t sequenceLength = Utf8SequenceLengthAt(bytes, i, input.size());
378 if (sequenceLength > 0)
379 {
380 result.append(input.data() + i, sequenceLength);
381 i += sequenceLength;
382 }
383 else
384 {
385 const uint32_t codepoint = kCp1252ToUnicode[bytes[i]];
386 const int len = EncodeUtf8Codepoint(codepoint, utf8Buf);
387 result.append(utf8Buf, len);
388 ++i;
389 }
390 }
391
392 return result;
393 }
394
395 void SanitizeToUtf8(std::string& s)
396 {
397 if (!IsValidUtf8(s))
398 {
399 s = ToUtf8(s);
400 }
401 }
402
403 void AssertAllStringFieldsAreValidUtf8(const google::protobuf::Message& msg, std::string_view msgName)
404 {
405#ifndef NDEBUG
406 std::string fieldPath(msgName);
407 AssertMessageStringsAreValidUtf8(msg, fieldPath);
408#else
409 (void)msg;
410 (void)msgName;
411#endif
412 }
413
414} // namespace Odb::Lib::Text
void AssertAllStringFieldsAreValidUtf8(const google::protobuf::Message &msg, std::string_view msgName)
Debug-only assertion to verify all string fields in a message are valid UTF-8.
std::string ToUtf8(std::string_view input)
Converts input to valid UTF-8.
void SanitizeToUtf8(std::string &s)
In-place convenience overload.
bool IsValidUtf8(const char *data, std::size_t size) noexcept
Validates that a byte sequence is valid UTF-8 per RFC 3629.
UTF-8 validation and Windows-1252 to UTF-8 transcoding utilities.