FTXUI 7.0.3
C++ functional terminal UI.
Loading...
Searching...
No Matches
string.cpp
Go to the documentation of this file.
1// Copyright 2020 Arthur Sonzogni. All rights reserved.
2// Use of this source code is governed by the MIT license that can be found in
3// the LICENSE file.
4//
5// Content of this file was created thanks to:
6// -
7// https://www.unicode.org/Public/UCD/latest/ucd/auxiliary/WordBreakProperty.txt
8// - Markus Kuhn -- 2007-05-26 (Unicode 5.0)
9// http://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c
10// Thanks you!
11
13
14#include <array> // for array
15#include <cstddef> // for size_t
16#include <cstdint> // for uint32_t, uint8_t, uint16_t, int32_t
17#include <string> // for string, basic_string, wstring
18#include <string_view> // for string_view
19#include <tuple> // for _Swallow_assign, ignore
20#include <vector>
21
22#include "ftxui/screen/deprecated.hpp" // for wchar_width, wstring_width
23#include "ftxui/screen/string_internal.hpp" // for WordBreakProperty, EatCodePoint, CodepointToWordBreakProperty, GlyphCount, GlyphIterate, GlyphNext, GlyphPrevious, IsCombining, IsControl, IsFullWidth, Utf8ToWordBreakProperty
24
25namespace {
26
27struct Interval {
28 uint32_t first;
29 uint32_t last;
30};
31
32using WBP = ftxui::WordBreakProperty;
33struct WordBreakPropertyInterval {
34 uint32_t first;
35 uint32_t last;
36 WBP property;
37};
38
39// g_full_width_characters y g_word_break_intervals, generados a partir de la
40// Base de Datos de Caracteres Unicode por tools/gen_unicode_tables.py.
42
43// Construir tabla de solo los intervalos de caracteres WBP::Extend
44constexpr auto g_extend_characters{[]() constexpr {
45 // Calcular número de intervalos de caracteres extend
46 constexpr size_t size = []() constexpr {
47 size_t count = 0;
48 for (auto interval : g_word_break_intervals) {
49 if (interval.property == WBP::Extend) {
50 count++;
51 }
52 }
53 return count;
54 }();
55
56 // Crear array de intervalos de caracteres extend
57 std::array<Interval, size> result{};
58 size_t index = 0;
59 for (auto interval : g_word_break_intervals) {
60 if (interval.property == WBP::Extend) {
61 result[index++] = {interval.first, interval.last}; // NOLINT
62 }
63 }
64 return result;
65}()};
66
67// Encontrar un punto de código dentro de una lista ordenada de Interval.
68template <size_t N>
69bool Bisearch(uint32_t ucs, const std::array<Interval, N>& table) {
70 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
71 return false;
72 }
73
74 int min = 0;
75 int max = N - 1;
76 while (max >= min) {
77 const int mid = (min + max) / 2;
78 if (ucs > table[mid].last) { // NOLINT
79 min = mid + 1;
80 } else if (ucs < table[mid].first) { // NOLINT
81 max = mid - 1;
82 } else {
83 return true;
84 }
85 }
86
87 return false;
88}
89
90// Encontrar un valor dentro de una lista ordenada de Interval + propiedad.
91template <class C, size_t N>
92bool Bisearch(uint32_t ucs, const std::array<C, N>& table, C* out) {
93 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
94 return false;
95 }
96
97 int min = 0;
98 int max = N - 1;
99 while (max >= min) {
100 const int mid = (min + max) / 2;
101 if (ucs > table[mid].last) { // NOLINT
102 min = mid + 1;
103 } else if (ucs < table[mid].first) { // NOLINT
104 max = mid - 1;
105 } else {
106 *out = table[mid]; // NOLINT
107 return true;
108 }
109 }
110
111 return false;
112}
113
114int codepoint_width(uint32_t ucs) {
115 if (ftxui::IsControl(ucs)) {
116 return -1;
117 }
118
119 if (ftxui::IsCombining(ucs)) {
120 return 0;
121 }
122
123 if (ftxui::IsFullWidth(ucs)) {
124 return 2;
125 }
126
127 return 1;
128}
129
130} // namespace
131
132namespace ftxui {
133
134/// From UTF8 encoded string |input|, eat in between 1 and 4 byte representing
135/// one codepoint. Put the codepoint into |ucs|. Start at |start| and update
136/// |end| to represent the beginning of the next byte to eat for consecutive
137/// executions.
138// A partir de la cadena codificada en UTF8 |input|, lee entre 1 y 4 bytes que representan
139// un punto de código. Coloca el punto de código en |ucs|. Comienza en |start| y actualiza
140// |end| para representar el comienzo del siguiente byte a leer para ejecuciones
141// consecutivas.
142bool EatCodePoint(std::string_view input,
143 size_t start,
144 size_t* end,
145 uint32_t* ucs) {
146 if (start >= input.size()) {
147 *end = start + 1;
148 return false;
149 }
150 const uint8_t C0 = input[start];
151
152 // Cadena de 1 byte.
153 if ((C0 & 0b1000'0000) == 0b0000'0000) { // NOLINT
154 *ucs = C0 & 0b0111'1111; // NOLINT
155 *end = start + 1;
156 return true;
157 }
158
159 // Cadena de 2 bytes.
160 if ((C0 & 0b1110'0000) == 0b1100'0000 && // NOLINT
161 start + 1 < input.size()) {
162 const uint8_t C1 = input[start + 1];
163 *ucs = 0;
164 *ucs += C0 & 0b0001'1111; // NOLINT
165 *ucs <<= 6; // NOLINT
166 *ucs += C1 & 0b0011'1111; // NOLINT
167 *end = start + 2;
168 return true;
169 }
170
171 // Cadena de 3 bytes.
172 if ((C0 & 0b1111'0000) == 0b1110'0000 && // NOLINT
173 start + 2 < input.size()) {
174 const uint8_t C1 = input[start + 1];
175 const uint8_t C2 = input[start + 2];
176 *ucs = 0;
177 *ucs += C0 & 0b0000'1111; // NOLINT
178 *ucs <<= 6; // NOLINT
179 *ucs += C1 & 0b0011'1111; // NOLINT
180 *ucs <<= 6; // NOLINT
181 *ucs += C2 & 0b0011'1111; // NOLINT
182 *end = start + 3;
183 return true;
184 }
185
186 // Cadena de 4 bytes.
187 if ((C0 & 0b1111'1000) == 0b1111'0000 && // NOLINT
188 start + 3 < input.size()) {
189 const uint8_t C1 = input[start + 1];
190 const uint8_t C2 = input[start + 2];
191 const uint8_t C3 = input[start + 3];
192 *ucs = 0;
193 *ucs += C0 & 0b0000'0111; // NOLINT
194 *ucs <<= 6; // NOLINT
195 *ucs += C1 & 0b0011'1111; // NOLINT
196 *ucs <<= 6; // NOLINT
197 *ucs += C2 & 0b0011'1111; // NOLINT
198 *ucs <<= 6; // NOLINT
199 *ucs += C3 & 0b0011'1111; // NOLINT
200 *end = start + 4;
201 return true;
202 }
203
204 *end = start + 1;
205 return false;
206}
207
208/// From UTF16 encoded string |input|, eat in between 1 and 4 byte representing
209/// one codepoint. Put the codepoint into |ucs|. Start at |start| and update
210/// |end| to represent the beginning of the next byte to eat for consecutive
211/// executions.
212// A partir de la cadena codificada en UTF16 |input|, lee entre 1 y 4 bytes que representan
213// un punto de código. Coloca el punto de código en |ucs|. Comienza en |start| y actualiza
214// |end| para representar el comienzo del siguiente byte a leer para ejecuciones
215// consecutivas.
216bool EatCodePoint(std::wstring_view input,
217 size_t start,
218 size_t* end,
219 uint32_t* ucs) {
220 if (start >= input.size()) {
221 *end = start + 1;
222 return false;
223 }
224
225 // En linux wstring usa la codificación UTF32:
226 if constexpr (sizeof(wchar_t) == 4) {
227 *ucs = input[start]; // NOLINT
228 *end = start + 1;
229 return true;
230 }
231
232 // En windows, wstring usa la codificación UTF16:
233 int32_t C0 = input[start]; // NOLINT
234
235 // tamaño de palabra 1:
236 if (C0 < 0xd800 || C0 >= 0xdc00) { // NOLINT
237 *ucs = C0;
238 *end = start + 1;
239 return true;
240 }
241
242 // tamaño de palabra 2:
243 if (start + 1 >= input.size()) {
244 *end = start + 2;
245 return false;
246 }
247
248 int32_t C1 = input[start + 1]; // NOLINT
249 *ucs = ((C0 & 0x3ff) << 10) + (C1 & 0x3ff) + 0x10000; // NOLINT
250 *end = start + 2;
251 return true;
252}
253
254bool IsCombining(uint32_t ucs) {
255 return Bisearch(ucs, g_extend_characters);
256}
257
258bool IsFullWidth(uint32_t ucs) {
259 if (ucs < 0x0300) { // Ruta rápida: // NOLINT
260 return false;
261 }
262
263 return Bisearch(ucs, g_full_width_characters);
264}
265
266bool IsControl(uint32_t ucs) {
267 if (ucs == 0) {
268 return true;
269 }
270 if (ucs < 32) { // NOLINT
271 const uint32_t LINE_FEED = 10;
272 return ucs != LINE_FEED;
273 }
274 if (ucs >= 0x7f && ucs < 0xa0) { // NOLINT
275 return true;
276 }
277 return false;
278}
279
281 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
282 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
283 return interval.property;
284}
285
286int wchar_width(wchar_t ucs) {
287 return codepoint_width(uint32_t(ucs));
288}
289
290int wstring_width(const std::wstring& text) {
291 int width = 0;
292
293 for (const wchar_t& it : text) {
294 const int w = wchar_width(it);
295 if (w < 0) {
296 return -1;
297 }
298 width += w;
299 }
300 return width;
301}
302
303// Devuelve cuántas celdas ocupa la cadena codificada en UTF8 |input| al imprimirse.
304// Los caracteres de control no ocupan espacio, los caracteres combinantes
305// modifican el carácter anterior y no ocupan espacio, los caracteres de
306// ancho completo ocupan dos celdas y todos los demás caracteres ocupan una
307// celda.
308int string_width(std::string_view input) {
309 // Optimización de 1 byte: Esta función se llama a menudo con un único carácter
310 // ASCII, así que podemos optimizar este caso saltando la decodificación UTF8.
311 if (input.size() == 1) {
312 const char c = input[0];
313 if (c >= 32 && c < 127) { // NOLINT
314 return 1;
315 }
316 }
317
318 // Optimización ASCII: Si la cadena es ASCII puro, podemos saltar la decodificación
319 // UTF8 y simplemente contar el número de caracteres, ignorando los
320 // caracteres de control.
321 bool is_pure_ascii = true;
322 for (const char c : input) {
323 if (c < 31 || c >= 127) { // NOLINT
324 is_pure_ascii = false;
325 break;
326 }
327 }
328 if (is_pure_ascii) {
329 return static_cast<int>(input.size());
330 }
331
332 int width = 0;
333 size_t start = 0;
334 while (start < input.size()) {
335 uint32_t codepoint = 0;
336 if (!EatCodePoint(input, start, &start, &codepoint)) {
337 continue;
338 }
339
340 if (IsControl(codepoint)) {
341 continue;
342 }
343
344 if (IsCombining(codepoint)) {
345 continue;
346 }
347
348 if (IsFullWidth(codepoint)) {
349 width += 2;
350 continue;
351 }
352
353 width += 1;
354 }
355 return width;
356}
357
358std::vector<std::string> Utf8ToGlyphs(std::string_view input) {
359 std::vector<std::string> out;
360 out.reserve(input.size());
361 size_t start = 0;
362 size_t end = 0;
363 while (start < input.size()) {
364 uint32_t codepoint = 0;
365 if (!EatCodePoint(input, start, &end, &codepoint)) {
366 start = end;
367 continue;
368 }
369
370 const auto append = input.substr(start, end - start);
371 start = end;
372
373 // Ignora los caracteres de control.
374 if (IsControl(codepoint)) {
375 continue;
376 }
377
378 // Los caracteres combinatorios se colocan con el glifo anterior que están modificando.
379 if (IsCombining(codepoint)) {
380 if (!out.empty()) {
381 out.back() += append;
382 }
383 continue;
384 }
385
386 // Fullwidth characters take two cells. The second is made of the empty
387 // string to reserve the space the first is taking.
388 // Los caracteres de ancho completo ocupan dos celdas. La segunda se compone de una
389 // cadena vacía para reservar el espacio que ocupa la primera.
390 if (IsFullWidth(codepoint)) {
391 out.emplace_back(append);
392 out.emplace_back("");
393 continue;
394 }
395
396 // Caracteres normales:
397 out.emplace_back(append);
398 }
399 return out;
400}
401
402size_t GlyphPrevious(std::string_view input, size_t start) {
403 while (true) {
404 if (start == 0) {
405 return 0;
406 }
407 start--;
408
409 // Skip the UTF8 continuation bytes.
410 // Omite los bytes de continuación UTF8.
411 if ((input[start] & 0b1100'0000) == 0b1000'0000) {
412 continue;
413 }
414
415 uint32_t codepoint = 0;
416 size_t end = 0;
417 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
418
419 // Ignora los caracteres inválidos, de control y combinatorios.
420 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
421 continue;
422 }
423
424 return start;
425 }
426}
427
428size_t GlyphNext(std::string_view input, size_t start) {
429 bool glyph_found = false;
430 while (start < input.size()) {
431 size_t end = 0;
432 uint32_t codepoint = 0;
433 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
434
435 // Ignora los caracteres inválidos, de control y combinatorios.
436 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
437 start = end;
438 continue;
439 }
440
441 // We eat the beginning of the next glyph. If we are eating the one
442 // requested, return its start position immediately.
443 // Leemos el comienzo del siguiente glifo. Si estamos leyendo el glifo
444 // solicitado, devuelve su posición de inicio inmediatamente.
445 if (glyph_found) {
446 return static_cast<int>(start);
447 }
448
449 // Otherwise, skip this glyph and iterate:
450 // De lo contrario, omite este glifo e itera:
451 glyph_found = true;
452 start = end;
453 }
454 return static_cast<int>(input.size());
455}
456
457size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start) {
458 if (glyph_offset >= 0) {
459 for (int i = 0; i < glyph_offset; ++i) {
460 start = GlyphNext(input, start);
461 }
462 return start;
463 } else {
464 for (int i = 0; i < -glyph_offset; ++i) {
465 start = GlyphPrevious(input, start);
466 }
467 return start;
468 }
469}
470
471std::vector<int> CellToGlyphIndex(std::string_view input) {
472 int x = -1;
473 std::vector<int> out;
474 out.reserve(input.size());
475 size_t start = 0;
476 size_t end = 0;
477 while (start < input.size()) {
478 uint32_t codepoint = 0;
479 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
480 start = end;
481
482 // Ignorar caracteres inválidos / de control.
483 if (!eaten || IsControl(codepoint)) {
484 continue;
485 }
486
487 // Los caracteres combinatorios se colocan con el glifo anterior que están modificando.
488 if (IsCombining(codepoint)) {
489 if (x == -1) {
490 ++x;
491 out.push_back(x);
492 }
493 continue;
494 }
495
496 // Fullwidth characters take two cells. The second is made of the empty
497 // string to reserve the space the first is taking.
498 // Los caracteres de ancho completo ocupan dos celdas. La segunda se compone de una
499 // cadena vacía para reservar el espacio que ocupa la primera.
500 if (IsFullWidth(codepoint)) {
501 ++x;
502 out.push_back(x);
503 out.push_back(x);
504 continue;
505 }
506
507 // Caracteres normales:
508 ++x;
509 out.push_back(x);
510 }
511 return out;
512}
513
514int GlyphCount(std::string_view input) {
515 int size = 0;
516 size_t start = 0;
517 size_t end = 0;
518 while (start < input.size()) {
519 uint32_t codepoint = 0;
520 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
521 start = end;
522
523 // Ignorar caracteres inválidos:
524 if (!eaten || IsControl(codepoint)) {
525 continue;
526 }
527
528 // Ignore combining characters, except when they don\'t have a preceding to
529 // combine with.
530 // Ignora los caracteres combinatorios, excepto cuando no tienen un precedente con
531 // el que combinar.
532 if (IsCombining(codepoint)) {
533 if (size == 0) {
534 size++;
535 }
536 continue;
537 }
538
539 size++;
540 }
541 return size;
542}
543
544std::vector<WordBreakProperty> Utf8ToWordBreakProperty(std::string_view input) {
545 std::vector<WordBreakProperty> out;
546 out.reserve(input.size());
547 size_t start = 0;
548 size_t end = 0;
549 while (start < input.size()) {
550 uint32_t codepoint = 0;
551 if (!EatCodePoint(input, start, &end, &codepoint)) {
552 start = end;
553 continue;
554 }
555 start = end;
556
557 // Ignora los caracteres de control.
558 if (IsControl(codepoint)) {
559 continue;
560 }
561
562 // Ignorar caracteres combinantes.
563 if (IsCombining(codepoint)) {
564 continue;
565 }
566
567 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
568 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
569 out.push_back(interval.property);
570 }
571 return out;
572}
573
574/// Convert a std::wstring into a UTF8 std::string.
575/// Convierte un std::wstring en un std::string UTF8.
576std::string to_string(std::wstring_view s) {
577 std::string out;
578
579 size_t i = 0;
580 uint32_t codepoint = 0;
581 while (EatCodePoint(s, i, &i, &codepoint)) {
582 // Code point <-> UTF-8 conversion
583 // Conversión de punto de código <-> UTF-8
584 //
585 // ┏━━━━━━━━┳━━━━━━━━┳━━━━━━━━┳━━━━━━━━┓
586 // ┃Byte 1 ┃Byte 2 ┃Byte 3 ┃Byte 4 ┃
587 // ┡━━━━━━━━╇━━━━━━━━╇━━━━━━━━╇━━━━━━━━┩
588 // │0xxxxxxx│ │ │ │
589 // ├────────┼────────┼────────┼────────┤
590 // │110xxxxx│10xxxxxx│ │ │
591 // ├────────┼────────┼────────┼────────┤
592 // │1110xxxx│10xxxxxx│10xxxxxx│ │
593 // ├────────┼────────┼────────┼────────┤
594 // │11110xxx│10xxxxxx│10xxxxxx│10xxxxxx│
595 // └────────┴────────┴────────┴────────┘
596
597 // UTF8 de 1 byte
598 if (codepoint <= 0b000'0000'0111'1111) { // NOLINT
599 const uint8_t p1 = codepoint;
600 out.push_back(p1); // NOLINT
601 continue;
602 }
603
604 // UTF8 de 2 bytes
605 if (codepoint <= 0b000'0111'1111'1111) { // NOLINT
606 uint8_t p2 = codepoint & 0b111111; // NOLINT
607 codepoint >>= 6; // NOLINT
608 uint8_t p1 = codepoint; // NOLINT
609 out.push_back(0b11000000 + p1); // NOLINT
610 out.push_back(0b10000000 + p2); // NOLINT
611 continue;
612 }
613
614 // UTF8 de 3 bytes
615 if (codepoint <= 0b1111'1111'1111'1111) { // NOLINT
616 uint8_t p3 = codepoint & 0b111111; // NOLINT
617 codepoint >>= 6; // NOLINT
618 uint8_t p2 = codepoint & 0b111111; // NOLINT
619 codepoint >>= 6; // NOLINT
620 uint8_t p1 = codepoint; // NOLINT
621 out.push_back(0b11100000 + p1); // NOLINT
622 out.push_back(0b10000000 + p2); // NOLINT
623 out.push_back(0b10000000 + p3); // NOLINT
624 continue;
625 }
626
627 // UTF8 de 4 bytes
628 if (codepoint <= 0b1'0000'1111'1111'1111'1111) { // NOLINT
629 uint8_t p4 = codepoint & 0b111111; // NOLINT
630 codepoint >>= 6; // NOLINT
631 uint8_t p3 = codepoint & 0b111111; // NOLINT
632 codepoint >>= 6; // NOLINT
633 uint8_t p2 = codepoint & 0b111111; // NOLINT
634 codepoint >>= 6; // NOLINT
635 uint8_t p1 = codepoint; // NOLINT
636 out.push_back(0b11110000 + p1); // NOLINT
637 out.push_back(0b10000000 + p2); // NOLINT
638 out.push_back(0b10000000 + p3); // NOLINT
639 out.push_back(0b10000000 + p4); // NOLINT
640 continue;
641 }
642
643 // ¿Algo más?
644 }
645 return out;
646}
647
648/// Convert a UTF8 std::string into a std::wstring.
649/// Convierte un std::string UTF8 en un std::wstring.
650std::wstring to_wstring(std::string_view s) {
651 std::wstring out;
652
653 size_t i = 0;
654 uint32_t codepoint = 0;
655 while (EatCodePoint(s, i, &i, &codepoint)) {
656 // On linux wstring are UTF32 encoded:
657 // En Linux, wstring se codifican en UTF32:
658 if constexpr (sizeof(wchar_t) == 4) {
659 out.push_back(codepoint); // NOLINT
660 continue;
661 }
662
663 // On Windows, wstring are UTF16 encoded:
664 // En Windows, wstring se codifican en UTF16:
665
666 // Codepoint encoded using 1 word:
667 // Punto de código codificado usando 1 palabra:
668 // NOLINTNEXTLINE
669 if (codepoint < 0xD800 || (codepoint > 0xDFFF && codepoint < 0x10000)) {
670 uint16_t p0 = codepoint; // NOLINT
671 out.push_back(p0); // NOLINT
672 continue;
673 }
674
675 // Codepoint encoded using 2 words:
676 // Punto de código codificado usando 2 palabras:
677 codepoint -= 0x010000; // NOLINT
678 uint16_t p0 = (((codepoint << 12) >> 22) + 0xD800); // NOLINT
679 uint16_t p1 = (((codepoint << 22) >> 22) + 0xDC00); // NOLINT
680 out.push_back(p0); // NOLINT
681 out.push_back(p1); // NOLINT
682 }
683 return out;
684}
685
686} // namespace ftxui
Decorator size(WidthOrHeight direction, Constraint constraint, int value)
Aplica una restricción al tamaño de un elemento.
El espacio de nombres ftxui:: de FTXUI.
Definition animation.hpp:11
bool IsControl(uint32_t ucs)
Definition string.cpp:266
WordBreakProperty CodepointToWordBreakProperty(uint32_t codepoint)
Definition string.cpp:280
size_t GlyphPrevious(std::string_view input, size_t start)
Definition string.cpp:402
FTXUI_EXPORT(SCREEN) int string_width(std std::vector< std::string > Utf8ToGlyphs(std::string_view input)
Definition string.cpp:358
int string_width(std::string_view input)
Definition string.cpp:308
bool IsCombining(uint32_t ucs)
Definition string.cpp:254
int wchar_width(wchar_t ucs)
Definition string.cpp:286
bool EatCodePoint(std::string_view input, size_t start, size_t *end, uint32_t *ucs)
Definition string.cpp:142
std::string to_string(std::wstring_view s)
Definition string.cpp:576
int GlyphCount(std::string_view input)
Definition string.cpp:514
std::vector< WordBreakProperty > Utf8ToWordBreakProperty(std::string_view input)
Definition string.cpp:544
int wstring_width(const std::wstring &text)
Definition string.cpp:290
std::vector< int > CellToGlyphIndex(std::string_view input)
Definition string.cpp:471
FTXUI_EXPORT(SCREEN) std FTXUI_EXPORT(SCREEN) std std::wstring to_wstring(T s)
Definition string.hpp:18
size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start)
Definition string.cpp:457
bool IsFullWidth(uint32_t ucs)
Definition string.cpp:258
size_t GlyphNext(std::string_view input, size_t start)
Definition string.cpp:428
constexpr std::array< Interval, 123 > g_full_width_characters
constexpr std::array< WordBreakPropertyInterval, 1100 > g_word_break_intervals