FTXUI 7.0.3
C++ functional terminal UI.
Loading...
Searching...
No Matches
string.cpp
Go to the documentation of this file.
1// Copyright 2020 Arthur Sonzogni. All rights reserved.
2// Use of this source code is governed by the MIT license that can be found in
3// the LICENSE file.
4//
5// Content of this file was created thanks to:
6// -
7// https://www.unicode.org/Public/UCD/latest/ucd/auxiliary/WordBreakProperty.txt
8// - Markus Kuhn -- 2007-05-26 (Unicode 5.0)
9// http://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c
10// Thanks you!
11
13
14#include <array> // for array
15#include <cstddef> // for size_t
16#include <cstdint> // for uint32_t, uint8_t, uint16_t, int32_t
17#include <string> // for string, basic_string, wstring
18#include <string_view> // for string_view
19#include <tuple> // for _Swallow_assign, ignore
20#include <vector>
21
22#include "ftxui/screen/deprecated.hpp" // for wchar_width, wstring_width
23#include "ftxui/screen/string_internal.hpp" // for WordBreakProperty, EatCodePoint, CodepointToWordBreakProperty, GlyphCount, GlyphIterate, GlyphNext, GlyphPrevious, IsCombining, IsControl, IsFullWidth, Utf8ToWordBreakProperty
24
25namespace {
26
27struct Interval {
28 uint32_t first;
29 uint32_t last;
30};
31
32using WBP = ftxui::WordBreakProperty;
33struct WordBreakPropertyInterval {
34 uint32_t first;
35 uint32_t last;
36 WBP property;
37};
38
39// g_full_width_characters et g_word_break_intervals, générés à partir de la
40// base de données de caractères Unicode par tools/gen_unicode_tables.py.
42
43// Construit une table contenant uniquement les intervalles de caractères
44// WBP::Extend
45constexpr auto g_extend_characters{[]() constexpr {
46 // Calcule le nombre d'intervalles de caractères « extend »
47 constexpr size_t size = []() constexpr {
48 size_t count = 0;
49 for (auto interval : g_word_break_intervals) {
50 if (interval.property == WBP::Extend) {
51 count++;
52 }
53 }
54 return count;
55 }();
56
57 // Crée le tableau des intervalles de caractères « extend »
58 std::array<Interval, size> result{};
59 size_t index = 0;
60 for (auto interval : g_word_break_intervals) {
61 if (interval.property == WBP::Extend) {
62 result[index++] = {interval.first, interval.last}; // NOLINT
63 }
64 }
65 return result;
66}()};
67
68// Recherche un point de code dans une liste triée d'Interval.
69template <size_t N>
70bool Bisearch(uint32_t ucs, const std::array<Interval, N>& table) {
71 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
72 return false;
73 }
74
75 int min = 0;
76 int max = N - 1;
77 while (max >= min) {
78 const int mid = (min + max) / 2;
79 if (ucs > table[mid].last) { // NOLINT
80 min = mid + 1;
81 } else if (ucs < table[mid].first) { // NOLINT
82 max = mid - 1;
83 } else {
84 return true;
85 }
86 }
87
88 return false;
89}
90
91// Recherche une valeur dans une liste triée d'Interval + propriété.
92template <class C, size_t N>
93bool Bisearch(uint32_t ucs, const std::array<C, N>& table, C* out) {
94 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
95 return false;
96 }
97
98 int min = 0;
99 int max = N - 1;
100 while (max >= min) {
101 const int mid = (min + max) / 2;
102 if (ucs > table[mid].last) { // NOLINT
103 min = mid + 1;
104 } else if (ucs < table[mid].first) { // NOLINT
105 max = mid - 1;
106 } else {
107 *out = table[mid]; // NOLINT
108 return true;
109 }
110 }
111
112 return false;
113}
114
115int codepoint_width(uint32_t ucs) {
116 if (ftxui::IsControl(ucs)) {
117 return -1;
118 }
119
120 if (ftxui::IsCombining(ucs)) {
121 return 0;
122 }
123
124 if (ftxui::IsFullWidth(ucs)) {
125 return 2;
126 }
127
128 return 1;
129}
130
131} // namespace
132
133namespace ftxui {
134
135// À partir de la chaîne encodée en UTF8 |input|, consomme entre 1 et 4 octets
136// représentant un point de code. Place le point de code dans |ucs|. Démarre à
137// |start| et met à jour |end| pour représenter le début du prochain octet à
138// consommer lors d'exécutions consécutives.
139bool EatCodePoint(std::string_view input,
140 size_t start,
141 size_t* end,
142 uint32_t* ucs) {
143 if (start >= input.size()) {
144 *end = start + 1;
145 return false;
146 }
147 const uint8_t C0 = input[start];
148
149 // Chaîne de 1 octet.
150 if ((C0 & 0b1000'0000) == 0b0000'0000) { // NOLINT
151 *ucs = C0 & 0b0111'1111; // NOLINT
152 *end = start + 1;
153 return true;
154 }
155
156 // Chaîne de 2 octets.
157 if ((C0 & 0b1110'0000) == 0b1100'0000 && // NOLINT
158 start + 1 < input.size()) {
159 const uint8_t C1 = input[start + 1];
160 *ucs = 0;
161 *ucs += C0 & 0b0001'1111; // NOLINT
162 *ucs <<= 6; // NOLINT
163 *ucs += C1 & 0b0011'1111; // NOLINT
164 *end = start + 2;
165 return true;
166 }
167
168 // Chaîne de 3 octets.
169 if ((C0 & 0b1111'0000) == 0b1110'0000 && // NOLINT
170 start + 2 < input.size()) {
171 const uint8_t C1 = input[start + 1];
172 const uint8_t C2 = input[start + 2];
173 *ucs = 0;
174 *ucs += C0 & 0b0000'1111; // NOLINT
175 *ucs <<= 6; // NOLINT
176 *ucs += C1 & 0b0011'1111; // NOLINT
177 *ucs <<= 6; // NOLINT
178 *ucs += C2 & 0b0011'1111; // NOLINT
179 *end = start + 3;
180 return true;
181 }
182
183 // Chaîne de 4 octets.
184 if ((C0 & 0b1111'1000) == 0b1111'0000 && // NOLINT
185 start + 3 < input.size()) {
186 const uint8_t C1 = input[start + 1];
187 const uint8_t C2 = input[start + 2];
188 const uint8_t C3 = input[start + 3];
189 *ucs = 0;
190 *ucs += C0 & 0b0000'0111; // NOLINT
191 *ucs <<= 6; // NOLINT
192 *ucs += C1 & 0b0011'1111; // NOLINT
193 *ucs <<= 6; // NOLINT
194 *ucs += C2 & 0b0011'1111; // NOLINT
195 *ucs <<= 6; // NOLINT
196 *ucs += C3 & 0b0011'1111; // NOLINT
197 *end = start + 4;
198 return true;
199 }
200
201 *end = start + 1;
202 return false;
203}
204
205// À partir de la chaîne encodée en UTF16 |input|, consomme entre 1 et 4
206// octets représentant un point de code. Place le point de code dans |ucs|.
207// Démarre à |start| et met à jour |end| pour représenter le début du
208// prochain octet à consommer lors d'exécutions consécutives.
209bool EatCodePoint(std::wstring_view input,
210 size_t start,
211 size_t* end,
212 uint32_t* ucs) {
213 if (start >= input.size()) {
214 *end = start + 1;
215 return false;
216 }
217
218 // Sous Linux, wstring utilise l'encodage UTF32 :
219 if constexpr (sizeof(wchar_t) == 4) {
220 *ucs = input[start]; // NOLINT
221 *end = start + 1;
222 return true;
223 }
224
225 // Sous Windows, wstring utilise l'encodage UTF16 :
226 int32_t C0 = input[start]; // NOLINT
227
228 // Taille de 1 mot :
229 if (C0 < 0xd800 || C0 >= 0xdc00) { // NOLINT
230 *ucs = C0;
231 *end = start + 1;
232 return true;
233 }
234
235 // Taille de 2 mots :
236 if (start + 1 >= input.size()) {
237 *end = start + 2;
238 return false;
239 }
240
241 int32_t C1 = input[start + 1]; // NOLINT
242 *ucs = ((C0 & 0x3ff) << 10) + (C1 & 0x3ff) + 0x10000; // NOLINT
243 *end = start + 2;
244 return true;
245}
246
247bool IsCombining(uint32_t ucs) {
248 return Bisearch(ucs, g_extend_characters);
249}
250
251bool IsFullWidth(uint32_t ucs) {
252 if (ucs < 0x0300) { // Chemin rapide : // NOLINT
253 return false;
254 }
255
256 return Bisearch(ucs, g_full_width_characters);
257}
258
259bool IsControl(uint32_t ucs) {
260 if (ucs == 0) {
261 return true;
262 }
263 if (ucs < 32) { // NOLINT
264 const uint32_t LINE_FEED = 10;
265 return ucs != LINE_FEED;
266 }
267 if (ucs >= 0x7f && ucs < 0xa0) { // NOLINT
268 return true;
269 }
270 return false;
271}
272
274 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
275 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
276 return interval.property;
277}
278
279int wchar_width(wchar_t ucs) {
280 return codepoint_width(uint32_t(ucs));
281}
282
283int wstring_width(const std::wstring& text) {
284 int width = 0;
285
286 for (const wchar_t& it : text) {
287 const int w = wchar_width(it);
288 if (w < 0) {
289 return -1;
290 }
291 width += w;
292 }
293 return width;
294}
295
296// Retourne le nombre de cellules occupées par la chaîne encodée en UTF8
297// |input| lorsqu'elle est affichée. Les caractères de contrôle n'occupent
298// aucun espace, les caractères combinants modifient le caractère précédent et
299// n'occupent aucun espace, les caractères pleine largeur occupent deux
300// cellules et tous les autres caractères occupent une cellule.
301int string_width(std::string_view input) {
302 // Optimisation 1 octet : cette fonction est souvent appelée sur un seul
303 // caractère ASCII, on peut donc optimiser ce cas en sautant le décodage
304 // UTF8.
305 if (input.size() == 1) {
306 const char c = input[0];
307 if (c >= 32 && c < 127) { // NOLINT
308 return 1;
309 }
310 }
311
312 // Optimisation ASCII : si la chaîne est purement ASCII, on peut sauter le
313 // décodage UTF8 et simplement compter le nombre de caractères, en ignorant
314 // les caractères de contrôle.
315 bool is_pure_ascii = true;
316 for (const char c : input) {
317 if (c < 31 || c >= 127) { // NOLINT
318 is_pure_ascii = false;
319 break;
320 }
321 }
322 if (is_pure_ascii) {
323 return static_cast<int>(input.size());
324 }
325
326 int width = 0;
327 size_t start = 0;
328 while (start < input.size()) {
329 uint32_t codepoint = 0;
330 if (!EatCodePoint(input, start, &start, &codepoint)) {
331 continue;
332 }
333
334 if (IsControl(codepoint)) {
335 continue;
336 }
337
338 if (IsCombining(codepoint)) {
339 continue;
340 }
341
342 if (IsFullWidth(codepoint)) {
343 width += 2;
344 continue;
345 }
346
347 width += 1;
348 }
349 return width;
350}
351
352std::vector<std::string> Utf8ToGlyphs(std::string_view input) {
353 std::vector<std::string> out;
354 out.reserve(input.size());
355 size_t start = 0;
356 size_t end = 0;
357 while (start < input.size()) {
358 uint32_t codepoint = 0;
359 if (!EatCodePoint(input, start, &end, &codepoint)) {
360 start = end;
361 continue;
362 }
363
364 const auto append = input.substr(start, end - start);
365 start = end;
366
367 // Ignore les caractères de contrôle.
368 if (IsControl(codepoint)) {
369 continue;
370 }
371
372 // Les caractères combinants sont ajoutés au glyphe précédent qu'ils
373 // modifient.
374 if (IsCombining(codepoint)) {
375 if (!out.empty()) {
376 out.back() += append;
377 }
378 continue;
379 }
380
381 // Les caractères pleine largeur occupent deux cellules. La seconde est
382 // constituée d'une chaîne vide afin de réserver l'espace occupé par la
383 // première.
384 if (IsFullWidth(codepoint)) {
385 out.emplace_back(append);
386 out.emplace_back("");
387 continue;
388 }
389
390 // Caractères normaux :
391 out.emplace_back(append);
392 }
393 return out;
394}
395
396size_t GlyphPrevious(std::string_view input, size_t start) {
397 while (true) {
398 if (start == 0) {
399 return 0;
400 }
401 start--;
402
403 // Saute les octets de continuation UTF8.
404 if ((input[start] & 0b1100'0000) == 0b1000'0000) {
405 continue;
406 }
407
408 uint32_t codepoint = 0;
409 size_t end = 0;
410 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
411
412 // Ignore les caractères invalides, de contrôle et combinants.
413 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
414 continue;
415 }
416
417 return start;
418 }
419}
420
421size_t GlyphNext(std::string_view input, size_t start) {
422 bool glyph_found = false;
423 while (start < input.size()) {
424 size_t end = 0;
425 uint32_t codepoint = 0;
426 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
427
428 // Ignore les caractères invalides, de contrôle et combinants.
429 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
430 start = end;
431 continue;
432 }
433
434 // On consomme le début du glyphe suivant. Si c'est celui demandé, on
435 // retourne immédiatement sa position de départ.
436 if (glyph_found) {
437 return static_cast<int>(start);
438 }
439
440 // Sinon, on saute ce glyphe et on itère :
441 glyph_found = true;
442 start = end;
443 }
444 return static_cast<int>(input.size());
445}
446
447size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start) {
448 if (glyph_offset >= 0) {
449 for (int i = 0; i < glyph_offset; ++i) {
450 start = GlyphNext(input, start);
451 }
452 return start;
453 } else {
454 for (int i = 0; i < -glyph_offset; ++i) {
455 start = GlyphPrevious(input, start);
456 }
457 return start;
458 }
459}
460
461std::vector<int> CellToGlyphIndex(std::string_view input) {
462 int x = -1;
463 std::vector<int> out;
464 out.reserve(input.size());
465 size_t start = 0;
466 size_t end = 0;
467 while (start < input.size()) {
468 uint32_t codepoint = 0;
469 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
470 start = end;
471
472 // Ignore les caractères invalides / de contrôle.
473 if (!eaten || IsControl(codepoint)) {
474 continue;
475 }
476
477 // Les caractères combinants sont ajoutés au glyphe précédent qu'ils
478 // modifient.
479 if (IsCombining(codepoint)) {
480 if (x == -1) {
481 ++x;
482 out.push_back(x);
483 }
484 continue;
485 }
486
487 // Les caractères pleine largeur occupent deux cellules. La seconde est
488 // constituée d'une chaîne vide afin de réserver l'espace occupé par la
489 // première.
490 if (IsFullWidth(codepoint)) {
491 ++x;
492 out.push_back(x);
493 out.push_back(x);
494 continue;
495 }
496
497 // Caractères normaux :
498 ++x;
499 out.push_back(x);
500 }
501 return out;
502}
503
504int GlyphCount(std::string_view input) {
505 int size = 0;
506 size_t start = 0;
507 size_t end = 0;
508 while (start < input.size()) {
509 uint32_t codepoint = 0;
510 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
511 start = end;
512
513 // Ignore les caractères invalides :
514 if (!eaten || IsControl(codepoint)) {
515 continue;
516 }
517
518 // Ignore les caractères combinants, sauf s'ils n'ont pas de prédécesseur
519 // avec lequel se combiner.
520 if (IsCombining(codepoint)) {
521 if (size == 0) {
522 size++;
523 }
524 continue;
525 }
526
527 size++;
528 }
529 return size;
530}
531
532std::vector<WordBreakProperty> Utf8ToWordBreakProperty(std::string_view input) {
533 std::vector<WordBreakProperty> out;
534 out.reserve(input.size());
535 size_t start = 0;
536 size_t end = 0;
537 while (start < input.size()) {
538 uint32_t codepoint = 0;
539 if (!EatCodePoint(input, start, &end, &codepoint)) {
540 start = end;
541 continue;
542 }
543 start = end;
544
545 // Ignore les caractères de contrôle.
546 if (IsControl(codepoint)) {
547 continue;
548 }
549
550 // Ignore les caractères combinants.
551 if (IsCombining(codepoint)) {
552 continue;
553 }
554
555 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
556 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
557 out.push_back(interval.property);
558 }
559 return out;
560}
561
562/// Convertit un std::wstring en std::string UTF8.
563std::string to_string(std::wstring_view s) {
564 std::string out;
565
566 size_t i = 0;
567 uint32_t codepoint = 0;
568 while (EatCodePoint(s, i, &i, &codepoint)) {
569 // Conversion point de code <-> UTF-8
570 //
571 // ┏━━━━━━━━┳━━━━━━━━┳━━━━━━━━┳━━━━━━━━┓
572 // ┃Octet 1 ┃Octet 2 ┃Octet 3 ┃Octet 4 ┃
573 // ┡━━━━━━━━╇━━━━━━━━╇━━━━━━━━╇━━━━━━━━┩
574 // │0xxxxxxx│ │ │ │
575 // ├────────┼────────┼────────┼────────┤
576 // │110xxxxx│10xxxxxx│ │ │
577 // ├────────┼────────┼────────┼────────┤
578 // │1110xxxx│10xxxxxx│10xxxxxx│ │
579 // ├────────┼────────┼────────┼────────┤
580 // │11110xxx│10xxxxxx│10xxxxxx│10xxxxxx│
581 // └────────┴────────┴────────┴────────┘
582
583 // UTF8 sur 1 octet
584 if (codepoint <= 0b000'0000'0111'1111) { // NOLINT
585 const uint8_t p1 = codepoint;
586 out.push_back(p1); // NOLINT
587 continue;
588 }
589
590 // UTF8 sur 2 octets
591 if (codepoint <= 0b000'0111'1111'1111) { // NOLINT
592 uint8_t p2 = codepoint & 0b111111; // NOLINT
593 codepoint >>= 6; // NOLINT
594 uint8_t p1 = codepoint; // NOLINT
595 out.push_back(0b11000000 + p1); // NOLINT
596 out.push_back(0b10000000 + p2); // NOLINT
597 continue;
598 }
599
600 // UTF8 sur 3 octets
601 if (codepoint <= 0b1111'1111'1111'1111) { // NOLINT
602 uint8_t p3 = codepoint & 0b111111; // NOLINT
603 codepoint >>= 6; // NOLINT
604 uint8_t p2 = codepoint & 0b111111; // NOLINT
605 codepoint >>= 6; // NOLINT
606 uint8_t p1 = codepoint; // NOLINT
607 out.push_back(0b11100000 + p1); // NOLINT
608 out.push_back(0b10000000 + p2); // NOLINT
609 out.push_back(0b10000000 + p3); // NOLINT
610 continue;
611 }
612
613 // UTF8 sur 4 octets
614 if (codepoint <= 0b1'0000'1111'1111'1111'1111) { // NOLINT
615 uint8_t p4 = codepoint & 0b111111; // NOLINT
616 codepoint >>= 6; // NOLINT
617 uint8_t p3 = codepoint & 0b111111; // NOLINT
618 codepoint >>= 6; // NOLINT
619 uint8_t p2 = codepoint & 0b111111; // NOLINT
620 codepoint >>= 6; // NOLINT
621 uint8_t p1 = codepoint; // NOLINT
622 out.push_back(0b11110000 + p1); // NOLINT
623 out.push_back(0b10000000 + p2); // NOLINT
624 out.push_back(0b10000000 + p3); // NOLINT
625 out.push_back(0b10000000 + p4); // NOLINT
626 continue;
627 }
628
629 // Autre chose ?
630 }
631 return out;
632}
633
634/// Convertit un std::string UTF8 en std::wstring.
635std::wstring to_wstring(std::string_view s) {
636 std::wstring out;
637
638 size_t i = 0;
639 uint32_t codepoint = 0;
640 while (EatCodePoint(s, i, &i, &codepoint)) {
641 // Sous Linux, wstring est encodé en UTF32 :
642 if constexpr (sizeof(wchar_t) == 4) {
643 out.push_back(codepoint); // NOLINT
644 continue;
645 }
646
647 // Sous Windows, wstring est encodé en UTF16 :
648
649 // Point de code encodé sur 1 mot :
650 // NOLINTNEXTLINE
651 if (codepoint < 0xD800 || (codepoint > 0xDFFF && codepoint < 0x10000)) {
652 uint16_t p0 = codepoint; // NOLINT
653 out.push_back(p0); // NOLINT
654 continue;
655 }
656
657 // Point de code encodé sur 2 mots :
658 codepoint -= 0x010000; // NOLINT
659 uint16_t p0 = (((codepoint << 12) >> 22) + 0xD800); // NOLINT
660 uint16_t p1 = (((codepoint << 22) >> 22) + 0xDC00); // NOLINT
661 out.push_back(p0); // NOLINT
662 out.push_back(p1); // NOLINT
663 }
664 return out;
665}
666
667} // namespace ftxui
Decorator size(WidthOrHeight direction, Constraint constraint, int value)
Applique une contrainte sur la taille d'un élément.
L'espace de noms FTXUI ftxui::
Definition animation.hpp:11
bool IsControl(uint32_t ucs)
Definition string.cpp:259
WordBreakProperty CodepointToWordBreakProperty(uint32_t codepoint)
Definition string.cpp:273
size_t GlyphPrevious(std::string_view input, size_t start)
Definition string.cpp:396
FTXUI_EXPORT(SCREEN) int string_width(std std::vector< std::string > Utf8ToGlyphs(std::string_view input)
Definition string.cpp:352
int string_width(std::string_view input)
Definition string.cpp:301
bool IsCombining(uint32_t ucs)
Definition string.cpp:247
int wchar_width(wchar_t ucs)
Definition string.cpp:279
bool EatCodePoint(std::string_view input, size_t start, size_t *end, uint32_t *ucs)
Definition string.cpp:139
std::string to_string(std::wstring_view s)
Convertit un std::wstring en std::string UTF8.
Definition string.cpp:563
int GlyphCount(std::string_view input)
Definition string.cpp:504
std::vector< WordBreakProperty > Utf8ToWordBreakProperty(std::string_view input)
Definition string.cpp:532
int wstring_width(const std::wstring &text)
Definition string.cpp:283
std::vector< int > CellToGlyphIndex(std::string_view input)
Definition string.cpp:461
FTXUI_EXPORT(SCREEN) std FTXUI_EXPORT(SCREEN) std std::wstring to_wstring(T s)
Definition string.hpp:18
size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start)
Definition string.cpp:447
bool IsFullWidth(uint32_t ucs)
Definition string.cpp:251
size_t GlyphNext(std::string_view input, size_t start)
Definition string.cpp:421
constexpr std::array< Interval, 123 > g_full_width_characters
constexpr std::array< WordBreakPropertyInterval, 1100 > g_word_break_intervals