FTXUI 7.0.3
C++ functional terminal UI.
Loading...
Searching...
No Matches
string.cpp
Go to the documentation of this file.
1// Copyright 2020 Arthur Sonzogni. All rights reserved.
2// Use of this source code is governed by the MIT license that can be found in
3// the LICENSE file.
4//
5// Content of this file was created thanks to:
6// -
7// https://www.unicode.org/Public/UCD/latest/ucd/auxiliary/WordBreakProperty.txt
8// - Markus Kuhn -- 2007-05-26 (Unicode 5.0)
9// http://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c
10// Thanks you!
11
13
14#include <array> // for array
15#include <cstddef> // for size_t
16#include <cstdint> // for uint32_t, uint8_t, uint16_t, int32_t
17#include <string> // for string, basic_string, wstring
18#include <string_view> // for string_view
19#include <tuple> // for _Swallow_assign, ignore
20#include <vector>
21
22#include "ftxui/screen/deprecated.hpp" // for wchar_width, wstring_width
23#include "ftxui/screen/string_internal.hpp" // for WordBreakProperty, EatCodePoint, CodepointToWordBreakProperty, GlyphCount, GlyphIterate, GlyphNext, GlyphPrevious, IsCombining, IsControl, IsFullWidth, Utf8ToWordBreakProperty
24
25namespace {
26
27struct Interval {
28 uint32_t first;
29 uint32_t last;
30};
31
32using WBP = ftxui::WordBreakProperty;
33struct WordBreakPropertyInterval {
34 uint32_t first;
35 uint32_t last;
36 WBP property;
37};
38
39// g_full_width_characters 及 g_word_break_intervals,由
40// tools/gen_unicode_tables.py 從 Unicode Character Database 產生。
42
43// 建構僅包含 WBP::Extend 字元區間的表格
44constexpr auto g_extend_characters{[]() constexpr {
45 // 計算延伸字元區間的數量
46 constexpr size_t size = []() constexpr {
47 size_t count = 0;
48 for (auto interval : g_word_break_intervals) {
49 if (interval.property == WBP::Extend) {
50 count++;
51 }
52 }
53 return count;
54 }();
55
56 // 建立延伸字元區間的陣列
57 std::array<Interval, size> result{};
58 size_t index = 0;
59 for (auto interval : g_word_break_intervals) {
60 if (interval.property == WBP::Extend) {
61 result[index++] = {interval.first, interval.last}; // NOLINT
62 }
63 }
64 return result;
65}()};
66
67// 在 Interval 的排序列表中查找一個碼點。
68template <size_t N>
69bool Bisearch(uint32_t ucs, const std::array<Interval, N>& table) {
70 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
71 return false;
72 }
73
74 int min = 0;
75 int max = N - 1;
76 while (max >= min) {
77 const int mid = (min + max) / 2;
78 if (ucs > table[mid].last) { // NOLINT
79 min = mid + 1;
80 } else if (ucs < table[mid].first) { // NOLINT
81 max = mid - 1;
82 } else {
83 return true;
84 }
85 }
86
87 return false;
88}
89
90// 在 Interval + 屬性的排序列表中查找一個值。
91template <class C, size_t N>
92bool Bisearch(uint32_t ucs, const std::array<C, N>& table, C* out) {
93 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
94 return false;
95 }
96
97 int min = 0;
98 int max = N - 1;
99 while (max >= min) {
100 const int mid = (min + max) / 2;
101 if (ucs > table[mid].last) { // NOLINT
102 min = mid + 1;
103 } else if (ucs < table[mid].first) { // NOLINT
104 max = mid - 1;
105 } else {
106 *out = table[mid]; // NOLINT
107 return true;
108 }
109 }
110
111 return false;
112}
113
114int codepoint_width(uint32_t ucs) {
115 if (ftxui::IsControl(ucs)) {
116 return -1;
117 }
118
119 if (ftxui::IsCombining(ucs)) {
120 return 0;
121 }
122
123 if (ftxui::IsFullWidth(ucs)) {
124 return 2;
125 }
126
127 return 1;
128}
129
130} // namespace
131
132namespace ftxui {
133
134// 從 UTF8 編碼的字串 |input| 中,讀取介於 1 到 4 個位元組,表示一個碼點。
135// 將碼點放入 |ucs| 中。從 |start| 開始,並更新 |end| 以表示下一個要讀取的位元組的開頭,
136// 以便連續執行。
137bool EatCodePoint(std::string_view input,
138 size_t start,
139 size_t* end,
140 uint32_t* ucs) {
141 if (start >= input.size()) {
142 *end = start + 1;
143 return false;
144 }
145 const uint8_t C0 = input[start];
146
147 // 1 位元組字串。
148 if ((C0 & 0b1000'0000) == 0b0000'0000) { // NOLINT
149 *ucs = C0 & 0b0111'1111; // NOLINT
150 *end = start + 1;
151 return true;
152 }
153
154 // 2 位元組字串。
155 if ((C0 & 0b1110'0000) == 0b1100'0000 && // NOLINT
156 start + 1 < input.size()) {
157 const uint8_t C1 = input[start + 1];
158 *ucs = 0;
159 *ucs += C0 & 0b0001'1111; // NOLINT
160 *ucs <<= 6; // NOLINT
161 *ucs += C1 & 0b0011'1111; // NOLINT
162 *end = start + 2;
163 return true;
164 }
165
166 // 3 位元組字串。
167 if ((C0 & 0b1111'0000) == 0b1110'0000 && // NOLINT
168 start + 2 < input.size()) {
169 const uint8_t C1 = input[start + 1];
170 const uint8_t C2 = input[start + 2];
171 *ucs = 0;
172 *ucs += C0 & 0b0000'1111; // NOLINT
173 *ucs <<= 6; // NOLINT
174 *ucs += C1 & 0b0011'1111; // NOLINT
175 *ucs <<= 6; // NOLINT
176 *ucs += C2 & 0b0011'1111; // NOLINT
177 *end = start + 3;
178 return true;
179 }
180
181 // 4 位元組字串。
182 if ((C0 & 0b1111'1000) == 0b1111'0000 && // NOLINT
183 start + 3 < input.size()) {
184 const uint8_t C1 = input[start + 1];
185 const uint8_t C2 = input[start + 2];
186 const uint8_t C3 = input[start + 3];
187 *ucs = 0;
188 *ucs += C0 & 0b0000'0111; // NOLINT
189 *ucs <<= 6; // NOLINT
190 *ucs += C1 & 0b0011'1111; // NOLINT
191 *ucs <<= 6; // NOLINT
192 *ucs += C2 & 0b0011'1111; // NOLINT
193 *ucs <<= 6; // NOLINT
194 *ucs += C3 & 0b0011'1111; // NOLINT
195 *end = start + 4;
196 return true;
197 }
198
199 *end = start + 1;
200 return false;
201}
202
203// 從 UTF16 編碼的字串 |input| 中,讀取介於 1 到 4 個位元組,表示一個碼點。
204// 將碼點放入 |ucs| 中。從 |start| 開始,並更新 |end| 以表示下一個要讀取的位元組的開頭,
205// 以便連續執行。
206bool EatCodePoint(std::wstring_view input,
207 size_t start,
208 size_t* end,
209 uint32_t* ucs) {
210 if (start >= input.size()) {
211 *end = start + 1;
212 return false;
213 }
214
215 // 在 linux 上,wstring 使用 UTF32 編碼:
216 if constexpr (sizeof(wchar_t) == 4) {
217 *ucs = input[start]; // NOLINT
218 *end = start + 1;
219 return true;
220 }
221
222 // 在 windows 上,wstring 使用 UTF16 編碼:
223 int32_t C0 = input[start]; // NOLINT
224
225 // 1 word 大小:
226 if (C0 < 0xd800 || C0 >= 0xdc00) { // NOLINT
227 *ucs = C0;
228 *end = start + 1;
229 return true;
230 }
231
232 // 2 word 大小:
233 if (start + 1 >= input.size()) {
234 *end = start + 2;
235 return false;
236 }
237
238 int32_t C1 = input[start + 1]; // NOLINT
239 *ucs = ((C0 & 0x3ff) << 10) + (C1 & 0x3ff) + 0x10000; // NOLINT
240 *end = start + 2;
241 return true;
242}
243
244bool IsCombining(uint32_t ucs) {
245 return Bisearch(ucs, g_extend_characters);
246}
247
248bool IsFullWidth(uint32_t ucs) {
249 if (ucs < 0x0300) { // 快速路徑: // NOLINT
250 return false;
251 }
252
253 return Bisearch(ucs, g_full_width_characters);
254}
255
256bool IsControl(uint32_t ucs) {
257 if (ucs == 0) {
258 return true;
259 }
260 if (ucs < 32) { // NOLINT
261 const uint32_t LINE_FEED = 10;
262 return ucs != LINE_FEED;
263 }
264 if (ucs >= 0x7f && ucs < 0xa0) { // NOLINT
265 return true;
266 }
267 return false;
268}
269
271 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
272 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
273 return interval.property;
274}
275
276int wchar_width(wchar_t ucs) {
277 return codepoint_width(uint32_t(ucs));
278}
279
280int wstring_width(const std::wstring& text) {
281 int width = 0;
282
283 for (const wchar_t& it : text) {
284 const int w = wchar_width(it);
285 if (w < 0) {
286 return -1;
287 }
288 width += w;
289 }
290 return width;
291}
292
293// 傳回 UTF8 編碼字串 |input| 在印出時佔用的 cell 數量。
294// 控制字元不佔用任何空間,組合字元
295// 會修改前一個字元且不佔用任何空間,全形
296// 字元佔用兩個 cell,而其他所有字元則佔用一個
297// cell。
298int string_width(std::string_view input) {
299 // 1 位元組最佳化:此函式通常在單一 ASCII
300 // 字元上被呼叫,因此我們可以透過跳過 UTF8 解碼來最佳化這種情況。
301 if (input.size() == 1) {
302 const char c = input[0];
303 if (c >= 32 && c < 127) { // NOLINT
304 return 1;
305 }
306 }
307
308 // ASCII 最佳化:如果字串為純 ASCII,我們可以跳過 UTF8
309 // 解碼,只計算字元的數量,並忽略控制
310 // 字元。
311 bool is_pure_ascii = true;
312 for (const char c : input) {
313 if (c < 31 || c >= 127) { // NOLINT
314 is_pure_ascii = false;
315 break;
316 }
317 }
318 if (is_pure_ascii) {
319 return static_cast<int>(input.size());
320 }
321
322 int width = 0;
323 size_t start = 0;
324 while (start < input.size()) {
325 uint32_t codepoint = 0;
326 if (!EatCodePoint(input, start, &start, &codepoint)) {
327 continue;
328 }
329
330 if (IsControl(codepoint)) {
331 continue;
332 }
333
334 if (IsCombining(codepoint)) {
335 continue;
336 }
337
338 if (IsFullWidth(codepoint)) {
339 width += 2;
340 continue;
341 }
342
343 width += 1;
344 }
345 return width;
346}
347
348std::vector<std::string> Utf8ToGlyphs(std::string_view input) {
349 std::vector<std::string> out;
350 out.reserve(input.size());
351 size_t start = 0;
352 size_t end = 0;
353 while (start < input.size()) {
354 uint32_t codepoint = 0;
355 if (!EatCodePoint(input, start, &end, &codepoint)) {
356 start = end;
357 continue;
358 }
359
360 const auto append = input.substr(start, end - start);
361 start = end;
362
363 // 忽略控制字元。
364 if (IsControl(codepoint)) {
365 continue;
366 }
367
368 // 組合字元會與它們正在修改的前一個字符放在一起。
369 if (IsCombining(codepoint)) {
370 if (!out.empty()) {
371 out.back() += append;
372 }
373 continue;
374 }
375
376 // Fullwidth characters take two cells. The second is made of the empty
377 // 全形字元佔用兩個單元格。第二個是空字串,用於保留第一個佔用的空間。
378 if (IsFullWidth(codepoint)) {
379 out.emplace_back(append);
380 out.emplace_back("");
381 continue;
382 }
383
384 // 一般字元:
385 out.emplace_back(append);
386 }
387 return out;
388}
389
390size_t GlyphPrevious(std::string_view input, size_t start) {
391 while (true) {
392 if (start == 0) {
393 return 0;
394 }
395 start--;
396
397 // 跳過 UTF8 延續位元組。
398 if ((input[start] & 0b1100'0000) == 0b1000'0000) {
399 continue;
400 }
401
402 uint32_t codepoint = 0;
403 size_t end = 0;
404 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
405
406 // 忽略無效、控制字元及組合字元。
407 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
408 continue;
409 }
410
411 return start;
412 }
413}
414
415size_t GlyphNext(std::string_view input, size_t start) {
416 bool glyph_found = false;
417 while (start < input.size()) {
418 size_t end = 0;
419 uint32_t codepoint = 0;
420 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
421
422 // 忽略無效、控制字元及組合字元。
423 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
424 start = end;
425 continue;
426 }
427
428 // 我們吃掉下一個字符的開頭。如果我們正在吃掉的正是
429 // 請求的那一個,立即回傳它的起始位置。
430 if (glyph_found) {
431 return static_cast<int>(start);
432 }
433
434 // 否則,跳過這個字符並繼續迭代:
435 glyph_found = true;
436 start = end;
437 }
438 return static_cast<int>(input.size());
439}
440
441size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start) {
442 if (glyph_offset >= 0) {
443 for (int i = 0; i < glyph_offset; ++i) {
444 start = GlyphNext(input, start);
445 }
446 return start;
447 } else {
448 for (int i = 0; i < -glyph_offset; ++i) {
449 start = GlyphPrevious(input, start);
450 }
451 return start;
452 }
453}
454
455std::vector<int> CellToGlyphIndex(std::string_view input) {
456 int x = -1;
457 std::vector<int> out;
458 out.reserve(input.size());
459 size_t start = 0;
460 size_t end = 0;
461 while (start < input.size()) {
462 uint32_t codepoint = 0;
463 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
464 start = end;
465
466 // 忽略無效/控制字元。
467 if (!eaten || IsControl(codepoint)) {
468 continue;
469 }
470
471 // 組合字元會與它們正在修改的前一個字符放在一起。
472 if (IsCombining(codepoint)) {
473 if (x == -1) {
474 ++x;
475 out.push_back(x);
476 }
477 continue;
478 }
479
480 // Fullwidth characters take two cells. The second is made of the empty
481 // 全形字元佔用兩個單元格。第二個是空字串,用於保留第一個佔用的空間。
482 if (IsFullWidth(codepoint)) {
483 ++x;
484 out.push_back(x);
485 out.push_back(x);
486 continue;
487 }
488
489 // 一般字元:
490 ++x;
491 out.push_back(x);
492 }
493 return out;
494}
495
496int GlyphCount(std::string_view input) {
497 int size = 0;
498 size_t start = 0;
499 size_t end = 0;
500 while (start < input.size()) {
501 uint32_t codepoint = 0;
502 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
503 start = end;
504
505 // 忽略無效字元:
506 if (!eaten || IsControl(codepoint)) {
507 continue;
508 }
509
510 // 忽略組合字元,除非它們沒有可
511 // 組合的前置字元。
512 if (IsCombining(codepoint)) {
513 if (size == 0) {
514 size++;
515 }
516 continue;
517 }
518
519 size++;
520 }
521 return size;
522}
523
524std::vector<WordBreakProperty> Utf8ToWordBreakProperty(std::string_view input) {
525 std::vector<WordBreakProperty> out;
526 out.reserve(input.size());
527 size_t start = 0;
528 size_t end = 0;
529 while (start < input.size()) {
530 uint32_t codepoint = 0;
531 if (!EatCodePoint(input, start, &end, &codepoint)) {
532 start = end;
533 continue;
534 }
535 start = end;
536
537 // 忽略控制字元。
538 if (IsControl(codepoint)) {
539 continue;
540 }
541
542 // 忽略組合字元。
543 if (IsCombining(codepoint)) {
544 continue;
545 }
546
547 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
548 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
549 out.push_back(interval.property);
550 }
551 return out;
552}
553
554/// 將 std::wstring 轉換為 UTF8 std::string。
555std::string to_string(std::wstring_view s) {
556 std::string out;
557
558 size_t i = 0;
559 uint32_t codepoint = 0;
560 while (EatCodePoint(s, i, &i, &codepoint)) {
561 // 碼位 <-> UTF-8 轉換
562 //
563 // ┏━━━━━━━━┳━━━━━━━━┳━━━━━━━━┳━━━━━━━━┓
564 // ┃Byte 1 ┃Byte 2 ┃Byte 3 ┃Byte 4 ┃
565 // ┡━━━━━━━━╇━━━━━━━━╇━━━━━━━━╇━━━━━━━━┩
566 // │0xxxxxxx│ │ │ │
567 // ├────────┼────────┼────────┼────────┤
568 // │110xxxxx│10xxxxxx│ │ │
569 // ├────────┼────────┼────────┼────────┤
570 // │1110xxxx│10xxxxxx│10xxxxxx│ │
571 // ├────────┼────────┼────────┼────────┤
572 // │11110xxx│10xxxxxx│10xxxxxx│10xxxxxx│
573 // └────────┴────────┴────────┴────────┘
574
575 // 1 位元組 UTF8
576 if (codepoint <= 0b000'0000'0111'1111) { // NOLINT
577 const uint8_t p1 = codepoint;
578 out.push_back(p1); // NOLINT
579 continue;
580 }
581
582 // 2 位元組 UTF8
583 if (codepoint <= 0b000'0111'1111'1111) { // NOLINT
584 uint8_t p2 = codepoint & 0b111111; // NOLINT
585 codepoint >>= 6; // NOLINT
586 uint8_t p1 = codepoint; // NOLINT
587 out.push_back(0b11000000 + p1); // NOLINT
588 out.push_back(0b10000000 + p2); // NOLINT
589 continue;
590 }
591
592 // 3 位元組 UTF8
593 if (codepoint <= 0b1111'1111'1111'1111) { // NOLINT
594 uint8_t p3 = codepoint & 0b111111; // NOLINT
595 codepoint >>= 6; // NOLINT
596 uint8_t p2 = codepoint & 0b111111; // NOLINT
597 codepoint >>= 6; // NOLINT
598 uint8_t p1 = codepoint; // NOLINT
599 out.push_back(0b11100000 + p1); // NOLINT
600 out.push_back(0b10000000 + p2); // NOLINT
601 out.push_back(0b10000000 + p3); // NOLINT
602 continue;
603 }
604
605 // 4 位元組 UTF8
606 if (codepoint <= 0b1'0000'1111'1111'1111'1111) { // NOLINT
607 uint8_t p4 = codepoint & 0b111111; // NOLINT
608 codepoint >>= 6; // NOLINT
609 uint8_t p3 = codepoint & 0b111111; // NOLINT
610 codepoint >>= 6; // NOLINT
611 uint8_t p2 = codepoint & 0b111111; // NOLINT
612 codepoint >>= 6; // NOLINT
613 uint8_t p1 = codepoint; // NOLINT
614 out.push_back(0b11110000 + p1); // NOLINT
615 out.push_back(0b10000000 + p2); // NOLINT
616 out.push_back(0b10000000 + p3); // NOLINT
617 out.push_back(0b10000000 + p4); // NOLINT
618 continue;
619 }
620
621 // 其他情況?
622 }
623 return out;
624}
625
626/// 將 UTF8 std::string 轉換為 std::wstring。
627std::wstring to_wstring(std::string_view s) {
628 std::wstring out;
629
630 size_t i = 0;
631 uint32_t codepoint = 0;
632 while (EatCodePoint(s, i, &i, &codepoint)) {
633 // 在 Linux 上,wstring 使用 UTF32 編碼:
634 if constexpr (sizeof(wchar_t) == 4) {
635 out.push_back(codepoint); // NOLINT
636 continue;
637 }
638
639 // 在 Windows 上,wstring 使用 UTF16 編碼:
640
641 // 使用 1 個字組編碼的碼點:
642 // NOLINTNEXTLINE
643 if (codepoint < 0xD800 || (codepoint > 0xDFFF && codepoint < 0x10000)) {
644 uint16_t p0 = codepoint; // NOLINT
645 out.push_back(p0); // NOLINT
646 continue;
647 }
648
649 // 使用 2 個 word 編碼的碼位:
650 codepoint -= 0x010000; // NOLINT
651 uint16_t p0 = (((codepoint << 12) >> 22) + 0xD800); // NOLINT
652 uint16_t p1 = (((codepoint << 22) >> 22) + 0xDC00); // NOLINT
653 out.push_back(p0); // NOLINT
654 out.push_back(p1); // NOLINT
655 }
656 return out;
657}
658
659} // namespace ftxui
Decorator size(WidthOrHeight direction, Constraint constraint, int value)
對元素的大小套用限制。
FTXUI ftxui:: 命名空間
Definition animation.hpp:11
bool IsControl(uint32_t ucs)
Definition string.cpp:256
WordBreakProperty CodepointToWordBreakProperty(uint32_t codepoint)
Definition string.cpp:270
size_t GlyphPrevious(std::string_view input, size_t start)
Definition string.cpp:390
FTXUI_EXPORT(SCREEN) int string_width(std std::vector< std::string > Utf8ToGlyphs(std::string_view input)
Definition string.cpp:348
int string_width(std::string_view input)
Definition string.cpp:298
bool IsCombining(uint32_t ucs)
Definition string.cpp:244
int wchar_width(wchar_t ucs)
Definition string.cpp:276
bool EatCodePoint(std::string_view input, size_t start, size_t *end, uint32_t *ucs)
Definition string.cpp:137
std::string to_string(std::wstring_view s)
將 std::wstring 轉換為 UTF8 std::string。
Definition string.cpp:555
int GlyphCount(std::string_view input)
Definition string.cpp:496
std::vector< WordBreakProperty > Utf8ToWordBreakProperty(std::string_view input)
Definition string.cpp:524
int wstring_width(const std::wstring &text)
Definition string.cpp:280
std::vector< int > CellToGlyphIndex(std::string_view input)
Definition string.cpp:455
FTXUI_EXPORT(SCREEN) std FTXUI_EXPORT(SCREEN) std std::wstring to_wstring(T s)
Definition string.hpp:18
size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start)
Definition string.cpp:441
bool IsFullWidth(uint32_t ucs)
Definition string.cpp:248
size_t GlyphNext(std::string_view input, size_t start)
Definition string.cpp:415
constexpr std::array< Interval, 123 > g_full_width_characters
constexpr std::array< WordBreakPropertyInterval, 1100 > g_word_break_intervals