FTXUI 7.0.3
C++ functional terminal UI.
Loading...
Searching...
No Matches
string.cpp
Go to the documentation of this file.
1// Copyright 2020 Arthur Sonzogni. All rights reserved.
2// Use of this source code is governed by the MIT license that can be found in
3// the LICENSE file.
4//
5// Content of this file was created thanks to:
6// -
7// https://www.unicode.org/Public/UCD/latest/ucd/auxiliary/WordBreakProperty.txt
8// - Markus Kuhn -- 2007-05-26 (Unicode 5.0)
9// http://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c
10// Thanks you!
11
13
14#include <array> // for array
15#include <cstddef> // for size_t
16#include <cstdint> // for uint32_t, uint8_t, uint16_t, int32_t
17#include <string> // for string, basic_string, wstring
18#include <string_view> // for string_view
19#include <tuple> // for _Swallow_assign, ignore
20#include <vector>
21
22#include "ftxui/screen/deprecated.hpp" // for wchar_width, wstring_width
23#include "ftxui/screen/string_internal.hpp" // for WordBreakProperty, EatCodePoint, CodepointToWordBreakProperty, GlyphCount, GlyphIterate, GlyphNext, GlyphPrevious, IsCombining, IsControl, IsFullWidth, Utf8ToWordBreakProperty
24
25namespace {
26
27struct Interval {
28 uint32_t first;
29 uint32_t last;
30};
31
32using WBP = ftxui::WordBreakProperty;
33struct WordBreakPropertyInterval {
34 uint32_t first;
35 uint32_t last;
36 WBP property;
37};
38
39// g_full_width_charactersとg_word_break_intervalsは、tools/gen_unicode_tables.py
40// によってUnicode Character Databaseから生成される。
42
43// WBP::Extend文字間隔のみのテーブルを構築します。
44constexpr auto g_extend_characters{[]() constexpr {
45 // 拡張文字区間の数を計算する
46 constexpr size_t size = []() constexpr {
47 size_t count = 0;
48 for (auto interval : g_word_break_intervals) {
49 if (interval.property == WBP::Extend) {
50 count++;
51 }
52 }
53 return count;
54 }();
55
56 // 拡張文字区間の配列を作成する
57 std::array<Interval, size> result{};
58 size_t index = 0;
59 for (auto interval : g_word_break_intervals) {
60 if (interval.property == WBP::Extend) {
61 result[index++] = {interval.first, interval.last}; // NOLINT
62 }
63 }
64 return result;
65}()};
66
67// ソートされたIntervalのリスト内でコードポイントを検索します。
68template <size_t N>
69bool Bisearch(uint32_t ucs, const std::array<Interval, N>& table) {
70 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
71 return false;
72 }
73
74 int min = 0;
75 int max = N - 1;
76 while (max >= min) {
77 const int mid = (min + max) / 2;
78 if (ucs > table[mid].last) { // NOLINT
79 min = mid + 1;
80 } else if (ucs < table[mid].first) { // NOLINT
81 max = mid - 1;
82 } else {
83 return true;
84 }
85 }
86
87 return false;
88}
89
90// ソートされたInterval + プロパティのリスト内で値を検索します。
91template <class C, size_t N>
92bool Bisearch(uint32_t ucs, const std::array<C, N>& table, C* out) {
93 if (ucs < table.front().first || ucs > table.back().last) { // NOLINT
94 return false;
95 }
96
97 int min = 0;
98 int max = N - 1;
99 while (max >= min) {
100 const int mid = (min + max) / 2;
101 if (ucs > table[mid].last) { // NOLINT
102 min = mid + 1;
103 } else if (ucs < table[mid].first) { // NOLINT
104 max = mid - 1;
105 } else {
106 *out = table[mid]; // NOLINT
107 return true;
108 }
109 }
110
111 return false;
112}
113
114int codepoint_width(uint32_t ucs) {
115 if (ftxui::IsControl(ucs)) {
116 return -1;
117 }
118
119 if (ftxui::IsCombining(ucs)) {
120 return 0;
121 }
122
123 if (ftxui::IsFullWidth(ucs)) {
124 return 2;
125 }
126
127 return 1;
128}
129
130} // namespace
131
132namespace ftxui {
133
134// UTF8エンコードされた文字列|input|から、1つのコードポイントを表す1〜4バイトを読み取ります。// コードポイントを|ucs|に格納します。|start|から開始し、連続する実行のために次のバイトの// 開始位置を表すように|end|を更新します。
135bool EatCodePoint(std::string_view input,
136 size_t start,
137 size_t* end,
138 uint32_t* ucs) {
139 if (start >= input.size()) {
140 *end = start + 1;
141 return false;
142 }
143 const uint8_t C0 = input[start];
144
145 // 1バイト文字列。
146 if ((C0 & 0b1000'0000) == 0b0000'0000) { // NOLINT
147 *ucs = C0 & 0b0111'1111; // NOLINT
148 *end = start + 1;
149 return true;
150 }
151
152 // 2バイト文字列。
153 if ((C0 & 0b1110'0000) == 0b1100'0000 && // NOLINT
154 start + 1 < input.size()) {
155 const uint8_t C1 = input[start + 1];
156 *ucs = 0;
157 *ucs += C0 & 0b0001'1111; // NOLINT
158 *ucs <<= 6; // NOLINT
159 *ucs += C1 & 0b0011'1111; // NOLINT
160 *end = start + 2;
161 return true;
162 }
163
164 // 3バイト文字列。
165 if ((C0 & 0b1111'0000) == 0b1110'0000 && // NOLINT
166 start + 2 < input.size()) {
167 const uint8_t C1 = input[start + 1];
168 const uint8_t C2 = input[start + 2];
169 *ucs = 0;
170 *ucs += C0 & 0b0000'1111; // NOLINT
171 *ucs <<= 6; // NOLINT
172 *ucs += C1 & 0b0011'1111; // NOLINT
173 *ucs <<= 6; // NOLINT
174 *ucs += C2 & 0b0011'1111; // NOLINT
175 *end = start + 3;
176 return true;
177 }
178
179 // 4バイト文字列。
180 if ((C0 & 0b1111'1000) == 0b1111'0000 && // NOLINT
181 start + 3 < input.size()) {
182 const uint8_t C1 = input[start + 1];
183 const uint8_t C2 = input[start + 2];
184 const uint8_t C3 = input[start + 3];
185 *ucs = 0;
186 *ucs += C0 & 0b0000'0111; // NOLINT
187 *ucs <<= 6; // NOLINT
188 *ucs += C1 & 0b0011'1111; // NOLINT
189 *ucs <<= 6; // NOLINT
190 *ucs += C2 & 0b0011'1111; // NOLINT
191 *ucs <<= 6; // NOLINT
192 *ucs += C3 & 0b0011'1111; // NOLINT
193 *end = start + 4;
194 return true;
195 }
196
197 *end = start + 1;
198 return false;
199}
200
201// UTF16エンコードされた文字列|input|から、1つのコードポイントを表す1〜4バイトを読み取ります。// コードポイントを|ucs|に格納します。|start|から開始し、連続する実行のために次のバイトの// 開始位置を表すように|end|を更新します。
202bool EatCodePoint(std::wstring_view input,
203 size_t start,
204 size_t* end,
205 uint32_t* ucs) {
206 if (start >= input.size()) {
207 *end = start + 1;
208 return false;
209 }
210
211 // LinuxではwstringはUTF32エンコーディングを使用します。
212 if constexpr (sizeof(wchar_t) == 4) {
213 *ucs = input[start]; // NOLINT
214 *end = start + 1;
215 return true;
216 }
217
218 // Windowsでは、wstringはUTF16エンコーディングを使用します。
219 int32_t C0 = input[start]; // NOLINT
220
221 // 1ワードサイズ:
222 if (C0 < 0xd800 || C0 >= 0xdc00) { // NOLINT
223 *ucs = C0;
224 *end = start + 1;
225 return true;
226 }
227
228 // 2ワードサイズ:
229 if (start + 1 >= input.size()) {
230 *end = start + 2;
231 return false;
232 }
233
234 int32_t C1 = input[start + 1]; // NOLINT
235 *ucs = ((C0 & 0x3ff) << 10) + (C1 & 0x3ff) + 0x10000; // NOLINT
236 *end = start + 2;
237 return true;
238}
239
240bool IsCombining(uint32_t ucs) {
241 return Bisearch(ucs, g_extend_characters);
242}
243
244bool IsFullWidth(uint32_t ucs) {
245 if (ucs < 0x0300) { // 高速パス: // NOLINT
246 return false;
247 }
248
249 return Bisearch(ucs, g_full_width_characters);
250}
251
252bool IsControl(uint32_t ucs) {
253 if (ucs == 0) {
254 return true;
255 }
256 if (ucs < 32) { // NOLINT
257 const uint32_t LINE_FEED = 10;
258 return ucs != LINE_FEED;
259 }
260 if (ucs >= 0x7f && ucs < 0xa0) { // NOLINT
261 return true;
262 }
263 return false;
264}
265
267 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
268 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
269 return interval.property;
270}
271
272int wchar_width(wchar_t ucs) {
273 return codepoint_width(uint32_t(ucs));
274}
275
276int wstring_width(const std::wstring& text) {
277 int width = 0;
278
279 for (const wchar_t& it : text) {
280 const int w = wchar_width(it);
281 if (w < 0) {
282 return -1;
283 }
284 width += w;
285 }
286 return width;
287}
288
289// UTF8でエンコードされた文字列|input|が印刷される際に占めるセル数を返す。
290// 制御文字はスペースを占めず、結合文字は前の文字を修飾し
291// スペースを占めず、全角文字は2セルを占め、その他すべての文字は1セルを
292// 占める。
293int string_width(std::string_view input) {
294 // 1バイト最適化: この関数は単一のASCII文字に対して呼ばれることが多いため、
295 // UTF8デコードをスキップすることでこのケースを最適化できる。
296 if (input.size() == 1) {
297 const char c = input[0];
298 if (c >= 32 && c < 127) { // NOLINT
299 return 1;
300 }
301 }
302
303 // ASCII最適化: 文字列が純粋なASCIIの場合、UTF8デコードをスキップして
304 // 制御文字を無視しながら文字数を数えることができる。
305 bool is_pure_ascii = true;
306 for (const char c : input) {
307 if (c < 31 || c >= 127) { // NOLINT
308 is_pure_ascii = false;
309 break;
310 }
311 }
312 if (is_pure_ascii) {
313 return static_cast<int>(input.size());
314 }
315
316 int width = 0;
317 size_t start = 0;
318 while (start < input.size()) {
319 uint32_t codepoint = 0;
320 if (!EatCodePoint(input, start, &start, &codepoint)) {
321 continue;
322 }
323
324 if (IsControl(codepoint)) {
325 continue;
326 }
327
328 if (IsCombining(codepoint)) {
329 continue;
330 }
331
332 if (IsFullWidth(codepoint)) {
333 width += 2;
334 continue;
335 }
336
337 width += 1;
338 }
339 return width;
340}
341
342std::vector<std::string> Utf8ToGlyphs(std::string_view input) {
343 std::vector<std::string> out;
344 out.reserve(input.size());
345 size_t start = 0;
346 size_t end = 0;
347 while (start < input.size()) {
348 uint32_t codepoint = 0;
349 if (!EatCodePoint(input, start, &end, &codepoint)) {
350 start = end;
351 continue;
352 }
353
354 const auto append = input.substr(start, end - start);
355 start = end;
356
357 // 制御文字を無視する。
358 if (IsControl(codepoint)) {
359 continue;
360 }
361
362 // 結合文字は、それらが修飾する直前のグリフと一緒に配置される。
363 if (IsCombining(codepoint)) {
364 if (!out.empty()) {
365 out.back() += append;
366 }
367 continue;
368 }
369
370 // 全角文字は2セルを占有します。2番目のセルは、最初のセルが
371 // 占有するスペースを確保するための空文字列で構成されます。
372 if (IsFullWidth(codepoint)) {
373 out.emplace_back(append);
374 out.emplace_back("");
375 continue;
376 }
377
378 // 通常の文字:
379 out.emplace_back(append);
380 }
381 return out;
382}
383
384size_t GlyphPrevious(std::string_view input, size_t start) {
385 while (true) {
386 if (start == 0) {
387 return 0;
388 }
389 start--;
390
391 // UTF8の継続バイトをスキップします。
392 if ((input[start] & 0b1100'0000) == 0b1000'0000) {
393 continue;
394 }
395
396 uint32_t codepoint = 0;
397 size_t end = 0;
398 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
399
400 // 無効な文字、制御文字、結合文字は無視します。
401 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
402 continue;
403 }
404
405 return start;
406 }
407}
408
409size_t GlyphNext(std::string_view input, size_t start) {
410 bool glyph_found = false;
411 while (start < input.size()) {
412 size_t end = 0;
413 uint32_t codepoint = 0;
414 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
415
416 // 無効な文字、制御文字、結合文字は無視します。
417 if (!eaten || IsControl(codepoint) || IsCombining(codepoint)) {
418 start = end;
419 continue;
420 }
421
422 // 次のグリフの開始を読み取ります。要求されたものを読み取っている場合は、// その開始位置を直ちに戻します。 if (glyph_found) {
423 if (glyph_found) {
424 return static_cast<int>(start);
425 }
426
427 // それ以外の場合、このグリフをスキップして反復処理します:
428 glyph_found = true;
429 start = end;
430 }
431 return static_cast<int>(input.size());
432}
433
434size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start) {
435 if (glyph_offset >= 0) {
436 for (int i = 0; i < glyph_offset; ++i) {
437 start = GlyphNext(input, start);
438 }
439 return start;
440 } else {
441 for (int i = 0; i < -glyph_offset; ++i) {
442 start = GlyphPrevious(input, start);
443 }
444 return start;
445 }
446}
447
448std::vector<int> CellToGlyphIndex(std::string_view input) {
449 int x = -1;
450 std::vector<int> out;
451 out.reserve(input.size());
452 size_t start = 0;
453 size_t end = 0;
454 while (start < input.size()) {
455 uint32_t codepoint = 0;
456 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
457 start = end;
458
459 // 無効な文字・制御文字は無視します。
460 if (!eaten || IsControl(codepoint)) {
461 continue;
462 }
463
464 // 結合文字は、それらが修飾する直前のグリフと一緒に配置される。
465 if (IsCombining(codepoint)) {
466 if (x == -1) {
467 ++x;
468 out.push_back(x);
469 }
470 continue;
471 }
472
473 // 全角文字は2セルを占有します。2番目のセルは、最初のセルが
474 // 占有するスペースを確保するための空文字列で構成されます。
475 if (IsFullWidth(codepoint)) {
476 ++x;
477 out.push_back(x);
478 out.push_back(x);
479 continue;
480 }
481
482 // 通常の文字:
483 ++x;
484 out.push_back(x);
485 }
486 return out;
487}
488
489int GlyphCount(std::string_view input) {
490 int size = 0;
491 size_t start = 0;
492 size_t end = 0;
493 while (start < input.size()) {
494 uint32_t codepoint = 0;
495 const bool eaten = EatCodePoint(input, start, &end, &codepoint);
496 start = end;
497
498 // 無効な文字は無視します:
499 if (!eaten || IsControl(codepoint)) {
500 continue;
501 }
502
503 // 結合文字を無視します。ただし、先行する結合文字がない場合は除きます。 if (IsCombining(codepoint)) {
504 if (IsCombining(codepoint)) {
505 if (size == 0) {
506 size++;
507 }
508 continue;
509 }
510
511 size++;
512 }
513 return size;
514}
515
516std::vector<WordBreakProperty> Utf8ToWordBreakProperty(std::string_view input) {
517 std::vector<WordBreakProperty> out;
518 out.reserve(input.size());
519 size_t start = 0;
520 size_t end = 0;
521 while (start < input.size()) {
522 uint32_t codepoint = 0;
523 if (!EatCodePoint(input, start, &end, &codepoint)) {
524 start = end;
525 continue;
526 }
527 start = end;
528
529 // 制御文字を無視する。
530 if (IsControl(codepoint)) {
531 continue;
532 }
533
534 // 結合文字は無視します。
535 if (IsCombining(codepoint)) {
536 continue;
537 }
538
539 WordBreakPropertyInterval interval = {0, 0, WBP::ALetter};
540 std::ignore = Bisearch(codepoint, g_word_break_intervals, &interval);
541 out.push_back(interval.property);
542 }
543 return out;
544}
545
546/// std::wstringをUTF8 std::stringに変換します。
547std::string to_string(std::wstring_view s) {
548 std::string out;
549
550 size_t i = 0;
551 uint32_t codepoint = 0;
552 while (EatCodePoint(s, i, &i, &codepoint)) {
553 // コードポイント <-> UTF-8 変換
554 //
555 // ┏━━━━━━━━┳━━━━━━━━┳━━━━━━━━┳━━━━━━━━┓
556 // ┃Byte 1 ┃Byte 2 ┃Byte 3 ┃Byte 4 ┃
557 // ┡━━━━━━━━╇━━━━━━━━╇━━━━━━━━╇━━━━━━━━┩
558 // │0xxxxxxx│ │ │ │
559 // ├────────┼────────┼────────┼────────┤
560 // │110xxxxx│10xxxxxx│ │ │
561 // ├────────┼────────┼────────┼────────┤
562 // │1110xxxx│10xxxxxx│10xxxxxx│ │
563 // ├────────┼────────┼────────┼────────┤
564 // │11110xxx│10xxxxxx│10xxxxxx│10xxxxxx│
565 // └────────┴────────┴────────┴────────┘
566
567 // 1バイトUTF8
568 if (codepoint <= 0b000'0000'0111'1111) { // NOLINT
569 const uint8_t p1 = codepoint;
570 out.push_back(p1); // NOLINT
571 continue;
572 }
573
574 // 2バイトUTF8
575 if (codepoint <= 0b000'0111'1111'1111) { // NOLINT
576 uint8_t p2 = codepoint & 0b111111; // NOLINT
577 codepoint >>= 6; // NOLINT
578 uint8_t p1 = codepoint; // NOLINT
579 out.push_back(0b11000000 + p1); // NOLINT
580 out.push_back(0b10000000 + p2); // NOLINT
581 continue;
582 }
583
584 // 3バイトUTF8
585 if (codepoint <= 0b1111'1111'1111'1111) { // NOLINT
586 uint8_t p3 = codepoint & 0b111111; // NOLINT
587 codepoint >>= 6; // NOLINT
588 uint8_t p2 = codepoint & 0b111111; // NOLINT
589 codepoint >>= 6; // NOLINT
590 uint8_t p1 = codepoint; // NOLINT
591 out.push_back(0b11100000 + p1); // NOLINT
592 out.push_back(0b10000000 + p2); // NOLINT
593 out.push_back(0b10000000 + p3); // NOLINT
594 continue;
595 }
596
597 // 4バイトUTF8
598 if (codepoint <= 0b1'0000'1111'1111'1111'1111) { // NOLINT
599 uint8_t p4 = codepoint & 0b111111; // NOLINT
600 codepoint >>= 6; // NOLINT
601 uint8_t p3 = codepoint & 0b111111; // NOLINT
602 codepoint >>= 6; // NOLINT
603 uint8_t p2 = codepoint & 0b111111; // NOLINT
604 codepoint >>= 6; // NOLINT
605 uint8_t p1 = codepoint; // NOLINT
606 out.push_back(0b11110000 + p1); // NOLINT
607 out.push_back(0b10000000 + p2); // NOLINT
608 out.push_back(0b10000000 + p3); // NOLINT
609 out.push_back(0b10000000 + p4); // NOLINT
610 continue;
611 }
612
613 // それ以外の何か?
614 }
615 return out;
616}
617
618/// UTF8 std::stringをstd::wstringに変換します。
619std::wstring to_wstring(std::string_view s) {
620 std::wstring out;
621
622 size_t i = 0;
623 uint32_t codepoint = 0;
624 while (EatCodePoint(s, i, &i, &codepoint)) {
625 // LinuxではwstringはUTF32でエンコードされています:
626 if constexpr (sizeof(wchar_t) == 4) {
627 out.push_back(codepoint); // NOLINT
628 continue;
629 }
630
631 // Windowsでは、wstringはUTF16でエンコードされています:
632
633 // 1ワードを使用してエンコードされたコードポイント:
634 // NOLINTNEXTLINE
635 if (codepoint < 0xD800 || (codepoint > 0xDFFF && codepoint < 0x10000)) {
636 uint16_t p0 = codepoint; // NOLINT
637 out.push_back(p0); // NOLINT
638 continue;
639 }
640
641 // 2ワードでエンコードされたコードポイント:
642 codepoint -= 0x010000; // NOLINT
643 uint16_t p0 = (((codepoint << 12) >> 22) + 0xD800); // NOLINT
644 uint16_t p1 = (((codepoint << 22) >> 22) + 0xDC00); // NOLINT
645 out.push_back(p0); // NOLINT
646 out.push_back(p1); // NOLINT
647 }
648 return out;
649}
650
651} // namespace ftxui
Decorator size(WidthOrHeight direction, Constraint constraint, int value)
要素のサイズに制約を適用する。
FTXUI ftxui::名前空間
Definition animation.hpp:11
bool IsControl(uint32_t ucs)
Definition string.cpp:252
WordBreakProperty CodepointToWordBreakProperty(uint32_t codepoint)
Definition string.cpp:266
size_t GlyphPrevious(std::string_view input, size_t start)
Definition string.cpp:384
FTXUI_EXPORT(SCREEN) int string_width(std std::vector< std::string > Utf8ToGlyphs(std::string_view input)
Definition string.cpp:342
int string_width(std::string_view input)
Definition string.cpp:293
bool IsCombining(uint32_t ucs)
Definition string.cpp:240
int wchar_width(wchar_t ucs)
Definition string.cpp:272
bool EatCodePoint(std::string_view input, size_t start, size_t *end, uint32_t *ucs)
Definition string.cpp:135
std::string to_string(std::wstring_view s)
std::wstringをUTF8 std::stringに変換します。
Definition string.cpp:547
int GlyphCount(std::string_view input)
Definition string.cpp:489
std::vector< WordBreakProperty > Utf8ToWordBreakProperty(std::string_view input)
Definition string.cpp:516
int wstring_width(const std::wstring &text)
Definition string.cpp:276
std::vector< int > CellToGlyphIndex(std::string_view input)
Definition string.cpp:448
FTXUI_EXPORT(SCREEN) std FTXUI_EXPORT(SCREEN) std std::wstring to_wstring(T s)
Definition string.hpp:18
size_t GlyphIterate(std::string_view input, int glyph_offset, size_t start)
Definition string.cpp:434
bool IsFullWidth(uint32_t ucs)
Definition string.cpp:244
size_t GlyphNext(std::string_view input, size_t start)
Definition string.cpp:409
constexpr std::array< Interval, 123 > g_full_width_characters
constexpr std::array< WordBreakPropertyInterval, 1100 > g_word_break_intervals