Sourcemeta Core 0.0.0
Loading...
Searching...
No Matches
unicode.h
1#ifndef SOURCEMETA_CORE_UNICODE_H_
2#define SOURCEMETA_CORE_UNICODE_H_
3
4#ifndef SOURCEMETA_CORE_UNICODE_EXPORT
5#include <sourcemeta/core/unicode_export.h>
6#endif
7
8#include <sourcemeta/core/unicode_ucd.h>
9
10#include <cstddef> // std::size_t
11#include <cstdint> // std::uint8_t, std::uint64_t
12#include <cstring> // std::memcpy
13#include <istream> // std::istream
14#include <optional> // std::optional
15#include <ostream> // std::ostream
16#include <string> // std::string, std::u32string, std::wstring
17#include <string_view> // std::string_view, std::u32string_view, std::wstring_view
18#include <utility> // std::pair, std::make_pair
19
28
29namespace sourcemeta::core {
30
42SOURCEMETA_CORE_UNICODE_EXPORT
43auto codepoint_to_utf8(const char32_t codepoint) -> std::string;
44
59SOURCEMETA_CORE_UNICODE_EXPORT
60auto codepoint_to_utf8(const char32_t codepoint, std::ostream &output) -> void;
61
75SOURCEMETA_CORE_UNICODE_EXPORT
76auto codepoint_to_utf8(const char32_t codepoint, std::string &output) -> void;
77
92SOURCEMETA_CORE_UNICODE_EXPORT
93auto utf8_to_utf32(std::istream &input) -> std::optional<std::u32string>;
94
107SOURCEMETA_CORE_UNICODE_EXPORT
108auto utf8_to_utf32(const std::string_view input)
109 -> std::optional<std::u32string>;
110
123SOURCEMETA_CORE_UNICODE_EXPORT
124auto utf32_to_utf8(const std::u32string_view input) -> std::string;
125
145SOURCEMETA_CORE_UNICODE_EXPORT
146auto utf32_to_utf8_lenient(const std::u32string_view input) -> std::string;
147
163SOURCEMETA_CORE_UNICODE_EXPORT
164auto to_valid_utf8(const std::string_view input) -> std::string;
165
177SOURCEMETA_CORE_UNICODE_EXPORT
178auto utf8_to_wide(const std::string_view input) -> std::wstring;
179
190SOURCEMETA_CORE_UNICODE_EXPORT
191auto wide_to_utf8(const std::wstring_view input) -> std::string;
192
210constexpr auto utf8_lead_byte_size(const unsigned char byte) -> std::uint8_t {
211 if (byte < 0x80) {
212 return 1;
213 }
214 if (byte >= 0xC2 && byte <= 0xDF) {
215 return 2;
216 }
217 if (byte >= 0xE0 && byte <= 0xEF) {
218 return 3;
219 }
220 if (byte >= 0xF0 && byte <= 0xF4) {
221 return 4;
222 }
223 return 0;
224}
225
248constexpr auto is_utf8_tail(const unsigned char lead,
249 const std::size_t position,
250 const unsigned char byte) -> bool {
251 const unsigned char lower{static_cast<unsigned char>(
252 position > 1 ? 0x80
253 : (lead == 0xE0 ? 0xA0 : (lead == 0xF0 ? 0x90 : 0x80)))};
254 const unsigned char upper{static_cast<unsigned char>(
255 position > 1 ? 0xBF
256 : (lead == 0xED ? 0x9F : (lead == 0xF4 ? 0x8F : 0xBF)))};
257 return byte >= lower && byte <= upper;
258}
259
273constexpr auto utf8_sequence_size(const std::string_view input) -> std::size_t {
274 if (input.empty()) {
275 return 0;
276 }
277
278 const auto lead{static_cast<unsigned char>(input.front())};
279 const std::size_t size{utf8_lead_byte_size(lead)};
280 if (size == 0 || input.size() < size) {
281 return 0;
282 }
283
284 for (std::size_t index{1}; index < size; index += 1) {
285 if (!is_utf8_tail(lead, index, static_cast<unsigned char>(input[index]))) {
286 return 0;
287 }
288 }
289
290 return size;
291}
292
305inline auto is_valid_utf8(const std::string_view input) noexcept -> bool {
306 constexpr std::uint64_t HIGH_BITS{0x8080808080808080ULL};
307 const auto size{input.size()};
308 std::size_t position{0};
309 while (position < size) {
310 // Comparing what is left rather than the position past the word keeps the
311 // sum of a position and a word from wrapping around
312 if (size - position >= 8) {
313 std::uint64_t word{0};
314 std::memcpy(&word, input.data() + position, 8);
315 if ((word & HIGH_BITS) == 0) {
316 position += 8;
317 continue;
318 }
319 }
320
321 if (static_cast<unsigned char>(input[position]) < 0x80) {
322 position += 1;
323 continue;
324 }
325
326 const auto length{utf8_sequence_size(input.substr(position))};
327 if (length == 0) {
328 return false;
329 }
330
331 position += length;
332 }
333
334 return true;
335}
336
350constexpr auto is_utf8_continuation(const unsigned char byte) -> bool {
351 return byte >= 0x80 && byte <= 0xBF;
352}
353
366constexpr auto utf8_codepoint_count(const std::string_view input)
367 -> std::size_t {
368 std::size_t count{0};
369 for (const auto byte : input) {
370 if (!is_utf8_continuation(static_cast<unsigned char>(byte))) {
371 count += 1;
372 }
373 }
374
375 return count;
376}
377
393constexpr auto utf8_codepoint_within(const std::string_view input,
394 const std::size_t minimum,
395 const std::size_t maximum) -> bool {
396 const auto bytes{input.size()};
397
398 // A code point is at least one byte, so the count never exceeds the byte
399 // length: fewer bytes than the minimum cannot reach it
400 if (bytes < minimum) {
401 return false;
402 }
403
404 // A code point is at most four bytes, so the count is at least the byte
405 // length divided by four (rounded up): too many bytes cannot fit the maximum
406 const auto lower_bound{(bytes + 3) / 4};
407 if (lower_bound > maximum) {
408 return false;
409 }
410
411 // If the byte length already satisfies the maximum and the rounded-up lower
412 // bound already satisfies the minimum, the count must be in range
413 if (bytes <= maximum && lower_bound >= minimum) {
414 return true;
415 }
416
417 // Otherwise count, stopping as soon as the maximum is exceeded. Reaching here
418 // implies the byte length is at most four times the maximum, so this is
419 // bounded regardless of how long the input is
420 std::size_t count{0};
421 for (const auto byte : input) {
422 if (!is_utf8_continuation(static_cast<unsigned char>(byte))) {
423 count += 1;
424 if (count > maximum) {
425 return false;
426 }
427 }
428 }
429
430 return count >= minimum;
431}
432
447constexpr auto is_surrogate(const char32_t codepoint) -> bool {
448 return codepoint >= 0xD800 && codepoint <= 0xDFFF;
449}
450
465constexpr auto is_valid_codepoint(const char32_t codepoint) -> bool {
466 return codepoint <= 0x10FFFF && !is_surrogate(codepoint);
467}
468
484constexpr auto is_ucschar(const char32_t codepoint) -> bool {
485 if (codepoint >= 0xA0 && codepoint <= 0xD7FF) {
486 return true;
487 }
488 if (codepoint >= 0xF900 && codepoint <= 0xFDCF) {
489 return true;
490 }
491 if (codepoint >= 0xFDF0 && codepoint <= 0xFFEF) {
492 return true;
493 }
494 // Supplementary planes 1 through 14. Each plane allows 0..FFFD;
495 // FFFE and FFFF are noncharacters. Plane 14 starts at offset 0x1000
496 // rather than 0x0000.
497 if (codepoint < 0x10000 || codepoint > 0xEFFFD) {
498 return false;
499 }
500 if (codepoint >= 0xE0000 && codepoint < 0xE1000) {
501 return false;
502 }
503 return (codepoint & 0xFFFFU) <= 0xFFFDU;
504}
505
520constexpr auto is_iprivate(const char32_t codepoint) -> bool {
521 if (codepoint >= 0xE000 && codepoint <= 0xF8FF) {
522 return true;
523 }
524 if (codepoint >= 0xF0000 && codepoint <= 0xFFFFD) {
525 return true;
526 }
527 if (codepoint >= 0x100000 && codepoint <= 0x10FFFD) {
528 return true;
529 }
530 return false;
531}
532
548constexpr auto utf8_codepoint_byte_count(const char32_t codepoint)
549 -> std::uint8_t {
550 if (codepoint < 0x80) {
551 return 1;
552 }
553 if (codepoint < 0x800) {
554 return 2;
555 }
556 if (codepoint < 0x10000) {
557 return 3;
558 }
559 return 4;
560}
561
575SOURCEMETA_CORE_UNICODE_EXPORT
576auto combining_class(const char32_t codepoint) noexcept -> std::uint8_t;
577
594SOURCEMETA_CORE_UNICODE_EXPORT
595auto joining_type(const char32_t codepoint) noexcept -> JoiningType;
596
613SOURCEMETA_CORE_UNICODE_EXPORT
614auto bidi_class(const char32_t codepoint) noexcept -> BidiClass;
615
632SOURCEMETA_CORE_UNICODE_EXPORT
633auto script(const char32_t codepoint) noexcept -> UnicodeScript;
634
652SOURCEMETA_CORE_UNICODE_EXPORT
653auto general_category(const char32_t codepoint) noexcept -> GeneralCategory;
654
669SOURCEMETA_CORE_UNICODE_EXPORT
670auto is_combining_mark(const char32_t codepoint) noexcept -> bool;
671
685SOURCEMETA_CORE_UNICODE_EXPORT
686auto is_id_start(const char32_t codepoint) noexcept -> bool;
687
701SOURCEMETA_CORE_UNICODE_EXPORT
702auto is_id_continue(const char32_t codepoint) noexcept -> bool;
703
720SOURCEMETA_CORE_UNICODE_EXPORT
721auto nfc_quick_check(const char32_t codepoint) noexcept -> NFCQuickCheck;
722
740SOURCEMETA_CORE_UNICODE_EXPORT
741auto canonical_decomposition(const char32_t codepoint) noexcept
742 -> std::u32string_view;
743
761SOURCEMETA_CORE_UNICODE_EXPORT
762auto canonical_composition(const char32_t starter,
763 const char32_t combining) noexcept
764 -> std::optional<char32_t>;
765
778SOURCEMETA_CORE_UNICODE_EXPORT
779auto nfc(const std::u32string_view input) -> std::u32string;
780
795SOURCEMETA_CORE_UNICODE_EXPORT
796auto is_nfc(const std::u32string_view input) -> bool;
797
815SOURCEMETA_CORE_UNICODE_EXPORT
816auto case_fold(const char32_t codepoint) noexcept -> std::u32string_view;
817
829SOURCEMETA_CORE_UNICODE_EXPORT
830auto case_fold(const std::u32string_view input) -> std::u32string;
831
849constexpr auto utf8_codepoint_length(const std::string_view input,
850 const std::string_view::size_type position)
851 -> std::size_t {
852 if (position >= input.size()) {
853 return 0;
854 }
855 const auto byte_0{static_cast<unsigned char>(input[position])};
856 const auto size{utf8_lead_byte_size(byte_0)};
857 if (size == 0 || position + size > input.size()) {
858 return 0;
859 }
860 if (size == 1) {
861 return 1;
862 }
863
864 // The second byte after the lead has tighter sub-ranges for specific leads
865 // (RFC 3629 ยง4) that exclude overlong encodings, surrogates, and code
866 // points above U+10FFFF
867 const auto byte_1{static_cast<unsigned char>(input[position + 1])};
868 bool byte_1_ok{false};
869 if (size == 2) {
870 byte_1_ok = is_utf8_continuation(byte_1);
871 } else if (size == 3) {
872 if (byte_0 == 0xE0) {
873 byte_1_ok = byte_1 >= 0xA0 && byte_1 <= 0xBF;
874 } else if (byte_0 == 0xED) {
875 byte_1_ok = byte_1 >= 0x80 && byte_1 <= 0x9F;
876 } else {
877 byte_1_ok = is_utf8_continuation(byte_1);
878 }
879 } else {
880 if (byte_0 == 0xF0) {
881 byte_1_ok = byte_1 >= 0x90 && byte_1 <= 0xBF;
882 } else if (byte_0 == 0xF4) {
883 byte_1_ok = byte_1 >= 0x80 && byte_1 <= 0x8F;
884 } else {
885 byte_1_ok = is_utf8_continuation(byte_1);
886 }
887 }
888
889 if (!byte_1_ok) {
890 return 0;
891 }
892
893 // Remaining continuation bytes (if any) are unconstrained beyond the
894 // continuation byte range
895 for (std::size_t index{2}; index < size; ++index) {
897 static_cast<unsigned char>(input[position + index]))) {
898 return 0;
899 }
900 }
901
902 return size;
903}
904
922constexpr auto utf8_decode(const std::string_view input,
923 const std::string_view::size_type position)
924 -> std::optional<std::pair<char32_t, std::size_t>> {
925 const auto size{utf8_codepoint_length(input, position)};
926 if (size == 0) {
927 return std::nullopt;
928 }
929
930 const auto lead{static_cast<unsigned char>(input[position])};
931 char32_t codepoint{0};
932 if (size == 1) {
933 codepoint = static_cast<char32_t>(lead);
934 } else if (size == 2) {
935 codepoint = static_cast<char32_t>(lead & 0x1FU);
936 } else if (size == 3) {
937 codepoint = static_cast<char32_t>(lead & 0x0FU);
938 } else {
939 codepoint = static_cast<char32_t>(lead & 0x07U);
940 }
941
942 for (std::size_t index{1}; index < size; ++index) {
943 const auto continuation{
944 static_cast<unsigned char>(input[position + index])};
945 codepoint = (codepoint << 6) | static_cast<char32_t>(continuation & 0x3FU);
946 }
947
948 return std::make_pair(codepoint, size);
949}
950
951} // namespace sourcemeta::core
952
953#endif
constexpr auto utf8_lead_byte_size(const unsigned char byte) -> std::uint8_t
Definition unicode.h:210
SOURCEMETA_CORE_UNICODE_EXPORT auto case_fold(const char32_t codepoint) noexcept -> std::u32string_view
SOURCEMETA_CORE_UNICODE_EXPORT auto nfc(const std::u32string_view input) -> std::u32string
SOURCEMETA_CORE_UNICODE_EXPORT auto utf8_to_wide(const std::string_view input) -> std::wstring
constexpr auto is_ucschar(const char32_t codepoint) -> bool
Definition unicode.h:484
SOURCEMETA_CORE_UNICODE_EXPORT auto codepoint_to_utf8(const char32_t codepoint) -> std::string
constexpr auto utf8_codepoint_count(const std::string_view input) -> std::size_t
Definition unicode.h:366
constexpr auto utf8_sequence_size(const std::string_view input) -> std::size_t
Definition unicode.h:273
SOURCEMETA_CORE_UNICODE_EXPORT auto canonical_decomposition(const char32_t codepoint) noexcept -> std::u32string_view
SOURCEMETA_CORE_UNICODE_EXPORT auto utf32_to_utf8_lenient(const std::u32string_view input) -> std::string
JoiningType
Definition unicode_ucd.h:21
constexpr auto is_surrogate(const char32_t codepoint) -> bool
Definition unicode.h:447
SOURCEMETA_CORE_UNICODE_EXPORT auto is_nfc(const std::u32string_view input) -> bool
SOURCEMETA_CORE_UNICODE_EXPORT auto wide_to_utf8(const std::wstring_view input) -> std::string
constexpr auto utf8_codepoint_byte_count(const char32_t codepoint) -> std::uint8_t
Definition unicode.h:548
SOURCEMETA_CORE_UNICODE_EXPORT auto utf8_to_utf32(std::istream &input) -> std::optional< std::u32string >
SOURCEMETA_CORE_UNICODE_EXPORT auto bidi_class(const char32_t codepoint) noexcept -> BidiClass
constexpr auto is_utf8_tail(const unsigned char lead, const std::size_t position, const unsigned char byte) -> bool
Definition unicode.h:248
UnicodeScript
Definition unicode_ucd.h:252
SOURCEMETA_CORE_UNICODE_EXPORT auto joining_type(const char32_t codepoint) noexcept -> JoiningType
constexpr auto utf8_decode(const std::string_view input, const std::string_view::size_type position) -> std::optional< std::pair< char32_t, std::size_t > >
Definition unicode.h:922
BidiClass
Definition unicode_ucd.h:59
constexpr auto is_utf8_continuation(const unsigned char byte) -> bool
Definition unicode.h:350
SOURCEMETA_CORE_UNICODE_EXPORT auto general_category(const char32_t codepoint) noexcept -> GeneralCategory
SOURCEMETA_CORE_UNICODE_EXPORT auto script(const char32_t codepoint) noexcept -> UnicodeScript
constexpr auto utf8_codepoint_length(const std::string_view input, const std::string_view::size_type position) -> std::size_t
Definition unicode.h:849
SOURCEMETA_CORE_UNICODE_EXPORT auto is_combining_mark(const char32_t codepoint) noexcept -> bool
constexpr auto is_iprivate(const char32_t codepoint) -> bool
Definition unicode.h:520
constexpr auto is_valid_codepoint(const char32_t codepoint) -> bool
Definition unicode.h:465
GeneralCategory
Definition unicode_ucd.h:297
SOURCEMETA_CORE_UNICODE_EXPORT auto is_id_continue(const char32_t codepoint) noexcept -> bool
NFCQuickCheck
Definition unicode_ucd.h:315
SOURCEMETA_CORE_UNICODE_EXPORT auto is_id_start(const char32_t codepoint) noexcept -> bool
SOURCEMETA_CORE_UNICODE_EXPORT auto to_valid_utf8(const std::string_view input) -> std::string
constexpr auto utf8_codepoint_within(const std::string_view input, const std::size_t minimum, const std::size_t maximum) -> bool
Definition unicode.h:393
SOURCEMETA_CORE_UNICODE_EXPORT auto nfc_quick_check(const char32_t codepoint) noexcept -> NFCQuickCheck
SOURCEMETA_CORE_UNICODE_EXPORT auto combining_class(const char32_t codepoint) noexcept -> std::uint8_t
SOURCEMETA_CORE_UNICODE_EXPORT auto utf32_to_utf8(const std::u32string_view input) -> std::string
auto is_valid_utf8(const std::string_view input) noexcept -> bool
Definition unicode.h:305
SOURCEMETA_CORE_UNICODE_EXPORT auto canonical_composition(const char32_t starter, const char32_t combining) noexcept -> std::optional< char32_t >