1#ifndef SOURCEMETA_CORE_UNICODE_H_
2#define SOURCEMETA_CORE_UNICODE_H_
4#ifndef SOURCEMETA_CORE_UNICODE_EXPORT
5#include <sourcemeta/core/unicode_export.h>
8#include <sourcemeta/core/unicode_ucd.h>
29namespace sourcemeta::core {
42SOURCEMETA_CORE_UNICODE_EXPORT
59SOURCEMETA_CORE_UNICODE_EXPORT
75SOURCEMETA_CORE_UNICODE_EXPORT
92SOURCEMETA_CORE_UNICODE_EXPORT
93auto utf8_to_utf32(std::istream &input) -> std::optional<std::u32string>;
107SOURCEMETA_CORE_UNICODE_EXPORT
109 -> std::optional<std::u32string>;
123SOURCEMETA_CORE_UNICODE_EXPORT
145SOURCEMETA_CORE_UNICODE_EXPORT
163SOURCEMETA_CORE_UNICODE_EXPORT
177SOURCEMETA_CORE_UNICODE_EXPORT
190SOURCEMETA_CORE_UNICODE_EXPORT
214 if (
byte >= 0xC2 &&
byte <= 0xDF) {
217 if (
byte >= 0xE0 &&
byte <= 0xEF) {
220 if (
byte >= 0xF0 &&
byte <= 0xF4) {
249 const std::size_t position,
250 const unsigned char byte) ->
bool {
251 const unsigned char lower{
static_cast<unsigned char>(
253 : (lead == 0xE0 ? 0xA0 : (lead == 0xF0 ? 0x90 : 0x80)))};
254 const unsigned char upper{
static_cast<unsigned char>(
256 : (lead == 0xED ? 0x9F : (lead == 0xF4 ? 0x8F : 0xBF)))};
257 return byte >= lower &&
byte <= upper;
278 const auto lead{
static_cast<unsigned char>(input.front())};
280 if (size == 0 || input.size() < size) {
284 for (std::size_t index{1}; index < size; index += 1) {
285 if (!
is_utf8_tail(lead, index,
static_cast<unsigned char>(input[index]))) {
306 constexpr std::uint64_t HIGH_BITS{0x8080808080808080ULL};
307 const auto size{input.size()};
308 std::size_t position{0};
309 while (position < size) {
312 if (size - position >= 8) {
313 std::uint64_t word{0};
314 std::memcpy(&word, input.data() + position, 8);
315 if ((word & HIGH_BITS) == 0) {
321 if (
static_cast<unsigned char>(input[position]) < 0x80) {
351 return byte >= 0x80 &&
byte <= 0xBF;
368 std::size_t count{0};
369 for (
const auto byte : input) {
394 const std::size_t minimum,
395 const std::size_t maximum) ->
bool {
396 const auto bytes{input.size()};
400 if (bytes < minimum) {
406 const auto lower_bound{(bytes + 3) / 4};
407 if (lower_bound > maximum) {
413 if (bytes <= maximum && lower_bound >= minimum) {
420 std::size_t count{0};
421 for (
const auto byte : input) {
424 if (count > maximum) {
430 return count >= minimum;
448 return codepoint >= 0xD800 && codepoint <= 0xDFFF;
466 return codepoint <= 0x10FFFF && !
is_surrogate(codepoint);
485 if (codepoint >= 0xA0 && codepoint <= 0xD7FF) {
488 if (codepoint >= 0xF900 && codepoint <= 0xFDCF) {
491 if (codepoint >= 0xFDF0 && codepoint <= 0xFFEF) {
497 if (codepoint < 0x10000 || codepoint > 0xEFFFD) {
500 if (codepoint >= 0xE0000 && codepoint < 0xE1000) {
503 return (codepoint & 0xFFFFU) <= 0xFFFDU;
521 if (codepoint >= 0xE000 && codepoint <= 0xF8FF) {
524 if (codepoint >= 0xF0000 && codepoint <= 0xFFFFD) {
527 if (codepoint >= 0x100000 && codepoint <= 0x10FFFD) {
550 if (codepoint < 0x80) {
553 if (codepoint < 0x800) {
556 if (codepoint < 0x10000) {
575SOURCEMETA_CORE_UNICODE_EXPORT
594SOURCEMETA_CORE_UNICODE_EXPORT
613SOURCEMETA_CORE_UNICODE_EXPORT
632SOURCEMETA_CORE_UNICODE_EXPORT
652SOURCEMETA_CORE_UNICODE_EXPORT
669SOURCEMETA_CORE_UNICODE_EXPORT
685SOURCEMETA_CORE_UNICODE_EXPORT
701SOURCEMETA_CORE_UNICODE_EXPORT
720SOURCEMETA_CORE_UNICODE_EXPORT
740SOURCEMETA_CORE_UNICODE_EXPORT
742 -> std::u32string_view;
761SOURCEMETA_CORE_UNICODE_EXPORT
763 const char32_t combining)
noexcept
764 -> std::optional<char32_t>;
778SOURCEMETA_CORE_UNICODE_EXPORT
779auto nfc(
const std::u32string_view input) -> std::u32string;
795SOURCEMETA_CORE_UNICODE_EXPORT
796auto is_nfc(
const std::u32string_view input) -> bool;
815SOURCEMETA_CORE_UNICODE_EXPORT
816auto case_fold(
const char32_t codepoint)
noexcept -> std::u32string_view;
829SOURCEMETA_CORE_UNICODE_EXPORT
830auto case_fold(
const std::u32string_view input) -> std::u32string;
850 const std::string_view::size_type position)
852 if (position >= input.size()) {
855 const auto byte_0{
static_cast<unsigned char>(input[position])};
857 if (size == 0 || position + size > input.size()) {
867 const auto byte_1{
static_cast<unsigned char>(input[position + 1])};
868 bool byte_1_ok{
false};
871 }
else if (size == 3) {
872 if (byte_0 == 0xE0) {
873 byte_1_ok = byte_1 >= 0xA0 && byte_1 <= 0xBF;
874 }
else if (byte_0 == 0xED) {
875 byte_1_ok = byte_1 >= 0x80 && byte_1 <= 0x9F;
880 if (byte_0 == 0xF0) {
881 byte_1_ok = byte_1 >= 0x90 && byte_1 <= 0xBF;
882 }
else if (byte_0 == 0xF4) {
883 byte_1_ok = byte_1 >= 0x80 && byte_1 <= 0x8F;
895 for (std::size_t index{2}; index < size; ++index) {
897 static_cast<unsigned char>(input[position + index]))) {
923 const std::string_view::size_type position)
924 -> std::optional<std::pair<char32_t, std::size_t>> {
930 const auto lead{
static_cast<unsigned char>(input[position])};
931 char32_t codepoint{0};
933 codepoint =
static_cast<char32_t>(lead);
934 }
else if (size == 2) {
935 codepoint =
static_cast<char32_t>(lead & 0x1FU);
936 }
else if (size == 3) {
937 codepoint =
static_cast<char32_t>(lead & 0x0FU);
939 codepoint =
static_cast<char32_t>(lead & 0x07U);
942 for (std::size_t index{1}; index < size; ++index) {
943 const auto continuation{
944 static_cast<unsigned char>(input[position + index])};
945 codepoint = (codepoint << 6) | static_cast<char32_t>(continuation & 0x3FU);
948 return std::make_pair(codepoint, size);
constexpr auto utf8_lead_byte_size(const unsigned char byte) -> std::uint8_t
Definition unicode.h:210
SOURCEMETA_CORE_UNICODE_EXPORT auto case_fold(const char32_t codepoint) noexcept -> std::u32string_view
SOURCEMETA_CORE_UNICODE_EXPORT auto nfc(const std::u32string_view input) -> std::u32string
SOURCEMETA_CORE_UNICODE_EXPORT auto utf8_to_wide(const std::string_view input) -> std::wstring
constexpr auto is_ucschar(const char32_t codepoint) -> bool
Definition unicode.h:484
SOURCEMETA_CORE_UNICODE_EXPORT auto codepoint_to_utf8(const char32_t codepoint) -> std::string
constexpr auto utf8_codepoint_count(const std::string_view input) -> std::size_t
Definition unicode.h:366
constexpr auto utf8_sequence_size(const std::string_view input) -> std::size_t
Definition unicode.h:273
SOURCEMETA_CORE_UNICODE_EXPORT auto canonical_decomposition(const char32_t codepoint) noexcept -> std::u32string_view
SOURCEMETA_CORE_UNICODE_EXPORT auto utf32_to_utf8_lenient(const std::u32string_view input) -> std::string
JoiningType
Definition unicode_ucd.h:21
constexpr auto is_surrogate(const char32_t codepoint) -> bool
Definition unicode.h:447
SOURCEMETA_CORE_UNICODE_EXPORT auto is_nfc(const std::u32string_view input) -> bool
SOURCEMETA_CORE_UNICODE_EXPORT auto wide_to_utf8(const std::wstring_view input) -> std::string
constexpr auto utf8_codepoint_byte_count(const char32_t codepoint) -> std::uint8_t
Definition unicode.h:548
SOURCEMETA_CORE_UNICODE_EXPORT auto utf8_to_utf32(std::istream &input) -> std::optional< std::u32string >
SOURCEMETA_CORE_UNICODE_EXPORT auto bidi_class(const char32_t codepoint) noexcept -> BidiClass
constexpr auto is_utf8_tail(const unsigned char lead, const std::size_t position, const unsigned char byte) -> bool
Definition unicode.h:248
UnicodeScript
Definition unicode_ucd.h:252
SOURCEMETA_CORE_UNICODE_EXPORT auto joining_type(const char32_t codepoint) noexcept -> JoiningType
constexpr auto utf8_decode(const std::string_view input, const std::string_view::size_type position) -> std::optional< std::pair< char32_t, std::size_t > >
Definition unicode.h:922
BidiClass
Definition unicode_ucd.h:59
constexpr auto is_utf8_continuation(const unsigned char byte) -> bool
Definition unicode.h:350
SOURCEMETA_CORE_UNICODE_EXPORT auto general_category(const char32_t codepoint) noexcept -> GeneralCategory
SOURCEMETA_CORE_UNICODE_EXPORT auto script(const char32_t codepoint) noexcept -> UnicodeScript
constexpr auto utf8_codepoint_length(const std::string_view input, const std::string_view::size_type position) -> std::size_t
Definition unicode.h:849
SOURCEMETA_CORE_UNICODE_EXPORT auto is_combining_mark(const char32_t codepoint) noexcept -> bool
constexpr auto is_iprivate(const char32_t codepoint) -> bool
Definition unicode.h:520
constexpr auto is_valid_codepoint(const char32_t codepoint) -> bool
Definition unicode.h:465
GeneralCategory
Definition unicode_ucd.h:297
SOURCEMETA_CORE_UNICODE_EXPORT auto is_id_continue(const char32_t codepoint) noexcept -> bool
NFCQuickCheck
Definition unicode_ucd.h:315
SOURCEMETA_CORE_UNICODE_EXPORT auto is_id_start(const char32_t codepoint) noexcept -> bool
SOURCEMETA_CORE_UNICODE_EXPORT auto to_valid_utf8(const std::string_view input) -> std::string
constexpr auto utf8_codepoint_within(const std::string_view input, const std::size_t minimum, const std::size_t maximum) -> bool
Definition unicode.h:393
SOURCEMETA_CORE_UNICODE_EXPORT auto nfc_quick_check(const char32_t codepoint) noexcept -> NFCQuickCheck
SOURCEMETA_CORE_UNICODE_EXPORT auto combining_class(const char32_t codepoint) noexcept -> std::uint8_t
SOURCEMETA_CORE_UNICODE_EXPORT auto utf32_to_utf8(const std::u32string_view input) -> std::string
auto is_valid_utf8(const std::string_view input) noexcept -> bool
Definition unicode.h:305
SOURCEMETA_CORE_UNICODE_EXPORT auto canonical_composition(const char32_t starter, const char32_t combining) noexcept -> std::optional< char32_t >