diff --git a/examples/readme_examples.cpp b/examples/readme_examples.cpp index 4a41e4bd..4a0856b4 100644 --- a/examples/readme_examples.cpp +++ b/examples/readme_examples.cpp @@ -58,23 +58,23 @@ std::u8string as_char32_t_example() { return input_utf8; } -std::string enum_to_string(transcoding_error ec) { +std::string enum_to_string(utf_transcoding_error ec) { switch (ec) { - case transcoding_error::truncated_utf8_sequence: + case utf_transcoding_error::truncated_utf8_sequence: return "truncated_utf8_sequence"; - case transcoding_error::unpaired_high_surrogate: + case utf_transcoding_error::unpaired_high_surrogate: return "unpaired_high_surrogate"; - case transcoding_error::unpaired_low_surrogate: + case utf_transcoding_error::unpaired_low_surrogate: return "unpaired_low_surrogate"; - case transcoding_error::unexpected_utf8_continuation_byte: + case utf_transcoding_error::unexpected_utf8_continuation_byte: return "unexpected_utf8_continuation_byte"; - case transcoding_error::overlong: + case utf_transcoding_error::overlong: return "overlong"; - case transcoding_error::encoded_surrogate: + case utf_transcoding_error::encoded_surrogate: return "encoded_surrogate"; - case transcoding_error::out_of_range: + case utf_transcoding_error::out_of_range: return "out_of_range"; - case transcoding_error::invalid_utf8_leading_byte: + case utf_transcoding_error::invalid_utf8_leading_byte: return "invalid_utf8_leading_byte"; } std::unreachable(); diff --git a/include/beman/utf_view/to_utf_view.hpp b/include/beman/utf_view/to_utf_view.hpp index 7d3db695..1ddd2bdf 100644 --- a/include/beman/utf_view/to_utf_view.hpp +++ b/include/beman/utf_view/to_utf_view.hpp @@ -90,7 +90,7 @@ template using exposition_only_bidirectional_at_most_t = decltype(exposition_only_bidirectional_at_most()); // @*exposition only*@ -enum class transcoding_error { +enum class utf_transcoding_error { truncated_utf8_sequence, unpaired_high_surrogate, unpaired_low_surrogate, @@ -277,11 +277,11 @@ class exposition_only_to_utf_view_impl return std::move(*this).curr(); } - /* PAPER: constexpr expected success() const; */ + /* PAPER: constexpr expected success() const; */ /* !PAPER */ - constexpr std::expected success() const { + constexpr std::expected success() const { return success_; } @@ -411,7 +411,7 @@ class exposition_only_to_utf_view_impl struct decode_code_point_result { char32_t c; std::uint8_t to_incr; - std::expected success; + std::expected success; }; template @@ -450,9 +450,9 @@ class exposition_only_to_utf_view_impl ++it; const std::uint8_t lo_bound = 0x80, hi_bound = 0xBF; std::uint8_t to_incr = 1; - std::expected success{}; + std::expected success{}; - auto const error{[&](transcoding_error const error_enum_in) { + auto const error{[&](utf_transcoding_error const error_enum_in) { success = std::unexpected{error_enum_in}; c = U'\uFFFD'; }}; @@ -460,18 +460,18 @@ class exposition_only_to_utf_view_impl if (u <= 0x7F) [[likely]] // 0x00 to 0x7F c = u; else if (u < 0xC0) [[unlikely]] { - error(transcoding_error::unexpected_utf8_continuation_byte); + error(utf_transcoding_error::unexpected_utf8_continuation_byte); } else if (u < 0xC2 || u > 0xF4) [[unlikely]] { - error(transcoding_error::invalid_utf8_leading_byte); + error(utf_transcoding_error::invalid_utf8_leading_byte); } else if (it == last) [[unlikely]] { - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); } else if (u <= 0xDF) // 0xC2 to 0xDF { c = u & 0x1F; u = *it; if (u < lo_bound || u > hi_bound) [[unlikely]] - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); else { c = (c << 6) | (u & 0x3F); ++it; @@ -484,21 +484,21 @@ class exposition_only_to_utf_view_impl u = *it; if (orig == 0xE0 && 0x80 <= u && u < 0xA0) [[unlikely]] - error(transcoding_error::overlong); + error(utf_transcoding_error::overlong); else if (orig == 0xED && 0xA0 <= u && u < 0xC0) [[unlikely]] - error(transcoding_error::encoded_surrogate); + error(utf_transcoding_error::encoded_surrogate); else if (u < lo_bound || u > hi_bound) [[unlikely]] - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); else if (++it == last) { [[unlikely]]++ to_incr; - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); } else { ++to_incr; c = (c << 6) | (u & 0x3F); u = *it; if (u < lo_bound || u > hi_bound) [[unlikely]] - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); else { c = (c << 6) | (u & 0x3F); ++it; @@ -512,31 +512,31 @@ class exposition_only_to_utf_view_impl u = *it; if (orig == 0xF0 && 0x80 <= u && u < 0x90) [[unlikely]] - error(transcoding_error::overlong); + error(utf_transcoding_error::overlong); else if (orig == 0xF4 && 0x90 <= u && u < 0xC0) [[unlikely]] - error(transcoding_error::out_of_range); + error(utf_transcoding_error::out_of_range); else if (u < lo_bound || u > hi_bound) [[unlikely]] - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); else if (++it == last) { [[unlikely]]++ to_incr; - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); } else { ++to_incr; c = (c << 6) | (u & 0x3F); u = *it; if (u < lo_bound || u > hi_bound) [[unlikely]] - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); else if (++it == last) { [[unlikely]]++ to_incr; - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); } else { ++to_incr; c = (c << 6) | (u & 0x3F); u = *it; if (u < lo_bound || u > hi_bound) [[unlikely]] - error(transcoding_error::truncated_utf8_sequence); + error(utf_transcoding_error::truncated_utf8_sequence); else { c = (c << 6) | (u & 0x3F); ++it; @@ -560,9 +560,9 @@ class exposition_only_to_utf_view_impl std::uint16_t u = *it; ++it; std::uint8_t to_incr = 1; - std::expected success{}; + std::expected success{}; - auto const error{[&](transcoding_error const error_enum_in) { + auto const error{[&](utf_transcoding_error const error_enum_in) { success = std::unexpected{error_enum_in}; c = U'\uFFFD'; }}; @@ -571,11 +571,11 @@ class exposition_only_to_utf_view_impl c = u; else if (u < 0xDC00) { if (it == last) [[unlikely]] { - error(transcoding_error::unpaired_high_surrogate); + error(utf_transcoding_error::unpaired_high_surrogate); } else { std::uint16_t u2 = *it; if (u2 < 0xDC00 || u2 > 0xDFFF) [[unlikely]] - error(transcoding_error::unpaired_high_surrogate); + error(utf_transcoding_error::unpaired_high_surrogate); else { ++it; to_incr = 2; @@ -585,7 +585,7 @@ class exposition_only_to_utf_view_impl } } } else - error(transcoding_error::unpaired_low_surrogate); + error(utf_transcoding_error::unpaired_low_surrogate); return {.c{c}, .to_incr{to_incr}, .success{success}}; } @@ -598,18 +598,18 @@ class exposition_only_to_utf_view_impl static constexpr decode_code_point_result decode_code_point_utf32_impl( exposition_only_innermost_iter& it) { char32_t c = *it; - std::expected success{}; + std::expected success{}; ++it; - auto const error{[&](transcoding_error const error_enum_in) { + auto const error{[&](utf_transcoding_error const error_enum_in) { success = std::unexpected{error_enum_in}; c = U'\uFFFD'; }}; if (c >= 0xD800) { if (c < 0xE000) { - error(transcoding_error::encoded_surrogate); + error(utf_transcoding_error::encoded_surrogate); } if (c > 0x10FFFF) { - error(transcoding_error::out_of_range); + error(utf_transcoding_error::out_of_range); } } return {.c{c}, .to_incr{1}, .success{success}}; @@ -710,7 +710,7 @@ class exposition_only_to_utf_view_impl .decode_result{.c{U'\uFFFD'}, .to_incr{1}, .success{std::unexpected{ - transcoding_error::unexpected_utf8_continuation_byte}}}, + utf_transcoding_error::unexpected_utf8_continuation_byte}}}, .new_curr{new_curr}}; } else if (detail::is_ascii(*it) || detail::lead_code_unit(*it)) { int const expected_reversed{detail::utf8_code_units(*it)}; @@ -722,7 +722,7 @@ class exposition_only_to_utf_view_impl .decode_result{.c{U'\uFFFD'}, .to_incr{1}, .success{std::unexpected{ - transcoding_error::unexpected_utf8_continuation_byte}}}, + utf_transcoding_error::unexpected_utf8_continuation_byte}}}, .new_curr{new_curr}}; } else { auto lead{it}; @@ -730,7 +730,7 @@ class exposition_only_to_utf_view_impl decode_code_point_utf8_impl(it, end())}; if (decode_result.success || decode_result.success == - std::unexpected{transcoding_error::truncated_utf8_sequence}) { + std::unexpected{utf_transcoding_error::truncated_utf8_sequence}) { assert(decode_result.to_incr == reversed); return {.decode_result{decode_result}, .new_curr{lead}}; } else { @@ -743,7 +743,7 @@ class exposition_only_to_utf_view_impl .success{ reversed == 1 ? decode_result.success - : std::unexpected{transcoding_error:: + : std::unexpected{utf_transcoding_error:: unexpected_utf8_continuation_byte}}}, .new_curr{new_curr}}; } @@ -758,8 +758,8 @@ class exposition_only_to_utf_view_impl .to_incr{1}, .success{ reversed == 1 - ? std::unexpected{transcoding_error::invalid_utf8_leading_byte} - : std::unexpected{transcoding_error:: + ? std::unexpected{utf_transcoding_error::invalid_utf8_leading_byte} + : std::unexpected{utf_transcoding_error:: unexpected_utf8_continuation_byte}}}, .new_curr{it}}; } @@ -774,14 +774,14 @@ class exposition_only_to_utf_view_impl return {.decode_result{.c{U'\uFFFD'}, .to_incr{1}, .success{std::unexpected{ - transcoding_error::unpaired_high_surrogate}}}, + utf_transcoding_error::unpaired_high_surrogate}}}, .new_curr{it}}; } else if (detail::low_surrogate(*it)) { if (it == begin()) { return {.decode_result{.c{U'\uFFFD'}, .to_incr{1}, .success{std::unexpected{ - transcoding_error::unpaired_low_surrogate}}}, + utf_transcoding_error::unpaired_low_surrogate}}}, .new_curr{it}}; } else { --it; @@ -795,7 +795,7 @@ class exposition_only_to_utf_view_impl return {.decode_result{.c{U'\uFFFD'}, .to_incr{1}, .success{std::unexpected{ - transcoding_error::unpaired_low_surrogate}}}, + utf_transcoding_error::unpaired_low_surrogate}}}, .new_curr{new_curr}}; } } @@ -866,7 +866,7 @@ class exposition_only_to_utf_view_impl std::uint8_t to_increment_ = 0; // @*exposition only*@ /* !PAPER */ - std::expected success_{}; // @*exposition only*@ + std::expected success_{}; // @*exposition only*@ /* PAPER */ }; diff --git a/paper/P2728.md b/paper/P2728.md index 5b4bea3e..1ccd9876 100644 --- a/paper/P2728.md +++ b/paper/P2728.md @@ -27,7 +27,7 @@ monofont: "DejaVu Sans Mono" - Remove all make-functions. - Replace the misbegotten `as_utfN()` functions with the `as_utfN` view adaptors that should have been there all along. -- Add missing `transcoding_error_handler` concept. +- Add missing `utf_transcoding_error_handler` concept. - Turn `unpack_iterator_and_sentinel` into a CPO. - Lower the UTF iterator concepts from bidirectional to input. @@ -98,8 +98,8 @@ monofont: "DejaVu Sans Mono" implement unpacking for user-defined UTF iterators. - Remove `std::uc::format`. - Make all concepts exposition-only. -- Remove `transcoding_error_handler` mechanism. -- Introduce new error handling mechanism based on a new `transcoding_error` +- Remove `utf_transcoding_error_handler` mechanism. +- Introduce new error handling mechanism based on a new `utf_transcoding_error` enumeration which is returned by an `success()` member function of the transcoding view's iterator. - Remove ability to pass pointers to range adaptor closure objects, which @@ -442,8 +442,8 @@ The UTF transcoding views in this paper provide such a basis operation by adding an `success()` member function to the iterator of the transcoding view, which informs users whether the current code point is a U+FFFD that was inserted in response to an invalid code unit sequence. The `success()` member -function returns a `std::expected`, where -`std::transcoding_error` is a new enum class containing enumerators for +function returns a `std::expected`, where +`std::utf_transcoding_error` is a new enum class containing enumerators for every category of transcoding error. Users who choose not to implement error handling will simply sanitize any @@ -457,12 +457,12 @@ by wrapping the iterator or by iterating with a traditional for loop: - Producing an error log message - Collecting statistics on transcoding errors - Implementing a custom transcoding view whose `value_type` is - `std::expected` + `std::expected` ### Why `std::expected`? The main alternative to consider here would be to specify that -default-constructed `std::transcoding_error` values represent success, or +default-constructed `std::utf_transcoding_error` values represent success, or add a `success` enumerator whose value is zero. There is precedent for doing this in the standard in the error handling approach of `std::from_chars`, which returns a `std::from_chars_result` containing a `std::errc` that has an @@ -521,7 +521,7 @@ but not vice versa. See [Appendix: Implementing Existing Practice for Error Handling](#appendix-implementing-existing-practice-for-error-handling) for code examples which demonstrate this. -### `std::transcoding_error` enumerators +### `std::utf_transcoding_error` enumerators - `truncated_utf8_sequence` - An ill-formed subsequence that matches the beginning of some well-formed @@ -4863,7 +4863,7 @@ namespace std::ranges { ```c++ namespace std::ranges { - enum class transcoding_error { + enum class utf_transcoding_error { truncated_utf8_sequence, unpaired_high_surrogate, unpaired_low_surrogate, @@ -4974,7 +4974,7 @@ namespace std::ranges { constexpr @*iter*@ base() && requires (!forward_iterator<@*innermost-iter*@>) { return move(*this).curr(); } - constexpr expected success() const; + constexpr expected success() const; constexpr value_type operator*() const; @@ -5379,13 +5379,13 @@ In the following paragraph, `@*utf-error(foo)*@` refers to the result of the exposition-only function: ```cpp -expected @*utf-error-func*@(transcoding_error err) { +expected @*utf-error-func*@(utf_transcoding_error err) { return unexpected{err}; } ``` When the `@*utf-iterator*@` is at the end of the underlying range, `success()` -returns a default-constructed `expected`. When the +returns a default-constructed `expected`. When the `@*utf-iterator*@` has a code unit, derived from a code point `c`, which is itself derived from a particular input subsequence (the "current input subsequence"), the result of the `success()` method corresponds to the @@ -5395,46 +5395,46 @@ values of code units below are inclusive.) - If the encoding corresponding to `@*from-type*@` is UTF-8: - If the current input subsequence is valid UTF-8, `success()` returns - `expected{}`. + `expected{}`. - If the current input subsequence is a code unit between 0x80 and 0xBF, `success()` returns - `@*utf-error*@(transcoding_error::unexpected_utf8_continuation_byte)`. + `@*utf-error*@(utf_transcoding_error::unexpected_utf8_continuation_byte)`. - If the current input subsequence is a code unit between 0xC0 and 0xC2, or between 0xF5 and 0xFF, `success()` returns - `@*utf-error*@(transcoding_error::invalid_utf8_leading_byte)`. + `@*utf-error*@(utf_transcoding_error::invalid_utf8_leading_byte)`. - If the current input subsequence is 0xE0, and the subsequent input subsequence is between 0x80 and 0x9F; or if the current input subsequence is 0xF0, and the subsequent input subsequence is between 0x80 and 0x8F; - then `success()` returns `@*utf-error*@(transcoding_error::overlong)`. + then `success()` returns `@*utf-error*@(utf_transcoding_error::overlong)`. - If the current input subsequence is 0xED, and the subsequent input subsequence is between 0xA0 and 0xBF, then `success()` returns - `@*utf-error*@(transcoding_error::encoded_surrogate)`. + `@*utf-error*@(utf_transcoding_error::encoded_surrogate)`. - If the the current input subsequence is 0xF4, and the subsequent input subsequence is between 0x90 and 0xBF, then `success()` returns - `@*utf-error*@(transcoding_error::out_of_range)` + `@*utf-error*@(utf_transcoding_error::out_of_range)` - Otherwise, if the current input subsequence is invalid UTF-8, begins with a code unit between 0xC2 and 0xF4, and there exists some hypothetical sequence of code units which would make the current input subsequence well-formed if concatenated to the end of it, `success()` returns - `@*utf-error*@(transcoding_error::truncated_utf8_sequence)`. + `@*utf-error*@(utf_transcoding_error::truncated_utf8_sequence)`. - If the encoding corresponding to `@*from-type*@` is UTF-16: - If the current input subsequence is valid UTF-16, `success()` returns - `expected{}`. + `expected{}`. - If the current input subsequence is between 0xD800 and 0xDBFF, `success()` - returns `@*utf-error*@(transcoding_error::unpaired_high_surrogate)`. + returns `@*utf-error*@(utf_transcoding_error::unpaired_high_surrogate)`. - If the current input subsequence is between 0xDC00 and 0xDFFF, `success()` - returns `@*utf-error*@(transcoding_error::unpaired_low_surrogate)`. + returns `@*utf-error*@(utf_transcoding_error::unpaired_low_surrogate)`. - If the encoding corresponding to `@*from-type*@` is UTF-32: - If the current input subsequence is valid UTF-32, `success()` returns - `expected{}`. + `expected{}`. - If the current input subsequence is between 0xD800 and 0xDFFF, `success()` - returns `@*utf-error*@(transcoding_error::encoded_surrogate)`. + returns `@*utf-error*@(utf_transcoding_error::encoded_surrogate)`. - If the current input subsequence is between 0x110000 and 0xFFFFFFFF, - `success()` returns `@*utf-error*@(transcoding_error::out_of_range)`. + `success()` returns `@*utf-error*@(utf_transcoding_error::out_of_range)`. `@*utf-iterator*@`'s exposition-only type alias `@*innermost-iter*@` is `@*iter*@::@*innermost-iter*@` if `@*iter*@` is @@ -5748,20 +5748,20 @@ size_t iconv(iconv_t cd, const char** inbuf, size_t* inbytesleft, char** outbuf, *inbytesleft -= bytes_converted; *inbuf = it.base(); } else { - transcoding_error e = it.success().error(); + utf_transcoding_error e = it.success().error(); switch (e) { - case transcoding_error::truncated_utf8_sequence: { + case utf_transcoding_error::truncated_utf8_sequence: { errno = EINVAL; } break; - case transcoding_error::unexpected_utf8_continuation_byte: - case transcoding_error::overlong: - case transcoding_error::encoded_surrogate: - case transcoding_error::out_of_range: - case transcoding_error::invalid_utf8_leading_byte: { + case utf_transcoding_error::unexpected_utf8_continuation_byte: + case utf_transcoding_error::overlong: + case utf_transcoding_error::encoded_surrogate: + case utf_transcoding_error::out_of_range: + case utf_transcoding_error::invalid_utf8_leading_byte: { errno = EILSEQ; } break; - case transcoding_error::unpaired_high_surrogate: - case transcoding_error::unpaired_low_surrogate: { + case utf_transcoding_error::unpaired_high_surrogate: + case utf_transcoding_error::unpaired_low_surrogate: { std::unreachable(); } } @@ -5972,23 +5972,23 @@ std::basic_string decode(std::basic_string_view input) { ss << ": "; ss << [&] { switch (it.success().error()) { - case transcoding_error::truncated_utf8_sequence: + case utf_transcoding_error::truncated_utf8_sequence: return "unexpected end of data"; - case transcoding_error::unpaired_high_surrogate: - case transcoding_error::unpaired_low_surrogate: + case utf_transcoding_error::unpaired_high_surrogate: + case utf_transcoding_error::unpaired_low_surrogate: return "illegal UTF-16 surrogate"; - case transcoding_error::unexpected_utf8_continuation_byte: - case transcoding_error::invalid_utf8_leading_byte: + case utf_transcoding_error::unexpected_utf8_continuation_byte: + case utf_transcoding_error::invalid_utf8_leading_byte: return "invalid start byte"; - case transcoding_error::encoded_surrogate: + case utf_transcoding_error::encoded_surrogate: if constexpr (std::same_as) { return "code point in surrogate code point range(0xd800, 0xe000)"; } - case transcoding_error::overlong: + case utf_transcoding_error::overlong: if constexpr (std::same_as) { return "code point not in range(0x110000)"; } - case transcoding_error::out_of_range: + case utf_transcoding_error::out_of_range: return "invalid continuation byte"; } std::unreachable(); diff --git a/tests/beman/utf_view/to_utf_view.t.cpp b/tests/beman/utf_view/to_utf_view.t.cpp index 497cb726..8f5f2d40 100644 --- a/tests/beman/utf_view/to_utf_view.t.cpp +++ b/tests/beman/utf_view/to_utf_view.t.cpp @@ -78,7 +78,7 @@ static_assert( template struct test_case_code_unit_result { CharT code_unit; - std::expected success; + std::expected success; }; template @@ -93,19 +93,19 @@ CONSTEXPR_UNLESS_MSVC test_case table3_8{ static_cast('\xbf'), static_cast('\xf0'), static_cast('\x81'), static_cast('\x82'), static_cast('A')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::invalid_utf8_leading_byte}}, + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::invalid_utf8_leading_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, - {U'\uFFFD', std::unexpected{transcoding_error::overlong}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::overlong}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, - {U'\uFFFD', std::unexpected{transcoding_error::overlong}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::overlong}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'A', {}}}}; CONSTEXPR_UNLESS_MSVC test_case table3_9{ @@ -114,19 +114,19 @@ CONSTEXPR_UNLESS_MSVC test_case table3_9{ static_cast('\xbf'), static_cast('\xbf'), static_cast('\xed'), static_cast('\xaf'), static_cast('A')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::encoded_surrogate}}, + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::encoded_surrogate}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, - {U'\uFFFD', std::unexpected{transcoding_error::encoded_surrogate}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::encoded_surrogate}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, - {U'\uFFFD', std::unexpected{transcoding_error::encoded_surrogate}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::encoded_surrogate}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'A', {}}}}; CONSTEXPR_UNLESS_MSVC test_case table3_10{ @@ -135,19 +135,19 @@ CONSTEXPR_UNLESS_MSVC test_case table3_10{ static_cast('\xff'), static_cast('\x41'), static_cast('\x80'), static_cast('\xbf'), static_cast('B')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::out_of_range}}, + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::out_of_range}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, - {U'\uFFFD', std::unexpected{transcoding_error::invalid_utf8_leading_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::invalid_utf8_leading_byte}}, {U'A', {}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}, + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}, {U'B', {}}}}; CONSTEXPR_UNLESS_MSVC test_case table3_11{ @@ -156,10 +156,10 @@ CONSTEXPR_UNLESS_MSVC test_case table3_11{ static_cast('\x91'), static_cast('\x92'), static_cast('\xf1'), static_cast('\xbf'), static_cast('A')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}, - {U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}, - {U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}, - {U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}, + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}, + {U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}, {U'A', {}}}}; CONSTEXPR_UNLESS_MSVC test_case two_byte{ @@ -168,11 +168,11 @@ CONSTEXPR_UNLESS_MSVC test_case two_byte{ CONSTEXPR_UNLESS_MSVC test_case two_byte_truncated_by_end{ .input{static_cast('\xc2')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}}}; CONSTEXPR_UNLESS_MSVC test_case two_byte_truncated_by_non_continuation{ .input{static_cast('\xc2'), static_cast('A')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}, + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}, {U'A', {}}}}; CONSTEXPR_UNLESS_MSVC test_case three_byte{ @@ -182,36 +182,36 @@ CONSTEXPR_UNLESS_MSVC test_case three_byte{ CONSTEXPR_UNLESS_MSVC test_case three_byte_truncated_at_third_by_end{ .input{static_cast('\xe4'), static_cast('\xba')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}}}; CONSTEXPR_UNLESS_MSVC test_case four_byte_truncated_at_second_by_non_continuation{ .input{static_cast('\xf0'), static_cast('A')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}, + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}, {U'A', {}}}}; CONSTEXPR_UNLESS_MSVC test_case four_byte_truncated_at_third_by_end{ .input{static_cast('\xf0'), static_cast('\x90')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}}}; CONSTEXPR_UNLESS_MSVC test_case four_byte_truncated_at_fourth_by_end{ .input{static_cast('\xf0'), static_cast('\x90'), static_cast('\x80')}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::truncated_utf8_sequence}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::truncated_utf8_sequence}}}}; CONSTEXPR_UNLESS_MSVC test_case single_high{ .input{u'\xD800'}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::unpaired_high_surrogate}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::unpaired_high_surrogate}}}}; CONSTEXPR_UNLESS_MSVC test_case surrogate_pair_truncated_by_non_low_surrogate{ .input{u'\xD800', u'A'}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::unpaired_high_surrogate}}, + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::unpaired_high_surrogate}}, {U'A', {}}}}; CONSTEXPR_UNLESS_MSVC test_case single_low{ .input{u'\xDC00'}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::unpaired_low_surrogate}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::unpaired_low_surrogate}}}}; CONSTEXPR_UNLESS_MSVC test_case surrogates{ .input{u'\xD800', u'\xDC00'}, .output{{U'\U00010000', {}}}}; @@ -221,11 +221,11 @@ CONSTEXPR_UNLESS_MSVC test_case nonsurrogates{ CONSTEXPR_UNLESS_MSVC test_case encoded_surrogate{ .input{U'\x0000DC00'}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::encoded_surrogate}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::encoded_surrogate}}}}; CONSTEXPR_UNLESS_MSVC test_case out_of_range{ .input{U'\x00110000'}, - .output{{U'\uFFFD', std::unexpected{transcoding_error::out_of_range}}}}; + .output{{U'\uFFFD', std::unexpected{utf_transcoding_error::out_of_range}}}}; CONSTEXPR_UNLESS_MSVC test_case valid_utf8_identity{ .input{static_cast('A'), static_cast('\xc2'), @@ -273,18 +273,18 @@ CONSTEXPR_UNLESS_MSVC test_case four_continuations{ static_cast('\x80')}, .output{{U'\x00010000', {}}, {U'\uFFFD', - std::unexpected{transcoding_error::unexpected_utf8_continuation_byte}}}}; + std::unexpected{utf_transcoding_error::unexpected_utf8_continuation_byte}}}}; CONSTEXPR_UNLESS_MSVC test_case two_low_surrogates{ .input{u'\xD800', u'\xDC00', u'\xDC00'}, .output{{U'\x00010000', {}}, - {U'\uFFFD', std::unexpected{transcoding_error::unpaired_low_surrogate}}}}; + {U'\uFFFD', std::unexpected{utf_transcoding_error::unpaired_low_surrogate}}}}; CONSTEXPR_UNLESS_MSVC test_case ff_at_end{ .input{static_cast('\xc3'), static_cast('\xa9'), static_cast('\xff')}, .output{{U'\u00E9', {}}, - {U'\uFFFD', std::unexpected{transcoding_error::invalid_utf8_leading_byte}}}}; + {U'\uFFFD', std::unexpected{utf_transcoding_error::invalid_utf8_leading_byte}}}}; template @@ -735,7 +735,7 @@ CONSTEXPR_UNLESS_MSVC bool wrapped_view_mid_code_point_test_impl() { } auto u8_begin{u8v.begin()}; if (u8_begin.success() != - std::unexpected{transcoding_error::unpaired_low_surrogate}) { + std::unexpected{utf_transcoding_error::unpaired_low_surrogate}) { return false; } if (base_testing == base_test::iterator_mid_code_point) { @@ -1015,20 +1015,20 @@ size_t iconv([[maybe_unused]] iconv_t cd, const char** inbuf, size_t* inbyteslef *inbytesleft -= bytes_converted; *inbuf = it.base(); } else { - transcoding_error e = it.success().error(); + utf_transcoding_error e = it.success().error(); switch (e) { - case transcoding_error::truncated_utf8_sequence: { + case utf_transcoding_error::truncated_utf8_sequence: { errno = EINVAL; } break; - case transcoding_error::unexpected_utf8_continuation_byte: - case transcoding_error::overlong: - case transcoding_error::encoded_surrogate: - case transcoding_error::out_of_range: - case transcoding_error::invalid_utf8_leading_byte: { + case utf_transcoding_error::unexpected_utf8_continuation_byte: + case utf_transcoding_error::overlong: + case utf_transcoding_error::encoded_surrogate: + case utf_transcoding_error::out_of_range: + case utf_transcoding_error::invalid_utf8_leading_byte: { errno = EILSEQ; } break; - case transcoding_error::unpaired_high_surrogate: - case transcoding_error::unpaired_low_surrogate: { + case utf_transcoding_error::unpaired_high_surrogate: + case utf_transcoding_error::unpaired_low_surrogate: { std::unreachable(); } } @@ -1468,23 +1468,23 @@ std::basic_string decode(std::basic_string_view input) { ss << ": "; ss << [&] { switch (it.success().error()) { - case transcoding_error::truncated_utf8_sequence: + case utf_transcoding_error::truncated_utf8_sequence: return "unexpected end of data"; - case transcoding_error::unpaired_high_surrogate: - case transcoding_error::unpaired_low_surrogate: + case utf_transcoding_error::unpaired_high_surrogate: + case utf_transcoding_error::unpaired_low_surrogate: return "illegal UTF-16 surrogate"; - case transcoding_error::unexpected_utf8_continuation_byte: - case transcoding_error::invalid_utf8_leading_byte: + case utf_transcoding_error::unexpected_utf8_continuation_byte: + case utf_transcoding_error::invalid_utf8_leading_byte: return "invalid start byte"; - case transcoding_error::encoded_surrogate: + case utf_transcoding_error::encoded_surrogate: if constexpr (std::same_as) { return "code point in surrogate code point range(0xd800, 0xe000)"; } - case transcoding_error::overlong: + case utf_transcoding_error::overlong: if constexpr (std::same_as) { return "code point not in range(0x110000)"; } - case transcoding_error::out_of_range: + case utf_transcoding_error::out_of_range: return "invalid continuation byte"; } std::unreachable();