From 0d11df08885b9011f37ef12f1a989f778185da35 Mon Sep 17 00:00:00 2001 From: Charlie Vieth Date: Sat, 18 Apr 2026 11:33:36 -0600 Subject: [PATCH] deps: update bundled library version from v8.0.0 to v8.2.0 Update the bundled simdutf library from version v8.0.0 to v8.2.0. --- README.md | 2 +- SIMDUTF_VERSION | 2 +- simdutf.cpp | 1400 ++++++++++++++++++++++++++++++++++++----------- simdutf.h | 374 ++++++++++++- 4 files changed, 1459 insertions(+), 319 deletions(-) diff --git a/README.md b/README.md index bf6341e..8e6517d 100644 --- a/README.md +++ b/README.md @@ -16,7 +16,7 @@ simdutf library build this library with the `libsimdutf` build tag. ## simdutf version This library bundles [simdutf](https://github.com/simdutf/simdutf/) version -[v8.0.0](https://github.com/simdutf/simdutf/releases/tag/v8.0.0). +[v8.2.0](https://github.com/simdutf/simdutf/releases/tag/v8.2.0). The [SIMDUTF_VERSION](./SIMDUTF_VERSION) file contains the current version of the bundled simdutf version. diff --git a/SIMDUTF_VERSION b/SIMDUTF_VERSION index 5f4f91f..7c330f2 100644 --- a/SIMDUTF_VERSION +++ b/SIMDUTF_VERSION @@ -1 +1 @@ -v8.0.0 +v8.2.0 diff --git a/simdutf.cpp b/simdutf.cpp index e11ac8b..07b847f 100644 --- a/simdutf.cpp +++ b/simdutf.cpp @@ -1,6 +1,6 @@ //go:build !libsimdutf -/* auto-generated on 2026-01-13 09:03:21 +0100. Do not edit! */ +/* auto-generated on 2026-03-12 20:42:59 -0400. Do not edit! */ /* begin file src/simdutf.cpp */ #include "simdutf.h" @@ -2007,6 +2007,15 @@ class implementation final : public simdutf::implementation { simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 simdutf_warn_unused size_t utf8_length_from_utf32( @@ -2057,6 +2066,10 @@ class implementation final : public simdutf::implementation { char character) const noexcept override; const char16_t *find(const char16_t *start, const char16_t *end, char16_t character) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char *input, size_t length) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_BASE64 }; @@ -2633,6 +2646,12 @@ template struct simd8x64 { this->chunks[2] > mask, this->chunks[3] > mask) .to_bitmask(); } + simdutf_really_inline uint64_t gteq(const T m) const { + const simd8 mask = simd8::splat(m); + return simd8x64(this->chunks[0] >= mask, this->chunks[1] >= mask, + this->chunks[2] >= mask, this->chunks[3] >= mask) + .to_bitmask(); + } simdutf_really_inline uint64_t gteq_unsigned(const uint8_t m) const { const simd8 mask = simd8::splat(m); return simd8x64(simd8(uint8x16_t(this->chunks[0])) >= mask, @@ -3002,7 +3021,12 @@ template struct simd16x32 { this->chunks[2] > mask, this->chunks[3] > mask) .to_bitmask(); } - + simdutf_really_inline uint64_t gteq(const T m) const { + const simd16 mask = simd16::splat(m); + return simd16x32(this->chunks[0] >= mask, this->chunks[1] >= mask, + this->chunks[2] >= mask, this->chunks[3] >= mask) + .to_bitmask(); + } simdutf_really_inline uint64_t lteq(const T m) const { const simd16 mask = simd16::splat(m); return simd16x32(this->chunks[0] <= mask, this->chunks[1] <= mask, @@ -3703,6 +3727,15 @@ class implementation final : public simdutf::implementation { simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -3759,6 +3792,10 @@ class implementation final : public simdutf::implementation { char character) const noexcept override; const char16_t *find(const char16_t *start, const char16_t *end, char16_t character) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char *input, size_t length) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_BASE64 }; @@ -4321,6 +4358,15 @@ class implementation final : public simdutf::implementation { simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -4377,6 +4423,10 @@ class implementation final : public simdutf::implementation { char character) const noexcept override; const char16_t *find(const char16_t *start, const char16_t *end, char16_t character) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char *input, size_t length) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_BASE64 }; @@ -5611,6 +5661,15 @@ class implementation final : public simdutf::implementation { simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -5667,6 +5726,10 @@ class implementation final : public simdutf::implementation { char character) const noexcept override; const char16_t *find(const char16_t *start, const char16_t *end, char16_t character) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char *input, size_t length) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_BASE64 }; @@ -6110,7 +6173,12 @@ template struct simd8x64 { this->chunks[2] > mask, this->chunks[3] > mask) .to_bitmask(); } - + simdutf_really_inline uint64_t gteq(const T m) const { + const simd8 mask = simd8::splat(m); + return simd8x64(this->chunks[0] >= mask, this->chunks[1] >= mask, + this->chunks[2] >= mask, this->chunks[3] >= mask) + .to_bitmask(); + } simdutf_really_inline uint64_t eq(const T m) const { const simd8 mask = simd8::splat(m); return simd8x64(this->chunks[0] == mask, this->chunks[1] == mask, @@ -6342,6 +6410,13 @@ template struct simd16x32 { .to_bitmask(); } + simdutf_really_inline uint64_t gteq(const T m) const { + const simd16 mask = simd16::splat(m); + return simd16x32(this->chunks[0] >= mask, this->chunks[1] >= mask, + this->chunks[2] >= mask, this->chunks[3] >= mask) + .to_bitmask(); + } + simdutf_really_inline uint64_t eq(const T m) const { const simd16 mask = simd16::splat(m); return simd16x32(this->chunks[0] == mask, this->chunks[1] == mask, @@ -6835,40 +6910,49 @@ class implementation final : public simdutf::implementation { #if SIMDUTF_FEATURE_UTF16 void change_endianness_utf16(const char16_t *buf, size_t length, char16_t *output) const noexcept final; - simdutf_warn_unused size_t count_utf16le(const char16_t *buf, - size_t length) const noexcept; - simdutf_warn_unused size_t count_utf16be(const char16_t *buf, - size_t length) const noexcept; + simdutf_warn_unused size_t + count_utf16le(const char16_t *buf, size_t length) const noexcept override; + simdutf_warn_unused size_t + count_utf16be(const char16_t *buf, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 simdutf_warn_unused size_t count_utf8(const char *buf, - size_t length) const noexcept; + size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 #if SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf8_length_from_utf16le(const char16_t *input, size_t length) const noexcept; - simdutf_warn_unused size_t - utf8_length_from_utf16be(const char16_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_utf16le( + const char16_t *input, size_t length) const noexcept override; + simdutf_warn_unused size_t utf8_length_from_utf16be( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 simdutf_warn_unused size_t utf32_length_from_utf16le( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; simdutf_warn_unused size_t utf32_length_from_utf16be( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf16_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf16_length_from_utf8( + const char *input, size_t length) const noexcept override; simdutf_warn_unused result utf8_length_from_utf16le_with_replacement( const char16_t *input, size_t length) const noexcept; ; simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( const char16_t *input, size_t length) const noexcept; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -8631,86 +8715,96 @@ class implementation final : public simdutf::implementation { #if SIMDUTF_FEATURE_UTF16 void change_endianness_utf16(const char16_t *buf, size_t length, char16_t *output) const noexcept final; - simdutf_warn_unused size_t count_utf16le(const char16_t *buf, - size_t length) const noexcept; - simdutf_warn_unused size_t count_utf16be(const char16_t *buf, - size_t length) const noexcept; + simdutf_warn_unused size_t + count_utf16le(const char16_t *buf, size_t length) const noexcept override; + simdutf_warn_unused size_t + count_utf16be(const char16_t *buf, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 simdutf_warn_unused size_t count_utf8(const char *buf, - size_t length) const noexcept; + size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf8_length_from_utf16le(const char16_t *input, size_t length) const noexcept; - simdutf_warn_unused size_t - utf8_length_from_utf16be(const char16_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_utf16le( + const char16_t *input, size_t length) const noexcept override; + simdutf_warn_unused size_t utf8_length_from_utf16be( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF16 || SIMDUTF_FEATURE_UTF32 simdutf_warn_unused size_t utf32_length_from_utf16le( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; simdutf_warn_unused size_t utf32_length_from_utf16be( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 || SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf16_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf16_length_from_utf8( + const char *input, size_t length) const noexcept override; simdutf_warn_unused result utf8_length_from_utf16le_with_replacement( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; ; simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf8_length_from_utf32(const char32_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_utf32( + const char32_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf16_length_from_utf32(const char32_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf16_length_from_utf32( + const char32_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 || SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf32_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf32_length_from_utf8( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 || SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 - simdutf_warn_unused size_t - latin1_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t latin1_length_from_utf8( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 - simdutf_warn_unused size_t - utf8_length_from_latin1(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_latin1( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_BASE64 simdutf_warn_unused result base64_to_binary( const char *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; simdutf_warn_unused full_result base64_to_binary_details( const char *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; - simdutf_warn_unused result - base64_to_binary(const char16_t *input, size_t length, char *output, - base64_options options, - last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; + simdutf_warn_unused result base64_to_binary( + const char16_t *input, size_t length, char *output, + base64_options options, + last_chunk_handling_options last_chunk_options = + last_chunk_handling_options::loose) const noexcept override; simdutf_warn_unused full_result base64_to_binary_details( const char16_t *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; size_t binary_to_base64(const char *input, size_t length, char *output, - base64_options options) const noexcept; + base64_options options) const noexcept override; - size_t binary_to_base64_with_lines(const char *input, size_t length, - char *output, size_t line_length, - base64_options options) const noexcept; + size_t + binary_to_base64_with_lines(const char *input, size_t length, char *output, + size_t line_length, + base64_options options) const noexcept override; const char *find(const char *start, const char *end, - char character) const noexcept; + char character) const noexcept override; const char16_t *find(const char16_t *start, const char16_t *end, - char16_t character) const noexcept; + char16_t character) const noexcept override; #endif // SIMDUTF_FEATURE_BASE64 private: const bool _supports_zvbb; @@ -9127,85 +9221,99 @@ class implementation final : public simdutf::implementation { #if SIMDUTF_FEATURE_UTF16 void change_endianness_utf16(const char16_t *buf, size_t length, char16_t *output) const noexcept final; - simdutf_warn_unused size_t count_utf16le(const char16_t *buf, - size_t length) const noexcept; - simdutf_warn_unused size_t count_utf16be(const char16_t *buf, - size_t length) const noexcept; + simdutf_warn_unused size_t + count_utf16le(const char16_t *buf, size_t length) const noexcept override; + simdutf_warn_unused size_t + count_utf16be(const char16_t *buf, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 simdutf_warn_unused size_t count_utf8(const char *buf, - size_t length) const noexcept; + size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf8_length_from_utf16le(const char16_t *input, size_t length) const noexcept; - simdutf_warn_unused size_t - utf8_length_from_utf16be(const char16_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_utf16le( + const char16_t *input, size_t length) const noexcept override; + simdutf_warn_unused size_t utf8_length_from_utf16be( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 simdutf_warn_unused size_t utf32_length_from_utf16le( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; simdutf_warn_unused size_t utf32_length_from_utf16be( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf16_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf16_length_from_utf8( + const char *input, size_t length) const noexcept override; simdutf_warn_unused result utf8_length_from_utf16le_with_replacement( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; ; simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf8_length_from_utf32(const char32_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_utf32( + const char32_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf16_length_from_utf32(const char32_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf16_length_from_utf32( + const char32_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf32_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf32_length_from_utf8( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 - simdutf_warn_unused size_t - latin1_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t latin1_length_from_utf8( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 - simdutf_warn_unused size_t - utf8_length_from_latin1(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_latin1( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_BASE64 simdutf_warn_unused result base64_to_binary( const char *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; simdutf_warn_unused full_result base64_to_binary_details( const char *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; - simdutf_warn_unused result - base64_to_binary(const char16_t *input, size_t length, char *output, - base64_options options, - last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; + simdutf_warn_unused result base64_to_binary( + const char16_t *input, size_t length, char *output, + base64_options options, + last_chunk_handling_options last_chunk_options = + last_chunk_handling_options::loose) const noexcept override; simdutf_warn_unused full_result base64_to_binary_details( const char16_t *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; size_t binary_to_base64(const char *input, size_t length, char *output, - base64_options options) const noexcept; - size_t binary_to_base64_with_lines(const char *input, size_t length, - char *output, size_t line_length, - base64_options options) const noexcept; + base64_options options) const noexcept override; + size_t + binary_to_base64_with_lines(const char *input, size_t length, char *output, + size_t line_length, + base64_options options) const noexcept override; const char *find(const char *start, const char *end, - char character) const noexcept; + char character) const noexcept override; const char16_t *find(const char16_t *start, const char16_t *end, - char16_t character) const noexcept; + char16_t character) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char *input, size_t length) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_BASE64 }; @@ -9991,6 +10099,10 @@ template <> struct simd8 : base8_numeric { operator>=(const simd8 other) const { return __lasx_xvsle_bu(other, *this); } + simdutf_really_inline simd8 + operator>(const simd8 other) const { + return __lasx_xvslt_bu(other, *this); + } simdutf_really_inline simd8 &operator-=(const simd8 other) { value = __lasx_xvsub_b(value, other.value); return *this; @@ -10091,6 +10203,12 @@ template struct simd8x64 { .to_bitmask(); } + simdutf_really_inline uint64_t gteq(const T m) const { + const simd8 mask = simd8::splat(m); + return simd8x64(this->chunks[0] >= mask, this->chunks[1] >= mask) + .to_bitmask(); + } + simdutf_really_inline uint64_t gt(const T m) const { const simd8 mask = simd8::splat(m); return simd8x64(this->chunks[0] > mask, this->chunks[1] > mask) @@ -10291,6 +10409,11 @@ template struct simd16x32 { uint64_t r_hi = this->chunks[1].to_bitmask(); return r_lo | (r_hi << 32); } + simdutf_really_inline uint64_t gteq(const T m) const { + const simd16 mask = simd16::splat(m); + return simd16x32(this->chunks[0] >= mask, this->chunks[1] >= mask) + .to_bitmask(); + } simdutf_really_inline uint64_t lteq(const T m) const { const simd16 mask = simd16::splat(m); return simd16x32(this->chunks[0] <= mask, this->chunks[1] <= mask) @@ -10733,85 +10856,99 @@ class implementation final : public simdutf::implementation { #if SIMDUTF_FEATURE_UTF16 void change_endianness_utf16(const char16_t *buf, size_t length, char16_t *output) const noexcept final; - simdutf_warn_unused size_t count_utf16le(const char16_t *buf, - size_t length) const noexcept; - simdutf_warn_unused size_t count_utf16be(const char16_t *buf, - size_t length) const noexcept; + simdutf_warn_unused size_t + count_utf16le(const char16_t *buf, size_t length) const noexcept override; + simdutf_warn_unused size_t + count_utf16be(const char16_t *buf, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 simdutf_warn_unused size_t count_utf8(const char *buf, - size_t length) const noexcept; + size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf8_length_from_utf16le(const char16_t *input, size_t length) const noexcept; - simdutf_warn_unused size_t - utf8_length_from_utf16be(const char16_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_utf16le( + const char16_t *input, size_t length) const noexcept override; + simdutf_warn_unused size_t utf8_length_from_utf16be( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 simdutf_warn_unused size_t utf32_length_from_utf16le( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; simdutf_warn_unused size_t utf32_length_from_utf16be( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 - simdutf_warn_unused size_t - utf16_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf16_length_from_utf8( + const char *input, size_t length) const noexcept override; simdutf_warn_unused result utf8_length_from_utf16le_with_replacement( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; ; simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( - const char16_t *input, size_t length) const noexcept; + const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf8_length_from_utf32(const char32_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_utf32( + const char32_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf16_length_from_utf32(const char32_t *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf16_length_from_utf32( + const char32_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF16 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 - simdutf_warn_unused size_t - utf32_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf32_length_from_utf8( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 - simdutf_warn_unused size_t - latin1_length_from_utf8(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t latin1_length_from_utf8( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 - simdutf_warn_unused size_t - utf8_length_from_latin1(const char *input, size_t length) const noexcept; + simdutf_warn_unused size_t utf8_length_from_latin1( + const char *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 #if SIMDUTF_FEATURE_BASE64 simdutf_warn_unused result base64_to_binary( const char *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; simdutf_warn_unused full_result base64_to_binary_details( const char *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; - simdutf_warn_unused result - base64_to_binary(const char16_t *input, size_t length, char *output, - base64_options options, - last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; + simdutf_warn_unused result base64_to_binary( + const char16_t *input, size_t length, char *output, + base64_options options, + last_chunk_handling_options last_chunk_options = + last_chunk_handling_options::loose) const noexcept override; simdutf_warn_unused full_result base64_to_binary_details( const char16_t *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_options = - last_chunk_handling_options::loose) const noexcept; + last_chunk_handling_options::loose) const noexcept override; size_t binary_to_base64(const char *input, size_t length, char *output, - base64_options options) const noexcept; - size_t binary_to_base64_with_lines(const char *input, size_t length, - char *output, size_t line_length, - base64_options options) const noexcept; + base64_options options) const noexcept override; + size_t + binary_to_base64_with_lines(const char *input, size_t length, char *output, + size_t line_length, + base64_options options) const noexcept override; const char *find(const char *start, const char *end, - char character) const noexcept; + char character) const noexcept override; const char16_t *find(const char16_t *start, const char16_t *end, - char16_t character) const noexcept; + char16_t character) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char *input, size_t length) const noexcept override; + simdutf_warn_unused size_t binary_length_from_base64( + const char16_t *input, size_t length) const noexcept override; #endif // SIMDUTF_FEATURE_BASE64 }; @@ -11445,6 +11582,12 @@ template struct simd8x64 { this->chunks[2] > mask, this->chunks[3] > mask) .to_bitmask(); } + simdutf_really_inline uint64_t gteq(const T m) const { + const simd8 mask = simd8::splat(m); + return simd8x64(this->chunks[0] >= mask, this->chunks[1] >= mask, + this->chunks[2] >= mask, this->chunks[3] >= mask) + .to_bitmask(); + } simdutf_really_inline uint64_t gteq_unsigned(const uint8_t m) const { const simd8 mask = simd8::splat(m); return simd8x64(simd8(this->chunks[0].value) >= mask, @@ -11665,6 +11808,12 @@ template struct simd16x32 { uint64_t r3 = this->chunks[3].to_bitmask(); return r0 | (r1 << 16) | (r2 << 32) | (r3 << 48); } + simdutf_really_inline uint64_t gteq(const T m) const { + const simd16 mask = simd16::splat(m); + return simd16x32(this->chunks[0] >= mask, this->chunks[1] >= mask, + this->chunks[2] >= mask, this->chunks[3] >= mask) + .to_bitmask(); + } simdutf_really_inline uint64_t lteq(const T m) const { const simd16 mask = simd16::splat(m); return simd16x32(this->chunks[0] <= mask, this->chunks[1] <= mask, @@ -12123,6 +12272,15 @@ class implementation final : public simdutf::implementation { simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( const char16_t *input, size_t length) const noexcept override; ; + + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept override; + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -12371,6 +12529,17 @@ simdutf_warn_unused size_t implementation::maximal_binary_length_from_base64( const char16_t *input, size_t length) const noexcept { return scalar::base64::maximal_binary_length_from_base64(input, length); } + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char *input, size_t length) const noexcept { + return scalar::base64::binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char16_t *input, size_t length) const noexcept { + return scalar::base64::binary_length_from_base64(input, length); +} + simdutf_warn_unused size_t implementation::base64_length_from_binary( size_t length, base64_options options) const noexcept { return scalar::base64::base64_length_from_binary(length, options); @@ -12447,7 +12616,7 @@ static const fallback::implementation *get_fallback_singleton() { #endif #if SIMDUTF_SINGLE_IMPLEMENTATION -static const implementation *get_single_implementation() { +simdutf_really_inline static const implementation *get_single_implementation() { return #if SIMDUTF_IMPLEMENTATION_ICELAKE get_icelake_singleton(); @@ -12683,6 +12852,20 @@ class detect_best_supported_implementation_on_first_use final return set_best()->utf8_length_from_utf16be_with_replacement(input, length); } + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept final override { + return set_best()->convert_utf16le_to_utf8_with_replacement(input, length, + utf8_buffer); + } + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept final override { + return set_best()->convert_utf16be_to_utf8_with_replacement(input, length, + utf8_buffer); + } + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -13052,6 +13235,16 @@ class detect_best_supported_implementation_on_first_use final char16_t character) const noexcept override { return set_best()->find(start, end, character); } + + simdutf_warn_unused size_t binary_length_from_base64( + const char *input, size_t length) const noexcept override { + return set_best()->binary_length_from_base64(input, length); + } + + simdutf_warn_unused size_t binary_length_from_base64( + const char16_t *input, size_t length) const noexcept override { + return set_best()->binary_length_from_base64(input, length); + } #endif // SIMDUTF_FEATURE_BASE64 simdutf_really_inline @@ -13290,6 +13483,16 @@ class unsupported_implementation final : public implementation { return {OTHER, 0}; // Not supported } + simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *, size_t, char *) const noexcept final override { + return 0; // Not supported + } + + simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *, size_t, char *) const noexcept final override { + return 0; // Not supported + } + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -13597,6 +13800,14 @@ class unsupported_implementation final : public implementation { char16_t) const noexcept override { return nullptr; } + simdutf_warn_unused size_t + binary_length_from_base64(const char *, size_t) const noexcept override { + return 0; + } + simdutf_warn_unused size_t + binary_length_from_base64(const char16_t *, size_t) const noexcept override { + return 0; + } #endif // SIMDUTF_FEATURE_BASE64 unsupported_implementation() @@ -13692,11 +13903,12 @@ get_active_implementation() { } #if SIMDUTF_SINGLE_IMPLEMENTATION -const implementation *get_default_implementation() { +simdutf_really_inline const implementation *get_default_implementation() { return internal::get_single_implementation(); } #else -internal::atomic_ptr &get_default_implementation() { +simdutf_really_inline internal::atomic_ptr & +get_default_implementation() { return get_active_implementation(); } #endif @@ -14494,6 +14706,27 @@ simdutf_warn_unused result utf8_length_from_utf16be_with_replacement( ->utf8_length_from_utf16be_with_replacement(input, length); } +simdutf_warn_unused size_t convert_utf16_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) noexcept { + #if SIMDUTF_IS_BIG_ENDIAN + return convert_utf16be_to_utf8_with_replacement(input, length, utf8_buffer); + #else + return convert_utf16le_to_utf8_with_replacement(input, length, utf8_buffer); + #endif +} + +simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) noexcept { + return get_default_implementation()->convert_utf16le_to_utf8_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) noexcept { + return get_default_implementation()->convert_utf16be_to_utf8_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -14557,6 +14790,16 @@ simdutf_warn_unused size_t maximal_binary_length_from_base64( input, length); } +simdutf_warn_unused size_t binary_length_from_base64(const char *input, + size_t length) noexcept { + return get_default_implementation()->binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t binary_length_from_base64(const char16_t *input, + size_t length) noexcept { + return get_default_implementation()->binary_length_from_base64(input, length); +} + simdutf_warn_unused result base64_to_binary( const char16_t *input, size_t length, char *output, base64_options options, last_chunk_handling_options last_chunk_handling_options) noexcept { @@ -19435,10 +19678,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -19523,10 +19766,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -20627,6 +20870,73 @@ simdutf_really_inline size_t convert_valid(const char *in, size_t size, // namespace simdutf /* end file src/generic/utf8_to_latin1/valid_utf8_to_latin1.h */ #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 +#if SIMDUTF_FEATURE_BASE64 +/* begin file src/generic/base64lengths.h */ +namespace simdutf { +namespace arm64 { +namespace { +namespace base64_lengths { + +simdutf_warn_unused size_t binary_length_from_base64(const char *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 64 <= length; pos += 64) { + simd8x64 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +simdutf_warn_unused size_t binary_length_from_base64(const char16_t *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 32 <= length; pos += 32) { + simd16x32 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char16_t c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +} // namespace base64_lengths +} // unnamed namespace +} // namespace arm64 +} // namespace simdutf +/* end file src/generic/base64lengths.h */ +#endif // SIMDUTF_FEATURE_BASE64 // // Implementation-specific overrides @@ -21620,6 +21930,20 @@ implementation::utf8_length_from_utf16be_with_replacement( length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -21838,6 +22162,16 @@ const char16_t *implementation::find(const char16_t *start, const char16_t *end, char16_t character) const noexcept { return util_find(start, end, character); } + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char16_t *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} #endif // SIMDUTF_FEATURE_BASE64 } // namespace arm64 @@ -22347,6 +22681,20 @@ implementation::utf8_length_from_utf16be_with_replacement( endianness::BIG>(input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -26902,6 +27250,70 @@ compress_decode_base64(char *dst, const chartype *src, size_t srclen, } return {SUCCESS, srclen, size_t(dst - dstinit)}; } + +simdutf_warn_unused size_t icelake_binary_length_from_base64(const char *input, + size_t length) { + size_t count = 0; + const char *ptr = input; + const char *end = input + length; + + __m512i spaces = _mm512_set1_epi8(0x20); + while (ptr + 64 <= end) { + __m512i data = _mm512_loadu_si512(reinterpret_cast(ptr)); + uint64_t mask = _mm512_cmpgt_epi8_mask(data, spaces); + count += count_ones(mask); + ptr += 64; + } + + while (ptr < end) { + count += (*ptr > 0x20) ? 1 : 0; + ptr++; + } + + size_t padding = 0; + size_t pos = length; + while (pos > 0 && padding < 2) { + char c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +simdutf_warn_unused size_t +icelake_binary_length_from_base64(const char16_t *input, size_t length) { + size_t count = 0; + const char16_t *ptr = input; + const char16_t *end = input + length; + + __m512i spaces = _mm512_set1_epi16(0x20); + while (ptr + 32 <= end) { + __m512i data = _mm512_loadu_si512(reinterpret_cast(ptr)); + __mmask32 mask = _mm512_cmpgt_epi16_mask(data, spaces); + count += _mm_popcnt_u32(mask); + ptr += 32; + } + + while (ptr < end) { + count += (*ptr > 0x20) ? 1 : 0; + ptr++; + } + + size_t padding = 0; + size_t pos = length; + while (pos > 0 && padding < 2) { + char16_t c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} /* end file src/icelake/icelake_base64.inl.cpp */ /* begin file src/icelake/icelake_find.inl.cpp */ simdutf_really_inline const char *util_find(const char *start, const char *end, @@ -28788,6 +29200,20 @@ implementation::utf8_length_from_utf16be_with_replacement( input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 simdutf_warn_unused size_t implementation::utf8_length_from_utf32( @@ -28979,6 +29405,16 @@ const char16_t *implementation::find(const char16_t *start, const char16_t *end, char16_t character) const noexcept { return util_find(start, end, character); } + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char *input, size_t length) const noexcept { + return icelake_binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char16_t *input, size_t length) const noexcept { + return icelake_binary_length_from_base64(input, length); +} #endif // SIMDUTF_FEATURE_BASE64 } // namespace icelake @@ -32346,6 +32782,73 @@ class block64 { return 63; } }; + +simdutf_warn_unused size_t avx2_binary_length_from_base64(const char *input, + size_t length) { + size_t count = 0; + const char *ptr = input; + const char *end = input + length; + + __m256i spaces = _mm256_set1_epi8(0x20); + while (ptr + 32 <= end) { + __m256i data = _mm256_loadu_si256(reinterpret_cast(ptr)); + __m256i gt_space = _mm256_cmpgt_epi8(data, spaces); + uint32_t mask = static_cast(_mm256_movemask_epi8(gt_space)); + count += count_ones(mask); + ptr += 32; + } + + while (ptr < end) { + count += (*ptr > 0x20) ? 1 : 0; + ptr++; + } + + size_t padding = 0; + size_t pos = length; + while (pos > 0 && padding < 2) { + char c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +simdutf_warn_unused size_t avx2_binary_length_from_base64(const char16_t *input, + size_t length) { + size_t count = 0; + const char16_t *ptr = input; + const char16_t *end = input + length; + + __m256i spaces = _mm256_set1_epi16(0x20); + while (ptr + 16 <= end) { + __m256i data = _mm256_loadu_si256(reinterpret_cast(ptr)); + __m256i gt_space = _mm256_cmpgt_epi16(data, spaces); + uint32_t mask = static_cast(_mm256_movemask_epi8(gt_space)); + count += count_ones(mask); + ptr += 16; + } + count /= 2; + + while (ptr < end) { + count += (*ptr > 0x20) ? 1 : 0; + ptr++; + } + + size_t padding = 0; + size_t pos = length; + while (pos > 0 && padding < 2) { + char16_t c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} /* end file src/haswell/avx2_base64.cpp */ #endif // SIMDUTF_FEATURE_BASE64 @@ -33072,10 +33575,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -33160,10 +33663,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -36197,6 +36700,20 @@ implementation::utf8_length_from_utf16be_with_replacement( input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_LATIN1 @@ -36444,6 +36961,16 @@ const char16_t *implementation::find(const char16_t *start, const char16_t *end, char16_t character) const noexcept { return util::find(start, end, character); } + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char *input, size_t length) const noexcept { + return avx2_binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char16_t *input, size_t length) const noexcept { + return avx2_binary_length_from_base64(input, length); +} #endif // SIMDUTF_FEATURE_BASE64 } // namespace haswell @@ -40119,10 +40646,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -40207,10 +40734,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -42815,6 +43342,20 @@ implementation::utf8_length_from_utf16be_with_replacement( endianness::BIG>(input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -44966,6 +45507,20 @@ implementation::utf8_length_from_utf16be_with_replacement( endianness::BIG>(input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 } // namespace rvv @@ -48848,10 +49403,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -48936,10 +49491,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -50943,6 +51498,71 @@ find(const char16_t *start, const char16_t *end, char16_t character) noexcept { } // namespace westmere } // namespace simdutf /* end file src/generic/find.h */ +/* begin file src/generic/base64lengths.h */ +namespace simdutf { +namespace westmere { +namespace { +namespace base64_lengths { + +simdutf_warn_unused size_t binary_length_from_base64(const char *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 64 <= length; pos += 64) { + simd8x64 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +simdutf_warn_unused size_t binary_length_from_base64(const char16_t *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 32 <= length; pos += 32) { + simd16x32 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char16_t c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +} // namespace base64_lengths +} // unnamed namespace +} // namespace westmere +} // namespace simdutf +/* end file src/generic/base64lengths.h */ #endif // SIMDUTF_FEATURE_BASE64 // @@ -52060,6 +52680,20 @@ implementation::utf8_length_from_utf16be_with_replacement( input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -52250,6 +52884,16 @@ const char16_t *implementation::find(const char16_t *start, const char16_t *end, char16_t character) const noexcept { return util::find(start, end, character); } + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char16_t *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} #endif // SIMDUTF_FEATURE_BASE64 } // namespace westmere @@ -56876,10 +57520,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -56964,10 +57608,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -58281,6 +58925,73 @@ simdutf_really_inline size_t utf8_length_from_utf32(const char32_t *input, } // namespace simdutf /* end file src/generic/utf32.h */ #endif // SIMDUTF_FEATURE_UTF32 +#if SIMDUTF_FEATURE_BASE64 +/* begin file src/generic/base64lengths.h */ +namespace simdutf { +namespace lasx { +namespace { +namespace base64_lengths { + +simdutf_warn_unused size_t binary_length_from_base64(const char *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 64 <= length; pos += 64) { + simd8x64 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +simdutf_warn_unused size_t binary_length_from_base64(const char16_t *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 32 <= length; pos += 32) { + simd16x32 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char16_t c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +} // namespace base64_lengths +} // unnamed namespace +} // namespace lasx +} // namespace simdutf +/* end file src/generic/base64lengths.h */ +#endif // SIMDUTF_FEATURE_BASE64 // // Implementation-specific overrides @@ -59377,6 +60088,20 @@ implementation::utf8_length_from_utf16be_with_replacement( endianness::BIG>(input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -59558,6 +60283,16 @@ const char16_t *implementation::find(const char16_t *start, const char16_t *end, char16_t character) const noexcept { return util_find(start, end, character); } + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char16_t *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} #endif // SIMDUTF_FEATURE_BASE64 } // namespace lasx @@ -63767,10 +64502,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -63855,10 +64590,10 @@ struct validating_transcoder { (simd8x64::NUM_CHUNKS == 4), "We support either two or four chunks per 64-byte block."); auto zero = simd8{uint8_t(0)}; - if (simd8x64::NUM_CHUNKS == 2) { + if simdutf_constexpr (simd8x64::NUM_CHUNKS == 2) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); - } else if (simd8x64::NUM_CHUNKS == 4) { + } else if simdutf_constexpr (simd8x64::NUM_CHUNKS == 4) { this->check_utf8_bytes(input.chunks[0], zero); this->check_utf8_bytes(input.chunks[1], input.chunks[0]); this->check_utf8_bytes(input.chunks[2], input.chunks[1]); @@ -65173,6 +65908,73 @@ simdutf_really_inline size_t utf8_length_from_utf32(const char32_t *input, } // namespace simdutf /* end file src/generic/utf32.h */ #endif // SIMDUTF_FEATURE_UTF32 +#if SIMDUTF_FEATURE_BASE64 +/* begin file src/generic/base64lengths.h */ +namespace simdutf { +namespace lsx { +namespace { +namespace base64_lengths { + +simdutf_warn_unused size_t binary_length_from_base64(const char *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 64 <= length; pos += 64) { + simd8x64 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +simdutf_warn_unused size_t binary_length_from_base64(const char16_t *input, + size_t length) { + size_t pos = 0; + size_t count = 0; + for (; pos + 32 <= length; pos += 32) { + simd16x32 block(reinterpret_cast(input + pos)); + uint64_t maybe_base64 = block.gteq(33); // >= 33 which is '!' in ASCII + count += count_ones(maybe_base64); + } + while (pos < length) { + count += (input[pos] > 0x20) ? 1 : 0; + pos++; + } + // Count padding at the end. + size_t padding = 0; + pos = length; + while (pos > 0 && padding < 2) { + char16_t c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + +} // namespace base64_lengths +} // unnamed namespace +} // namespace lsx +} // namespace simdutf +/* end file src/generic/base64lengths.h */ +#endif // SIMDUTF_FEATURE_BASE64 // // Implementation-specific overrides @@ -66156,6 +66958,20 @@ implementation::utf8_length_from_utf16be_with_replacement( endianness::BIG>(input, length); } +simdutf_warn_unused size_t +implementation::convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + +simdutf_warn_unused size_t +implementation::convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) const noexcept { + return scalar::utf16_to_utf8::convert_with_replacement( + input, length, utf8_buffer); +} + #endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 #if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF32 @@ -66337,6 +67153,16 @@ const char16_t *implementation::find(const char16_t *start, const char16_t *end, char16_t character) const noexcept { return util_find(start, end, character); } + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} + +simdutf_warn_unused size_t implementation::binary_length_from_base64( + const char16_t *input, size_t length) const noexcept { + return base64_lengths::binary_length_from_base64(input, length); +} #endif // SIMDUTF_FEATURE_BASE64 } // namespace lsx @@ -66467,6 +67293,7 @@ size_t simdutf_latin1_length_from_utf32(size_t length); size_t simdutf_utf16_length_from_utf8(const char *input, size_t length); size_t simdutf_utf32_length_from_utf8(const char *input, size_t length); size_t simdutf_utf8_length_from_utf16(const char16_t *input, size_t length); +size_t simdutf_utf8_length_from_utf32(const char32_t *input, size_t length); simdutf_result simdutf_utf8_length_from_utf16_with_replacement(const char16_t *input, size_t length); @@ -66488,6 +67315,8 @@ size_t simdutf_convert_latin1_to_utf16le(const char *input, size_t length, char16_t *output); size_t simdutf_convert_latin1_to_utf16be(const char *input, size_t length, char16_t *output); +size_t simdutf_convert_latin1_to_utf16(const char *input, size_t length, + char16_t *output); size_t simdutf_convert_latin1_to_utf32(const char *input, size_t length, char32_t *output); @@ -66733,81 +67562,66 @@ simdutf_result simdutf_validate_ascii_with_errors(const char *buf, size_t len) { } bool simdutf_validate_utf16_as_ascii(const char16_t *buf, size_t len) { - return simdutf::validate_utf16_as_ascii( - reinterpret_cast(buf), len); + return simdutf::validate_utf16_as_ascii(buf, len); } bool simdutf_validate_utf16be_as_ascii(const char16_t *buf, size_t len) { - return simdutf::validate_utf16be_as_ascii( - reinterpret_cast(buf), len); + return simdutf::validate_utf16be_as_ascii(buf, len); } bool simdutf_validate_utf16le_as_ascii(const char16_t *buf, size_t len) { - return simdutf::validate_utf16le_as_ascii( - reinterpret_cast(buf), len); + return simdutf::validate_utf16le_as_ascii(buf, len); } bool simdutf_validate_utf16(const char16_t *buf, size_t len) { - return simdutf::validate_utf16(reinterpret_cast(buf), len); + return simdutf::validate_utf16(buf, len); } bool simdutf_validate_utf16le(const char16_t *buf, size_t len) { - return simdutf::validate_utf16le(reinterpret_cast(buf), - len); + return simdutf::validate_utf16le(buf, len); } bool simdutf_validate_utf16be(const char16_t *buf, size_t len) { - return simdutf::validate_utf16be(reinterpret_cast(buf), - len); + return simdutf::validate_utf16be(buf, len); } simdutf_result simdutf_validate_utf16_with_errors(const char16_t *buf, size_t len) { - return to_c_result(simdutf::validate_utf16_with_errors( - reinterpret_cast(buf), len)); + return to_c_result(simdutf::validate_utf16_with_errors(buf, len)); } simdutf_result simdutf_validate_utf16le_with_errors(const char16_t *buf, size_t len) { - return to_c_result(simdutf::validate_utf16le_with_errors( - reinterpret_cast(buf), len)); + return to_c_result(simdutf::validate_utf16le_with_errors(buf, len)); } simdutf_result simdutf_validate_utf16be_with_errors(const char16_t *buf, size_t len) { - return to_c_result(simdutf::validate_utf16be_with_errors( - reinterpret_cast(buf), len)); + return to_c_result(simdutf::validate_utf16be_with_errors(buf, len)); } bool simdutf_validate_utf32(const char32_t *buf, size_t len) { - return simdutf::validate_utf32(reinterpret_cast(buf), len); + return simdutf::validate_utf32(buf, len); } simdutf_result simdutf_validate_utf32_with_errors(const char32_t *buf, size_t len) { - return to_c_result(simdutf::validate_utf32_with_errors( - reinterpret_cast(buf), len)); + return to_c_result(simdutf::validate_utf32_with_errors(buf, len)); } void simdutf_to_well_formed_utf16le(const char16_t *input, size_t len, char16_t *output) { - simdutf::to_well_formed_utf16le(reinterpret_cast(input), - len, reinterpret_cast(output)); + simdutf::to_well_formed_utf16le(input, len, output); } void simdutf_to_well_formed_utf16be(const char16_t *input, size_t len, char16_t *output) { - simdutf::to_well_formed_utf16be(reinterpret_cast(input), - len, reinterpret_cast(output)); + simdutf::to_well_formed_utf16be(input, len, output); } void simdutf_to_well_formed_utf16(const char16_t *input, size_t len, char16_t *output) { - simdutf::to_well_formed_utf16(reinterpret_cast(input), len, - reinterpret_cast(output)); + simdutf::to_well_formed_utf16(input, len, output); } size_t simdutf_count_utf16(const char16_t *input, size_t length) { - return simdutf::count_utf16(reinterpret_cast(input), - length); + return simdutf::count_utf16(input, length); } size_t simdutf_count_utf16le(const char16_t *input, size_t length) { - return simdutf::count_utf16le(reinterpret_cast(input), - length); + return simdutf::count_utf16le(input, length); } size_t simdutf_count_utf16be(const char16_t *input, size_t length) { - return simdutf::count_utf16be(reinterpret_cast(input), - length); + return simdutf::count_utf16be(input, length); } size_t simdutf_count_utf8(const char *input, size_t length) { return simdutf::count_utf8(input, length); @@ -66832,34 +67646,34 @@ size_t simdutf_utf32_length_from_utf8(const char *input, size_t length) { return simdutf::utf32_length_from_utf8(input, length); } size_t simdutf_utf8_length_from_utf16(const char16_t *input, size_t length) { - return simdutf::utf8_length_from_utf16( - reinterpret_cast(input), length); + return simdutf::utf8_length_from_utf16(input, length); +} +size_t simdutf_utf8_length_from_utf32(const char32_t *input, size_t length) { + return simdutf::utf8_length_from_utf32(input, length); } simdutf_result simdutf_utf8_length_from_utf16_with_replacement(const char16_t *input, size_t length) { - return to_c_result(simdutf::utf8_length_from_utf16_with_replacement( - reinterpret_cast(input), length)); + return to_c_result( + simdutf::utf8_length_from_utf16_with_replacement(input, length)); } size_t simdutf_utf8_length_from_utf16le(const char16_t *input, size_t length) { - return simdutf::utf8_length_from_utf16le( - reinterpret_cast(input), length); + return simdutf::utf8_length_from_utf16le(input, length); } size_t simdutf_utf8_length_from_utf16be(const char16_t *input, size_t length) { - return simdutf::utf8_length_from_utf16be( - reinterpret_cast(input), length); + return simdutf::utf8_length_from_utf16be(input, length); } simdutf_result simdutf_utf8_length_from_utf16le_with_replacement(const char16_t *input, size_t length) { - return to_c_result(simdutf::utf8_length_from_utf16le_with_replacement( - reinterpret_cast(input), length)); + return to_c_result( + simdutf::utf8_length_from_utf16le_with_replacement(input, length)); } simdutf_result simdutf_utf8_length_from_utf16be_with_replacement(const char16_t *input, size_t length) { - return to_c_result(simdutf::utf8_length_from_utf16be_with_replacement( - reinterpret_cast(input), length)); + return to_c_result( + simdutf::utf8_length_from_utf16be_with_replacement(input, length)); } /* Conversions: latin1 <-> utf8, utf8 <-> utf16/utf32, utf16 <-> utf8, etc. */ @@ -66874,18 +67688,19 @@ size_t simdutf_convert_latin1_to_utf8_safe(const char *input, size_t length, } size_t simdutf_convert_latin1_to_utf16le(const char *input, size_t length, char16_t *output) { - return simdutf::convert_latin1_to_utf16le( - input, length, reinterpret_cast(output)); + return simdutf::convert_latin1_to_utf16le(input, length, output); } size_t simdutf_convert_latin1_to_utf16be(const char *input, size_t length, char16_t *output) { - return simdutf::convert_latin1_to_utf16be( - input, length, reinterpret_cast(output)); + return simdutf::convert_latin1_to_utf16be(input, length, output); +} +size_t simdutf_convert_latin1_to_utf16(const char *input, size_t length, + char16_t *output) { + return simdutf::convert_latin1_to_utf16(input, length, output); } size_t simdutf_convert_latin1_to_utf32(const char *input, size_t length, char32_t *output) { - return simdutf::convert_latin1_to_utf32(input, length, - reinterpret_cast(output)); + return simdutf::convert_latin1_to_utf32(input, length, output); } size_t simdutf_convert_utf8_to_latin1(const char *input, size_t length, @@ -66894,23 +67709,19 @@ size_t simdutf_convert_utf8_to_latin1(const char *input, size_t length, } size_t simdutf_convert_utf8_to_utf16le(const char *input, size_t length, char16_t *output) { - return simdutf::convert_utf8_to_utf16le(input, length, - reinterpret_cast(output)); + return simdutf::convert_utf8_to_utf16le(input, length, output); } size_t simdutf_convert_utf8_to_utf16(const char *input, size_t length, char16_t *output) { - return simdutf::convert_utf8_to_utf16(input, length, - reinterpret_cast(output)); + return simdutf::convert_utf8_to_utf16(input, length, output); } size_t simdutf_convert_utf8_to_utf16be(const char *input, size_t length, char16_t *output) { - return simdutf::convert_utf8_to_utf16be(input, length, - reinterpret_cast(output)); + return simdutf::convert_utf8_to_utf16be(input, length, output); } size_t simdutf_convert_utf8_to_utf32(const char *input, size_t length, char32_t *output) { - return simdutf::convert_utf8_to_utf32(input, length, - reinterpret_cast(output)); + return simdutf::convert_utf8_to_utf32(input, length, output); } simdutf_result simdutf_convert_utf8_to_latin1_with_errors(const char *input, size_t length, @@ -66921,26 +67732,26 @@ simdutf_result simdutf_convert_utf8_to_latin1_with_errors(const char *input, simdutf_result simdutf_convert_utf8_to_utf16_with_errors(const char *input, size_t length, char16_t *output) { - return to_c_result(simdutf::convert_utf8_to_utf16_with_errors( - input, length, reinterpret_cast(output))); + return to_c_result( + simdutf::convert_utf8_to_utf16_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf8_to_utf16le_with_errors(const char *input, size_t length, char16_t *output) { - return to_c_result(simdutf::convert_utf8_to_utf16le_with_errors( - input, length, reinterpret_cast(output))); + return to_c_result( + simdutf::convert_utf8_to_utf16le_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf8_to_utf16be_with_errors(const char *input, size_t length, char16_t *output) { - return to_c_result(simdutf::convert_utf8_to_utf16be_with_errors( - input, length, reinterpret_cast(output))); + return to_c_result( + simdutf::convert_utf8_to_utf16be_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf8_to_utf32_with_errors(const char *input, size_t length, char32_t *output) { - return to_c_result(simdutf::convert_utf8_to_utf32_with_errors( - input, length, reinterpret_cast(output))); + return to_c_result( + simdutf::convert_utf8_to_utf32_with_errors(input, length, output)); } /* Conversions assuming valid input */ @@ -66950,229 +67761,190 @@ size_t simdutf_convert_valid_utf8_to_latin1(const char *input, size_t length, } size_t simdutf_convert_valid_utf8_to_utf16le(const char *input, size_t length, char16_t *output) { - return simdutf::convert_valid_utf8_to_utf16le( - input, length, reinterpret_cast(output)); + return simdutf::convert_valid_utf8_to_utf16le(input, length, output); } size_t simdutf_convert_valid_utf8_to_utf16be(const char *input, size_t length, char16_t *output) { - return simdutf::convert_valid_utf8_to_utf16be( - input, length, reinterpret_cast(output)); + return simdutf::convert_valid_utf8_to_utf16be(input, length, output); } size_t simdutf_convert_valid_utf8_to_utf32(const char *input, size_t length, char32_t *output) { - return simdutf::convert_valid_utf8_to_utf32( - input, length, reinterpret_cast(output)); + return simdutf::convert_valid_utf8_to_utf32(input, length, output); } /* UTF-16 -> UTF-8 and related conversions */ size_t simdutf_convert_utf16_to_utf8(const char16_t *input, size_t length, char *output) { - return simdutf::convert_utf16_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_utf16_to_utf8(input, length, output); } size_t simdutf_convert_utf16_to_utf8_safe(const char16_t *input, size_t length, char *output, size_t utf8_len) { - return simdutf::convert_utf16_to_utf8_safe( - reinterpret_cast(input), length, output, utf8_len); + return simdutf::convert_utf16_to_utf8_safe(input, length, output, utf8_len); } size_t simdutf_convert_utf16_to_latin1(const char16_t *input, size_t length, char *output) { - return simdutf::convert_utf16_to_latin1( - reinterpret_cast(input), length, output); + return simdutf::convert_utf16_to_latin1(input, length, output); } size_t simdutf_convert_utf16le_to_latin1(const char16_t *input, size_t length, char *output) { - return simdutf::convert_utf16le_to_latin1( - reinterpret_cast(input), length, output); + return simdutf::convert_utf16le_to_latin1(input, length, output); } size_t simdutf_convert_utf16be_to_latin1(const char16_t *input, size_t length, char *output) { - return simdutf::convert_utf16be_to_latin1( - reinterpret_cast(input), length, output); + return simdutf::convert_utf16be_to_latin1(input, length, output); } simdutf_result simdutf_convert_utf16_to_latin1_with_errors(const char16_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf16_to_latin1_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf16_to_latin1_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf16le_to_latin1_with_errors(const char16_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf16le_to_latin1_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf16le_to_latin1_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf16be_to_latin1_with_errors(const char16_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf16be_to_latin1_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf16be_to_latin1_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf16_to_utf8_with_errors(const char16_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf16_to_utf8_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf16_to_utf8_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf16le_to_utf8_with_errors(const char16_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf16le_to_utf8_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf16le_to_utf8_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf16be_to_utf8_with_errors(const char16_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf16be_to_utf8_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf16be_to_utf8_with_errors(input, length, output)); } size_t simdutf_convert_utf16le_to_utf8(const char16_t *input, size_t length, char *output) { - return simdutf::convert_utf16le_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_utf16le_to_utf8(input, length, output); } size_t simdutf_convert_utf16be_to_utf8(const char16_t *input, size_t length, char *output) { - return simdutf::convert_utf16be_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_utf16be_to_utf8(input, length, output); } size_t simdutf_convert_valid_utf16_to_utf8(const char16_t *input, size_t length, char *output) { - return simdutf::convert_valid_utf16_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_valid_utf16_to_utf8(input, length, output); } size_t simdutf_convert_valid_utf16_to_latin1(const char16_t *input, size_t length, char *output) { - return simdutf::convert_valid_utf16_to_latin1( - reinterpret_cast(input), length, output); + return simdutf::convert_valid_utf16_to_latin1(input, length, output); } size_t simdutf_convert_valid_utf16le_to_latin1(const char16_t *input, size_t length, char *output) { - return simdutf::convert_valid_utf16le_to_latin1( - reinterpret_cast(input), length, output); + return simdutf::convert_valid_utf16le_to_latin1(input, length, output); } size_t simdutf_convert_valid_utf16be_to_latin1(const char16_t *input, size_t length, char *output) { - return simdutf::convert_valid_utf16be_to_latin1( - reinterpret_cast(input), length, output); + return simdutf::convert_valid_utf16be_to_latin1(input, length, output); } size_t simdutf_convert_valid_utf16le_to_utf8(const char16_t *input, size_t length, char *output) { - return simdutf::convert_valid_utf16le_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_valid_utf16le_to_utf8(input, length, output); } size_t simdutf_convert_valid_utf16be_to_utf8(const char16_t *input, size_t length, char *output) { - return simdutf::convert_valid_utf16be_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_valid_utf16be_to_utf8(input, length, output); } /* UTF-16 <-> UTF-32 conversions */ size_t simdutf_convert_utf16_to_utf32(const char16_t *input, size_t length, char32_t *output) { - return simdutf::convert_utf16_to_utf32( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_utf16_to_utf32(input, length, output); } size_t simdutf_convert_utf16le_to_utf32(const char16_t *input, size_t length, char32_t *output) { - return simdutf::convert_utf16le_to_utf32( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_utf16le_to_utf32(input, length, output); } size_t simdutf_convert_utf16be_to_utf32(const char16_t *input, size_t length, char32_t *output) { - return simdutf::convert_utf16be_to_utf32( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_utf16be_to_utf32(input, length, output); } simdutf_result simdutf_convert_utf16_to_utf32_with_errors(const char16_t *input, size_t length, char32_t *output) { - return to_c_result(simdutf::convert_utf16_to_utf32_with_errors( - reinterpret_cast(input), length, - reinterpret_cast(output))); + return to_c_result( + simdutf::convert_utf16_to_utf32_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf16le_to_utf32_with_errors(const char16_t *input, size_t length, char32_t *output) { - return to_c_result(simdutf::convert_utf16le_to_utf32_with_errors( - reinterpret_cast(input), length, - reinterpret_cast(output))); + return to_c_result( + simdutf::convert_utf16le_to_utf32_with_errors(input, length, output)); } simdutf_result simdutf_convert_utf16be_to_utf32_with_errors(const char16_t *input, size_t length, char32_t *output) { - return to_c_result(simdutf::convert_utf16be_to_utf32_with_errors( - reinterpret_cast(input), length, - reinterpret_cast(output))); + return to_c_result( + simdutf::convert_utf16be_to_utf32_with_errors(input, length, output)); } /* Valid UTF-16 conversions */ size_t simdutf_convert_valid_utf16_to_utf32(const char16_t *input, size_t length, char32_t *output) { - return simdutf::convert_valid_utf16_to_utf32( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_valid_utf16_to_utf32(input, length, output); } size_t simdutf_convert_valid_utf16le_to_utf32(const char16_t *input, size_t length, char32_t *output) { - return simdutf::convert_valid_utf16le_to_utf32( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_valid_utf16le_to_utf32(input, length, output); } size_t simdutf_convert_valid_utf16be_to_utf32(const char16_t *input, size_t length, char32_t *output) { - return simdutf::convert_valid_utf16be_to_utf32( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_valid_utf16be_to_utf32(input, length, output); } /* UTF-32 -> ... conversions */ size_t simdutf_convert_utf32_to_utf8(const char32_t *input, size_t length, char *output) { - return simdutf::convert_utf32_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_utf32_to_utf8(input, length, output); } simdutf_result simdutf_convert_utf32_to_utf8_with_errors(const char32_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf32_to_utf8_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf32_to_utf8_with_errors(input, length, output)); } size_t simdutf_convert_valid_utf32_to_utf8(const char32_t *input, size_t length, char *output) { - return simdutf::convert_valid_utf32_to_utf8( - reinterpret_cast(input), length, output); + return simdutf::convert_valid_utf32_to_utf8(input, length, output); } size_t simdutf_convert_utf32_to_utf16(const char32_t *input, size_t length, char16_t *output) { - return simdutf::convert_utf32_to_utf16( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_utf32_to_utf16(input, length, output); } size_t simdutf_convert_utf32_to_utf16le(const char32_t *input, size_t length, char16_t *output) { - return simdutf::convert_utf32_to_utf16le( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_utf32_to_utf16le(input, length, output); } size_t simdutf_convert_utf32_to_utf16be(const char32_t *input, size_t length, char16_t *output) { - return simdutf::convert_utf32_to_utf16be( - reinterpret_cast(input), length, - reinterpret_cast(output)); + return simdutf::convert_utf32_to_utf16be(input, length, output); } simdutf_result simdutf_convert_utf32_to_latin1_with_errors(const char32_t *input, size_t length, char *output) { - return to_c_result(simdutf::convert_utf32_to_latin1_with_errors( - reinterpret_cast(input), length, output)); + return to_c_result( + simdutf::convert_utf32_to_latin1_with_errors(input, length, output)); } /* --- find helpers --- */ diff --git a/simdutf.h b/simdutf.h index 0c93439..2862384 100644 --- a/simdutf.h +++ b/simdutf.h @@ -1,6 +1,6 @@ //go:build !libsimdutf -/* auto-generated on 2026-01-13 09:03:21 +0100. Do not edit! */ +/* auto-generated on 2026-03-12 20:42:59 -0400. Do not edit! */ /* begin file include/simdutf.h */ #ifndef SIMDUTF_H #define SIMDUTF_H @@ -210,6 +210,22 @@ #define SIMDUTF_IS_LASX 1 // We can always run both #elif defined(__loongarch_sx) #define SIMDUTF_IS_LSX 1 + // Adjust for runtime dispatching support. + #if defined(__GNUC__) && !defined(__clang__) && \ + !defined(__INTEL_COMPILER) && !defined(__NVCOMPILER) + #if __GNUC__ > 15 || (__GNUC__ == 15 && __GNUC_MINOR__ >= 0) + // We are ok, we will support runtime dispatch for LASX. + #else + // We disable runtime dispatch for LASX, which means that we will not be + // able to use LASX even if it is supported by the hardware. Loongson + // users should update to GCC 15 or better. + #define SIMDUTF_IMPLEMENTATION_LASX 0 + #endif + #else + // We are not using GCC, so we assume that we can support runtime dispatch + // for LASX. https://godbolt.org/z/jcMnrjYhs + #define SIMDUTF_IMPLEMENTATION_LASX 0 + #endif #endif #else // The simdutf library is designed @@ -924,7 +940,7 @@ SIMDUTF_DISABLE_UNDESIRED_WARNINGS #define SIMDUTF_SIMDUTF_VERSION_H /** The version of simdutf being used (major.minor.revision) */ -#define SIMDUTF_VERSION "8.0.0" +#define SIMDUTF_VERSION "8.2.0" namespace simdutf { enum { @@ -935,7 +951,7 @@ enum { /** * The minor version (major.MINOR.revision) of simdutf being used. */ - SIMDUTF_VERSION_MINOR = 0, + SIMDUTF_VERSION_MINOR = 2, /** * The revision (major.minor.REVISION) of simdutf being used. */ @@ -2794,6 +2810,87 @@ inline result simple_convert_with_errors(const char16_t *buf, size_t len, return convert_with_errors(buf, len, utf8_output, 0); } +template +simdutf_constexpr23 size_t convert_with_replacement(const char16_t *data, + size_t len, + char *utf8_output) { + size_t pos = 0; + char *start = utf8_output; + while (pos < len) { +#if SIMDUTF_CPLUSPLUS23 + if !consteval +#endif + { + // try to convert the next block of 8 bytes + if (pos + 4 <= len) { // if it is safe to read 8 more bytes, check that + // they are ascii + uint64_t v; + ::memcpy(&v, data + pos, sizeof(uint64_t)); + if simdutf_constexpr (!match_system(big_endian)) { + v = (v >> 8) | (v << (64 - 8)); + } + if ((v & 0xFF80FF80FF80FF80) == 0) { + size_t final_pos = pos + 4; + while (pos < final_pos) { + *utf8_output++ = !match_system(big_endian) + ? char(u16_swap_bytes(data[pos])) + : char(data[pos]); + pos++; + } + continue; + } + } + } + uint16_t word = + !match_system(big_endian) ? u16_swap_bytes(data[pos]) : data[pos]; + if ((word & 0xFF80) == 0) { + // will generate one UTF-8 bytes + *utf8_output++ = char(word); + pos++; + } else if ((word & 0xF800) == 0) { + // will generate two UTF-8 bytes + // we have 0b110XXXXX 0b10XXXXXX + *utf8_output++ = char((word >> 6) | 0b11000000); + *utf8_output++ = char((word & 0b111111) | 0b10000000); + pos++; + } else if ((word & 0xF800) != 0xD800) { + // will generate three UTF-8 bytes + // we have 0b1110XXXX 0b10XXXXXX 0b10XXXXXX + *utf8_output++ = char((word >> 12) | 0b11100000); + *utf8_output++ = char(((word >> 6) & 0b111111) | 0b10000000); + *utf8_output++ = char((word & 0b111111) | 0b10000000); + pos++; + } else { + // surrogate range + uint16_t diff = uint16_t(word - 0xD800); + if (diff <= 0x3FF && pos + 1 < len) { + // high surrogate, check for valid pair + uint16_t next_word = !match_system(big_endian) + ? u16_swap_bytes(data[pos + 1]) + : data[pos + 1]; + uint16_t diff2 = uint16_t(next_word - 0xDC00); + if (diff2 <= 0x3FF) { + // valid surrogate pair + uint32_t value = (diff << 10) + diff2 + 0x10000; + // will generate four UTF-8 bytes + *utf8_output++ = char((value >> 18) | 0b11110000); + *utf8_output++ = char(((value >> 12) & 0b111111) | 0b10000000); + *utf8_output++ = char(((value >> 6) & 0b111111) | 0b10000000); + *utf8_output++ = char((value & 0b111111) | 0b10000000); + pos += 2; + continue; + } + } + // unpaired surrogate: replace with U+FFFD (0xEF 0xBF 0xBD) + *utf8_output++ = char(0xef); + *utf8_output++ = char(0xbf); + *utf8_output++ = char(0xbd); + pos++; + } + } + return utf8_output - start; +} + } // namespace utf16_to_utf8 } // unnamed namespace } // namespace scalar @@ -7048,6 +7145,113 @@ convert_utf16be_to_utf8_with_errors( } #endif // SIMDUTF_SPAN +/** + * Convert possibly broken UTF-16LE string into UTF-8 string, replacing + * unpaired surrogates with the Unicode replacement character U+FFFD. + * + * This function always succeeds: unpaired surrogates are replaced with + * U+FFFD (3 bytes in UTF-8: 0xEF 0xBF 0xBD). + * + * This function is not BOM-aware. + * + * @param input the UTF-16LE string to convert + * @param length the length of the string in 2-byte code units (char16_t) + * @param utf8_buffer the pointer to buffer that can hold conversion result + * @return number of written code units + */ +simdutf_warn_unused size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 size_t +convert_utf16le_to_utf8_with_replacement( + std::span utf16_input, + detail::output_span_of_byte_like auto &&utf8_output) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + return scalar::utf16_to_utf8::convert_with_replacement( + utf16_input.data(), utf16_input.size(), utf8_output.data()); + } else + #endif + { + return convert_utf16le_to_utf8_with_replacement( + utf16_input.data(), utf16_input.size(), + reinterpret_cast(utf8_output.data())); + } +} + #endif // SIMDUTF_SPAN + +/** + * Convert possibly broken UTF-16BE string into UTF-8 string, replacing + * unpaired surrogates with the Unicode replacement character U+FFFD. + * + * This function always succeeds: unpaired surrogates are replaced with + * U+FFFD (3 bytes in UTF-8: 0xEF 0xBF 0xBD). + * + * This function is not BOM-aware. + * + * @param input the UTF-16BE string to convert + * @param length the length of the string in 2-byte code units (char16_t) + * @param utf8_buffer the pointer to buffer that can hold conversion result + * @return number of written code units + */ +simdutf_warn_unused size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 size_t +convert_utf16be_to_utf8_with_replacement( + std::span utf16_input, + detail::output_span_of_byte_like auto &&utf8_output) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + return scalar::utf16_to_utf8::convert_with_replacement( + utf16_input.data(), utf16_input.size(), utf8_output.data()); + } else + #endif + { + return convert_utf16be_to_utf8_with_replacement( + utf16_input.data(), utf16_input.size(), + reinterpret_cast(utf8_output.data())); + } +} + #endif // SIMDUTF_SPAN + +/** + * Convert possibly broken UTF-16 string (native endianness) into UTF-8 string, + * replacing unpaired surrogates with the Unicode replacement character U+FFFD. + * + * This function always succeeds: unpaired surrogates are replaced with + * U+FFFD (3 bytes in UTF-8: 0xEF 0xBF 0xBD). + * + * This function is not BOM-aware. + * + * @param input the UTF-16 string to convert + * @param length the length of the string in 2-byte code units (char16_t) + * @param utf8_buffer the pointer to buffer that can hold conversion result + * @return number of written code units + */ +simdutf_warn_unused size_t convert_utf16_to_utf8_with_replacement( + const char16_t *input, size_t length, char *utf8_buffer) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 size_t +convert_utf16_to_utf8_with_replacement( + std::span utf16_input, + detail::output_span_of_byte_like auto &&utf8_output) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + return scalar::utf16_to_utf8::convert_with_replacement( + utf16_input.data(), utf16_input.size(), utf8_output.data()); + } else + #endif + { + return convert_utf16_to_utf8_with_replacement( + utf16_input.data(), utf16_input.size(), + reinterpret_cast(utf8_output.data())); + } +} + #endif // SIMDUTF_SPAN +#endif // SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 + +#if SIMDUTF_FEATURE_UTF8 && SIMDUTF_FEATURE_UTF16 /** * Using native endianness, convert valid UTF-16 string into UTF-8 string. * @@ -10433,6 +10637,35 @@ maximal_binary_length_from_base64(InputPtr input, size_t length) noexcept { return actual_length / 4 * 3 + (actual_length % 4) - 1; } +// This function computes the binary length by iterating through the input +// and counting non-whitespace characters (excluding padding characters). +// We use a simple check (c > ' ') which is easy to parallelize and matches +// SIMD behavior. Only the last few characters are checked for padding '='. +template +simdutf_warn_unused simdutf_constexpr23 size_t +binary_length_from_base64(const char_type *input, size_t length) noexcept { + // Count non-whitespace characters (c > ' ') with loop unrolling + size_t count = 0; + for (size_t i = 0; i < length; i++) { + count += (input[i] > ' '); + } + + // Check for padding '=' at the end (at most 2 padding characters) + // Scan backwards, skipping whitespace, to find padding + size_t padding = 0; + size_t pos = length; + // Skip trailing whitespace + while (pos > 0 && padding < 2) { + char_type c = input[--pos]; + if (c == '=') { + padding++; + } else if (c > ' ') { + break; + } + } + return ((count - padding) * 3) / 4; +} + template simdutf_warn_unused simdutf_constexpr23 full_result base64_to_binary_details_impl( @@ -10726,6 +10959,71 @@ maximal_binary_length_from_base64(std::span input) noexcept { } #endif // SIMDUTF_SPAN +/** + * Compute the binary length from a base64 input. + * This function is useful for base64 inputs that may contain ASCII whitespaces + * (such as line breaks). For such inputs, the result is exact, and for any + * inputs the result can be used to size the output buffer passed to + * `base64_to_binary`. + * + * The function ignores whitespace and does not require padding characters + * ('='). + * + * @param input the base64 input to process + * @param length the length of the base64 input in bytes + * @return number of binary bytes + */ +simdutf_warn_unused size_t binary_length_from_base64(const char *input, + size_t length) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 size_t +binary_length_from_base64( + const detail::input_span_of_byte_like auto &input) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + return scalar::base64::binary_length_from_base64(input.data(), + input.size()); + } else + #endif + { + return binary_length_from_base64( + reinterpret_cast(input.data()), input.size()); + } +} + #endif // SIMDUTF_SPAN + +/** + * Compute the binary length from a base64 input. + * This function is useful for base64 inputs that may contain ASCII whitespaces + * (such as line breaks). For such inputs, the result is exact, and for any + * inputs the result can be used to size the output buffer passed to + * `base64_to_binary`. + * + * The function ignores whitespace and does not require padding characters + * ('='). + * + * @param input the base64 input to process, in ASCII stored as 16-bit + * units + * @param length the length of the base64 input in 16-bit units + * @return number of binary bytes + */ +simdutf_warn_unused size_t binary_length_from_base64(const char16_t *input, + size_t length) noexcept; + #if SIMDUTF_SPAN +simdutf_really_inline simdutf_warn_unused simdutf_constexpr23 size_t +binary_length_from_base64(std::span input) noexcept { + #if SIMDUTF_CPLUSPLUS23 + if consteval { + return scalar::base64::binary_length_from_base64(input.data(), + input.size()); + } else + #endif + { + return binary_length_from_base64(input.data(), input.size()); + } +} + #endif // SIMDUTF_SPAN + /** * Convert a base64 input to a binary output. * @@ -12187,6 +12485,44 @@ class implementation { convert_utf16be_to_utf8_with_errors(const char16_t *input, size_t length, char *utf8_buffer) const noexcept = 0; + /** + * Convert possibly broken UTF-16LE string into UTF-8 string, replacing + * unpaired surrogates with the Unicode replacement character U+FFFD. + * + * This function always succeeds: unpaired surrogates are replaced with + * U+FFFD (3 bytes in UTF-8: 0xEF 0xBF 0xBD). + * + * This function is not BOM-aware. + * + * @param input the UTF-16LE string to convert + * @param length the length of the string in 2-byte code units + * (char16_t) + * @param utf8_buffer the pointer to buffer that can hold conversion result + * @return number of written code units + */ + simdutf_warn_unused virtual size_t convert_utf16le_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept = 0; + + /** + * Convert possibly broken UTF-16BE string into UTF-8 string, replacing + * unpaired surrogates with the Unicode replacement character U+FFFD. + * + * This function always succeeds: unpaired surrogates are replaced with + * U+FFFD (3 bytes in UTF-8: 0xEF 0xBF 0xBD). + * + * This function is not BOM-aware. + * + * @param input the UTF-16BE string to convert + * @param length the length of the string in 2-byte code units + * (char16_t) + * @param utf8_buffer the pointer to buffer that can hold conversion result + * @return number of written code units + */ + simdutf_warn_unused virtual size_t convert_utf16be_to_utf8_with_replacement( + const char16_t *input, size_t length, + char *utf8_buffer) const noexcept = 0; + /** * Convert valid UTF-16LE string into UTF-8 string. * @@ -12922,6 +13258,38 @@ class implementation { simdutf_warn_unused size_t maximal_binary_length_from_base64( const char16_t *input, size_t length) const noexcept; + /** + * Compute the binary length from a base64 input with ASCII spaces. + * This function is useful for well-formed base64 inputs that may contain + * ASCII spaces (such as line breaks). For such inputs, the result is exact. + * + * The function counts non-whitespace characters (ASCII value > 0x20) and + * subtracts padding characters ('=') found at the end. + * + * @param input the base64 input to process + * @param length the length of the base64 input in bytes + * @return number of binary bytes + */ + simdutf_warn_unused virtual size_t + binary_length_from_base64(const char *input, size_t length) const noexcept; + + /** + * Compute the binary length from a base64 input with ASCII spaces. + * This function is useful for well-formed base64 inputs that may contain + * ASCII spaces (such as line breaks). For such inputs, the result is exact. + * + * The function counts non-whitespace characters (ASCII value > 0x20) and + * subtracts padding characters ('=') found at the end. + * + * @param input the base64 input to process, in ASCII stored as 16-bit + * units + * @param length the length of the base64 input in 16-bit units + * @return number of binary bytes + */ + simdutf_warn_unused virtual size_t + binary_length_from_base64(const char16_t *input, + size_t length) const noexcept; + /** * Convert a base64 input to a binary output. *