From e03c0371c78283f6638cc36750127829089459ab Mon Sep 17 00:00:00 2001 From: Jakub Onderka Date: Sun, 13 Sep 2026 12:43:24 +0200 Subject: [PATCH] Bump simdutf to 9.1.2 --- src/simdutf.cpp | 404 ++++++++++++++++++++++++++++++++++++------------ src/simdutf.h | 8 +- 2 files changed, 309 insertions(+), 103 deletions(-) diff --git a/src/simdutf.cpp b/src/simdutf.cpp index 4f88cdc..1b21ab5 100644 --- a/src/simdutf.cpp +++ b/src/simdutf.cpp @@ -1,4 +1,4 @@ -/* auto-generated on 2026-08-20 18:57:03.026391. Do not edit! */ +/* auto-generated on 2026-09-13 12:42:40.305120. Do not edit! */ /* begin file src/simdutf.cpp */ #include "simdutf.h" @@ -12006,7 +12006,7 @@ compress_decode_base64(char *dst, const char_type *src, size_t srclen, size_t full_input_length = ri.full_input_length; if (srclen == 0) { if (!ignore_garbage && equalsigns > 0) { - return {INVALID_BASE64_CHARACTER, equallocation, 0}; + return {INVALID_BASE64_CHARACTER, equallocation, 0, true}; } return {SUCCESS, full_input_length, 0}; } @@ -12166,7 +12166,8 @@ compress_decode_base64(char *dst, const char_type *src, size_t srclen, if (equalsigns > 0 && !ignore_garbage) { if ((size_t(dst - dstinit) % 3 == 0) || ((size_t(dst - dstinit) % 3) + 1 + equalsigns != 4)) { - return {INVALID_BASE64_CHARACTER, equallocation, size_t(dst - dstinit)}; + return {INVALID_BASE64_CHARACTER, equallocation, size_t(dst - dstinit), + true}; } } return {SUCCESS, srclen, size_t(dst - dstinit)}; @@ -15709,57 +15710,15 @@ size_t encode_base64(char *dst, const char *src, size_t srclen, return encode_base64_impl(dst, src, srclen, options); } -template -static inline uint64_t to_base64_mask(block64 *b, uint64_t *error, - uint64_t input_mask = UINT64_MAX) { +template +static inline uint64_t +to_base64_mask(block64 *b, uint64_t *error, const __m512i lookup0, + const __m512i lookup1, uint64_t input_mask = UINT64_MAX) { __m512i input = b->chunks[0]; const __m512i ascii_space_tbl = _mm512_set_epi8( 0, 0, 13, 12, 0, 10, 9, 0, 0, 0, 0, 0, 0, 0, 0, 32, 0, 0, 13, 12, 0, 10, 9, 0, 0, 0, 0, 0, 0, 0, 0, 32, 0, 0, 13, 12, 0, 10, 9, 0, 0, 0, 0, 0, 0, 0, 0, 32, 0, 0, 13, 12, 0, 10, 9, 0, 0, 0, 0, 0, 0, 0, 0, 32); - __m512i lookup0; - if (default_or_url) { - lookup0 = _mm512_set_epi8( - -128, -128, -128, -128, -128, -128, 61, 60, 59, 58, 57, 56, 55, 54, 53, - 52, 63, -128, 62, -128, 62, -128, -128, -128, -128, -128, -128, -128, - -128, -128, -128, -1, -128, -128, -128, -128, -128, -128, -128, -128, - -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -1, -128, - -128, -1, -1, -128, -128, -128, -128, -128, -128, -128, -128, -1); - } else if (base64_url) { - lookup0 = _mm512_set_epi8( - -128, -128, -128, -128, -128, -128, 61, 60, 59, 58, 57, 56, 55, 54, 53, - 52, -128, -128, 62, -128, -128, -128, -128, -128, -128, -128, -128, - -128, -128, -128, -128, -1, -128, -128, -128, -128, -128, -128, -128, - -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -1, - -128, -128, -1, -1, -128, -128, -128, -128, -128, -128, -128, -128, -1); - } else { - lookup0 = _mm512_set_epi8( - -128, -128, -128, -128, -128, -128, 61, 60, 59, 58, 57, 56, 55, 54, 53, - 52, 63, -128, -128, -128, 62, -128, -128, -128, -128, -128, -128, -128, - -128, -128, -128, -1, -128, -128, -128, -128, -128, -128, -128, -128, - -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -1, -128, - -128, -1, -1, -128, -128, -128, -128, -128, -128, -128, -128, -128); - } - __m512i lookup1; - if (default_or_url) { - lookup1 = _mm512_set_epi8( - -128, -128, -128, -128, -128, 51, 50, 49, 48, 47, 46, 45, 44, 43, 42, - 41, 40, 39, 38, 37, 36, 35, 34, 33, 32, 31, 30, 29, 28, 27, 26, -128, - 63, -128, -128, -128, -128, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16, 15, - 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0, -128); - } else if (base64_url) { - lookup1 = _mm512_set_epi8( - -128, -128, -128, -128, -128, 51, 50, 49, 48, 47, 46, 45, 44, 43, 42, - 41, 40, 39, 38, 37, 36, 35, 34, 33, 32, 31, 30, 29, 28, 27, 26, -128, - 63, -128, -128, -128, -128, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16, 15, - 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0, -128); - } else { - lookup1 = _mm512_set_epi8( - -128, -128, -128, -128, -128, 51, 50, 49, 48, 47, 46, 45, 44, 43, 42, - 41, 40, 39, 38, 37, 36, 35, 34, 33, 32, 31, 30, 29, 28, 27, 26, -128, - -128, -128, -128, -128, -128, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16, - 15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0, -128); - } const __m512i translated = _mm512_permutex2var_epi8(lookup0, input, lookup1); const __m512i combined = _mm512_or_si512(translated, input); @@ -15821,7 +15780,8 @@ static inline void load_block_partial(block64 *b, const char16_t *src, _mm512_permutexvar_epi64(_mm512_setr_epi64(0, 2, 4, 6, 1, 3, 5, 7), p); } -static inline void base64_decode(char *out, __m512i str) { +// Pack 64 6-bit values into 48 output bytes (16 trailing bytes unused). +static inline __m512i base64_pack(__m512i str) { const __m512i merge_ab_and_bc = _mm512_maddubs_epi16(str, _mm512_set1_epi32(0x01400140)); const __m512i merged = @@ -15831,19 +15791,76 @@ static inline void base64_decode(char *out, __m512i str) { 52, 53, 54, 48, 49, 50, 44, 45, 46, 40, 41, 42, 36, 37, 38, 32, 33, 34, 28, 29, 30, 24, 25, 26, 20, 21, 22, 16, 17, 18, 12, 13, 14, 8, 9, 10, 4, 5, 6, 0, 1, 2); - const __m512i shuffled = _mm512_permutexvar_epi8(pack, merged); - _mm512_mask_storeu_epi8( - (__m512i *)out, 0xffffffffffff, - shuffled); // mask would be 0xffffffffffff since we write 48 bytes. + return _mm512_permutexvar_epi8(pack, merged); } + +// Write 64 bytes: 48 valid plus 16 that the next store (at out+48) overwrites. +// Callers must only use this when at least 16 more bytes of real output +// follow, so that a later store overwrites the 16 extra bytes. +static inline void base64_decode(char *out, __m512i str) { + _mm512_storeu_si512(reinterpret_cast<__m512i *>(out), base64_pack(str)); +} + +static inline void base64_decode_safe(char *out, __m512i str) { + _mm512_mask_storeu_epi8((__m512i *)out, 0xffffffffffff, base64_pack(str)); +} + // decode 64 bytes and output 48 bytes static inline void base64_decode_block(char *out, const char *src) { base64_decode(out, _mm512_loadu_si512(reinterpret_cast(src))); } +static inline void base64_decode_block_safe(char *out, const char *src) { + base64_decode_safe( + out, _mm512_loadu_si512(reinterpret_cast(src))); +} static inline void base64_decode_block(char *out, block64 *b) { base64_decode(out, b->chunks[0]); } +static inline void base64_decode_block_safe(char *out, block64 *b) { + base64_decode_safe(out, b->chunks[0]); +} + +template +static inline void load_base64_lookups(__m512i &lookup0, __m512i &lookup1) { + if (default_or_url) { + lookup0 = _mm512_set_epi8( + -128, -128, -128, -128, -128, -128, 61, 60, 59, 58, 57, 56, 55, 54, 53, + 52, 63, -128, 62, -128, 62, -128, -128, -128, -128, -128, -128, -128, + -128, -128, -128, -1, -128, -128, -128, -128, -128, -128, -128, -128, + -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -1, -128, + -128, -1, -1, -128, -128, -128, -128, -128, -128, -128, -128, -1); + lookup1 = _mm512_set_epi8( + -128, -128, -128, -128, -128, 51, 50, 49, 48, 47, 46, 45, 44, 43, 42, + 41, 40, 39, 38, 37, 36, 35, 34, 33, 32, 31, 30, 29, 28, 27, 26, -128, + 63, -128, -128, -128, -128, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16, 15, + 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0, -128); + } else if (base64_url) { + lookup0 = _mm512_set_epi8( + -128, -128, -128, -128, -128, -128, 61, 60, 59, 58, 57, 56, 55, 54, 53, + 52, -128, -128, 62, -128, -128, -128, -128, -128, -128, -128, -128, + -128, -128, -128, -128, -1, -128, -128, -128, -128, -128, -128, -128, + -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -1, + -128, -128, -1, -1, -128, -128, -128, -128, -128, -128, -128, -128, -1); + lookup1 = _mm512_set_epi8( + -128, -128, -128, -128, -128, 51, 50, 49, 48, 47, 46, 45, 44, 43, 42, + 41, 40, 39, 38, 37, 36, 35, 34, 33, 32, 31, 30, 29, 28, 27, 26, -128, + 63, -128, -128, -128, -128, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16, 15, + 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0, -128); + } else { + lookup0 = _mm512_set_epi8( + -128, -128, -128, -128, -128, -128, 61, 60, 59, 58, 57, 56, 55, 54, 53, + 52, 63, -128, -128, -128, 62, -128, -128, -128, -128, -128, -128, -128, + -128, -128, -128, -1, -128, -128, -128, -128, -128, -128, -128, -128, + -128, -128, -128, -128, -128, -128, -128, -128, -128, -128, -1, -128, + -128, -1, -1, -128, -128, -128, -128, -128, -128, -128, -128, -128); + lookup1 = _mm512_set_epi8( + -128, -128, -128, -128, -128, 51, 50, 49, 48, 47, 46, 45, 44, 43, 42, + 41, 40, 39, 38, 37, 36, 35, 34, 33, 32, 31, 30, 29, 28, 27, 26, -128, + -128, -128, -128, -128, -128, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16, + 15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0, -128); + } +} template @@ -15871,20 +15888,132 @@ compress_decode_base64(char *dst, const chartype *src, size_t srclen, const char *const dstinit = dst; const chartype *const srcend = src + srclen; + // A 64-byte store writes 16 bytes past the 48 valid bytes it produces, so it + // is only allowed when at least 16 more bytes of real output follow it. We + // establish that locally, by always closing a run of wide stores with a + // masked 48-byte store. Bounding it from srclen instead would be wrong: + // srclen still counts ignorable characters, so a whitespace-bearing input + // inflates the bound past the end of a correctly sized output buffer. + + __m512i lookup0, lookup1; + load_base64_lookups(lookup0, lookup1); + // figure out why block_size == 2 is sometimes best??? constexpr size_t block_size = 6; char buffer[block_size * 64]; char *bufferptr = buffer; if (srclen >= 64) { const chartype *const srcend64 = src + srclen - 64; + // 512 input bytes / iteration, matching Turbo's DS256×2 inner loop. + // DNS messages are ~350 bytes so they never enter; a whitespace hit in + // the first group would throw the work away. + constexpr size_t unroll = 8; + while (bufferptr == buffer && size_t(srcend - src) >= unroll * 64) { + block64 b0, b1, b2, b3, b4, b5, b6, b7; + load_block(&b0, src); + load_block(&b1, src + 64); + const __m512i t0 = + _mm512_permutex2var_epi8(lookup0, b0.chunks[0], lookup1); + const __m512i t1 = + _mm512_permutex2var_epi8(lookup0, b1.chunks[0], lookup1); + load_block(&b2, src + 128); + load_block(&b3, src + 192); + const __m512i t2 = + _mm512_permutex2var_epi8(lookup0, b2.chunks[0], lookup1); + const __m512i t3 = + _mm512_permutex2var_epi8(lookup0, b3.chunks[0], lookup1); + const __m512i c0 = _mm512_or_si512(t0, b0.chunks[0]); + const __m512i c1 = _mm512_or_si512(t1, b1.chunks[0]); + const __m512i c2 = _mm512_or_si512(t2, b2.chunks[0]); + const __m512i c3 = _mm512_or_si512(t3, b3.chunks[0]); + __m512i any = + _mm512_or_si512(_mm512_ternarylogic_epi32(c0, c1, c2, 0xfe), c3); + const __m512i p0 = base64_pack(t0); + const __m512i p1 = base64_pack(t1); + const __m512i p2 = base64_pack(t2); + const __m512i p3 = base64_pack(t3); + load_block(&b4, src + 256); + load_block(&b5, src + 320); + const __m512i t4 = + _mm512_permutex2var_epi8(lookup0, b4.chunks[0], lookup1); + const __m512i t5 = + _mm512_permutex2var_epi8(lookup0, b5.chunks[0], lookup1); + load_block(&b6, src + 384); + load_block(&b7, src + 448); + const __m512i t6 = + _mm512_permutex2var_epi8(lookup0, b6.chunks[0], lookup1); + const __m512i t7 = + _mm512_permutex2var_epi8(lookup0, b7.chunks[0], lookup1); + const __m512i c4 = _mm512_or_si512(t4, b4.chunks[0]); + const __m512i c5 = _mm512_or_si512(t5, b5.chunks[0]); + const __m512i c6 = _mm512_or_si512(t6, b6.chunks[0]); + const __m512i c7 = _mm512_or_si512(t7, b7.chunks[0]); + any = _mm512_or_si512( + any, + _mm512_or_si512(_mm512_ternarylogic_epi32(c4, c5, c6, 0xfe), c7)); + const __m512i p4 = base64_pack(t4); + const __m512i p5 = base64_pack(t5); + const __m512i p6 = base64_pack(t6); + const __m512i p7 = base64_pack(t7); + if (simdutf_unlikely(_mm512_movepi8_mask(any) != 0)) { + break; + } + // Overlapping 64-byte stores for all but the last of the group: a later + // store overwrites the extra 16. The last is masked so we do not write + // past the 384 valid bytes (error paths and base64_to_binary_safe + // require no garbage past the logical output). + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst), p0); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 48), p1); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 96), p2); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 144), p3); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 192), p4); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 240), p5); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 288), p6); + _mm512_mask_storeu_epi8(dst + 336, 0xffffffffffff, p7); + src += unroll * 64; + dst += unroll * 48; + } + // One leftover 256-byte group on large clean inputs (bing is 1808 B). + // DNS messages are ~350 B so they skip this probe. + if (srclen >= 1024 && bufferptr == buffer && size_t(srcend - src) >= 256) { + block64 b0, b1, b2, b3; + load_block(&b0, src); + load_block(&b1, src + 64); + const __m512i t0 = + _mm512_permutex2var_epi8(lookup0, b0.chunks[0], lookup1); + const __m512i t1 = + _mm512_permutex2var_epi8(lookup0, b1.chunks[0], lookup1); + load_block(&b2, src + 128); + load_block(&b3, src + 192); + const __m512i t2 = + _mm512_permutex2var_epi8(lookup0, b2.chunks[0], lookup1); + const __m512i t3 = + _mm512_permutex2var_epi8(lookup0, b3.chunks[0], lookup1); + const __m512i any = _mm512_or_si512( + _mm512_ternarylogic_epi32(_mm512_or_si512(t0, b0.chunks[0]), + _mm512_or_si512(t1, b1.chunks[0]), + _mm512_or_si512(t2, b2.chunks[0]), 0xfe), + _mm512_or_si512(t3, b3.chunks[0])); + const __m512i p0 = base64_pack(t0); + const __m512i p1 = base64_pack(t1); + const __m512i p2 = base64_pack(t2); + const __m512i p3 = base64_pack(t3); + if (simdutf_likely(_mm512_movepi8_mask(any) == 0)) { + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst), p0); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 48), p1); + _mm512_storeu_si512(reinterpret_cast<__m512i *>(dst + 96), p2); + _mm512_mask_storeu_epi8(dst + 144, 0xffffffffffff, p3); + src += 256; + dst += 192; + } + } while (src <= srcend64) { block64 b; load_block(&b, src); src += 64; uint64_t error = 0; uint64_t badcharmask = - to_base64_mask(&b, - &error); + to_base64_mask(&b, &error, lookup0, lookup1); if (!ignore_garbage && error) { src -= 64; size_t error_offset = _tzcnt_u64(error); @@ -15900,14 +16029,18 @@ compress_decode_base64(char *dst, const chartype *src, size_t srclen, copy_block(&b, bufferptr); bufferptr += 64; } else { - base64_decode_block(dst, &b); + // Always a masked 48-byte store on the 1-block path: a later invalid + // character must not leave the 16-byte overlap past outlen. + base64_decode_block_safe(dst, &b); dst += 48; } if (bufferptr >= (block_size - 1) * 64 + buffer) { - for (size_t i = 0; i < (block_size - 1); i++) { + for (size_t i = 0; i < (block_size - 2); i++) { base64_decode_block(dst, buffer + i * 64); dst += 48; } + base64_decode_block_safe(dst, buffer + (block_size - 2) * 64); + dst += 48; std::memcpy(buffer, buffer + (block_size - 1) * 64, 64); // 64 might be too much bufferptr -= (block_size - 1) * 64; @@ -15921,9 +16054,8 @@ compress_decode_base64(char *dst, const chartype *src, size_t srclen, block64 b; load_block_partial(&b, src, input_mask); uint64_t error = 0; - uint64_t badcharmask = - to_base64_mask(&b, &error, - input_mask); + uint64_t badcharmask = to_base64_mask(&b, &error, lookup0, + lookup1, input_mask); if (!ignore_garbage && error) { size_t error_offset = _tzcnt_u64(error); return {error_code::INVALID_BASE64_CHARACTER, @@ -15935,7 +16067,11 @@ compress_decode_base64(char *dst, const chartype *src, size_t srclen, char *buffer_start = buffer; for (; buffer_start + 64 <= bufferptr; buffer_start += 64) { - base64_decode_block(dst, buffer_start); + if (buffer_start + 128 <= bufferptr) { + base64_decode_block(dst, buffer_start); + } else { + base64_decode_block_safe(dst, buffer_start); + } dst += 48; } if ((bufferptr - buffer_start) != 0) { @@ -16116,7 +16252,7 @@ simdutf_warn_unused size_t icelake_binary_length_from_base64(const char *input, const char *end = input + length; __m512i spaces = _mm512_set1_epi8(0x20); - while (ptr + 64 <= end) { + while (size_t(end - ptr) >= 64) { __m512i data = _mm512_loadu_si512(reinterpret_cast(ptr)); uint64_t mask = _mm512_cmpgt_epi8_mask(data, spaces); count += count_ones(mask); @@ -16153,7 +16289,7 @@ icelake_binary_length_from_base64(const char16_t *input, size_t length) { const char16_t *end = input + length; __m512i spaces = _mm512_set1_epi16(0x20); - while (ptr + 32 <= end) { + while (size_t(end - ptr) >= 32) { __m512i data = _mm512_loadu_si512(reinterpret_cast(ptr)); __mmask32 mask = _mm512_cmpgt_epi16_mask(data, spaces); count += _mm_popcnt_u32(mask); @@ -16353,6 +16489,32 @@ implementation::validate_utf8(const char *buf, size_t len) const noexcept { avx512_utf8_checker checker{}; const char *ptr = buf; const char *end = ptr + len; + // Get the 512-bit reads onto a 64-byte boundary. A load whose address + // straddles a cache line costs two accesses, and callers rarely hand us an + // aligned buffer. + // + // We cannot simply mask-load a short head block to reach the boundary: the + // checker carries state from one block to the next, and zero padding in the + // middle of a character would read as a truncated sequence. Instead we + // consume one full (unaligned) block and re-seed the cross-block state from + // the three bytes preceding the aligned start. Those three bytes must lie + // inside the buffer, hence the requirement that the adjustment be at least + // three. Below a couple of kilobytes the fixed cost of the prologue is not + // repaid. + if (len >= 2048) { + const uintptr_t misalignment = reinterpret_cast(ptr) % 64; + if (misalignment != 0 && misalignment <= 61) { + const size_t adjustment = 64 - misalignment; + checker.check_next_input(_mm512_loadu_si512((const __m512i *)ptr)); + ptr += adjustment; + // Only the top three lanes are read. Masked-out lanes never fault, so + // this is safe even though ptr - 64 may point before buf. + const __m512i prev3 = _mm512_maskz_loadu_epi8( + UINT64_C(0xE000000000000000), (const __m512i *)(ptr - 64)); + checker.prev_input_block = prev3; + checker.prev_incomplete = is_incomplete(prev3); + } + } for (; end - ptr >= 64; ptr += 64) { const __m512i utf8 = _mm512_loadu_si512((const __m512i *)ptr); checker.check_next_input(utf8); @@ -16375,35 +16537,70 @@ simdutf_warn_unused result implementation::validate_utf8_with_errors( const char *ptr = buf; const char *end = ptr + len; size_t count{0}; + // Largest prefix that a clean error check has already cleared. On failure it + // is handed to the scalar rewind, which re-validates forward from there to + // the end of the buffer, so naming a position earlier than the error only + // costs scalar work on the error path. + size_t safe{0}; + // Get the 512-bit reads onto a 64-byte boundary. A load whose address is not + // aligned touches two cache lines and costs two accesses, and callers rarely + // hand us an aligned buffer. + // + // The head has to be a full block rather than a masked one: the checker + // carries state from one block to the next, and zero padding in the middle + // of a character would read as a truncated sequence. The cross-block state + // is then re-seeded from the three bytes preceding the aligned start, which + // must be inside the buffer, hence the misalignment <= 61 guard. + if (len >= 2048) { + const uintptr_t misalignment = reinterpret_cast(ptr) % 64; + if (misalignment != 0 && misalignment <= 61) { + const size_t adjustment = 64 - misalignment; + checker.check_next_input(_mm512_loadu_si512((const __m512i *)ptr)); + if (simdutf_unlikely(checker.errors())) { + return scalar::utf8::rewind_and_validate_with_errors(buf, buf, len); + } + ptr += adjustment; + count = adjustment; + // Only the top three lanes are read. Masked-out lanes never fault, so + // this is safe even though ptr - 64 may point before buf. + const __m512i prev3 = _mm512_maskz_loadu_epi8( + UINT64_C(0xE000000000000000), (const __m512i *)(ptr - 64)); + checker.prev_input_block = prev3; + checker.prev_incomplete = is_incomplete(prev3); + } + } + // checker.error is a sticky OR-accumulator, so it does not have to be tested + // every 64 bytes. Testing it every eighth block takes a vptestmb, a ktest + // and a branch out of the hot loop; an error is then handed to the scalar + // rewind at most nine blocks early, which only lengthens the rare error + // path. + unsigned since = 0; for (; end - ptr >= 64; ptr += 64) { const __m512i utf8 = _mm512_loadu_si512((const __m512i *)ptr); checker.check_next_input(utf8); - if (checker.errors()) { - if (count != 0) { - count--; - } // Sometimes the error is only detected in the next chunk - result res = scalar::utf8::rewind_and_validate_with_errors( - reinterpret_cast(buf), - reinterpret_cast(buf + count), len - count); - res.count += count; - return res; - } count += 64; + if (++since == 8) { + since = 0; + if (simdutf_unlikely(checker.errors())) { + break; + } + safe = count >= 64 ? count - 64 : 0; + } } - if (end != ptr) { + if (!checker.errors() && end != ptr) { const __m512i utf8 = _mm512_maskz_loadu_epi8( ~UINT64_C(0) >> (64 - (end - ptr)), (const __m512i *)ptr); checker.check_next_input(utf8); } checker.check_eof(); if (checker.errors()) { - if (count != 0) { - count--; + if (safe != 0) { + safe--; } // Sometimes the error is only detected in the next chunk result res = scalar::utf8::rewind_and_validate_with_errors( reinterpret_cast(buf), - reinterpret_cast(buf + count), len - count); - res.count += count; + reinterpret_cast(buf + safe), len - safe); + res.count += safe; return res; } return result(error_code::SUCCESS, len); @@ -17679,7 +17876,7 @@ simdutf_warn_unused size_t avx2_binary_length_from_base64(const char *input, const char *end = input + length; __m256i spaces = _mm256_set1_epi8(0x20); - while (ptr + 32 <= end) { + while (size_t(end - ptr) >= 32) { __m256i data = _mm256_loadu_si256(reinterpret_cast(ptr)); __m256i gt_space = _mm256_cmpgt_epi8(data, spaces); uint32_t mask = static_cast(_mm256_movemask_epi8(gt_space)); @@ -17712,7 +17909,7 @@ simdutf_warn_unused size_t avx2_binary_length_from_base64(const char16_t *input, const char16_t *end = input + length; __m256i spaces = _mm256_set1_epi16(0x20); - while (ptr + 16 <= end) { + while (size_t(end - ptr) >= 16) { __m256i data = _mm256_loadu_si256(reinterpret_cast(ptr)); __m256i gt_space = _mm256_cmpgt_epi16(data, spaces); uint32_t mask = static_cast(_mm256_movemask_epi8(gt_space)); @@ -21134,10 +21331,13 @@ static simdutf_really_inline vector_u8 decoding_pack(vector_u8 input) { const auto tmp = as_vector_u8(t4); + // The last four lanes are padding: pull them from a zero vector rather + // than from tmp, as the 16-byte store in base64_decode would otherwise + // write garbage past the 12 bytes we produce. const auto shuffle = - vector_u8(1, 2, 3, 5, 6, 7, 9, 10, 11, 13, 14, 15, 0, 0, 0, 0); + vector_u8(1, 2, 3, 5, 6, 7, 9, 10, 11, 13, 14, 15, 16, 16, 16, 16); - const auto t = shuffle.lookup_16(tmp); + const auto t = shuffle.lookup_32(tmp, vector_u8::zero()); return t; #else @@ -21158,10 +21358,13 @@ static simdutf_really_inline vector_u8 decoding_pack(vector_u8 input) { const auto tmp = as_vector_u8(t4); + // The last four lanes are padding: pull them from a zero vector rather + // than from tmp, as the 16-byte store in base64_decode would otherwise + // write garbage past the 12 bytes we produce. const auto shuffle = - vector_u8(2, 1, 0, 6, 5, 4, 10, 9, 8, 14, 13, 12, 0, 0, 0, 0); + vector_u8(2, 1, 0, 6, 5, 4, 10, 9, 8, 14, 13, 12, 16, 16, 16, 16); - const auto t = shuffle.lookup_16(tmp); + const auto t = shuffle.lookup_32(tmp, vector_u8::zero()); return t; #endif // SIMDUTF_IS_BIG_ENDIAN @@ -26594,8 +26797,12 @@ static inline void base64_decode(char *out, __m256i str) { __m256i pack_shuffle = ____m256i( (__m128i)v16u8{3, 2, 1, 7, 6, 5, 11, 10, 9, 15, 14, 13, 0, 0, 0, 0}); t3 = __lasx_xvshuf_b(t3, t3, (__m256i)pack_shuffle); + t3 = __lasx_xvinsgr2vr_w(t3, 0, 7); - // Store the output: + // Two 16-byte stores write 28 bytes: the 24 bytes of output followed by four + // zero bytes. Callers that cannot spare those four bytes must go through + // base64_decode_block_safe. The bulk loop can: it only takes this path while + // dst is below end_of_safe_64byte_zone, which leaves 63 bytes of room. __lsx_vst(lasx_extracti128_lo(t3), out, 0); __lsx_vst(lasx_extracti128_hi(t3), out, 12); } @@ -26607,11 +26814,9 @@ static inline void base64_decode_block(char *out, const char *src) { } static inline void base64_decode_block_safe(char *out, const char *src) { - base64_decode(out, __lasx_xvld(reinterpret_cast(src), 0)); - alignas(32) char buffer[32]; - base64_decode(buffer, - __lasx_xvld(reinterpret_cast(src), 32)); - std::memcpy(out + 24, buffer, 24); + alignas(32) char buffer[64]; + base64_decode_block(buffer, src); + std::memcpy(out, buffer, 48); } static inline void base64_decode_block(char *out, block64 *b) { @@ -26619,10 +26824,9 @@ static inline void base64_decode_block(char *out, block64 *b) { base64_decode(out + 24, b->chunks[1]); } static inline void base64_decode_block_safe(char *out, block64 *b) { - base64_decode(out, b->chunks[0]); - alignas(32) char buffer[32]; - base64_decode(buffer, b->chunks[1]); - std::memcpy(out + 24, buffer, 24); + alignas(32) char buffer[64]; + base64_decode_block(buffer, b); + std::memcpy(out, buffer, 48); } template 0) { - return {INVALID_BASE64_CHARACTER, equallocation, 0}; + return {INVALID_BASE64_CHARACTER, equallocation, 0, true}; } return {SUCCESS, full_input_length, 0}; } @@ -26812,7 +27016,8 @@ compress_decode_base64(char *dst, const chartype *src, size_t srclen, if (equalsigns > 0 && !ignore_garbage) { if ((size_t(dst - dstinit) % 3 == 0) || ((size_t(dst - dstinit) % 3) + 1 + equalsigns != 4)) { - return {INVALID_BASE64_CHARACTER, equallocation, size_t(dst - dstinit)}; + return {INVALID_BASE64_CHARACTER, equallocation, size_t(dst - dstinit), + true}; } } return {SUCCESS, srclen, size_t(dst - dstinit)}; @@ -28903,7 +29108,7 @@ compress_decode_base64(char *dst, const char_type *src, size_t srclen, size_t full_input_length = ri.full_input_length; if (srclen == 0) { if (!ignore_garbage && equalsigns > 0) { - return {INVALID_BASE64_CHARACTER, equallocation, 0}; + return {INVALID_BASE64_CHARACTER, equallocation, 0, true}; } return {SUCCESS, full_input_length, 0}; } @@ -29058,7 +29263,8 @@ compress_decode_base64(char *dst, const char_type *src, size_t srclen, if (equalsigns > 0 && !ignore_garbage) { if ((size_t(dst - dstinit) % 3 == 0) || ((size_t(dst - dstinit) % 3) + 1 + equalsigns != 4)) { - return {INVALID_BASE64_CHARACTER, equallocation, size_t(dst - dstinit)}; + return {INVALID_BASE64_CHARACTER, equallocation, size_t(dst - dstinit), + true}; } } return {SUCCESS, srclen, size_t(dst - dstinit)}; diff --git a/src/simdutf.h b/src/simdutf.h index 8a8b4c5..774ce3c 100644 --- a/src/simdutf.h +++ b/src/simdutf.h @@ -1,4 +1,4 @@ -/* auto-generated on 2026-08-20 18:57:03.026391. Do not edit! */ +/* auto-generated on 2026-09-13 12:42:40.305120. Do not edit! */ /* begin file include/simdutf.h */ #ifndef SIMDUTF_H #define SIMDUTF_H @@ -172,7 +172,7 @@ #elif defined(__aarch64__) || defined(_M_ARM64) || defined(_M_ARM64EC) #define SIMDUTF_IS_ARM64 1 #elif defined(__PPC64__) || defined(_M_PPC64) - #if defined(__VEC__) && defined(__ALTIVEC__) + #if defined(__VEC__) && defined(__ALTIVEC__) && defined(__POWER8_VECTOR__) #define SIMDUTF_IS_PPC64 1 #endif #elif defined(__s390__) @@ -941,7 +941,7 @@ SIMDUTF_DISABLE_UNDESIRED_WARNINGS #define SIMDUTF_SIMDUTF_VERSION_H /** The version of simdutf being used (major.minor.revision) */ -#define SIMDUTF_VERSION "9.1.0" +#define SIMDUTF_VERSION "9.1.2" namespace simdutf { enum { @@ -956,7 +956,7 @@ enum { /** * The revision (major.minor.REVISION) of simdutf being used. */ - SIMDUTF_VERSION_REVISION = 0 + SIMDUTF_VERSION_REVISION = 2 }; } // namespace simdutf