1#include "native/simd/common_body.h"
3namespace native::detail::NATIVE_BACKEND {
4 template<
class T>
using swizzle_word = std::conditional_t<
sizeof(T)==1,std::uint8_t,
5 std::conditional_t<
sizeof(T)==2,std::uint16_t,std::conditional_t<
sizeof(T)==4,std::uint32_t,std::uint64_t>>>;
6 template<
class T>
using swizzle_native = swizzle_word<T> __attribute__((ext_vector_type(4)));
7 template<std::size_t... I>
struct swizzle {
8 static constexpr std::size_t size=
sizeof...(I);
9 template<
class Self>
using result = std::conditional_t<size==1,
typename Self::value_type,
10 simd<typename Self::value_type,size,Self::architecture>>;
11 static constexpr bool unique=[] {
12 constexpr std::size_t indices[]{I...};
13 for(std::size_t i=0;i<size;++i)
14 for(std::size_t j=0;j<i;++j)
if(indices[i]==indices[j])
return false;
17 template<std::
size_t D>
static consteval int slot() {
18 constexpr std::size_t indices[]{I...};
19 for(std::size_t i=0;i<size;++i)
if(indices[i]==D)
return int(4+i);
22 template<
class Self>
static consteval bool readable() {
23 using T=
typename Self::value_type;
24 if constexpr(!((I<Self::lanes)&&...) || !std::is_trivially_copyable_v<T> ||
25 !std::default_initializable<T> || !(
sizeof(T)==1 ||
sizeof(T)==2 ||
sizeof(T)==4 ||
sizeof(T)==8))
return false;
26 else if constexpr(!
requires(Self
const & self,T * p) { self.template store_memory<1>(p); })
return false;
27 else if constexpr(size==1)
return std::copy_constructible<T>;
28 else return requires(T
const * p) {
30 { result<Self>::template load_memory<1>(p) } -> std::same_as<result<Self>>;
33 template<
class Self>
static consteval bool writable() {
34 if constexpr(std::is_const_v<Self> || !unique || !readable<Self>())
return false;
35 else if constexpr(size>1 && !
requires(result<Self>
const & rhs,
typename Self::value_type * p) { rhs.template store_memory<1>(p); })
return false;
36 else return std::is_copy_assignable_v<typename Self::value_type> &&
37 std::is_copy_constructible_v<result<Self>> &&
requires(Self & self,
typename Self::value_type
const * p) {
38 { Self::template load_memory<1>(p) } -> std::same_as<Self>;
39 self=Self::template load_memory<1>(p);
42 template<
class V>
static constexpr bool native_words=[] {
43 using T=
typename V::value_type;
44 if constexpr(simd_custom_element<T>)
return false;
45 else if constexpr(
requires {
typename V::native_type; })
return sizeof(
typename V::native_type)==
sizeof(swizzle_native<T>);
48 template<
class Self>
static native_inline constexpr auto words(Self
const & self) {
49 using T=
typename Self::value_type;
50 if constexpr(native_words<Self>) {
51 return __builtin_bit_cast(swizzle_native<T>,self.to_native());
53 std::array<T,4> values{};
54 self.template store_memory<1>(values.data());
55 return __builtin_bit_cast(swizzle_native<T>,values);
58 template<
class V>
static native_inline constexpr V from_words(swizzle_native<typename V::value_type> words) {
59 using T=
typename V::value_type;
60 if constexpr(native_words<V>) {
61 return V::from_native(__builtin_bit_cast(
typename V::native_type,words));
63 auto values=__builtin_bit_cast(std::array<T,4>,words);
64 return V::template load_memory<1>(values.data());
69 using T=
typename Self::value_type;
70 auto value=words(self);
71 if constexpr(size==1) {
72 constexpr std::size_t indices[]{I...};
73 return std::bit_cast<T>(swizzle_word<T>(value[indices[0]]));
75 constexpr int indices[]{int(I)...};
76 auto shuffled=__builtin_shufflevector(value,value,
77 indices[0],indices[1],indices[2%size],indices[3%size]);
78 return from_words<result<Self>>(shuffled);
82 native_inline constexpr static result<Self> write(Self & self,result<Self> rhs) {
83 using T=
typename Self::value_type;
84 auto before=words(self);
85 auto replacement=[&] {
86 if constexpr(size==1)
return swizzle_native<T>{std::bit_cast<swizzle_word<T>>(rhs),0,0,0};
87 else return words(rhs);
89 auto shuffled=__builtin_shufflevector(before,replacement,slot<0>(),slot<1>(),slot<2>(),slot<3>());
90 self=from_words<Self>(shuffled);
95namespace native::detail {
98 template<
class T,std::
size_t N,::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) &&(N<=4)
99 struct swizzle_access<T,N,Arch> {
100 template<std::
size_t K>
using result = std::conditional_t<K==1,T,simd<T,K,Arch>>;
101#define NATIVE_SWIZZLE_FIELD(NAME,K,...) \
102 template<class Self> requires(::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::template readable<Self>()) \
103 native_nodiscard native_inline constexpr result<K> get_##NAME(this Self const & self) { return ::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::read(self); } \
104 template<class Self> requires(::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::template writable<Self>()) \
105 native_inline constexpr result<K> set_##NAME(this Self & self,result<K> rhs) { return ::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::write(self,rhs); } \
106 __declspec(property(get=get_##NAME,put=set_##NAME)) result<K> NAME;
107#define NATIVE_SWIZZLE_ROW2(A,I,X,Y,Z,W) \
108 NATIVE_SWIZZLE_FIELD(A##X,2,I,0) NATIVE_SWIZZLE_FIELD(A##Y,2,I,1) \
109 NATIVE_SWIZZLE_FIELD(A##Z,2,I,2) NATIVE_SWIZZLE_FIELD(A##W,2,I,3)
110#define NATIVE_SWIZZLE_ROW3(A,I,B,J,X,Y,Z,W) \
111 NATIVE_SWIZZLE_FIELD(A##B##X,3,I,J,0) NATIVE_SWIZZLE_FIELD(A##B##Y,3,I,J,1) \
112 NATIVE_SWIZZLE_FIELD(A##B##Z,3,I,J,2) NATIVE_SWIZZLE_FIELD(A##B##W,3,I,J,3)
113#define NATIVE_SWIZZLE_PLANE3(A,I,X,Y,Z,W) \
114 NATIVE_SWIZZLE_ROW3(A,I,X,0,X,Y,Z,W) NATIVE_SWIZZLE_ROW3(A,I,Y,1,X,Y,Z,W) \
115 NATIVE_SWIZZLE_ROW3(A,I,Z,2,X,Y,Z,W) NATIVE_SWIZZLE_ROW3(A,I,W,3,X,Y,Z,W)
116#define NATIVE_SWIZZLE_ROW4(A,I,B,J,C,K,X,Y,Z,W) \
117 NATIVE_SWIZZLE_FIELD(A##B##C##X,4,I,J,K,0) NATIVE_SWIZZLE_FIELD(A##B##C##Y,4,I,J,K,1) \
118 NATIVE_SWIZZLE_FIELD(A##B##C##Z,4,I,J,K,2) NATIVE_SWIZZLE_FIELD(A##B##C##W,4,I,J,K,3)
119#define NATIVE_SWIZZLE_PLANE4(A,I,B,J,X,Y,Z,W) \
120 NATIVE_SWIZZLE_ROW4(A,I,B,J,X,0,X,Y,Z,W) NATIVE_SWIZZLE_ROW4(A,I,B,J,Y,1,X,Y,Z,W) \
121 NATIVE_SWIZZLE_ROW4(A,I,B,J,Z,2,X,Y,Z,W) NATIVE_SWIZZLE_ROW4(A,I,B,J,W,3,X,Y,Z,W)
122#define NATIVE_SWIZZLE_CUBE4(A,I,X,Y,Z,W) \
123 NATIVE_SWIZZLE_PLANE4(A,I,X,0,X,Y,Z,W) NATIVE_SWIZZLE_PLANE4(A,I,Y,1,X,Y,Z,W) \
124 NATIVE_SWIZZLE_PLANE4(A,I,Z,2,X,Y,Z,W) NATIVE_SWIZZLE_PLANE4(A,I,W,3,X,Y,Z,W)
125#define NATIVE_SWIZZLE4(X,Y,Z,W) \
126 NATIVE_SWIZZLE_FIELD(X,1,0) NATIVE_SWIZZLE_FIELD(Y,1,1) NATIVE_SWIZZLE_FIELD(Z,1,2) NATIVE_SWIZZLE_FIELD(W,1,3) \
127 NATIVE_SWIZZLE_ROW2(X,0,X,Y,Z,W) NATIVE_SWIZZLE_ROW2(Y,1,X,Y,Z,W) NATIVE_SWIZZLE_ROW2(Z,2,X,Y,Z,W) NATIVE_SWIZZLE_ROW2(W,3,X,Y,Z,W) \
128 NATIVE_SWIZZLE_PLANE3(X,0,X,Y,Z,W) NATIVE_SWIZZLE_PLANE3(Y,1,X,Y,Z,W) NATIVE_SWIZZLE_PLANE3(Z,2,X,Y,Z,W) NATIVE_SWIZZLE_PLANE3(W,3,X,Y,Z,W) \
129 NATIVE_SWIZZLE_CUBE4(X,0,X,Y,Z,W) NATIVE_SWIZZLE_CUBE4(Y,1,X,Y,Z,W) NATIVE_SWIZZLE_CUBE4(Z,2,X,Y,Z,W) NATIVE_SWIZZLE_CUBE4(W,3,X,Y,Z,W)
130 NATIVE_SWIZZLE4(x,y,z,w)
131#undef NATIVE_SWIZZLE4
132#undef NATIVE_SWIZZLE_CUBE4
133#undef NATIVE_SWIZZLE_PLANE4
134#undef NATIVE_SWIZZLE_ROW4
135#undef NATIVE_SWIZZLE_PLANE3
136#undef NATIVE_SWIZZLE_ROW3
137#undef NATIVE_SWIZZLE_ROW2
138#undef NATIVE_SWIZZLE_FIELD
147#if NATIVE_HAS_AVX2 || NATIVE_HAS_AVX512F
149#if NATIVE_HAS_ARM_NEON
153 namespace detail::NATIVE_BACKEND {
155 template<std::
size_t N>
inline constexpr bool float_shape = N==1
156#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
157 || N==2 || N==3 || N==4
162#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
166 template <
class T>
inline constexpr std::size_t mask_lane_bytes = [] {
167 if constexpr (::native::simd_custom_element<T>)
return sizeof(typename ::native::simd_traits<T>::storage_type);
168 else return sizeof(T);
170 template <
class T>
concept mask_element = ::native::simd_custom_element<T> ||
171 (std::is_arithmetic_v<T> && !std::is_const_v<T> && !std::is_volatile_v<T> &&
172 (
sizeof(T)==1 ||
sizeof(T)==2 ||
sizeof(T)==4 ||
sizeof(T)==8));
173 template <std::
size_t B,std::
size_t N>
inline constexpr bool mask_compact =
174#if NATIVE_HAS_AVX512F
176#if NATIVE_HAS_AVX512BW
180#
if NATIVE_HAS_AVX512VL
181 || B*N==16 || B*N==32
187 template <std::
size_t B,std::
size_t N>
inline constexpr bool mask_shape = N==1 || (B*N==64 && mask_compact<B,N>)
189 || B*N==16 || B*N==32
191#
if NATIVE_HAS_ARM_NEON
195 template <std::
size_t N>
inline constexpr std::uint64_t mask_low_bits = [] {
196 if constexpr (N==64)
return ~std::uint64_t(0);
197 else return (std::uint64_t(1)<<N)-1;
199 template <std::
size_t B>
using mask_word = std::conditional_t<B==1,std::uint8_t,
200 std::conditional_t<B==2,std::uint16_t,std::conditional_t<B==4,std::uint32_t,std::uint64_t>>>;
202 template<
class U>
struct mask_scalar_ops {
204 static native_inline native_const constexpr native_type normalize(native_type x)
noexcept {
return x?native_type(~U(0)):U(0); }
215 template <std::
size_t N>
struct mask_compact_ops {
216 using native_type = std::conditional_t<(N<=8),std::uint8_t,std::conditional_t<(N<=16),std::uint16_t,
217 std::conditional_t<(N<=32),std::uint32_t,std::uint64_t>>>;
222 if constexpr(N==8 || N==16 || N==32 || N==64)
return a;
223 else return native_type(std::uint64_t(a)&mask_low_bits<N>);
227 if (std::is_constant_evaluated())
return native_type(a&b);
228#if NATIVE_HAS_AVX512F
229#if NATIVE_HAS_AVX512DQ
230 if constexpr(N<=8)
return _kand_mask8(a,b);
233 if constexpr(N<=16)
return native_type(_kand_mask16(a,b));
234#if NATIVE_HAS_AVX512BW
235 else if constexpr(N<=32)
return _kand_mask32(a,b);
236 else return _kand_mask64(a,b);
238 else return native_type(a&b);
241 return native_type(a&b);
245 if (std::is_constant_evaluated())
return native_type(a|b);
246#if NATIVE_HAS_AVX512F
247#if NATIVE_HAS_AVX512DQ
248 if constexpr(N<=8)
return _kor_mask8(a,b);
251 if constexpr(N<=16)
return native_type(_kor_mask16(a,b));
252#if NATIVE_HAS_AVX512BW
253 else if constexpr(N<=32)
return _kor_mask32(a,b);
254 else return _kor_mask64(a,b);
256 else return native_type(a|b);
259 return native_type(a|b);
263 if (std::is_constant_evaluated())
return native_type(a^b);
264#if NATIVE_HAS_AVX512F
265#if NATIVE_HAS_AVX512DQ
266 if constexpr(N<=8)
return _kxor_mask8(a,b);
269 if constexpr(N<=16)
return native_type(_kxor_mask16(a,b));
270#if NATIVE_HAS_AVX512BW
271 else if constexpr(N<=32)
return _kxor_mask32(a,b);
272 else return _kxor_mask64(a,b);
274 else return native_type(a^b);
277 return native_type(a^b);
281 if(std::is_constant_evaluated())
return native_type((~std::uint64_t(a))&mask_low_bits<N>);
283 if constexpr(N<8)
return bit_xor(a,native_type(mask_low_bits<N>));
284#if NATIVE_HAS_AVX512F
285#if NATIVE_HAS_AVX512DQ
286 else if constexpr(N==8)
return _knot_mask8(a);
288 else if constexpr(N<=16)
return native_type(_knot_mask16(a));
289#if NATIVE_HAS_AVX512BW
290 else if constexpr(N<=32)
return _knot_mask32(a);
291 else return _knot_mask64(a);
293 else return native_type(~a);
296 else return native_type(~a);
302 static native_inline native_const constexpr native_type from_bits(std::uint64_t a)
noexcept {
return normalize(native_type(a)); }
304 template <std::
size_t B,std::
size_t N>
struct mask_vector_ops {
306 using native_type = std::conditional_t<B*N==16,__m128i,__m256i>;
309 std::array<mask_word<B>,N> lanes{};
310 lanes.fill(a?mask_word<B>(~mask_word<B>(0)):0);
311 return __builtin_bit_cast(native_type,lanes);
313 if constexpr (B*N==16)
return _mm_set1_epi32(a?-1:0);
314 else return _mm256_set1_epi32(a?-1:0);
318 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
319 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
320 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]&y[i]);
321 return __builtin_bit_cast(native_type,x);
323 if constexpr (B*N==16)
return _mm_and_si128(a,b);
324 else return _mm256_and_si256(a,b);
328 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
329 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
330 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]|y[i]);
331 return __builtin_bit_cast(native_type,x);
333 if constexpr (B*N==16)
return _mm_or_si128(a,b);
334 else return _mm256_or_si256(a,b);
338 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
339 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
340 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]^y[i]);
341 return __builtin_bit_cast(native_type,x);
343 if constexpr (B*N==16)
return _mm_xor_si128(a,b);
344 else return _mm256_xor_si256(a,b);
349 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
350 for(
auto & lane:lanes) lane=lane?mask_word<B>(~mask_word<B>(0)):0;
351 return __builtin_bit_cast(native_type,lanes);
354 if constexpr (B*N==16) {
355 if constexpr (B==1)
return bit_not(_mm_cmpeq_epi8(a,z));
356 else if constexpr (B==2)
return bit_not(_mm_cmpeq_epi16(a,z));
357 else if constexpr (B==4)
return bit_not(_mm_cmpeq_epi32(a,z));
358 else return bit_not(_mm_cmpeq_epi64(a,z));
360 if constexpr (B==1)
return bit_not(_mm256_cmpeq_epi8(a,z));
361 else if constexpr (B==2)
return bit_not(_mm256_cmpeq_epi16(a,z));
362 else if constexpr (B==4)
return bit_not(_mm256_cmpeq_epi32(a,z));
363 else return bit_not(_mm256_cmpeq_epi64(a,z));
368 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
369 for(
auto lane:lanes)
if(lane!=0)
return true;
372 if constexpr (B*N==16)
return _mm_testz_si128(a,a)==0;
373 else return _mm256_testz_si256(a,a)==0;
377 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
378 for(
auto lane:lanes)
if(lane!=mask_word<B>(~mask_word<B>(0)))
return false;
381 if constexpr (B*N==16)
return _mm_movemask_epi8(a)==0xffff;
382 else return _mm256_movemask_epi8(a)==-1;
386 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
387 std::uint64_t result=0;
388 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(lanes[i]!=0)<<i;
392 if constexpr (B*N==16) bytes=std::uint32_t(_mm_movemask_epi8(a));
393 else bytes=std::uint32_t(_mm256_movemask_epi8(a));
394 std::uint64_t result=0;
395 for(std::size_t i=0;i<N;++i) result|=std::uint64_t((bytes>>(i*B))&1)<<i;
400 std::array<mask_word<B>,N> lanes{};
401 for(std::size_t i=0;i<N;++i) lanes[i]=((bits>>i)&1)?mask_word<B>(~mask_word<B>(0)):0;
402 return __builtin_bit_cast(native_type,lanes);
404 std::array<mask_word<B>,N> a{};
405 for(std::size_t i=0;i<N;++i) a[i]=((bits>>i)&1)?mask_word<B>(~mask_word<B>(0)):mask_word<B>(0);
406 if constexpr (B*N==16)
return _mm_loadu_si128(
reinterpret_cast<__m128i
const *
>(a.data()));
407 else return _mm256_loadu_si256(
reinterpret_cast<__m256i
const *
>(a.data()));
409#elif NATIVE_HAS_ARM_NEON
410 using native_type = uint8x16_t;
413 std::array<mask_word<B>,N> lanes{};
414 lanes.fill(a?mask_word<B>(~mask_word<B>(0)):0);
415 return __builtin_bit_cast(native_type,lanes);
417 return vdupq_n_u8(a?0xff:0);
421 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
422 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
423 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]&y[i]);
424 return __builtin_bit_cast(native_type,x);
426 return vandq_u8(a,b);
430 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
431 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
432 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]|y[i]);
433 return __builtin_bit_cast(native_type,x);
435 return vorrq_u8(a,b);
439 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
440 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
441 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]^y[i]);
442 return __builtin_bit_cast(native_type,x);
444 return veorq_u8(a,b);
449 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
450 for(
auto & lane:lanes) lane=lane?mask_word<B>(~mask_word<B>(0)):0;
451 return __builtin_bit_cast(native_type,lanes);
453 if constexpr(B==1)
return bit_not(vceqq_u8(a,vdupq_n_u8(0)));
454 else if constexpr(B==2)
return bit_not(vreinterpretq_u8_u16(vceqq_u16(vreinterpretq_u16_u8(a),vdupq_n_u16(0))));
455 else if constexpr(B==4)
return bit_not(vreinterpretq_u8_u32(vceqq_u32(vreinterpretq_u32_u8(a),vdupq_n_u32(0))));
456 else return bit_not(vreinterpretq_u8_u64(vceqq_u64(vreinterpretq_u64_u8(a),vdupq_n_u64(0))));
460 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
461 for(
auto lane:lanes)
if(lane!=0)
return true;
464 return vmaxvq_u8(a)!=0;
468 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
469 for(
auto lane:lanes)
if(lane!=mask_word<B>(~mask_word<B>(0)))
return false;
472 return vminvq_u8(a)==0xff;
476 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
477 std::uint64_t result=0;
478 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(lanes[i]!=0)<<i;
481 std::array<std::uint8_t,16> bytes{};vst1q_u8(bytes.data(),a);
482 std::uint64_t result=0;
483 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(bytes[i*B]!=0)<<i;
488 std::array<mask_word<B>,N> lanes{};
489 for(std::size_t i=0;i<N;++i) lanes[i]=((bits>>i)&1)?mask_word<B>(~mask_word<B>(0)):0;
490 return __builtin_bit_cast(native_type,lanes);
492 std::array<std::uint8_t,16> a{};
493 for(std::size_t i=0;i<N;++i)
for(std::size_t j=0;j<B;++j) a[i*B+j]=((bits>>i)&1)?0xff:0;
494 return vld1q_u8(a.data());
498 template<std::
size_t B,std::
size_t N>
struct mask_vector512_ops {
499#if NATIVE_HAS_AVX512F
500 using native_type=__m512i;
503 std::array<mask_word<B>,N> lanes{};
504 lanes.fill(a?mask_word<B>(~mask_word<B>(0)):0);
505 return __builtin_bit_cast(native_type,lanes);
507 return _mm512_set1_epi32(a?-1:0);
511 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
512 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
513 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]&y[i]);
514 return __builtin_bit_cast(native_type,x);
516 return _mm512_and_si512(a,b);
520 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
521 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
522 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]|y[i]);
523 return __builtin_bit_cast(native_type,x);
525 return _mm512_or_si512(a,b);
529 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
530 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
531 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]^y[i]);
532 return __builtin_bit_cast(native_type,x);
534 return _mm512_xor_si512(a,b);
539 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
540 std::uint64_t result=0;
541 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(lanes[i]!=0)<<i;
544 if constexpr(B==1)
return _mm512_cmpneq_epi8_mask(a,_mm512_setzero_si512());
545 else if constexpr(B==2)
return _mm512_cmpneq_epi16_mask(a,_mm512_setzero_si512());
546 else if constexpr(B==4)
return _mm512_cmpneq_epi32_mask(a,_mm512_setzero_si512());
547 else return _mm512_cmpneq_epi64_mask(a,_mm512_setzero_si512());
551 std::array<mask_word<B>,N> lanes{};
552 for(std::size_t i=0;i<N;++i) lanes[i]=((a>>i)&1)?mask_word<B>(~mask_word<B>(0)):0;
553 return __builtin_bit_cast(native_type,lanes);
555 if constexpr(B==1)
return _mm512_maskz_set1_epi8(__mmask64(a),-1);
556 else if constexpr(B==2)
return _mm512_maskz_set1_epi16(__mmask32(a),-1);
557 else if constexpr(B==4)
return _mm512_maskz_set1_epi32(__mmask16(a),-1);
558 else return _mm512_maskz_set1_epi64(__mmask8(a),-1);
562 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
563 for(
auto & lane:lanes) lane=lane?mask_word<B>(~mask_word<B>(0)):0;
564 return __builtin_bit_cast(native_type,lanes);
566 return from_bits(bits(a));
570 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
571 for(
auto lane:lanes)
if(lane!=0)
return true;
578 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
579 for(
auto lane:lanes)
if(lane!=mask_word<B>(~mask_word<B>(0)))
return false;
582 return bits(a)==mask_low_bits<N>;
586 template<std::
size_t B,std::
size_t N>
using mask_full_ops=std::conditional_t<N==1,mask_scalar_ops<mask_word<B>>,
587 std::conditional_t<B*N==64,mask_vector512_ops<B,N>,mask_vector_ops<B,N>>>;
588 template<std::
size_t N>
inline constexpr bool predicate_shape=
589#if NATIVE_HAS_AVX512F
590 (N==1 || N==2 || N==3 || N==4 || N==8 || N==16
591#if NATIVE_HAS_AVX512BW
600 template<
class U,std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::mask_shape<
sizeof(U),N>
605 struct alignas(typename ::NATIVE_BACKEND_NAMESPACE::mask_full_ops<
sizeof(U),N>::native_type)
simd<mask_lane<U>,N,Arch> : detail::swizzle_access<mask_lane<U>,N,Arch> {
609 template <std::
size_t A = 1>
612 template <std::
size_t A = 1>
616 using storage_type=U;
617 using ops=::NATIVE_BACKEND_NAMESPACE::mask_full_ops<
sizeof(U),N>;
618 using native_type=
typename ops::native_type;
619 using mask_type=
simd;
620 using mask = mask_type;
622 static constexpr std::size_t lanes=N;
623 static constexpr bool compact=
false;
627 explicit native_inline constexpr simd(
bool value) noexcept : value_(ops::broadcast(value)) {}
631 template<
class... X>
requires(N>1 &&
sizeof...(X)==N) && (std::same_as<X,value_type>&&...)
634 explicit native_inline constexpr simd(std::array<value_type,N>
const & values) noexcept :
simd(load(values.data())) {}
651 std::uint64_t bits=0;
652 for(std::size_t i=0;i<N;++i) bits|=std::uint64_t(p[i].
to_bool())<<i;
653 return from_bitset(bits);
655 native_type value;std::memcpy(&value,
static_cast<void const *
>(p),
sizeof(value));
return unsafe_from_native(value);
658 native_inline constexpr void store(value_type * p)
const noexcept {
660 auto bits=to_bitset();
661 for(std::size_t i=0;i<N;++i) p[i]=value_type(((bits>>i)&1)!=0);
664 std::memcpy(
static_cast<void *
>(p),&value_,
sizeof(value_));
704 template<std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N>
706 static constexpr isa<> architecture=Arch;
707 using ops=::NATIVE_BACKEND_NAMESPACE::mask_compact_ops<N>;
708 using native_type=
typename ops::native_type;
709 static constexpr std::size_t lanes=N;
710 static constexpr bool compact=
true;
774 native_inline constexpr predicate(raw,native_type value) noexcept : value_(value) {}
775 native_type value_=0;
778 namespace detail::NATIVE_BACKEND {
779 template<
class T>
using mask_lane_for=mask_lane<mask_word<mask_lane_bytes<T>>>;
780 template<
class T,std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
using comparison_mask=std::conditional_t<mask_compact<mask_lane_bytes<T>,N>,
781 predicate<N,Arch>,simd<mask_lane_for<T>,N,Arch>>;
783 namespace detail::NATIVE_BACKEND {
784 template<::native::isa<> Arch,
class T,std::
size_t N,
class U,std::
size_t A=1,simd_access Access=simd_access::ordinary>
785 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> &&
requires {
typename simd<T,N,Arch>::native_type; }
787 template<
class U,
class T,std::
size_t N,std::
size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
788 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> &&
requires {
typename simd<T,N,Arch>::native_type; }
789 native_inline constexpr void store_simd(U * p,simd<T,N,Arch> value,simd_memory<A,Access> = {})
noexcept { value.store(p); }
790 template<::native::isa<> Arch,
class T,std::
size_t N,
class U,std::
size_t A=1,simd_access Access=simd_access::ordinary>
791 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> &&
requires {
typename simd<T,N,Arch>::native_type; }
793 std::array<T,N> a;a.fill(fill);
for(std::size_t i=0;i<count;++i)a[i]=p[i];return simd<T,N,Arch>::load(a.data());
795 template<
class U,
class T,std::
size_t N,std::
size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
796 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> &&
requires {
typename simd<T,N,Arch>::native_type; }
798 std::array<T,N> a;value.store(a.data());
for(std::size_t i=0;i<count;++i)p[i]=a[i];
803 template<
class U,
class T,std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<U> && simd_mask_element<T> &&
811 template<
class T,std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> &&
requires {
typename simd<T,N,Arch>::native_type; }
814#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
815 if constexpr(
sizeof(T)*N==16 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<
sizeof(T),N>) {
816 auto x=value.to_native();
auto z=_mm_setzero_si128();
821 }
else if constexpr(
sizeof(T)*N==32 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<
sizeof(T),N>) {
822 auto x=value.to_native();
auto z=_mm256_setzero_si256();
833 template<
class T,std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> &&
requires {
typename simd<T,N,Arch>::native_type; }
836#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
837 if constexpr(
sizeof(T)*N==16 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<
sizeof(T),N>) {
838 auto k=value.to_native();
843 }
else if constexpr(
sizeof(T)*N==32 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<
sizeof(T),N>) {
844 auto k=value.to_native();
857 namespace detail::NATIVE_BACKEND {
860 using V=
typename mask_full_ops<1,N>::native_type;
861 std::array<std::uint8_t,N> lanes{}; lanes.fill(1);
862 return __builtin_bit_cast(V,lanes);
864 if constexpr(N==1)
return std::uint8_t(1);
866 else if constexpr(N==16)
return _mm_set1_epi8(1);
867 else if constexpr(N==32)
return _mm256_set1_epi8(1);
869#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512BW
870 else if constexpr(N==64)
return _mm512_set1_epi8(1);
872#if NATIVE_HAS_ARM_NEON
873 else return vdupq_n_u8(1);
880 template<std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::mask_shape<1,N>
881 struct simd<bool,N,Arch> : detail::swizzle_access<bool,N,Arch> {
882 static constexpr isa<> architecture=Arch;
885 template <std::
size_t A = 1>
888 template <std::
size_t A = 1>
891 using value_type=bool;
892 using storage_type=std::uint8_t;
893 using ops=::NATIVE_BACKEND_NAMESPACE::mask_full_ops<1,N>;
894 using native_type=
typename ops::native_type;
896 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<bool,N,Arch>;
897 using mask = mask_type;
899 static constexpr std::size_t lanes=N;
903 explicit native_inline constexpr simd(
bool value) noexcept : value_(value?::NATIVE_BACKEND_NAMESPACE::bool_ones<N>():ops::broadcast(
false)) {}
905 template<
class... X>
requires(N>1 &&
sizeof...(X)==N) && (std::same_as<X,bool>&&...)
911 return simd(raw{},ops::bit_and(ops::normalize(value),::NATIVE_BACKEND_NAMESPACE::bool_ones<N>()));
920 std::array<std::uint8_t,N> bytes{};
921 for(std::size_t i=0;i<N;++i) bytes[i]=p[i]?1:0;
923 native_type value;std::memcpy(&value,bytes.data(),
sizeof(value));
return unsafe_from_native(value);
927 std::array<std::uint8_t,N> bytes{};
928 if consteval { bytes=__builtin_bit_cast(
decltype(bytes),value_); }
929 else { std::memcpy(bytes.data(),&value_,
sizeof(value_)); }
930 for(std::size_t i=0;i<N;++i) p[i]=bytes[i]!=0;
935 std::array<bool,N> values;values.fill(fill);
936 for(std::size_t i=0;i<count;++i) values[i]=p[i];
937 return load(values.data());
941 std::array<bool,N> values;
store(values.data());
942 for(std::size_t i=0;i<count;++i) p[i]=values[i];
956 auto x=ops::bit_xor(a.value_,b.value_);
957 if consteval {
return mask_type::from_bitset(vector_mask_type::from_native(x).to_bitset()); }
958#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512BW
959 if constexpr(N==64)
return mask_type::from_native(_mm512_cmpneq_epi8_mask(x,_mm512_setzero_si512()));
960#if NATIVE_HAS_AVX512VL
961 else if constexpr(N==32)
return mask_type::from_native(_mm256_cmpneq_epi8_mask(x,_mm256_setzero_si256()));
962 else if constexpr(N==16)
return mask_type::from_native(_mm_cmpneq_epi8_mask(x,_mm_setzero_si128()));
966 return vector_mask_type::from_native(x);
983 template<
class M>
requires(std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
986 auto mask=vector_mask_type::from_bitset(m.to_bitset()).to_native();
987 return simd(raw{},ops::bit_or(ops::bit_and(mask,a.value_),ops::bit_and(ops::bit_not(mask),b.value_)));
989#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512BW
990 if constexpr(M::compact) {
991 if constexpr(N==64)
return simd(raw{},_mm512_mask_blend_epi8(m.to_native(),b.value_,a.value_));
992#if NATIVE_HAS_AVX512VL
993 else if constexpr(N==32)
return simd(raw{},_mm256_mask_blend_epi8(m.to_native(),b.value_,a.value_));
994 else return simd(raw{},_mm_mask_blend_epi8(m.to_native(),b.value_,a.value_));
998 return simd(raw{},ops::bit_or(ops::bit_and(m.to_native(),a.value_),ops::bit_and(ops::bit_not(m.to_native()),b.value_)));
1002 native_inline constexpr simd(raw,native_type value) noexcept : value_(value) {}
1006 namespace detail::NATIVE_BACKEND {
1007 template<::native::isa<> Arch,
class T,std::
size_t N,
class U,std::
size_t A=1,simd_access Access=simd_access::ordinary>
1008 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> &&
requires {
typename simd<T,N,Arch>::native_type; }
1010 template<
class U,
class T,std::
size_t N,std::
size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
1011 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> &&
requires {
typename simd<T,N,Arch>::native_type; }
1012 native_inline constexpr void store_simd(U * p,simd<T,N,Arch> value,simd_memory<A,Access> = {})
noexcept { value.store(p); }
1013 template<::native::isa<> Arch,
class T,std::
size_t N,
class U,std::
size_t A=1,simd_access Access=simd_access::ordinary>
1014 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> &&
requires {
typename simd<T,N,Arch>::native_type; }
1016 return simd<T,N,Arch>::load_partial(p,count,fill);
1018 template<
class U,
class T,std::
size_t N,std::
size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
1019 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> &&
requires {
typename simd<T,N,Arch>::native_type; }
1028 template<
class T,std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> &&
requires {
typename simd<bool,N,Arch>::native_type;
typename simd<T,N,Arch>::native_type; }
1034 template<std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> &&
requires {
typename simd<bool,N,Arch>::native_type; }
1037 template<std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> &&
requires {
typename simd<bool,N,Arch>::native_type; }
1040 if constexpr(
decltype(result)::compact)
return result;
1046namespace NATIVE_BACKEND_NAMESPACE {
1048 return std::bit_cast<std::make_unsigned_t<T>>(value);
1050 template <simd_
integer_element T,
class U>
1052 return std::bit_cast<T>(
static_cast<std::make_unsigned_t<T>
>(value));
1054 template <simd_
integer_element T>
1055 using integer_work_word = std::conditional_t<(
sizeof(T) < 4), uint32_t, std::make_unsigned_t<T>>;
1056 template <simd_
integer_element T>
1058 using U = integer_work_word<T>;
1059 return integer_wrap<T>(U(integer_word(a)) + U(integer_word(b)));
1061 template <simd_
integer_element T>
1063 using U = integer_work_word<T>;
1064 return integer_wrap<T>(U(integer_word(a)) - U(integer_word(b)));
1066 template <simd_
integer_element T>
1068 using U = integer_work_word<T>;
1069 return integer_wrap<T>(U(integer_word(a)) * U(integer_word(b)));
1073 template <simd_
integer_element T>
1075 if constexpr (
sizeof(T) == 1)
1076 return _mm_set1_epi8(std::bit_cast<int8_t>(x));
1077 else if constexpr (
sizeof(T) == 2)
1078 return _mm_set1_epi16(std::bit_cast<int16_t>(x));
1079 else if constexpr (
sizeof(T) == 4)
1080 return _mm_set1_epi32(std::bit_cast<int32_t>(x));
1081 else if constexpr (
sizeof(T) == 8)
1082 return _mm_set1_epi64x(std::bit_cast<int64_t>(x));
1084 template <simd_
integer_element T>
1086 if constexpr (
sizeof(T) == 1)
1087 return _mm_add_epi8(a, b);
1088 else if constexpr (
sizeof(T) == 2)
1089 return _mm_add_epi16(a, b);
1090 else if constexpr (
sizeof(T) == 4)
1091 return _mm_add_epi32(a, b);
1092 else if constexpr (
sizeof(T) == 8)
1093 return _mm_add_epi64(a, b);
1095 template <simd_
integer_element T>
1097 if constexpr (
sizeof(T) == 1)
1098 return _mm_sub_epi8(a, b);
1099 else if constexpr (
sizeof(T) == 2)
1100 return _mm_sub_epi16(a, b);
1101 else if constexpr (
sizeof(T) == 4)
1102 return _mm_sub_epi32(a, b);
1103 else if constexpr (
sizeof(T) == 8)
1104 return _mm_sub_epi64(a, b);
1106 template <simd_
integer_element T>
1108 if constexpr (
sizeof(T) == 1) {
1110 auto low = _mm_mullo_epi16(a, b);
1111 auto high = _mm_mullo_epi16(_mm_srli_epi16(a, 8), _mm_srli_epi16(b, 8));
1112 return _mm_or_si128(_mm_and_si128(low, _mm_set1_epi16(255)), _mm_slli_epi16(high, 8));
1113 }
else if constexpr (
sizeof(T) == 2)
1114 return _mm_mullo_epi16(a, b);
1115 if constexpr (
sizeof(T) == 4)
1116 return _mm_mullo_epi32(a, b);
1117 else if constexpr (
sizeof(T) == 8) {
1118#if NATIVE_HAS_AVX512DQ && NATIVE_HAS_AVX512VL
1119 return _mm_mullo_epi64(a, b);
1123 _mm_add_epi64(_mm_mul_epu32(a, _mm_srli_epi64(b, 32)), _mm_mul_epu32(_mm_srli_epi64(a, 32), b));
1124 return _mm_add_epi64(_mm_mul_epu32(a, b), _mm_slli_epi64(cross, 32));
1128 template <simd_
integer_element T,
unsigned S>
1129 requires(S <
sizeof(T) * 8)
1131 if constexpr (S == 0)
1134 if constexpr (
sizeof(T) == 1)
1135 return _mm_and_si128(_mm_slli_epi16(a, S),
1136 _mm_set1_epi8(std::bit_cast<int8_t>(uint8_t((255u << S) & 255u))));
1137 else if constexpr (
sizeof(T) == 2)
1138 return _mm_slli_epi16(a, S);
1139 if constexpr (
sizeof(T) == 4)
1140 return _mm_slli_epi32(a, S);
1141 else if constexpr (
sizeof(T) == 8)
1142 return _mm_slli_epi64(a, S);
1145 template <simd_
integer_element T,
unsigned S>
1146 requires(S <
sizeof(T) * 8)
1148 if constexpr (S == 0)
1151 if constexpr (
sizeof(T) == 1) {
1152 auto low = _mm_and_si128(_mm_srli_epi16(a, S), _mm_set1_epi8(int8_t(255u >> S)));
1153 if constexpr (std::is_unsigned_v<T>)
1156 auto negative = _mm_cmpgt_epi8(_mm_setzero_si128(), a);
1157 return _mm_or_si128(
1158 low, _mm_and_si128(negative, _mm_set1_epi8(std::bit_cast<int8_t>(uint8_t(255u ^ (255u >> S))))));
1160 }
else if constexpr (
sizeof(T) == 2) {
1161 if constexpr (std::is_signed_v<T>)
1162 return _mm_srai_epi16(a, S);
1164 return _mm_srli_epi16(a, S);
1166 if constexpr (
sizeof(T) == 4) {
1167 if constexpr (std::is_signed_v<T>)
1168 return _mm_srai_epi32(a, S);
1170 return _mm_srli_epi32(a, S);
1171 }
else if constexpr (
sizeof(T) == 8) {
1172 if constexpr (std::is_unsigned_v<T>)
1173 return _mm_srli_epi64(a, S);
1175#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
1176 return _mm_srai_epi64(a, S);
1178 auto negative = _mm_cmpgt_epi64(_mm_setzero_si128(), a);
1179 return _mm_or_si128(
1180 _mm_srli_epi64(a, S),
1181 _mm_and_si128(negative, _mm_set1_epi64x(std::bit_cast<int64_t>(~uint64_t(0) << (64 - S)))));
1190 template <simd_
integer_element T>
1192 if constexpr (
sizeof(T) == 1)
1193 return _mm256_set1_epi8(std::bit_cast<int8_t>(x));
1194 else if constexpr (
sizeof(T) == 2)
1195 return _mm256_set1_epi16(std::bit_cast<int16_t>(x));
1196 else if constexpr (
sizeof(T) == 4)
1197 return _mm256_set1_epi32(std::bit_cast<int32_t>(x));
1198 else if constexpr (
sizeof(T) == 8)
1199 return _mm256_set1_epi64x(std::bit_cast<int64_t>(x));
1201 template <simd_
integer_element T>
1203 if constexpr (
sizeof(T) == 1)
1204 return _mm256_add_epi8(a, b);
1205 else if constexpr (
sizeof(T) == 2)
1206 return _mm256_add_epi16(a, b);
1207 else if constexpr (
sizeof(T) == 4)
1208 return _mm256_add_epi32(a, b);
1209 else if constexpr (
sizeof(T) == 8)
1210 return _mm256_add_epi64(a, b);
1212 template <simd_
integer_element T>
1214 if constexpr (
sizeof(T) == 1)
1215 return _mm256_sub_epi8(a, b);
1216 else if constexpr (
sizeof(T) == 2)
1217 return _mm256_sub_epi16(a, b);
1218 else if constexpr (
sizeof(T) == 4)
1219 return _mm256_sub_epi32(a, b);
1220 else if constexpr (
sizeof(T) == 8)
1221 return _mm256_sub_epi64(a, b);
1223 template <simd_
integer_element T>
1225 if constexpr (
sizeof(T) == 1) {
1227 auto low = _mm256_mullo_epi16(a, b);
1228 auto high = _mm256_mullo_epi16(_mm256_srli_epi16(a, 8), _mm256_srli_epi16(b, 8));
1229 return _mm256_or_si256(_mm256_and_si256(low, _mm256_set1_epi16(255)), _mm256_slli_epi16(high, 8));
1230 }
else if constexpr (
sizeof(T) == 2)
1231 return _mm256_mullo_epi16(a, b);
1232 if constexpr (
sizeof(T) == 4)
1233 return _mm256_mullo_epi32(a, b);
1234 else if constexpr (
sizeof(T) == 8) {
1235#if NATIVE_HAS_AVX512DQ && NATIVE_HAS_AVX512VL
1236 return _mm256_mullo_epi64(a, b);
1239 auto cross = _mm256_add_epi64(_mm256_mul_epu32(a, _mm256_srli_epi64(b, 32)),
1240 _mm256_mul_epu32(_mm256_srli_epi64(a, 32), b));
1241 return _mm256_add_epi64(_mm256_mul_epu32(a, b), _mm256_slli_epi64(cross, 32));
1245 template <simd_
integer_element T,
unsigned S>
1246 requires(S <
sizeof(T) * 8)
1248 if constexpr (S == 0)
1251 if constexpr (
sizeof(T) == 1)
1252 return _mm256_and_si256(_mm256_slli_epi16(a, S),
1253 _mm256_set1_epi8(std::bit_cast<int8_t>(uint8_t((255u << S) & 255u))));
1254 else if constexpr (
sizeof(T) == 2)
1255 return _mm256_slli_epi16(a, S);
1256 if constexpr (
sizeof(T) == 4)
1257 return _mm256_slli_epi32(a, S);
1258 else if constexpr (
sizeof(T) == 8)
1259 return _mm256_slli_epi64(a, S);
1263 return _mm256_sllv_epi32(a,counts);
1265 template <simd_
integer_element T,
unsigned S>
1266 requires(S <
sizeof(T) * 8)
1268 if constexpr (S == 0)
1271 if constexpr (
sizeof(T) == 1) {
1272 auto low = _mm256_and_si256(_mm256_srli_epi16(a, S), _mm256_set1_epi8(int8_t(255u >> S)));
1273 if constexpr (std::is_unsigned_v<T>)
1276 auto negative = _mm256_cmpgt_epi8(_mm256_setzero_si256(), a);
1277 return _mm256_or_si256(
1279 _mm256_and_si256(negative, _mm256_set1_epi8(std::bit_cast<int8_t>(uint8_t(255u ^ (255u >> S))))));
1281 }
else if constexpr (
sizeof(T) == 2) {
1282 if constexpr (std::is_signed_v<T>)
1283 return _mm256_srai_epi16(a, S);
1285 return _mm256_srli_epi16(a, S);
1287 if constexpr (
sizeof(T) == 4) {
1288 if constexpr (std::is_signed_v<T>)
1289 return _mm256_srai_epi32(a, S);
1291 return _mm256_srli_epi32(a, S);
1292 }
else if constexpr (
sizeof(T) == 8) {
1293 if constexpr (std::is_unsigned_v<T>)
1294 return _mm256_srli_epi64(a, S);
1296#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
1297 return _mm256_srai_epi64(a, S);
1299 auto negative = _mm256_cmpgt_epi64(_mm256_setzero_si256(), a);
1300 return _mm256_or_si256(
1301 _mm256_srli_epi64(a, S),
1302 _mm256_and_si256(negative, _mm256_set1_epi64x(std::bit_cast<int64_t>(~uint64_t(0) << (64 - S)))));
1310#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
1311 template <simd_
integer_element T>
1313#if NATIVE_HAS_AVX512BW
1314 if constexpr (
sizeof(T) == 1)
1315 return _mm512_set1_epi8(std::bit_cast<int8_t>(x));
1316 else if constexpr (
sizeof(T) == 2)
1317 return _mm512_set1_epi16(std::bit_cast<int16_t>(x));
1319 if constexpr (
sizeof(T) == 4)
1320 return _mm512_set1_epi32(std::bit_cast<int32_t>(x));
1321 else if constexpr (
sizeof(T) == 8)
1322 return _mm512_set1_epi64(std::bit_cast<int64_t>(x));
1324 template <simd_
integer_element T>
1326#if NATIVE_HAS_AVX512BW
1327 if constexpr (
sizeof(T) == 1)
1328 return _mm512_add_epi8(a, b);
1329 else if constexpr (
sizeof(T) == 2)
1330 return _mm512_add_epi16(a, b);
1332 if constexpr (
sizeof(T) == 4)
1333 return _mm512_add_epi32(a, b);
1334 else if constexpr (
sizeof(T) == 8)
1335 return _mm512_add_epi64(a, b);
1337 template <simd_
integer_element T>
1339#if NATIVE_HAS_AVX512BW
1340 if constexpr (
sizeof(T) == 1)
1341 return _mm512_sub_epi8(a, b);
1342 else if constexpr (
sizeof(T) == 2)
1343 return _mm512_sub_epi16(a, b);
1345 if constexpr (
sizeof(T) == 4)
1346 return _mm512_sub_epi32(a, b);
1347 else if constexpr (
sizeof(T) == 8)
1348 return _mm512_sub_epi64(a, b);
1350 template <simd_
integer_element T>
1352#if NATIVE_HAS_AVX512BW
1353 if constexpr (
sizeof(T) == 1) {
1355 auto low = _mm512_mullo_epi16(a, b);
1356 auto high = _mm512_mullo_epi16(_mm512_srli_epi16(a, 8), _mm512_srli_epi16(b, 8));
1357 return _mm512_or_si512(_mm512_and_si512(low, _mm512_set1_epi16(255)), _mm512_slli_epi16(high, 8));
1358 }
else if constexpr (
sizeof(T) == 2)
1359 return _mm512_mullo_epi16(a, b);
1361 if constexpr (
sizeof(T) == 4)
1362 return _mm512_mullo_epi32(a, b);
1363 else if constexpr (
sizeof(T) == 8) {
1364 return _mm512_mullo_epi64(a, b);
1367 template <simd_
integer_element T,
unsigned S>
1368 requires(S <
sizeof(T) * 8)
1370 if constexpr (S == 0)
1373#if NATIVE_HAS_AVX512BW
1374 if constexpr (
sizeof(T) == 1)
1375 return _mm512_and_si512(_mm512_slli_epi16(a, S),
1376 _mm512_set1_epi8(std::bit_cast<int8_t>(uint8_t((255u << S) & 255u))));
1377 else if constexpr (
sizeof(T) == 2)
1378 return _mm512_slli_epi16(a, S);
1380 if constexpr (
sizeof(T) == 4)
1381 return _mm512_slli_epi32(a, S);
1382 else if constexpr (
sizeof(T) == 8)
1383 return _mm512_slli_epi64(a, S);
1387 return _mm512_sllv_epi32(a,counts);
1389 template <simd_
integer_element T,
unsigned S>
1390 requires(S <
sizeof(T) * 8)
1392 if constexpr (S == 0)
1395#if NATIVE_HAS_AVX512BW
1396 if constexpr (
sizeof(T) == 1) {
1397 auto low = _mm512_and_si512(_mm512_srli_epi16(a, S), _mm512_set1_epi8(int8_t(255u >> S)));
1398 if constexpr (std::is_unsigned_v<T>)
1401 auto negative = _mm512_movm_epi8(_mm512_cmp_epi8_mask(a, _mm512_setzero_si512(), _MM_CMPINT_LT));
1402 return _mm512_or_si512(
1404 _mm512_and_si512(negative, _mm512_set1_epi8(std::bit_cast<int8_t>(uint8_t(255u ^ (255u >> S))))));
1406 }
else if constexpr (
sizeof(T) == 2) {
1407 if constexpr (std::is_signed_v<T>)
1408 return _mm512_srai_epi16(a, S);
1410 return _mm512_srli_epi16(a, S);
1413 if constexpr (
sizeof(T) == 4) {
1414 if constexpr (std::is_signed_v<T>)
1415 return _mm512_srai_epi32(a, S);
1417 return _mm512_srli_epi32(a, S);
1418 }
else if constexpr (
sizeof(T) == 8) {
1419 if constexpr (std::is_unsigned_v<T>)
1420 return _mm512_srli_epi64(a, S);
1422 return _mm512_srai_epi64(a, S);
1430 return _mm_and_si128(a, b);
1433 return _mm_or_si128(a, b);
1436 return _mm_xor_si128(a, b);
1438 template <simd_
integer_element T,
bool Greater>
1440 if constexpr (mask_compact<
sizeof(T), 16 /
sizeof(T)>) {
1441 if constexpr (
sizeof(T) == 1) {
1442 if constexpr (Greater) {
1443 if constexpr (std::is_signed_v<T>)
1444 return _mm_cmp_epi8_mask(a, b, _MM_CMPINT_GT);
1446 return _mm_cmp_epu8_mask(a, b, _MM_CMPINT_GT);
1448 return _mm_cmp_epi8_mask(a, b, _MM_CMPINT_EQ);
1449 }
else if constexpr (
sizeof(T) == 2) {
1450 if constexpr (Greater) {
1451 if constexpr (std::is_signed_v<T>)
1452 return _mm_cmp_epi16_mask(a, b, _MM_CMPINT_GT);
1454 return _mm_cmp_epu16_mask(a, b, _MM_CMPINT_GT);
1456 return _mm_cmp_epi16_mask(a, b, _MM_CMPINT_EQ);
1457 }
else if constexpr (
sizeof(T) == 4) {
1458 if constexpr (Greater) {
1459 if constexpr (std::is_signed_v<T>)
1460 return _mm_cmp_epi32_mask(a, b, _MM_CMPINT_GT);
1462 return _mm_cmp_epu32_mask(a, b, _MM_CMPINT_GT);
1464 return _mm_cmp_epi32_mask(a, b, _MM_CMPINT_EQ);
1465 }
else if constexpr (
sizeof(T) == 8) {
1466 if constexpr (Greater) {
1467 if constexpr (std::is_signed_v<T>)
1468 return _mm_cmp_epi64_mask(a, b, _MM_CMPINT_GT);
1470 return _mm_cmp_epu64_mask(a, b, _MM_CMPINT_GT);
1472 return _mm_cmp_epi64_mask(a, b, _MM_CMPINT_EQ);
1475 if constexpr (
sizeof(T) == 1) {
1476 if constexpr (!Greater)
1477 return _mm_cmpeq_epi8(a, b);
1478 else if constexpr (std::is_signed_v<T>)
1479 return _mm_cmpgt_epi8(a, b);
1481 auto bias = _mm_set1_epi8(int8_t(-128));
1482 return _mm_cmpgt_epi8(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1484 }
else if constexpr (
sizeof(T) == 2) {
1485 if constexpr (!Greater)
1486 return _mm_cmpeq_epi16(a, b);
1487 else if constexpr (std::is_signed_v<T>)
1488 return _mm_cmpgt_epi16(a, b);
1490 auto bias = _mm_set1_epi16(int16_t(-32768));
1491 return _mm_cmpgt_epi16(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1493 }
else if constexpr (
sizeof(T) == 4) {
1494 if constexpr (!Greater)
1495 return _mm_cmpeq_epi32(a, b);
1496 else if constexpr (std::is_signed_v<T>)
1497 return _mm_cmpgt_epi32(a, b);
1499 auto bias = _mm_set1_epi32(std::bit_cast<int32_t>(uint32_t(0x80000000u)));
1500 return _mm_cmpgt_epi32(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1502 }
else if constexpr (
sizeof(T) == 8) {
1503 if constexpr (!Greater)
1504 return _mm_cmpeq_epi64(a, b);
1505 else if constexpr (std::is_signed_v<T>)
1506 return _mm_cmpgt_epi64(a, b);
1508 auto bias = _mm_set1_epi64x(std::bit_cast<int64_t>(uint64_t(1) << 63));
1509 return _mm_cmpgt_epi64(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1514 template <simd_
integer_element T,
class M>
1516 if constexpr (M::compact) {
1517 if constexpr (
sizeof(T) == 1)
1518 return _mm_mask_blend_epi8(m.to_native(), b, a);
1519 else if constexpr (
sizeof(T) == 2)
1520 return _mm_mask_blend_epi16(m.to_native(), b, a);
1521 else if constexpr (
sizeof(T) == 4)
1522 return _mm_mask_blend_epi32(m.to_native(), b, a);
1523 else if constexpr (
sizeof(T) == 8)
1524 return _mm_mask_blend_epi64(m.to_native(), b, a);
1526 return _mm_or_si128(_mm_and_si128(m.to_native(), a), _mm_andnot_si128(m.to_native(), b));
1528 template <simd_
integer_element T,
class M>
1530 __m128i b)
noexcept {
1531 if constexpr (M::compact) {
1532 if constexpr (
sizeof(T) == 1)
1533 return _mm_mask_add_epi8(prior, m.to_native(), a, b);
1534 else if constexpr (
sizeof(T) == 2)
1535 return _mm_mask_add_epi16(prior, m.to_native(), a, b);
1536 else if constexpr (
sizeof(T) == 4)
1537 return _mm_mask_add_epi32(prior, m.to_native(), a, b);
1538 else if constexpr (
sizeof(T) == 8)
1539 return _mm_mask_add_epi64(prior, m.to_native(), a, b);
1541 return integer_select<T>(m, integer_add<T>(a, b), prior);
1543 template <simd_
integer_element T,
class M>
1545 __m128i b)
noexcept {
1546 if constexpr (M::compact) {
1547 if constexpr (
sizeof(T) == 1)
1548 return _mm_mask_sub_epi8(prior, m.to_native(), a, b);
1549 else if constexpr (
sizeof(T) == 2)
1550 return _mm_mask_sub_epi16(prior, m.to_native(), a, b);
1551 else if constexpr (
sizeof(T) == 4)
1552 return _mm_mask_sub_epi32(prior, m.to_native(), a, b);
1553 else if constexpr (
sizeof(T) == 8)
1554 return _mm_mask_sub_epi64(prior, m.to_native(), a, b);
1556 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1562 return _mm256_and_si256(a, b);
1565 return _mm256_or_si256(a, b);
1568 return _mm256_xor_si256(a, b);
1570 template <simd_
integer_element T,
bool Greater>
1572 if constexpr (mask_compact<
sizeof(T), 32 /
sizeof(T)>) {
1573 if constexpr (
sizeof(T) == 1) {
1574 if constexpr (Greater) {
1575 if constexpr (std::is_signed_v<T>)
1576 return _mm256_cmp_epi8_mask(a, b, _MM_CMPINT_GT);
1578 return _mm256_cmp_epu8_mask(a, b, _MM_CMPINT_GT);
1580 return _mm256_cmp_epi8_mask(a, b, _MM_CMPINT_EQ);
1581 }
else if constexpr (
sizeof(T) == 2) {
1582 if constexpr (Greater) {
1583 if constexpr (std::is_signed_v<T>)
1584 return _mm256_cmp_epi16_mask(a, b, _MM_CMPINT_GT);
1586 return _mm256_cmp_epu16_mask(a, b, _MM_CMPINT_GT);
1588 return _mm256_cmp_epi16_mask(a, b, _MM_CMPINT_EQ);
1589 }
else if constexpr (
sizeof(T) == 4) {
1590 if constexpr (Greater) {
1591 if constexpr (std::is_signed_v<T>)
1592 return _mm256_cmp_epi32_mask(a, b, _MM_CMPINT_GT);
1594 return _mm256_cmp_epu32_mask(a, b, _MM_CMPINT_GT);
1596 return _mm256_cmp_epi32_mask(a, b, _MM_CMPINT_EQ);
1597 }
else if constexpr (
sizeof(T) == 8) {
1598 if constexpr (Greater) {
1599 if constexpr (std::is_signed_v<T>)
1600 return _mm256_cmp_epi64_mask(a, b, _MM_CMPINT_GT);
1602 return _mm256_cmp_epu64_mask(a, b, _MM_CMPINT_GT);
1604 return _mm256_cmp_epi64_mask(a, b, _MM_CMPINT_EQ);
1607 if constexpr (
sizeof(T) == 1) {
1608 if constexpr (!Greater)
1609 return _mm256_cmpeq_epi8(a, b);
1610 else if constexpr (std::is_signed_v<T>)
1611 return _mm256_cmpgt_epi8(a, b);
1613 auto bias = _mm256_set1_epi8(int8_t(-128));
1614 return _mm256_cmpgt_epi8(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1616 }
else if constexpr (
sizeof(T) == 2) {
1617 if constexpr (!Greater)
1618 return _mm256_cmpeq_epi16(a, b);
1619 else if constexpr (std::is_signed_v<T>)
1620 return _mm256_cmpgt_epi16(a, b);
1622 auto bias = _mm256_set1_epi16(int16_t(-32768));
1623 return _mm256_cmpgt_epi16(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1625 }
else if constexpr (
sizeof(T) == 4) {
1626 if constexpr (!Greater)
1627 return _mm256_cmpeq_epi32(a, b);
1628 else if constexpr (std::is_signed_v<T>)
1629 return _mm256_cmpgt_epi32(a, b);
1631 auto bias = _mm256_set1_epi32(std::bit_cast<int32_t>(uint32_t(0x80000000u)));
1632 return _mm256_cmpgt_epi32(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1634 }
else if constexpr (
sizeof(T) == 8) {
1635 if constexpr (!Greater)
1636 return _mm256_cmpeq_epi64(a, b);
1637 else if constexpr (std::is_signed_v<T>)
1638 return _mm256_cmpgt_epi64(a, b);
1640 auto bias = _mm256_set1_epi64x(std::bit_cast<int64_t>(uint64_t(1) << 63));
1641 return _mm256_cmpgt_epi64(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1646 template <simd_
integer_element T,
class M>
1648 if constexpr (M::compact) {
1649 if constexpr (
sizeof(T) == 1)
1650 return _mm256_mask_blend_epi8(m.to_native(), b, a);
1651 else if constexpr (
sizeof(T) == 2)
1652 return _mm256_mask_blend_epi16(m.to_native(), b, a);
1653 else if constexpr (
sizeof(T) == 4)
1654 return _mm256_mask_blend_epi32(m.to_native(), b, a);
1655 else if constexpr (
sizeof(T) == 8)
1656 return _mm256_mask_blend_epi64(m.to_native(), b, a);
1658 return _mm256_or_si256(_mm256_and_si256(m.to_native(), a), _mm256_andnot_si256(m.to_native(), b));
1660 template <simd_
integer_element T,
class M>
1662 __m256i b)
noexcept {
1663 if constexpr (M::compact) {
1664 if constexpr (
sizeof(T) == 1)
1665 return _mm256_mask_add_epi8(prior, m.to_native(), a, b);
1666 else if constexpr (
sizeof(T) == 2)
1667 return _mm256_mask_add_epi16(prior, m.to_native(), a, b);
1668 else if constexpr (
sizeof(T) == 4)
1669 return _mm256_mask_add_epi32(prior, m.to_native(), a, b);
1670 else if constexpr (
sizeof(T) == 8)
1671 return _mm256_mask_add_epi64(prior, m.to_native(), a, b);
1673 return integer_select<T>(m, integer_add<T>(a, b), prior);
1675 template <simd_
integer_element T,
class M>
1677 __m256i b)
noexcept {
1678 if constexpr (M::compact) {
1679 if constexpr (
sizeof(T) == 1)
1680 return _mm256_mask_sub_epi8(prior, m.to_native(), a, b);
1681 else if constexpr (
sizeof(T) == 2)
1682 return _mm256_mask_sub_epi16(prior, m.to_native(), a, b);
1683 else if constexpr (
sizeof(T) == 4)
1684 return _mm256_mask_sub_epi32(prior, m.to_native(), a, b);
1685 else if constexpr (
sizeof(T) == 8)
1686 return _mm256_mask_sub_epi64(prior, m.to_native(), a, b);
1688 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1692#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
1694 return _mm512_and_si512(a, b);
1697 return _mm512_or_si512(a, b);
1700 return _mm512_xor_si512(a, b);
1702 template <simd_
integer_element T,
bool Greater>
1704 if constexpr (mask_compact<
sizeof(T), 64 /
sizeof(T)>) {
1705#if NATIVE_HAS_AVX512BW
1706 if constexpr (
sizeof(T) == 1) {
1707 if constexpr (Greater) {
1708 if constexpr (std::is_signed_v<T>)
1709 return _mm512_cmp_epi8_mask(a, b, _MM_CMPINT_GT);
1711 return _mm512_cmp_epu8_mask(a, b, _MM_CMPINT_GT);
1713 return _mm512_cmp_epi8_mask(a, b, _MM_CMPINT_EQ);
1714 }
else if constexpr (
sizeof(T) == 2) {
1715 if constexpr (Greater) {
1716 if constexpr (std::is_signed_v<T>)
1717 return _mm512_cmp_epi16_mask(a, b, _MM_CMPINT_GT);
1719 return _mm512_cmp_epu16_mask(a, b, _MM_CMPINT_GT);
1721 return _mm512_cmp_epi16_mask(a, b, _MM_CMPINT_EQ);
1724 if constexpr (
sizeof(T) == 4) {
1725 if constexpr (Greater) {
1726 if constexpr (std::is_signed_v<T>)
1727 return _mm512_cmp_epi32_mask(a, b, _MM_CMPINT_GT);
1729 return _mm512_cmp_epu32_mask(a, b, _MM_CMPINT_GT);
1731 return _mm512_cmp_epi32_mask(a, b, _MM_CMPINT_EQ);
1732 }
else if constexpr (
sizeof(T) == 8) {
1733 if constexpr (Greater) {
1734 if constexpr (std::is_signed_v<T>)
1735 return _mm512_cmp_epi64_mask(a, b, _MM_CMPINT_GT);
1737 return _mm512_cmp_epu64_mask(a, b, _MM_CMPINT_GT);
1739 return _mm512_cmp_epi64_mask(a, b, _MM_CMPINT_EQ);
1742 static_assert(
sizeof(T) == 0,
"512-bit integer comparisons require native predicates");
1745 template <simd_
integer_element T,
class M>
1747 if constexpr (M::compact) {
1748#if NATIVE_HAS_AVX512BW
1749 if constexpr (
sizeof(T) == 1)
1750 return _mm512_mask_blend_epi8(m.to_native(), b, a);
1751 else if constexpr (
sizeof(T) == 2)
1752 return _mm512_mask_blend_epi16(m.to_native(), b, a);
1754 if constexpr (
sizeof(T) == 4)
1755 return _mm512_mask_blend_epi32(m.to_native(), b, a);
1756 else if constexpr (
sizeof(T) == 8)
1757 return _mm512_mask_blend_epi64(m.to_native(), b, a);
1759 return _mm512_or_si512(_mm512_and_si512(m.to_native(), a), _mm512_andnot_si512(m.to_native(), b));
1761 template <simd_
integer_element T,
class M>
1763 __m512i b)
noexcept {
1764 if constexpr (M::compact) {
1765#if NATIVE_HAS_AVX512BW
1766 if constexpr (
sizeof(T) == 1)
1767 return _mm512_mask_add_epi8(prior, m.to_native(), a, b);
1768 else if constexpr (
sizeof(T) == 2)
1769 return _mm512_mask_add_epi16(prior, m.to_native(), a, b);
1771 if constexpr (
sizeof(T) == 4)
1772 return _mm512_mask_add_epi32(prior, m.to_native(), a, b);
1773 else if constexpr (
sizeof(T) == 8)
1774 return _mm512_mask_add_epi64(prior, m.to_native(), a, b);
1776 return integer_select<T>(m, integer_add<T>(a, b), prior);
1778 template <simd_
integer_element T,
class M>
1780 __m512i b)
noexcept {
1781 if constexpr (M::compact) {
1782#if NATIVE_HAS_AVX512BW
1783 if constexpr (
sizeof(T) == 1)
1784 return _mm512_mask_sub_epi8(prior, m.to_native(), a, b);
1785 else if constexpr (
sizeof(T) == 2)
1786 return _mm512_mask_sub_epi16(prior, m.to_native(), a, b);
1788 if constexpr (
sizeof(T) == 4)
1789 return _mm512_mask_sub_epi32(prior, m.to_native(), a, b);
1790 else if constexpr (
sizeof(T) == 8)
1791 return _mm512_mask_sub_epi64(prior, m.to_native(), a, b);
1793 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1797#if NATIVE_HAS_ARM_NEON
1799 return vandq_u8(a, b);
1802 return vorrq_u8(a, b);
1805 return veorq_u8(a, b);
1807 template <simd_
integer_element T>
1809 if constexpr (
sizeof(T) == 1) {
1810 return vaddq_u8(a, b);
1811 }
else if constexpr (
sizeof(T) == 2) {
1812 return vreinterpretq_u8_u16(vaddq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1813 }
else if constexpr (
sizeof(T) == 4) {
1814 return vreinterpretq_u8_u32(vaddq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1815 }
else if constexpr (
sizeof(T) == 8) {
1816 return vreinterpretq_u8_u64(vaddq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1819 template <simd_
integer_element T>
1821 if constexpr (
sizeof(T) == 1) {
1822 return vsubq_u8(a, b);
1823 }
else if constexpr (
sizeof(T) == 2) {
1824 return vreinterpretq_u8_u16(vsubq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1825 }
else if constexpr (
sizeof(T) == 4) {
1826 return vreinterpretq_u8_u32(vsubq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1827 }
else if constexpr (
sizeof(T) == 8) {
1828 return vreinterpretq_u8_u64(vsubq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1831 template <simd_
integer_element T>
1833 if constexpr (
sizeof(T) == 1) {
1834 return vmulq_u8(a, b);
1835 }
else if constexpr (
sizeof(T) == 2) {
1836 return vreinterpretq_u8_u16(vmulq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1837 }
else if constexpr (
sizeof(T) == 4) {
1838 return vreinterpretq_u8_u32(vmulq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1839 }
else if constexpr (
sizeof(T) == 8) {
1840 auto x = vreinterpretq_u64_u8(a), y = vreinterpretq_u64_u8(b);
1841 auto xl = vmovn_u64(x), yl = vmovn_u64(y);
1843 vadd_u32(vmul_u32(xl, vmovn_u64(vshrq_n_u64(y, 32))), vmul_u32(vmovn_u64(vshrq_n_u64(x, 32)), yl));
1844 return vreinterpretq_u8_u64(vaddq_u64(vmull_u32(xl, yl), vshlq_n_u64(vmovl_u32(cross), 32)));
1847 template <simd_
integer_element T>
1849 if constexpr (
sizeof(T) == 1) {
1850 return vdupq_n_u8(integer_word(x));
1851 }
else if constexpr (
sizeof(T) == 2) {
1852 return vreinterpretq_u8_u16(vdupq_n_u16(integer_word(x)));
1853 }
else if constexpr (
sizeof(T) == 4) {
1854 return vreinterpretq_u8_u32(vdupq_n_u32(integer_word(x)));
1855 }
else if constexpr (
sizeof(T) == 8) {
1856 return vreinterpretq_u8_u64(vdupq_n_u64(integer_word(x)));
1859 template <simd_
integer_element T,
unsigned S>
1860 requires(S <
sizeof(T) * 8)
1862 if constexpr (S == 0)
1865 if constexpr (
sizeof(T) == 1) {
1866 return vshlq_n_u8(a, S);
1867 }
else if constexpr (
sizeof(T) == 2) {
1868 return vreinterpretq_u8_u16(vshlq_n_u16(vreinterpretq_u16_u8(a), S));
1869 }
else if constexpr (
sizeof(T) == 4) {
1870 return vreinterpretq_u8_u32(vshlq_n_u32(vreinterpretq_u32_u8(a), S));
1871 }
else if constexpr (
sizeof(T) == 8) {
1872 return vreinterpretq_u8_u64(vshlq_n_u64(vreinterpretq_u64_u8(a), S));
1877 return vreinterpretq_u8_u32(vshlq_u32(vreinterpretq_u32_u8(a),vreinterpretq_s32_u32(counts)));
1879 template <simd_
integer_element T,
unsigned S>
1880 requires(S <
sizeof(T) * 8)
1882 if constexpr (S == 0)
1885 if constexpr (
sizeof(T) == 1) {
1886 if constexpr (std::is_unsigned_v<T>)
1887 return vshrq_n_u8(a, S);
1889 return vreinterpretq_u8_s8(vshrq_n_s8(vreinterpretq_s8_u8(a), S));
1890 }
else if constexpr (
sizeof(T) == 2) {
1891 if constexpr (std::is_unsigned_v<T>)
1892 return vreinterpretq_u8_u16(vshrq_n_u16(vreinterpretq_u16_u8(a), S));
1894 return vreinterpretq_u8_s16(vshrq_n_s16(vreinterpretq_s16_u8(a), S));
1895 }
else if constexpr (
sizeof(T) == 4) {
1896 if constexpr (std::is_unsigned_v<T>)
1897 return vreinterpretq_u8_u32(vshrq_n_u32(vreinterpretq_u32_u8(a), S));
1899 return vreinterpretq_u8_s32(vshrq_n_s32(vreinterpretq_s32_u8(a), S));
1900 }
else if constexpr (
sizeof(T) == 8) {
1901 if constexpr (std::is_unsigned_v<T>)
1902 return vreinterpretq_u8_u64(vshrq_n_u64(vreinterpretq_u64_u8(a), S));
1904 return vreinterpretq_u8_s64(vshrq_n_s64(vreinterpretq_s64_u8(a), S));
1908 template <simd_
integer_element T,
bool Greater>
1910 if constexpr (
sizeof(T) == 1) {
1911 if constexpr (!Greater)
1912 return vceqq_u8(a, b);
1913 else if constexpr (std::is_unsigned_v<T>)
1914 return vcgtq_u8(a, b);
1916 return vcgtq_s8(vreinterpretq_s8_u8(a), vreinterpretq_s8_u8(b));
1917 }
else if constexpr (
sizeof(T) == 2) {
1918 if constexpr (!Greater)
1919 return vreinterpretq_u8_u16(vceqq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1920 else if constexpr (std::is_unsigned_v<T>)
1921 return vreinterpretq_u8_u16(vcgtq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1923 return vreinterpretq_u8_u16(vcgtq_s16(vreinterpretq_s16_u8(a), vreinterpretq_s16_u8(b)));
1924 }
else if constexpr (
sizeof(T) == 4) {
1925 if constexpr (!Greater)
1926 return vreinterpretq_u8_u32(vceqq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1927 else if constexpr (std::is_unsigned_v<T>)
1928 return vreinterpretq_u8_u32(vcgtq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1930 return vreinterpretq_u8_u32(vcgtq_s32(vreinterpretq_s32_u8(a), vreinterpretq_s32_u8(b)));
1931 }
else if constexpr (
sizeof(T) == 8) {
1932 if constexpr (!Greater)
1933 return vreinterpretq_u8_u64(vceqq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1934 else if constexpr (std::is_unsigned_v<T>)
1935 return vreinterpretq_u8_u64(vcgtq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1937 return vreinterpretq_u8_u64(vcgtq_s64(vreinterpretq_s64_u8(a), vreinterpretq_s64_u8(b)));
1940 template <simd_
integer_element T,
class M>
1942 return vbslq_u8(m.to_native(), a, b);
1944 template <simd_
integer_element T,
class M>
1946 uint8x16_t b)
noexcept {
1947 return integer_select<T>(m, integer_add<T>(a, b), prior);
1949 template <simd_
integer_element T,
class M>
1951 uint8x16_t b)
noexcept {
1952 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1957 template <simd_
integer_element T,
class M>
1959 __m128i b)
noexcept {
1960 if constexpr (M::compact &&
sizeof(T) > 1) {
1961 if constexpr (
sizeof(T) == 2)
1962 return _mm_mask_mullo_epi16(prior, m.to_native(), a, b);
1963 if constexpr (
sizeof(T) == 4)
1964 return _mm_mask_mullo_epi32(prior, m.to_native(), a, b);
1965 else if constexpr (
sizeof(T) == 8)
1966 return _mm_mask_mullo_epi64(prior, m.to_native(), a, b);
1968 return integer_select<T>(m, integer_mul<T>(a, b), prior);
1972 template <simd_
integer_element T,
class M>
1974 __m256i b)
noexcept {
1975 if constexpr (M::compact &&
sizeof(T) > 1) {
1976 if constexpr (
sizeof(T) == 2)
1977 return _mm256_mask_mullo_epi16(prior, m.to_native(), a, b);
1978 if constexpr (
sizeof(T) == 4)
1979 return _mm256_mask_mullo_epi32(prior, m.to_native(), a, b);
1980 else if constexpr (
sizeof(T) == 8)
1981 return _mm256_mask_mullo_epi64(prior, m.to_native(), a, b);
1983 return integer_select<T>(m, integer_mul<T>(a, b), prior);
1986#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
1987 template <simd_
integer_element T,
class M>
1989 __m512i b)
noexcept {
1990 if constexpr (M::compact &&
sizeof(T) > 1) {
1991#if NATIVE_HAS_AVX512BW
1992 if constexpr (
sizeof(T) == 2)
1993 return _mm512_mask_mullo_epi16(prior, m.to_native(), a, b);
1995 if constexpr (
sizeof(T) == 4)
1996 return _mm512_mask_mullo_epi32(prior, m.to_native(), a, b);
1997 else if constexpr (
sizeof(T) == 8)
1998 return _mm512_mask_mullo_epi64(prior, m.to_native(), a, b);
2000 return integer_select<T>(m, integer_mul<T>(a, b), prior);
2003#if NATIVE_HAS_ARM_NEON
2004 template <simd_
integer_element T,
class M>
2006 uint8x16_t b)
noexcept {
2007 return integer_select<T>(m, integer_mul<T>(a, b), prior);
2012 return _mm_or_si128(_mm_and_si128(m,a),_mm_andnot_si128(m,b));
2017 return _mm256_or_si256(_mm256_and_si256(m,a),_mm256_andnot_si256(m,b));
2020#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2022 return _mm512_or_si512(_mm512_and_si512(m,a),_mm512_andnot_si512(m,b));
2025#if NATIVE_HAS_ARM_NEON
2032 namespace detail::NATIVE_BACKEND {
2033 template <std::
size_t Bytes>
struct integer_storage;
2035 template <>
struct integer_storage<16> {
2036 using type = __m128i;
2038 template <>
struct integer_storage<32> {
2039 using type = __m256i;
2042#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2043 template <>
struct integer_storage<64> {
2044 using type = __m512i;
2047#if NATIVE_HAS_ARM_NEON
2048 template <>
struct integer_storage<16> {
2049 using type = uint8x16_t;
2052 template <
class T, std::
size_t N>
2053 inline constexpr bool integer_shape =
2054 simd_integer_element<T> && (N == 1
2056 ||
sizeof(T) * N == 16 ||
sizeof(T) * N == 32
2058#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2059 || (
sizeof(T) * N == 64 && (
sizeof(T) >= 4
2060#if NATIVE_HAS_AVX512BW
2065#if NATIVE_HAS_ARM_NEON
2066 ||
sizeof(T) * N == 16
2071 namespace detail::NATIVE_BACKEND {
2072 template <::native::isa<> Arch, simd_
integer_element T, std::
size_t N, std::
size_t A = 1,
2073 simd_access Access = simd_access::ordinary>
2074 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2076 template <simd_integer_element T, std::size_t N, std::size_t A = 1,
2077 simd_access Access = simd_access::ordinary, ::native::isa<> Arch>
2078 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2080 template <::native::isa<> Arch, simd_
integer_element T, std::
size_t N, std::
size_t A = 1,
2081 simd_access Access = simd_access::ordinary>
2082 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2084 simd_memory<A, Access> = {})
noexcept native_diagnose_if(count > N,
"partial SIMD count exceeds the lane count");
2085 template <simd_integer_element T, std::size_t N, std::size_t A = 1,
2086 simd_access Access = simd_access::ordinary, ::native::isa<> Arch>
2087 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2089 simd_memory<A, Access> = {})
noexcept native_diagnose_if(count > N,
"partial SIMD count exceeds the lane count");
2092 template <simd_
integer_element T, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
struct simd<T, 1,Arch> : detail::swizzle_access<T,1,Arch> {
2094 template <
class U>
using rebind = simd<U,1,Arch>;
2095 template <std::
size_t A = 1>
2098 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T,1>(p,simd_memory<A>{});
2100 template <std::
size_t A = 1>
2102 native_inline constexpr void store_memory(T * p)
const noexcept {
2103 ::NATIVE_BACKEND_NAMESPACE::store_simd(p,*
this,simd_memory<A>{});
2106 using value_type = T;
2107 using native_type = T;
2108 using mask_type = ::NATIVE_BACKEND_NAMESPACE::comparison_mask<T, 1,Arch>;
2109 using mask = mask_type;
2110 using predicate_type = predicate<1,Arch>;
2111 using unsigned_register_tag = void;
2112 static constexpr std::size_t lanes = 1;
2115 constexpr simd() noexcept = default;
2117 template <simd_integer_element U> constexpr simd(U x) noexcept : value(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(x)) {}
2119 constexpr simd(std::array<T, 1>
const &x) noexcept : value(x[0]) {}
2128 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_scalar_add(a.value, b.value));
2132 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_scalar_sub(a.value, b.value));
2136 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_scalar_mul(a.value, b.value));
2141 ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(::NATIVE_BACKEND_NAMESPACE::integer_word(a.value) & ::NATIVE_BACKEND_NAMESPACE::integer_word(b.value)));
2146 ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(::NATIVE_BACKEND_NAMESPACE::integer_word(a.value) | ::NATIVE_BACKEND_NAMESPACE::integer_word(b.value)));
2151 ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(::NATIVE_BACKEND_NAMESPACE::integer_word(a.value) ^ ::NATIVE_BACKEND_NAMESPACE::integer_word(b.value)));
2155 return a ^ simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(~std::make_unsigned_t<T>(0)));
2163 return mask_type(a.value == b.value);
2167 return mask_type(a.value > b.value);
2186 template <simd_
integer_element U>
2188 return a + simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2191 template <simd_
integer_element U>
2193 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) + b;
2196 template <simd_
integer_element U>
2198 return a - simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2201 template <simd_
integer_element U>
2203 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) - b;
2206 template <simd_
integer_element U>
2208 return a * simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2211 template <simd_
integer_element U>
2213 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) * b;
2216 template <simd_
integer_element U>
2218 return a & simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2221 template <simd_
integer_element U>
2223 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) & b;
2226 template <simd_
integer_element U>
2228 return a | simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2231 template <simd_
integer_element U>
2233 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) | b;
2236 template <simd_
integer_element U>
2238 return a ^ simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2241 template <simd_
integer_element U>
2243 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) ^ b;
2246 template <simd_
integer_element U>
2248 return a == simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2251 template <simd_
integer_element U>
2253 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) == b;
2256 template <simd_
integer_element U>
2258 return a != simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2261 template <simd_
integer_element U>
2263 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) != b;
2266 template <simd_
integer_element U>
2268 return a < simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2271 template <simd_
integer_element U>
2273 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) < b;
2276 template <simd_
integer_element U>
2278 return a > simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2281 template <simd_
integer_element U>
2283 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) > b;
2286 template <simd_
integer_element U>
2288 return a <= simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2291 template <simd_
integer_element U>
2293 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) <= b;
2296 template <simd_
integer_element U>
2298 return a >= simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2301 template <simd_
integer_element U>
2303 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) >= b;
2308 template <simd_
integer_element U>
native_inline constexpr simd &
operator+=(U b)
noexcept {
return *
this = *
this + b; }
2312 template <simd_
integer_element U>
native_inline constexpr simd &
operator-=(U b)
noexcept {
return *
this = *
this - b; }
2316 template <simd_
integer_element U>
native_inline constexpr simd &
operator*=(U b)
noexcept {
return *
this = *
this * b; }
2320 template <simd_
integer_element U>
native_inline constexpr simd &
operator&=(U b)
noexcept {
return *
this = *
this & b; }
2324 template <simd_
integer_element U>
native_inline constexpr simd &
operator|=(U b)
noexcept {
return *
this = *
this | b; }
2328 template <simd_
integer_element U>
native_inline constexpr simd &
operator^=(U b)
noexcept {
return *
this = *
this ^ b; }
2330 template <
unsigned K>
2331 requires(K <
sizeof(T) * 8)
2333 using W = ::NATIVE_BACKEND_NAMESPACE::integer_work_word<T>;
2334 auto u = ::NATIVE_BACKEND_NAMESPACE::integer_word(value);
2335 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(W(u) << K));
2338 template <
unsigned K>
2339 requires(K <
sizeof(T) * 8)
2341 using U = std::make_unsigned_t<T>;
2342 using W = ::NATIVE_BACKEND_NAMESPACE::integer_work_word<T>;
2343 auto u = ::NATIVE_BACKEND_NAMESPACE::integer_word(value);
2344 if constexpr (K == 0)
2347 U result = U(W(u) >> K);
2348 if constexpr (std::is_signed_v<T>) {
2350 result = U(result | U(std::numeric_limits<U>::max() ^ (std::numeric_limits<U>::max() >> K)));
2352 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(result));
2356 template <std::
size_t K>
2357 requires(K <
sizeof(T) * 8)
2359 return a.template left<K>();
2362 template <std::
size_t K>
2363 requires(K <
sizeof(T) * 8)
2365 return a.template right<K>();
2369 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2373 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2376 native_inline constexpr void store(T *p)
const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *
this); }
2378 native_inline constexpr void storeu(T *p)
const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *
this); }
2381 T fill = T(0)) noexcept
native_diagnose_if(n > simd::lanes,
"partial SIMD count exceeds the lane count") {
2383 std::array<T, lanes> data;
2385 if consteval {
for(std::size_t i=0;i<n;++i) data[i]=p[i]; }
2386 else {
if (n) std::memcpy(data.data(),
static_cast<void const *
>(p), n *
sizeof(T)); }
2387 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(data.data());
2390 native_inline constexpr void store_partial(T *p, std::size_t n)
const noexcept native_diagnose_if(n > simd::lanes,
"partial SIMD count exceeds the lane count") { ::NATIVE_BACKEND_NAMESPACE::store_simd_partial(p, *
this, n); }
2392 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator+(simd, U) =
delete;
2394 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator+(U, simd) =
delete;
2396 template <simd_
integer_element U, std::
size_t M>
2397 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2398 friend void operator+(simd, ::native::simd<U, M,Arch>) =
delete;
2400 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator-(simd, U) =
delete;
2402 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator-(U, simd) =
delete;
2404 template <simd_
integer_element U, std::
size_t M>
2405 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2406 friend void operator-(simd, ::native::simd<U, M,Arch>) =
delete;
2408 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator*(simd, U) =
delete;
2410 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator*(U, simd) =
delete;
2412 template <simd_
integer_element U, std::
size_t M>
2413 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2414 friend void operator*(simd, ::native::simd<U, M,Arch>) =
delete;
2416 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator/(simd, U) =
delete;
2418 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator/(U, simd) =
delete;
2420 template <simd_
integer_element U, std::
size_t M>
2421 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2422 friend void operator/(simd, ::native::simd<U, M,Arch>) =
delete;
2424 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator%(simd, U) =
delete;
2426 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator%(U, simd) =
delete;
2428 template <simd_
integer_element U, std::
size_t M>
2429 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2430 friend void operator%(simd, ::native::simd<U, M,Arch>) =
delete;
2432 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator&(simd, U) =
delete;
2434 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator&(U, simd) =
delete;
2436 template <simd_
integer_element U, std::
size_t M>
2437 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2438 friend void operator&(simd, ::native::simd<U, M,Arch>) =
delete;
2440 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator|(simd, U) =
delete;
2442 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator|(U, simd) =
delete;
2444 template <simd_
integer_element U, std::
size_t M>
2445 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2446 friend void operator|(simd, ::native::simd<U, M,Arch>) =
delete;
2448 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator^(simd, U) =
delete;
2450 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator^(U, simd) =
delete;
2452 template <simd_
integer_element U, std::
size_t M>
2453 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2454 friend void operator^(simd, ::native::simd<U, M,Arch>) =
delete;
2456 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator==(simd, U) =
delete;
2458 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator==(U, simd) =
delete;
2460 template <simd_
integer_element U, std::
size_t M>
2461 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2462 friend void operator==(simd, ::native::simd<U, M,Arch>) =
delete;
2464 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator!=(simd, U) =
delete;
2466 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator!=(U, simd) =
delete;
2468 template <simd_
integer_element U, std::
size_t M>
2469 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2470 friend void operator!=(simd, ::native::simd<U, M,Arch>) =
delete;
2472 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<(simd, U) =
delete;
2474 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<(U, simd) =
delete;
2476 template <simd_
integer_element U, std::
size_t M>
2477 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2478 friend void operator<(simd, ::native::simd<U, M,Arch>) =
delete;
2480 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>(simd, U) =
delete;
2482 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>(U, simd) =
delete;
2484 template <simd_
integer_element U, std::
size_t M>
2485 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2486 friend void operator>(simd, ::native::simd<U, M,Arch>) =
delete;
2488 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<=(simd, U) =
delete;
2490 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<=(U, simd) =
delete;
2492 template <simd_
integer_element U, std::
size_t M>
2493 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2494 friend void operator<=(simd, ::native::simd<U, M,Arch>) =
delete;
2496 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>=(simd, U) =
delete;
2498 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>=(U, simd) =
delete;
2500 template <simd_
integer_element U, std::
size_t M>
2501 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2502 friend void operator>=(simd, ::native::simd<U, M,Arch>) =
delete;
2504 friend simd
operator/(simd, simd) =
delete;
2506 friend simd
operator%(simd, simd) =
delete;
2508 template <simd_
integer_element U>
friend simd
operator/(simd, U) =
delete;
2510 template <simd_
integer_element U>
friend simd
operator/(U, simd) =
delete;
2512 template <simd_
integer_element U>
friend simd
operator%(simd, U) =
delete;
2514 template <simd_
integer_element U>
friend simd
operator%(U, simd) =
delete;
2516 template <simd_
integer_element U>
friend simd
operator<<(simd, U) =
delete;
2518 template <simd_
integer_element U>
friend simd
operator>>(simd, U) =
delete;
2521#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
2525 template <simd_
integer_element T, std::
size_t N, ::native::isa<> Arch>
2526 requires NATIVE_ARCH_REQUIRES(Arch) &&(N > 1 && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>)
2530 template <std::
size_t A = 1>
2533 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T,N>(p,
simd_memory<A>{});
2535 template <std::
size_t A = 1>
2538 ::NATIVE_BACKEND_NAMESPACE::store_simd(p,*
this,
simd_memory<A>{});
2541 using value_type = T;
2542 using native_type = typename ::NATIVE_BACKEND_NAMESPACE::integer_storage<
sizeof(T) * N>
::type;
2543 using mask_type = ::NATIVE_BACKEND_NAMESPACE::comparison_mask<T, N,Arch>;
2544 using mask = mask_type;
2546 using unsigned_register_tag = void;
2547 static constexpr std::size_t lanes = N;
2548 native_type value{};
2550 constexpr simd() noexcept = default;
2553 T x = ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(input);
2555 std::array<T,N> data;
2557 value=__builtin_bit_cast(native_type,data);
2559 if constexpr (
sizeof(T) * N == 16)
2560 value = ::NATIVE_BACKEND_NAMESPACE::integer_broadcast_16(x);
2562 else if constexpr (
sizeof(T) * N == 32)
2563 value = ::NATIVE_BACKEND_NAMESPACE::integer_broadcast_32(x);
2565#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2566 else if constexpr (
sizeof(T) * N == 64)
2567 value = ::NATIVE_BACKEND_NAMESPACE::integer_broadcast_64(x);
2583 template <
class... U>
2584 requires(
sizeof...(U) == N && (simd_integer_element<U> && ...))
2586 std::array<T, N> data{::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(xs)...};
2587 if consteval { value=__builtin_bit_cast(native_type,data); }
2588 else { std::memcpy(&value, data.data(),
sizeof(value)); }
2592 if consteval { value=__builtin_bit_cast(native_type,data); }
2593 else { std::memcpy(&value, data.data(),
sizeof(value)); }
2598 operator native_type() const noexcept requires(sizeof(native_type)==16) {
return value; }
2604 native_type
to_native() const noexcept requires(sizeof(native_type)==16) {
return value; }
2607 operator native_type() const noexcept requires(sizeof(native_type)==32) {
return value; }
2613 native_type
to_native() const noexcept requires(sizeof(native_type)==32) {
return value; }
2616 operator native_type() const noexcept requires(sizeof(native_type)==64) {
return value; }
2622 native_type
to_native() const noexcept requires(sizeof(native_type)==64) {
return value; }
2634 using U=std::make_unsigned_t<T>;
2635 using V=U __attribute__((ext_vector_type(N)));
2636 return from_native(__builtin_bit_cast(native_type,
2637 __builtin_bit_cast(V,a.value) + __builtin_bit_cast(V,b.value)));
2639 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_add<T>(a.value, b.value));
2644 using U=std::make_unsigned_t<T>;
2645 using V=U __attribute__((ext_vector_type(N)));
2646 return from_native(__builtin_bit_cast(native_type,
2647 __builtin_bit_cast(V,a.value) - __builtin_bit_cast(V,b.value)));
2649 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_sub<T>(a.value, b.value));
2654 using U=std::make_unsigned_t<T>;
2655 using V=U __attribute__((ext_vector_type(N)));
2656 return from_native(__builtin_bit_cast(native_type,
2657 __builtin_bit_cast(V,a.value) * __builtin_bit_cast(V,b.value)));
2659 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_mul<T>(a.value, b.value));
2664 using U=std::make_unsigned_t<T>;
2665 using V=U __attribute__((ext_vector_type(N)));
2666 return from_native(__builtin_bit_cast(native_type,
2667 __builtin_bit_cast(V,a.value) & __builtin_bit_cast(V,b.value)));
2669 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_and(a.value, b.value));
2674 using U=std::make_unsigned_t<T>;
2675 using V=U __attribute__((ext_vector_type(N)));
2676 return from_native(__builtin_bit_cast(native_type,
2677 __builtin_bit_cast(V,a.value) | __builtin_bit_cast(V,b.value)));
2679 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_or(a.value, b.value));
2684 using U=std::make_unsigned_t<T>;
2685 using V=U __attribute__((ext_vector_type(N)));
2686 return from_native(__builtin_bit_cast(native_type,
2687 __builtin_bit_cast(V,a.value) ^ __builtin_bit_cast(V,b.value)));
2689 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_xor(a.value, b.value));
2693 return a ^
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(~std::make_unsigned_t<T>(0)));
2702 auto x=__builtin_bit_cast(std::array<T,N>,a.value);
2703 auto y=__builtin_bit_cast(std::array<T,N>,b.value);
2704 std::uint64_t
bits=0;
2705 for(std::size_t i=0;i<N;++i)
bits|=std::uint64_t(x[i] == y[i])<<i;
2706 return mask_type::from_bitset(
bits);
2708 return mask_type::unsafe_from_native(::NATIVE_BACKEND_NAMESPACE::integer_compare<T, false>(a.value, b.value));
2713 auto x=__builtin_bit_cast(std::array<T,N>,a.value);
2714 auto y=__builtin_bit_cast(std::array<T,N>,b.value);
2715 std::uint64_t
bits=0;
2716 for(std::size_t i=0;i<N;++i)
bits|=std::uint64_t(x[i] > y[i])<<i;
2717 return mask_type::from_bitset(
bits);
2719 return mask_type::unsafe_from_native(::NATIVE_BACKEND_NAMESPACE::integer_compare<T, true>(a.value, b.value));
2738 template <simd_
integer_element U>
2740 return a +
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2743 template <simd_
integer_element U>
2745 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) + b;
2748 template <simd_
integer_element U>
2750 return a -
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2753 template <simd_
integer_element U>
2755 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) - b;
2758 template <simd_
integer_element U>
2760 return a *
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2763 template <simd_
integer_element U>
2765 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) * b;
2768 template <simd_
integer_element U>
2770 return a &
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2773 template <simd_
integer_element U>
2775 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) & b;
2778 template <simd_
integer_element U>
2780 return a |
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2783 template <simd_
integer_element U>
2785 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) | b;
2788 template <simd_
integer_element U>
2790 return a ^
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2793 template <simd_
integer_element U>
2795 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) ^ b;
2798 template <simd_
integer_element U>
2800 return a ==
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2803 template <simd_
integer_element U>
2805 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) == b;
2808 template <simd_
integer_element U>
2810 return a !=
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2813 template <simd_
integer_element U>
2815 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) != b;
2818 template <simd_
integer_element U>
2820 return a < simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2823 template <simd_
integer_element U>
2825 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) < b;
2828 template <simd_
integer_element U>
2830 return a >
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2833 template <simd_
integer_element U>
2835 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) > b;
2838 template <simd_
integer_element U>
2840 return a <= simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2843 template <simd_
integer_element U>
2845 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) <= b;
2848 template <simd_
integer_element U>
2850 return a >=
simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2853 template <simd_
integer_element U>
2855 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) >= b;
2882 template <
unsigned K>
2883 requires(K <
sizeof(T) * 8)
2886 using E=std::make_unsigned_t<T>;
2887 using V=E __attribute__((ext_vector_type(N)));
2888 return from_native(__builtin_bit_cast(native_type,__builtin_bit_cast(V,value) << K));
2890 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_left<T, K>(value));
2893 template <
unsigned K>
2894 requires(K <
sizeof(T) * 8)
2898 using V=E __attribute__((ext_vector_type(N)));
2899 return from_native(__builtin_bit_cast(native_type,__builtin_bit_cast(V,value) >> K));
2901 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_right<T, K>(value));
2904 template <std::
size_t K>
2905 requires(K <
sizeof(T) * 8)
2911 requires std::same_as<T,std::uint32_t> &&
requires(native_type x) {
2912 ::NATIVE_BACKEND_NAMESPACE::integer_shift_left_variable(x,x);
2916 auto values=__builtin_bit_cast(std::array<std::uint32_t,N>,a.value);
2917 auto shifts=__builtin_bit_cast(std::array<std::uint32_t,N>,counts.value);
2918 for(std::size_t i=0;i<N;++i) values[i]=shifts[i]<32 ? values[i]<<shifts[i] : 0;
2919 return simd(values);
2921 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_shift_left_variable(a.value,counts.value));
2924 template <std::
size_t K>
2925 requires(K <
sizeof(T) * 8)
2931 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2935 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2938 native_inline constexpr void store(T *p)
const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *
this); }
2940 native_inline constexpr void storeu(T *p)
const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *
this); }
2943 T fill = T(0)) noexcept
native_diagnose_if(n >
simd::lanes,
"partial SIMD count exceeds the lane count") {
2946 std::array<T,N> data; data.fill(fill);
2947 for(std::size_t i=0;i<n;++i) data[i]=p[i];
2950 if (n == lanes)
return load(p);
2951 if (!n)
return simd(fill);
2954#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2955 if constexpr (
sizeof(T) == 4 ||
sizeof(T) == 8) {
2956 auto active = std::uint64_t((std::uint64_t(1) << n) - 1);
2957 if constexpr (
sizeof(T) * N == 64) {
2958 if constexpr (
sizeof(T) == 4)
2959 return from_native(_mm512_mask_loadu_epi32(
simd(fill).value, __mmask16(active), p));
2960 else return from_native(_mm512_mask_loadu_epi64(
simd(fill).value, __mmask8(active), p));
2962#if NATIVE_HAS_AVX512VL
2963 else if constexpr (
sizeof(T) * N == 32) {
2964 if constexpr (
sizeof(T) == 4)
2965 return from_native(_mm256_mask_loadu_epi32(
simd(fill).value, __mmask8(active), p));
2966 else return from_native(_mm256_mask_loadu_epi64(
simd(fill).value, __mmask8(active), p));
2968 if constexpr (
sizeof(T) == 4)
2969 return from_native(_mm_mask_loadu_epi32(
simd(fill).value, __mmask8(active), p));
2970 else return from_native(_mm_mask_loadu_epi64(
simd(fill).value, __mmask8(active), p));
2976 if constexpr ((
sizeof(T) == 4 ||
sizeof(T) == 8) && !mask_type::compact &&
sizeof(T) * N <= 32) {
2977 native_type active, loaded;
2978 if constexpr (
sizeof(T) * N == 32) {
2979 if constexpr (
sizeof(T) == 4) {
2980 active = _mm256_cmpgt_epi32(_mm256_set1_epi32(
int(n)), _mm256_setr_epi32(0,1,2,3,4,5,6,7));
2981 loaded = _mm256_maskload_epi32(
reinterpret_cast<int const *
>(p), active);
2983 active = _mm256_cmpgt_epi64(_mm256_set1_epi64x(
static_cast<long long>(n)), _mm256_setr_epi64x(0,1,2,3));
2984 loaded = _mm256_maskload_epi64(
reinterpret_cast<long long const *
>(p), active);
2986 }
else if constexpr (
sizeof(T) * N == 16) {
2987 if constexpr (
sizeof(T) == 4) {
2988 active = _mm_cmpgt_epi32(_mm_set1_epi32(
int(n)), _mm_setr_epi32(0,1,2,3));
2989 loaded = _mm_maskload_epi32(
reinterpret_cast<int const *
>(p), active);
2991 active = _mm_cmpgt_epi64(_mm_set1_epi64x(
static_cast<long long>(n)), _mm_set_epi64x(1,0));
2992 loaded = _mm_maskload_epi64(
reinterpret_cast<long long const *
>(p), active);
2995 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_bit_select(active, loaded,
simd(fill).value));
2998#if NATIVE_HAS_ARM_NEON
2999 if constexpr (
sizeof(T) == 4) {
3000 auto result = vreinterpretq_u32_u8(
simd(fill).value);
3002 auto bytes =
reinterpret_cast<unsigned char const *
>(p);
3004 case 3: std::memcpy(&word, bytes + 8, 4); result = vsetq_lane_u32(word, result, 2); [[fallthrough]];
3005 case 2: std::memcpy(&word, bytes + 4, 4); result = vsetq_lane_u32(word, result, 1); [[fallthrough]];
3006 case 1: std::memcpy(&word, bytes, 4); result = vsetq_lane_u32(word, result, 0);
3009 }
else if constexpr (
sizeof(T) == 8) {
3011 std::memcpy(&word,
static_cast<void const *
>(p), 8);
3012 return from_native(vreinterpretq_u8_u64(vsetq_lane_u64(word, vreinterpretq_u64_u8(
simd(fill).value), 0)));
3015 std::array<T, lanes> data;
3018 std::memcpy(data.data(),
static_cast<void const *
>(p), n *
sizeof(T));
3019 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(data.data());
3025 auto data=__builtin_bit_cast(std::array<T,N>,value);
3026 for(std::size_t i=0;i<n;++i) p[i]=data[i];
3028 } ::NATIVE_BACKEND_NAMESPACE::store_simd_partial(p, *
this, n); }
3030 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator+(
simd, U) =
delete;
3032 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator+(U,
simd) =
delete;
3034 template <simd_
integer_element U, std::
size_t M>
3035 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3038 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator-(
simd, U) =
delete;
3040 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator-(U,
simd) =
delete;
3042 template <simd_
integer_element U, std::
size_t M>
3043 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3046 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator*(
simd, U) =
delete;
3048 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator*(U,
simd) =
delete;
3050 template <simd_
integer_element U, std::
size_t M>
3051 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3054 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator/(
simd, U) =
delete;
3056 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator/(U,
simd) =
delete;
3058 template <simd_
integer_element U, std::
size_t M>
3059 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3062 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator%(
simd, U) =
delete;
3064 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator%(U,
simd) =
delete;
3066 template <simd_
integer_element U, std::
size_t M>
3067 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3070 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator&(
simd, U) =
delete;
3072 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator&(U,
simd) =
delete;
3074 template <simd_
integer_element U, std::
size_t M>
3075 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3078 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator|(
simd, U) =
delete;
3080 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator|(U,
simd) =
delete;
3082 template <simd_
integer_element U, std::
size_t M>
3083 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3086 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator^(
simd, U) =
delete;
3088 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator^(U,
simd) =
delete;
3090 template <simd_
integer_element U, std::
size_t M>
3091 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3094 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator==(
simd, U) =
delete;
3096 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator==(U,
simd) =
delete;
3098 template <simd_
integer_element U, std::
size_t M>
3099 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3102 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator!=(
simd, U) =
delete;
3104 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator!=(U,
simd) =
delete;
3106 template <simd_
integer_element U, std::
size_t M>
3107 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3110 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<(
simd, U) =
delete;
3112 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<(U,
simd) =
delete;
3114 template <simd_
integer_element U, std::
size_t M>
3115 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3118 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>(
simd, U) =
delete;
3120 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>(U,
simd) =
delete;
3122 template <simd_
integer_element U, std::
size_t M>
3123 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3126 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<=(
simd, U) =
delete;
3128 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator<=(U,
simd) =
delete;
3130 template <simd_
integer_element U, std::
size_t M>
3131 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3134 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>=(
simd, U) =
delete;
3136 template <
class U>
requires (std::is_arithmetic_v<U> && !simd_integer_element<U>)
friend void operator>=(U,
simd) =
delete;
3138 template <simd_
integer_element U, std::
size_t M>
3139 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3161 namespace detail::NATIVE_BACKEND {
3162 template <::native::isa<> Arch, simd_
integer_element T, std::
size_t N, std::
size_t A, simd_access Access>
3163 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3167 if constexpr(N==1)
return V(*p);
3169 std::array<T,N> data;
3170 for(std::size_t i=0;i<N;++i) data[i]=p[i];
3174 typename V::native_type value;
3178#if defined(__GNUC__) || defined(__clang__)
3179 if constexpr (A > 1)
3180 p =
static_cast<T
const *
>(__builtin_assume_aligned(p, A));
3181#elif defined(_MSC_VER)
3182 if constexpr (A > 1)
3183 __assume((
reinterpret_cast<std::uintptr_t
>(p) & (A - 1)) == 0);
3185 std::memcpy(&value,
static_cast<void const *
>(p),
sizeof(value));
3186 return V::from_native(value);
3188 template <simd_
integer_element T, std::
size_t N, std::
size_t A, simd_access Access, ::native::isa<> Arch>
3189 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3190 native_inline constexpr void store_simd(T *p, simd<T, N,Arch> v, simd_memory<A, Access>)
noexcept {
3192 if constexpr(N==1) *p=v.value;
3194 auto data=__builtin_bit_cast(std::array<T,N>,v.value);
3195 for(std::size_t i=0;i<N;++i) p[i]=data[i];
3199#if defined(__GNUC__) || defined(__clang__)
3200 if constexpr (A > 1)
3201 p =
static_cast<T *
>(__builtin_assume_aligned(p, A));
3202#elif defined(_MSC_VER)
3203 if constexpr (A > 1)
3204 __assume((
reinterpret_cast<std::uintptr_t
>(p) & (A - 1)) == 0);
3206 std::memcpy(
static_cast<void *
>(p), &v.value,
sizeof(v.value));
3208 template <::native::isa<> Arch, simd_
integer_element T, std::
size_t N, std::
size_t A, simd_access Access>
3209 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3211 simd_memory<A, Access>)
noexcept native_diagnose_if(count > N,
"partial SIMD count exceeds the lane count") {
3213 std::array<T, N> data{};
3214 if consteval {
for(std::size_t i=0;i<count;++i) data[i]=p[i]; }
3215 else {
if (count) std::memcpy(data.data(),
static_cast<void const *
>(p), count *
sizeof(T)); }
3216 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, N>(data.data());
3218 template <simd_
integer_element T, std::
size_t N, std::
size_t A, simd_access Access, ::native::isa<> Arch>
3219 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3221 simd_memory<A, Access>)
noexcept native_diagnose_if(count > N,
"partial SIMD count exceeds the lane count") {
3223 std::array<T, N> data;
3225 if consteval {
for(std::size_t i=0;i<count;++i) p[i]=data[i]; }
3226 else {
if (count) std::memcpy(
static_cast<void *
>(p), data.data(), count *
sizeof(T)); }
3228 template <::native::isa<> Arch, simd_
integer_element T, std::
size_t N>
3229 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3231 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, N>(p.data());
3233 template <::native::isa<> Arch, simd_
integer_element T, std::
size_t N>
3234 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3236 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, N>(p.data());
3241 namespace detail::NATIVE_BACKEND {
3242 template <
class M,
class T, std::
size_t N, isa<> Arch>
3243 concept integer_mask_for =
3244 std::same_as<M, typename simd<T, N,Arch>::mask_type> || std::same_as<M, simd<::NATIVE_BACKEND_NAMESPACE::mask_lane_for<T>, N,Arch>>;
3248 template <simd_
integer_element T, std::
size_t N,
class M, ::native::isa<> Arch>
3249 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3252 std::array<T,N> first{},second{};
3253 a.store(first.data()); b.store(second.data());
3254 auto bits=m.to_bitset();
3255 for(std::size_t i=0;i<N;++i)
if(!((bits>>i)&1)) first[i]=second[i];
3258 if constexpr (N == 1)
3259 return any(m) ? a : b;
3260#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3267 template <simd_
integer_element T, std::
size_t N, ::native::isa<> Arch>
3268 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3271 if consteval {
return (bits & a) | (~bits & b); }
3272 if constexpr(N==1)
return (bits & a) | (~bits & b);
3273#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3280 template <simd_
integer_element T, std::
size_t N, ::native::isa<> Arch>
3281 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3288 template <simd_mask_element M, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
3290 using U =
typename M::storage_type;
3295 template <simd_
integer_element T, simd_mask_element M, std::
size_t N, ::native::isa<> Arch>
3296 requires NATIVE_ARCH_REQUIRES(Arch) &&(
sizeof(T) ==
sizeof(M))
3302 template <simd_
integer_element T, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) &&
3303 requires(predicate<N,Arch> m) { to_vector_mask<::NATIVE_BACKEND_NAMESPACE::mask_lane_for<T>>(m); }
3309 template <simd_
integer_element T, std::
size_t N,
class M, ::native::isa<> Arch>
3310 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3313 if consteval {
return select(m,a + b,prior); }
3314 if constexpr (N == 1)
3315 return select(m, a + b, prior);
3316#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3323 template <simd_
integer_element T, std::
size_t N,
class M, ::native::isa<> Arch>
3324 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3330 template <simd_
integer_element T, std::
size_t N,
class M, ::native::isa<> Arch>
3331 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3334 if consteval {
return select(m,a - b,prior); }
3335 if constexpr (N == 1)
3336 return select(m, a - b, prior);
3337#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3344 template <simd_
integer_element T, std::
size_t N,
class M, ::native::isa<> Arch>
3345 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3351 template <simd_
integer_element T, std::
size_t N,
class M, ::native::isa<> Arch>
3352 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3355 if consteval {
return select(m,a * b,prior); }
3356 if constexpr (N == 1)
3357 return select(m, a * b, prior);
3358#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3365 template <simd_
integer_element T, std::
size_t N,
class M, ::native::isa<> Arch>
3366 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3376 template<simd_
integer_element T, std::
size_t N, std::
size_t K, ::native::isa<> Arch>
3377 requires NATIVE_ARCH_REQUIRES(Arch) && (K >=
sizeof(T) * 8)
3380 template<simd_
integer_element T, std::
size_t N, std::
size_t K, ::native::isa<> Arch>
3381 requires NATIVE_ARCH_REQUIRES(Arch) && (K >=
sizeof(T) * 8)
3385namespace NATIVE_BACKEND_NAMESPACE::native {
3387 using uint32x1=::native::simd<::native::uint32_t,1,NATIVE_DEFAULT_ARCH>;
3388#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
3389 using uint32x4=::native::simd<::native::uint32_t,4,NATIVE_DEFAULT_ARCH>;
3392 using uint32x8=::native::simd<::native::uint32_t,8,NATIVE_DEFAULT_ARCH>;
3394#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
3395 using uint32x16=::native::simd<::native::uint32_t,16,NATIVE_DEFAULT_ARCH>;
3397 template<
class V>
concept unsigned_register =
requires {
3398 typename V::unsigned_register_tag;
3399 typename V::value_type;
3400 } && std::same_as<typename V::value_type,::native::uint32_t>;
3401 template<
unsigned S,
unsigned_register V>
requires(S<32)
3403 template<
unsigned S,
unsigned_register V>
requires(S<32)
3407 struct integer_target {
3408#if defined(NATIVE_INTEGER_SCALAR)
3409 using unsigned_type=uint32x1;
3410#elif NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
3411 using unsigned_type=uint32x16;
3412#elif NATIVE_HAS_AVX2
3413 using unsigned_type=uint32x8;
3414#elif NATIVE_HAS_ARM_NEON
3415 using unsigned_type=uint32x4;
3417 using unsigned_type=uint32x1;
3419 static constexpr std::size_t lanes=unsigned_type::lanes;
3424 namespace detail::NATIVE_BACKEND {
3428 namespace detail::NATIVE_BACKEND {
3429 template <
class T>
inline constexpr bool custom_argument = ::native::simd_custom_element<std::remove_cvref_t<T>>;
3430 template <
class T, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
inline constexpr bool custom_argument<simd<T,N,Arch>> = ::native::simd_custom_element<T>;
3432 namespace detail::NATIVE_BACKEND {
3434 using ::native::detail::register_memory;
3436 template <::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
struct native_empty_bases
simd<float, 1,Arch> : ::NATIVE_BACKEND_NAMESPACE::register_memory<simd<float,1,Arch>, 1>, detail::swizzle_access<float,1,Arch> {
3438 template <
class T>
using rebind = simd<T,1,Arch>;
3439 using vector_mask_type=simd<mask32,1,Arch>;
3440 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<float,1,Arch>;
3441 using mask = mask_type;
3442 using predicate_type = predicate<1,Arch>;
3451 native_inline constexpr void store(
native_noescape float * p)
const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*
this); }
3454 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3455 return simd(a.value + b.value);
3459 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
3460 return simd(a.value - b.value);
3464 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
3465 return simd(a.value * b.value);
3469 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
3470 return simd(a.value / b.value);
3474 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
3475 return simd(-a.value);
3479 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
3480 return mask_type::from_native(a.value < b.value ? ~std::uint32_t(0) : 0u);
3484 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
3485 return mask_type::from_native(a.value > b.value ? ~std::uint32_t(0) : 0u);
3489 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
3490 return mask_type::from_native(a.value == b.value ? ~std::uint32_t(0) : 0u);
3493 template<
class M>
requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
3495 if consteval { return ::native::detail::float_constant::select(m,a,b); }
3496 return m.to_native()!=0 ? a : b;
3500 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
3501 return simd(std::fma(a.value, b.value, c.value));
3505 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
3506 return simd(std::sqrt(a.value));
3510 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
3511 if (!std::isfinite(a.value) || std::abs(a.value) >= 0x1p23f)
return a;
3512 float lo = std::floor(a.value), delta = a.value - lo;
3513 float r = lo + float(delta > .5f || (delta == .5f && std::fmod(lo, 2.f) != 0));
3514 return simd(r == 0 ? std::copysign(0.f, a.value) : r);
3519 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
3520 return simd(std::bit_cast<float>(std::uint32_t(
int(n.value) + 127) << 23));
3523 template <std::
size_t Alignment = 1>
3526 return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,Alignment>(p);
3528 template <std::
size_t Alignment = 1>
3530 native_inline constexpr void store_memory(
float * p)
const noexcept {
3531 ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,Alignment>(p,*
this);
3533 using value_type = float;
3534 using register_type = simd;
3535 using native_type = float;
3536 using bits_type = simd<uint32_t,1,Arch>;
3558 native_inline constexpr void storeu(
native_noescape float * p)
const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*
this); }
3568 native_inline constexpr simd(std::array<float,1>
const & values) noexcept : simd(loadu(values.data())) {}
3590 template <::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
struct native_empty_bases
simd<float, 4,Arch> : detail::register_memory<simd<float,4,Arch>, 4>, detail::swizzle_access<float,4,Arch> {
3591 static constexpr isa<> architecture=Arch;
3595 using mask = mask_type;
3606 if consteval { std::array<float,
sizeof(native_type)/
sizeof(
float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
3607 else { value=_mm_set1_ps(x); }
3617 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3618 return simd(_mm_add_ps(a.value, b.value));
3622 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
3623 return simd(_mm_sub_ps(a.value, b.value));
3627 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
3628 return simd(_mm_mul_ps(a.value, b.value));
3632 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
3633 return simd(_mm_div_ps(a.value, b.value));
3637 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
3638 return simd(_mm_xor_ps(a.value, _mm_set1_ps(-0.f)));
3642 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
3643 if constexpr(bool(NATIVE_HAS_AVX512VL))
3644 return mask_type::from_native(_mm_cmp_ps_mask(a.value,b.value,_CMP_LT_OQ));
3646 return mask_type::unsafe_from_native(_mm_castps_si128(_mm_cmp_ps(a.value,b.value,_CMP_LT_OQ)));
3650 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
3651 if constexpr(bool(NATIVE_HAS_AVX512VL))
3652 return mask_type::from_native(_mm_cmp_ps_mask(a.value,b.value,_CMP_GT_OQ));
3654 return mask_type::unsafe_from_native(_mm_castps_si128(_mm_cmp_ps(a.value,b.value,_CMP_GT_OQ)));
3658 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
3659 if constexpr(bool(NATIVE_HAS_AVX512VL))
3660 return mask_type::from_native(_mm_cmp_ps_mask(a.value,b.value,_CMP_EQ_OQ));
3662 return mask_type::unsafe_from_native(_mm_castps_si128(_mm_cmp_ps(a.value,b.value,_CMP_EQ_OQ)));
3665 template<
class M>
requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
3667 if consteval { return ::native::detail::float_constant::select(m,a,b); }
3668 if constexpr(M::compact)
return simd(_mm_mask_blend_ps(m.to_native(),b.value,a.value));
3670 return simd(_mm_blendv_ps(b.value,a.value,_mm_castsi128_ps(m.to_native()))); }
3673 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
3674 return simd(_mm_fmadd_ps(a.value, b.value, c.value));
3678 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
3679 return simd(_mm_sqrt_ps(a.value));
3683 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
3684 return simd(_mm_round_ps(a.value, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
3688 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
3689 return simd(_mm_castsi128_ps(_mm_slli_epi32(_mm_add_epi32(_mm_cvttps_epi32(n.value), _mm_set1_epi32(127)), 23)));
3692 template <std::
size_t Alignment = 1>
3696 std::array<float,
sizeof(native_type)/
sizeof(
float)> values{};
3697 for (std::size_t i=0;i<
sizeof(native_type)/
sizeof(float);++i) values[i]=p[i];
3698 return from_native(__builtin_bit_cast(native_type,values));
3700 if constexpr(Alignment>=16)
return simd(_mm_load_ps(p));
3701 else return simd(_mm_loadu_ps(p));
3703 template <std::
size_t Alignment = 1>
3707 auto values=__builtin_bit_cast(std::array<
float,
sizeof(native_type)/
sizeof(
float)>,value);
3708 for (std::size_t i=0;i<
sizeof(native_type)/
sizeof(float);++i) p[i]=values[i];
3711 if constexpr(Alignment>=16) _mm_store_ps(p,value);
3712 else _mm_storeu_ps(p,value);
3714 using value_type = float;
3715 using register_type =
simd;
3716 using native_type = __m128;
3750#if defined(__clang__)
3752 template <
class... X>
requires (
sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
3753 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) : value{
static_cast<float>(x)...} {}
3756 template <
class... X>
requires (
sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
3757 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) :
simd(
loadu(std::array<float,4>{
static_cast<float>(x)...}.data())) {}
3777 template <::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
struct simd<float, 8,Arch> : detail::register_memory<simd<float,8,Arch>, 8> {
3778 static constexpr isa<> architecture=Arch;
3782 using mask = mask_type;
3793 if consteval { std::array<float,
sizeof(native_type)/
sizeof(
float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
3794 else { value=_mm256_set1_ps(x); }
3804 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3805 return simd(_mm256_add_ps(a.value, b.value));
3809 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
3810 return simd(_mm256_sub_ps(a.value, b.value));
3814 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
3815 return simd(_mm256_mul_ps(a.value, b.value));
3819 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
3820 return simd(_mm256_div_ps(a.value, b.value));
3824 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
3825 return simd(_mm256_xor_ps(a.value, _mm256_set1_ps(-0.f)));
3829 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
3830 if constexpr(bool(NATIVE_HAS_AVX512VL))
3831 return mask_type::from_native(_mm256_cmp_ps_mask(a.value,b.value,_CMP_LT_OQ));
3833 return mask_type::unsafe_from_native(_mm256_castps_si256(_mm256_cmp_ps(a.value,b.value,_CMP_LT_OQ)));
3837 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
3838 if constexpr(bool(NATIVE_HAS_AVX512VL))
3839 return mask_type::from_native(_mm256_cmp_ps_mask(a.value,b.value,_CMP_GT_OQ));
3841 return mask_type::unsafe_from_native(_mm256_castps_si256(_mm256_cmp_ps(a.value,b.value,_CMP_GT_OQ)));
3845 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
3846 if constexpr(bool(NATIVE_HAS_AVX512VL))
3847 return mask_type::from_native(_mm256_cmp_ps_mask(a.value,b.value,_CMP_EQ_OQ));
3849 return mask_type::unsafe_from_native(_mm256_castps_si256(_mm256_cmp_ps(a.value,b.value,_CMP_EQ_OQ)));
3852 template<
class M>
requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
3854 if consteval { return ::native::detail::float_constant::select(m,a,b); }
3855 if constexpr(M::compact)
return simd(_mm256_mask_blend_ps(m.to_native(),b.value,a.value));
3857 return simd(_mm256_blendv_ps(b.value,a.value,_mm256_castsi256_ps(m.to_native()))); }
3860 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
3861 return simd(_mm256_fmadd_ps(a.value, b.value, c.value));
3865 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
3866 return simd(_mm256_sqrt_ps(a.value));
3870 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
3871 return simd(_mm256_round_ps(a.value, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
3875 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
3876 return simd(_mm256_castsi256_ps(_mm256_slli_epi32(_mm256_add_epi32(_mm256_cvttps_epi32(n.value), _mm256_set1_epi32(127)), 23)));
3879 template <std::
size_t Alignment = 1>
3883 std::array<float,
sizeof(native_type)/
sizeof(
float)> values{};
3884 for (std::size_t i=0;i<
sizeof(native_type)/
sizeof(float);++i) values[i]=p[i];
3885 return from_native(__builtin_bit_cast(native_type,values));
3887 if constexpr(Alignment>=32)
return simd(_mm256_load_ps(p));
3888 else return simd(_mm256_loadu_ps(p));
3890 template <std::
size_t Alignment = 1>
3894 auto values=__builtin_bit_cast(std::array<
float,
sizeof(native_type)/
sizeof(
float)>,value);
3895 for (std::size_t i=0;i<
sizeof(native_type)/
sizeof(float);++i) p[i]=values[i];
3898 if constexpr(Alignment>=32) _mm256_store_ps(p,value);
3899 else _mm256_storeu_ps(p,value);
3901 using value_type = float;
3902 using register_type =
simd;
3903 using native_type = __m256;
3937#if defined(__clang__)
3939 template <
class... X>
requires (
sizeof...(X)==8) && (std::convertible_to<X,float> && ...)
3940 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) : value{
static_cast<float>(x)...} {}
3943 template <
class... X>
requires (
sizeof...(X)==8) && (std::convertible_to<X,float> && ...)
3944 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) :
simd(
loadu(std::array<float,8>{
static_cast<float>(x)...}.data())) {}
3968#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
3969 template <::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
struct simd<float, 16,Arch> : ::NATIVE_BACKEND_NAMESPACE::register_memory<simd<float,16,Arch>, 16> {
3971 template <
class T>
using rebind = simd<T,16,Arch>;
3972 using vector_mask_type=simd<mask32,16,Arch>;
3973 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<float,16,Arch>;
3974 using mask = mask_type;
3975 using predicate_type = predicate<16,Arch>;
3985 if consteval { std::array<float,
sizeof(native_type)/
sizeof(
float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
3986 else { value=_mm512_set1_ps(x); }
3993 native_inline constexpr void store(
native_noescape float * p)
const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*
this); }
3996 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3997 return simd(_mm512_add_ps(a.value, b.value));
4001 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
4002 return simd(_mm512_sub_ps(a.value, b.value));
4006 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
4007 return simd(_mm512_mul_ps(a.value, b.value));
4011 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
4012 return simd(_mm512_div_ps(a.value, b.value));
4016 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
4017 return simd(_mm512_xor_ps(a.value, _mm512_set1_ps(-0.f)));
4021 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
4022 return mask_type::from_native(_mm512_cmp_ps_mask(a.value,b.value,_CMP_LT_OQ));
4026 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
4027 return mask_type::from_native(_mm512_cmp_ps_mask(a.value,b.value,_CMP_GT_OQ));
4031 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
4032 return mask_type::from_native(_mm512_cmp_ps_mask(a.value,b.value,_CMP_EQ_OQ));
4035 template<
class M>
requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
4037 if consteval { return ::native::detail::float_constant::select(m,a,b); }
4038 if constexpr(M::compact)
return simd(_mm512_mask_blend_ps(m.to_native(),b.value,a.value));
4039 else return simd(_mm512_castsi512_ps(_mm512_or_si512(_mm512_and_si512(m.to_native(),_mm512_castps_si512(a.value)),
4040 _mm512_andnot_si512(m.to_native(),_mm512_castps_si512(b.value))))); }
4043 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
4044 return simd(_mm512_fmadd_ps(a.value, b.value, c.value));
4048 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
4049 return simd(_mm512_sqrt_ps(a.value));
4053 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
4054 return simd(_mm512_roundscale_ps(a.value, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
4058 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
4059 return simd(_mm512_castsi512_ps(_mm512_slli_epi32(_mm512_add_epi32(_mm512_cvttps_epi32(n.value), _mm512_set1_epi32(127)), 23)));
4062 template <std::
size_t Alignment = 1>
4065 return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,Alignment>(p);
4067 template <std::
size_t Alignment = 1>
4069 native_inline constexpr void store_memory(
float * p)
const noexcept {
4070 ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,Alignment>(p,*
this);
4072 using value_type = float;
4073 using register_type = simd;
4074 using native_type = __m512;
4075 using bits_type = simd<uint32_t,16,Arch>;
4097 native_inline constexpr void storeu(
native_noescape float * p)
const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*
this); }
4107 native_inline constexpr simd(std::array<float,16>
const & values) noexcept : simd(loadu(values.data())) {}
4108#if defined(__clang__)
4110 template <
class... X>
requires (
sizeof...(X)==16) && (std::convertible_to<X,float> && ...)
4111 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) : value{
static_cast<float>(x)...} {}
4114 template <
class... X>
requires (
sizeof...(X)==16) && (std::convertible_to<X,float> && ...)
4115 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) : simd(loadu(std::array<float,16>{
static_cast<float>(x)...}.data())) {}
4133#if NATIVE_HAS_ARM_NEON
4134 template <::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
struct alignas(float32x4_t) native_empty_bases simd<float, 4,Arch> : ::NATIVE_BACKEND_NAMESPACE::register_memory<simd<float,4,Arch>, 4>, detail::swizzle_access<float,4,Arch> {
4136 template <
class T>
using rebind = simd<T,4,Arch>;
4137 using vector_mask_type=simd<mask32,4,Arch>;
4138 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<float,4,Arch>;
4139 using mask = mask_type;
4140 using predicate_type = predicate<4,Arch>;
4150 if consteval { std::array<float,
sizeof(native_type)/
sizeof(
float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
4151 else { value=vdupq_n_f32(x); }
4158 native_inline constexpr void store(
native_noescape float * p)
const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*
this); }
4161 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
4162 return simd(vaddq_f32(a.value, b.value));
4166 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
4167 return simd(vsubq_f32(a.value, b.value));
4171 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
4172 return simd(vmulq_f32(a.value, b.value));
4176 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
4177 return simd(vdivq_f32(a.value, b.value));
4181 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
4182 return simd(vnegq_f32(a.value));
4193 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
4194 auto x=::native::detail::arm_register_order(a.value);
4195 auto y=::native::detail::arm_register_order(b.value);
4197 asm volatile(
"fcmgt %0.4s, %1.4s, %2.4s" :
"=w"(bits) :
"w"(x),
"w"(y) :
"memory");
4198 return mask_type::unsafe_from_native(vreinterpretq_u8_u32(::native::detail::arm_register_order(bits)));
4202 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
4203 auto x=::native::detail::arm_register_order(a.value);
4204 auto y=::native::detail::arm_register_order(b.value);
4206 asm volatile(
"fcmeq %0.4s, %1.4s, %2.4s" :
"=w"(bits) :
"w"(x),
"w"(y) :
"memory");
4207 return mask_type::unsafe_from_native(vreinterpretq_u8_u32(::native::detail::arm_register_order(bits)));
4210 template<
class M>
requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
4212 if consteval { return ::native::detail::float_constant::select(m,a,b); }
4213 return simd(vbslq_f32(vreinterpretq_u32_u8(m.to_native()),a.value,b.value));
4217 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
4218 return simd(vfmaq_f32(c.value, a.value, b.value));
4222 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
4223 return simd(vsqrtq_f32(a.value));
4227 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
4228 return simd(vrndnq_f32(a.value));
4232 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
4233 return simd(vreinterpretq_f32_s32(vshlq_n_s32(vaddq_s32(vcvtq_s32_f32(n.value), vdupq_n_s32(127)), 23)));
4236 template <std::
size_t Alignment = 1>
4239 return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,Alignment>(p);
4241 template <std::
size_t Alignment = 1>
4243 native_inline constexpr void store_memory(
float * p)
const noexcept {
4244 ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,Alignment>(p,*
this);
4246 using value_type = float;
4247 using register_type = simd;
4248 using native_type = float32x4_t;
4249 using bits_type = simd<uint32_t,4,Arch>;
4271 native_inline constexpr void storeu(
native_noescape float * p)
const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*
this); }
4281 native_inline constexpr simd(std::array<float,4>
const & values) noexcept : simd(loadu(values.data())) {}
4282#if defined(__clang__)
4284 template <
class... X>
requires (
sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
4285 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) : value{
static_cast<float>(x)...} {}
4287 template <
class... X>
requires (
sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
4289 native_inline constexpr simd(X... x)
noexcept((
noexcept(
static_cast<float>(x)) && ...)) : simd(loadu(std::array<float,4>{
static_cast<float>(x)...}.data())) {}
4306 return ::native::detail::float_constant::compare([](
auto x,
auto y) {
4307 return ::native::detail::float_constant::less(y,x) || ::native::detail::float_constant::equal(x,y);
4310 auto x=::native::detail::arm_register_order(a.value);
4311 auto y=::native::detail::arm_register_order(b.value);
4313 asm volatile(
"fcmge %0.4s, %1.4s, %2.4s" :
"=w"(bits) :
"w"(x),
"w"(y) :
"memory");
4314 return mask_type::unsafe_from_native(vreinterpretq_u8_u32(::native::detail::arm_register_order(bits)));
4318 namespace detail::NATIVE_BACKEND {
4322 template<std::
size_t N>
inline constexpr bool native_scaleb_shape =
4323#if NATIVE_HAS_AVX512F
4325#if NATIVE_HAS_AVX512VL
4326 || N==2 || N==3 || N==4 || N==8
4342 template <std::
size_t N,
class M, ::native::isa<> Arch>
4343 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::native_scaleb_shape<N> &&(std::same_as<M,typename simd<float,N,Arch>::mask_type> ||
4344 std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
4348 auto a=::native::detail::float_constant::words(value),b=::native::detail::float_constant::words(exponent);
4349 auto result=::native::detail::float_constant::words(prior);
4350 auto active=
mask.to_bitset();
4351 for(std::size_t i=0;i<N;++i)
if((active>>i)&1) result[i]=::native::detail::float_constant::scale(a[i],b[i]);
4354#if NATIVE_HAS_AVX512F
4355 if constexpr(N==2 || N==3) {
4359 return __builtin_bit_cast(V,_mm_mask_scalef_ps(__builtin_bit_cast(__m128,prior.to_native()),
4360 __mmask8(
mask.to_bitset()),__builtin_bit_cast(__m128,value.to_native()),
4361 __builtin_bit_cast(__m128,exponent.to_native())));
4365 _mm_set_ss(prior.value),__mmask8(native_mask()),_mm_set_ss(value.value),_mm_set_ss(exponent.value))));
4366 else if constexpr(N==16)
return simd<float,N,Arch>(_mm512_mask_scalef_ps(prior.value,native_mask(),value.value,exponent.value));
4367#if NATIVE_HAS_AVX512VL
4368 else if constexpr(N==4)
return simd<float,N,Arch>(_mm_mask_scalef_ps(prior.value,native_mask(),value.value,exponent.value));
4369 else if constexpr(N==8)
return simd<float,N,Arch>(_mm256_mask_scalef_ps(prior.value,native_mask(),value.value,exponent.value));
4376 template <std::
size_t N,
class M, ::native::isa<> Arch>
4377 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::native_scaleb_shape<N> &&(std::same_as<M,typename simd<float,N,Arch>::mask_type> ||
4378 std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
4382#if NATIVE_HAS_AVX512F
4385 __mmask8(native_mask()),_mm_set_ss(value.value),_mm_set_ss(exponent.value))));
4386 else if constexpr(N==16)
return simd<float,N,Arch>(_mm512_maskz_scalef_ps(native_mask(),value.value,exponent.value));
4387#if NATIVE_HAS_AVX512VL
4388 else if constexpr(N==2 || N==3)
return __builtin_bit_cast(
simd<float,N,Arch>,_mm_maskz_scalef_ps(
4389 __mmask8(
mask.to_bitset()),__builtin_bit_cast(__m128,value.to_native()),__builtin_bit_cast(__m128,exponent.to_native())));
4390 else if constexpr(N==4)
return simd<float,N,Arch>(_mm_maskz_scalef_ps(native_mask(),value.value,exponent.value));
4391 else if constexpr(N==8)
return simd<float,N,Arch>(_mm256_maskz_scalef_ps(native_mask(),value.value,exponent.value));
4398 template <std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::native_scaleb_shape<N>
4404 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4407 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4410 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4413 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4416 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4419 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4422 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4425 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4428 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4431 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4434 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4437 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4440 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4443 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4446 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4449 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4452 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4455 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4458 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4461 template <std::
size_t N,
class U, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4464 template <std::
size_t N,
class A,
class B, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<A> && !::NATIVE_BACKEND_NAMESPACE::custom_argument<B>) && std::convertible_to<A,simd<float,N,Arch>> && std::convertible_to<B,simd<float,N,Arch>>
4465 native_nodiscard native_inline constexpr auto fma(
simd<float,N,Arch> a,A b,B c)
noexcept(
noexcept(
simd<float,N,Arch>(b)) &&
noexcept(
simd<float,N,Arch>(c))) {
return fma(a,
simd<float,N,Arch>(b),
simd<float,N,Arch>(c)); }
4467 template <std::
size_t N,
class A,
class B, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<A> && !::NATIVE_BACKEND_NAMESPACE::custom_argument<B>) && (!std::same_as<A,simd<float,N,Arch>>) && std::convertible_to<A,simd<float,N,Arch>> && std::convertible_to<B,simd<float,N,Arch>>
4468 native_nodiscard native_inline constexpr auto fma(A a,
simd<float,N,Arch> b,B c)
noexcept(
noexcept(
simd<float,N,Arch>(a)) &&
noexcept(
simd<float,N,Arch>(c))) {
return fma(
simd<float,N,Arch>(a),b,
simd<float,N,Arch>(c)); }
4470 template <std::
size_t N,
class A,
class B, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<A> && !::NATIVE_BACKEND_NAMESPACE::custom_argument<B>) && (!std::same_as<A,simd<float,N,Arch>>) && (!std::same_as<B,simd<float,N,Arch>>) && std::convertible_to<A,simd<float,N,Arch>> && std::convertible_to<B,simd<float,N,Arch>>
4471 native_nodiscard native_inline constexpr auto fma(A a,B b,
simd<float,N,Arch> c)
noexcept(
noexcept(
simd<float,N,Arch>(a)) &&
noexcept(
simd<float,N,Arch>(b))) {
return fma(
simd<float,N,Arch>(a),
simd<float,N,Arch>(b),c); }
4474 template <std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
4484 template<std::
size_t N, ::native::isa<> Arch>
4485 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
4487 fcvtzs(simd<float,N,Arch> x)
noexcept {
4488 using I = simd<std::int32_t,N,Arch>;
4490 std::array<float,N> values{};
4491 std::array<std::int32_t,N> result{};
4492 x.store(values.data());
4493 for (std::size_t i = 0; i < N; ++i)
4494 result[i] = detail::float_constant::fcvtzs(std::bit_cast<std::uint32_t>(values[i]));
4495 return I::load(result.data());
4497 if constexpr (N == 1)
return I(vcvts_s32_f32(x.value));
4498 else if constexpr (N == 2 || N == 3) {
4501 result.value = __builtin_bit_cast(
typename I::native_type,
4502 fcvtzs(x.to_storage()).to_native());
4505#if NATIVE_HAS_ARM_NEON
4506 else if constexpr (N == 4)
4507 return I::from_native(vreinterpretq_u8_s32(vcvtq_s32_f32(x.value)));
4513 template<::native::isa<> Arch = NATIVE_BA
SELINE,
class T>
4514 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,float>
4516 return fcvtzs(simd<float,1,Arch>(x)).value;
4522 template<std::
size_t N, ::native::isa<> Arch>
4523 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
4525 fcvtzu(simd<float,N,Arch> x)
noexcept {
4526 using I = simd<std::uint32_t,N,Arch>;
4528 std::array<float,N> values{};
4529 std::array<std::uint32_t,N> result{};
4530 x.store(values.data());
4531 for (std::size_t i = 0; i < N; ++i)
4532 result[i] = detail::float_constant::fcvtzu(std::bit_cast<std::uint32_t>(values[i]));
4533 return I::load(result.data());
4535 if constexpr (N == 1)
return I(vcvts_u32_f32(x.value));
4536 else if constexpr (N == 2 || N == 3) {
4538 result.value = __builtin_bit_cast(
typename I::native_type,
4539 fcvtzu(x.to_storage()).to_native());
4542#if NATIVE_HAS_ARM_NEON
4543 else if constexpr (N == 4)
4544 return I::from_native(vreinterpretq_u8_u32(vcvtq_u32_f32(x.value)));
4550 template<::native::isa<> Arch = NATIVE_BA
SELINE,
class T>
4551 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,float>
4553 return fcvtzu(simd<float,1,Arch>(x)).value;
4561 template <
class To, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && std::same_as<To,std::int32_t>
4564 std::array<float,N> a{}; std::array<To,N> b{}; x.store(a.data());
4565 for(std::size_t i=0;i<N;++i) b[i]=static_cast<To>(a[i]);
4569 else if constexpr (N==1)
return simd<To,N,Arch>(
static_cast<std::int32_t
>(x.value));
4574#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4577#if NATIVE_HAS_ARM_NEON
4584 template <
class To, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && std::same_as<To,float>
4587 std::array<std::int32_t,N> a{}; std::array<To,N> b{}; x.store(a.data());
4588 for(std::size_t i=0;i<N;++i) b[i]=static_cast<To>(a[i]);
4592 else if constexpr (N==1)
return simd<To,N,Arch>(
static_cast<float>(x.value));
4594 else if constexpr (N==4)
return simd<To,N,Arch>(_mm_cvtepi32_ps(x.value));
4595 else if constexpr (N==8)
return simd<To,N,Arch>(_mm256_cvtepi32_ps(x.value));
4597#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4598 else if constexpr (N==16)
return simd<To,N,Arch>(_mm512_cvtepi32_ps(x.value));
4600#if NATIVE_HAS_ARM_NEON
4601 else if constexpr (N==4)
return simd<To,N,Arch>(vcvtq_f32_s32(vreinterpretq_s32_u8(x.value)));
4605 namespace detail::NATIVE_BACKEND {
4608 std::array<float,V::lanes> values{};
4609 for (std::size_t i=0;i<V::lanes;++i) values[i]=p[i];
4610 return V::from_native(__builtin_bit_cast(
typename V::native_type,values));
4612 if constexpr(V::lanes==1)
return V(*p);
4614 else if constexpr(V::lanes==4) {
if constexpr(Alignment>=16)
return V(_mm_load_ps(p));
else return V(_mm_loadu_ps(p)); }
4615 else if constexpr(V::lanes==8) {
if constexpr(Alignment>=32)
return V(_mm256_load_ps(p));
else return V(_mm256_loadu_ps(p)); }
4617#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4618 else if constexpr(V::lanes==16) {
if constexpr(Alignment>=64)
return V(_mm512_load_ps(p));
else return V(_mm512_loadu_ps(p)); }
4620#if NATIVE_HAS_ARM_NEON
4621 else if constexpr(V::lanes==4)
return V(vld1q_f32(p));
4626 auto values=__builtin_bit_cast(std::array<float,V::lanes>,value.to_native());
4627 for (std::size_t i=0;i<V::lanes;++i) p[i]=values[i];
4630 if constexpr(V::lanes==1)*p=value.value;
4632 else if constexpr(V::lanes==4) {
if constexpr(Alignment>=16)_mm_store_ps(p,value.value);
else _mm_storeu_ps(p,value.value);}
4633 else if constexpr(V::lanes==8) {
if constexpr(Alignment>=32)_mm256_store_ps(p,value.value);
else _mm256_storeu_ps(p,value.value);}
4635#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4636 else if constexpr(V::lanes==16) {
if constexpr(Alignment>=64)_mm512_store_ps(p,value.value);
else _mm512_storeu_ps(p,value.value);}
4638#if NATIVE_HAS_ARM_NEON
4639 else if constexpr(V::lanes==4)vst1q_f32(p,value.value);
4643 namespace detail::NATIVE_BACKEND {
4646 template <::native::isa<> Arch,
class T,std::
size_t N,
class U,std::
size_t A=1,simd_access Access=simd_access::ordinary>
4647 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4648 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4650 if constexpr (::native::simd_custom_element<T>)
return simd<T,N,Arch>::template load_memory<A>(p);
4651 else if constexpr (::native::simd_custom_element<U>)
return simd<U,N,Arch>::template load_memory<A>(p).to_native();
4652 else return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd<float,N,Arch>,A>(p);
4654 template <
class U,
class T,std::
size_t N,std::
size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
4655 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4656 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4658 if constexpr (::native::simd_custom_element<U>) simd<U,N,Arch>(value).template store_memory<A>(p);
4660 else if constexpr (::native::simd_custom_element<T>) value.to_native().template store_memory<A>(p);
4661 else ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd<float,N,Arch>,A>(p,value);
4663 template <::native::isa<> Arch,
class T,std::
size_t N,
class U,std::
size_t A=1,simd_access Access=simd_access::ordinary>
4664 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4665 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4667 T fill=T{},simd_memory<A,Access> = {})
noexcept native_diagnose_if(count > N,
"partial SIMD count exceeds the lane count") {
4668 std::array<U,N> temporary;temporary.fill(U(fill));
4669 for(std::size_t i=0;i<count;++i)temporary[i]=p[i];
4670 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T,N>(temporary.data());
4672 template <
class U,
class T,std::
size_t N,std::
size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
4673 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4674 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4676 std::array<U,N> temporary;
store_simd(temporary.data(),value);
4677 for(std::size_t i=0;i<count;++i)p[i]=temporary[i];
4679 template <::native::isa<> Arch,
class T,std::
size_t N>
requires NATIVE_ARCH_REQUIRES(Arch)
4681 template <::native::isa<> Arch,
class T,std::
size_t N>
requires NATIVE_ARCH_REQUIRES(Arch) && (N!=std::dynamic_extent)
4686namespace NATIVE_BACKEND_NAMESPACE::native {
4687 template <
class V,std::
size_t N>
using register_memory = ::NATIVE_BACKEND_NAMESPACE::register_memory<V,N>;
4688 using fp32x1 = ::native::simd<float,1,NATIVE_DEFAULT_ARCH>;
4689#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
4690 using fp32x4 = ::native::simd<float,4,NATIVE_DEFAULT_ARCH>;
4693 using fp32x8 = ::native::simd<float,8,NATIVE_DEFAULT_ARCH>;
4695#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4696 using fp32x16 = ::native::simd<float,16,NATIVE_DEFAULT_ARCH>;
4698 template<
class V>
concept float_register =
requires { V::lanes; V::architecture;
typename V::value_type; } &&
4699 std::same_as<typename V::value_type,float> && ::NATIVE_BACKEND_NAMESPACE::float_shape<V::lanes> && (::native::abi_lookup<V::architecture,::native::detail::raw_kernel_policies>::index == NATIVE_RAW_TARGET);
4705 std::array<float, V::lanes> a; v.storeu(a.data());
float sum = 0;
4706 for (
float x : a) sum += x;
4710 std::array<float, V::lanes> a; v.storeu(a.data());
4711 for (
float & x : a) x = std::acos(x);
4712 return V::loadu(a.data());
4715#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4716 using float_type = fp32x16;
4717 static constexpr std::size_t preferred_registers = 6;
4718#elif NATIVE_HAS_AVX2
4719 using float_type = fp32x8;
4720 static constexpr std::size_t preferred_registers = 6;
4721#elif NATIVE_HAS_ARM_NEON
4722 using float_type = fp32x4;
4724 static constexpr std::size_t preferred_registers = 15;
4726 using float_type = fp32x1;
4727 static constexpr std::size_t preferred_registers = 1;
4730 static constexpr std::size_t lanes = float_type::lanes;
4736#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
4737#if defined(__x86_64__) || defined(_M_X64)
4738#elif defined(__aarch64__) || defined(_M_ARM64)
4742 namespace detail::NATIVE_BACKEND {
4743 template<
class T>
concept short_element = std::same_as<T,float> ||
4744 std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t> || std::same_as<T,mask32>;
4745 template<
class T>
using short_lane = std::conditional_t<simd_mask_element<T>,std::uint32_t,T>;
4746 template<
class T>
using short_native = T __attribute__((ext_vector_type(4)));
4747 template<
class T>
constexpr short_lane<T> short_word(T value)
noexcept {
4748 if constexpr(simd_mask_element<T>)
return value.to_bits();
4760 template<::NATIVE_BACKEND_NAMESPACE::
short_element T,std::
size_t N,::native::isa<> Arch>
4764 struct alignas(typename simd<T,4,Arch>::native_type)
simd<T,N,Arch> : detail::swizzle_access<T,N,Arch> {
4767 using storage_type=simd<T,4,Arch>;
4768 using native_type=::NATIVE_BACKEND_NAMESPACE::short_native<::NATIVE_BACKEND_NAMESPACE::short_lane<T>>;
4769 using register_type=simd;
4770 using unsigned_register_tag=void;
4771 using bits_type=simd<std::uint32_t,N,Arch>;
4772 using vector_mask_type=simd<mask32,N,Arch>;
4773 using mask_type=std::conditional_t<simd_mask_element<T>,simd,
4774 std::conditional_t<bool(NATIVE_HAS_AVX512VL),predicate<N,Arch>,vector_mask_type>>;
4775 using mask=mask_type;
4776 using predicate_type=predicate<N,Arch>;
4777 template<
class U>
using rebind=simd<U,N,Arch>;
4778 static constexpr std::size_t lanes=N;
4779 static constexpr std::size_t storage_lanes=4;
4780 static constexpr bool compact=
false;
4781 static constexpr std::uint64_t lane_mask=(std::uint64_t(1)<<N)-1;
4791 native_inline constexpr simd(T x) noexcept : value{::NATIVE_BACKEND_NAMESPACE::short_word(x),::NATIVE_BACKEND_NAMESPACE::short_word(x),
4792 N==3?::NATIVE_BACKEND_NAMESPACE::short_word(x):NATIVE_BACKEND_NAMESPACE::short_lane<T>(0),0} {}
4797 : value(__builtin_shufflevector(x,native_type{},0,1,N==3?2:4,4)) {}
4799 template<
class... X>
requires(
sizeof...(X)==N) && (std::convertible_to<X,T> && ...)
4801 : value{::NATIVE_BACKEND_NAMESPACE::short_word(
static_cast<T
>(x))...} {}
4811 if constexpr(simd_mask_element<T>)
return from_storage(storage_type::from_native(__builtin_bit_cast(
typename storage_type::native_type,x)));
4812 else return simd(x);
4818 if constexpr(simd_mask_element<T>)
return storage_type::unsafe_from_native(__builtin_bit_cast(
typename storage_type::native_type,value));
4819 else return storage_type::from_native(__builtin_bit_cast(
typename storage_type::native_type,value));
4823 return simd(__builtin_bit_cast(native_type,x.to_native()));
4828 template<std::
size_t Alignment=1>
4831 native_type words{};
4832 for (std::size_t i=0;i<N;++i) words[i]=::NATIVE_BACKEND_NAMESPACE::short_word(p[i]);
4833 return simd(unchecked{},words);
4835#if defined(__x86_64__) || defined(_M_X64)
4836 if constexpr(bool(NATIVE_HAS_AVX2)) {
4837 if constexpr(N==2)
return simd(unchecked{},__builtin_bit_cast(native_type,_mm_loadl_epi64(
reinterpret_cast<__m128i
const *
>(p))));
4838 else if constexpr(bool(NATIVE_HAS_AVX512VL)) {
4839 if constexpr(std::same_as<T,float>)
return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskz_loadu_ps(7,p)));
4840 else return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskz_loadu_epi32(7,p)));
4842 auto active=_mm_set_epi32(0,-1,-1,-1);
4843 if constexpr(std::same_as<T,float>)
return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskload_ps(p,active)));
4844 else return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskload_epi32(
reinterpret_cast<int const *
>(p),active)));
4847#elif defined(__aarch64__) || defined(_M_ARM64)
4848 if constexpr((::native::arm_feature::neon <= Arch)) {
4849 if constexpr(std::same_as<T,float>) {
4850 auto x=vcombine_f32(vld1_f32(p),vdup_n_f32(0.f));
4851 if constexpr(N==3) x=vld1q_lane_f32(p+2,x,2);
4852 return simd(unchecked{},__builtin_bit_cast(native_type,x));
4854 auto q=
reinterpret_cast<std::uint32_t
const *
>(p);
4855 auto x=vcombine_u32(vld1_u32(q),vdup_n_u32(0));
4856 if constexpr(N==3) x=vld1q_lane_u32(q+2,x,2);
4857 return simd(unchecked{},__builtin_bit_cast(native_type,x));
4863 template<std::
size_t Alignment=1>
4866 for (std::size_t i=0;i<N;++i) {
4867 if constexpr(simd_mask_element<T>) p[i]=T::from_bits(value[i]);
4872#if defined(__x86_64__) || defined(_M_X64)
4873 if constexpr(bool(NATIVE_HAS_AVX2)) {
4874 if constexpr(N==2) _mm_storel_epi64(
reinterpret_cast<__m128i *
>(p),__builtin_bit_cast(__m128i,value));
4875 else if constexpr(bool(NATIVE_HAS_AVX512VL)) {
4876 if constexpr(std::same_as<T,float>) _mm_mask_storeu_ps(p,7,__builtin_bit_cast(__m128,value));
4877 else _mm_mask_storeu_epi32(p,7,__builtin_bit_cast(__m128i,value));
4879 auto active=_mm_set_epi32(0,-1,-1,-1);
4880 if constexpr(std::same_as<T,float>) _mm_maskstore_ps(p,active,__builtin_bit_cast(__m128,value));
4881 else _mm_maskstore_epi32(
reinterpret_cast<int *
>(p),active,__builtin_bit_cast(__m128i,value));
4884#elif defined(__aarch64__) || defined(_M_ARM64)
4885 if constexpr((::native::arm_feature::neon <= Arch)) {
4886 if constexpr(std::same_as<T,float>) {
4887 auto x=__builtin_bit_cast(float32x4_t,value);vst1_f32(p,vget_low_f32(x));
4888 if constexpr(N==3) vst1q_lane_f32(p+2,x,2);
4890 auto q=
reinterpret_cast<std::uint32_t *
>(p);
auto x=__builtin_bit_cast(uint32x4_t,value);
4891 vst1_u32(q,vget_low_u32(x));
4892 if constexpr(N==3) vst1q_lane_u32(q+2,x,2);
4907 std::array<T,N> values;values.fill(fill);
4908 if consteval {
for (std::size_t i=0;i<n;++i) values[i]=p[i]; }
4909 else {
if(n) std::memcpy(values.data(),p,n*
sizeof(T)); }
4910 return load(values.data());
4914 std::array<T,N> values;
store(values.data());
4915 if consteval {
for (std::size_t i=0;i<n;++i) p[i]=values[i]; }
4916 else {
if(n) std::memcpy(p,values.data(),n*
sizeof(T)); }
4926 return from_storage(storage_type::from_bits(x.to_storage()));
4965 auto padded=__builtin_shufflevector(b.value,native_type{1.f,1.f,1.f,1.f},0,1,N==3?2:4,4);
4966 auto divisor=storage_type::from_native(__builtin_bit_cast(
typename storage_type::native_type,padded));
4967 return clean(a.to_storage()/divisor);
4994 template<
class M>
requires(std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
4996 using FM=
typename storage_type::mask_type;
4997 return clean(
select(FM::from_bitset(mask.to_bitset()),a.to_storage(),b.to_storage()));
5008 template<std::
size_t K>
requires(K<32)
5011 template<std::
size_t K>
requires(K<32)
5014 template<
unsigned K>
requires(K<32) && simd_integer_element<T>
5017 template<
unsigned K>
requires(K<32) && simd_integer_element<T>
5029 template<simd_
integer_element U>
friend simd operator<<(
simd,U)
requires simd_integer_element<T> =
delete;
5031 template<simd_
integer_element U>
friend simd operator>>(
simd,U)
requires simd_integer_element<T> =
delete;
5033 template<simd_
integer_element U>
friend simd operator/(
simd,U)
requires simd_integer_element<T> =
delete;
5035 template<simd_
integer_element U>
friend simd operator/(U,
simd)
requires simd_integer_element<T> =
delete;
5037 template<simd_
integer_element U>
friend simd operator%(
simd,U)
requires simd_integer_element<T> =
delete;
5039 template<simd_
integer_element U>
friend simd operator%(U,
simd)
requires simd_integer_element<T> =
delete;
5055 struct unchecked {};
5056 native_inline constexpr simd(unchecked,native_type x) noexcept : value(x) {}
5057 native_nodiscard static native_inline constexpr simd clean(storage_type x)
noexcept {
return simd(unchecked{},__builtin_bit_cast(native_type,x.to_native())); }
5059 if constexpr(mask_type::compact)
return mask_type::from_native(x.to_native());
5060 else return mask_type::from_storage(x);
5065 template<simd_
integer_element T,std::
size_t N,::native::isa<> Arch>
5066 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (
sizeof(T)==4)
5067 native_nodiscard native_inline constexpr simd<T,N,Arch> bit_select(
simd<T,N,Arch> bits,
simd<T,N,Arch> a,
simd<T,N,Arch> b)
noexcept {
return (bits&a)|(~bits&b); }
5069 template<simd_
integer_element T,std::
size_t N,::native::isa<> Arch,
class M>
5070 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (
sizeof(T)==4) &&
5071 (std::same_as<M,typename simd<T,N,Arch>::mask> || std::same_as<M,simd<mask32,N,Arch>>)
5072 native_nodiscard native_inline constexpr simd<T,N,Arch> masked_add(M m,
simd<T,N,Arch> prior,
simd<T,N,Arch> a,
simd<T,N,Arch> b)
noexcept {
return select(m,a+b,prior); }
5074 template<simd_
integer_element T,std::
size_t N,::native::isa<> Arch,
class M>
5075 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (
sizeof(T)==4) &&
5076 (std::same_as<M,typename simd<T,N,Arch>::mask> || std::same_as<M,simd<mask32,N,Arch>>)
5077 native_nodiscard native_inline constexpr simd<T,N,Arch> masked_sub(M m,
simd<T,N,Arch> prior,
simd<T,N,Arch> a,
simd<T,N,Arch> b)
noexcept {
return select(m,a-b,prior); }
5079 template<simd_
integer_element T,std::
size_t N,::native::isa<> Arch,
class M>
5080 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (
sizeof(T)==4) &&
5081 (std::same_as<M,typename simd<T,N,Arch>::mask> || std::same_as<M,simd<mask32,N,Arch>>)
5082 native_nodiscard native_inline constexpr simd<T,N,Arch> masked_mul(M m,
simd<T,N,Arch> prior,
simd<T,N,Arch> a,
simd<T,N,Arch> b)
noexcept {
return select(m,a*b,prior); }
5089 namespace detail::NATIVE_BACKEND {
5090 enum class rounding_direction { down, up, zero };
5092 template<rounding_direction Direction,std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch)
5093 native_inline constexpr simd<float,N,Arch> round_integral(simd<float,N,Arch> x)
noexcept {
5094 using V = simd<float,N,Arch>;
5096 if constexpr(Direction==rounding_direction::down) return ::native::detail::float_constant::map(::native::detail::float_constant::floor,x);
5097 else if constexpr(Direction==rounding_direction::up) return ::native::detail::float_constant::map(::native::detail::float_constant::ceil,x);
5098 else return ::native::detail::float_constant::map(::native::detail::float_constant::trunc,x);
5100 if constexpr (N == 2 || N == 3) {
5101 return V::from_storage(round_integral<Direction>(x.to_storage()));
5104 constexpr int mode = (Direction == rounding_direction::down ? _MM_FROUND_TO_NEG_INF :
5105 Direction == rounding_direction::up ? _MM_FROUND_TO_POS_INF : _MM_FROUND_TO_ZERO) | _MM_FROUND_NO_EXC;
5106 if constexpr (N == 1)
5107 return V::from_native(_mm_cvtss_f32(_mm_round_ss(_mm_setzero_ps(),_mm_set_ss(x.to_native()),mode)));
5108 else if constexpr (N == 4)
return V::from_native(_mm_round_ps(x.to_native(),mode));
5109 else if constexpr (N == 8)
return V::from_native(_mm256_round_ps(x.to_native(),mode));
5110#if NATIVE_HAS_AVX512F
5111 else if constexpr (N == 16)
return V::from_native(_mm512_roundscale_ps(x.to_native(),mode));
5113#elif NATIVE_HAS_ARM_NEON
5114 if constexpr (N == 1) {
5115 auto a = vdup_n_f32(x.to_native());
5116 if constexpr (Direction == rounding_direction::down)
return V::from_native(vget_lane_f32(vrndm_f32(a),0));
5117 else if constexpr (Direction == rounding_direction::up)
return V::from_native(vget_lane_f32(vrndp_f32(a),0));
5118 else return V::from_native(vget_lane_f32(vrnd_f32(a),0));
5120 if constexpr (Direction == rounding_direction::down)
return V::from_native(__builtin_elementwise_floor(x.to_native()));
5121 else if constexpr (Direction == rounding_direction::up)
return V::from_native(vrndpq_f32(x.to_native()));
5122 else return V::from_native(vrndq_f32(x.to_native()));
5125 if constexpr (Direction == rounding_direction::down)
return V::from_native(std::floor(x.to_native()));
5126 else if constexpr (Direction == rounding_direction::up)
return V::from_native(std::ceil(x.to_native()));
5127 else return V::from_native(std::trunc(x.to_native()));
5141 template<std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5143 return detail::NATIVE_BACKEND::round_integral<detail::NATIVE_BACKEND::rounding_direction::down>(x);
5147 template<std::
size_t N,std::
size_t M, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5149 auto const & [...x] = input;
5150 return {{
floor(x)...}};
5157 template<std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5159 return detail::NATIVE_BACKEND::round_integral<detail::NATIVE_BACKEND::rounding_direction::up>(x);
5163 template<std::
size_t N,std::
size_t M, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5165 auto const & [...x] = input;
5166 return {{
ceil(x)...}};
5173 template<std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5175 return detail::NATIVE_BACKEND::round_integral<detail::NATIVE_BACKEND::rounding_direction::zero>(x);
5179 template<std::
size_t N,std::
size_t M, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5181 auto const & [...x] = input;
5182 return {{
trunc(x)...}};
5190namespace native::detail::NATIVE_BACKEND {
5191 template<std::
size_t N,
class M>
5192 native_inline std::uint32_t compaction_mask_bits(M mask)
noexcept {
5194 if constexpr (N > 1 && !M::compact) {
5195 if constexpr (
sizeof(M) == 32)
5196 return std::uint32_t(_mm256_movemask_ps(__builtin_bit_cast(__m256,
mask.to_native())));
5197 else return std::uint32_t(_mm_movemask_ps(__builtin_bit_cast(__m128,
mask.to_native()))) & ((1u << N) - 1);
5199#elif NATIVE_HAS_ARM_NEON
5200 if constexpr (N > 1) {
5201 constexpr std::array<std::uint32_t,4> weights{1,2,4,8};
5202 return vaddvq_u32(vandq_u32(__builtin_bit_cast(uint32x4_t,
mask.to_native()),vld1q_u32(weights.data()))) & ((1u << N) - 1);
5205 return std::uint32_t(
mask.to_bitset()) & ((1u << N) - 1);
5210 template<
bool Expand, std::
size_t W
idth>
inline constexpr auto compaction_indices = [] {
5211 constexpr auto bytes = Width == 4 ? 16 : 8;
5212 std::array<std::array<std::uint8_t,bytes>,std::size_t(1) << Width> table{};
5213 for (std::size_t mask = 0;
mask != table.size(); ++
mask) {
5214 std::size_t packed = 0;
5215 for (std::size_t lane = 0; lane != Width; ++lane)
if (mask & (std::size_t(1) << lane)) {
5216 auto destination = Expand ? lane : packed;
5217 auto source = Expand ? packed : lane;
5218 if constexpr (Width == 4)
5219 for (std::size_t
byte = 0;
byte != 4; ++byte)
5220 table[mask][4 * destination +
byte] = std::uint8_t(4 * source +
byte);
5221 else table[
mask][destination] = std::uint8_t(source);
5228 template<
bool Expand,
class V>
5229 native_inline V compact_register(std::uint32_t mask, V input, V prior)
noexcept {
5230 if constexpr (V::lanes == 1)
return mask ? input : prior;
5232#if NATIVE_HAS_AVX512F
5233 if constexpr (
sizeof(V) == 64) {
5234 auto x = __builtin_bit_cast(__m512i,input), merge = __builtin_bit_cast(__m512i,prior);
5235 if constexpr (Expand)
return __builtin_bit_cast(V,_mm512_mask_expand_epi32(merge, __mmask16(mask), x));
5236 else return __builtin_bit_cast(V,_mm512_mask_compress_epi32(merge, __mmask16(mask), x));
5238#if NATIVE_HAS_AVX512VL
5239 else if constexpr (
sizeof(V) == 32) {
5240 auto x = __builtin_bit_cast(__m256i,input), merge = __builtin_bit_cast(__m256i,prior);
5241 if constexpr (Expand)
return __builtin_bit_cast(V,_mm256_mask_expand_epi32(merge, __mmask8(mask), x));
5242 else return __builtin_bit_cast(V,_mm256_mask_compress_epi32(merge, __mmask8(mask), x));
5244 auto x = __builtin_bit_cast(__m128i,input), merge = __builtin_bit_cast(__m128i,prior);
5245 if constexpr (Expand)
return __builtin_bit_cast(V,_mm_mask_expand_epi32(merge, __mmask8(mask), x));
5246 else return __builtin_bit_cast(V,_mm_mask_compress_epi32(merge, __mmask8(mask), x));
5252#if (!NATIVE_HAS_AVX512F || !NATIVE_HAS_AVX512VL) && (NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON)
5256 if constexpr (
sizeof(V) == 32) {
5257 auto const & row = compaction_indices<Expand,8>[
mask];
5258 auto indices = _mm256_cvtepu8_epi32(_mm_loadl_epi64(
reinterpret_cast<__m128i
const *
>(row.data())));
5259 permuted = __builtin_bit_cast(V,_mm256_permutevar8x32_epi32(__builtin_bit_cast(__m256i,input), indices));
5261 auto const & row = compaction_indices<Expand,4>[
mask];
5262 auto indices = _mm_loadu_si128(
reinterpret_cast<__m128i
const *
>(row.data()));
5263 permuted = __builtin_bit_cast(V,_mm_shuffle_epi8(__builtin_bit_cast(__m128i,input), indices));
5266 auto const & row = compaction_indices<Expand,4>[
mask];
5267 permuted = __builtin_bit_cast(V,vqtbl1q_u8(__builtin_bit_cast(uint8x16_t,input), vld1q_u8(row.data())));
5272 if constexpr (
sizeof(V) == 32) {
5274 if constexpr (Expand) {
5275 auto bits = _mm256_setr_epi32(1,2,4,8,16,32,64,128);
5276 live = _mm256_cmpeq_epi32(_mm256_and_si256(_mm256_set1_epi32(
int(mask)),bits),bits);
5277 }
else live = _mm256_cmpgt_epi32(_mm256_set1_epi32(std::popcount(mask)),_mm256_setr_epi32(0,1,2,3,4,5,6,7));
5278 return __builtin_bit_cast(V,_mm256_blendv_epi8(__builtin_bit_cast(__m256i,prior),__builtin_bit_cast(__m256i,permuted),live));
5281 if constexpr (Expand) {
5282 auto bits = _mm_setr_epi32(1,2,4,8);
5283 live = _mm_cmpeq_epi32(_mm_and_si128(_mm_set1_epi32(
int(mask)),bits),bits);
5284 }
else live = _mm_cmpgt_epi32(_mm_set1_epi32(std::popcount(mask)),_mm_setr_epi32(0,1,2,3));
5285 return __builtin_bit_cast(V,_mm_blendv_epi8(__builtin_bit_cast(__m128i,prior),__builtin_bit_cast(__m128i,permuted),live));
5289 if constexpr (Expand) {
5290 constexpr std::array<std::uint32_t,4> weights{1,2,4,8};
5291 live = vtstq_u32(vdupq_n_u32(mask),vld1q_u32(weights.data()));
5293 constexpr std::array<std::uint32_t,4> lanes{0,1,2,3};
5294 live = vcltq_u32(vld1q_u32(lanes.data()),vdupq_n_u32(std::uint32_t(std::popcount(mask))));
5296 return __builtin_bit_cast(V,vbslq_u8(vreinterpretq_u8_u32(live),__builtin_bit_cast(uint8x16_t,permuted),__builtin_bit_cast(uint8x16_t,prior)));
5310 template<
class T, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) &&
5311 (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
5312 ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5316 std::array<T,N> input{},output{}; value.store(input.data()); output.fill(fill);
5317 auto bits=
mask.to_bitset(); std::size_t count=0;
5318 for(std::size_t i=0;i<N;++i)
if((bits>>i)&1) output[count++]=input[i];
5322 auto bits = detail::NATIVE_BACKEND::compaction_mask_bits<N>(
mask);
5325 V prior = __builtin_bit_cast(V,U(std::bit_cast<std::uint32_t>(fill)));
5326 return {detail::NATIVE_BACKEND::compact_register<false>(bits, value, prior), std::size_t(std::popcount(bits))};
5332 template<
class T, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) &&
5333 (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
5334 ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5339 std::array<T,N> input{},output{}; packed.store(input.data()); prior.store(output.data());
5340 auto bits=
mask.to_bitset(); std::size_t count=0;
5341 for(std::size_t i=0;i<N;++i)
if((bits>>i)&1) output[i]=input[count++];
5344 auto bits = detail::NATIVE_BACKEND::compaction_mask_bits<N>(
mask);
5345 auto result = detail::NATIVE_BACKEND::compact_register<true>(bits, packed, prior);
5354 template<
class T, std::
size_t N, ::native::isa<> Arch>
requires NATIVE_ARCH_REQUIRES(Arch) &&
5355 (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
5356 ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5360 std::array<T,N> input{}; value.store(input.data());
5361 auto bits=
mask.to_bitset(); std::size_t written=0;
5362 for(std::size_t i=0;i<N && written<capacity;++i)
5363 if((bits>>i)&1) destination[written++]=input[i];
5366 auto bits = detail::NATIVE_BACKEND::compaction_mask_bits<N>(
mask);
5367 auto selected = std::size_t(std::popcount(bits));
5368 auto written = capacity < selected ? capacity : selected;
5369 if (!written)
return 0;
5370#if NATIVE_HAS_AVX512F
5371 if constexpr (N > 1 && (
sizeof(value) == 64 || NATIVE_HAS_AVX512VL)) {
5374 auto prefix = (std::uint32_t(1) << written) - 1;
5375 if constexpr (
sizeof(value) == 64) {
5376 auto packed = _mm512_maskz_compress_epi32(__mmask16(bits),__builtin_bit_cast(__m512i,value));
5377 _mm512_mask_storeu_epi32(destination,__mmask16(prefix),packed);
5378 }
else if constexpr (
sizeof(value) == 32) {
5379 auto packed = _mm256_maskz_compress_epi32(__mmask8(bits),__builtin_bit_cast(__m256i,value));
5380 _mm256_mask_storeu_epi32(destination,__mmask8(prefix),packed);
5382 auto packed = _mm_maskz_compress_epi32(__mmask8(bits),__builtin_bit_cast(__m128i,value));
5383 _mm_mask_storeu_epi32(destination,__mmask8(prefix),packed);
5388 std::array<T,N> packed;
5390 std::memcpy(destination, packed.data(), written *
sizeof(T));
5395#if NATIVE_HAS_WASM_SIMD128 || defined(NATIVE_DOXYGEN)
5401 concept wasm_number =
5402 simd_integer_element<T> || std::same_as<T, float> || std::same_as<T, double>;
5404 using wasm_word = std::conditional_t<
5405 sizeof(T) == 1, std::uint8_t,
5406 std::conditional_t<
sizeof(T) == 2, std::uint16_t,
5407 std::conditional_t<
sizeof(T) == 4, std::uint32_t, std::uint64_t>>>;
5410 std::conditional_t<
sizeof(T) == 4, constexpr_float::binary32, constexpr_float::binary64>;
5413 constexpr auto wasm_lanes(V v)
noexcept {
5414 std::array<typename V::value_type, V::lanes> a{};
5419 template<
class V,
class F,
class... W>
5420 constexpr V wasm_map(F f, V v, W... w)
noexcept {
5421 auto inputs = std::tuple{wasm_lanes(v), wasm_lanes(w)...};
5422 std::array<typename V::value_type, V::lanes> r{};
5423 for (std::size_t i = 0; i < V::lanes; ++i) {
5424 r[i] = std::apply([&](
auto const &... a) {
return f(a[i]...); }, inputs);
5426 return V::load(r.data());
5429 template<
class T, std::
size_t N, isa<> A>
5430 requires ordinary_simd_element<T> && (A.has(wasm_feature::simd128)) &&
5431 (wasm_number<T> || simd_mask_element<T>) && (
sizeof(T) * N == 16)
5432 struct value_traits<simd<T, N, A>> {
5433 static constexpr isa<> value = A;
5434 static constexpr bool known =
true;
5435 static constexpr bool aggregate_default =
false;
5440 template<
class U, std::
size_t N, isa<> A>
5441 requires(A.has(wasm_feature::simd128)) && (
sizeof(U) * N == 16)
5444 using native_type = v128_t;
5445 using mask_type =
simd;
5449 static constexpr auto architecture = A;
5450 static constexpr std::size_t lanes = N;
5451 static constexpr bool compact =
false;
5454 native_type value_{};
5458 constexpr simd() noexcept = default;
5479 auto words = __builtin_bit_cast(std::array<U, N>, x);
5480 for (
auto & w : words) {
5481 w = w ? U(~U(0)) : U(0);
5485 if constexpr (
sizeof(U) == 1) {
5487 }
else if constexpr (
sizeof(U) == 2) {
5489 }
else if constexpr (
sizeof(U) == 4) {
5491 }
else if constexpr (
sizeof(U) == 8) {
5499 std::array<U, N> a{};
5500 for (std::size_t i = 0; i < N; ++i) {
5501 a[i] = ((
bits >> i) & 1) ? U(~U(0)) : U(0);
5509 auto a = __builtin_bit_cast(std::array<U, N>, value_);
5510 std::uint64_t r = 0;
5511 for (std::size_t i = 0; i < N; ++i) {
5512 r |= std::uint64_t(a[i] != 0) << i;
5516 if constexpr (
sizeof(U) == 1) {
5517 return wasm_i8x16_bitmask(value_);
5518 }
else if constexpr (
sizeof(U) == 2) {
5519 return wasm_i16x8_bitmask(value_);
5520 }
else if constexpr (
sizeof(U) == 4) {
5521 return wasm_i32x4_bitmask(value_);
5522 }
else if constexpr (
sizeof(U) == 8) {
5523 return wasm_i64x2_bitmask(value_);
5535 std::array<U, N> a{};
5536 for (std::size_t i = 0; i < N; ++i) {
5537 a[i] = p[i].to_bits();
5544 auto a = __builtin_bit_cast(std::array<U, N>, value_);
5545 for (std::size_t i = 0; i < N; ++i) {
5551 template<std::
size_t Align>
5557 template<std::
size_t Align>
5565 return v.to_bitset() != 0;
5567 return wasm_v128_any_true(v.value_);
5574 return v.to_bitset() == ((std::uint64_t{1} << N) - 1);
5576 if constexpr (
sizeof(U) == 1) {
5577 return wasm_i8x16_all_true(v.value_);
5578 }
else if constexpr (
sizeof(U) == 2) {
5579 return wasm_i16x8_all_true(v.value_);
5580 }
else if constexpr (
sizeof(U) == 4) {
5581 return wasm_i32x4_all_true(v.value_);
5583 return wasm_i64x2_all_true(v.value_);
5596 return from_bitset(a.to_bitset() & b.to_bitset());
5604 return *
this = *
this & b;
5610 return from_bitset(a.to_bitset() | b.to_bitset());
5618 return *
this = *
this | b;
5624 return from_bitset(a.to_bitset() ^ b.to_bitset());
5632 return *
this = *
this ^ b;
5661 return (m & a) | (~m & b);
5666 template<detail::wasm_number T, std::
size_t N, isa<> A>
5667 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
5668 struct alignas(16) simd<T, N, A> {
5669 using value_type = T;
5670 using native_type = v128_t;
5671 using word_type = detail::wasm_word<T>;
5672 using mask_type = simd<mask_lane<word_type>, N, A>;
5673 using mask = mask_type;
5674 using bits_type = simd<word_type, N, A>;
5675 using register_type = simd;
5677 using rebind = simd<U, N, A>;
5678 static constexpr auto architecture = A;
5679 static constexpr std::size_t lanes = N;
5682 native_type value_{};
5686 constexpr simd() noexcept = default;
5690 requires
std::is_floating_point_v<T>
5693 std::array<T, N> a{};
5695 value_ = __builtin_bit_cast(native_type, a);
5697 if constexpr (std::same_as<T, float>) {
5698 value_ = wasm_f32x4_splat(x);
5700 value_ = wasm_f64x2_splat(x);
5706 template<simd_
integer_element U>
5707 requires simd_integer_element<T>
5709 auto x =
static_cast<T
>(input);
5711 std::array<T, N> a{};
5713 value_ = __builtin_bit_cast(native_type, a);
5715 if constexpr (std::same_as<T, std::int8_t>) {
5716 value_ = wasm_i8x16_splat(x);
5717 }
else if constexpr (std::same_as<T, std::uint8_t>) {
5718 value_ = wasm_u8x16_splat(x);
5719 }
else if constexpr (std::same_as<T, std::int16_t>) {
5720 value_ = wasm_i16x8_splat(x);
5721 }
else if constexpr (std::same_as<T, std::uint16_t>) {
5722 value_ = wasm_u16x8_splat(x);
5723 }
else if constexpr (std::same_as<T, std::int32_t>) {
5724 value_ = wasm_i32x4_splat(x);
5725 }
else if constexpr (std::same_as<T, std::uint32_t>) {
5726 value_ = wasm_u32x4_splat(x);
5727 }
else if constexpr (std::same_as<T, std::int64_t>) {
5728 value_ = wasm_i64x2_splat(x);
5729 }
else if constexpr (std::same_as<T, std::uint64_t>) {
5730 value_ = wasm_u64x2_splat(x);
5740 template<
class... U>
5741 requires(
sizeof...(U) == N && ((std::same_as<U, T> && ...) ||
5742 (simd_integer_element<T> && (simd_integer_element<U> && ...))))
5765 std::array<T, N> a{};
5766 for (std::size_t i = 0; i < N; ++i) {
5769 return from_native(__builtin_bit_cast(native_type, a));
5778 auto a = __builtin_bit_cast(std::array<T, N>, value_);
5779 for (std::size_t i = 0; i < N; ++i) {
5783 wasm_v128_store(p, value_);
5788 template<std::
size_t Align>
5793 return load(
static_cast<T
const *
>(__builtin_assume_aligned(p, Align)));
5798 template<std::
size_t Align>
5803 store(
static_cast<T *
>(__builtin_assume_aligned(p, Align)));
5809 T fill = {})
noexcept {
5810 std::array<T, N> a{};
5812 for (std::size_t i = 0; i < n; ++i) {
5815 return load(a.data());
5820 auto a = detail::wasm_lanes(*
this);
5821 for (std::size_t i = 0; i < n; ++i) {
5838 return bits_type::from_native(value_);
5847 template<std::
size_t I>
5851 return __builtin_bit_cast(std::array<T, N>, value_)[I];
5853 if constexpr (std::same_as<T, float>) {
5854 return wasm_f32x4_extract_lane(value_, I);
5855 }
else if constexpr (std::same_as<T, double>) {
5856 return wasm_f64x2_extract_lane(value_, I);
5857 }
else if constexpr (std::same_as<T, std::int8_t>) {
5858 return wasm_i8x16_extract_lane(value_, I);
5859 }
else if constexpr (std::same_as<T, std::uint8_t>) {
5860 return wasm_u8x16_extract_lane(value_, I);
5861 }
else if constexpr (std::same_as<T, std::int16_t>) {
5862 return wasm_i16x8_extract_lane(value_, I);
5863 }
else if constexpr (std::same_as<T, std::uint16_t>) {
5864 return wasm_u16x8_extract_lane(value_, I);
5865 }
else if constexpr (std::same_as<T, std::int32_t>) {
5866 return wasm_i32x4_extract_lane(value_, I);
5867 }
else if constexpr (std::same_as<T, std::uint32_t>) {
5868 return wasm_u32x4_extract_lane(value_, I);
5869 }
else if constexpr (std::same_as<T, std::int64_t>) {
5870 return wasm_i64x2_extract_lane(value_, I);
5871 }
else if constexpr (std::same_as<T, std::uint64_t>) {
5872 return wasm_u64x2_extract_lane(value_, I);
5878 template<std::
size_t I>
5882 auto a = __builtin_bit_cast(std::array<T, N>, value_);
5884 return load(a.data());
5886 if constexpr (std::same_as<T, float>) {
5887 return from_native(wasm_f32x4_replace_lane(value_, I, x));
5888 }
else if constexpr (std::same_as<T, double>) {
5889 return from_native(wasm_f64x2_replace_lane(value_, I, x));
5890 }
else if constexpr (std::same_as<T, std::int8_t>) {
5891 return from_native(wasm_i8x16_replace_lane(value_, I, x));
5892 }
else if constexpr (std::same_as<T, std::uint8_t>) {
5893 return from_native(wasm_u8x16_replace_lane(value_, I, x));
5894 }
else if constexpr (std::same_as<T, std::int16_t>) {
5895 return from_native(wasm_i16x8_replace_lane(value_, I, x));
5896 }
else if constexpr (std::same_as<T, std::uint16_t>) {
5897 return from_native(wasm_u16x8_replace_lane(value_, I, x));
5898 }
else if constexpr (std::same_as<T, std::int32_t>) {
5899 return from_native(wasm_i32x4_replace_lane(value_, I, x));
5900 }
else if constexpr (std::same_as<T, std::uint32_t>) {
5901 return from_native(wasm_u32x4_replace_lane(value_, I, x));
5902 }
else if constexpr (std::same_as<T, std::int64_t>) {
5903 return from_native(wasm_i64x2_replace_lane(value_, I, x));
5904 }
else if constexpr (std::same_as<T, std::uint64_t>) {
5905 return from_native(wasm_u64x2_replace_lane(value_, I, x));
5913 return detail::wasm_map(
5915 if constexpr (std::is_floating_point_v<T>) {
5916 using format_type = detail::wasm_format<T>;
5917 return std::bit_cast<T>(detail::constexpr_float::add_bits<format_type>(
5918 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
5920 return std::bit_cast<T>(
5921 word_type(std::uint64_t(word_type(x)) + std::uint64_t(word_type(y))));
5926 if constexpr (std::same_as<T, float>) {
5927 return from_native(wasm_f32x4_add(a.value_, b.value_));
5928 }
else if constexpr (std::same_as<T, double>) {
5929 return from_native(wasm_f64x2_add(a.value_, b.value_));
5930 }
else if constexpr (std::same_as<T, std::int8_t>) {
5931 return from_native(wasm_i8x16_add(a.value_, b.value_));
5932 }
else if constexpr (std::same_as<T, std::uint8_t>) {
5933 return from_native(wasm_i8x16_add(a.value_, b.value_));
5934 }
else if constexpr (std::same_as<T, std::int16_t>) {
5935 return from_native(wasm_i16x8_add(a.value_, b.value_));
5936 }
else if constexpr (std::same_as<T, std::uint16_t>) {
5937 return from_native(wasm_i16x8_add(a.value_, b.value_));
5938 }
else if constexpr (std::same_as<T, std::int32_t>) {
5939 return from_native(wasm_i32x4_add(a.value_, b.value_));
5940 }
else if constexpr (std::same_as<T, std::uint32_t>) {
5941 return from_native(wasm_i32x4_add(a.value_, b.value_));
5942 }
else if constexpr (std::same_as<T, std::int64_t>) {
5943 return from_native(wasm_i64x2_add(a.value_, b.value_));
5944 }
else if constexpr (std::same_as<T, std::uint64_t>) {
5945 return from_native(wasm_i64x2_add(a.value_, b.value_));
5952 return *
this = *
this + b;
5958 return detail::wasm_map(
5960 if constexpr (std::is_floating_point_v<T>) {
5961 using format_type = detail::wasm_format<T>;
5962 return std::bit_cast<T>(detail::constexpr_float::sub_bits<format_type>(
5963 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
5965 return std::bit_cast<T>(
5966 word_type(std::uint64_t(word_type(x)) - std::uint64_t(word_type(y))));
5971 if constexpr (std::same_as<T, float>) {
5972 return from_native(wasm_f32x4_sub(a.value_, b.value_));
5973 }
else if constexpr (std::same_as<T, double>) {
5974 return from_native(wasm_f64x2_sub(a.value_, b.value_));
5975 }
else if constexpr (std::same_as<T, std::int8_t>) {
5976 return from_native(wasm_i8x16_sub(a.value_, b.value_));
5977 }
else if constexpr (std::same_as<T, std::uint8_t>) {
5978 return from_native(wasm_i8x16_sub(a.value_, b.value_));
5979 }
else if constexpr (std::same_as<T, std::int16_t>) {
5980 return from_native(wasm_i16x8_sub(a.value_, b.value_));
5981 }
else if constexpr (std::same_as<T, std::uint16_t>) {
5982 return from_native(wasm_i16x8_sub(a.value_, b.value_));
5983 }
else if constexpr (std::same_as<T, std::int32_t>) {
5984 return from_native(wasm_i32x4_sub(a.value_, b.value_));
5985 }
else if constexpr (std::same_as<T, std::uint32_t>) {
5986 return from_native(wasm_i32x4_sub(a.value_, b.value_));
5987 }
else if constexpr (std::same_as<T, std::int64_t>) {
5988 return from_native(wasm_i64x2_sub(a.value_, b.value_));
5989 }
else if constexpr (std::same_as<T, std::uint64_t>) {
5990 return from_native(wasm_i64x2_sub(a.value_, b.value_));
5997 return *
this = *
this - b;
6002 requires(
sizeof(T) > 1)
6005 return detail::wasm_map(
6007 if constexpr (std::is_floating_point_v<T>) {
6008 using format_type = detail::wasm_format<T>;
6009 return std::bit_cast<T>(detail::constexpr_float::mul_bits<format_type>(
6010 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
6012 return std::bit_cast<T>(
6013 word_type(std::uint64_t(word_type(x)) * std::uint64_t(word_type(y))));
6018 if constexpr (std::same_as<T, float>) {
6019 return from_native(wasm_f32x4_mul(a.value_, b.value_));
6020 }
else if constexpr (std::same_as<T, double>) {
6021 return from_native(wasm_f64x2_mul(a.value_, b.value_));
6022 }
else if constexpr (std::same_as<T, std::int16_t>) {
6023 return from_native(wasm_i16x8_mul(a.value_, b.value_));
6024 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6025 return from_native(wasm_i16x8_mul(a.value_, b.value_));
6026 }
else if constexpr (std::same_as<T, std::int32_t>) {
6027 return from_native(wasm_i32x4_mul(a.value_, b.value_));
6028 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6029 return from_native(wasm_i32x4_mul(a.value_, b.value_));
6030 }
else if constexpr (std::same_as<T, std::int64_t>) {
6031 return from_native(wasm_i64x2_mul(a.value_, b.value_));
6032 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6033 return from_native(wasm_i64x2_mul(a.value_, b.value_));
6040 requires(
sizeof(T) > 1)
6042 return *
this = *
this * b;
6047 requires std::is_floating_point_v<T>
6050 return detail::wasm_map(
6052 using format_type = detail::wasm_format<T>;
6053 return std::bit_cast<T>(detail::constexpr_float::div_bits<format_type>(
6054 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
6058 if constexpr (std::same_as<T, float>) {
6059 return from_native(wasm_f32x4_div(a.value_, b.value_));
6060 }
else if constexpr (std::same_as<T, double>) {
6061 return from_native(wasm_f64x2_div(a.value_, b.value_));
6068 requires std::is_floating_point_v<T>
6070 return *
this = *
this / b;
6076 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6077 auto y = __builtin_bit_cast(std::array<word_type, N>, b.value_);
6078 for (std::size_t i = 0; i < N; ++i) {
6081 return from_native(__builtin_bit_cast(native_type, x));
6083 return from_native(wasm_v128_and(a.value_, b.value_));
6089 return *
this = *
this & b;
6095 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6096 auto y = __builtin_bit_cast(std::array<word_type, N>, b.value_);
6097 for (std::size_t i = 0; i < N; ++i) {
6100 return from_native(__builtin_bit_cast(native_type, x));
6102 return from_native(wasm_v128_or(a.value_, b.value_));
6108 return *
this = *
this | b;
6114 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6115 auto y = __builtin_bit_cast(std::array<word_type, N>, b.value_);
6116 for (std::size_t i = 0; i < N; ++i) {
6119 return from_native(__builtin_bit_cast(native_type, x));
6121 return from_native(wasm_v128_xor(a.value_, b.value_));
6127 return *
this = *
this ^ b;
6133 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6134 for (
auto & w : x) {
6137 return from_native(__builtin_bit_cast(native_type, x));
6146 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6147 for (
auto & w : x) {
6148 if constexpr (std::is_floating_point_v<T>) {
6149 w ^= word_type{1} << (
sizeof(T) * 8 - 1);
6151 w = word_type(0 - w);
6154 return from_native(__builtin_bit_cast(native_type, x));
6156 if constexpr (std::same_as<T, float>) {
6158 }
else if constexpr (std::same_as<T, double>) {
6160 }
else if constexpr (std::same_as<T, std::int8_t>) {
6162 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6164 }
else if constexpr (std::same_as<T, std::int16_t>) {
6166 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6168 }
else if constexpr (std::same_as<T, std::int32_t>) {
6170 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6172 }
else if constexpr (std::same_as<T, std::int64_t>) {
6174 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6183 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6184 std::uint64_t
bits = 0;
6185 for (std::size_t i = 0; i < N; ++i) {
6186 bits |= std::uint64_t(x[i] == y[i]) << i;
6190 if constexpr (std::same_as<T, float>) {
6191 return mask_type::unsafe_from_native(wasm_f32x4_eq(a.value_, b.value_));
6192 }
else if constexpr (std::same_as<T, double>) {
6193 return mask_type::unsafe_from_native(wasm_f64x2_eq(a.value_, b.value_));
6194 }
else if constexpr (std::same_as<T, std::int8_t>) {
6195 return mask_type::unsafe_from_native(wasm_i8x16_eq(a.value_, b.value_));
6196 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6197 return mask_type::unsafe_from_native(wasm_i8x16_eq(a.value_, b.value_));
6198 }
else if constexpr (std::same_as<T, std::int16_t>) {
6199 return mask_type::unsafe_from_native(wasm_i16x8_eq(a.value_, b.value_));
6200 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6201 return mask_type::unsafe_from_native(wasm_i16x8_eq(a.value_, b.value_));
6202 }
else if constexpr (std::same_as<T, std::int32_t>) {
6203 return mask_type::unsafe_from_native(wasm_i32x4_eq(a.value_, b.value_));
6204 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6205 return mask_type::unsafe_from_native(wasm_i32x4_eq(a.value_, b.value_));
6206 }
else if constexpr (std::same_as<T, std::int64_t>) {
6207 return mask_type::unsafe_from_native(wasm_i64x2_eq(a.value_, b.value_));
6208 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6209 return mask_type::unsafe_from_native(wasm_i64x2_eq(a.value_, b.value_));
6217 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6218 std::uint64_t
bits = 0;
6219 for (std::size_t i = 0; i < N; ++i) {
6220 bits |= std::uint64_t(x[i] != y[i]) << i;
6224 if constexpr (std::same_as<T, float>) {
6225 return mask_type::unsafe_from_native(wasm_f32x4_ne(a.value_, b.value_));
6226 }
else if constexpr (std::same_as<T, double>) {
6227 return mask_type::unsafe_from_native(wasm_f64x2_ne(a.value_, b.value_));
6228 }
else if constexpr (std::same_as<T, std::int8_t>) {
6229 return mask_type::unsafe_from_native(wasm_i8x16_ne(a.value_, b.value_));
6230 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6231 return mask_type::unsafe_from_native(wasm_i8x16_ne(a.value_, b.value_));
6232 }
else if constexpr (std::same_as<T, std::int16_t>) {
6233 return mask_type::unsafe_from_native(wasm_i16x8_ne(a.value_, b.value_));
6234 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6235 return mask_type::unsafe_from_native(wasm_i16x8_ne(a.value_, b.value_));
6236 }
else if constexpr (std::same_as<T, std::int32_t>) {
6237 return mask_type::unsafe_from_native(wasm_i32x4_ne(a.value_, b.value_));
6238 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6239 return mask_type::unsafe_from_native(wasm_i32x4_ne(a.value_, b.value_));
6240 }
else if constexpr (std::same_as<T, std::int64_t>) {
6241 return mask_type::unsafe_from_native(wasm_i64x2_ne(a.value_, b.value_));
6242 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6243 return mask_type::unsafe_from_native(wasm_i64x2_ne(a.value_, b.value_));
6251 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6252 std::uint64_t
bits = 0;
6253 for (std::size_t i = 0; i < N; ++i) {
6254 bits |= std::uint64_t(x[i] < y[i]) << i;
6258 if constexpr (std::same_as<T, float>) {
6259 return mask_type::unsafe_from_native(wasm_f32x4_lt(a.value_, b.value_));
6260 }
else if constexpr (std::same_as<T, double>) {
6261 return mask_type::unsafe_from_native(wasm_f64x2_lt(a.value_, b.value_));
6262 }
else if constexpr (std::same_as<T, std::int8_t>) {
6263 return mask_type::unsafe_from_native(wasm_i8x16_lt(a.value_, b.value_));
6264 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6265 return mask_type::unsafe_from_native(wasm_u8x16_lt(a.value_, b.value_));
6266 }
else if constexpr (std::same_as<T, std::int16_t>) {
6267 return mask_type::unsafe_from_native(wasm_i16x8_lt(a.value_, b.value_));
6268 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6269 return mask_type::unsafe_from_native(wasm_u16x8_lt(a.value_, b.value_));
6270 }
else if constexpr (std::same_as<T, std::int32_t>) {
6271 return mask_type::unsafe_from_native(wasm_i32x4_lt(a.value_, b.value_));
6272 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6273 return mask_type::unsafe_from_native(wasm_u32x4_lt(a.value_, b.value_));
6274 }
else if constexpr (std::same_as<T, std::int64_t>) {
6275 return mask_type::unsafe_from_native(wasm_i64x2_lt(a.value_, b.value_));
6276 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6277 return mask_type::unsafe_from_native(
6278 wasm_i64x2_lt(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6279 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6287 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6288 std::uint64_t
bits = 0;
6289 for (std::size_t i = 0; i < N; ++i) {
6290 bits |= std::uint64_t(x[i] <= y[i]) << i;
6294 if constexpr (std::same_as<T, float>) {
6295 return mask_type::unsafe_from_native(wasm_f32x4_le(a.value_, b.value_));
6296 }
else if constexpr (std::same_as<T, double>) {
6297 return mask_type::unsafe_from_native(wasm_f64x2_le(a.value_, b.value_));
6298 }
else if constexpr (std::same_as<T, std::int8_t>) {
6299 return mask_type::unsafe_from_native(wasm_i8x16_le(a.value_, b.value_));
6300 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6301 return mask_type::unsafe_from_native(wasm_u8x16_le(a.value_, b.value_));
6302 }
else if constexpr (std::same_as<T, std::int16_t>) {
6303 return mask_type::unsafe_from_native(wasm_i16x8_le(a.value_, b.value_));
6304 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6305 return mask_type::unsafe_from_native(wasm_u16x8_le(a.value_, b.value_));
6306 }
else if constexpr (std::same_as<T, std::int32_t>) {
6307 return mask_type::unsafe_from_native(wasm_i32x4_le(a.value_, b.value_));
6308 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6309 return mask_type::unsafe_from_native(wasm_u32x4_le(a.value_, b.value_));
6310 }
else if constexpr (std::same_as<T, std::int64_t>) {
6311 return mask_type::unsafe_from_native(wasm_i64x2_le(a.value_, b.value_));
6312 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6313 return mask_type::unsafe_from_native(
6314 wasm_i64x2_le(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6315 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6323 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6324 std::uint64_t
bits = 0;
6325 for (std::size_t i = 0; i < N; ++i) {
6326 bits |= std::uint64_t(x[i] > y[i]) << i;
6330 if constexpr (std::same_as<T, float>) {
6331 return mask_type::unsafe_from_native(wasm_f32x4_gt(a.value_, b.value_));
6332 }
else if constexpr (std::same_as<T, double>) {
6333 return mask_type::unsafe_from_native(wasm_f64x2_gt(a.value_, b.value_));
6334 }
else if constexpr (std::same_as<T, std::int8_t>) {
6335 return mask_type::unsafe_from_native(wasm_i8x16_gt(a.value_, b.value_));
6336 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6337 return mask_type::unsafe_from_native(wasm_u8x16_gt(a.value_, b.value_));
6338 }
else if constexpr (std::same_as<T, std::int16_t>) {
6339 return mask_type::unsafe_from_native(wasm_i16x8_gt(a.value_, b.value_));
6340 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6341 return mask_type::unsafe_from_native(wasm_u16x8_gt(a.value_, b.value_));
6342 }
else if constexpr (std::same_as<T, std::int32_t>) {
6343 return mask_type::unsafe_from_native(wasm_i32x4_gt(a.value_, b.value_));
6344 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6345 return mask_type::unsafe_from_native(wasm_u32x4_gt(a.value_, b.value_));
6346 }
else if constexpr (std::same_as<T, std::int64_t>) {
6347 return mask_type::unsafe_from_native(wasm_i64x2_gt(a.value_, b.value_));
6348 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6349 return mask_type::unsafe_from_native(
6350 wasm_i64x2_gt(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6351 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6359 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6360 std::uint64_t
bits = 0;
6361 for (std::size_t i = 0; i < N; ++i) {
6362 bits |= std::uint64_t(x[i] >= y[i]) << i;
6366 if constexpr (std::same_as<T, float>) {
6367 return mask_type::unsafe_from_native(wasm_f32x4_ge(a.value_, b.value_));
6368 }
else if constexpr (std::same_as<T, double>) {
6369 return mask_type::unsafe_from_native(wasm_f64x2_ge(a.value_, b.value_));
6370 }
else if constexpr (std::same_as<T, std::int8_t>) {
6371 return mask_type::unsafe_from_native(wasm_i8x16_ge(a.value_, b.value_));
6372 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6373 return mask_type::unsafe_from_native(wasm_u8x16_ge(a.value_, b.value_));
6374 }
else if constexpr (std::same_as<T, std::int16_t>) {
6375 return mask_type::unsafe_from_native(wasm_i16x8_ge(a.value_, b.value_));
6376 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6377 return mask_type::unsafe_from_native(wasm_u16x8_ge(a.value_, b.value_));
6378 }
else if constexpr (std::same_as<T, std::int32_t>) {
6379 return mask_type::unsafe_from_native(wasm_i32x4_ge(a.value_, b.value_));
6380 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6381 return mask_type::unsafe_from_native(wasm_u32x4_ge(a.value_, b.value_));
6382 }
else if constexpr (std::same_as<T, std::int64_t>) {
6383 return mask_type::unsafe_from_native(wasm_i64x2_ge(a.value_, b.value_));
6384 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6385 return mask_type::unsafe_from_native(
6386 wasm_i64x2_ge(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6387 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6394 requires simd_integer_element<T>
6396 count %=
sizeof(T) * 8;
6398 return detail::wasm_map(
6399 [&](T x) {
return std::bit_cast<T>(word_type(std::uint64_t(word_type(x)) << count)); },
6402 if constexpr (std::same_as<T, std::int8_t>) {
6403 return from_native(wasm_i8x16_shl(value_, count));
6404 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6405 return from_native(wasm_i8x16_shl(value_, count));
6406 }
else if constexpr (std::same_as<T, std::int16_t>) {
6407 return from_native(wasm_i16x8_shl(value_, count));
6408 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6409 return from_native(wasm_i16x8_shl(value_, count));
6410 }
else if constexpr (std::same_as<T, std::int32_t>) {
6411 return from_native(wasm_i32x4_shl(value_, count));
6412 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6413 return from_native(wasm_i32x4_shl(value_, count));
6414 }
else if constexpr (std::same_as<T, std::int64_t>) {
6415 return from_native(wasm_i64x2_shl(value_, count));
6416 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6417 return from_native(wasm_i64x2_shl(value_, count));
6423 template<std::
size_t K>
6425 requires simd_integer_element<T>
6432 requires simd_integer_element<T>
6438 template<std::
size_t K>
6440 requires simd_integer_element<T>
6447 requires simd_integer_element<T>
6449 count %=
sizeof(T) * 8;
6451 return detail::wasm_map([&](T x) {
return T(x >> count); }, *
this);
6453 if constexpr (std::same_as<T, std::int8_t>) {
6454 return from_native(wasm_i8x16_shr(value_, count));
6455 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6456 return from_native(wasm_u8x16_shr(value_, count));
6457 }
else if constexpr (std::same_as<T, std::int16_t>) {
6458 return from_native(wasm_i16x8_shr(value_, count));
6459 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6460 return from_native(wasm_u16x8_shr(value_, count));
6461 }
else if constexpr (std::same_as<T, std::int32_t>) {
6462 return from_native(wasm_i32x4_shr(value_, count));
6463 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6464 return from_native(wasm_u32x4_shr(value_, count));
6465 }
else if constexpr (std::same_as<T, std::int64_t>) {
6466 return from_native(wasm_i64x2_shr(value_, count));
6467 }
else if constexpr (std::same_as<T, std::uint64_t>) {
6468 return from_native(wasm_u64x2_shr(value_, count));
6474 template<std::
size_t K>
6476 requires simd_integer_element<T>
6483 requires simd_integer_element<T>
6489 template<std::
size_t K>
6491 requires simd_integer_element<T>
6498 template<detail::wasm_number T, std::
size_t N, isa<> A>
6499 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6503 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6504 auto bits = m.to_bitset();
6505 for (std::size_t i = 0; i < N; ++i) {
6506 if (!((bits >> i) & 1)) {
6513 wasm_v128_bitselect(a.to_native(), b.to_native(), m.to_native()));
6518 template<detail::wasm_number T, std::
size_t N, isa<> A>
6519 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6526 template<
class To,
class U, std::
size_t N, isa<> A>
6527 requires(A.has(wasm_feature::simd128)) && (
sizeof(U) * N == 16) && (
sizeof(To) ==
sizeof(U)) &&
6528 simd_integer_element<To>
6534 template<std::
size_t I, detail::wasm_number T, std::
size_t N, isa<> A>
6535 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16) && (I < N)
6541 template<std::size_t... I, detail::wasm_number T, std::size_t N, isa<> A>
6542 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16) && (
sizeof...(I) == N) &&
6543 ((I < 2 * N) && ...)
6546 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6547 std::array<T, N> r{(I < N ? x[I] : y[I - N])...};
6550 using vector_type = T __attribute__((ext_vector_type(N)));
6552 v128_t, __builtin_shufflevector(__builtin_bit_cast(vector_type, a.to_native()),
6553 __builtin_bit_cast(vector_type, b.to_native()), I...)));
6559 requires(A.has(wasm_feature::simd128))
6563 auto x = detail::wasm_lanes(v), i = detail::wasm_lanes(indices);
6564 for (
auto & n : i) {
6565 n = n < 16 ? x[n] : 0;
6570 wasm_i8x16_swizzle(v.to_native(), indices.to_native()));
6575 template<std::
floating_po
int T, std::
size_t N, isa<> A>
6576 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6579 return detail::wasm_map(
6581 using format_type = detail::wasm_format<T>;
6582 using bits_type =
typename format_type::bits_type;
6583 auto w = std::bit_cast<bits_type>(x);
6584 return std::bit_cast<T>(detail::constexpr_float::sqrt_bits<format_type>(w));
6588 if constexpr (
sizeof(T) == 4) {
6597 template<std::
floating_po
int T, std::
size_t N, isa<> A>
6598 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6601 return detail::wasm_map(
6603 using format_type = detail::wasm_format<T>;
6604 using bits_type =
typename format_type::bits_type;
6605 auto w = std::bit_cast<bits_type>(x);
6606 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(
6607 w, detail::constexpr_float::rounding::downward));
6611 if constexpr (
sizeof(T) == 4) {
6620 template<std::
floating_po
int T, std::
size_t N, isa<> A>
6621 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6624 return detail::wasm_map(
6626 using format_type = detail::wasm_format<T>;
6627 using bits_type =
typename format_type::bits_type;
6628 auto w = std::bit_cast<bits_type>(x);
6629 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(
6630 w, detail::constexpr_float::rounding::upward));
6634 if constexpr (
sizeof(T) == 4) {
6643 template<std::
floating_po
int T, std::
size_t N, isa<> A>
6644 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6647 return detail::wasm_map(
6649 using format_type = detail::wasm_format<T>;
6650 using bits_type =
typename format_type::bits_type;
6651 auto w = std::bit_cast<bits_type>(x);
6652 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(
6653 w, detail::constexpr_float::rounding::toward_zero));
6657 if constexpr (
sizeof(T) == 4) {
6666 template<std::
floating_po
int T, std::
size_t N, isa<> A>
6667 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6670 return detail::wasm_map(
6672 using format_type = detail::wasm_format<T>;
6673 using bits_type =
typename format_type::bits_type;
6674 auto w = std::bit_cast<bits_type>(x);
6675 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(w));
6679 if constexpr (
sizeof(T) == 4) {
6688 template<std::
floating_po
int T, std::
size_t N, isa<> A>
6689 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6692 return detail::wasm_map(
6694 using format_type = detail::wasm_format<T>;
6695 using bits_type =
typename format_type::bits_type;
6696 auto w = std::bit_cast<bits_type>(x);
6697 return std::bit_cast<T>(bits_type(w & ~format_type::sign_mask));
6701 if constexpr (
sizeof(T) == 4) {
6714 constexpr T wasm_saturate(std::int64_t x)
noexcept {
6715 if (x < std::int64_t(std::numeric_limits<T>::min())) {
6716 return std::numeric_limits<T>::min();
6718 if (x > std::int64_t(std::numeric_limits<T>::max())) {
6719 return std::numeric_limits<T>::max();
6724 template<
class To,
class From>
6725 constexpr To wasm_trunc_sat(From x)
noexcept {
6729 if (x <=
double(std::numeric_limits<To>::min())) {
6730 return std::numeric_limits<To>::min();
6732 if (x >=
double(std::numeric_limits<To>::max())) {
6733 return std::numeric_limits<To>::max();
6740 template<simd_
integer_element T, std::
size_t N, isa<> A>
6741 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 &&
sizeof(T) <= 2)
6745 return detail::wasm_map(
6746 [](T x, T y) {
return detail::wasm_saturate<T>(std::int64_t(x) + std::int64_t(y)); }, a, b);
6748 if constexpr (std::same_as<T, std::int8_t>) {
6749 return vector_type::from_native(wasm_i8x16_add_sat(a.to_native(), b.to_native()));
6750 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6751 return vector_type::from_native(wasm_u8x16_add_sat(a.to_native(), b.to_native()));
6752 }
else if constexpr (std::same_as<T, std::int16_t>) {
6753 return vector_type::from_native(wasm_i16x8_add_sat(a.to_native(), b.to_native()));
6754 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6755 return vector_type::from_native(wasm_u16x8_add_sat(a.to_native(), b.to_native()));
6761 template<simd_
integer_element T, std::
size_t N, isa<> A>
6762 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 &&
sizeof(T) <= 2)
6766 return detail::wasm_map(
6767 [](T x, T y) {
return detail::wasm_saturate<T>(std::int64_t(x) - std::int64_t(y)); }, a, b);
6769 if constexpr (std::same_as<T, std::int8_t>) {
6770 return vector_type::from_native(wasm_i8x16_sub_sat(a.to_native(), b.to_native()));
6771 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6772 return vector_type::from_native(wasm_u8x16_sub_sat(a.to_native(), b.to_native()));
6773 }
else if constexpr (std::same_as<T, std::int16_t>) {
6774 return vector_type::from_native(wasm_i16x8_sub_sat(a.to_native(), b.to_native()));
6775 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6776 return vector_type::from_native(wasm_u16x8_sub_sat(a.to_native(), b.to_native()));
6782 template<simd_
integer_element T, std::
size_t N, isa<> A>
6783 requires(A.has(wasm_feature::simd128)) &&
6784 (std::is_unsigned_v<T> &&
sizeof(T) * N == 16 &&
sizeof(T) <= 2)
6787 return detail::wasm_map([](T x, T y) {
return T((
unsigned(x) + y + 1) / 2); }, a, b);
6789 if constexpr (
sizeof(T) == 1) {
6798 template<simd_
integer_element T, std::
size_t N, isa<> A>
6799 requires(A.has(wasm_feature::simd128)) && (std::is_signed_v<T> &&
sizeof(T) * N == 16)
6802 using unsigned_type = std::make_unsigned_t<T>;
6804 return detail::wasm_map(
6805 [](T x) {
return x < 0 ? std::bit_cast<T>(unsigned_type(0 - unsigned_type(x))) : x; }, a);
6807 if constexpr (
sizeof(T) == 1) {
6808 return vector_type::from_native(wasm_i8x16_abs(a.to_native()));
6809 }
else if constexpr (
sizeof(T) == 2) {
6810 return vector_type::from_native(wasm_i16x8_abs(a.to_native()));
6811 }
else if constexpr (
sizeof(T) == 4) {
6812 return vector_type::from_native(wasm_i32x4_abs(a.to_native()));
6813 }
else if constexpr (
sizeof(T) == 8) {
6814 return vector_type::from_native(wasm_i64x2_abs(a.to_native()));
6820 template<detail::wasm_number T, std::
size_t N, isa<> A>
6821 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6825 return detail::wasm_map(
6827 if constexpr (std::is_floating_point_v<T>) {
6828 using format_type = detail::wasm_format<T>;
6829 using bits_type =
typename format_type::bits_type;
6830 auto xx = std::bit_cast<bits_type>(x), yy = std::bit_cast<bits_type>(y);
6831 if (detail::constexpr_float::is_nan<format_type>(xx) ||
6832 detail::constexpr_float::is_nan<format_type>(yy)) {
6833 return std::bit_cast<T>(detail::constexpr_float::default_nan<format_type>({}));
6835 if (detail::constexpr_float::is_zero<format_type>(xx) &&
6836 detail::constexpr_float::is_zero<format_type>(yy)) {
6837 return std::bit_cast<T>(bits_type(xx | yy));
6840 return y < x ? y : x;
6844 if constexpr (std::same_as<T, float>) {
6845 return vector_type::from_native(wasm_f32x4_min(a.to_native(), b.to_native()));
6846 }
else if constexpr (std::same_as<T, double>) {
6847 return vector_type::from_native(wasm_f64x2_min(a.to_native(), b.to_native()));
6848 }
else if constexpr (std::same_as<T, std::int8_t>) {
6849 return vector_type::from_native(wasm_i8x16_min(a.to_native(), b.to_native()));
6850 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6851 return vector_type::from_native(wasm_u8x16_min(a.to_native(), b.to_native()));
6852 }
else if constexpr (std::same_as<T, std::int16_t>) {
6853 return vector_type::from_native(wasm_i16x8_min(a.to_native(), b.to_native()));
6854 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6855 return vector_type::from_native(wasm_u16x8_min(a.to_native(), b.to_native()));
6856 }
else if constexpr (std::same_as<T, std::int32_t>) {
6857 return vector_type::from_native(wasm_i32x4_min(a.to_native(), b.to_native()));
6858 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6859 return vector_type::from_native(wasm_u32x4_min(a.to_native(), b.to_native()));
6861 return select(a < b, a, b);
6867 template<detail::wasm_number T, std::
size_t N, isa<> A>
6868 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
6872 return detail::wasm_map(
6874 if constexpr (std::is_floating_point_v<T>) {
6875 using format_type = detail::wasm_format<T>;
6876 using bits_type =
typename format_type::bits_type;
6877 auto xx = std::bit_cast<bits_type>(x), yy = std::bit_cast<bits_type>(y);
6878 if (detail::constexpr_float::is_nan<format_type>(xx) ||
6879 detail::constexpr_float::is_nan<format_type>(yy)) {
6880 return std::bit_cast<T>(detail::constexpr_float::default_nan<format_type>({}));
6882 if (detail::constexpr_float::is_zero<format_type>(xx) &&
6883 detail::constexpr_float::is_zero<format_type>(yy)) {
6884 return std::bit_cast<T>(bits_type(xx & yy));
6887 return y > x ? y : x;
6891 if constexpr (std::same_as<T, float>) {
6892 return vector_type::from_native(wasm_f32x4_max(a.to_native(), b.to_native()));
6893 }
else if constexpr (std::same_as<T, double>) {
6894 return vector_type::from_native(wasm_f64x2_max(a.to_native(), b.to_native()));
6895 }
else if constexpr (std::same_as<T, std::int8_t>) {
6896 return vector_type::from_native(wasm_i8x16_max(a.to_native(), b.to_native()));
6897 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6898 return vector_type::from_native(wasm_u8x16_max(a.to_native(), b.to_native()));
6899 }
else if constexpr (std::same_as<T, std::int16_t>) {
6900 return vector_type::from_native(wasm_i16x8_max(a.to_native(), b.to_native()));
6901 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6902 return vector_type::from_native(wasm_u16x8_max(a.to_native(), b.to_native()));
6903 }
else if constexpr (std::same_as<T, std::int32_t>) {
6904 return vector_type::from_native(wasm_i32x4_max(a.to_native(), b.to_native()));
6905 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6906 return vector_type::from_native(wasm_u32x4_max(a.to_native(), b.to_native()));
6908 return select(a > b, a, b);
6914 template<detail::wasm_number T, std::
size_t N, isa<> A>
6915 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 && std::is_floating_point_v<T>)
6919 return detail::wasm_map([](T x, T y) {
return y < x ? y : x; }, a, b);
6921 if constexpr (std::same_as<T, float>) {
6922 return vector_type::from_native(wasm_f32x4_pmin(a.to_native(), b.to_native()));
6923 }
else if constexpr (std::same_as<T, double>) {
6924 return vector_type::from_native(wasm_f64x2_pmin(a.to_native(), b.to_native()));
6930 template<detail::wasm_number T, std::
size_t N, isa<> A>
6931 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 && std::is_floating_point_v<T>)
6935 return detail::wasm_map([](T x, T y) {
return y > x ? y : x; }, a, b);
6937 if constexpr (std::same_as<T, float>) {
6938 return vector_type::from_native(wasm_f32x4_pmax(a.to_native(), b.to_native()));
6939 }
else if constexpr (std::same_as<T, double>) {
6940 return vector_type::from_native(wasm_f64x2_pmax(a.to_native(), b.to_native()));
6946 template<simd_
integer_element T, std::
size_t N, isa<> A>
6947 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 &&
sizeof(T) <= 4)
6949 using wide_unsigned_type =
6950 std::conditional_t<
sizeof(T) == 1, std::uint16_t,
6951 std::conditional_t<
sizeof(T) == 2, std::uint32_t, std::uint64_t>>;
6953 std::conditional_t<std::is_signed_v<T>, std::make_signed_t<wide_unsigned_type>,
6954 wide_unsigned_type>;
6955 using vector_type =
simd<wide_type, N / 2, A>;
6957 auto x = detail::wasm_lanes(a);
6958 std::array<wide_type, N / 2> r{};
6959 for (std::size_t i = 0; i < N / 2; ++i) {
6962 return vector_type::load(r.data());
6964 if constexpr (std::same_as<T, std::int8_t>) {
6965 return vector_type::from_native(wasm_i16x8_extend_low_i8x16(a.to_native()));
6966 }
else if constexpr (std::same_as<T, std::uint8_t>) {
6967 return vector_type::from_native(wasm_u16x8_extend_low_u8x16(a.to_native()));
6968 }
else if constexpr (std::same_as<T, std::int16_t>) {
6969 return vector_type::from_native(wasm_i32x4_extend_low_i16x8(a.to_native()));
6970 }
else if constexpr (std::same_as<T, std::uint16_t>) {
6971 return vector_type::from_native(wasm_u32x4_extend_low_u16x8(a.to_native()));
6972 }
else if constexpr (std::same_as<T, std::int32_t>) {
6973 return vector_type::from_native(wasm_i64x2_extend_low_i32x4(a.to_native()));
6974 }
else if constexpr (std::same_as<T, std::uint32_t>) {
6975 return vector_type::from_native(wasm_u64x2_extend_low_u32x4(a.to_native()));
6981 template<simd_
integer_element T, std::
size_t N, isa<> A>
6982 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 &&
sizeof(T) <= 4)
6988 template<simd_
integer_element T, std::
size_t N, isa<> A>
6989 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 &&
sizeof(T) <= 4)
6991 using wide_unsigned_type =
6992 std::conditional_t<
sizeof(T) == 1, std::uint16_t,
6993 std::conditional_t<
sizeof(T) == 2, std::uint32_t, std::uint64_t>>;
6995 std::conditional_t<std::is_signed_v<T>, std::make_signed_t<wide_unsigned_type>,
6996 wide_unsigned_type>;
6997 using vector_type =
simd<wide_type, N / 2, A>;
6999 auto x = detail::wasm_lanes(a);
7000 std::array<wide_type, N / 2> r{};
7001 for (std::size_t i = 0; i < N / 2; ++i) {
7002 r[i] = x[i + N / 2];
7004 return vector_type::load(r.data());
7006 if constexpr (std::same_as<T, std::int8_t>) {
7007 return vector_type::from_native(wasm_i16x8_extend_high_i8x16(a.to_native()));
7008 }
else if constexpr (std::same_as<T, std::uint8_t>) {
7009 return vector_type::from_native(wasm_u16x8_extend_high_u8x16(a.to_native()));
7010 }
else if constexpr (std::same_as<T, std::int16_t>) {
7011 return vector_type::from_native(wasm_i32x4_extend_high_i16x8(a.to_native()));
7012 }
else if constexpr (std::same_as<T, std::uint16_t>) {
7013 return vector_type::from_native(wasm_u32x4_extend_high_u16x8(a.to_native()));
7014 }
else if constexpr (std::same_as<T, std::int32_t>) {
7015 return vector_type::from_native(wasm_i64x2_extend_high_i32x4(a.to_native()));
7016 }
else if constexpr (std::same_as<T, std::uint32_t>) {
7017 return vector_type::from_native(wasm_u64x2_extend_high_u32x4(a.to_native()));
7023 template<simd_
integer_element T, std::
size_t N, isa<> A>
7024 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 &&
sizeof(T) <= 4)
7030 template<simd_
integer_element To, simd_
integer_element From, std::
size_t N, isa<> A>
7031 requires(A.has(wasm_feature::simd128)) &&
7032 (
sizeof(From) * N == 16 &&
sizeof(From) == 2 *
sizeof(To) &&
sizeof(To) <= 2 &&
7033 std::is_signed_v<From>)
7038 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
7039 std::array<To, N * 2> r{};
7040 for (std::size_t i = 0; i < N; ++i) {
7041 r[i] = detail::wasm_saturate<To>(x[i]);
7042 r[i + N] = detail::wasm_saturate<To>(y[i]);
7044 return vector_type::load(r.data());
7046 if constexpr (std::same_as<To, std::int8_t>) {
7047 return vector_type::from_native(wasm_i8x16_narrow_i16x8(a.to_native(), b.to_native()));
7048 }
else if constexpr (std::same_as<To, std::uint8_t>) {
7049 return vector_type::from_native(wasm_u8x16_narrow_i16x8(a.to_native(), b.to_native()));
7050 }
else if constexpr (std::same_as<To, std::int16_t>) {
7051 return vector_type::from_native(wasm_i16x8_narrow_i32x4(a.to_native(), b.to_native()));
7052 }
else if constexpr (std::same_as<To, std::uint16_t>) {
7053 return vector_type::from_native(wasm_u16x8_narrow_i32x4(a.to_native(), b.to_native()));
7060 requires(A.has(wasm_feature::simd128))
7064 return detail::wasm_map(
7065 [](std::int16_t x, std::int16_t y) {
7066 return detail::wasm_saturate<std::int16_t>((std::int64_t(x) * y + 16384) >> 15);
7071 wasm_i16x8_q15mulr_sat(a.to_native(), b.to_native()));
7077 requires(A.has(wasm_feature::simd128))
7082 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
7083 std::array<std::int32_t, 4> r{};
7084 for (std::size_t i = 0; i < 4; ++i) {
7085 r[i] = std::bit_cast<std::int32_t>(std::uint32_t(
7086 std::int64_t(x[2 * i]) * y[2 * i] + std::int64_t(x[2 * i + 1]) * y[2 * i + 1]));
7088 return vector_type::load(r.data());
7090 return vector_type::from_native(wasm_i32x4_dot_i16x8(a.to_native(), b.to_native()));
7096 template<simd_
integer_element To, std::
floating_po
int From, std::
size_t N, isa<> A>
7097 requires(A.has(wasm_feature::simd128)) && (
sizeof(From) * N == 16 &&
sizeof(To) == 4)
7101 auto x = detail::wasm_lanes(a);
7102 std::array<To, 4> r{};
7103 for (std::size_t i = 0; i < N; ++i) {
7104 r[i] = detail::wasm_trunc_sat<To>(x[i]);
7106 return vector_type::load(r.data());
7108 if constexpr (
sizeof(From) == 4 && std::is_signed_v<To>) {
7109 return vector_type::from_native(wasm_i32x4_trunc_sat_f32x4(a.to_native()));
7110 }
else if constexpr (
sizeof(From) == 4) {
7111 return vector_type::from_native(wasm_u32x4_trunc_sat_f32x4(a.to_native()));
7112 }
else if constexpr (std::is_signed_v<To>) {
7113 return vector_type::from_native(wasm_i32x4_trunc_sat_f64x2_zero(a.to_native()));
7115 return vector_type::from_native(wasm_u32x4_trunc_sat_f64x2_zero(a.to_native()));
7122 template<std::
floating_po
int To, detail::wasm_number From, std::
size_t N, isa<> A>
7123 requires(A.has(wasm_feature::simd128)) &&
7124 (
sizeof(From) * N == 16 && (
sizeof(From) == 4 || std::is_floating_point_v<From>) &&
7125 !std::same_as<To, From>)
7127 using vector_type =
simd<To, 16 /
sizeof(To), A>;
7129 auto x = detail::wasm_lanes(a);
7130 std::array<To, vector_type::lanes> r{};
7131 for (std::size_t i = 0; i < std::min(N, vector_type::lanes); ++i) {
7132 if constexpr (std::is_floating_point_v<From>) {
7133 using format_type = detail::wasm_format<From>;
7134 using result_format = detail::wasm_format<To>;
7136 std::bit_cast<To>(detail::constexpr_float::convert_bits<result_format, format_type>(
7137 std::bit_cast<typename format_type::bits_type>(x[i])));
7142 return vector_type::load(r.data());
7144 if constexpr (std::same_as<From, float>) {
7145 return vector_type::from_native(wasm_f64x2_promote_low_f32x4(a.to_native()));
7146 }
else if constexpr (std::same_as<From, double>) {
7147 return vector_type::from_native(wasm_f32x4_demote_f64x2_zero(a.to_native()));
7148 }
else if constexpr (
sizeof(To) == 4 && std::is_signed_v<From>) {
7149 return vector_type::from_native(wasm_f32x4_convert_i32x4(a.to_native()));
7150 }
else if constexpr (
sizeof(To) == 4) {
7151 return vector_type::from_native(wasm_f32x4_convert_u32x4(a.to_native()));
7152 }
else if constexpr (std::is_signed_v<From>) {
7153 return vector_type::from_native(wasm_f64x2_convert_low_i32x4(a.to_native()));
7155 return vector_type::from_native(wasm_f64x2_convert_low_u32x4(a.to_native()));
7162 requires detail::wasm_number<typename V::value_type> &&
7163 (V::architecture.has(wasm_feature::simd128)) &&
7164 (V::lanes *
sizeof(
typename V::value_type) == 16)
7170 template<std::
size_t I, detail::wasm_number T, std::
size_t N, isa<> A>
7171 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 && I < N)
7173 return a.template replace<I>(*p);
7177 template<std::
size_t I, detail::wasm_number T, std::
size_t N, isa<> A>
7178 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16 && I < N)
7180 *p = a.template get<I>();
7185 requires(V::architecture.has(wasm_feature::simd128)) &&
7186 (V::lanes *
sizeof(
typename V::value_type) == 16 &&
sizeof(
typename V::value_type) >= 4)
7189 std::array<typename V::value_type, V::lanes> a{};
7191 return V::load(a.data());
7193 if constexpr (
sizeof(
typename V::value_type) == 4) {
7194 return V::from_native(wasm_v128_load32_zero(p));
7196 return V::from_native(wasm_v128_load64_zero(p));
7202 template<simd_
integer_element To, simd_
integer_element From, isa<> A>
7203 requires(A.has(wasm_feature::simd128)) &&
7204 (
sizeof(To) == 2 *
sizeof(From) && std::is_signed_v<To> == std::is_signed_v<From>)
7206 using vector_type =
simd<To, 16 /
sizeof(To), A>;
7208 std::array<To, vector_type::lanes> a{};
7209 for (std::size_t i = 0; i < vector_type::lanes; ++i) {
7212 return vector_type::load(a.data());
7214 if constexpr (std::same_as<To, std::int16_t>) {
7215 return vector_type::from_native(wasm_i16x8_load8x8(p));
7216 }
else if constexpr (std::same_as<To, std::uint16_t>) {
7217 return vector_type::from_native(wasm_u16x8_load8x8(p));
7218 }
else if constexpr (std::same_as<To, std::int32_t>) {
7219 return vector_type::from_native(wasm_i32x4_load16x4(p));
7220 }
else if constexpr (std::same_as<To, std::uint32_t>) {
7221 return vector_type::from_native(wasm_u32x4_load16x4(p));
7222 }
else if constexpr (std::same_as<To, std::int64_t>) {
7223 return vector_type::from_native(wasm_i64x2_load32x2(p));
7224 }
else if constexpr (std::same_as<To, std::uint64_t>) {
7225 return vector_type::from_native(wasm_u64x2_load32x2(p));
7231 template<simd_
integer_element T, std::
size_t N, isa<> A>
7232 requires(A.has(wasm_feature::simd128)) &&
7233 (std::is_signed_v<T> &&
sizeof(T) * N == 16 &&
sizeof(T) <= 2)
7235 using wide_type = std::conditional_t<
sizeof(T) == 1, std::int16_t, std::int32_t>;
7236 using vector_type =
simd<wide_type, N / 2, A>;
7238 auto x = detail::wasm_lanes(a);
7239 std::array<wide_type, N / 2> r{};
7240 for (std::size_t i = 0; i < N / 2; ++i) {
7241 r[i] = wide_type(x[2 * i]) + x[2 * i + 1];
7243 return vector_type::load(r.data());
7245 if constexpr (
sizeof(T) == 1) {
7246 return vector_type::from_native(wasm_i16x8_extadd_pairwise_i8x16(a.to_native()));
7248 return vector_type::from_native(wasm_i32x4_extadd_pairwise_i16x8(a.to_native()));
7254 template<simd_
integer_element T, std::
size_t N, isa<> A>
7255 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
7258 auto x = detail::wasm_lanes(a);
7259 std::uint32_t r = 0;
7260 for (std::size_t i = 0; i < N; ++i) {
7261 r |= std::uint32_t(std::make_unsigned_t<T>(x[i]) >> (
sizeof(T) * 8 - 1)) << i;
7265 if constexpr (
sizeof(T) == 1) {
7266 return wasm_i8x16_bitmask(a.to_native());
7267 }
else if constexpr (
sizeof(T) == 2) {
7268 return wasm_i16x8_bitmask(a.to_native());
7269 }
else if constexpr (
sizeof(T) == 4) {
7270 return wasm_i32x4_bitmask(a.to_native());
7272 return wasm_i64x2_bitmask(a.to_native());
7278 template<simd_
integer_element T, std::
size_t N, isa<> A>
7279 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
7282 for (
auto x : detail::wasm_lanes(a)) {
7289 return wasm_v128_any_true(a.to_native());
7294 template<simd_
integer_element T, std::
size_t N, isa<> A>
7295 requires(A.has(wasm_feature::simd128)) && (
sizeof(T) * N == 16)
7298 for (
auto x : detail::wasm_lanes(a)) {
7305 if constexpr (
sizeof(T) == 1) {
7306 return wasm_i8x16_all_true(a.to_native());
7307 }
else if constexpr (
sizeof(T) == 2) {
7308 return wasm_i16x8_all_true(a.to_native());
7309 }
else if constexpr (
sizeof(T) == 4) {
7310 return wasm_i32x4_all_true(a.to_native());
7312 return wasm_i64x2_all_true(a.to_native());
#define native_diagnose_if(condition, message)
Reject a call when Clang can prove that its arguments violate a precondition.
#define native_reinitializes
[[clang::reinitializes]]
#define native_artificial
[[artificial]].
#define native_inline
inline [[always_inline]]
#define native_noescape
portable __attribute__((noescape))
#define native_nodiscard
C++17 [[nodiscard]].
constexpr auto mask_bits(simd< M, N, Arch > m) noexcept
constexpr predicate< N, Arch > to_predicate(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > select(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > to_vector_mask(predicate< N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_mul(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_mul_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< U, N, Arch > mask_cast(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > bit_select(simd< T, N, Arch > bits, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
#define native_const
[[const]] is not const
#define native_pure
[[pure]]
#define native_target(x)
this indicates a required feature set for the current multiversioned function.
std::string_view const type
typename mask_traits< std::remove_cvref_t< T > >::type mask
constexpr simd< float, N, Arch > masked_scaleb_zero(M mask, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > scaleb(simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > ceil(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > masked_scaleb(M mask, simd< float, N, Arch > prior, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< To, N, Arch > convert(simd< float, N, Arch > x) noexcept
constexpr auto abs(simd< float, N, Arch > a) noexcept
constexpr simd< float, N, Arch > trunc(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > floor(simd< float, N, Arch > x) noexcept
constexpr void store_simd(U *p, V value, simd_memory< A, Access >={}) noexcept(noexcept(value.template store_memory< A >(p)))
constexpr std::size_t compress_store(T *destination, std::size_t capacity, typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value) noexcept
constexpr V load_simd(U const *p, simd_memory< A, Access >={}) noexcept(noexcept(V::template load_memory< A >(p)))
constexpr V load_simd_partial(U const *p, std::size_t count, typename V::value_type fill={}, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_constructible_v< U, typename V::value_type & > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::load_simd< V >(p)))
constexpr compaction_result< simd< T, N, Arch > > compress(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value, T fill=T{}) noexcept
constexpr simd< T, N, Arch > expand(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > packed, simd< T, N, Arch > prior) noexcept
constexpr void store_simd_partial(U *p, V value, std::size_t count, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::store_simd(p, value)))
constexpr auto operator!(wide< T, N > const &a)
constexpr auto operator~(wide< T, N > const &a)
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr simd< std::int16_t, 8, A > q15mulr_sat(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Multiply signed Q15 lanes, round by adding 2^14, shift by 15 and saturate.
constexpr simd< T, N, A > average_round(simd< T, N, A > a, simd< T, N, A > b) noexcept
Rounded unsigned average: (a+b+1)/2 without intermediate overflow.
void operator-(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto multiply_widened_high(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened upper integer lanes; the complete product fits.
constexpr simd< T, N, A > sqrt(simd< T, N, A > v) noexcept
Compute correctly rounded square roots; negative finite lanes produce NaN.
void operator/(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr bool all(simd< T, N, A > a) noexcept
Test whether every integer lane is nonzero.
void operator>>(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< To, N *2, A > narrow_sat(simd< From, N, A > a, simd< From, N, A > b) noexcept
Saturating concatenate from signed source lanes, including unsigned destinations.
void operator*=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< bool, N, Arch > to_bool(simd< T, N, Arch > value) noexcept
Convert lane truth into Boolean data lanes represented as zero or one, preserving the lane count.
void operator==(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator%(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > pmax(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
constexpr simd< T, N, A > shuffle(simd< T, N, A > a, simd< T, N, A > b) noexcept
Select N lanes from the concatenation of two inputs, using constant indices.
void operator>=(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator>(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator&=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > round_even(simd< T, N, A > v) noexcept
Round floating lanes to nearest integers, choosing even at ties.
constexpr simd< T, N, A > add_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
architecture
Instruction-set families; an ISA value belongs to exactly one family.
void operator/=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto extend_high(simd< T, N, A > a) noexcept
Widen the upper half of integer lanes, preserving signedness.
void operator|(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > pmin(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
void operator+=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator*(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator-=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto multiply_widened_low(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened lower integer lanes; the complete product fits.
constexpr simd< fp16, 32, Arch > fma(simd< fp16, 32, Arch > a, simd< fp16, 32, Arch > b, simd< fp16, 32, Arch > c) noexcept
constexpr simd< To, 4, A > trunc_sat(simd< From, N, A > a) noexcept
void operator<=(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< std::uint8_t, 16, A > swizzle(simd< std::uint8_t, 16, A > v, simd< std::uint8_t, 16, A > indices) noexcept
Look up byte indices 0..15; every other index produces zero.
constexpr simd< T, N, A > max(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
void operator^=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > broadcast(simd< T, N, A > v, imm_t< I >) noexcept
Broadcast one compile-time-selected lane.
constexpr std::uint32_t bitmask(simd< T, N, A > a) noexcept
Gather each integer lane's sign bit into bit i of the scalar result.
constexpr auto pairwise_add_widened(simd< T, N, Arch > value) noexcept
constexpr V load_splat(typename V::value_type const *p) noexcept
Read one scalar and broadcast it; the access is exactly sizeof(T) bytes.
constexpr simd< std::int32_t, 4, A > dot(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Sum adjacent signed halfword products modulo 2^32.
void operator<<(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto extend_low(simd< T, N, A > a) noexcept
Widen the lower half of integer lanes, preserving signedness.
void operator&(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator^(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator+(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< To, 16/sizeof(To), A > load_widened(From const *p) noexcept
Read exactly eight bytes of source lanes and widen them with their signedness.
void operator!=(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr void store_lane(T *p, simd< T, N, A > a) noexcept
Write only lane I to one scalar object.
constexpr V load_zero(typename V::value_type const *p) noexcept
Read one 32- or 64-bit lane and zero the other lanes.
void operator<(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > sub_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
void operator|=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr bool any(simd< T, N, A > a) noexcept
Test whether at least one integer lane is nonzero.
constexpr simd< T, N, A > load_lane(T const *p, simd< T, N, A > a) noexcept
Read one scalar into lane I, preserving the other lanes.
constexpr simd< T, N, A > min(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
Standard-library adaptations documented here for SIMD value types.
static constexpr mask_lane from_bits(U value) noexcept
Normalize nonzero bits to a true, all-one lane.
static constexpr predicate from_bitset(std::uint64_t bits) noexcept
Construct from the logical lane bitset.
constexpr predicate & operator&=(predicate b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
constexpr predicate & operator^=(predicate b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
friend constexpr predicate operator==(predicate a, predicate b) noexcept
Return a mask whose lanes are true where a == b holds.
constexpr std::uint64_t to_bitset() const noexcept
Pack lane truth into low bits, with lane zero in bit zero.
friend constexpr bool none(predicate a) noexcept
Return true exactly when no logical lane is true.
constexpr predicate & operator|=(predicate b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr predicate operator~(predicate a) noexcept
Invert each lane truth value, preserving the mask representation.
static constexpr predicate from_native(native_type value) noexcept
Import compact mask bits and clear bits above the lane count.
friend constexpr bool all(predicate a) noexcept
Return whether every logical lane is true.
friend constexpr predicate operator!(predicate a) noexcept
Return the lane-wise logical complement, retaining this mask type.
friend constexpr predicate operator&(predicate a, predicate b) noexcept
Bitwise AND of corresponding lane representations.
static constexpr predicate from_bitset(std::uint64_t value) noexcept
Import lane truth from the low logical-lane bits.
constexpr native_type to_native() const noexcept
Return the native storage representation.
constexpr predicate() noexcept=default
Initialize every logical lane to false.
friend constexpr predicate operator^(predicate a, predicate b) noexcept
Bitwise XOR of corresponding lane representations.
friend constexpr predicate select(predicate p, predicate a, predicate b) noexcept
Choose a in true lanes and b in false lanes; both values are already evaluated.
friend constexpr predicate operator!=(predicate a, predicate b) noexcept
Return a mask whose lanes are true where a != b holds.
static constexpr predicate unsafe_from_native(native_type value) noexcept
Import compact bits and clear bits above the lane count, just like from_native.
friend constexpr predicate operator|(predicate a, predicate b) noexcept
Bitwise OR of corresponding lane representations.
friend constexpr bool any(predicate a) noexcept
Return whether at least one logical lane is true.
friend constexpr simd operator>>(simd a, imm_t< K >) noexcept
Shift integer lanes right, extending signed lanes and reducing the count modulo their width.
friend constexpr simd operator<<(simd a, unsigned n) noexcept
Shift integer lanes left, reducing the count modulo the lane width.
constexpr simd replace(T x) const noexcept
Replace one compile-time-selected lane and preserve every other lane.
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
static constexpr simd load_bits(word_type const *p) noexcept
Load lane representations from corresponding unsigned words.
static constexpr simd load(T const *p) noexcept
Read exactly N elements with their natural alignment.
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator|(simd a, simd b) noexcept
Unite representation bits.
constexpr void store_memory(T *p) const noexcept
Store all lanes; Align is the caller-provided pointer alignment.
constexpr simd(std::array< T, N > const &a) noexcept
Copy the array in lane order.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane operation and update this value.
constexpr simd left() const noexcept
Shift lanes left by K modulo the lane width.
constexpr T get() const noexcept
Extract the compile-time-selected lane as its scalar element type.
constexpr simd() noexcept=default
Initialize every lane to zero.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
constexpr native_type to_native() const noexcept
Return the implementation register without changing its bits.
friend constexpr simd operator&(simd a, simd b) noexcept
Intersect representation bits.
constexpr void store_partial(T *p, std::size_t n) const noexcept
Write only the first n lanes; require n <= N. Null is valid when n is zero.
static constexpr simd from_native(native_type x) noexcept
Adopt implementation storage; mask specializations normalize nonzero lanes.
constexpr native_type to_native() const noexcept
Bridge to the implementation register without numerical conversion.
constexpr simd left(unsigned count) const noexcept
WebAssembly shifts reduce the count modulo the lane width.
friend constexpr simd operator>>(simd a, unsigned n) noexcept
Shift integer lanes right, extending signed lanes and reducing the count modulo their width.
static constexpr simd load_partial(T const *p, std::size_t n, T fill={}) noexcept
Read n lanes and fill the rest; require n <= N. Null is valid when n is zero.
friend constexpr simd operator-(simd a) noexcept
Subtract or negate lanes; integer results wrap and floating signs are preserved.
constexpr bits_type bits() const noexcept
Return each half lane as its unchanged unsigned representation.
friend constexpr mask_type operator>(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
friend constexpr simd operator~(simd a) noexcept
Complement every representation bit (or lane truth for masks).
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator<<(simd a, imm_t< K >) noexcept
Shift integer lanes left, reducing the count modulo the lane width.
static constexpr simd from_native(native_type value) noexcept
Adopt register bits unchanged; unused physical bytes are unspecified.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Compare lane inequality and return canonical mask lanes.
static constexpr simd load_memory(T const *p) noexcept
Load all lanes; Align is the caller-provided pointer alignment.
static constexpr simd unsafe_from_native(native_type x) noexcept
Adopt implementation bits; mask callers must supply canonical lanes.
constexpr simd right(unsigned count) const noexcept
WebAssembly shifts reduce the count modulo the lane width.
constexpr void store(T *p) const noexcept
Store exactly 16 unaligned bytes in lane order.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator+(simd a, simd b) noexcept
Add lanes; integer results wrap and floating results round to nearest-even.
friend constexpr simd operator-(simd a, simd b) noexcept
Subtract or negate lanes; integer results wrap and floating signs are preserved.
constexpr void store_bits(word_type *p) const noexcept
Store lane representations as corresponding unsigned words.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane operation and update this value.
constexpr simd right() const noexcept
Shift lanes right by K modulo the lane width.
constexpr simd(U input) noexcept
Broadcast an integer, retaining its low lane-width bits in every lane.
friend constexpr mask_type operator<(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
constexpr bits_type bits() const noexcept
Reinterpret each lane as an unsigned word of the same width.
friend constexpr simd operator/(simd a, simd b) noexcept
Divide floating lanes with WebAssembly IEEE semantics.
friend constexpr mask_type operator==(simd a, simd b) noexcept
Compare lane equality and return canonical mask lanes.
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator^(simd a, simd b) noexcept
Exclusive-or representation bits.
static constexpr simd from_bits(bits_type x) noexcept
Adopt unsigned lane representations without numerical conversion.
friend constexpr simd operator*(simd a, simd b) noexcept
Multiply lanes; integer results wrap and floating results round to nearest-even.
friend constexpr simd operator!(simd a) noexcept
Return the lane-wise logical complement, retaining this mask type.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
friend constexpr mask_type operator<(simd a, U b) noexcept
Return a mask whose lanes are true where a < b holds. Ordering follows the signedness of T....
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds.
friend simd operator%(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend simd operator%(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr simd operator^(U a, simd b) noexcept
Bitwise XOR of corresponding lane representations. Scalar operands are reduced to the lane width and ...
constexpr simd & operator^=(U b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this. Scalar operands are reduce...
friend constexpr simd operator+(simd a, U b) noexcept
Add corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width and bro...
friend constexpr mask_type operator<=(U a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. Ordering follows the signedness of T....
constexpr simd & operator+=(U b) noexcept
Apply the corresponding lane-wise add operation in place and return *this. Scalar operands are reduce...
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. Ordering follows the signedness of T.
friend constexpr simd operator+(simd a, simd b) noexcept
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane-wise divide operation in place and return *this.
static constexpr simd loadu(T const *p) noexcept
Read all logical lanes without an extra alignment promise.
friend simd operator<<(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
static constexpr simd unsafe_from_native(native_type x) noexcept
Adopt native bits and clear physical padding. If T is a mask element, every logical lane must already...
friend simd operator/(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
static constexpr simd load_memory(T const *p) noexcept
Load logical lanes, assuming the template alignment in bytes.
friend simd operator%(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator|(simd a, simd b) noexcept
Bitwise OR of corresponding lane representations.
friend simd operator>>(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
friend constexpr simd operator|(simd a, U b) noexcept
Bitwise OR of corresponding lane representations. Scalar operands are reduced to the lane width and b...
friend constexpr simd round_even(simd a) noexcept
Round each logical lane to an integral value, ties to even, independently of the ambient rounding dir...
friend constexpr mask_type operator!=(simd a, U b) noexcept
Return a mask whose lanes are true where a != b holds. Ordering follows the signedness of T....
friend constexpr simd fma(simd a, simd b, simd c) noexcept
Compute a*b+c with one fused rounding per logical lane in the caller's floating-point environment.
static constexpr simd load_partial(T const *p, std::size_t n, T fill=T(0)) noexcept
Read exactly n logical lanes and fill the remainder; require n <= lanes. For n == 0,...
static constexpr simd unsafe_from_float32(native_type x) noexcept
Adopt raw float storage without numerical conversion or normalization.
friend constexpr simd operator*(U a, simd b) noexcept
Multiply corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
friend constexpr simd operator&(U a, simd b) noexcept
Bitwise AND of corresponding lane representations. Scalar operands are reduced to the lane width and ...
constexpr simd left() const noexcept
Shift each lane left by compile-time K, discarding high bits; require K below the lane bit width.
friend simd operator/(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator/(simd a, simd b) noexcept
constexpr std::uint64_t to_bitset() const noexcept
Pack each logical lane truth value into bit i; higher bits are zero.
constexpr native_type to_native() const noexcept
Return the native storage representation under its minimal register ABI.
constexpr simd & operator-=(U b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this. Scalar operands are r...
friend constexpr simd operator+(simd a) noexcept
Return the unchanged vector value.
static constexpr simd load_bits_partial(std::uint32_t const *p, std::size_t n, std::uint32_t fill=0) noexcept
Read n representation words and fill the remaining logical lanes; require n <= lanes....
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
friend constexpr simd operator^(simd a, simd b) noexcept
Bitwise XOR of corresponding lane representations.
friend constexpr mask_type operator<=(simd a, U b) noexcept
Return a mask whose lanes are true where a <= b holds. Ordering follows the signedness of T....
constexpr simd(bool x) noexcept
Broadcast the supplied value to each logical lane. Unused physical lanes are zero.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. Ordering follows the signedness of T.
friend constexpr simd normal_pow2(simd a) noexcept
Construct normal powers of two; require each input to be an integral exponent in [-126,...
static constexpr simd from_bitset(std::uint64_t bits) noexcept
Import lane i from bit i, clearing bits above the logical lane count.
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exactly the logical count of binary32 words without normalizing their representations.
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
constexpr simd()=default
Default-construct the element customization; its initialization contract is retained.
friend constexpr simd operator&(simd a, U b) noexcept
Bitwise AND of corresponding lane representations. Scalar operands are reduced to the lane width and ...
friend simd operator/(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
static constexpr simd from_native(native_type x) noexcept
Adopt native storage without numerical conversion.
constexpr simd(std::array< T, N > const &values) noexcept
Copy one value per logical lane in array order.
constexpr simd & operator&=(U b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this. Scalar operands are reduce...
friend constexpr mask_type operator>(simd a, U b) noexcept
Return a mask whose lanes are true where a > b holds. Ordering follows the signedness of T....
friend simd operator<<(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
friend simd operator%(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
constexpr simd(std::array< T, N > const &data) noexcept
Copy one value per logical lane in array order.
friend constexpr simd operator&(simd a, simd b) noexcept
Bitwise AND of corresponding lane representations.
constexpr simd & operator*=(U b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this. Scalar operands are r...
friend constexpr simd operator+(U a, simd b) noexcept
Add corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width and bro...
constexpr void store_bits(std::uint32_t *p) const noexcept
Store the exact binary32 words for every logical lane.
friend simd operator>>(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
friend simd operator%(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr mask_type operator!=(U a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. Ordering follows the signedness of T....
constexpr void store_partial(T *p, std::size_t n) const noexcept
Write exactly n logical lanes; require n <= lanes. For n == 0, p may be null.
friend simd operator>>(simd, simd)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
constexpr simd & operator=(simd const &)=default
Copy the stored value and return *this; no numerical conversion is performed.
constexpr simd left() const noexcept
Shift each lane left by compile-time K, discarding high bits; require K below the lane bit width.
friend constexpr mask_type operator<(simd a, simd b) noexcept
Return a mask whose lanes are true where a < b holds.
constexpr void storeu(T *p) const noexcept
Write all logical lanes without an extra alignment promise.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
friend simd operator<<(simd, simd)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
friend constexpr simd operator-(simd a) noexcept
Negate each lane modulo 2^(sizeof(T)*8).
static constexpr simd from_float(float x) noexcept
Broadcast the float value without adding an FTZ or other normalization policy.
friend constexpr mask_type operator>(simd a, simd b) noexcept
Return a mask whose lanes are true where a > b holds. Ordering follows the signedness of T.
friend constexpr simd operator~(simd a) noexcept
Complement every bit in every lane.
friend constexpr simd operator~(simd a) noexcept
Complement every bit in every lane.
friend constexpr mask_type operator==(U a, simd b) noexcept
Return a mask whose lanes are true where a == b holds. Ordering follows the signedness of T....
static constexpr simd load(T const *p) noexcept
Read all logical lanes from an unaligned element pointer.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
friend constexpr mask_type operator>(U a, simd b) noexcept
Return a mask whose lanes are true where a > b holds. Ordering follows the signedness of T....
constexpr void store_memory(T *p) const noexcept
Store logical lanes, assuming the template alignment in bytes.
constexpr simd right() const noexcept
Shift each lane right by compile-time K; signed lanes extend their sign. Require K below the lane bit...
friend constexpr simd operator*(simd a, simd b) noexcept
friend constexpr simd operator^(simd a, U b) noexcept
Bitwise XOR of corresponding lane representations. Scalar operands are reduced to the lane width and ...
friend constexpr simd operator*(simd a, simd b) noexcept
Multiply corresponding lanes modulo 2^(sizeof(T)*8).
constexpr void store_bits_partial(std::uint32_t *p, std::size_t n) const noexcept
Write n exact representation words; require n <= lanes. A zero count permits null.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds.
friend constexpr simd operator|(simd a, simd b) noexcept
Bitwise OR of corresponding lane representations.
friend constexpr bool any(simd x) noexcept
Return true when at least one logical lane is true.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. Ordering follows the signedness of T.
constexpr simd() noexcept=default
Initialize the stored lane values to zero.
constexpr simd & operator|=(U b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this. Scalar operands are reduced...
friend constexpr mask_type operator<(U a, simd b) noexcept
Return a mask whose lanes are true where a < b holds. Ordering follows the signedness of T....
constexpr bits_type bits() const noexcept
Return the exact binary32 lane representations in the unsigned vector.
friend constexpr simd operator|(U a, simd b) noexcept
Bitwise OR of corresponding lane representations. Scalar operands are reduced to the lane width and b...
constexpr simd(simd const &)=default
Copy the stored value without arithmetic or normalization.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
friend constexpr mask_type operator==(simd a, U b) noexcept
Return a mask whose lanes are true where a == b holds. Ordering follows the signedness of T....
friend constexpr simd sqrt(simd a) noexcept
Compute the native square root of each logical lane in the caller's floating-point environment.
friend simd operator%(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator-(simd a, U b) noexcept
Subtract corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
static constexpr simd from_bits(std::uint32_t x) noexcept
Reinterpret binary32 words as lane values without normalization; a scalar word is broadcast.
friend constexpr simd operator+(simd a, simd b) noexcept
Add corresponding lanes modulo 2^(sizeof(T)*8).
friend constexpr simd operator<<(simd a, simd counts) noexcept
Shift each unsigned 32-bit lane by its corresponding count, which must be less than 32.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr simd operator-(simd a, simd b) noexcept
Subtract corresponding lanes modulo 2^(sizeof(T)*8).
friend constexpr mask_type operator>=(U a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. Ordering follows the signedness of T....
constexpr simd(T x) noexcept
Broadcast the supplied value to each logical lane. Unused physical lanes are zero.
friend constexpr simd operator-(simd a) noexcept
Negate every logical lane; floating-point lanes change sign.
static constexpr simd from_bits(bits_type x) noexcept
Reinterpret binary32 words as lane values without normalization; a scalar word is broadcast.
friend constexpr mask_type operator>(simd a, simd b) noexcept
Return a mask whose lanes are true where a > b holds.
friend constexpr mask_type operator<(simd a, simd b) noexcept
Return a mask whose lanes are true where a < b holds. Ordering follows the signedness of T.
friend simd operator/(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator-(U a, simd b) noexcept
Subtract corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
friend constexpr simd select(M mask, simd a, simd b) noexcept
Choose a in true mask lanes and b in false lanes; both values are already evaluated.
constexpr bits_type to_bits() const noexcept
Return the exact binary32 lane representations in the unsigned vector.
static constexpr simd from_native(native_type x) noexcept
Import native lanes, normalizing mask elements; other elements retain their bits. Clear physical padd...
constexpr simd(native_type x) noexcept
Adopt native lane storage and clear physical padding. Mask elements must already be canonical.
friend constexpr simd operator*(simd a, U b) noexcept
Multiply corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
friend constexpr mask_type operator>=(simd a, U b) noexcept
Return a mask whose lanes are true where a >= b holds. Ordering follows the signedness of T....
friend constexpr mask_type operator==(simd a, simd b) noexcept
Return a mask whose lanes are true where a == b holds. Ordering follows the signedness of T.
constexpr storage_type to_storage() const noexcept
Return the corresponding four-lane storage vector without changing logical lane bits.
friend simd operator/(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
static constexpr simd from_storage(storage_type x) noexcept
Copy the first N lanes from a four-lane storage value and clear physical padding.
static constexpr simd load_partial(T const *p, std::size_t n, T fill={}) noexcept
Read exactly n logical lanes and fill the remainder; require n <= lanes. For n == 0,...
friend constexpr simd operator^(simd a, simd b) noexcept
Bitwise XOR of corresponding lane representations.
friend constexpr simd operator-(simd a, simd b) noexcept
constexpr simd right() const noexcept
Shift each lane right by compile-time K; signed lanes extend their sign. Require K below the lane bit...
friend constexpr simd operator&(simd a, simd b) noexcept
Bitwise AND of corresponding lane representations.
friend constexpr bool none(simd x) noexcept
Return true exactly when no logical lane is true.
constexpr native_type to_native() const noexcept
Return the native storage representation, including physical padding when present.
friend constexpr bool all(simd x) noexcept
Return true exactly when every logical lane is true.
constexpr void store(T *p) const noexcept
Write all logical lanes to an unaligned element pointer.
friend simd operator/(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
friend constexpr bool any(simd a) noexcept
Return whether at least one logical lane is true.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr simd operator|(simd a, simd b) noexcept
Bitwise OR of corresponding lane representations.
friend constexpr simd operator!(simd a) noexcept
Return the lane-wise logical complement, retaining this mask type.
constexpr native_type to_native() const noexcept
Return the native storage representation.
static constexpr simd load_partial(bool const *p, std::size_t count, bool fill=false) noexcept
Read exactly n logical lanes and fill the remainder; require n <= lanes. For n == 0,...
friend constexpr bool none(simd a) noexcept
Return true exactly when no logical lane is true.
constexpr void store_partial(bool *p, std::size_t count) const noexcept
Write exactly n logical lanes; require n <= lanes. For n == 0, p may be null.
friend constexpr simd operator&(simd a, simd b) noexcept
Bitwise AND of corresponding lane representations.
static constexpr simd load_memory(bool const *p) noexcept
Load exactly the logical lanes; the template alignment is a caller promise, never permission to read ...
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
constexpr simd(std::array< bool, N > const &value) noexcept
Read all logical lanes from an unaligned element pointer.
friend constexpr simd select(M m, simd a, simd b) noexcept
Choose a in true lanes and b in false lanes; both values are already evaluated.
static constexpr simd unsafe_from_native(native_type value) noexcept
Adopt storage with the precondition that every Boolean byte is zero or one.
friend constexpr bool all(simd a) noexcept
Return whether every logical lane is true.
friend constexpr simd operator~(simd a) noexcept
Invert each lane truth value, preserving the mask representation.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds.
constexpr simd() noexcept
Initialize every logical lane to false.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
static constexpr simd from_native(native_type value) noexcept
Import native byte lanes, converting each nonzero byte to Boolean one.
constexpr void store_memory(bool *p) const noexcept
Store exactly the logical lanes; the template alignment is a caller promise, never permission to writ...
friend constexpr mask_type operator==(simd a, simd b) noexcept
Return a mask whose lanes are true where a == b holds.
friend constexpr simd operator^(simd a, simd b) noexcept
Bitwise XOR of corresponding lane representations.
constexpr simd(bool value) noexcept
Broadcast the supplied truth value to every logical lane.
static constexpr simd load(bool const *p) noexcept
Read all logical lanes from an unaligned element pointer.
constexpr void store(bool *p) const noexcept
Write all logical lanes to an unaligned element pointer.
friend constexpr simd normal_pow2(simd n)
Construct normal powers of two; require integral exponents in [-126,127].
static constexpr simd load(float const *p)
Load every logical lane; no extra alignment is required.
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. NaN lanes yield false.
friend constexpr simd operator-(simd a)
Negate every logical lane; floating-point lanes change sign.
static constexpr simd loadu(float const *p)
Synonym for an unaligned full-vector load.
static constexpr simd from_bits(bits_type bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
friend constexpr mask_type operator<(simd a, simd b)
Return a mask whose lanes are true where a < b holds. NaN lanes yield false.
constexpr void store_bits(std::uint32_t *p) const noexcept
Store exact binary32 representations as uint32_t words.
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exact binary32 representations from uint32_t words.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
static constexpr simd from_float(float x) noexcept
Broadcast one binary32 value to every lane.
friend constexpr mask_type operator==(simd a, simd b)
Return a mask whose lanes are true where a == b holds. NaN lanes yield false.
static constexpr simd load_memory(float const *p) noexcept
Load full lanes, assuming Alignment-byte pointer alignment.
friend constexpr simd sqrt(simd a)
Compute the native square root in every lane.
constexpr void storeu(float *p) const
Synonym for an unaligned full-vector store.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. NaN lanes yield false.
constexpr bits_type to_bits() const noexcept
Synonym for bits(): preserve all binary32 representation bits.
friend constexpr simd operator+(simd a, simd b)
Add corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr simd & operator=(simd const &)=default
Copy the stored value and return *this; no numerical conversion is performed.
friend constexpr simd round_even(simd a)
Round to an integral value, ties to even, independent of ambient direction.
friend constexpr simd operator*(simd a, simd b)
Multiply corresponding floating-point lanes using the caller's rounding and denormal environment.
friend constexpr simd select(M m, simd a, simd b)
Choose a where the canonical mask is true, otherwise b; both operands are evaluated.
static constexpr simd from_bits(std::uint32_t bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
friend constexpr simd operator/(simd a, simd b)
Divide corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr void store_bits_partial(std::uint32_t *p, std::size_t n) const noexcept
Store exactly n representation words; require n <= lanes.
constexpr simd(simd const &)=default
Copy the stored value without arithmetic or normalization.
static constexpr simd load_bits_partial(std::uint32_t const *p, std::size_t n, std::uint32_t fill=0) noexcept
Load n words and fill the remaining lanes; require n <= lanes.
constexpr simd(float x)
Broadcast x to all lanes.
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane-wise divide operation in place and return *this.
constexpr bits_type bits() const noexcept
Project exact binary32 lane words into the unsigned vector.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. NaN lanes compare unequal.
static constexpr simd unsafe_from_float32(native_type x) noexcept
Adopt native raw float storage; this raw type adds no normalization.
constexpr void store_memory(float *p) const noexcept
Store full lanes, assuming Alignment-byte pointer alignment.
friend constexpr mask_type operator>(simd a, simd b)
Return a mask whose lanes are true where a > b holds. NaN lanes yield false.
static constexpr simd from_native(native_type x) noexcept
Adopt native register storage without changing its bits.
constexpr native_type to_native() const noexcept
Project native register storage without a numerical conversion.
constexpr simd(std::array< float, 4 > const &values) noexcept
Load one lane from each array element, in array order.
constexpr void store(float *p) const
Store every logical lane; no extra alignment is required.
simd()=default
Default initialization leaves storage unspecified; value initialization with braces zero-initializes ...
friend constexpr simd fma(simd a, simd b, simd c)
Compute a*b+c with one fused rounding per lane.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
constexpr simd(X... x) noexcept((noexcept(static_cast< float >(x)) &&...))
Convert one argument per lane; exceptions follow those named-lvalue conversions.
friend constexpr simd operator-(simd a, simd b)
Subtract corresponding floating-point lanes using the caller's rounding and denormal environment.
friend constexpr simd normal_pow2(simd n)
Construct normal powers of two; require integral exponents in [-126,127].
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. NaN lanes yield false.
friend constexpr simd operator-(simd a)
Negate every logical lane; floating-point lanes change sign.
simd()=default
Default initialization leaves storage unspecified; value initialization with braces zero-initializes ...
friend constexpr mask_type operator<(simd a, simd b)
Return a mask whose lanes are true where a < b holds. NaN lanes yield false.
static constexpr simd load_memory(float const *p) noexcept
Load full lanes, assuming Alignment-byte pointer alignment.
constexpr void store(float *p) const
Store every logical lane; no extra alignment is required.
static constexpr simd load_bits_partial(std::uint32_t const *p, std::size_t n, std::uint32_t fill=0) noexcept
Load n words and fill the remaining lanes; require n <= lanes.
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane-wise divide operation in place and return *this.
constexpr void store_memory(float *p) const noexcept
Store full lanes, assuming Alignment-byte pointer alignment.
friend constexpr mask_type operator==(simd a, simd b)
Return a mask whose lanes are true where a == b holds. NaN lanes yield false.
constexpr native_type to_native() const noexcept
Project native register storage without a numerical conversion.
friend constexpr simd sqrt(simd a)
Compute the native square root in every lane.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. NaN lanes yield false.
friend constexpr simd operator+(simd a, simd b)
Add corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr simd(float x)
Broadcast x to all lanes.
static constexpr simd load(float const *p)
Load every logical lane; no extra alignment is required.
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exact binary32 representations from uint32_t words.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
friend constexpr simd round_even(simd a)
Round to an integral value, ties to even, independent of ambient direction.
static constexpr simd from_float(float x) noexcept
Broadcast one binary32 value to every lane.
friend constexpr simd operator*(simd a, simd b)
Multiply corresponding floating-point lanes using the caller's rounding and denormal environment.
static constexpr simd loadu(float const *p)
Synonym for an unaligned full-vector load.
friend constexpr simd select(M m, simd a, simd b)
Choose a where the canonical mask is true, otherwise b; both operands are evaluated.
friend constexpr simd operator/(simd a, simd b)
Divide corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
constexpr void storeu(float *p) const
Synonym for an unaligned full-vector store.
static constexpr simd unsafe_from_float32(native_type x) noexcept
Adopt native raw float storage; this raw type adds no normalization.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. NaN lanes compare unequal.
constexpr simd(simd const &)=default
Copy the stored value without arithmetic or normalization.
constexpr bits_type bits() const noexcept
Project exact binary32 lane words into the unsigned vector.
constexpr void store_bits_partial(std::uint32_t *p, std::size_t n) const noexcept
Store exactly n representation words; require n <= lanes.
constexpr simd(X... x) noexcept((noexcept(static_cast< float >(x)) &&...))
Convert one argument per lane; exceptions follow those named-lvalue conversions.
static constexpr simd from_native(native_type x) noexcept
Adopt native register storage without changing its bits.
constexpr bits_type to_bits() const noexcept
Synonym for bits(): preserve all binary32 representation bits.
friend constexpr mask_type operator>(simd a, simd b)
Return a mask whose lanes are true where a > b holds. NaN lanes yield false.
constexpr simd(std::array< float, 8 > const &values) noexcept
Load one lane from each array element, in array order.
static constexpr simd from_bits(bits_type bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
constexpr simd & operator=(simd const &)=default
Copy the stored value and return *this; no numerical conversion is performed.
constexpr void store_bits(std::uint32_t *p) const noexcept
Store exact binary32 representations as uint32_t words.
static constexpr simd from_bits(std::uint32_t bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
friend constexpr simd fma(simd a, simd b, simd c)
Compute a*b+c with one fused rounding per lane.
friend constexpr simd operator-(simd a, simd b)
Subtract corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr std::uint64_t to_bitset() const noexcept
Gather lane truth values into low scalar bits.
friend constexpr simd operator|(simd a, simd b) noexcept
Unite representation bits.
friend constexpr simd operator!(simd a) noexcept
Complement every mask lane.
friend constexpr bool any(simd v) noexcept
Test whether at least one lane is true.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane operation and update this value.
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator!=(simd a, simd b) noexcept
Compare lane inequality and return canonical mask lanes.
friend constexpr bool all(simd v) noexcept
Test whether every lane is true.
static constexpr simd load_memory(value_type const *p) noexcept
Load all lanes; Align is the caller-provided pointer alignment.
constexpr native_type to_native() const noexcept
Return the implementation register without changing its bits.
static constexpr simd from_native(native_type x) noexcept
Adopt implementation storage; mask specializations normalize nonzero lanes.
friend constexpr simd operator&(simd a, simd b) noexcept
Intersect representation bits.
constexpr void store(value_type *p) const noexcept
Write exactly N canonical mask lane objects.
friend constexpr bool none(simd v) noexcept
Test whether no lane is true.
constexpr std::uint64_t bits() const noexcept
Return the compact lane truth bitset.
static constexpr simd unsafe_from_native(native_type x) noexcept
Adopt implementation bits; mask callers must supply canonical lanes.
constexpr simd() noexcept=default
Initialize every lane to zero.
friend constexpr simd operator==(simd a, simd b) noexcept
Compare lane equality and return canonical mask lanes.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator~(simd a) noexcept
Complement every representation bit (or lane truth for masks).
constexpr void store_memory(value_type *p) const noexcept
Store all lanes; Align is the caller-provided pointer alignment.
static constexpr simd load(value_type const *p) noexcept
Read exactly N canonical mask lane objects.
static constexpr simd from_bitset(std::uint64_t bits) noexcept
Expand low scalar bits to canonical mask lanes, ignoring excess bits.
friend constexpr simd select(simd m, simd a, simd b) noexcept
Choose mask lanes from a where m is true, otherwise from b.
friend constexpr simd operator^(simd a, simd b) noexcept
Exclusive-or representation bits.
Omitted architecture arguments use the native.simd provider's baseline.