native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
simd_family.h
1#include "native/simd/common_body.h"
2
3namespace native::detail::NATIVE_BACKEND {
4 template<class T> using swizzle_word = std::conditional_t<sizeof(T)==1,std::uint8_t,
5 std::conditional_t<sizeof(T)==2,std::uint16_t,std::conditional_t<sizeof(T)==4,std::uint32_t,std::uint64_t>>>;
6 template<class T> using swizzle_native = swizzle_word<T> __attribute__((ext_vector_type(4)));
7 template<std::size_t... I> struct swizzle {
8 static constexpr std::size_t size=sizeof...(I);
9 template<class Self> using result = std::conditional_t<size==1,typename Self::value_type,
10 simd<typename Self::value_type,size,Self::architecture>>;
11 static constexpr bool unique=[] {
12 constexpr std::size_t indices[]{I...};
13 for(std::size_t i=0;i<size;++i)
14 for(std::size_t j=0;j<i;++j) if(indices[i]==indices[j]) return false;
15 return true;
16 }();
17 template<std::size_t D> static consteval int slot() {
18 constexpr std::size_t indices[]{I...};
19 for(std::size_t i=0;i<size;++i) if(indices[i]==D) return int(4+i);
20 return int(D);
21 }
22 template<class Self> static consteval bool readable() {
23 using T=typename Self::value_type;
24 if constexpr(!((I<Self::lanes)&&...) || !std::is_trivially_copyable_v<T> ||
25 !std::default_initializable<T> || !(sizeof(T)==1 || sizeof(T)==2 || sizeof(T)==4 || sizeof(T)==8)) return false;
26 else if constexpr(!requires(Self const & self,T * p) { self.template store_memory<1>(p); }) return false;
27 else if constexpr(size==1) return std::copy_constructible<T>;
28 else return requires(T const * p) {
29 sizeof(result<Self>);
30 { result<Self>::template load_memory<1>(p) } -> std::same_as<result<Self>>;
31 };
32 }
33 template<class Self> static consteval bool writable() {
34 if constexpr(std::is_const_v<Self> || !unique || !readable<Self>()) return false;
35 else if constexpr(size>1 && !requires(result<Self> const & rhs,typename Self::value_type * p) { rhs.template store_memory<1>(p); }) return false;
36 else return std::is_copy_assignable_v<typename Self::value_type> &&
37 std::is_copy_constructible_v<result<Self>> && requires(Self & self,typename Self::value_type const * p) {
38 { Self::template load_memory<1>(p) } -> std::same_as<Self>;
39 self=Self::template load_memory<1>(p);
40 };
41 }
42 template<class V> static constexpr bool native_words=[] {
43 using T=typename V::value_type;
44 if constexpr(simd_custom_element<T>) return false;
45 else if constexpr(requires { typename V::native_type; }) return sizeof(typename V::native_type)==sizeof(swizzle_native<T>);
46 else return false;
47 }();
48 template<class Self> static native_inline constexpr auto words(Self const & self) {
49 using T=typename Self::value_type;
50 if constexpr(native_words<Self>) {
51 return __builtin_bit_cast(swizzle_native<T>,self.to_native());
52 } else {
53 std::array<T,4> values{};
54 self.template store_memory<1>(values.data());
55 return __builtin_bit_cast(swizzle_native<T>,values);
56 }
57 }
58 template<class V> static native_inline constexpr V from_words(swizzle_native<typename V::value_type> words) {
59 using T=typename V::value_type;
60 if constexpr(native_words<V>) {
61 return V::from_native(__builtin_bit_cast(typename V::native_type,words));
62 } else {
63 auto values=__builtin_bit_cast(std::array<T,4>,words);
64 return V::template load_memory<1>(values.data());
65 }
66 }
67 template<class Self>
68 native_nodiscard static native_inline constexpr result<Self> read(Self const & self) {
69 using T=typename Self::value_type;
70 auto value=words(self);
71 if constexpr(size==1) {
72 constexpr std::size_t indices[]{I...};
73 return std::bit_cast<T>(swizzle_word<T>(value[indices[0]]));
74 } else {
75 constexpr int indices[]{int(I)...};
76 auto shuffled=__builtin_shufflevector(value,value,
77 indices[0],indices[1],indices[2%size],indices[3%size]);
78 return from_words<result<Self>>(shuffled);
79 }
80 }
81 template<class Self>
82 native_inline constexpr static result<Self> write(Self & self,result<Self> rhs) {
83 using T=typename Self::value_type;
84 auto before=words(self);
85 auto replacement=[&] {
86 if constexpr(size==1) return swizzle_native<T>{std::bit_cast<swizzle_word<T>>(rhs),0,0,0};
87 else return words(rhs);
88 }();
89 auto shuffled=__builtin_shufflevector(before,replacement,slot<0>(),slot<1>(),slot<2>(),slot<3>());
90 self=from_words<Self>(shuffled);
91 return rhs;
92 }
93 };
94}
95namespace native::detail {
96 // Properties are compiler accessors, not proxy objects: a read owns its lanes,
97 // and assignment materializes the complete right side before any scatter.
98 template<class T,std::size_t N,::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) &&(N<=4)
99 struct swizzle_access<T,N,Arch> {
100 template<std::size_t K> using result = std::conditional_t<K==1,T,simd<T,K,Arch>>;
101#define NATIVE_SWIZZLE_FIELD(NAME,K,...) \
102 template<class Self> requires(::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::template readable<Self>()) \
103 native_nodiscard native_inline constexpr result<K> get_##NAME(this Self const & self) { return ::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::read(self); } \
104 template<class Self> requires(::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::template writable<Self>()) \
105 native_inline constexpr result<K> set_##NAME(this Self & self,result<K> rhs) { return ::NATIVE_BACKEND_NAMESPACE::swizzle<__VA_ARGS__>::write(self,rhs); } \
106 __declspec(property(get=get_##NAME,put=set_##NAME)) result<K> NAME;
107#define NATIVE_SWIZZLE_ROW2(A,I,X,Y,Z,W) \
108 NATIVE_SWIZZLE_FIELD(A##X,2,I,0) NATIVE_SWIZZLE_FIELD(A##Y,2,I,1) \
109 NATIVE_SWIZZLE_FIELD(A##Z,2,I,2) NATIVE_SWIZZLE_FIELD(A##W,2,I,3)
110#define NATIVE_SWIZZLE_ROW3(A,I,B,J,X,Y,Z,W) \
111 NATIVE_SWIZZLE_FIELD(A##B##X,3,I,J,0) NATIVE_SWIZZLE_FIELD(A##B##Y,3,I,J,1) \
112 NATIVE_SWIZZLE_FIELD(A##B##Z,3,I,J,2) NATIVE_SWIZZLE_FIELD(A##B##W,3,I,J,3)
113#define NATIVE_SWIZZLE_PLANE3(A,I,X,Y,Z,W) \
114 NATIVE_SWIZZLE_ROW3(A,I,X,0,X,Y,Z,W) NATIVE_SWIZZLE_ROW3(A,I,Y,1,X,Y,Z,W) \
115 NATIVE_SWIZZLE_ROW3(A,I,Z,2,X,Y,Z,W) NATIVE_SWIZZLE_ROW3(A,I,W,3,X,Y,Z,W)
116#define NATIVE_SWIZZLE_ROW4(A,I,B,J,C,K,X,Y,Z,W) \
117 NATIVE_SWIZZLE_FIELD(A##B##C##X,4,I,J,K,0) NATIVE_SWIZZLE_FIELD(A##B##C##Y,4,I,J,K,1) \
118 NATIVE_SWIZZLE_FIELD(A##B##C##Z,4,I,J,K,2) NATIVE_SWIZZLE_FIELD(A##B##C##W,4,I,J,K,3)
119#define NATIVE_SWIZZLE_PLANE4(A,I,B,J,X,Y,Z,W) \
120 NATIVE_SWIZZLE_ROW4(A,I,B,J,X,0,X,Y,Z,W) NATIVE_SWIZZLE_ROW4(A,I,B,J,Y,1,X,Y,Z,W) \
121 NATIVE_SWIZZLE_ROW4(A,I,B,J,Z,2,X,Y,Z,W) NATIVE_SWIZZLE_ROW4(A,I,B,J,W,3,X,Y,Z,W)
122#define NATIVE_SWIZZLE_CUBE4(A,I,X,Y,Z,W) \
123 NATIVE_SWIZZLE_PLANE4(A,I,X,0,X,Y,Z,W) NATIVE_SWIZZLE_PLANE4(A,I,Y,1,X,Y,Z,W) \
124 NATIVE_SWIZZLE_PLANE4(A,I,Z,2,X,Y,Z,W) NATIVE_SWIZZLE_PLANE4(A,I,W,3,X,Y,Z,W)
125#define NATIVE_SWIZZLE4(X,Y,Z,W) \
126 NATIVE_SWIZZLE_FIELD(X,1,0) NATIVE_SWIZZLE_FIELD(Y,1,1) NATIVE_SWIZZLE_FIELD(Z,1,2) NATIVE_SWIZZLE_FIELD(W,1,3) \
127 NATIVE_SWIZZLE_ROW2(X,0,X,Y,Z,W) NATIVE_SWIZZLE_ROW2(Y,1,X,Y,Z,W) NATIVE_SWIZZLE_ROW2(Z,2,X,Y,Z,W) NATIVE_SWIZZLE_ROW2(W,3,X,Y,Z,W) \
128 NATIVE_SWIZZLE_PLANE3(X,0,X,Y,Z,W) NATIVE_SWIZZLE_PLANE3(Y,1,X,Y,Z,W) NATIVE_SWIZZLE_PLANE3(Z,2,X,Y,Z,W) NATIVE_SWIZZLE_PLANE3(W,3,X,Y,Z,W) \
129 NATIVE_SWIZZLE_CUBE4(X,0,X,Y,Z,W) NATIVE_SWIZZLE_CUBE4(Y,1,X,Y,Z,W) NATIVE_SWIZZLE_CUBE4(Z,2,X,Y,Z,W) NATIVE_SWIZZLE_CUBE4(W,3,X,Y,Z,W)
130 NATIVE_SWIZZLE4(x,y,z,w)
131#undef NATIVE_SWIZZLE4
132#undef NATIVE_SWIZZLE_CUBE4
133#undef NATIVE_SWIZZLE_PLANE4
134#undef NATIVE_SWIZZLE_ROW4
135#undef NATIVE_SWIZZLE_PLANE3
136#undef NATIVE_SWIZZLE_ROW3
137#undef NATIVE_SWIZZLE_ROW2
138#undef NATIVE_SWIZZLE_FIELD
139 };
140}
141
142// SPDX-FileCopyrightText: 2026 Edward Kmett <ekmett@gmail.com>
143// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
144// Included once per disjoint backend family under its function target scope.
145// Raw SIMD types and operations for the selected compile-time ISA profile.
146
147#if NATIVE_HAS_AVX2 || NATIVE_HAS_AVX512F
148#endif
149#if NATIVE_HAS_ARM_NEON
150#endif
151
152namespace native {
153 namespace detail::NATIVE_BACKEND {
154 // Register storage can exist without this backend's arithmetic operations.
155 template<std::size_t N> inline constexpr bool float_shape = N==1
156#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
157 || N==2 || N==3 || N==4
158#endif
159#if NATIVE_HAS_AVX2
160 || N==8
161#endif
162#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
163 || N==16
164#endif
165 ;
166 template <class T> inline constexpr std::size_t mask_lane_bytes = [] {
167 if constexpr (::native::simd_custom_element<T>) return sizeof(typename ::native::simd_traits<T>::storage_type);
168 else return sizeof(T);
169 }();
170 template <class T> concept mask_element = ::native::simd_custom_element<T> ||
171 (std::is_arithmetic_v<T> && !std::is_const_v<T> && !std::is_volatile_v<T> &&
172 (sizeof(T)==1 || sizeof(T)==2 || sizeof(T)==4 || sizeof(T)==8));
173 template <std::size_t B,std::size_t N> inline constexpr bool mask_compact =
174#if NATIVE_HAS_AVX512F
175 N>1 && (B>=4
176#if NATIVE_HAS_AVX512BW
177 || B==1 || B==2
178#endif
179 ) && (B*N==64
180#if NATIVE_HAS_AVX512VL
181 || B*N==16 || B*N==32
182#endif
183 );
184#else
185 false;
186#endif
187 template <std::size_t B,std::size_t N> inline constexpr bool mask_shape = N==1 || (B*N==64 && mask_compact<B,N>)
188#if NATIVE_HAS_AVX2
189 || B*N==16 || B*N==32
190#endif
191#if NATIVE_HAS_ARM_NEON
192 || B*N==16
193#endif
194 ;
195 template <std::size_t N> inline constexpr std::uint64_t mask_low_bits = [] {
196 if constexpr (N==64) return ~std::uint64_t(0);
197 else return (std::uint64_t(1)<<N)-1;
198 }();
199 template <std::size_t B> using mask_word = std::conditional_t<B==1,std::uint8_t,
200 std::conditional_t<B==2,std::uint16_t,std::conditional_t<B==4,std::uint32_t,std::uint64_t>>>;
201
202 template<class U> struct mask_scalar_ops {
203 using native_type=U;
204 static native_inline native_const constexpr native_type normalize(native_type x) noexcept { return x?native_type(~U(0)):U(0); }
205 static native_inline native_const constexpr native_type broadcast(bool x) noexcept { return normalize(U(x)); }
206 static native_inline native_const constexpr native_type bit_and(native_type a,native_type b) noexcept { return U(a&b); }
207 static native_inline native_const constexpr native_type bit_or(native_type a,native_type b) noexcept { return U(a|b); }
208 static native_inline native_const constexpr native_type bit_xor(native_type a,native_type b) noexcept { return U(a^b); }
209 static native_inline native_const constexpr native_type bit_not(native_type a) noexcept { return U(~a); }
210 static native_inline native_const constexpr bool any(native_type a) noexcept { return a!=0; }
211 static native_inline native_const constexpr bool all(native_type a) noexcept { return a==U(~U(0)); }
212 static native_inline native_const constexpr std::uint64_t bits(native_type a) noexcept { return a?1:0; }
213 static native_inline native_const constexpr native_type from_bits(std::uint64_t a) noexcept { return broadcast((a&1)!=0); }
214 };
215 template <std::size_t N> struct mask_compact_ops {
216 using native_type = std::conditional_t<(N<=8),std::uint8_t,std::conditional_t<(N<=16),std::uint16_t,
217 std::conditional_t<(N<=32),std::uint32_t,std::uint64_t>>>;
218#if NATIVE_HOST_X86
219 native_target("sse2")
220#endif
221 static native_inline native_const constexpr native_type normalize(native_type a) noexcept {
222 if constexpr(N==8 || N==16 || N==32 || N==64) return a;
223 else return native_type(std::uint64_t(a)&mask_low_bits<N>);
224 }
225 static native_inline native_const constexpr native_type broadcast(bool a) noexcept { return a?native_type(mask_low_bits<N>):native_type(0); }
226 static native_inline native_const constexpr native_type bit_and(native_type a,native_type b) noexcept {
227 if (std::is_constant_evaluated()) return native_type(a&b);
228#if NATIVE_HAS_AVX512F
229#if NATIVE_HAS_AVX512DQ
230 if constexpr(N<=8) return _kand_mask8(a,b);
231 else
232#endif
233 if constexpr(N<=16) return native_type(_kand_mask16(a,b));
234#if NATIVE_HAS_AVX512BW
235 else if constexpr(N<=32) return _kand_mask32(a,b);
236 else return _kand_mask64(a,b);
237#else
238 else return native_type(a&b);
239#endif
240#else
241 return native_type(a&b);
242#endif
243 }
244 static native_inline native_const constexpr native_type bit_or(native_type a,native_type b) noexcept {
245 if (std::is_constant_evaluated()) return native_type(a|b);
246#if NATIVE_HAS_AVX512F
247#if NATIVE_HAS_AVX512DQ
248 if constexpr(N<=8) return _kor_mask8(a,b);
249 else
250#endif
251 if constexpr(N<=16) return native_type(_kor_mask16(a,b));
252#if NATIVE_HAS_AVX512BW
253 else if constexpr(N<=32) return _kor_mask32(a,b);
254 else return _kor_mask64(a,b);
255#else
256 else return native_type(a|b);
257#endif
258#else
259 return native_type(a|b);
260#endif
261 }
262 static native_inline native_const constexpr native_type bit_xor(native_type a,native_type b) noexcept {
263 if (std::is_constant_evaluated()) return native_type(a^b);
264#if NATIVE_HAS_AVX512F
265#if NATIVE_HAS_AVX512DQ
266 if constexpr(N<=8) return _kxor_mask8(a,b);
267 else
268#endif
269 if constexpr(N<=16) return native_type(_kxor_mask16(a,b));
270#if NATIVE_HAS_AVX512BW
271 else if constexpr(N<=32) return _kxor_mask32(a,b);
272 else return _kxor_mask64(a,b);
273#else
274 else return native_type(a^b);
275#endif
276#else
277 return native_type(a^b);
278#endif
279 }
280 static native_inline native_const constexpr native_type bit_not(native_type a) noexcept {
281 if(std::is_constant_evaluated()) return native_type((~std::uint64_t(a))&mask_low_bits<N>);
282 // Partial mask8 values complement only their N logical lanes.
283 if constexpr(N<8) return bit_xor(a,native_type(mask_low_bits<N>));
284#if NATIVE_HAS_AVX512F
285#if NATIVE_HAS_AVX512DQ
286 else if constexpr(N==8) return _knot_mask8(a);
287#endif
288 else if constexpr(N<=16) return native_type(_knot_mask16(a));
289#if NATIVE_HAS_AVX512BW
290 else if constexpr(N<=32) return _knot_mask32(a);
291 else return _knot_mask64(a);
292#else
293 else return native_type(~a);
294#endif
295#else
296 else return native_type(~a);
297#endif
298 }
299 static native_inline native_const constexpr bool any(native_type a) noexcept { return a!=0; }
300 static native_inline native_const constexpr bool all(native_type a) noexcept { return a==native_type(mask_low_bits<N>); }
301 static native_inline native_const constexpr std::uint64_t bits(native_type a) noexcept { return a; }
302 static native_inline native_const constexpr native_type from_bits(std::uint64_t a) noexcept { return normalize(native_type(a)); }
303 };
304 template <std::size_t B,std::size_t N> struct mask_vector_ops {
305#if NATIVE_HAS_AVX2
306 using native_type = std::conditional_t<B*N==16,__m128i,__m256i>;
307 static native_inline constexpr native_const native_type broadcast(bool a) noexcept {
308 if consteval {
309 std::array<mask_word<B>,N> lanes{};
310 lanes.fill(a?mask_word<B>(~mask_word<B>(0)):0);
311 return __builtin_bit_cast(native_type,lanes);
312 }
313 if constexpr (B*N==16) return _mm_set1_epi32(a?-1:0);
314 else return _mm256_set1_epi32(a?-1:0);
315 }
316 static native_inline constexpr native_const native_type bit_and(native_type a,native_type b) noexcept {
317 if consteval {
318 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
319 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
320 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]&y[i]);
321 return __builtin_bit_cast(native_type,x);
322 }
323 if constexpr (B*N==16) return _mm_and_si128(a,b);
324 else return _mm256_and_si256(a,b);
325 }
326 static native_inline constexpr native_const native_type bit_or(native_type a,native_type b) noexcept {
327 if consteval {
328 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
329 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
330 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]|y[i]);
331 return __builtin_bit_cast(native_type,x);
332 }
333 if constexpr (B*N==16) return _mm_or_si128(a,b);
334 else return _mm256_or_si256(a,b);
335 }
336 static native_inline constexpr native_const native_type bit_xor(native_type a,native_type b) noexcept {
337 if consteval {
338 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
339 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
340 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]^y[i]);
341 return __builtin_bit_cast(native_type,x);
342 }
343 if constexpr (B*N==16) return _mm_xor_si128(a,b);
344 else return _mm256_xor_si256(a,b);
345 }
346 static native_inline constexpr native_const native_type bit_not(native_type a) noexcept { return bit_xor(a,broadcast(true)); }
347 static native_inline constexpr native_const native_type normalize(native_type a) noexcept {
348 if consteval {
349 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
350 for(auto & lane:lanes) lane=lane?mask_word<B>(~mask_word<B>(0)):0;
351 return __builtin_bit_cast(native_type,lanes);
352 }
353 auto z=broadcast(false);
354 if constexpr (B*N==16) {
355 if constexpr (B==1) return bit_not(_mm_cmpeq_epi8(a,z));
356 else if constexpr (B==2) return bit_not(_mm_cmpeq_epi16(a,z));
357 else if constexpr (B==4) return bit_not(_mm_cmpeq_epi32(a,z));
358 else return bit_not(_mm_cmpeq_epi64(a,z));
359 } else {
360 if constexpr (B==1) return bit_not(_mm256_cmpeq_epi8(a,z));
361 else if constexpr (B==2) return bit_not(_mm256_cmpeq_epi16(a,z));
362 else if constexpr (B==4) return bit_not(_mm256_cmpeq_epi32(a,z));
363 else return bit_not(_mm256_cmpeq_epi64(a,z));
364 }
365 }
366 static native_inline constexpr native_const bool any(native_type a) noexcept {
367 if consteval {
368 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
369 for(auto lane:lanes) if(lane!=0) return true;
370 return false;
371 }
372 if constexpr (B*N==16) return _mm_testz_si128(a,a)==0;
373 else return _mm256_testz_si256(a,a)==0;
374 }
375 static native_inline constexpr native_const bool all(native_type a) noexcept {
376 if consteval {
377 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
378 for(auto lane:lanes) if(lane!=mask_word<B>(~mask_word<B>(0))) return false;
379 return true;
380 }
381 if constexpr (B*N==16) return _mm_movemask_epi8(a)==0xffff;
382 else return _mm256_movemask_epi8(a)==-1;
383 }
384 static native_inline constexpr native_const std::uint64_t bits(native_type a) noexcept {
385 if consteval {
386 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
387 std::uint64_t result=0;
388 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(lanes[i]!=0)<<i;
389 return result;
390 }
391 std::uint32_t bytes;
392 if constexpr (B*N==16) bytes=std::uint32_t(_mm_movemask_epi8(a));
393 else bytes=std::uint32_t(_mm256_movemask_epi8(a));
394 std::uint64_t result=0;
395 for(std::size_t i=0;i<N;++i) result|=std::uint64_t((bytes>>(i*B))&1)<<i;
396 return result;
397 }
398 static native_inline constexpr native_const native_type from_bits(std::uint64_t bits) noexcept {
399 if consteval {
400 std::array<mask_word<B>,N> lanes{};
401 for(std::size_t i=0;i<N;++i) lanes[i]=((bits>>i)&1)?mask_word<B>(~mask_word<B>(0)):0;
402 return __builtin_bit_cast(native_type,lanes);
403 }
404 std::array<mask_word<B>,N> a{};
405 for(std::size_t i=0;i<N;++i) a[i]=((bits>>i)&1)?mask_word<B>(~mask_word<B>(0)):mask_word<B>(0);
406 if constexpr (B*N==16) return _mm_loadu_si128(reinterpret_cast<__m128i const *>(a.data()));
407 else return _mm256_loadu_si256(reinterpret_cast<__m256i const *>(a.data()));
408 }
409#elif NATIVE_HAS_ARM_NEON
410 using native_type = uint8x16_t;
411 static native_inline constexpr native_const native_type broadcast(bool a) noexcept {
412 if consteval {
413 std::array<mask_word<B>,N> lanes{};
414 lanes.fill(a?mask_word<B>(~mask_word<B>(0)):0);
415 return __builtin_bit_cast(native_type,lanes);
416 }
417 return vdupq_n_u8(a?0xff:0);
418 }
419 static native_inline constexpr native_const native_type bit_and(native_type a,native_type b) noexcept {
420 if consteval {
421 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
422 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
423 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]&y[i]);
424 return __builtin_bit_cast(native_type,x);
425 }
426 return vandq_u8(a,b);
427 }
428 static native_inline constexpr native_const native_type bit_or(native_type a,native_type b) noexcept {
429 if consteval {
430 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
431 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
432 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]|y[i]);
433 return __builtin_bit_cast(native_type,x);
434 }
435 return vorrq_u8(a,b);
436 }
437 static native_inline constexpr native_const native_type bit_xor(native_type a,native_type b) noexcept {
438 if consteval {
439 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
440 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
441 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]^y[i]);
442 return __builtin_bit_cast(native_type,x);
443 }
444 return veorq_u8(a,b);
445 }
446 static native_inline constexpr native_const native_type bit_not(native_type a) noexcept { if consteval { return bit_xor(a,broadcast(true)); } return vmvnq_u8(a); }
447 static native_inline constexpr native_const native_type normalize(native_type a) noexcept {
448 if consteval {
449 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
450 for(auto & lane:lanes) lane=lane?mask_word<B>(~mask_word<B>(0)):0;
451 return __builtin_bit_cast(native_type,lanes);
452 }
453 if constexpr(B==1) return bit_not(vceqq_u8(a,vdupq_n_u8(0)));
454 else if constexpr(B==2) return bit_not(vreinterpretq_u8_u16(vceqq_u16(vreinterpretq_u16_u8(a),vdupq_n_u16(0))));
455 else if constexpr(B==4) return bit_not(vreinterpretq_u8_u32(vceqq_u32(vreinterpretq_u32_u8(a),vdupq_n_u32(0))));
456 else return bit_not(vreinterpretq_u8_u64(vceqq_u64(vreinterpretq_u64_u8(a),vdupq_n_u64(0))));
457 }
458 static native_inline constexpr native_const bool any(native_type a) noexcept {
459 if consteval {
460 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
461 for(auto lane:lanes) if(lane!=0) return true;
462 return false;
463 }
464 return vmaxvq_u8(a)!=0;
465 }
466 static native_inline constexpr native_const bool all(native_type a) noexcept {
467 if consteval {
468 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
469 for(auto lane:lanes) if(lane!=mask_word<B>(~mask_word<B>(0))) return false;
470 return true;
471 }
472 return vminvq_u8(a)==0xff;
473 }
474 static native_inline constexpr native_const std::uint64_t bits(native_type a) noexcept {
475 if consteval {
476 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
477 std::uint64_t result=0;
478 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(lanes[i]!=0)<<i;
479 return result;
480 }
481 std::array<std::uint8_t,16> bytes{};vst1q_u8(bytes.data(),a);
482 std::uint64_t result=0;
483 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(bytes[i*B]!=0)<<i;
484 return result;
485 }
486 static native_inline constexpr native_const native_type from_bits(std::uint64_t bits) noexcept {
487 if consteval {
488 std::array<mask_word<B>,N> lanes{};
489 for(std::size_t i=0;i<N;++i) lanes[i]=((bits>>i)&1)?mask_word<B>(~mask_word<B>(0)):0;
490 return __builtin_bit_cast(native_type,lanes);
491 }
492 std::array<std::uint8_t,16> a{};
493 for(std::size_t i=0;i<N;++i) for(std::size_t j=0;j<B;++j) a[i*B+j]=((bits>>i)&1)?0xff:0;
494 return vld1q_u8(a.data());
495 }
496#endif
497 };
498 template<std::size_t B,std::size_t N> struct mask_vector512_ops {
499#if NATIVE_HAS_AVX512F
500 using native_type=__m512i;
501 static native_inline constexpr native_const native_type broadcast(bool a) noexcept {
502 if consteval {
503 std::array<mask_word<B>,N> lanes{};
504 lanes.fill(a?mask_word<B>(~mask_word<B>(0)):0);
505 return __builtin_bit_cast(native_type,lanes);
506 }
507 return _mm512_set1_epi32(a?-1:0);
508 }
509 static native_inline constexpr native_const native_type bit_and(native_type a,native_type b) noexcept {
510 if consteval {
511 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
512 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
513 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]&y[i]);
514 return __builtin_bit_cast(native_type,x);
515 }
516 return _mm512_and_si512(a,b);
517 }
518 static native_inline constexpr native_const native_type bit_or(native_type a,native_type b) noexcept {
519 if consteval {
520 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
521 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
522 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]|y[i]);
523 return __builtin_bit_cast(native_type,x);
524 }
525 return _mm512_or_si512(a,b);
526 }
527 static native_inline constexpr native_const native_type bit_xor(native_type a,native_type b) noexcept {
528 if consteval {
529 auto x=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
530 auto y=__builtin_bit_cast(std::array<mask_word<B>,N>,b);
531 for(std::size_t i=0;i<N;++i) x[i]=mask_word<B>(x[i]^y[i]);
532 return __builtin_bit_cast(native_type,x);
533 }
534 return _mm512_xor_si512(a,b);
535 }
536 static native_inline constexpr native_const native_type bit_not(native_type a) noexcept { return bit_xor(a,broadcast(true)); }
537 static native_inline constexpr native_const std::uint64_t bits(native_type a) noexcept {
538 if consteval {
539 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
540 std::uint64_t result=0;
541 for(std::size_t i=0;i<N;++i) result|=std::uint64_t(lanes[i]!=0)<<i;
542 return result;
543 }
544 if constexpr(B==1) return _mm512_cmpneq_epi8_mask(a,_mm512_setzero_si512());
545 else if constexpr(B==2) return _mm512_cmpneq_epi16_mask(a,_mm512_setzero_si512());
546 else if constexpr(B==4) return _mm512_cmpneq_epi32_mask(a,_mm512_setzero_si512());
547 else return _mm512_cmpneq_epi64_mask(a,_mm512_setzero_si512());
548 }
549 static native_inline constexpr native_const native_type from_bits(std::uint64_t a) noexcept {
550 if consteval {
551 std::array<mask_word<B>,N> lanes{};
552 for(std::size_t i=0;i<N;++i) lanes[i]=((a>>i)&1)?mask_word<B>(~mask_word<B>(0)):0;
553 return __builtin_bit_cast(native_type,lanes);
554 }
555 if constexpr(B==1) return _mm512_maskz_set1_epi8(__mmask64(a),-1);
556 else if constexpr(B==2) return _mm512_maskz_set1_epi16(__mmask32(a),-1);
557 else if constexpr(B==4) return _mm512_maskz_set1_epi32(__mmask16(a),-1);
558 else return _mm512_maskz_set1_epi64(__mmask8(a),-1);
559 }
560 static native_inline constexpr native_const native_type normalize(native_type a) noexcept {
561 if consteval {
562 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
563 for(auto & lane:lanes) lane=lane?mask_word<B>(~mask_word<B>(0)):0;
564 return __builtin_bit_cast(native_type,lanes);
565 }
566 return from_bits(bits(a));
567 }
568 static native_inline constexpr native_const bool any(native_type a) noexcept {
569 if consteval {
570 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
571 for(auto lane:lanes) if(lane!=0) return true;
572 return false;
573 }
574 return bits(a)!=0;
575 }
576 static native_inline constexpr native_const bool all(native_type a) noexcept {
577 if consteval {
578 auto lanes=__builtin_bit_cast(std::array<mask_word<B>,N>,a);
579 for(auto lane:lanes) if(lane!=mask_word<B>(~mask_word<B>(0))) return false;
580 return true;
581 }
582 return bits(a)==mask_low_bits<N>;
583 }
584#endif
585 };
586 template<std::size_t B,std::size_t N> using mask_full_ops=std::conditional_t<N==1,mask_scalar_ops<mask_word<B>>,
587 std::conditional_t<B*N==64,mask_vector512_ops<B,N>,mask_vector_ops<B,N>>>;
588 template<std::size_t N> inline constexpr bool predicate_shape=
589#if NATIVE_HAS_AVX512F
590 (N==1 || N==2 || N==3 || N==4 || N==8 || N==16
591#if NATIVE_HAS_AVX512BW
592 || N==32 || N==64
593#endif
594 );
595#else
596 false;
597#endif
598 }
599
600 template<class U,std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::mask_shape<sizeof(U),N>
605 struct alignas(typename ::NATIVE_BACKEND_NAMESPACE::mask_full_ops<sizeof(U),N>::native_type) simd<mask_lane<U>,N,Arch> : detail::swizzle_access<mask_lane<U>,N,Arch> {
606 static constexpr isa<> architecture=Arch;
607 template <class X> using rebind = simd<X,N,Arch>;
609 template <std::size_t A = 1>
610 native_nodiscard static native_inline constexpr simd load_memory(mask_lane<U> const * p) noexcept { return load(p); }
612 template <std::size_t A = 1>
613 native_inline constexpr void store_memory(mask_lane<U> * p) const noexcept { store(p); }
614
615 using value_type=mask_lane<U>;
616 using storage_type=U;
617 using ops=::NATIVE_BACKEND_NAMESPACE::mask_full_ops<sizeof(U),N>;
618 using native_type=typename ops::native_type;
619 using mask_type=simd;
620 using mask = mask_type;
621 using predicate_type = predicate<N,Arch>;
622 static constexpr std::size_t lanes=N;
623 static constexpr bool compact=false;
625 native_inline constexpr simd() noexcept : value_(ops::broadcast(false)) {}
627 explicit native_inline constexpr simd(bool value) noexcept : value_(ops::broadcast(value)) {}
629 native_inline constexpr simd(value_type value) noexcept : simd(value.to_bool()) {}
631 template<class... X> requires(N>1 && sizeof...(X)==N) && (std::same_as<X,value_type>&&...)
632 native_inline constexpr simd(X... values) noexcept : simd(std::array<value_type,N>{values...}) {}
634 explicit native_inline constexpr simd(std::array<value_type,N> const & values) noexcept : simd(load(values.data())) {}
635 // Safe native import interprets each whole U-sized lane as nonzero truth.
637 native_nodiscard static native_inline native_const constexpr simd from_native(native_type value) noexcept { return simd(raw{},ops::normalize(value)); }
638 // Caller promises zero/all-ones for EVERY lane. Arbitrary bitselect masks
639 // must use bit_select instead, never this canonical predicate domain.
641 native_nodiscard static native_inline native_const constexpr simd unsafe_from_native(native_type value) noexcept { return simd(raw{},value); }
643 native_nodiscard native_inline native_pure constexpr native_type to_native() const noexcept { return value_; }
645 native_nodiscard static native_inline native_const constexpr simd from_bitset(std::uint64_t value) noexcept { return simd(raw{},ops::from_bits(value)); }
647 native_nodiscard native_inline native_pure constexpr std::uint64_t to_bitset() const noexcept { return ops::bits(value_); }
649 native_nodiscard static native_inline constexpr native_pure simd load(value_type const * p) noexcept {
650 if consteval {
651 std::uint64_t bits=0;
652 for(std::size_t i=0;i<N;++i) bits|=std::uint64_t(p[i].to_bool())<<i;
653 return from_bitset(bits);
654 }
655 native_type value;std::memcpy(&value,static_cast<void const *>(p),sizeof(value));return unsafe_from_native(value);
656 }
658 native_inline constexpr void store(value_type * p) const noexcept {
659 if consteval {
660 auto bits=to_bitset();
661 for(std::size_t i=0;i<N;++i) p[i]=value_type(((bits>>i)&1)!=0);
662 return;
663 }
664 std::memcpy(static_cast<void *>(p),&value_,sizeof(value_));
665 }
667 native_nodiscard friend native_inline native_const constexpr simd operator~(simd a) noexcept { return simd(raw{},ops::bit_not(a.value_)); }
669 native_nodiscard friend native_inline native_const constexpr simd operator!(simd a) noexcept { return ~a; }
671 native_nodiscard friend native_inline native_const constexpr simd operator&(simd a,simd b) noexcept { return simd(raw{},ops::bit_and(a.value_,b.value_)); }
673 native_nodiscard friend native_inline native_const constexpr simd operator|(simd a,simd b) noexcept { return simd(raw{},ops::bit_or(a.value_,b.value_)); }
675 native_nodiscard friend native_inline native_const constexpr simd operator^(simd a,simd b) noexcept { return simd(raw{},ops::bit_xor(a.value_,b.value_)); }
677 native_nodiscard friend native_inline native_const constexpr simd operator==(simd a,simd b) noexcept { return ~(a^b); }
679 native_nodiscard friend native_inline native_const constexpr simd operator!=(simd a,simd b) noexcept { return a^b; }
681 native_inline constexpr simd & operator&=(simd b) noexcept { return *this=*this&b; }
683 native_inline constexpr simd & operator|=(simd b) noexcept { return *this=*this|b; }
685 native_inline constexpr simd & operator^=(simd b) noexcept { return *this=*this^b; }
687 native_nodiscard friend native_inline native_const constexpr bool any(simd a) noexcept { return ops::any(a.value_); }
689 native_nodiscard friend native_inline native_const constexpr bool all(simd a) noexcept { return ops::all(a.value_); }
691 native_nodiscard friend native_inline native_const constexpr bool none(simd a) noexcept { return !any(a); }
693 native_nodiscard friend native_inline native_const constexpr simd select(simd predicate,simd a,simd b) noexcept { return (predicate&a)|(~predicate&b); }
694 private:
695 struct raw {};
696 native_inline constexpr simd(raw,native_type value) noexcept : value_(value) {}
697 native_type value_;
698 };
699
703
704 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N>
705 struct predicate<N,Arch> {
706 static constexpr isa<> architecture=Arch;
707 using ops=::NATIVE_BACKEND_NAMESPACE::mask_compact_ops<N>;
708 using native_type=typename ops::native_type;
709 static constexpr std::size_t lanes=N;
710 static constexpr bool compact=true;
712 native_inline constexpr predicate() noexcept = default;
714 explicit native_inline constexpr predicate(bool value) noexcept : value_(ops::broadcast(value)) {}
716#if NATIVE_HOST_X86
717 native_target("sse2")
718#endif
719 native_nodiscard static native_inline native_const constexpr predicate from_native(native_type value) noexcept { return predicate(raw{},ops::normalize(value)); }
721#if NATIVE_HOST_X86
722 native_target("sse2")
723#endif
724 native_nodiscard static native_inline native_const constexpr predicate unsafe_from_native(native_type value) noexcept { return from_native(value); }
726#if NATIVE_HOST_X86
727 native_target("sse2")
728#endif
729 native_nodiscard native_inline native_pure constexpr native_type to_native() const noexcept { return value_; }
731#if NATIVE_HOST_X86
732 native_target("sse2")
733#endif
734 native_nodiscard static native_inline native_const constexpr predicate from_bitset(std::uint64_t value) noexcept { return from_native(native_type(value)); }
736#if NATIVE_HOST_X86
737 // Compact masks hold scalar bits even when their vector profile is stronger.
738 native_target("sse2")
739#endif
740 native_nodiscard native_inline native_pure constexpr std::uint64_t to_bitset() const noexcept { return value_; }
742 native_nodiscard friend native_inline native_const constexpr predicate operator~(predicate a) noexcept { return predicate(raw{},ops::bit_not(a.value_)); }
744 native_nodiscard friend native_inline native_const constexpr predicate operator!(predicate a) noexcept { return ~a; }
746 native_nodiscard friend native_inline native_const constexpr predicate operator&(predicate a,predicate b) noexcept { return predicate(raw{},ops::bit_and(a.value_,b.value_)); }
748 native_nodiscard friend native_inline native_const constexpr predicate operator|(predicate a,predicate b) noexcept { return predicate(raw{},ops::bit_or(a.value_,b.value_)); }
750 native_nodiscard friend native_inline native_const constexpr predicate operator^(predicate a,predicate b) noexcept { return predicate(raw{},ops::bit_xor(a.value_,b.value_)); }
752 native_nodiscard friend native_inline native_const constexpr predicate operator==(predicate a,predicate b) noexcept { return ~(a^b); }
756 native_inline constexpr predicate & operator&=(predicate b) noexcept { return *this=*this&b; }
758 native_inline constexpr predicate & operator|=(predicate b) noexcept { return *this=*this|b; }
760 native_inline constexpr predicate & operator^=(predicate b) noexcept { return *this=*this^b; }
762 native_nodiscard friend native_inline native_const constexpr bool any(predicate a) noexcept { return a.value_!=0; }
764 native_nodiscard friend native_inline native_const constexpr bool all(predicate a) noexcept { return ops::all(a.value_); }
766 native_nodiscard friend native_inline native_const constexpr bool none(predicate a) noexcept { return !any(a); }
768 native_nodiscard friend native_inline native_const constexpr predicate select(predicate p,predicate a,predicate b) noexcept { return (p&a)|(~p&b); }
769 private:
770 struct raw {};
771#if NATIVE_HOST_X86
772 native_target("sse2")
773#endif
774 native_inline constexpr predicate(raw,native_type value) noexcept : value_(value) {}
775 native_type value_=0;
776 };
777
778 namespace detail::NATIVE_BACKEND {
779 template<class T> using mask_lane_for=mask_lane<mask_word<mask_lane_bytes<T>>>;
780 template<class T,std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) using comparison_mask=std::conditional_t<mask_compact<mask_lane_bytes<T>,N>,
781 predicate<N,Arch>,simd<mask_lane_for<T>,N,Arch>>;
782 }
783 namespace detail::NATIVE_BACKEND {
784 template<::native::isa<> Arch, class T,std::size_t N,class U,std::size_t A=1,simd_access Access=simd_access::ordinary>
785 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> && requires { typename simd<T,N,Arch>::native_type; }
786 native_nodiscard native_inline constexpr native_pure simd<T,N,Arch> load_simd(U const * p,simd_memory<A,Access> = {}) noexcept { return simd<T,N,Arch>::load(p); }
787 template<class U,class T,std::size_t N,std::size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
788 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> && requires { typename simd<T,N,Arch>::native_type; }
789 native_inline constexpr void store_simd(U * p,simd<T,N,Arch> value,simd_memory<A,Access> = {}) noexcept { value.store(p); }
790 template<::native::isa<> Arch, class T,std::size_t N,class U,std::size_t A=1,simd_access Access=simd_access::ordinary>
791 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> && requires { typename simd<T,N,Arch>::native_type; }
792 native_nodiscard native_inline constexpr native_pure simd<T,N,Arch> load_simd_partial(U const * p,std::size_t count,T fill=T{},simd_memory<A,Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") {
793 std::array<T,N> a;a.fill(fill);for(std::size_t i=0;i<count;++i)a[i]=p[i];return simd<T,N,Arch>::load(a.data());
794 }
795 template<class U,class T,std::size_t N,std::size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
796 requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && std::same_as<T,U> && requires { typename simd<T,N,Arch>::native_type; }
797 native_inline constexpr void store_simd_partial(U * p,simd<T,N,Arch> value,std::size_t count,simd_memory<A,Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") {
798 std::array<T,N> a;value.store(a.data());for(std::size_t i=0;i<count;++i)p[i]=a[i];
799 }
800 }
803 template<class U,class T,std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<U> && simd_mask_element<T> &&
804 requires { typename simd<U,N,Arch>::native_type; typename simd<T,N,Arch>::native_type; }
806 if constexpr(sizeof(U)==sizeof(T)) return simd<U,N,Arch>::unsafe_from_native(value.to_native());
807 else return simd<U,N,Arch>::from_bitset(value.to_bitset());
808 }
809
811 template<class T,std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> && requires { typename simd<T,N,Arch>::native_type; }
813 if consteval { return predicate<N,Arch>::from_bitset(value.to_bitset()); }
814#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
815 if constexpr(sizeof(T)*N==16 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<sizeof(T),N>) {
816 auto x=value.to_native();auto z=_mm_setzero_si128();
817 if constexpr(sizeof(T)==1) return predicate<N,Arch>::from_native(_mm_cmpneq_epi8_mask(x,z));
818 else if constexpr(sizeof(T)==2) return predicate<N,Arch>::from_native(_mm_cmpneq_epi16_mask(x,z));
819 else if constexpr(sizeof(T)==4) return predicate<N,Arch>::from_native(_mm_cmpneq_epi32_mask(x,z));
820 else return predicate<N,Arch>::from_native(_mm_cmpneq_epi64_mask(x,z));
821 } else if constexpr(sizeof(T)*N==32 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<sizeof(T),N>) {
822 auto x=value.to_native();auto z=_mm256_setzero_si256();
823 if constexpr(sizeof(T)==1) return predicate<N,Arch>::from_native(_mm256_cmpneq_epi8_mask(x,z));
824 else if constexpr(sizeof(T)==2) return predicate<N,Arch>::from_native(_mm256_cmpneq_epi16_mask(x,z));
825 else if constexpr(sizeof(T)==4) return predicate<N,Arch>::from_native(_mm256_cmpneq_epi32_mask(x,z));
826 else return predicate<N,Arch>::from_native(_mm256_cmpneq_epi64_mask(x,z));
827 } else
828#endif
829 return predicate<N,Arch>::from_bitset(value.to_bitset());
830 }
831
833 template<class T,std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> && requires { typename simd<T,N,Arch>::native_type; }
835 if consteval { return simd<T,N,Arch>::from_bitset(value.to_bitset()); }
836#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
837 if constexpr(sizeof(T)*N==16 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<sizeof(T),N>) {
838 auto k=value.to_native();
839 if constexpr(sizeof(T)==1) return simd<T,N,Arch>::unsafe_from_native(_mm_maskz_set1_epi8(k,-1));
840 else if constexpr(sizeof(T)==2) return simd<T,N,Arch>::unsafe_from_native(_mm_maskz_set1_epi16(k,-1));
841 else if constexpr(sizeof(T)==4) return simd<T,N,Arch>::unsafe_from_native(_mm_maskz_set1_epi32(k,-1));
842 else return simd<T,N,Arch>::unsafe_from_native(_mm_maskz_set1_epi64(k,-1));
843 } else if constexpr(sizeof(T)*N==32 && ::NATIVE_BACKEND_NAMESPACE::mask_compact<sizeof(T),N>) {
844 auto k=value.to_native();
845 if constexpr(sizeof(T)==1) return simd<T,N,Arch>::unsafe_from_native(_mm256_maskz_set1_epi8(k,-1));
846 else if constexpr(sizeof(T)==2) return simd<T,N,Arch>::unsafe_from_native(_mm256_maskz_set1_epi16(k,-1));
847 else if constexpr(sizeof(T)==4) return simd<T,N,Arch>::unsafe_from_native(_mm256_maskz_set1_epi32(k,-1));
848 else return simd<T,N,Arch>::unsafe_from_native(_mm256_maskz_set1_epi64(k,-1));
849 } else
850#endif
851 return simd<T,N,Arch>::from_bitset(value.to_bitset());
852 }
853}
854
855
856namespace native {
857 namespace detail::NATIVE_BACKEND {
858 template<std::size_t N> native_nodiscard native_inline native_const constexpr auto bool_ones() noexcept {
859 if consteval {
860 using V=typename mask_full_ops<1,N>::native_type;
861 std::array<std::uint8_t,N> lanes{}; lanes.fill(1);
862 return __builtin_bit_cast(V,lanes);
863 }
864 if constexpr(N==1) return std::uint8_t(1);
865#if NATIVE_HAS_AVX2
866 else if constexpr(N==16) return _mm_set1_epi8(1);
867 else if constexpr(N==32) return _mm256_set1_epi8(1);
868#endif
869#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512BW
870 else if constexpr(N==64) return _mm512_set1_epi8(1);
871#endif
872#if NATIVE_HAS_ARM_NEON
873 else return vdupq_n_u8(1);
874#endif
875 }
876 }
880 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::mask_shape<1,N>
881 struct simd<bool,N,Arch> : detail::swizzle_access<bool,N,Arch> {
882 static constexpr isa<> architecture=Arch;
883 template <class X> using rebind = simd<X,N,Arch>;
885 template <std::size_t A = 1>
886 native_nodiscard static native_inline constexpr simd load_memory(bool const * p) noexcept { return load(p); }
888 template <std::size_t A = 1>
889 native_inline constexpr void store_memory(bool * p) const noexcept { store(p); }
890
891 using value_type=bool;
892 using storage_type=std::uint8_t;
893 using ops=::NATIVE_BACKEND_NAMESPACE::mask_full_ops<1,N>;
894 using native_type=typename ops::native_type;
895 using vector_mask_type=simd<mask8,N,Arch>;
896 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<bool,N,Arch>;
897 using mask = mask_type;
898 using predicate_type = predicate<N,Arch>;
899 static constexpr std::size_t lanes=N;
901 native_inline constexpr simd() noexcept : value_(ops::broadcast(false)) {}
903 explicit native_inline constexpr simd(bool value) noexcept : value_(value?::NATIVE_BACKEND_NAMESPACE::bool_ones<N>():ops::broadcast(false)) {}
905 template<class... X> requires(N>1 && sizeof...(X)==N) && (std::same_as<X,bool>&&...)
906 native_inline constexpr simd(X... value) noexcept : simd(std::array<bool,N>{value...}) {}
908 explicit native_inline constexpr simd(std::array<bool,N> const & value) noexcept : simd(load(value.data())) {}
910 native_nodiscard static native_inline native_const constexpr simd from_native(native_type value) noexcept {
911 return simd(raw{},ops::bit_and(ops::normalize(value),::NATIVE_BACKEND_NAMESPACE::bool_ones<N>()));
912 }
913 // Caller promises EVERY byte is 0 or 1, never an all-ones mask byte.
915 native_nodiscard static native_inline native_const constexpr simd unsafe_from_native(native_type value) noexcept { return simd(raw{},value); }
917 native_nodiscard native_inline native_pure constexpr native_type to_native() const noexcept { return value_; }
919 native_nodiscard static native_inline constexpr native_pure simd load(bool const * p) noexcept {
920 std::array<std::uint8_t,N> bytes{};
921 for(std::size_t i=0;i<N;++i) bytes[i]=p[i]?1:0;
922 if consteval { return unsafe_from_native(__builtin_bit_cast(native_type,bytes)); }
923 native_type value;std::memcpy(&value,bytes.data(),sizeof(value));return unsafe_from_native(value);
924 }
925
926 native_inline constexpr void store(bool * p) const noexcept {
927 std::array<std::uint8_t,N> bytes{};
928 if consteval { bytes=__builtin_bit_cast(decltype(bytes),value_); }
929 else { std::memcpy(bytes.data(),&value_,sizeof(value_)); }
930 for(std::size_t i=0;i<N;++i) p[i]=bytes[i]!=0;
931 }
932 // The caller supplies 0 <= count <= N. Zero touches no pointer, even null.
934 native_nodiscard static native_inline constexpr native_pure simd load_partial(bool const * p,std::size_t count,bool fill=false) noexcept native_diagnose_if(count > simd::lanes,"partial SIMD count exceeds the lane count") {
935 std::array<bool,N> values;values.fill(fill);
936 for(std::size_t i=0;i<count;++i) values[i]=p[i];
937 return load(values.data());
938 }
939
940 native_inline constexpr void store_partial(bool * p,std::size_t count) const noexcept native_diagnose_if(count > simd::lanes,"partial SIMD count exceeds the lane count") {
941 std::array<bool,N> values;store(values.data());
942 for(std::size_t i=0;i<count;++i) p[i]=values[i];
943 }
944
945 native_nodiscard friend native_inline native_const constexpr simd operator!(simd a) noexcept { return simd(raw{},ops::bit_xor(a.value_,::NATIVE_BACKEND_NAMESPACE::bool_ones<N>())); }
947 native_nodiscard friend native_inline native_const constexpr simd operator~(simd a) noexcept { return !a; }
949 native_nodiscard friend native_inline native_const constexpr simd operator&(simd a,simd b) noexcept { return simd(raw{},ops::bit_and(a.value_,b.value_)); }
951 native_nodiscard friend native_inline native_const constexpr simd operator|(simd a,simd b) noexcept { return simd(raw{},ops::bit_or(a.value_,b.value_)); }
953 native_nodiscard friend native_inline native_const constexpr simd operator^(simd a,simd b) noexcept { return simd(raw{},ops::bit_xor(a.value_,b.value_)); }
955 native_nodiscard friend native_inline native_const constexpr mask_type operator!=(simd a,simd b) noexcept {
956 auto x=ops::bit_xor(a.value_,b.value_);
957 if consteval { return mask_type::from_bitset(vector_mask_type::from_native(x).to_bitset()); }
958#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512BW
959 if constexpr(N==64) return mask_type::from_native(_mm512_cmpneq_epi8_mask(x,_mm512_setzero_si512()));
960#if NATIVE_HAS_AVX512VL
961 else if constexpr(N==32) return mask_type::from_native(_mm256_cmpneq_epi8_mask(x,_mm256_setzero_si256()));
962 else if constexpr(N==16) return mask_type::from_native(_mm_cmpneq_epi8_mask(x,_mm_setzero_si128()));
963#endif
964 else
965#endif
966 return vector_mask_type::from_native(x);
967 }
968
969 native_nodiscard friend native_inline native_const constexpr mask_type operator==(simd a,simd b) noexcept { return ~(a!=b); }
971 native_inline constexpr simd & operator&=(simd b) noexcept { return *this=*this&b; }
973 native_inline constexpr simd & operator|=(simd b) noexcept { return *this=*this|b; }
975 native_inline constexpr simd & operator^=(simd b) noexcept { return *this=*this^b; }
977 native_nodiscard friend native_inline native_const constexpr bool any(simd a) noexcept { return ops::any(a.value_); }
979 native_nodiscard friend native_inline native_const constexpr bool all(simd a) noexcept { return none(!a); }
981 native_nodiscard friend native_inline native_const constexpr bool none(simd a) noexcept { return !any(a); }
983 template<class M> requires(std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
984 native_nodiscard friend native_inline native_const constexpr simd select(M m,simd a,simd b) noexcept {
985 if consteval {
986 auto mask=vector_mask_type::from_bitset(m.to_bitset()).to_native();
987 return simd(raw{},ops::bit_or(ops::bit_and(mask,a.value_),ops::bit_and(ops::bit_not(mask),b.value_)));
988 }
989#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512BW
990 if constexpr(M::compact) {
991 if constexpr(N==64) return simd(raw{},_mm512_mask_blend_epi8(m.to_native(),b.value_,a.value_));
992#if NATIVE_HAS_AVX512VL
993 else if constexpr(N==32) return simd(raw{},_mm256_mask_blend_epi8(m.to_native(),b.value_,a.value_));
994 else return simd(raw{},_mm_mask_blend_epi8(m.to_native(),b.value_,a.value_));
995#endif
996 } else
997#endif
998 return simd(raw{},ops::bit_or(ops::bit_and(m.to_native(),a.value_),ops::bit_and(ops::bit_not(m.to_native()),b.value_)));
999 }
1000 private:
1001 struct raw {};
1002 native_inline constexpr simd(raw,native_type value) noexcept : value_(value) {}
1003 native_type value_;
1004 };
1005
1006 namespace detail::NATIVE_BACKEND {
1007 template<::native::isa<> Arch, class T,std::size_t N,class U,std::size_t A=1,simd_access Access=simd_access::ordinary>
1008 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> && requires { typename simd<T,N,Arch>::native_type; }
1009 native_nodiscard native_inline constexpr native_pure simd<T,N,Arch> load_simd(U const * p,simd_memory<A,Access> = {}) noexcept { return simd<T,N,Arch>::load(p); }
1010 template<class U,class T,std::size_t N,std::size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
1011 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> && requires { typename simd<T,N,Arch>::native_type; }
1012 native_inline constexpr void store_simd(U * p,simd<T,N,Arch> value,simd_memory<A,Access> = {}) noexcept { value.store(p); }
1013 template<::native::isa<> Arch, class T,std::size_t N,class U,std::size_t A=1,simd_access Access=simd_access::ordinary>
1014 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> && requires { typename simd<T,N,Arch>::native_type; }
1015 native_nodiscard native_inline constexpr native_pure simd<T,N,Arch> load_simd_partial(U const * p,std::size_t count,T fill=false,simd_memory<A,Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") {
1016 return simd<T,N,Arch>::load_partial(p,count,fill);
1017 }
1018 template<class U,class T,std::size_t N,std::size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
1019 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,bool> && std::same_as<U,bool> && requires { typename simd<T,N,Arch>::native_type; }
1020 native_inline constexpr void store_simd_partial(U * p,simd<T,N,Arch> value,std::size_t count,simd_memory<A,Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") { value.store_partial(p,count); }
1021 }
1023 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && requires { typename simd<bool,N,Arch>::native_type; }
1027
1028 template<class T,std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && simd_mask_element<T> && requires { typename simd<bool,N,Arch>::native_type; typename simd<T,N,Arch>::native_type; }
1030 auto byte_mask=mask_cast<mask8>(value);
1031 return simd<bool,N,Arch>::unsafe_from_native(simd<bool,N,Arch>::ops::bit_and(byte_mask.to_native(),::NATIVE_BACKEND_NAMESPACE::bool_ones<N>()));
1032 }
1033
1034 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> && requires { typename simd<bool,N,Arch>::native_type; }
1037 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::predicate_shape<N> && requires { typename simd<bool,N,Arch>::native_type; }
1039 auto result=value!=simd<bool,N,Arch>(false);
1040 if constexpr(decltype(result)::compact) return result;
1041 else return to_predicate(result);
1042 }
1043}
1044
1045
1046namespace NATIVE_BACKEND_NAMESPACE {
1047 template <simd_integer_element T> native_nodiscard native_inline constexpr auto integer_word(T value) noexcept {
1048 return std::bit_cast<std::make_unsigned_t<T>>(value);
1049 }
1050 template <simd_integer_element T, class U>
1051 native_nodiscard native_inline constexpr T integer_wrap(U value) noexcept {
1052 return std::bit_cast<T>(static_cast<std::make_unsigned_t<T>>(value));
1053 }
1054 template <simd_integer_element T>
1055 using integer_work_word = std::conditional_t<(sizeof(T) < 4), uint32_t, std::make_unsigned_t<T>>;
1056 template <simd_integer_element T>
1057 native_nodiscard native_inline constexpr T integer_scalar_add(T a, T b) noexcept {
1058 using U = integer_work_word<T>;
1059 return integer_wrap<T>(U(integer_word(a)) + U(integer_word(b)));
1060 }
1061 template <simd_integer_element T>
1062 native_nodiscard native_inline constexpr T integer_scalar_sub(T a, T b) noexcept {
1063 using U = integer_work_word<T>;
1064 return integer_wrap<T>(U(integer_word(a)) - U(integer_word(b)));
1065 }
1066 template <simd_integer_element T>
1067 native_nodiscard native_inline constexpr T integer_scalar_mul(T a, T b) noexcept {
1068 using U = integer_work_word<T>;
1069 return integer_wrap<T>(U(integer_word(a)) * U(integer_word(b)));
1070 }
1071
1072#if NATIVE_HAS_AVX2
1073 template <simd_integer_element T>
1074 native_nodiscard native_inline native_const __m128i integer_broadcast_16(T x) noexcept {
1075 if constexpr (sizeof(T) == 1)
1076 return _mm_set1_epi8(std::bit_cast<int8_t>(x));
1077 else if constexpr (sizeof(T) == 2)
1078 return _mm_set1_epi16(std::bit_cast<int16_t>(x));
1079 else if constexpr (sizeof(T) == 4)
1080 return _mm_set1_epi32(std::bit_cast<int32_t>(x));
1081 else if constexpr (sizeof(T) == 8)
1082 return _mm_set1_epi64x(std::bit_cast<int64_t>(x));
1083 }
1084 template <simd_integer_element T>
1085 native_nodiscard native_inline native_const __m128i integer_add(__m128i a, __m128i b) noexcept {
1086 if constexpr (sizeof(T) == 1)
1087 return _mm_add_epi8(a, b);
1088 else if constexpr (sizeof(T) == 2)
1089 return _mm_add_epi16(a, b);
1090 else if constexpr (sizeof(T) == 4)
1091 return _mm_add_epi32(a, b);
1092 else if constexpr (sizeof(T) == 8)
1093 return _mm_add_epi64(a, b);
1094 }
1095 template <simd_integer_element T>
1096 native_nodiscard native_inline native_const __m128i integer_sub(__m128i a, __m128i b) noexcept {
1097 if constexpr (sizeof(T) == 1)
1098 return _mm_sub_epi8(a, b);
1099 else if constexpr (sizeof(T) == 2)
1100 return _mm_sub_epi16(a, b);
1101 else if constexpr (sizeof(T) == 4)
1102 return _mm_sub_epi32(a, b);
1103 else if constexpr (sizeof(T) == 8)
1104 return _mm_sub_epi64(a, b);
1105 }
1106 template <simd_integer_element T>
1107 native_nodiscard native_inline native_const __m128i integer_mul(__m128i a, __m128i b) noexcept {
1108 if constexpr (sizeof(T) == 1) {
1109 // Independent low-byte products inside each word, with cross-byte bits removed.
1110 auto low = _mm_mullo_epi16(a, b);
1111 auto high = _mm_mullo_epi16(_mm_srli_epi16(a, 8), _mm_srli_epi16(b, 8));
1112 return _mm_or_si128(_mm_and_si128(low, _mm_set1_epi16(255)), _mm_slli_epi16(high, 8));
1113 } else if constexpr (sizeof(T) == 2)
1114 return _mm_mullo_epi16(a, b);
1115 if constexpr (sizeof(T) == 4)
1116 return _mm_mullo_epi32(a, b);
1117 else if constexpr (sizeof(T) == 8) {
1118#if NATIVE_HAS_AVX512DQ && NATIVE_HAS_AVX512VL
1119 return _mm_mullo_epi64(a, b);
1120#else
1121 // Low 64 bits: low32*low32 plus both cross products shifted by 32.
1122 auto cross =
1123 _mm_add_epi64(_mm_mul_epu32(a, _mm_srli_epi64(b, 32)), _mm_mul_epu32(_mm_srli_epi64(a, 32), b));
1124 return _mm_add_epi64(_mm_mul_epu32(a, b), _mm_slli_epi64(cross, 32));
1125#endif
1126 }
1127 }
1128 template <simd_integer_element T, unsigned S>
1129 requires(S < sizeof(T) * 8)
1130 native_nodiscard native_inline native_const __m128i integer_left(__m128i a) noexcept {
1131 if constexpr (S == 0)
1132 return a;
1133 else {
1134 if constexpr (sizeof(T) == 1)
1135 return _mm_and_si128(_mm_slli_epi16(a, S),
1136 _mm_set1_epi8(std::bit_cast<int8_t>(uint8_t((255u << S) & 255u))));
1137 else if constexpr (sizeof(T) == 2)
1138 return _mm_slli_epi16(a, S);
1139 if constexpr (sizeof(T) == 4)
1140 return _mm_slli_epi32(a, S);
1141 else if constexpr (sizeof(T) == 8)
1142 return _mm_slli_epi64(a, S);
1143 }
1144 }
1145 template <simd_integer_element T, unsigned S>
1146 requires(S < sizeof(T) * 8)
1147 native_nodiscard native_inline native_const __m128i integer_right(__m128i a) noexcept {
1148 if constexpr (S == 0)
1149 return a;
1150 else {
1151 if constexpr (sizeof(T) == 1) {
1152 auto low = _mm_and_si128(_mm_srli_epi16(a, S), _mm_set1_epi8(int8_t(255u >> S)));
1153 if constexpr (std::is_unsigned_v<T>)
1154 return low;
1155 else {
1156 auto negative = _mm_cmpgt_epi8(_mm_setzero_si128(), a);
1157 return _mm_or_si128(
1158 low, _mm_and_si128(negative, _mm_set1_epi8(std::bit_cast<int8_t>(uint8_t(255u ^ (255u >> S))))));
1159 }
1160 } else if constexpr (sizeof(T) == 2) {
1161 if constexpr (std::is_signed_v<T>)
1162 return _mm_srai_epi16(a, S);
1163 else
1164 return _mm_srli_epi16(a, S);
1165 }
1166 if constexpr (sizeof(T) == 4) {
1167 if constexpr (std::is_signed_v<T>)
1168 return _mm_srai_epi32(a, S);
1169 else
1170 return _mm_srli_epi32(a, S);
1171 } else if constexpr (sizeof(T) == 8) {
1172 if constexpr (std::is_unsigned_v<T>)
1173 return _mm_srli_epi64(a, S);
1174 else {
1175#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
1176 return _mm_srai_epi64(a, S);
1177#else
1178 auto negative = _mm_cmpgt_epi64(_mm_setzero_si128(), a);
1179 return _mm_or_si128(
1180 _mm_srli_epi64(a, S),
1181 _mm_and_si128(negative, _mm_set1_epi64x(std::bit_cast<int64_t>(~uint64_t(0) << (64 - S)))));
1182#endif
1183 }
1184 }
1185 }
1186 }
1187#endif
1188
1189#if NATIVE_HAS_AVX2
1190 template <simd_integer_element T>
1191 native_nodiscard native_inline native_const __m256i integer_broadcast_32(T x) noexcept {
1192 if constexpr (sizeof(T) == 1)
1193 return _mm256_set1_epi8(std::bit_cast<int8_t>(x));
1194 else if constexpr (sizeof(T) == 2)
1195 return _mm256_set1_epi16(std::bit_cast<int16_t>(x));
1196 else if constexpr (sizeof(T) == 4)
1197 return _mm256_set1_epi32(std::bit_cast<int32_t>(x));
1198 else if constexpr (sizeof(T) == 8)
1199 return _mm256_set1_epi64x(std::bit_cast<int64_t>(x));
1200 }
1201 template <simd_integer_element T>
1202 native_nodiscard native_inline native_const __m256i integer_add(__m256i a, __m256i b) noexcept {
1203 if constexpr (sizeof(T) == 1)
1204 return _mm256_add_epi8(a, b);
1205 else if constexpr (sizeof(T) == 2)
1206 return _mm256_add_epi16(a, b);
1207 else if constexpr (sizeof(T) == 4)
1208 return _mm256_add_epi32(a, b);
1209 else if constexpr (sizeof(T) == 8)
1210 return _mm256_add_epi64(a, b);
1211 }
1212 template <simd_integer_element T>
1213 native_nodiscard native_inline native_const __m256i integer_sub(__m256i a, __m256i b) noexcept {
1214 if constexpr (sizeof(T) == 1)
1215 return _mm256_sub_epi8(a, b);
1216 else if constexpr (sizeof(T) == 2)
1217 return _mm256_sub_epi16(a, b);
1218 else if constexpr (sizeof(T) == 4)
1219 return _mm256_sub_epi32(a, b);
1220 else if constexpr (sizeof(T) == 8)
1221 return _mm256_sub_epi64(a, b);
1222 }
1223 template <simd_integer_element T>
1224 native_nodiscard native_inline native_const __m256i integer_mul(__m256i a, __m256i b) noexcept {
1225 if constexpr (sizeof(T) == 1) {
1226 // Independent low-byte products inside each word, with cross-byte bits removed.
1227 auto low = _mm256_mullo_epi16(a, b);
1228 auto high = _mm256_mullo_epi16(_mm256_srli_epi16(a, 8), _mm256_srli_epi16(b, 8));
1229 return _mm256_or_si256(_mm256_and_si256(low, _mm256_set1_epi16(255)), _mm256_slli_epi16(high, 8));
1230 } else if constexpr (sizeof(T) == 2)
1231 return _mm256_mullo_epi16(a, b);
1232 if constexpr (sizeof(T) == 4)
1233 return _mm256_mullo_epi32(a, b);
1234 else if constexpr (sizeof(T) == 8) {
1235#if NATIVE_HAS_AVX512DQ && NATIVE_HAS_AVX512VL
1236 return _mm256_mullo_epi64(a, b);
1237#else
1238 // Low 64 bits: low32*low32 plus both cross products shifted by 32.
1239 auto cross = _mm256_add_epi64(_mm256_mul_epu32(a, _mm256_srli_epi64(b, 32)),
1240 _mm256_mul_epu32(_mm256_srli_epi64(a, 32), b));
1241 return _mm256_add_epi64(_mm256_mul_epu32(a, b), _mm256_slli_epi64(cross, 32));
1242#endif
1243 }
1244 }
1245 template <simd_integer_element T, unsigned S>
1246 requires(S < sizeof(T) * 8)
1247 native_nodiscard native_inline native_const __m256i integer_left(__m256i a) noexcept {
1248 if constexpr (S == 0)
1249 return a;
1250 else {
1251 if constexpr (sizeof(T) == 1)
1252 return _mm256_and_si256(_mm256_slli_epi16(a, S),
1253 _mm256_set1_epi8(std::bit_cast<int8_t>(uint8_t((255u << S) & 255u))));
1254 else if constexpr (sizeof(T) == 2)
1255 return _mm256_slli_epi16(a, S);
1256 if constexpr (sizeof(T) == 4)
1257 return _mm256_slli_epi32(a, S);
1258 else if constexpr (sizeof(T) == 8)
1259 return _mm256_slli_epi64(a, S);
1260 }
1261 }
1262 native_nodiscard native_inline native_const __m256i integer_shift_left_variable(__m256i a,__m256i counts) noexcept {
1263 return _mm256_sllv_epi32(a,counts);
1264 }
1265 template <simd_integer_element T, unsigned S>
1266 requires(S < sizeof(T) * 8)
1267 native_nodiscard native_inline native_const __m256i integer_right(__m256i a) noexcept {
1268 if constexpr (S == 0)
1269 return a;
1270 else {
1271 if constexpr (sizeof(T) == 1) {
1272 auto low = _mm256_and_si256(_mm256_srli_epi16(a, S), _mm256_set1_epi8(int8_t(255u >> S)));
1273 if constexpr (std::is_unsigned_v<T>)
1274 return low;
1275 else {
1276 auto negative = _mm256_cmpgt_epi8(_mm256_setzero_si256(), a);
1277 return _mm256_or_si256(
1278 low,
1279 _mm256_and_si256(negative, _mm256_set1_epi8(std::bit_cast<int8_t>(uint8_t(255u ^ (255u >> S))))));
1280 }
1281 } else if constexpr (sizeof(T) == 2) {
1282 if constexpr (std::is_signed_v<T>)
1283 return _mm256_srai_epi16(a, S);
1284 else
1285 return _mm256_srli_epi16(a, S);
1286 }
1287 if constexpr (sizeof(T) == 4) {
1288 if constexpr (std::is_signed_v<T>)
1289 return _mm256_srai_epi32(a, S);
1290 else
1291 return _mm256_srli_epi32(a, S);
1292 } else if constexpr (sizeof(T) == 8) {
1293 if constexpr (std::is_unsigned_v<T>)
1294 return _mm256_srli_epi64(a, S);
1295 else {
1296#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512VL
1297 return _mm256_srai_epi64(a, S);
1298#else
1299 auto negative = _mm256_cmpgt_epi64(_mm256_setzero_si256(), a);
1300 return _mm256_or_si256(
1301 _mm256_srli_epi64(a, S),
1302 _mm256_and_si256(negative, _mm256_set1_epi64x(std::bit_cast<int64_t>(~uint64_t(0) << (64 - S)))));
1303#endif
1304 }
1305 }
1306 }
1307 }
1308#endif
1309
1310#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
1311 template <simd_integer_element T>
1312 native_nodiscard native_inline native_const __m512i integer_broadcast_64(T x) noexcept {
1313#if NATIVE_HAS_AVX512BW
1314 if constexpr (sizeof(T) == 1)
1315 return _mm512_set1_epi8(std::bit_cast<int8_t>(x));
1316 else if constexpr (sizeof(T) == 2)
1317 return _mm512_set1_epi16(std::bit_cast<int16_t>(x));
1318#endif
1319 if constexpr (sizeof(T) == 4)
1320 return _mm512_set1_epi32(std::bit_cast<int32_t>(x));
1321 else if constexpr (sizeof(T) == 8)
1322 return _mm512_set1_epi64(std::bit_cast<int64_t>(x));
1323 }
1324 template <simd_integer_element T>
1325 native_nodiscard native_inline native_const __m512i integer_add(__m512i a, __m512i b) noexcept {
1326#if NATIVE_HAS_AVX512BW
1327 if constexpr (sizeof(T) == 1)
1328 return _mm512_add_epi8(a, b);
1329 else if constexpr (sizeof(T) == 2)
1330 return _mm512_add_epi16(a, b);
1331#endif
1332 if constexpr (sizeof(T) == 4)
1333 return _mm512_add_epi32(a, b);
1334 else if constexpr (sizeof(T) == 8)
1335 return _mm512_add_epi64(a, b);
1336 }
1337 template <simd_integer_element T>
1338 native_nodiscard native_inline native_const __m512i integer_sub(__m512i a, __m512i b) noexcept {
1339#if NATIVE_HAS_AVX512BW
1340 if constexpr (sizeof(T) == 1)
1341 return _mm512_sub_epi8(a, b);
1342 else if constexpr (sizeof(T) == 2)
1343 return _mm512_sub_epi16(a, b);
1344#endif
1345 if constexpr (sizeof(T) == 4)
1346 return _mm512_sub_epi32(a, b);
1347 else if constexpr (sizeof(T) == 8)
1348 return _mm512_sub_epi64(a, b);
1349 }
1350 template <simd_integer_element T>
1351 native_nodiscard native_inline native_const __m512i integer_mul(__m512i a, __m512i b) noexcept {
1352#if NATIVE_HAS_AVX512BW
1353 if constexpr (sizeof(T) == 1) {
1354 // Independent low-byte products inside each word, with cross-byte bits removed.
1355 auto low = _mm512_mullo_epi16(a, b);
1356 auto high = _mm512_mullo_epi16(_mm512_srli_epi16(a, 8), _mm512_srli_epi16(b, 8));
1357 return _mm512_or_si512(_mm512_and_si512(low, _mm512_set1_epi16(255)), _mm512_slli_epi16(high, 8));
1358 } else if constexpr (sizeof(T) == 2)
1359 return _mm512_mullo_epi16(a, b);
1360#endif
1361 if constexpr (sizeof(T) == 4)
1362 return _mm512_mullo_epi32(a, b);
1363 else if constexpr (sizeof(T) == 8) {
1364 return _mm512_mullo_epi64(a, b);
1365 }
1366 }
1367 template <simd_integer_element T, unsigned S>
1368 requires(S < sizeof(T) * 8)
1369 native_nodiscard native_inline native_const __m512i integer_left(__m512i a) noexcept {
1370 if constexpr (S == 0)
1371 return a;
1372 else {
1373#if NATIVE_HAS_AVX512BW
1374 if constexpr (sizeof(T) == 1)
1375 return _mm512_and_si512(_mm512_slli_epi16(a, S),
1376 _mm512_set1_epi8(std::bit_cast<int8_t>(uint8_t((255u << S) & 255u))));
1377 else if constexpr (sizeof(T) == 2)
1378 return _mm512_slli_epi16(a, S);
1379#endif
1380 if constexpr (sizeof(T) == 4)
1381 return _mm512_slli_epi32(a, S);
1382 else if constexpr (sizeof(T) == 8)
1383 return _mm512_slli_epi64(a, S);
1384 }
1385 }
1386 native_nodiscard native_inline native_const __m512i integer_shift_left_variable(__m512i a,__m512i counts) noexcept {
1387 return _mm512_sllv_epi32(a,counts);
1388 }
1389 template <simd_integer_element T, unsigned S>
1390 requires(S < sizeof(T) * 8)
1391 native_nodiscard native_inline native_const __m512i integer_right(__m512i a) noexcept {
1392 if constexpr (S == 0)
1393 return a;
1394 else {
1395#if NATIVE_HAS_AVX512BW
1396 if constexpr (sizeof(T) == 1) {
1397 auto low = _mm512_and_si512(_mm512_srli_epi16(a, S), _mm512_set1_epi8(int8_t(255u >> S)));
1398 if constexpr (std::is_unsigned_v<T>)
1399 return low;
1400 else {
1401 auto negative = _mm512_movm_epi8(_mm512_cmp_epi8_mask(a, _mm512_setzero_si512(), _MM_CMPINT_LT));
1402 return _mm512_or_si512(
1403 low,
1404 _mm512_and_si512(negative, _mm512_set1_epi8(std::bit_cast<int8_t>(uint8_t(255u ^ (255u >> S))))));
1405 }
1406 } else if constexpr (sizeof(T) == 2) {
1407 if constexpr (std::is_signed_v<T>)
1408 return _mm512_srai_epi16(a, S);
1409 else
1410 return _mm512_srli_epi16(a, S);
1411 }
1412#endif
1413 if constexpr (sizeof(T) == 4) {
1414 if constexpr (std::is_signed_v<T>)
1415 return _mm512_srai_epi32(a, S);
1416 else
1417 return _mm512_srli_epi32(a, S);
1418 } else if constexpr (sizeof(T) == 8) {
1419 if constexpr (std::is_unsigned_v<T>)
1420 return _mm512_srli_epi64(a, S);
1421 else {
1422 return _mm512_srai_epi64(a, S);
1423 }
1424 }
1425 }
1426 }
1427#endif
1428#if NATIVE_HAS_AVX2
1429 native_nodiscard native_inline native_const __m128i integer_and(__m128i a, __m128i b) noexcept {
1430 return _mm_and_si128(a, b);
1431 }
1432 native_nodiscard native_inline native_const __m128i integer_or(__m128i a, __m128i b) noexcept {
1433 return _mm_or_si128(a, b);
1434 }
1435 native_nodiscard native_inline native_const __m128i integer_xor(__m128i a, __m128i b) noexcept {
1436 return _mm_xor_si128(a, b);
1437 }
1438 template <simd_integer_element T, bool Greater>
1439 native_nodiscard native_inline native_const auto integer_compare(__m128i a, __m128i b) noexcept {
1440 if constexpr (mask_compact<sizeof(T), 16 / sizeof(T)>) {
1441 if constexpr (sizeof(T) == 1) {
1442 if constexpr (Greater) {
1443 if constexpr (std::is_signed_v<T>)
1444 return _mm_cmp_epi8_mask(a, b, _MM_CMPINT_GT);
1445 else
1446 return _mm_cmp_epu8_mask(a, b, _MM_CMPINT_GT);
1447 } else
1448 return _mm_cmp_epi8_mask(a, b, _MM_CMPINT_EQ);
1449 } else if constexpr (sizeof(T) == 2) {
1450 if constexpr (Greater) {
1451 if constexpr (std::is_signed_v<T>)
1452 return _mm_cmp_epi16_mask(a, b, _MM_CMPINT_GT);
1453 else
1454 return _mm_cmp_epu16_mask(a, b, _MM_CMPINT_GT);
1455 } else
1456 return _mm_cmp_epi16_mask(a, b, _MM_CMPINT_EQ);
1457 } else if constexpr (sizeof(T) == 4) {
1458 if constexpr (Greater) {
1459 if constexpr (std::is_signed_v<T>)
1460 return _mm_cmp_epi32_mask(a, b, _MM_CMPINT_GT);
1461 else
1462 return _mm_cmp_epu32_mask(a, b, _MM_CMPINT_GT);
1463 } else
1464 return _mm_cmp_epi32_mask(a, b, _MM_CMPINT_EQ);
1465 } else if constexpr (sizeof(T) == 8) {
1466 if constexpr (Greater) {
1467 if constexpr (std::is_signed_v<T>)
1468 return _mm_cmp_epi64_mask(a, b, _MM_CMPINT_GT);
1469 else
1470 return _mm_cmp_epu64_mask(a, b, _MM_CMPINT_GT);
1471 } else
1472 return _mm_cmp_epi64_mask(a, b, _MM_CMPINT_EQ);
1473 }
1474 } else {
1475 if constexpr (sizeof(T) == 1) {
1476 if constexpr (!Greater)
1477 return _mm_cmpeq_epi8(a, b);
1478 else if constexpr (std::is_signed_v<T>)
1479 return _mm_cmpgt_epi8(a, b);
1480 else {
1481 auto bias = _mm_set1_epi8(int8_t(-128));
1482 return _mm_cmpgt_epi8(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1483 }
1484 } else if constexpr (sizeof(T) == 2) {
1485 if constexpr (!Greater)
1486 return _mm_cmpeq_epi16(a, b);
1487 else if constexpr (std::is_signed_v<T>)
1488 return _mm_cmpgt_epi16(a, b);
1489 else {
1490 auto bias = _mm_set1_epi16(int16_t(-32768));
1491 return _mm_cmpgt_epi16(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1492 }
1493 } else if constexpr (sizeof(T) == 4) {
1494 if constexpr (!Greater)
1495 return _mm_cmpeq_epi32(a, b);
1496 else if constexpr (std::is_signed_v<T>)
1497 return _mm_cmpgt_epi32(a, b);
1498 else {
1499 auto bias = _mm_set1_epi32(std::bit_cast<int32_t>(uint32_t(0x80000000u)));
1500 return _mm_cmpgt_epi32(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1501 }
1502 } else if constexpr (sizeof(T) == 8) {
1503 if constexpr (!Greater)
1504 return _mm_cmpeq_epi64(a, b);
1505 else if constexpr (std::is_signed_v<T>)
1506 return _mm_cmpgt_epi64(a, b);
1507 else {
1508 auto bias = _mm_set1_epi64x(std::bit_cast<int64_t>(uint64_t(1) << 63));
1509 return _mm_cmpgt_epi64(_mm_xor_si128(a, bias), _mm_xor_si128(b, bias));
1510 }
1511 }
1512 }
1513 }
1514 template <simd_integer_element T, class M>
1515 native_nodiscard native_inline native_const __m128i integer_select(M m, __m128i a, __m128i b) noexcept {
1516 if constexpr (M::compact) {
1517 if constexpr (sizeof(T) == 1)
1518 return _mm_mask_blend_epi8(m.to_native(), b, a);
1519 else if constexpr (sizeof(T) == 2)
1520 return _mm_mask_blend_epi16(m.to_native(), b, a);
1521 else if constexpr (sizeof(T) == 4)
1522 return _mm_mask_blend_epi32(m.to_native(), b, a);
1523 else if constexpr (sizeof(T) == 8)
1524 return _mm_mask_blend_epi64(m.to_native(), b, a);
1525 } else
1526 return _mm_or_si128(_mm_and_si128(m.to_native(), a), _mm_andnot_si128(m.to_native(), b));
1527 }
1528 template <simd_integer_element T, class M>
1529 native_nodiscard native_inline native_const __m128i integer_masked_add(M m, __m128i prior, __m128i a,
1530 __m128i b) noexcept {
1531 if constexpr (M::compact) {
1532 if constexpr (sizeof(T) == 1)
1533 return _mm_mask_add_epi8(prior, m.to_native(), a, b);
1534 else if constexpr (sizeof(T) == 2)
1535 return _mm_mask_add_epi16(prior, m.to_native(), a, b);
1536 else if constexpr (sizeof(T) == 4)
1537 return _mm_mask_add_epi32(prior, m.to_native(), a, b);
1538 else if constexpr (sizeof(T) == 8)
1539 return _mm_mask_add_epi64(prior, m.to_native(), a, b);
1540 } else
1541 return integer_select<T>(m, integer_add<T>(a, b), prior);
1542 }
1543 template <simd_integer_element T, class M>
1544 native_nodiscard native_inline native_const __m128i integer_masked_sub(M m, __m128i prior, __m128i a,
1545 __m128i b) noexcept {
1546 if constexpr (M::compact) {
1547 if constexpr (sizeof(T) == 1)
1548 return _mm_mask_sub_epi8(prior, m.to_native(), a, b);
1549 else if constexpr (sizeof(T) == 2)
1550 return _mm_mask_sub_epi16(prior, m.to_native(), a, b);
1551 else if constexpr (sizeof(T) == 4)
1552 return _mm_mask_sub_epi32(prior, m.to_native(), a, b);
1553 else if constexpr (sizeof(T) == 8)
1554 return _mm_mask_sub_epi64(prior, m.to_native(), a, b);
1555 } else
1556 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1557 }
1558#endif
1559
1560#if NATIVE_HAS_AVX2
1561 native_nodiscard native_inline native_const __m256i integer_and(__m256i a, __m256i b) noexcept {
1562 return _mm256_and_si256(a, b);
1563 }
1564 native_nodiscard native_inline native_const __m256i integer_or(__m256i a, __m256i b) noexcept {
1565 return _mm256_or_si256(a, b);
1566 }
1567 native_nodiscard native_inline native_const __m256i integer_xor(__m256i a, __m256i b) noexcept {
1568 return _mm256_xor_si256(a, b);
1569 }
1570 template <simd_integer_element T, bool Greater>
1571 native_nodiscard native_inline native_const auto integer_compare(__m256i a, __m256i b) noexcept {
1572 if constexpr (mask_compact<sizeof(T), 32 / sizeof(T)>) {
1573 if constexpr (sizeof(T) == 1) {
1574 if constexpr (Greater) {
1575 if constexpr (std::is_signed_v<T>)
1576 return _mm256_cmp_epi8_mask(a, b, _MM_CMPINT_GT);
1577 else
1578 return _mm256_cmp_epu8_mask(a, b, _MM_CMPINT_GT);
1579 } else
1580 return _mm256_cmp_epi8_mask(a, b, _MM_CMPINT_EQ);
1581 } else if constexpr (sizeof(T) == 2) {
1582 if constexpr (Greater) {
1583 if constexpr (std::is_signed_v<T>)
1584 return _mm256_cmp_epi16_mask(a, b, _MM_CMPINT_GT);
1585 else
1586 return _mm256_cmp_epu16_mask(a, b, _MM_CMPINT_GT);
1587 } else
1588 return _mm256_cmp_epi16_mask(a, b, _MM_CMPINT_EQ);
1589 } else if constexpr (sizeof(T) == 4) {
1590 if constexpr (Greater) {
1591 if constexpr (std::is_signed_v<T>)
1592 return _mm256_cmp_epi32_mask(a, b, _MM_CMPINT_GT);
1593 else
1594 return _mm256_cmp_epu32_mask(a, b, _MM_CMPINT_GT);
1595 } else
1596 return _mm256_cmp_epi32_mask(a, b, _MM_CMPINT_EQ);
1597 } else if constexpr (sizeof(T) == 8) {
1598 if constexpr (Greater) {
1599 if constexpr (std::is_signed_v<T>)
1600 return _mm256_cmp_epi64_mask(a, b, _MM_CMPINT_GT);
1601 else
1602 return _mm256_cmp_epu64_mask(a, b, _MM_CMPINT_GT);
1603 } else
1604 return _mm256_cmp_epi64_mask(a, b, _MM_CMPINT_EQ);
1605 }
1606 } else {
1607 if constexpr (sizeof(T) == 1) {
1608 if constexpr (!Greater)
1609 return _mm256_cmpeq_epi8(a, b);
1610 else if constexpr (std::is_signed_v<T>)
1611 return _mm256_cmpgt_epi8(a, b);
1612 else {
1613 auto bias = _mm256_set1_epi8(int8_t(-128));
1614 return _mm256_cmpgt_epi8(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1615 }
1616 } else if constexpr (sizeof(T) == 2) {
1617 if constexpr (!Greater)
1618 return _mm256_cmpeq_epi16(a, b);
1619 else if constexpr (std::is_signed_v<T>)
1620 return _mm256_cmpgt_epi16(a, b);
1621 else {
1622 auto bias = _mm256_set1_epi16(int16_t(-32768));
1623 return _mm256_cmpgt_epi16(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1624 }
1625 } else if constexpr (sizeof(T) == 4) {
1626 if constexpr (!Greater)
1627 return _mm256_cmpeq_epi32(a, b);
1628 else if constexpr (std::is_signed_v<T>)
1629 return _mm256_cmpgt_epi32(a, b);
1630 else {
1631 auto bias = _mm256_set1_epi32(std::bit_cast<int32_t>(uint32_t(0x80000000u)));
1632 return _mm256_cmpgt_epi32(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1633 }
1634 } else if constexpr (sizeof(T) == 8) {
1635 if constexpr (!Greater)
1636 return _mm256_cmpeq_epi64(a, b);
1637 else if constexpr (std::is_signed_v<T>)
1638 return _mm256_cmpgt_epi64(a, b);
1639 else {
1640 auto bias = _mm256_set1_epi64x(std::bit_cast<int64_t>(uint64_t(1) << 63));
1641 return _mm256_cmpgt_epi64(_mm256_xor_si256(a, bias), _mm256_xor_si256(b, bias));
1642 }
1643 }
1644 }
1645 }
1646 template <simd_integer_element T, class M>
1647 native_nodiscard native_inline native_const __m256i integer_select(M m, __m256i a, __m256i b) noexcept {
1648 if constexpr (M::compact) {
1649 if constexpr (sizeof(T) == 1)
1650 return _mm256_mask_blend_epi8(m.to_native(), b, a);
1651 else if constexpr (sizeof(T) == 2)
1652 return _mm256_mask_blend_epi16(m.to_native(), b, a);
1653 else if constexpr (sizeof(T) == 4)
1654 return _mm256_mask_blend_epi32(m.to_native(), b, a);
1655 else if constexpr (sizeof(T) == 8)
1656 return _mm256_mask_blend_epi64(m.to_native(), b, a);
1657 } else
1658 return _mm256_or_si256(_mm256_and_si256(m.to_native(), a), _mm256_andnot_si256(m.to_native(), b));
1659 }
1660 template <simd_integer_element T, class M>
1661 native_nodiscard native_inline native_const __m256i integer_masked_add(M m, __m256i prior, __m256i a,
1662 __m256i b) noexcept {
1663 if constexpr (M::compact) {
1664 if constexpr (sizeof(T) == 1)
1665 return _mm256_mask_add_epi8(prior, m.to_native(), a, b);
1666 else if constexpr (sizeof(T) == 2)
1667 return _mm256_mask_add_epi16(prior, m.to_native(), a, b);
1668 else if constexpr (sizeof(T) == 4)
1669 return _mm256_mask_add_epi32(prior, m.to_native(), a, b);
1670 else if constexpr (sizeof(T) == 8)
1671 return _mm256_mask_add_epi64(prior, m.to_native(), a, b);
1672 } else
1673 return integer_select<T>(m, integer_add<T>(a, b), prior);
1674 }
1675 template <simd_integer_element T, class M>
1676 native_nodiscard native_inline native_const __m256i integer_masked_sub(M m, __m256i prior, __m256i a,
1677 __m256i b) noexcept {
1678 if constexpr (M::compact) {
1679 if constexpr (sizeof(T) == 1)
1680 return _mm256_mask_sub_epi8(prior, m.to_native(), a, b);
1681 else if constexpr (sizeof(T) == 2)
1682 return _mm256_mask_sub_epi16(prior, m.to_native(), a, b);
1683 else if constexpr (sizeof(T) == 4)
1684 return _mm256_mask_sub_epi32(prior, m.to_native(), a, b);
1685 else if constexpr (sizeof(T) == 8)
1686 return _mm256_mask_sub_epi64(prior, m.to_native(), a, b);
1687 } else
1688 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1689 }
1690#endif
1691
1692#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
1693 native_nodiscard native_inline native_const __m512i integer_and(__m512i a, __m512i b) noexcept {
1694 return _mm512_and_si512(a, b);
1695 }
1696 native_nodiscard native_inline native_const __m512i integer_or(__m512i a, __m512i b) noexcept {
1697 return _mm512_or_si512(a, b);
1698 }
1699 native_nodiscard native_inline native_const __m512i integer_xor(__m512i a, __m512i b) noexcept {
1700 return _mm512_xor_si512(a, b);
1701 }
1702 template <simd_integer_element T, bool Greater>
1703 native_nodiscard native_inline native_const auto integer_compare(__m512i a, __m512i b) noexcept {
1704 if constexpr (mask_compact<sizeof(T), 64 / sizeof(T)>) {
1705#if NATIVE_HAS_AVX512BW
1706 if constexpr (sizeof(T) == 1) {
1707 if constexpr (Greater) {
1708 if constexpr (std::is_signed_v<T>)
1709 return _mm512_cmp_epi8_mask(a, b, _MM_CMPINT_GT);
1710 else
1711 return _mm512_cmp_epu8_mask(a, b, _MM_CMPINT_GT);
1712 } else
1713 return _mm512_cmp_epi8_mask(a, b, _MM_CMPINT_EQ);
1714 } else if constexpr (sizeof(T) == 2) {
1715 if constexpr (Greater) {
1716 if constexpr (std::is_signed_v<T>)
1717 return _mm512_cmp_epi16_mask(a, b, _MM_CMPINT_GT);
1718 else
1719 return _mm512_cmp_epu16_mask(a, b, _MM_CMPINT_GT);
1720 } else
1721 return _mm512_cmp_epi16_mask(a, b, _MM_CMPINT_EQ);
1722 }
1723#endif
1724 if constexpr (sizeof(T) == 4) {
1725 if constexpr (Greater) {
1726 if constexpr (std::is_signed_v<T>)
1727 return _mm512_cmp_epi32_mask(a, b, _MM_CMPINT_GT);
1728 else
1729 return _mm512_cmp_epu32_mask(a, b, _MM_CMPINT_GT);
1730 } else
1731 return _mm512_cmp_epi32_mask(a, b, _MM_CMPINT_EQ);
1732 } else if constexpr (sizeof(T) == 8) {
1733 if constexpr (Greater) {
1734 if constexpr (std::is_signed_v<T>)
1735 return _mm512_cmp_epi64_mask(a, b, _MM_CMPINT_GT);
1736 else
1737 return _mm512_cmp_epu64_mask(a, b, _MM_CMPINT_GT);
1738 } else
1739 return _mm512_cmp_epi64_mask(a, b, _MM_CMPINT_EQ);
1740 }
1741 } else {
1742 static_assert(sizeof(T) == 0, "512-bit integer comparisons require native predicates");
1743 }
1744 }
1745 template <simd_integer_element T, class M>
1746 native_nodiscard native_inline native_const __m512i integer_select(M m, __m512i a, __m512i b) noexcept {
1747 if constexpr (M::compact) {
1748#if NATIVE_HAS_AVX512BW
1749 if constexpr (sizeof(T) == 1)
1750 return _mm512_mask_blend_epi8(m.to_native(), b, a);
1751 else if constexpr (sizeof(T) == 2)
1752 return _mm512_mask_blend_epi16(m.to_native(), b, a);
1753#endif
1754 if constexpr (sizeof(T) == 4)
1755 return _mm512_mask_blend_epi32(m.to_native(), b, a);
1756 else if constexpr (sizeof(T) == 8)
1757 return _mm512_mask_blend_epi64(m.to_native(), b, a);
1758 } else
1759 return _mm512_or_si512(_mm512_and_si512(m.to_native(), a), _mm512_andnot_si512(m.to_native(), b));
1760 }
1761 template <simd_integer_element T, class M>
1762 native_nodiscard native_inline native_const __m512i integer_masked_add(M m, __m512i prior, __m512i a,
1763 __m512i b) noexcept {
1764 if constexpr (M::compact) {
1765#if NATIVE_HAS_AVX512BW
1766 if constexpr (sizeof(T) == 1)
1767 return _mm512_mask_add_epi8(prior, m.to_native(), a, b);
1768 else if constexpr (sizeof(T) == 2)
1769 return _mm512_mask_add_epi16(prior, m.to_native(), a, b);
1770#endif
1771 if constexpr (sizeof(T) == 4)
1772 return _mm512_mask_add_epi32(prior, m.to_native(), a, b);
1773 else if constexpr (sizeof(T) == 8)
1774 return _mm512_mask_add_epi64(prior, m.to_native(), a, b);
1775 } else
1776 return integer_select<T>(m, integer_add<T>(a, b), prior);
1777 }
1778 template <simd_integer_element T, class M>
1779 native_nodiscard native_inline native_const __m512i integer_masked_sub(M m, __m512i prior, __m512i a,
1780 __m512i b) noexcept {
1781 if constexpr (M::compact) {
1782#if NATIVE_HAS_AVX512BW
1783 if constexpr (sizeof(T) == 1)
1784 return _mm512_mask_sub_epi8(prior, m.to_native(), a, b);
1785 else if constexpr (sizeof(T) == 2)
1786 return _mm512_mask_sub_epi16(prior, m.to_native(), a, b);
1787#endif
1788 if constexpr (sizeof(T) == 4)
1789 return _mm512_mask_sub_epi32(prior, m.to_native(), a, b);
1790 else if constexpr (sizeof(T) == 8)
1791 return _mm512_mask_sub_epi64(prior, m.to_native(), a, b);
1792 } else
1793 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1794 }
1795#endif
1796
1797#if NATIVE_HAS_ARM_NEON
1798 native_nodiscard native_inline native_const uint8x16_t integer_and(uint8x16_t a, uint8x16_t b) noexcept {
1799 return vandq_u8(a, b);
1800 }
1801 native_nodiscard native_inline native_const uint8x16_t integer_or(uint8x16_t a, uint8x16_t b) noexcept {
1802 return vorrq_u8(a, b);
1803 }
1804 native_nodiscard native_inline native_const uint8x16_t integer_xor(uint8x16_t a, uint8x16_t b) noexcept {
1805 return veorq_u8(a, b);
1806 }
1807 template <simd_integer_element T>
1808 native_nodiscard native_inline native_const uint8x16_t integer_add(uint8x16_t a, uint8x16_t b) noexcept {
1809 if constexpr (sizeof(T) == 1) {
1810 return vaddq_u8(a, b);
1811 } else if constexpr (sizeof(T) == 2) {
1812 return vreinterpretq_u8_u16(vaddq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1813 } else if constexpr (sizeof(T) == 4) {
1814 return vreinterpretq_u8_u32(vaddq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1815 } else if constexpr (sizeof(T) == 8) {
1816 return vreinterpretq_u8_u64(vaddq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1817 }
1818 }
1819 template <simd_integer_element T>
1820 native_nodiscard native_inline native_const uint8x16_t integer_sub(uint8x16_t a, uint8x16_t b) noexcept {
1821 if constexpr (sizeof(T) == 1) {
1822 return vsubq_u8(a, b);
1823 } else if constexpr (sizeof(T) == 2) {
1824 return vreinterpretq_u8_u16(vsubq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1825 } else if constexpr (sizeof(T) == 4) {
1826 return vreinterpretq_u8_u32(vsubq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1827 } else if constexpr (sizeof(T) == 8) {
1828 return vreinterpretq_u8_u64(vsubq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1829 }
1830 }
1831 template <simd_integer_element T>
1832 native_nodiscard native_inline native_const uint8x16_t integer_mul(uint8x16_t a, uint8x16_t b) noexcept {
1833 if constexpr (sizeof(T) == 1) {
1834 return vmulq_u8(a, b);
1835 } else if constexpr (sizeof(T) == 2) {
1836 return vreinterpretq_u8_u16(vmulq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1837 } else if constexpr (sizeof(T) == 4) {
1838 return vreinterpretq_u8_u32(vmulq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1839 } else if constexpr (sizeof(T) == 8) {
1840 auto x = vreinterpretq_u64_u8(a), y = vreinterpretq_u64_u8(b);
1841 auto xl = vmovn_u64(x), yl = vmovn_u64(y);
1842 auto cross =
1843 vadd_u32(vmul_u32(xl, vmovn_u64(vshrq_n_u64(y, 32))), vmul_u32(vmovn_u64(vshrq_n_u64(x, 32)), yl));
1844 return vreinterpretq_u8_u64(vaddq_u64(vmull_u32(xl, yl), vshlq_n_u64(vmovl_u32(cross), 32)));
1845 }
1846 }
1847 template <simd_integer_element T>
1848 native_nodiscard native_inline native_const uint8x16_t integer_broadcast_16(T x) noexcept {
1849 if constexpr (sizeof(T) == 1) {
1850 return vdupq_n_u8(integer_word(x));
1851 } else if constexpr (sizeof(T) == 2) {
1852 return vreinterpretq_u8_u16(vdupq_n_u16(integer_word(x)));
1853 } else if constexpr (sizeof(T) == 4) {
1854 return vreinterpretq_u8_u32(vdupq_n_u32(integer_word(x)));
1855 } else if constexpr (sizeof(T) == 8) {
1856 return vreinterpretq_u8_u64(vdupq_n_u64(integer_word(x)));
1857 }
1858 }
1859 template <simd_integer_element T, unsigned S>
1860 requires(S < sizeof(T) * 8)
1861 native_nodiscard native_inline native_const uint8x16_t integer_left(uint8x16_t a) noexcept {
1862 if constexpr (S == 0)
1863 return a;
1864 else {
1865 if constexpr (sizeof(T) == 1) {
1866 return vshlq_n_u8(a, S);
1867 } else if constexpr (sizeof(T) == 2) {
1868 return vreinterpretq_u8_u16(vshlq_n_u16(vreinterpretq_u16_u8(a), S));
1869 } else if constexpr (sizeof(T) == 4) {
1870 return vreinterpretq_u8_u32(vshlq_n_u32(vreinterpretq_u32_u8(a), S));
1871 } else if constexpr (sizeof(T) == 8) {
1872 return vreinterpretq_u8_u64(vshlq_n_u64(vreinterpretq_u64_u8(a), S));
1873 }
1874 }
1875 }
1876 native_nodiscard native_inline native_const uint8x16_t integer_shift_left_variable(uint8x16_t a,uint8x16_t counts) noexcept {
1877 return vreinterpretq_u8_u32(vshlq_u32(vreinterpretq_u32_u8(a),vreinterpretq_s32_u32(counts)));
1878 }
1879 template <simd_integer_element T, unsigned S>
1880 requires(S < sizeof(T) * 8)
1881 native_nodiscard native_inline native_const uint8x16_t integer_right(uint8x16_t a) noexcept {
1882 if constexpr (S == 0)
1883 return a;
1884 else {
1885 if constexpr (sizeof(T) == 1) {
1886 if constexpr (std::is_unsigned_v<T>)
1887 return vshrq_n_u8(a, S);
1888 else
1889 return vreinterpretq_u8_s8(vshrq_n_s8(vreinterpretq_s8_u8(a), S));
1890 } else if constexpr (sizeof(T) == 2) {
1891 if constexpr (std::is_unsigned_v<T>)
1892 return vreinterpretq_u8_u16(vshrq_n_u16(vreinterpretq_u16_u8(a), S));
1893 else
1894 return vreinterpretq_u8_s16(vshrq_n_s16(vreinterpretq_s16_u8(a), S));
1895 } else if constexpr (sizeof(T) == 4) {
1896 if constexpr (std::is_unsigned_v<T>)
1897 return vreinterpretq_u8_u32(vshrq_n_u32(vreinterpretq_u32_u8(a), S));
1898 else
1899 return vreinterpretq_u8_s32(vshrq_n_s32(vreinterpretq_s32_u8(a), S));
1900 } else if constexpr (sizeof(T) == 8) {
1901 if constexpr (std::is_unsigned_v<T>)
1902 return vreinterpretq_u8_u64(vshrq_n_u64(vreinterpretq_u64_u8(a), S));
1903 else
1904 return vreinterpretq_u8_s64(vshrq_n_s64(vreinterpretq_s64_u8(a), S));
1905 }
1906 }
1907 }
1908 template <simd_integer_element T, bool Greater>
1909 native_nodiscard native_inline native_const uint8x16_t integer_compare(uint8x16_t a, uint8x16_t b) noexcept {
1910 if constexpr (sizeof(T) == 1) {
1911 if constexpr (!Greater)
1912 return vceqq_u8(a, b);
1913 else if constexpr (std::is_unsigned_v<T>)
1914 return vcgtq_u8(a, b);
1915 else
1916 return vcgtq_s8(vreinterpretq_s8_u8(a), vreinterpretq_s8_u8(b));
1917 } else if constexpr (sizeof(T) == 2) {
1918 if constexpr (!Greater)
1919 return vreinterpretq_u8_u16(vceqq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1920 else if constexpr (std::is_unsigned_v<T>)
1921 return vreinterpretq_u8_u16(vcgtq_u16(vreinterpretq_u16_u8(a), vreinterpretq_u16_u8(b)));
1922 else
1923 return vreinterpretq_u8_u16(vcgtq_s16(vreinterpretq_s16_u8(a), vreinterpretq_s16_u8(b)));
1924 } else if constexpr (sizeof(T) == 4) {
1925 if constexpr (!Greater)
1926 return vreinterpretq_u8_u32(vceqq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1927 else if constexpr (std::is_unsigned_v<T>)
1928 return vreinterpretq_u8_u32(vcgtq_u32(vreinterpretq_u32_u8(a), vreinterpretq_u32_u8(b)));
1929 else
1930 return vreinterpretq_u8_u32(vcgtq_s32(vreinterpretq_s32_u8(a), vreinterpretq_s32_u8(b)));
1931 } else if constexpr (sizeof(T) == 8) {
1932 if constexpr (!Greater)
1933 return vreinterpretq_u8_u64(vceqq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1934 else if constexpr (std::is_unsigned_v<T>)
1935 return vreinterpretq_u8_u64(vcgtq_u64(vreinterpretq_u64_u8(a), vreinterpretq_u64_u8(b)));
1936 else
1937 return vreinterpretq_u8_u64(vcgtq_s64(vreinterpretq_s64_u8(a), vreinterpretq_s64_u8(b)));
1938 }
1939 }
1940 template <simd_integer_element T, class M>
1941 native_nodiscard native_inline native_const uint8x16_t integer_select(M m, uint8x16_t a, uint8x16_t b) noexcept {
1942 return vbslq_u8(m.to_native(), a, b);
1943 }
1944 template <simd_integer_element T, class M>
1945 native_nodiscard native_inline native_const uint8x16_t integer_masked_add(M m, uint8x16_t prior, uint8x16_t a,
1946 uint8x16_t b) noexcept {
1947 return integer_select<T>(m, integer_add<T>(a, b), prior);
1948 }
1949 template <simd_integer_element T, class M>
1950 native_nodiscard native_inline native_const uint8x16_t integer_masked_sub(M m, uint8x16_t prior, uint8x16_t a,
1951 uint8x16_t b) noexcept {
1952 return integer_select<T>(m, integer_sub<T>(a, b), prior);
1953 }
1954#endif
1955
1956#if NATIVE_HAS_AVX2
1957 template <simd_integer_element T, class M>
1958 native_nodiscard native_inline native_const __m128i integer_masked_mul(M m, __m128i prior, __m128i a,
1959 __m128i b) noexcept {
1960 if constexpr (M::compact && sizeof(T) > 1) {
1961 if constexpr (sizeof(T) == 2)
1962 return _mm_mask_mullo_epi16(prior, m.to_native(), a, b);
1963 if constexpr (sizeof(T) == 4)
1964 return _mm_mask_mullo_epi32(prior, m.to_native(), a, b);
1965 else if constexpr (sizeof(T) == 8)
1966 return _mm_mask_mullo_epi64(prior, m.to_native(), a, b);
1967 } else
1968 return integer_select<T>(m, integer_mul<T>(a, b), prior);
1969 }
1970#endif
1971#if NATIVE_HAS_AVX2
1972 template <simd_integer_element T, class M>
1973 native_nodiscard native_inline native_const __m256i integer_masked_mul(M m, __m256i prior, __m256i a,
1974 __m256i b) noexcept {
1975 if constexpr (M::compact && sizeof(T) > 1) {
1976 if constexpr (sizeof(T) == 2)
1977 return _mm256_mask_mullo_epi16(prior, m.to_native(), a, b);
1978 if constexpr (sizeof(T) == 4)
1979 return _mm256_mask_mullo_epi32(prior, m.to_native(), a, b);
1980 else if constexpr (sizeof(T) == 8)
1981 return _mm256_mask_mullo_epi64(prior, m.to_native(), a, b);
1982 } else
1983 return integer_select<T>(m, integer_mul<T>(a, b), prior);
1984 }
1985#endif
1986#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
1987 template <simd_integer_element T, class M>
1988 native_nodiscard native_inline native_const __m512i integer_masked_mul(M m, __m512i prior, __m512i a,
1989 __m512i b) noexcept {
1990 if constexpr (M::compact && sizeof(T) > 1) {
1991#if NATIVE_HAS_AVX512BW
1992 if constexpr (sizeof(T) == 2)
1993 return _mm512_mask_mullo_epi16(prior, m.to_native(), a, b);
1994#endif
1995 if constexpr (sizeof(T) == 4)
1996 return _mm512_mask_mullo_epi32(prior, m.to_native(), a, b);
1997 else if constexpr (sizeof(T) == 8)
1998 return _mm512_mask_mullo_epi64(prior, m.to_native(), a, b);
1999 } else
2000 return integer_select<T>(m, integer_mul<T>(a, b), prior);
2001 }
2002#endif
2003#if NATIVE_HAS_ARM_NEON
2004 template <simd_integer_element T, class M>
2005 native_nodiscard native_inline native_const uint8x16_t integer_masked_mul(M m, uint8x16_t prior, uint8x16_t a,
2006 uint8x16_t b) noexcept {
2007 return integer_select<T>(m, integer_mul<T>(a, b), prior);
2008 }
2009#endif
2010#if NATIVE_HAS_AVX2
2011 native_nodiscard native_inline native_const __m128i integer_bit_select(__m128i m,__m128i a,__m128i b) noexcept {
2012 return _mm_or_si128(_mm_and_si128(m,a),_mm_andnot_si128(m,b));
2013 }
2014#endif
2015#if NATIVE_HAS_AVX2
2016 native_nodiscard native_inline native_const __m256i integer_bit_select(__m256i m,__m256i a,__m256i b) noexcept {
2017 return _mm256_or_si256(_mm256_and_si256(m,a),_mm256_andnot_si256(m,b));
2018 }
2019#endif
2020#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2021 native_nodiscard native_inline native_const __m512i integer_bit_select(__m512i m,__m512i a,__m512i b) noexcept {
2022 return _mm512_or_si512(_mm512_and_si512(m,a),_mm512_andnot_si512(m,b));
2023 }
2024#endif
2025#if NATIVE_HAS_ARM_NEON
2026 native_nodiscard native_inline native_const uint8x16_t integer_bit_select(uint8x16_t m,uint8x16_t a,uint8x16_t b) noexcept { return vbslq_u8(m,a,b); }
2027#endif
2028} // namespace NATIVE_BACKEND_NAMESPACE
2029
2030
2031namespace native {
2032 namespace detail::NATIVE_BACKEND {
2033 template <std::size_t Bytes> struct integer_storage;
2034#if NATIVE_HAS_AVX2
2035 template <> struct integer_storage<16> {
2036 using type = __m128i;
2037 };
2038 template <> struct integer_storage<32> {
2039 using type = __m256i;
2040 };
2041#endif
2042#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2043 template <> struct integer_storage<64> {
2044 using type = __m512i;
2045 };
2046#endif
2047#if NATIVE_HAS_ARM_NEON
2048 template <> struct integer_storage<16> {
2049 using type = uint8x16_t;
2050 };
2051#endif
2052 template <class T, std::size_t N>
2053 inline constexpr bool integer_shape =
2054 simd_integer_element<T> && (N == 1
2055#if NATIVE_HAS_AVX2
2056 || sizeof(T) * N == 16 || sizeof(T) * N == 32
2057#endif
2058#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2059 || (sizeof(T) * N == 64 && (sizeof(T) >= 4
2060#if NATIVE_HAS_AVX512BW
2061 || sizeof(T) <= 2
2062#endif
2063 ))
2064#endif
2065#if NATIVE_HAS_ARM_NEON
2066 || sizeof(T) * N == 16
2067#endif
2068 );
2069 } // namespace detail
2070
2071 namespace detail::NATIVE_BACKEND {
2072 template <::native::isa<> Arch, simd_integer_element T, std::size_t N, std::size_t A = 1,
2073 simd_access Access = simd_access::ordinary>
2074 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2075 native_nodiscard native_inline native_pure constexpr simd<T, N,Arch> load_simd(T const *p, simd_memory<A, Access> = {}) noexcept;
2076 template <simd_integer_element T, std::size_t N, std::size_t A = 1,
2077 simd_access Access = simd_access::ordinary, ::native::isa<> Arch>
2078 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2079 native_inline constexpr void store_simd(T *p, simd<T, N,Arch> v, simd_memory<A, Access> = {}) noexcept;
2080 template <::native::isa<> Arch, simd_integer_element T, std::size_t N, std::size_t A = 1,
2081 simd_access Access = simd_access::ordinary>
2082 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2083 native_nodiscard native_inline constexpr native_pure simd<T, N,Arch> load_simd_partial(T const *p, std::size_t count,
2084 simd_memory<A, Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count");
2085 template <simd_integer_element T, std::size_t N, std::size_t A = 1,
2086 simd_access Access = simd_access::ordinary, ::native::isa<> Arch>
2087 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
2088 native_inline constexpr void store_simd_partial(T *p, simd<T, N,Arch> v, std::size_t count,
2089 simd_memory<A, Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count");
2090
2091 }
2092 template <simd_integer_element T, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) struct simd<T, 1,Arch> : detail::swizzle_access<T,1,Arch> {
2093 static constexpr isa<> architecture=Arch;
2094 template <class U> using rebind = simd<U,1,Arch>;
2095 template <std::size_t A = 1>
2097 native_nodiscard static native_inline constexpr simd load_memory(T const * p) noexcept {
2098 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T,1>(p,simd_memory<A>{});
2099 }
2100 template <std::size_t A = 1>
2102 native_inline constexpr void store_memory(T * p) const noexcept {
2103 ::NATIVE_BACKEND_NAMESPACE::store_simd(p,*this,simd_memory<A>{});
2104 }
2105
2106 using value_type = T;
2107 using native_type = T;
2108 using mask_type = ::NATIVE_BACKEND_NAMESPACE::comparison_mask<T, 1,Arch>;
2109 using mask = mask_type;
2110 using predicate_type = predicate<1,Arch>;
2111 using unsigned_register_tag = void;
2112 static constexpr std::size_t lanes = 1;
2113 T value{};
2115 constexpr simd() noexcept = default;
2117 template <simd_integer_element U> constexpr simd(U x) noexcept : value(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(x)) {}
2119 constexpr simd(std::array<T, 1> const &x) noexcept : value(x[0]) {}
2121 native_nodiscard native_inline constexpr operator T() const noexcept { return value; }
2123 native_nodiscard native_inline static constexpr simd from_native(T x) noexcept { return simd(x); }
2125 native_nodiscard native_inline constexpr T to_native() const noexcept { return value; }
2127 native_nodiscard friend native_inline native_const constexpr simd operator+(simd a, simd b) noexcept {
2128 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_scalar_add(a.value, b.value));
2129 }
2131 native_nodiscard friend native_inline native_const constexpr simd operator-(simd a, simd b) noexcept {
2132 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_scalar_sub(a.value, b.value));
2133 }
2135 native_nodiscard friend native_inline native_const constexpr simd operator*(simd a, simd b) noexcept {
2136 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_scalar_mul(a.value, b.value));
2137 }
2139 native_nodiscard friend native_inline native_const constexpr simd operator&(simd a, simd b) noexcept {
2140 return from_native(
2141 ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(::NATIVE_BACKEND_NAMESPACE::integer_word(a.value) & ::NATIVE_BACKEND_NAMESPACE::integer_word(b.value)));
2142 }
2144 native_nodiscard friend native_inline native_const constexpr simd operator|(simd a, simd b) noexcept {
2145 return from_native(
2146 ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(::NATIVE_BACKEND_NAMESPACE::integer_word(a.value) | ::NATIVE_BACKEND_NAMESPACE::integer_word(b.value)));
2147 }
2149 native_nodiscard friend native_inline native_const constexpr simd operator^(simd a, simd b) noexcept {
2150 return from_native(
2151 ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(::NATIVE_BACKEND_NAMESPACE::integer_word(a.value) ^ ::NATIVE_BACKEND_NAMESPACE::integer_word(b.value)));
2152 }
2154 native_nodiscard friend native_inline native_const constexpr simd operator~(simd a) noexcept {
2155 return a ^ simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(~std::make_unsigned_t<T>(0)));
2156 }
2158 native_nodiscard friend native_inline native_const constexpr simd operator-(simd a) noexcept { return simd(T(0)) - a; }
2160 native_nodiscard friend native_inline native_const constexpr simd operator+(simd a) noexcept { return a; }
2162 native_nodiscard friend native_inline native_const constexpr mask_type operator==(simd a, simd b) noexcept {
2163 return mask_type(a.value == b.value);
2164 }
2166 native_nodiscard friend native_inline native_const constexpr mask_type operator>(simd a, simd b) noexcept {
2167 return mask_type(a.value > b.value);
2168 }
2170 native_nodiscard friend native_inline native_const constexpr mask_type operator!=(simd a, simd b) noexcept {
2171 return ~(a == b);
2172 }
2174 native_nodiscard friend native_inline native_const constexpr mask_type operator<(simd a, simd b) noexcept {
2175 return b > a;
2176 }
2178 native_nodiscard friend native_inline native_const constexpr mask_type operator<=(simd a, simd b) noexcept {
2179 return ~(a > b);
2180 }
2182 native_nodiscard friend native_inline native_const constexpr mask_type operator>=(simd a, simd b) noexcept {
2183 return ~(b > a);
2184 }
2186 template <simd_integer_element U>
2187 native_nodiscard friend native_inline native_const constexpr simd operator+(simd a, U b) noexcept {
2188 return a + simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2189 }
2191 template <simd_integer_element U>
2192 native_nodiscard friend native_inline native_const constexpr simd operator+(U a, simd b) noexcept {
2193 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) + b;
2194 }
2196 template <simd_integer_element U>
2197 native_nodiscard friend native_inline native_const constexpr simd operator-(simd a, U b) noexcept {
2198 return a - simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2199 }
2201 template <simd_integer_element U>
2202 native_nodiscard friend native_inline native_const constexpr simd operator-(U a, simd b) noexcept {
2203 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) - b;
2204 }
2206 template <simd_integer_element U>
2207 native_nodiscard friend native_inline native_const constexpr simd operator*(simd a, U b) noexcept {
2208 return a * simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2209 }
2211 template <simd_integer_element U>
2212 native_nodiscard friend native_inline native_const constexpr simd operator*(U a, simd b) noexcept {
2213 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) * b;
2214 }
2216 template <simd_integer_element U>
2217 native_nodiscard friend native_inline native_const constexpr simd operator&(simd a, U b) noexcept {
2218 return a & simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2219 }
2221 template <simd_integer_element U>
2222 native_nodiscard friend native_inline native_const constexpr simd operator&(U a, simd b) noexcept {
2223 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) & b;
2224 }
2226 template <simd_integer_element U>
2227 native_nodiscard friend native_inline native_const constexpr simd operator|(simd a, U b) noexcept {
2228 return a | simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2229 }
2231 template <simd_integer_element U>
2232 native_nodiscard friend native_inline native_const constexpr simd operator|(U a, simd b) noexcept {
2233 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) | b;
2234 }
2236 template <simd_integer_element U>
2237 native_nodiscard friend native_inline native_const constexpr simd operator^(simd a, U b) noexcept {
2238 return a ^ simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2239 }
2241 template <simd_integer_element U>
2242 native_nodiscard friend native_inline native_const constexpr simd operator^(U a, simd b) noexcept {
2243 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) ^ b;
2244 }
2246 template <simd_integer_element U>
2247 native_nodiscard friend native_inline native_const constexpr mask_type operator==(simd a, U b) noexcept {
2248 return a == simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2249 }
2251 template <simd_integer_element U>
2252 native_nodiscard friend native_inline native_const constexpr mask_type operator==(U a, simd b) noexcept {
2253 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) == b;
2254 }
2256 template <simd_integer_element U>
2257 native_nodiscard friend native_inline native_const constexpr mask_type operator!=(simd a, U b) noexcept {
2258 return a != simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2259 }
2261 template <simd_integer_element U>
2262 native_nodiscard friend native_inline native_const constexpr mask_type operator!=(U a, simd b) noexcept {
2263 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) != b;
2264 }
2266 template <simd_integer_element U>
2267 native_nodiscard friend native_inline native_const constexpr mask_type operator<(simd a, U b) noexcept {
2268 return a < simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2269 }
2271 template <simd_integer_element U>
2272 native_nodiscard friend native_inline native_const constexpr mask_type operator<(U a, simd b) noexcept {
2273 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) < b;
2274 }
2276 template <simd_integer_element U>
2277 native_nodiscard friend native_inline native_const constexpr mask_type operator>(simd a, U b) noexcept {
2278 return a > simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2279 }
2281 template <simd_integer_element U>
2282 native_nodiscard friend native_inline native_const constexpr mask_type operator>(U a, simd b) noexcept {
2283 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) > b;
2284 }
2286 template <simd_integer_element U>
2287 native_nodiscard friend native_inline native_const constexpr mask_type operator<=(simd a, U b) noexcept {
2288 return a <= simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2289 }
2291 template <simd_integer_element U>
2292 native_nodiscard friend native_inline native_const constexpr mask_type operator<=(U a, simd b) noexcept {
2293 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) <= b;
2294 }
2296 template <simd_integer_element U>
2297 native_nodiscard friend native_inline native_const constexpr mask_type operator>=(simd a, U b) noexcept {
2298 return a >= simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2299 }
2301 template <simd_integer_element U>
2302 native_nodiscard friend native_inline native_const constexpr mask_type operator>=(U a, simd b) noexcept {
2303 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) >= b;
2304 }
2306 native_inline constexpr simd &operator+=(simd b) noexcept { return *this = *this + b; }
2308 template <simd_integer_element U> native_inline constexpr simd &operator+=(U b) noexcept { return *this = *this + b; }
2310 native_inline constexpr simd &operator-=(simd b) noexcept { return *this = *this - b; }
2312 template <simd_integer_element U> native_inline constexpr simd &operator-=(U b) noexcept { return *this = *this - b; }
2314 native_inline constexpr simd &operator*=(simd b) noexcept { return *this = *this * b; }
2316 template <simd_integer_element U> native_inline constexpr simd &operator*=(U b) noexcept { return *this = *this * b; }
2318 native_inline constexpr simd &operator&=(simd b) noexcept { return *this = *this & b; }
2320 template <simd_integer_element U> native_inline constexpr simd &operator&=(U b) noexcept { return *this = *this & b; }
2322 native_inline constexpr simd &operator|=(simd b) noexcept { return *this = *this | b; }
2324 template <simd_integer_element U> native_inline constexpr simd &operator|=(U b) noexcept { return *this = *this | b; }
2326 native_inline constexpr simd &operator^=(simd b) noexcept { return *this = *this ^ b; }
2328 template <simd_integer_element U> native_inline constexpr simd &operator^=(U b) noexcept { return *this = *this ^ b; }
2330 template <unsigned K>
2331 requires(K < sizeof(T) * 8)
2332 native_nodiscard native_inline native_const constexpr simd left() const noexcept {
2333 using W = ::NATIVE_BACKEND_NAMESPACE::integer_work_word<T>;
2334 auto u = ::NATIVE_BACKEND_NAMESPACE::integer_word(value);
2335 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(W(u) << K));
2336 }
2338 template <unsigned K>
2339 requires(K < sizeof(T) * 8)
2340 native_nodiscard native_inline native_const constexpr simd right() const noexcept {
2341 using U = std::make_unsigned_t<T>;
2342 using W = ::NATIVE_BACKEND_NAMESPACE::integer_work_word<T>;
2343 auto u = ::NATIVE_BACKEND_NAMESPACE::integer_word(value);
2344 if constexpr (K == 0)
2345 return *this;
2346 else {
2347 U result = U(W(u) >> K);
2348 if constexpr (std::is_signed_v<T>) {
2349 if (value < T(0))
2350 result = U(result | U(std::numeric_limits<U>::max() ^ (std::numeric_limits<U>::max() >> K)));
2351 }
2352 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(result));
2353 }
2354 }
2356 template <std::size_t K>
2357 requires(K < sizeof(T) * 8)
2358 native_nodiscard friend native_inline native_const constexpr simd operator<<(simd a, imm_t<K>) noexcept {
2359 return a.template left<K>();
2360 }
2361
2362 template <std::size_t K>
2363 requires(K < sizeof(T) * 8)
2364 native_nodiscard friend native_inline native_const constexpr simd operator>>(simd a, imm_t<K>) noexcept {
2365 return a.template right<K>();
2366 }
2367
2368 native_nodiscard native_inline static native_pure constexpr simd load(T const *p) noexcept {
2369 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2370 }
2372 native_nodiscard native_inline static native_pure constexpr simd loadu(T const *p) noexcept {
2373 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2374 }
2376 native_inline constexpr void store(T *p) const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *this); }
2378 native_inline constexpr void storeu(T *p) const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *this); }
2380 native_nodiscard native_inline constexpr static native_pure simd load_partial(T const *p, std::size_t n,
2381 T fill = T(0)) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") {
2382 assert(n <= lanes);
2383 std::array<T, lanes> data;
2384 data.fill(fill);
2385 if consteval { for(std::size_t i=0;i<n;++i) data[i]=p[i]; }
2386 else { if (n) std::memcpy(data.data(), static_cast<void const *>(p), n * sizeof(T)); }
2387 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(data.data());
2388 }
2390 native_inline constexpr void store_partial(T *p, std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { ::NATIVE_BACKEND_NAMESPACE::store_simd_partial(p, *this, n); }
2392 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator+(simd, U) = delete;
2394 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator+(U, simd) = delete;
2396 template <simd_integer_element U, std::size_t M>
2397 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2398 friend void operator+(simd, ::native::simd<U, M,Arch>) = delete;
2400 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator-(simd, U) = delete;
2402 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator-(U, simd) = delete;
2404 template <simd_integer_element U, std::size_t M>
2405 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2406 friend void operator-(simd, ::native::simd<U, M,Arch>) = delete;
2408 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator*(simd, U) = delete;
2410 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator*(U, simd) = delete;
2412 template <simd_integer_element U, std::size_t M>
2413 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2414 friend void operator*(simd, ::native::simd<U, M,Arch>) = delete;
2416 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator/(simd, U) = delete;
2418 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator/(U, simd) = delete;
2420 template <simd_integer_element U, std::size_t M>
2421 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2422 friend void operator/(simd, ::native::simd<U, M,Arch>) = delete;
2424 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator%(simd, U) = delete;
2426 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator%(U, simd) = delete;
2428 template <simd_integer_element U, std::size_t M>
2429 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2430 friend void operator%(simd, ::native::simd<U, M,Arch>) = delete;
2432 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator&(simd, U) = delete;
2434 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator&(U, simd) = delete;
2436 template <simd_integer_element U, std::size_t M>
2437 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2438 friend void operator&(simd, ::native::simd<U, M,Arch>) = delete;
2440 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator|(simd, U) = delete;
2442 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator|(U, simd) = delete;
2444 template <simd_integer_element U, std::size_t M>
2445 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2446 friend void operator|(simd, ::native::simd<U, M,Arch>) = delete;
2448 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator^(simd, U) = delete;
2450 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator^(U, simd) = delete;
2452 template <simd_integer_element U, std::size_t M>
2453 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2454 friend void operator^(simd, ::native::simd<U, M,Arch>) = delete;
2456 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator==(simd, U) = delete;
2458 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator==(U, simd) = delete;
2460 template <simd_integer_element U, std::size_t M>
2461 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2462 friend void operator==(simd, ::native::simd<U, M,Arch>) = delete;
2464 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator!=(simd, U) = delete;
2466 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator!=(U, simd) = delete;
2468 template <simd_integer_element U, std::size_t M>
2469 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2470 friend void operator!=(simd, ::native::simd<U, M,Arch>) = delete;
2472 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<(simd, U) = delete;
2474 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<(U, simd) = delete;
2476 template <simd_integer_element U, std::size_t M>
2477 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2478 friend void operator<(simd, ::native::simd<U, M,Arch>) = delete;
2480 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>(simd, U) = delete;
2482 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>(U, simd) = delete;
2484 template <simd_integer_element U, std::size_t M>
2485 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2486 friend void operator>(simd, ::native::simd<U, M,Arch>) = delete;
2488 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<=(simd, U) = delete;
2490 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<=(U, simd) = delete;
2492 template <simd_integer_element U, std::size_t M>
2493 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2494 friend void operator<=(simd, ::native::simd<U, M,Arch>) = delete;
2496 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>=(simd, U) = delete;
2498 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>=(U, simd) = delete;
2500 template <simd_integer_element U, std::size_t M>
2501 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
2502 friend void operator>=(simd, ::native::simd<U, M,Arch>) = delete;
2504 friend simd operator/(simd, simd) = delete;
2506 friend simd operator%(simd, simd) = delete;
2508 template <simd_integer_element U> friend simd operator/(simd, U) = delete;
2510 template <simd_integer_element U> friend simd operator/(U, simd) = delete;
2512 template <simd_integer_element U> friend simd operator%(simd, U) = delete;
2514 template <simd_integer_element U> friend simd operator%(U, simd) = delete;
2516 template <simd_integer_element U> friend simd operator<<(simd, U) = delete;
2518 template <simd_integer_element U> friend simd operator>>(simd, U) = delete;
2519 };
2520
2521#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
2522 // Qualified native helper names must exist even before constraints are checked.
2523 // A real integer specialization owns one intrinsic register. The storage trait
2524 // chooses representation only; every operation below has integer semantics.
2525 template <simd_integer_element T, std::size_t N, ::native::isa<> Arch>
2526 requires NATIVE_ARCH_REQUIRES(Arch) &&(N > 1 && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>)
2527 struct simd<T, N,Arch> : detail::swizzle_access<T,N,Arch> {
2528 static constexpr isa<> architecture=Arch;
2529 template <class U> using rebind = simd<U,N,Arch>;
2530 template <std::size_t A = 1>
2532 native_nodiscard static native_inline constexpr simd load_memory(T const * p) noexcept {
2533 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T,N>(p,simd_memory<A>{});
2534 }
2535 template <std::size_t A = 1>
2537 native_inline constexpr void store_memory(T * p) const noexcept {
2538 ::NATIVE_BACKEND_NAMESPACE::store_simd(p,*this,simd_memory<A>{});
2539 }
2540
2541 using value_type = T;
2542 using native_type = typename ::NATIVE_BACKEND_NAMESPACE::integer_storage<sizeof(T) * N>::type;
2543 using mask_type = ::NATIVE_BACKEND_NAMESPACE::comparison_mask<T, N,Arch>;
2544 using mask = mask_type;
2545 using predicate_type = predicate<N,Arch>;
2546 using unsigned_register_tag = void;
2547 static constexpr std::size_t lanes = N;
2548 native_type value{};
2550 constexpr simd() noexcept = default;
2552 template <simd_integer_element U> native_inline constexpr simd(U input) noexcept {
2553 T x = ::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(input);
2554 if consteval {
2555 std::array<T,N> data;
2556 data.fill(x);
2557 value=__builtin_bit_cast(native_type,data);
2558 } else {
2559 if constexpr (sizeof(T) * N == 16)
2560 value = ::NATIVE_BACKEND_NAMESPACE::integer_broadcast_16(x);
2561#if NATIVE_HAS_AVX2
2562 else if constexpr (sizeof(T) * N == 32)
2563 value = ::NATIVE_BACKEND_NAMESPACE::integer_broadcast_32(x);
2564#endif
2565#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2566 else if constexpr (sizeof(T) * N == 64)
2567 value = ::NATIVE_BACKEND_NAMESPACE::integer_broadcast_64(x);
2568#endif
2569 }
2570 }
2571#if NATIVE_HOST_X86
2573 native_inline constexpr native_target("sse2") simd(native_type x) noexcept requires(sizeof(native_type)==16) : value(x) {}
2575 native_inline constexpr native_target("avx") simd(native_type x) noexcept requires(sizeof(native_type)==32) : value(x) {}
2577 native_inline constexpr native_target("avx512f") simd(native_type x) noexcept requires(sizeof(native_type)==64) : value(x) {}
2578#else
2580 native_inline constexpr simd(native_type x) noexcept : value(x) {}
2581#endif
2583 template <class... U>
2584 requires(sizeof...(U) == N && (simd_integer_element<U> && ...))
2585 native_inline constexpr simd(U... xs) noexcept {
2586 std::array<T, N> data{::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(xs)...};
2587 if consteval { value=__builtin_bit_cast(native_type,data); }
2588 else { std::memcpy(&value, data.data(), sizeof(value)); }
2589 }
2590
2591 native_inline constexpr simd(std::array<T, N> const &data) noexcept {
2592 if consteval { value=__builtin_bit_cast(native_type,data); }
2593 else { std::memcpy(&value, data.data(), sizeof(value)); }
2594 }
2595#if NATIVE_HOST_X86
2598 operator native_type() const noexcept requires(sizeof(native_type)==16) { return value; }
2601 simd from_native(native_type x) noexcept requires(sizeof(native_type)==16) { return simd(x); }
2604 native_type to_native() const noexcept requires(sizeof(native_type)==16) { return value; }
2607 operator native_type() const noexcept requires(sizeof(native_type)==32) { return value; }
2610 simd from_native(native_type x) noexcept requires(sizeof(native_type)==32) { return simd(x); }
2613 native_type to_native() const noexcept requires(sizeof(native_type)==32) { return value; }
2616 operator native_type() const noexcept requires(sizeof(native_type)==64) { return value; }
2618 native_nodiscard static native_inline constexpr native_const native_target("avx512f")
2619 simd from_native(native_type x) noexcept requires(sizeof(native_type)==64) { return simd(x); }
2622 native_type to_native() const noexcept requires(sizeof(native_type)==64) { return value; }
2623#else
2625 native_nodiscard native_inline constexpr native_const operator native_type() const noexcept { return value; }
2627 native_nodiscard native_inline constexpr static native_const simd from_native(native_type x) noexcept { return simd(x); }
2629 native_nodiscard native_inline constexpr native_const native_type to_native() const noexcept { return value; }
2630#endif
2633 if consteval {
2634 using U=std::make_unsigned_t<T>;
2635 using V=U __attribute__((ext_vector_type(N)));
2636 return from_native(__builtin_bit_cast(native_type,
2637 __builtin_bit_cast(V,a.value) + __builtin_bit_cast(V,b.value)));
2638 }
2639 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_add<T>(a.value, b.value));
2640 }
2641
2643 if consteval {
2644 using U=std::make_unsigned_t<T>;
2645 using V=U __attribute__((ext_vector_type(N)));
2646 return from_native(__builtin_bit_cast(native_type,
2647 __builtin_bit_cast(V,a.value) - __builtin_bit_cast(V,b.value)));
2648 }
2649 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_sub<T>(a.value, b.value));
2650 }
2651
2653 if consteval {
2654 using U=std::make_unsigned_t<T>;
2655 using V=U __attribute__((ext_vector_type(N)));
2656 return from_native(__builtin_bit_cast(native_type,
2657 __builtin_bit_cast(V,a.value) * __builtin_bit_cast(V,b.value)));
2658 }
2659 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_mul<T>(a.value, b.value));
2660 }
2661
2663 if consteval {
2664 using U=std::make_unsigned_t<T>;
2665 using V=U __attribute__((ext_vector_type(N)));
2666 return from_native(__builtin_bit_cast(native_type,
2667 __builtin_bit_cast(V,a.value) & __builtin_bit_cast(V,b.value)));
2668 }
2669 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_and(a.value, b.value));
2670 }
2671
2673 if consteval {
2674 using U=std::make_unsigned_t<T>;
2675 using V=U __attribute__((ext_vector_type(N)));
2676 return from_native(__builtin_bit_cast(native_type,
2677 __builtin_bit_cast(V,a.value) | __builtin_bit_cast(V,b.value)));
2678 }
2679 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_or(a.value, b.value));
2680 }
2681
2683 if consteval {
2684 using U=std::make_unsigned_t<T>;
2685 using V=U __attribute__((ext_vector_type(N)));
2686 return from_native(__builtin_bit_cast(native_type,
2687 __builtin_bit_cast(V,a.value) ^ __builtin_bit_cast(V,b.value)));
2688 }
2689 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_xor(a.value, b.value));
2690 }
2691
2693 return a ^ simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(~std::make_unsigned_t<T>(0)));
2694 }
2695
2696 native_nodiscard friend native_inline constexpr native_const simd operator-(simd a) noexcept { return simd(T(0)) - a; }
2698 native_nodiscard friend native_inline constexpr native_const simd operator+(simd a) noexcept { return a; }
2700 native_nodiscard friend native_inline constexpr native_const mask_type operator==(simd a, simd b) noexcept {
2701 if consteval {
2702 auto x=__builtin_bit_cast(std::array<T,N>,a.value);
2703 auto y=__builtin_bit_cast(std::array<T,N>,b.value);
2704 std::uint64_t bits=0;
2705 for(std::size_t i=0;i<N;++i) bits|=std::uint64_t(x[i] == y[i])<<i;
2706 return mask_type::from_bitset(bits);
2707 }
2708 return mask_type::unsafe_from_native(::NATIVE_BACKEND_NAMESPACE::integer_compare<T, false>(a.value, b.value));
2709 }
2710
2711 native_nodiscard friend native_inline constexpr native_const mask_type operator>(simd a, simd b) noexcept {
2712 if consteval {
2713 auto x=__builtin_bit_cast(std::array<T,N>,a.value);
2714 auto y=__builtin_bit_cast(std::array<T,N>,b.value);
2715 std::uint64_t bits=0;
2716 for(std::size_t i=0;i<N;++i) bits|=std::uint64_t(x[i] > y[i])<<i;
2717 return mask_type::from_bitset(bits);
2718 }
2719 return mask_type::unsafe_from_native(::NATIVE_BACKEND_NAMESPACE::integer_compare<T, true>(a.value, b.value));
2720 }
2721
2722 native_nodiscard friend native_inline constexpr native_const mask_type operator!=(simd a, simd b) noexcept {
2723 return ~(a == b);
2724 }
2725
2726 native_nodiscard friend native_inline constexpr native_const mask_type operator<(simd a, simd b) noexcept {
2727 return b > a;
2728 }
2729
2730 native_nodiscard friend native_inline constexpr native_const mask_type operator<=(simd a, simd b) noexcept {
2731 return ~(a > b);
2732 }
2733
2734 native_nodiscard friend native_inline constexpr native_const mask_type operator>=(simd a, simd b) noexcept {
2735 return ~(b > a);
2736 }
2737
2738 template <simd_integer_element U>
2739 native_nodiscard friend native_inline constexpr native_const simd operator+(simd a, U b) noexcept {
2740 return a + simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2741 }
2742
2743 template <simd_integer_element U>
2744 native_nodiscard friend native_inline constexpr native_const simd operator+(U a, simd b) noexcept {
2745 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) + b;
2746 }
2747
2748 template <simd_integer_element U>
2749 native_nodiscard friend native_inline constexpr native_const simd operator-(simd a, U b) noexcept {
2750 return a - simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2751 }
2752
2753 template <simd_integer_element U>
2754 native_nodiscard friend native_inline constexpr native_const simd operator-(U a, simd b) noexcept {
2755 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) - b;
2756 }
2757
2758 template <simd_integer_element U>
2759 native_nodiscard friend native_inline constexpr native_const simd operator*(simd a, U b) noexcept {
2760 return a * simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2761 }
2762
2763 template <simd_integer_element U>
2764 native_nodiscard friend native_inline constexpr native_const simd operator*(U a, simd b) noexcept {
2765 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) * b;
2766 }
2767
2768 template <simd_integer_element U>
2769 native_nodiscard friend native_inline constexpr native_const simd operator&(simd a, U b) noexcept {
2770 return a & simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2771 }
2772
2773 template <simd_integer_element U>
2774 native_nodiscard friend native_inline constexpr native_const simd operator&(U a, simd b) noexcept {
2775 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) & b;
2776 }
2777
2778 template <simd_integer_element U>
2779 native_nodiscard friend native_inline constexpr native_const simd operator|(simd a, U b) noexcept {
2780 return a | simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2781 }
2782
2783 template <simd_integer_element U>
2784 native_nodiscard friend native_inline constexpr native_const simd operator|(U a, simd b) noexcept {
2785 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) | b;
2786 }
2787
2788 template <simd_integer_element U>
2789 native_nodiscard friend native_inline constexpr native_const simd operator^(simd a, U b) noexcept {
2790 return a ^ simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2791 }
2792
2793 template <simd_integer_element U>
2794 native_nodiscard friend native_inline constexpr native_const simd operator^(U a, simd b) noexcept {
2795 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) ^ b;
2796 }
2797
2798 template <simd_integer_element U>
2799 native_nodiscard friend native_inline constexpr native_const mask_type operator==(simd a, U b) noexcept {
2800 return a == simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2801 }
2802
2803 template <simd_integer_element U>
2804 native_nodiscard friend native_inline constexpr native_const mask_type operator==(U a, simd b) noexcept {
2805 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) == b;
2806 }
2807
2808 template <simd_integer_element U>
2809 native_nodiscard friend native_inline constexpr native_const mask_type operator!=(simd a, U b) noexcept {
2810 return a != simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2811 }
2812
2813 template <simd_integer_element U>
2814 native_nodiscard friend native_inline constexpr native_const mask_type operator!=(U a, simd b) noexcept {
2815 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) != b;
2816 }
2817
2818 template <simd_integer_element U>
2819 native_nodiscard friend native_inline constexpr native_const mask_type operator<(simd a, U b) noexcept {
2820 return a < simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2821 }
2822
2823 template <simd_integer_element U>
2824 native_nodiscard friend native_inline constexpr native_const mask_type operator<(U a, simd b) noexcept {
2825 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) < b;
2826 }
2827
2828 template <simd_integer_element U>
2829 native_nodiscard friend native_inline constexpr native_const mask_type operator>(simd a, U b) noexcept {
2830 return a > simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2831 }
2832
2833 template <simd_integer_element U>
2834 native_nodiscard friend native_inline constexpr native_const mask_type operator>(U a, simd b) noexcept {
2835 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) > b;
2836 }
2837
2838 template <simd_integer_element U>
2839 native_nodiscard friend native_inline constexpr native_const mask_type operator<=(simd a, U b) noexcept {
2840 return a <= simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2841 }
2842
2843 template <simd_integer_element U>
2844 native_nodiscard friend native_inline constexpr native_const mask_type operator<=(U a, simd b) noexcept {
2845 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) <= b;
2846 }
2847
2848 template <simd_integer_element U>
2849 native_nodiscard friend native_inline constexpr native_const mask_type operator>=(simd a, U b) noexcept {
2850 return a >= simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(b));
2851 }
2852
2853 template <simd_integer_element U>
2854 native_nodiscard friend native_inline constexpr native_const mask_type operator>=(U a, simd b) noexcept {
2855 return simd(::NATIVE_BACKEND_NAMESPACE::integer_wrap<T>(a)) >= b;
2856 }
2857
2858 native_inline constexpr simd &operator+=(simd b) noexcept { return *this = *this + b; }
2860 template <simd_integer_element U> native_inline constexpr simd &operator+=(U b) noexcept { return *this = *this + b; }
2862 native_inline constexpr simd &operator-=(simd b) noexcept { return *this = *this - b; }
2864 template <simd_integer_element U> native_inline constexpr simd &operator-=(U b) noexcept { return *this = *this - b; }
2866 native_inline constexpr simd &operator*=(simd b) noexcept { return *this = *this * b; }
2868 template <simd_integer_element U> native_inline constexpr simd &operator*=(U b) noexcept { return *this = *this * b; }
2870 native_inline constexpr simd &operator&=(simd b) noexcept { return *this = *this & b; }
2872 template <simd_integer_element U> native_inline constexpr simd &operator&=(U b) noexcept { return *this = *this & b; }
2874 native_inline constexpr simd &operator|=(simd b) noexcept { return *this = *this | b; }
2876 template <simd_integer_element U> native_inline constexpr simd &operator|=(U b) noexcept { return *this = *this | b; }
2878 native_inline constexpr simd &operator^=(simd b) noexcept { return *this = *this ^ b; }
2880 template <simd_integer_element U> native_inline constexpr simd &operator^=(U b) noexcept { return *this = *this ^ b; }
2882 template <unsigned K>
2883 requires(K < sizeof(T) * 8)
2885 if consteval {
2886 using E=std::make_unsigned_t<T>;
2887 using V=E __attribute__((ext_vector_type(N)));
2888 return from_native(__builtin_bit_cast(native_type,__builtin_bit_cast(V,value) << K));
2889 }
2890 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_left<T, K>(value));
2891 }
2892
2893 template <unsigned K>
2894 requires(K < sizeof(T) * 8)
2896 if consteval {
2897 using E=T;
2898 using V=E __attribute__((ext_vector_type(N)));
2899 return from_native(__builtin_bit_cast(native_type,__builtin_bit_cast(V,value) >> K));
2900 }
2901 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_right<T, K>(value));
2902 }
2903
2904 template <std::size_t K>
2905 requires(K < sizeof(T) * 8)
2906 native_nodiscard friend native_inline constexpr native_const simd operator<<(simd a, imm_t<K>) noexcept {
2907 return a.template left<K>();
2908 }
2909
2911 requires std::same_as<T,std::uint32_t> && requires(native_type x) {
2912 ::NATIVE_BACKEND_NAMESPACE::integer_shift_left_variable(x,x);
2913 }
2914 {
2915 if consteval {
2916 auto values=__builtin_bit_cast(std::array<std::uint32_t,N>,a.value);
2917 auto shifts=__builtin_bit_cast(std::array<std::uint32_t,N>,counts.value);
2918 for(std::size_t i=0;i<N;++i) values[i]=shifts[i]<32 ? values[i]<<shifts[i] : 0;
2919 return simd(values);
2920 }
2921 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_shift_left_variable(a.value,counts.value));
2922 }
2923
2924 template <std::size_t K>
2925 requires(K < sizeof(T) * 8)
2926 native_nodiscard friend native_inline constexpr native_const simd operator>>(simd a, imm_t<K>) noexcept {
2927 return a.template right<K>();
2928 }
2929
2930 native_nodiscard native_inline static native_pure constexpr simd load(T const *p) noexcept {
2931 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2932 }
2933
2934 native_nodiscard native_inline static native_pure constexpr simd loadu(T const *p) noexcept {
2935 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(p);
2936 }
2937
2938 native_inline constexpr void store(T *p) const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *this); }
2940 native_inline constexpr void storeu(T *p) const noexcept { ::NATIVE_BACKEND_NAMESPACE::store_simd(p, *this); }
2942 native_nodiscard native_inline constexpr static native_pure simd load_partial(T const *p, std::size_t n,
2943 T fill = T(0)) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") {
2944 assert(n <= lanes);
2945 if consteval {
2946 std::array<T,N> data; data.fill(fill);
2947 for(std::size_t i=0;i<n;++i) data[i]=p[i];
2948 return simd(data);
2949 }
2950 if (n == lanes) return load(p);
2951 if (!n) return simd(fill);
2952 // Native 32/64-bit lane tails must not touch masked-off addresses. This
2953 // also accepts byte-unaligned sources; scalar tail loads use memcpy.
2954#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
2955 if constexpr (sizeof(T) == 4 || sizeof(T) == 8) {
2956 auto active = std::uint64_t((std::uint64_t(1) << n) - 1);
2957 if constexpr (sizeof(T) * N == 64) {
2958 if constexpr (sizeof(T) == 4)
2959 return from_native(_mm512_mask_loadu_epi32(simd(fill).value, __mmask16(active), p));
2960 else return from_native(_mm512_mask_loadu_epi64(simd(fill).value, __mmask8(active), p));
2961 }
2962#if NATIVE_HAS_AVX512VL
2963 else if constexpr (sizeof(T) * N == 32) {
2964 if constexpr (sizeof(T) == 4)
2965 return from_native(_mm256_mask_loadu_epi32(simd(fill).value, __mmask8(active), p));
2966 else return from_native(_mm256_mask_loadu_epi64(simd(fill).value, __mmask8(active), p));
2967 } else {
2968 if constexpr (sizeof(T) == 4)
2969 return from_native(_mm_mask_loadu_epi32(simd(fill).value, __mmask8(active), p));
2970 else return from_native(_mm_mask_loadu_epi64(simd(fill).value, __mmask8(active), p));
2971 }
2972#endif
2973 }
2974#endif
2975#if NATIVE_HAS_AVX2
2976 if constexpr ((sizeof(T) == 4 || sizeof(T) == 8) && !mask_type::compact && sizeof(T) * N <= 32) {
2977 native_type active, loaded;
2978 if constexpr (sizeof(T) * N == 32) {
2979 if constexpr (sizeof(T) == 4) {
2980 active = _mm256_cmpgt_epi32(_mm256_set1_epi32(int(n)), _mm256_setr_epi32(0,1,2,3,4,5,6,7));
2981 loaded = _mm256_maskload_epi32(reinterpret_cast<int const *>(p), active);
2982 } else {
2983 active = _mm256_cmpgt_epi64(_mm256_set1_epi64x(static_cast<long long>(n)), _mm256_setr_epi64x(0,1,2,3));
2984 loaded = _mm256_maskload_epi64(reinterpret_cast<long long const *>(p), active);
2985 }
2986 } else if constexpr (sizeof(T) * N == 16) {
2987 if constexpr (sizeof(T) == 4) {
2988 active = _mm_cmpgt_epi32(_mm_set1_epi32(int(n)), _mm_setr_epi32(0,1,2,3));
2989 loaded = _mm_maskload_epi32(reinterpret_cast<int const *>(p), active);
2990 } else {
2991 active = _mm_cmpgt_epi64(_mm_set1_epi64x(static_cast<long long>(n)), _mm_set_epi64x(1,0));
2992 loaded = _mm_maskload_epi64(reinterpret_cast<long long const *>(p), active);
2993 }
2994 }
2995 return from_native(::NATIVE_BACKEND_NAMESPACE::integer_bit_select(active, loaded, simd(fill).value));
2996 }
2997#endif
2998#if NATIVE_HAS_ARM_NEON
2999 if constexpr (sizeof(T) == 4) {
3000 auto result = vreinterpretq_u32_u8(simd(fill).value);
3001 std::uint32_t word;
3002 auto bytes = reinterpret_cast<unsigned char const *>(p);
3003 switch (n) {
3004 case 3: std::memcpy(&word, bytes + 8, 4); result = vsetq_lane_u32(word, result, 2); [[fallthrough]];
3005 case 2: std::memcpy(&word, bytes + 4, 4); result = vsetq_lane_u32(word, result, 1); [[fallthrough]];
3006 case 1: std::memcpy(&word, bytes, 4); result = vsetq_lane_u32(word, result, 0);
3007 }
3008 return from_native(vreinterpretq_u8_u32(result));
3009 } else if constexpr (sizeof(T) == 8) {
3010 std::uint64_t word;
3011 std::memcpy(&word, static_cast<void const *>(p), 8);
3012 return from_native(vreinterpretq_u8_u64(vsetq_lane_u64(word, vreinterpretq_u64_u8(simd(fill).value), 0)));
3013 }
3014#endif
3015 std::array<T, lanes> data;
3016 data.fill(fill);
3017 if (n)
3018 std::memcpy(data.data(), static_cast<void const *>(p), n * sizeof(T));
3019 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, lanes>(data.data());
3020 }
3021
3022 native_inline constexpr void store_partial(T *p, std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") {
3023 if consteval {
3024 assert(n<=N);
3025 auto data=__builtin_bit_cast(std::array<T,N>,value);
3026 for(std::size_t i=0;i<n;++i) p[i]=data[i];
3027 return;
3028 } ::NATIVE_BACKEND_NAMESPACE::store_simd_partial(p, *this, n); }
3029
3030 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator+(simd, U) = delete;
3032 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator+(U, simd) = delete;
3034 template <simd_integer_element U, std::size_t M>
3035 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3036 friend void operator+(simd, ::native::simd<U, M,Arch>) = delete;
3038 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator-(simd, U) = delete;
3040 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator-(U, simd) = delete;
3042 template <simd_integer_element U, std::size_t M>
3043 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3044 friend void operator-(simd, ::native::simd<U, M,Arch>) = delete;
3046 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator*(simd, U) = delete;
3048 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator*(U, simd) = delete;
3050 template <simd_integer_element U, std::size_t M>
3051 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3052 friend void operator*(simd, ::native::simd<U, M,Arch>) = delete;
3054 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator/(simd, U) = delete;
3056 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator/(U, simd) = delete;
3058 template <simd_integer_element U, std::size_t M>
3059 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3060 friend void operator/(simd, ::native::simd<U, M,Arch>) = delete;
3062 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator%(simd, U) = delete;
3064 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator%(U, simd) = delete;
3066 template <simd_integer_element U, std::size_t M>
3067 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3068 friend void operator%(simd, ::native::simd<U, M,Arch>) = delete;
3070 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator&(simd, U) = delete;
3072 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator&(U, simd) = delete;
3074 template <simd_integer_element U, std::size_t M>
3075 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3076 friend void operator&(simd, ::native::simd<U, M,Arch>) = delete;
3078 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator|(simd, U) = delete;
3080 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator|(U, simd) = delete;
3082 template <simd_integer_element U, std::size_t M>
3083 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3084 friend void operator|(simd, ::native::simd<U, M,Arch>) = delete;
3086 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator^(simd, U) = delete;
3088 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator^(U, simd) = delete;
3090 template <simd_integer_element U, std::size_t M>
3091 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3092 friend void operator^(simd, ::native::simd<U, M,Arch>) = delete;
3094 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator==(simd, U) = delete;
3096 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator==(U, simd) = delete;
3098 template <simd_integer_element U, std::size_t M>
3099 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3100 friend void operator==(simd, ::native::simd<U, M,Arch>) = delete;
3102 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator!=(simd, U) = delete;
3104 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator!=(U, simd) = delete;
3106 template <simd_integer_element U, std::size_t M>
3107 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3108 friend void operator!=(simd, ::native::simd<U, M,Arch>) = delete;
3110 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<(simd, U) = delete;
3112 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<(U, simd) = delete;
3114 template <simd_integer_element U, std::size_t M>
3115 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3116 friend void operator<(simd, ::native::simd<U, M,Arch>) = delete;
3118 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>(simd, U) = delete;
3120 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>(U, simd) = delete;
3122 template <simd_integer_element U, std::size_t M>
3123 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3124 friend void operator>(simd, ::native::simd<U, M,Arch>) = delete;
3126 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<=(simd, U) = delete;
3128 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator<=(U, simd) = delete;
3130 template <simd_integer_element U, std::size_t M>
3131 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3132 friend void operator<=(simd, ::native::simd<U, M,Arch>) = delete;
3134 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>=(simd, U) = delete;
3136 template <class U> requires (std::is_arithmetic_v<U> && !simd_integer_element<U>) friend void operator>=(U, simd) = delete;
3138 template <simd_integer_element U, std::size_t M>
3139 requires(!std::same_as<simd, ::native::simd<U, M,Arch>>)
3140 friend void operator>=(simd, ::native::simd<U, M,Arch>) = delete;
3142 friend simd operator/(simd, simd) = delete;
3144 friend simd operator%(simd, simd) = delete;
3146 template <simd_integer_element U> friend simd operator/(simd, U) = delete;
3148 template <simd_integer_element U> friend simd operator/(U, simd) = delete;
3150 template <simd_integer_element U> friend simd operator%(simd, U) = delete;
3152 template <simd_integer_element U> friend simd operator%(U, simd) = delete;
3154 template <simd_integer_element U> friend simd operator<<(simd, U) = delete;
3156 template <simd_integer_element U> friend simd operator>>(simd, U) = delete;
3157 };
3158
3159#endif
3160
3161 namespace detail::NATIVE_BACKEND {
3162 template <::native::isa<> Arch, simd_integer_element T, std::size_t N, std::size_t A, simd_access Access>
3163 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3165 using V = simd<T, N,Arch>;
3166 if consteval {
3167 if constexpr(N==1) return V(*p);
3168 else {
3169 std::array<T,N> data;
3170 for(std::size_t i=0;i<N;++i) data[i]=p[i];
3171 return V(data);
3172 }
3173 }
3174 typename V::native_type value;
3175 // Erase the element type at the byte-copy boundary: p need not satisfy
3176 // alignof(T). Only an explicit alignment promise may select an aligned
3177 // move. Streaming currently takes this ordinary fallback.
3178#if defined(__GNUC__) || defined(__clang__)
3179 if constexpr (A > 1)
3180 p = static_cast<T const *>(__builtin_assume_aligned(p, A));
3181#elif defined(_MSC_VER)
3182 if constexpr (A > 1)
3183 __assume((reinterpret_cast<std::uintptr_t>(p) & (A - 1)) == 0);
3184#endif
3185 std::memcpy(&value, static_cast<void const *>(p), sizeof(value));
3186 return V::from_native(value);
3187 }
3188 template <simd_integer_element T, std::size_t N, std::size_t A, simd_access Access, ::native::isa<> Arch>
3189 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3190 native_inline constexpr void store_simd(T *p, simd<T, N,Arch> v, simd_memory<A, Access>) noexcept {
3191 if consteval {
3192 if constexpr(N==1) *p=v.value;
3193 else {
3194 auto data=__builtin_bit_cast(std::array<T,N>,v.value);
3195 for(std::size_t i=0;i<N;++i) p[i]=data[i];
3196 }
3197 return;
3198 }
3199#if defined(__GNUC__) || defined(__clang__)
3200 if constexpr (A > 1)
3201 p = static_cast<T *>(__builtin_assume_aligned(p, A));
3202#elif defined(_MSC_VER)
3203 if constexpr (A > 1)
3204 __assume((reinterpret_cast<std::uintptr_t>(p) & (A - 1)) == 0);
3205#endif
3206 std::memcpy(static_cast<void *>(p), &v.value, sizeof(v.value));
3207 }
3208 template <::native::isa<> Arch, simd_integer_element T, std::size_t N, std::size_t A, simd_access Access>
3209 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3210 native_nodiscard native_inline constexpr native_pure simd<T, N,Arch> load_simd_partial(T const *p, std::size_t count,
3211 simd_memory<A, Access>) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") {
3212 assert(count <= N);
3213 std::array<T, N> data{};
3214 if consteval { for(std::size_t i=0;i<count;++i) data[i]=p[i]; }
3215 else { if (count) std::memcpy(data.data(), static_cast<void const *>(p), count * sizeof(T)); }
3216 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, N>(data.data());
3217 }
3218 template <simd_integer_element T, std::size_t N, std::size_t A, simd_access Access, ::native::isa<> Arch>
3219 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3220 native_inline constexpr void store_simd_partial(T *p, simd<T, N,Arch> v, std::size_t count,
3221 simd_memory<A, Access>) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") {
3222 assert(count <= N);
3223 std::array<T, N> data;
3224 store_simd(data.data(), v);
3225 if consteval { for(std::size_t i=0;i<count;++i) p[i]=data[i]; }
3226 else { if (count) std::memcpy(static_cast<void *>(p), data.data(), count * sizeof(T)); }
3227 }
3228 template <::native::isa<> Arch, simd_integer_element T, std::size_t N>
3229 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3230 native_nodiscard native_inline constexpr native_pure simd<T, N,Arch> load_simd(std::array<T, N> const &p) noexcept {
3231 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, N>(p.data());
3232 }
3233 template <::native::isa<> Arch, simd_integer_element T, std::size_t N>
3234 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3235 native_nodiscard native_inline constexpr native_pure simd<T, N,Arch> load_simd(std::span<T const, N> p) noexcept {
3236 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T, N>(p.data());
3237 }
3238
3239
3240 }
3241 namespace detail::NATIVE_BACKEND {
3242 template <class M, class T, std::size_t N, isa<> Arch>
3243 concept integer_mask_for =
3244 std::same_as<M, typename simd<T, N,Arch>::mask_type> || std::same_as<M, simd<::NATIVE_BACKEND_NAMESPACE::mask_lane_for<T>, N,Arch>>;
3245 }
3248 template <simd_integer_element T, std::size_t N, class M, ::native::isa<> Arch>
3249 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3251 if consteval {
3252 std::array<T,N> first{},second{};
3253 a.store(first.data()); b.store(second.data());
3254 auto bits=m.to_bitset();
3255 for(std::size_t i=0;i<N;++i) if(!((bits>>i)&1)) first[i]=second[i];
3256 return simd<T,N,Arch>(first);
3257 }
3258 if constexpr (N == 1)
3259 return any(m) ? a : b;
3260#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3261 else
3262 return simd<T, N,Arch>::from_native(::NATIVE_BACKEND_NAMESPACE::integer_select<T>(m, a.value, b.value));
3263#endif
3264 }
3265
3267 template <simd_integer_element T, std::size_t N, ::native::isa<> Arch>
3268 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3270 simd<T, N,Arch> b) noexcept {
3271 if consteval { return (bits & a) | (~bits & b); }
3272 if constexpr(N==1) return (bits & a) | (~bits & b);
3273#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3274 else return simd<T,N,Arch>::from_native(::NATIVE_BACKEND_NAMESPACE::integer_bit_select(bits.value,a.value,b.value));
3275#endif
3276 }
3277 // Compatibility: numeric masks select individual bits, never lane truth values.
3280 template <simd_integer_element T, std::size_t N, ::native::isa<> Arch>
3281 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N>
3286
3288 template <simd_mask_element M, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch)
3289 native_nodiscard native_inline constexpr native_const auto mask_bits(simd<M, N,Arch> m) noexcept {
3290 using U = typename M::storage_type;
3291 return simd<U, N,Arch>::from_native(m.to_native());
3292 }
3293
3295 template <simd_integer_element T, simd_mask_element M, std::size_t N, ::native::isa<> Arch>
3296 requires NATIVE_ARCH_REQUIRES(Arch) &&(sizeof(T) == sizeof(M))
3297 native_nodiscard native_inline constexpr native_const simd<std::make_unsigned_t<T>, N,Arch> mask_bits(simd<M, N,Arch> m) noexcept {
3298 return mask_bits(m);
3299 }
3300
3302 template <simd_integer_element T, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) &&
3303 requires(predicate<N,Arch> m) { to_vector_mask<::NATIVE_BACKEND_NAMESPACE::mask_lane_for<T>>(m); }
3305 return mask_bits(to_vector_mask<::NATIVE_BACKEND_NAMESPACE::mask_lane_for<T>>(m));
3306 }
3307
3309 template <simd_integer_element T, std::size_t N, class M, ::native::isa<> Arch>
3310 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3312 simd<T, N,Arch> b) noexcept {
3313 if consteval { return select(m,a + b,prior); }
3314 if constexpr (N == 1)
3315 return select(m, a + b, prior);
3316#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3317 else
3318 return simd<T, N,Arch>::from_native(::NATIVE_BACKEND_NAMESPACE::integer_masked_add<T>(m, prior.value, a.value, b.value));
3319#endif
3320 }
3321
3323 template <simd_integer_element T, std::size_t N, class M, ::native::isa<> Arch>
3324 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3328
3330 template <simd_integer_element T, std::size_t N, class M, ::native::isa<> Arch>
3331 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3333 simd<T, N,Arch> b) noexcept {
3334 if consteval { return select(m,a - b,prior); }
3335 if constexpr (N == 1)
3336 return select(m, a - b, prior);
3337#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3338 else
3339 return simd<T, N,Arch>::from_native(::NATIVE_BACKEND_NAMESPACE::integer_masked_sub<T>(m, prior.value, a.value, b.value));
3340#endif
3341 }
3342
3344 template <simd_integer_element T, std::size_t N, class M, ::native::isa<> Arch>
3345 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3349
3351 template <simd_integer_element T, std::size_t N, class M, ::native::isa<> Arch>
3352 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3354 simd<T, N,Arch> b) noexcept {
3355 if consteval { return select(m,a * b,prior); }
3356 if constexpr (N == 1)
3357 return select(m, a * b, prior);
3358#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON || (NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ)
3359 else
3360 return simd<T, N,Arch>::from_native(::NATIVE_BACKEND_NAMESPACE::integer_masked_mul<T>(m, prior.value, a.value, b.value));
3361#endif
3362 }
3363
3365 template <simd_integer_element T, std::size_t N, class M, ::native::isa<> Arch>
3366 requires NATIVE_ARCH_REQUIRES(Arch) &&(::NATIVE_BACKEND_NAMESPACE::integer_shape<T, N> && ::NATIVE_BACKEND_NAMESPACE::integer_mask_for<M, T, N, Arch>)
3370} // namespace native
3371
3372// Reject invalid immediate widths before implicit native conversion can select
3373// a builtin scalar shift. The immediate tag retains its public size conversion.
3374namespace native {
3376 template<simd_integer_element T, std::size_t N, std::size_t K, ::native::isa<> Arch>
3377 requires NATIVE_ARCH_REQUIRES(Arch) && (K >= sizeof(T) * 8)
3378 void operator<<(simd<T, N,Arch>, imm_t<K>) = delete;
3380 template<simd_integer_element T, std::size_t N, std::size_t K, ::native::isa<> Arch>
3381 requires NATIVE_ARCH_REQUIRES(Arch) && (K >= sizeof(T) * 8)
3382 void operator>>(simd<T, N,Arch>, imm_t<K>) = delete;
3383}
3384
3385namespace NATIVE_BACKEND_NAMESPACE::native {
3386 // Compatibility names point toward the public native class template.
3387 using uint32x1=::native::simd<::native::uint32_t,1,NATIVE_DEFAULT_ARCH>;
3388#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
3389 using uint32x4=::native::simd<::native::uint32_t,4,NATIVE_DEFAULT_ARCH>;
3390#endif
3391#if NATIVE_HAS_AVX2
3392 using uint32x8=::native::simd<::native::uint32_t,8,NATIVE_DEFAULT_ARCH>;
3393#endif
3394#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
3395 using uint32x16=::native::simd<::native::uint32_t,16,NATIVE_DEFAULT_ARCH>;
3396#endif
3397 template<class V> concept unsigned_register = requires {
3398 typename V::unsigned_register_tag;
3399 typename V::value_type;
3400 } && std::same_as<typename V::value_type,::native::uint32_t>;
3401 template<unsigned S,unsigned_register V> requires(S<32)
3402 native_nodiscard native_inline native_const V shift_left(V x) noexcept { return x.template left<S>(); }
3403 template<unsigned S,unsigned_register V> requires(S<32)
3404 native_nodiscard native_inline native_const V shift_right(V x) noexcept { return x.template right<S>(); }
3405 using ::native::mask_bits;
3406 using ::native::bit_select;
3407 struct integer_target {
3408#if defined(NATIVE_INTEGER_SCALAR)
3409 using unsigned_type=uint32x1;
3410#elif NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
3411 using unsigned_type=uint32x16;
3412#elif NATIVE_HAS_AVX2
3413 using unsigned_type=uint32x8;
3414#elif NATIVE_HAS_ARM_NEON
3415 using unsigned_type=uint32x4;
3416#else
3417 using unsigned_type=uint32x1;
3418#endif
3419 static constexpr std::size_t lanes=unsigned_type::lanes;
3420 };
3421}
3422
3423namespace native {
3424 namespace detail::NATIVE_BACKEND {
3425 template <class V,std::size_t Alignment> native_nodiscard native_artificial native_inline constexpr native_pure V simd_load_native(native_noescape float const *) noexcept;
3426 template <class V,std::size_t Alignment> native_artificial native_inline constexpr void simd_store_native(native_noescape float *,V) noexcept;
3427 }
3428 namespace detail::NATIVE_BACKEND {
3429 template <class T> inline constexpr bool custom_argument = ::native::simd_custom_element<std::remove_cvref_t<T>>;
3430 template <class T, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) inline constexpr bool custom_argument<simd<T,N,Arch>> = ::native::simd_custom_element<T>;
3431 }
3432 namespace detail::NATIVE_BACKEND {
3433
3434 using ::native::detail::register_memory;
3435 }
3436 template <::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) struct native_empty_bases simd<float, 1,Arch> : ::NATIVE_BACKEND_NAMESPACE::register_memory<simd<float,1,Arch>, 1>, detail::swizzle_access<float,1,Arch> {
3437 static constexpr isa<> architecture=Arch;
3438 template <class T> using rebind = simd<T,1,Arch>;
3439 using vector_mask_type=simd<mask32,1,Arch>;
3440 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<float,1,Arch>;
3441 using mask = mask_type;
3442 using predicate_type = predicate<1,Arch>;
3443 float value{};
3445 native_inline simd() = default;
3447 native_inline constexpr simd(float x) : value(x) {}
3449 native_nodiscard static native_inline constexpr native_pure simd load(native_noescape float const * p) { return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,1>(p); }
3451 native_inline constexpr void store(native_noescape float * p) const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*this); }
3453 native_nodiscard friend native_inline constexpr native_pure simd operator+(simd a, simd b) {
3454 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3455 return simd(a.value + b.value);
3456 }
3458 native_nodiscard friend native_inline constexpr native_pure simd operator-(simd a, simd b) {
3459 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
3460 return simd(a.value - b.value);
3461 }
3463 native_nodiscard friend native_inline constexpr native_pure simd operator*(simd a, simd b) {
3464 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
3465 return simd(a.value * b.value);
3466 }
3468 native_nodiscard friend native_inline constexpr native_pure simd operator/(simd a, simd b) {
3469 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
3470 return simd(a.value / b.value);
3471 }
3473 native_nodiscard friend native_inline constexpr native_const simd operator-(simd a) {
3474 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
3475 return simd(-a.value);
3476 }
3478 native_nodiscard friend native_inline constexpr native_const mask_type operator<(simd a, simd b) {
3479 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
3480 return mask_type::from_native(a.value < b.value ? ~std::uint32_t(0) : 0u);
3481 }
3483 native_nodiscard friend native_inline constexpr native_const mask_type operator>(simd a, simd b) {
3484 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
3485 return mask_type::from_native(a.value > b.value ? ~std::uint32_t(0) : 0u);
3486 }
3488 native_nodiscard friend native_inline constexpr native_const mask_type operator==(simd a, simd b) {
3489 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
3490 return mask_type::from_native(a.value == b.value ? ~std::uint32_t(0) : 0u);
3491 }
3493 template<class M> requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
3494 native_nodiscard friend native_inline constexpr native_const simd select(M m,simd a,simd b) {
3495 if consteval { return ::native::detail::float_constant::select(m,a,b); }
3496 return m.to_native()!=0 ? a : b;
3497 }
3499 native_nodiscard friend native_inline constexpr simd fma(simd a, simd b, simd c) {
3500 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
3501 return simd(std::fma(a.value, b.value, c.value));
3502 }
3504 native_nodiscard friend native_inline constexpr simd sqrt(simd a) {
3505 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
3506 return simd(std::sqrt(a.value));
3507 }
3509 native_nodiscard friend native_inline constexpr native_pure simd round_even(simd a) {
3510 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
3511 if (!std::isfinite(a.value) || std::abs(a.value) >= 0x1p23f) return a;
3512 float lo = std::floor(a.value), delta = a.value - lo;
3513 float r = lo + float(delta > .5f || (delta == .5f && std::fmod(lo, 2.f) != 0));
3514 return simd(r == 0 ? std::copysign(0.f, a.value) : r);
3515 }
3516 // Internal precondition: integral n in [-126,127].
3518 native_nodiscard friend native_inline constexpr native_const simd normal_pow2(simd n) {
3519 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
3520 return simd(std::bit_cast<float>(std::uint32_t(int(n.value) + 127) << 23));
3521 }
3522
3523 template <std::size_t Alignment = 1>
3525 native_nodiscard static native_inline constexpr native_pure simd load_memory(float const * p) noexcept {
3526 return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,Alignment>(p);
3527 }
3528 template <std::size_t Alignment = 1>
3530 native_inline constexpr void store_memory(float * p) const noexcept {
3531 ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,Alignment>(p,*this);
3532 }
3533 using value_type = float;
3534 using register_type = simd;
3535 using native_type = float;
3536 using bits_type = simd<uint32_t,1,Arch>;
3538 native_nodiscard native_inline constexpr native_pure operator native_type() const noexcept { return value; }
3540 native_nodiscard native_inline constexpr native_pure native_type to_native() const noexcept { return value; }
3542 native_nodiscard native_inline constexpr native_pure bits_type bits() const noexcept { return bits_type(std::bit_cast<std::uint32_t>(value)); }
3544 native_nodiscard native_inline constexpr native_pure bits_type to_bits() const noexcept { return bits(); }
3546 native_nodiscard static native_inline constexpr native_const simd from_bits(bits_type bits) noexcept { return simd(std::bit_cast<float>(bits.value)); }
3548 native_nodiscard static native_inline constexpr native_const simd from_bits(std::uint32_t bits) noexcept { return from_bits(bits_type(bits)); }
3550 native_nodiscard static native_inline constexpr native_const simd from_float(float x) noexcept { return simd(x); }
3552 native_nodiscard static native_inline constexpr native_const simd from_native(native_type x) noexcept { return simd(x); }
3554 native_nodiscard static native_inline constexpr native_const simd unsafe_from_float32(native_type x) noexcept { return simd(x); }
3556 native_nodiscard static native_inline constexpr native_pure simd loadu(native_noescape float const * p) { return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,1>(p); }
3558 native_inline constexpr void storeu(native_noescape float * p) const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*this); }
3560 native_nodiscard static native_inline constexpr native_pure simd load_bits(native_noescape std::uint32_t const * p) noexcept { return from_bits(bits_type::load(p)); }
3562 native_inline constexpr void store_bits(native_noescape std::uint32_t * p) const noexcept { bits().store(p); }
3564 native_nodiscard static native_inline constexpr native_pure simd load_bits_partial(native_noescape std::uint32_t const * p,std::size_t n,std::uint32_t fill=0) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { return from_bits(bits_type::load_partial(p,n,fill)); }
3566 native_inline constexpr void store_bits_partial(native_noescape std::uint32_t * p,std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { bits().store_partial(p,n); }
3568 native_inline constexpr simd(std::array<float,1> const & values) noexcept : simd(loadu(values.data())) {}
3570 native_inline constexpr simd & operator+=(simd b) noexcept { return *this=*this+b; }
3572 native_inline constexpr simd & operator-=(simd b) noexcept { return *this=*this-b; }
3574 native_inline constexpr simd & operator*=(simd b) noexcept { return *this=*this*b; }
3576 native_inline constexpr simd & operator/=(simd b) noexcept { return *this=*this/b; }
3578 native_nodiscard friend native_inline constexpr native_const mask_type operator!=(simd a,simd b) noexcept { return ~(a==b); }
3580 native_nodiscard friend native_inline constexpr native_const mask_type operator<=(simd a,simd b) noexcept { return (a<b)|(a==b); }
3582 native_nodiscard friend native_inline constexpr native_const mask_type operator>=(simd a,simd b) noexcept { return (a>b)|(a==b); }
3583 };
3584#if NATIVE_HAS_AVX2
3585}
3586
3587namespace native {
3590 template <::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) struct native_empty_bases simd<float, 4,Arch> : detail::register_memory<simd<float,4,Arch>, 4>, detail::swizzle_access<float,4,Arch> {
3591 static constexpr isa<> architecture=Arch;
3592 template <class T> using rebind = simd<T,4,Arch>;
3593 using vector_mask_type=simd<mask32,4,Arch>;
3594 using mask_type=std::conditional_t<bool(NATIVE_HAS_AVX512VL),predicate<4,Arch>,simd<mask32,4,Arch>>;
3595 using mask = mask_type;
3596 using predicate_type = predicate<4,Arch>;
3597 __m128 value;
3599 native_inline simd() = default;
3601 native_inline constexpr simd(simd const &) = default;
3603 native_reinitializes native_inline constexpr simd & operator=(simd const &) = default;
3605 native_inline constexpr simd(float x) {
3606 if consteval { std::array<float,sizeof(native_type)/sizeof(float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
3607 else { value=_mm_set1_ps(x); }
3608 }
3609
3610 native_inline native_target("sse") constexpr simd(__m128 x) : value(x) {}
3612 native_nodiscard static native_inline constexpr native_pure simd load(native_noescape float const * p) { return load_memory<1>(p); }
3614 native_inline constexpr void store(native_noescape float * p) const { store_memory<1>(p); }
3617 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3618 return simd(_mm_add_ps(a.value, b.value));
3619 }
3620
3622 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
3623 return simd(_mm_sub_ps(a.value, b.value));
3624 }
3625
3627 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
3628 return simd(_mm_mul_ps(a.value, b.value));
3629 }
3630
3632 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
3633 return simd(_mm_div_ps(a.value, b.value));
3634 }
3635
3637 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
3638 return simd(_mm_xor_ps(a.value, _mm_set1_ps(-0.f)));
3639 }
3640
3642 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
3643 if constexpr(bool(NATIVE_HAS_AVX512VL))
3644 return mask_type::from_native(_mm_cmp_ps_mask(a.value,b.value,_CMP_LT_OQ));
3645 else
3646 return mask_type::unsafe_from_native(_mm_castps_si128(_mm_cmp_ps(a.value,b.value,_CMP_LT_OQ)));
3647 }
3648
3650 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
3651 if constexpr(bool(NATIVE_HAS_AVX512VL))
3652 return mask_type::from_native(_mm_cmp_ps_mask(a.value,b.value,_CMP_GT_OQ));
3653 else
3654 return mask_type::unsafe_from_native(_mm_castps_si128(_mm_cmp_ps(a.value,b.value,_CMP_GT_OQ)));
3655 }
3656
3658 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
3659 if constexpr(bool(NATIVE_HAS_AVX512VL))
3660 return mask_type::from_native(_mm_cmp_ps_mask(a.value,b.value,_CMP_EQ_OQ));
3661 else
3662 return mask_type::unsafe_from_native(_mm_castps_si128(_mm_cmp_ps(a.value,b.value,_CMP_EQ_OQ)));
3663 }
3664
3665 template<class M> requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
3667 if consteval { return ::native::detail::float_constant::select(m,a,b); }
3668 if constexpr(M::compact) return simd(_mm_mask_blend_ps(m.to_native(),b.value,a.value));
3669 else
3670 return simd(_mm_blendv_ps(b.value,a.value,_mm_castsi128_ps(m.to_native()))); }
3671
3673 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
3674 return simd(_mm_fmadd_ps(a.value, b.value, c.value));
3675 }
3676
3678 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
3679 return simd(_mm_sqrt_ps(a.value));
3680 }
3681
3683 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
3684 return simd(_mm_round_ps(a.value, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
3685 }
3686
3688 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
3689 return simd(_mm_castsi128_ps(_mm_slli_epi32(_mm_add_epi32(_mm_cvttps_epi32(n.value), _mm_set1_epi32(127)), 23)));
3690 }
3691
3692 template <std::size_t Alignment = 1>
3694 native_nodiscard static native_inline constexpr native_pure simd load_memory(float const * p) noexcept {
3695 if consteval {
3696 std::array<float,sizeof(native_type)/sizeof(float)> values{};
3697 for (std::size_t i=0;i<sizeof(native_type)/sizeof(float);++i) values[i]=p[i];
3698 return from_native(__builtin_bit_cast(native_type,values));
3699 }
3700 if constexpr(Alignment>=16) return simd(_mm_load_ps(p));
3701 else return simd(_mm_loadu_ps(p));
3702 }
3703 template <std::size_t Alignment = 1>
3705 native_inline constexpr void store_memory(float * p) const noexcept {
3706 if consteval {
3707 auto values=__builtin_bit_cast(std::array<float,sizeof(native_type)/sizeof(float)>,value);
3708 for (std::size_t i=0;i<sizeof(native_type)/sizeof(float);++i) p[i]=values[i];
3709 return;
3710 }
3711 if constexpr(Alignment>=16) _mm_store_ps(p,value);
3712 else _mm_storeu_ps(p,value);
3713 }
3714 using value_type = float;
3715 using register_type = simd;
3716 using native_type = __m128;
3717 using bits_type = simd<uint32_t,4,Arch>;
3719 native_nodiscard native_inline constexpr native_pure native_target("sse") operator native_type() const noexcept { return value; }
3721 native_nodiscard native_inline constexpr native_pure native_target("sse") native_type to_native() const noexcept { return value; }
3723 native_artificial native_nodiscard native_inline constexpr native_pure bits_type bits() const noexcept { if consteval { return bits_type::from_native(__builtin_bit_cast(typename bits_type::native_type,value)); } return bits_type::from_native(_mm_castps_si128(value)); }
3725 native_nodiscard native_inline constexpr native_pure bits_type to_bits() const noexcept { return bits(); }
3727 native_artificial native_nodiscard static native_inline constexpr native_const simd from_bits(bits_type bits) noexcept { if consteval { return simd(__builtin_bit_cast(native_type,bits.to_native())); } return simd(_mm_castsi128_ps(bits.value)); }
3729 native_nodiscard static native_inline constexpr native_const simd from_bits(std::uint32_t bits) noexcept { return from_bits(bits_type(bits)); }
3731 native_nodiscard static native_inline constexpr native_const simd from_float(float x) noexcept { return simd(x); }
3733 native_nodiscard static native_inline constexpr native_const native_target("sse") simd from_native(native_type x) noexcept { return simd(x); }
3735 native_nodiscard static native_inline constexpr native_const native_target("sse") simd unsafe_from_float32(native_type x) noexcept { return simd(x); }
3737 native_nodiscard static native_inline constexpr native_pure simd loadu(native_noescape float const * p) { return load_memory<1>(p); }
3739 native_inline constexpr void storeu(native_noescape float * p) const { store_memory<1>(p); }
3741 native_nodiscard static native_inline constexpr native_pure simd load_bits(native_noescape std::uint32_t const * p) noexcept { return from_bits(bits_type::load(p)); }
3743 native_inline constexpr void store_bits(native_noescape std::uint32_t * p) const noexcept { bits().store(p); }
3745 native_nodiscard static native_inline constexpr native_pure simd load_bits_partial(native_noescape std::uint32_t const * p,std::size_t n,std::uint32_t fill=0) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { return from_bits(bits_type::load_partial(p,n,fill)); }
3747 native_inline constexpr void store_bits_partial(native_noescape std::uint32_t * p,std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { bits().store_partial(p,n); }
3749 native_inline constexpr simd(std::array<float,4> const & values) noexcept : simd(loadu(values.data())) {}
3750#if defined(__clang__)
3752 template <class... X> requires (sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
3753 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : value{static_cast<float>(x)...} {}
3754#else
3756 template <class... X> requires (sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
3757 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : simd(loadu(std::array<float,4>{static_cast<float>(x)...}.data())) {}
3758#endif
3760 native_inline constexpr simd & operator+=(simd b) noexcept { return *this=*this+b; }
3762 native_inline constexpr simd & operator-=(simd b) noexcept { return *this=*this-b; }
3764 native_inline constexpr simd & operator*=(simd b) noexcept { return *this=*this*b; }
3766 native_inline constexpr simd & operator/=(simd b) noexcept { return *this=*this/b; }
3768 native_nodiscard friend native_inline constexpr native_const mask_type operator!=(simd a,simd b) noexcept { return ~(a==b); }
3770 native_nodiscard friend native_inline constexpr native_const mask_type operator<=(simd a,simd b) noexcept { return (a<b)|(a==b); }
3772 native_nodiscard friend native_inline constexpr native_const mask_type operator>=(simd a,simd b) noexcept { return (a>b)|(a==b); }
3773 };
3774
3777 template <::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) struct simd<float, 8,Arch> : detail::register_memory<simd<float,8,Arch>, 8> {
3778 static constexpr isa<> architecture=Arch;
3779 template <class T> using rebind = simd<T,8,Arch>;
3780 using vector_mask_type=simd<mask32,8,Arch>;
3781 using mask_type=std::conditional_t<bool(NATIVE_HAS_AVX512VL),predicate<8,Arch>,simd<mask32,8,Arch>>;
3782 using mask = mask_type;
3783 using predicate_type = predicate<8,Arch>;
3784 __m256 value;
3786 native_inline simd() = default;
3788 native_inline constexpr simd(simd const &) = default;
3790 native_reinitializes native_inline constexpr simd & operator=(simd const &) = default;
3792 native_inline constexpr simd(float x) {
3793 if consteval { std::array<float,sizeof(native_type)/sizeof(float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
3794 else { value=_mm256_set1_ps(x); }
3795 }
3796
3797 native_inline native_target("avx") constexpr simd(__m256 x) : value(x) {}
3799 native_nodiscard static native_inline constexpr native_pure simd load(native_noescape float const * p) { return load_memory<1>(p); }
3801 native_inline constexpr void store(native_noescape float * p) const { store_memory<1>(p); }
3804 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3805 return simd(_mm256_add_ps(a.value, b.value));
3806 }
3807
3809 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
3810 return simd(_mm256_sub_ps(a.value, b.value));
3811 }
3812
3814 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
3815 return simd(_mm256_mul_ps(a.value, b.value));
3816 }
3817
3819 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
3820 return simd(_mm256_div_ps(a.value, b.value));
3821 }
3822
3824 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
3825 return simd(_mm256_xor_ps(a.value, _mm256_set1_ps(-0.f)));
3826 }
3827
3829 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
3830 if constexpr(bool(NATIVE_HAS_AVX512VL))
3831 return mask_type::from_native(_mm256_cmp_ps_mask(a.value,b.value,_CMP_LT_OQ));
3832 else
3833 return mask_type::unsafe_from_native(_mm256_castps_si256(_mm256_cmp_ps(a.value,b.value,_CMP_LT_OQ)));
3834 }
3835
3837 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
3838 if constexpr(bool(NATIVE_HAS_AVX512VL))
3839 return mask_type::from_native(_mm256_cmp_ps_mask(a.value,b.value,_CMP_GT_OQ));
3840 else
3841 return mask_type::unsafe_from_native(_mm256_castps_si256(_mm256_cmp_ps(a.value,b.value,_CMP_GT_OQ)));
3842 }
3843
3845 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
3846 if constexpr(bool(NATIVE_HAS_AVX512VL))
3847 return mask_type::from_native(_mm256_cmp_ps_mask(a.value,b.value,_CMP_EQ_OQ));
3848 else
3849 return mask_type::unsafe_from_native(_mm256_castps_si256(_mm256_cmp_ps(a.value,b.value,_CMP_EQ_OQ)));
3850 }
3851
3852 template<class M> requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
3854 if consteval { return ::native::detail::float_constant::select(m,a,b); }
3855 if constexpr(M::compact) return simd(_mm256_mask_blend_ps(m.to_native(),b.value,a.value));
3856 else
3857 return simd(_mm256_blendv_ps(b.value,a.value,_mm256_castsi256_ps(m.to_native()))); }
3858
3860 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
3861 return simd(_mm256_fmadd_ps(a.value, b.value, c.value));
3862 }
3863
3865 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
3866 return simd(_mm256_sqrt_ps(a.value));
3867 }
3868
3870 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
3871 return simd(_mm256_round_ps(a.value, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
3872 }
3873
3875 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
3876 return simd(_mm256_castsi256_ps(_mm256_slli_epi32(_mm256_add_epi32(_mm256_cvttps_epi32(n.value), _mm256_set1_epi32(127)), 23)));
3877 }
3878
3879 template <std::size_t Alignment = 1>
3881 native_nodiscard static native_inline constexpr native_pure simd load_memory(float const * p) noexcept {
3882 if consteval {
3883 std::array<float,sizeof(native_type)/sizeof(float)> values{};
3884 for (std::size_t i=0;i<sizeof(native_type)/sizeof(float);++i) values[i]=p[i];
3885 return from_native(__builtin_bit_cast(native_type,values));
3886 }
3887 if constexpr(Alignment>=32) return simd(_mm256_load_ps(p));
3888 else return simd(_mm256_loadu_ps(p));
3889 }
3890 template <std::size_t Alignment = 1>
3892 native_inline constexpr void store_memory(float * p) const noexcept {
3893 if consteval {
3894 auto values=__builtin_bit_cast(std::array<float,sizeof(native_type)/sizeof(float)>,value);
3895 for (std::size_t i=0;i<sizeof(native_type)/sizeof(float);++i) p[i]=values[i];
3896 return;
3897 }
3898 if constexpr(Alignment>=32) _mm256_store_ps(p,value);
3899 else _mm256_storeu_ps(p,value);
3900 }
3901 using value_type = float;
3902 using register_type = simd;
3903 using native_type = __m256;
3904 using bits_type = simd<uint32_t,8,Arch>;
3906 native_nodiscard native_inline constexpr native_pure native_target("avx") operator native_type() const noexcept { return value; }
3908 native_nodiscard native_inline constexpr native_pure native_target("avx") native_type to_native() const noexcept { return value; }
3910 native_artificial native_nodiscard native_inline constexpr native_pure bits_type bits() const noexcept { if consteval { return bits_type::from_native(__builtin_bit_cast(typename bits_type::native_type,value)); } return bits_type::from_native(_mm256_castps_si256(value)); }
3912 native_nodiscard native_inline constexpr native_pure bits_type to_bits() const noexcept { return bits(); }
3914 native_artificial native_nodiscard static native_inline constexpr native_const simd from_bits(bits_type bits) noexcept { if consteval { return simd(__builtin_bit_cast(native_type,bits.to_native())); } return simd(_mm256_castsi256_ps(bits.value)); }
3916 native_nodiscard static native_inline constexpr native_const simd from_bits(std::uint32_t bits) noexcept { return from_bits(bits_type(bits)); }
3918 native_nodiscard static native_inline constexpr native_const simd from_float(float x) noexcept { return simd(x); }
3920 native_nodiscard static native_inline constexpr native_const native_target("avx") simd from_native(native_type x) noexcept { return simd(x); }
3922 native_nodiscard static native_inline constexpr native_const native_target("avx") simd unsafe_from_float32(native_type x) noexcept { return simd(x); }
3924 native_nodiscard static native_inline constexpr native_pure simd loadu(native_noescape float const * p) { return load_memory<1>(p); }
3926 native_inline constexpr void storeu(native_noescape float * p) const { store_memory<1>(p); }
3928 native_nodiscard static native_inline constexpr native_pure simd load_bits(native_noescape std::uint32_t const * p) noexcept { return from_bits(bits_type::load(p)); }
3930 native_inline constexpr void store_bits(native_noescape std::uint32_t * p) const noexcept { bits().store(p); }
3932 native_nodiscard static native_inline constexpr native_pure simd load_bits_partial(native_noescape std::uint32_t const * p,std::size_t n,std::uint32_t fill=0) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { return from_bits(bits_type::load_partial(p,n,fill)); }
3934 native_inline constexpr void store_bits_partial(native_noescape std::uint32_t * p,std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { bits().store_partial(p,n); }
3936 native_inline constexpr simd(std::array<float,8> const & values) noexcept : simd(loadu(values.data())) {}
3937#if defined(__clang__)
3939 template <class... X> requires (sizeof...(X)==8) && (std::convertible_to<X,float> && ...)
3940 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : value{static_cast<float>(x)...} {}
3941#else
3943 template <class... X> requires (sizeof...(X)==8) && (std::convertible_to<X,float> && ...)
3944 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : simd(loadu(std::array<float,8>{static_cast<float>(x)...}.data())) {}
3945#endif
3947 native_inline constexpr simd & operator+=(simd b) noexcept { return *this=*this+b; }
3949 native_inline constexpr simd & operator-=(simd b) noexcept { return *this=*this-b; }
3951 native_inline constexpr simd & operator*=(simd b) noexcept { return *this=*this*b; }
3953 native_inline constexpr simd & operator/=(simd b) noexcept { return *this=*this/b; }
3955 native_nodiscard friend native_inline constexpr native_const mask_type operator!=(simd a,simd b) noexcept { return ~(a==b); }
3957 native_nodiscard friend native_inline constexpr native_const mask_type operator<=(simd a,simd b) noexcept { return (a<b)|(a==b); }
3959 native_nodiscard friend native_inline constexpr native_const mask_type operator>=(simd a,simd b) noexcept { return (a>b)|(a==b); }
3960 };
3961
3962}
3963
3964// SPDX-FileCopyrightText: 2026 Edward Kmett <ekmett@gmail.com>
3965// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
3966namespace native {
3967#endif
3968#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
3969 template <::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) struct simd<float, 16,Arch> : ::NATIVE_BACKEND_NAMESPACE::register_memory<simd<float,16,Arch>, 16> {
3970 static constexpr isa<> architecture=Arch;
3971 template <class T> using rebind = simd<T,16,Arch>;
3972 using vector_mask_type=simd<mask32,16,Arch>;
3973 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<float,16,Arch>;
3974 using mask = mask_type;
3975 using predicate_type = predicate<16,Arch>;
3976 __m512 value;
3978 native_inline simd() = default;
3980 native_inline constexpr simd(simd const &) = default;
3982 native_reinitializes native_inline constexpr simd & operator=(simd const &) = default;
3984 native_inline constexpr simd(float x) {
3985 if consteval { std::array<float,sizeof(native_type)/sizeof(float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
3986 else { value=_mm512_set1_ps(x); }
3987 }
3989 native_inline native_target("avx512f") constexpr simd(__m512 x) : value(x) {}
3991 native_nodiscard static native_inline constexpr native_pure simd load(native_noescape float const * p) { return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,1>(p); }
3993 native_inline constexpr void store(native_noescape float * p) const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*this); }
3995 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator+(simd a, simd b) {
3996 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
3997 return simd(_mm512_add_ps(a.value, b.value));
3998 }
4000 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator-(simd a, simd b) {
4001 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
4002 return simd(_mm512_sub_ps(a.value, b.value));
4003 }
4005 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator*(simd a, simd b) {
4006 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
4007 return simd(_mm512_mul_ps(a.value, b.value));
4008 }
4010 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator/(simd a, simd b) {
4011 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
4012 return simd(_mm512_div_ps(a.value, b.value));
4013 }
4015 native_nodiscard friend native_inline constexpr native_const simd operator-(simd a) {
4016 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
4017 return simd(_mm512_xor_ps(a.value, _mm512_set1_ps(-0.f)));
4018 }
4020 native_nodiscard friend native_inline constexpr native_const mask_type operator<(simd a, simd b) {
4021 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,a,b); }
4022 return mask_type::from_native(_mm512_cmp_ps_mask(a.value,b.value,_CMP_LT_OQ));
4023 }
4025 native_nodiscard friend native_inline constexpr native_const mask_type operator>(simd a, simd b) {
4026 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
4027 return mask_type::from_native(_mm512_cmp_ps_mask(a.value,b.value,_CMP_GT_OQ));
4028 }
4030 native_nodiscard friend native_inline constexpr native_const mask_type operator==(simd a, simd b) {
4031 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
4032 return mask_type::from_native(_mm512_cmp_ps_mask(a.value,b.value,_CMP_EQ_OQ));
4033 }
4035 template<class M> requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
4036 native_nodiscard friend native_inline constexpr native_const simd select(M m,simd a,simd b) {
4037 if consteval { return ::native::detail::float_constant::select(m,a,b); }
4038 if constexpr(M::compact) return simd(_mm512_mask_blend_ps(m.to_native(),b.value,a.value));
4039 else return simd(_mm512_castsi512_ps(_mm512_or_si512(_mm512_and_si512(m.to_native(),_mm512_castps_si512(a.value)),
4040 _mm512_andnot_si512(m.to_native(),_mm512_castps_si512(b.value))))); }
4042 native_artificial native_nodiscard friend native_inline constexpr native_pure simd fma(simd a, simd b, simd c) {
4043 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
4044 return simd(_mm512_fmadd_ps(a.value, b.value, c.value));
4045 }
4047 native_artificial native_nodiscard friend native_inline constexpr native_pure simd sqrt(simd a) {
4048 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
4049 return simd(_mm512_sqrt_ps(a.value));
4050 }
4053 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
4054 return simd(_mm512_roundscale_ps(a.value, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC));
4055 }
4057 native_nodiscard friend native_inline constexpr native_const simd normal_pow2(simd n) {
4058 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
4059 return simd(_mm512_castsi512_ps(_mm512_slli_epi32(_mm512_add_epi32(_mm512_cvttps_epi32(n.value), _mm512_set1_epi32(127)), 23)));
4060 }
4061
4062 template <std::size_t Alignment = 1>
4064 native_nodiscard static native_inline constexpr native_pure simd load_memory(float const * p) noexcept {
4065 return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,Alignment>(p);
4066 }
4067 template <std::size_t Alignment = 1>
4069 native_inline constexpr void store_memory(float * p) const noexcept {
4070 ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,Alignment>(p,*this);
4071 }
4072 using value_type = float;
4073 using register_type = simd;
4074 using native_type = __m512;
4075 using bits_type = simd<uint32_t,16,Arch>;
4077 native_nodiscard native_inline constexpr native_pure native_target("avx512f") operator native_type() const noexcept { return value; }
4079 native_nodiscard native_inline constexpr native_pure native_target("avx512f") native_type to_native() const noexcept { return value; }
4081 native_artificial native_nodiscard native_inline constexpr native_pure bits_type bits() const noexcept { if consteval { return bits_type::from_native(__builtin_bit_cast(typename bits_type::native_type,value)); } return bits_type::from_native(_mm512_castps_si512(value)); }
4083 native_nodiscard native_inline constexpr native_pure bits_type to_bits() const noexcept { return bits(); }
4085 native_artificial native_nodiscard static native_inline constexpr native_const simd from_bits(bits_type bits) noexcept { if consteval { return simd(__builtin_bit_cast(native_type,bits.to_native())); } return simd(_mm512_castsi512_ps(bits.value)); }
4087 native_nodiscard static native_inline constexpr native_const simd from_bits(std::uint32_t bits) noexcept { return from_bits(bits_type(bits)); }
4089 native_nodiscard static native_inline constexpr native_const simd from_float(float x) noexcept { return simd(x); }
4091 native_nodiscard static native_inline constexpr native_const native_target("avx512f") simd from_native(native_type x) noexcept { return simd(x); }
4093 native_nodiscard static native_inline constexpr native_const native_target("avx512f") simd unsafe_from_float32(native_type x) noexcept { return simd(x); }
4095 native_nodiscard static native_inline constexpr native_pure simd loadu(native_noescape float const * p) { return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,1>(p); }
4097 native_inline constexpr void storeu(native_noescape float * p) const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*this); }
4099 native_nodiscard static native_inline constexpr native_pure simd load_bits(native_noescape std::uint32_t const * p) noexcept { return from_bits(bits_type::load(p)); }
4101 native_inline constexpr void store_bits(native_noescape std::uint32_t * p) const noexcept { bits().store(p); }
4103 native_nodiscard static native_inline constexpr native_pure simd load_bits_partial(native_noescape std::uint32_t const * p,std::size_t n,std::uint32_t fill=0) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { return from_bits(bits_type::load_partial(p,n,fill)); }
4105 native_inline constexpr void store_bits_partial(native_noescape std::uint32_t * p,std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { bits().store_partial(p,n); }
4107 native_inline constexpr simd(std::array<float,16> const & values) noexcept : simd(loadu(values.data())) {}
4108#if defined(__clang__)
4110 template <class... X> requires (sizeof...(X)==16) && (std::convertible_to<X,float> && ...)
4111 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : value{static_cast<float>(x)...} {}
4112#else
4114 template <class... X> requires (sizeof...(X)==16) && (std::convertible_to<X,float> && ...)
4115 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : simd(loadu(std::array<float,16>{static_cast<float>(x)...}.data())) {}
4116#endif
4118 native_inline constexpr simd & operator+=(simd b) noexcept { return *this=*this+b; }
4120 native_inline constexpr simd & operator-=(simd b) noexcept { return *this=*this-b; }
4122 native_inline constexpr simd & operator*=(simd b) noexcept { return *this=*this*b; }
4124 native_inline constexpr simd & operator/=(simd b) noexcept { return *this=*this/b; }
4126 native_nodiscard friend native_inline constexpr native_const mask_type operator!=(simd a,simd b) noexcept { return ~(a==b); }
4128 native_nodiscard friend native_inline constexpr native_const mask_type operator<=(simd a,simd b) noexcept { return (a<b)|(a==b); }
4130 native_nodiscard friend native_inline constexpr native_const mask_type operator>=(simd a,simd b) noexcept { return (a>b)|(a==b); }
4131 };
4132#endif
4133#if NATIVE_HAS_ARM_NEON
4134 template <::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) struct alignas(float32x4_t) native_empty_bases simd<float, 4,Arch> : ::NATIVE_BACKEND_NAMESPACE::register_memory<simd<float,4,Arch>, 4>, detail::swizzle_access<float,4,Arch> {
4135 static constexpr isa<> architecture=Arch;
4136 template <class T> using rebind = simd<T,4,Arch>;
4137 using vector_mask_type=simd<mask32,4,Arch>;
4138 using mask_type=::NATIVE_BACKEND_NAMESPACE::comparison_mask<float,4,Arch>;
4139 using mask = mask_type;
4140 using predicate_type = predicate<4,Arch>;
4141 float32x4_t value;
4143 native_inline simd() = default;
4145 native_inline constexpr simd(simd const &) = default;
4147 native_reinitializes native_inline constexpr simd & operator=(simd const &) = default;
4149 native_inline constexpr simd(float x) {
4150 if consteval { std::array<float,sizeof(native_type)/sizeof(float)> values{}; values.fill(x); value=__builtin_bit_cast(native_type,values); }
4151 else { value=vdupq_n_f32(x); }
4152 }
4154 native_inline constexpr simd(float32x4_t x) : value(x) {}
4156 native_nodiscard static native_inline constexpr native_pure simd load(native_noescape float const * p) { return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,1>(p); }
4158 native_inline constexpr void store(native_noescape float * p) const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*this); }
4160 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator+(simd a, simd b) {
4161 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::add,a,b); }
4162 return simd(vaddq_f32(a.value, b.value));
4163 }
4165 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator-(simd a, simd b) {
4166 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::subtract,a,b); }
4167 return simd(vsubq_f32(a.value, b.value));
4168 }
4170 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator*(simd a, simd b) {
4171 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply,a,b); }
4172 return simd(vmulq_f32(a.value, b.value));
4173 }
4175 native_artificial native_nodiscard friend native_inline constexpr native_pure simd operator/(simd a, simd b) {
4176 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::divide,a,b); }
4177 return simd(vdivq_f32(a.value, b.value));
4178 }
4180 native_artificial native_nodiscard friend native_inline constexpr native_const simd operator-(simd a) {
4181 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::negate,a); }
4182 return simd(vnegq_f32(a.value));
4183 }
4184 // Clang's ACLE comparisons become scalar constrained fcmp under
4185 // -frounding-math. Keep the vector instruction and its FPCR/FPSR effects;
4186 // the memory clobber orders environment accesses without a CPU fence.
4188 native_nodiscard friend native_inline constexpr mask_type operator<(simd a, simd b) {
4189 return b>a;
4190 }
4192 native_nodiscard friend native_inline constexpr mask_type operator>(simd a, simd b) {
4193 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::less,b,a); }
4194 auto x=::native::detail::arm_register_order(a.value);
4195 auto y=::native::detail::arm_register_order(b.value);
4196 uint32x4_t bits;
4197 asm volatile("fcmgt %0.4s, %1.4s, %2.4s" : "=w"(bits) : "w"(x), "w"(y) : "memory");
4198 return mask_type::unsafe_from_native(vreinterpretq_u8_u32(::native::detail::arm_register_order(bits)));
4199 }
4201 native_nodiscard friend native_inline constexpr mask_type operator==(simd a, simd b) {
4202 if consteval { return ::native::detail::float_constant::compare(::native::detail::float_constant::equal,a,b); }
4203 auto x=::native::detail::arm_register_order(a.value);
4204 auto y=::native::detail::arm_register_order(b.value);
4205 uint32x4_t bits;
4206 asm volatile("fcmeq %0.4s, %1.4s, %2.4s" : "=w"(bits) : "w"(x), "w"(y) : "memory");
4207 return mask_type::unsafe_from_native(vreinterpretq_u8_u32(::native::detail::arm_register_order(bits)));
4208 }
4210 template<class M> requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
4211 native_nodiscard friend native_inline constexpr native_const simd select(M m,simd a,simd b) {
4212 if consteval { return ::native::detail::float_constant::select(m,a,b); }
4213 return simd(vbslq_f32(vreinterpretq_u32_u8(m.to_native()),a.value,b.value));
4214 }
4216 native_artificial native_nodiscard friend native_inline constexpr native_pure simd fma(simd a, simd b, simd c) {
4217 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::multiply_add,a,b,c); }
4218 return simd(vfmaq_f32(c.value, a.value, b.value));
4219 }
4221 native_artificial native_nodiscard friend native_inline constexpr native_pure simd sqrt(simd a) {
4222 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::square_root,a); }
4223 return simd(vsqrtq_f32(a.value));
4224 }
4227 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::nearest,a); }
4228 return simd(vrndnq_f32(a.value));
4229 }
4231 native_nodiscard friend native_inline constexpr native_const simd normal_pow2(simd n) {
4232 if consteval { return ::native::detail::float_constant::map(::native::detail::float_constant::power_of_two,n); }
4233 return simd(vreinterpretq_f32_s32(vshlq_n_s32(vaddq_s32(vcvtq_s32_f32(n.value), vdupq_n_s32(127)), 23)));
4234 }
4235
4236 template <std::size_t Alignment = 1>
4238 native_nodiscard static native_inline constexpr native_pure simd load_memory(float const * p) noexcept {
4239 return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,Alignment>(p);
4240 }
4241 template <std::size_t Alignment = 1>
4243 native_inline constexpr void store_memory(float * p) const noexcept {
4244 ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,Alignment>(p,*this);
4245 }
4246 using value_type = float;
4247 using register_type = simd;
4248 using native_type = float32x4_t;
4249 using bits_type = simd<uint32_t,4,Arch>;
4251 native_nodiscard native_inline constexpr native_pure operator native_type() const noexcept { return value; }
4253 native_nodiscard native_inline constexpr native_pure native_type to_native() const noexcept { return value; }
4255 native_artificial native_nodiscard native_inline constexpr native_pure bits_type bits() const noexcept { if consteval { return bits_type::from_native(__builtin_bit_cast(typename bits_type::native_type,value)); } return bits_type::from_native(vreinterpretq_u8_f32(value)); }
4257 native_nodiscard native_inline constexpr native_pure bits_type to_bits() const noexcept { return bits(); }
4259 native_artificial native_nodiscard static native_inline constexpr native_const simd from_bits(bits_type bits) noexcept { if consteval { return simd(__builtin_bit_cast(native_type,bits.to_native())); } return simd(vreinterpretq_f32_u8(bits.value)); }
4261 native_nodiscard static native_inline constexpr native_const simd from_bits(std::uint32_t bits) noexcept { return from_bits(bits_type(bits)); }
4263 native_nodiscard static native_inline constexpr native_const simd from_float(float x) noexcept { return simd(x); }
4265 native_nodiscard static native_inline constexpr native_const simd from_native(native_type x) noexcept { return simd(x); }
4267 native_nodiscard static native_inline constexpr native_const simd unsafe_from_float32(native_type x) noexcept { return simd(x); }
4269 native_nodiscard static native_inline constexpr native_pure simd loadu(native_noescape float const * p) { return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd,1>(p); }
4271 native_inline constexpr void storeu(native_noescape float * p) const { ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd,1>(p,*this); }
4273 native_nodiscard static native_inline constexpr native_pure simd load_bits(native_noescape std::uint32_t const * p) noexcept { return from_bits(bits_type::load(p)); }
4275 native_inline constexpr void store_bits(native_noescape std::uint32_t * p) const noexcept { bits().store(p); }
4277 native_nodiscard static native_inline constexpr native_pure simd load_bits_partial(native_noescape std::uint32_t const * p,std::size_t n,std::uint32_t fill=0) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { return from_bits(bits_type::load_partial(p,n,fill)); }
4279 native_inline constexpr void store_bits_partial(native_noescape std::uint32_t * p,std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { bits().store_partial(p,n); }
4281 native_inline constexpr simd(std::array<float,4> const & values) noexcept : simd(loadu(values.data())) {}
4282#if defined(__clang__)
4284 template <class... X> requires (sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
4285 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : value{static_cast<float>(x)...} {}
4286#else
4287 template <class... X> requires (sizeof...(X)==4) && (std::convertible_to<X,float> && ...)
4289 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<float>(x)) && ...)) : simd(loadu(std::array<float,4>{static_cast<float>(x)...}.data())) {}
4290#endif
4292 native_inline constexpr simd & operator+=(simd b) noexcept { return *this=*this+b; }
4294 native_inline constexpr simd & operator-=(simd b) noexcept { return *this=*this-b; }
4296 native_inline constexpr simd & operator*=(simd b) noexcept { return *this=*this*b; }
4298 native_inline constexpr simd & operator/=(simd b) noexcept { return *this=*this/b; }
4300 native_nodiscard friend native_inline constexpr mask_type operator!=(simd a,simd b) noexcept { return ~(a==b); }
4302 native_nodiscard friend native_inline constexpr mask_type operator<=(simd a,simd b) noexcept { return b>=a; }
4304 native_nodiscard friend native_inline constexpr mask_type operator>=(simd a,simd b) noexcept {
4305 if consteval {
4306 return ::native::detail::float_constant::compare([](auto x,auto y) {
4307 return ::native::detail::float_constant::less(y,x) || ::native::detail::float_constant::equal(x,y);
4308 },a,b);
4309 }
4310 auto x=::native::detail::arm_register_order(a.value);
4311 auto y=::native::detail::arm_register_order(b.value);
4312 uint32x4_t bits;
4313 asm volatile("fcmge %0.4s, %1.4s, %2.4s" : "=w"(bits) : "w"(x), "w"(y) : "memory");
4314 return mask_type::unsafe_from_native(vreinterpretq_u8_u32(::native::detail::arm_register_order(bits)));
4315 }
4316 };
4317#endif
4318 namespace detail::NATIVE_BACKEND {
4319 // Scalar VSCALEFSS and 512-bit VSCALEFPS require AVX512F; packed
4320 // 128/256-bit forms require AVX512VL. Short vectors mask their padding.
4321 // Keep names available for module exports without admitting software APIs.
4322 template<std::size_t N> inline constexpr bool native_scaleb_shape =
4323#if NATIVE_HAS_AVX512F
4324 N==1 || N==16
4325#if NATIVE_HAS_AVX512VL
4326 || N==2 || N==3 || N==4 || N==8
4327#endif
4328 ;
4329#else
4330 false;
4331#endif
4332 }
4333 // Full VSCALEFPS value semantics: floor(exponent), including nonfinite
4334 // operands. This raw operation follows the caller's FP environment.
4335 // Inactive lanes never execute scaling; merge preserves their exact bits.
4342 template <std::size_t N,class M, ::native::isa<> Arch>
4343 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::native_scaleb_shape<N> &&(std::same_as<M,typename simd<float,N,Arch>::mask_type> ||
4344 std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
4346 simd<float,N,Arch> prior,simd<float,N,Arch> value,simd<float,N,Arch> exponent) noexcept {
4347 if consteval {
4348 auto a=::native::detail::float_constant::words(value),b=::native::detail::float_constant::words(exponent);
4349 auto result=::native::detail::float_constant::words(prior);
4350 auto active=mask.to_bitset();
4351 for(std::size_t i=0;i<N;++i) if((active>>i)&1) result[i]=::native::detail::float_constant::scale(a[i],b[i]);
4352 return simd<float,N,Arch>::load_bits(result.data());
4353 }
4354#if NATIVE_HAS_AVX512F
4355 if constexpr(N==2 || N==3) {
4356 using V=simd<float,N,Arch>;
4357 // The compact mask excludes padding, whose prior bits are already zero.
4358 // A representation copy avoids re-normalizing the short-vector storage.
4359 return __builtin_bit_cast(V,_mm_mask_scalef_ps(__builtin_bit_cast(__m128,prior.to_native()),
4360 __mmask8(mask.to_bitset()),__builtin_bit_cast(__m128,value.to_native()),
4361 __builtin_bit_cast(__m128,exponent.to_native())));
4362 } else {
4363 auto native_mask=[&] {if constexpr(M::compact) return mask.to_native();else return to_predicate(mask).to_native();};
4364 if constexpr(N==1) return simd<float,N,Arch>(_mm_cvtss_f32(_mm_mask_scalef_ss(
4365 _mm_set_ss(prior.value),__mmask8(native_mask()),_mm_set_ss(value.value),_mm_set_ss(exponent.value))));
4366 else if constexpr(N==16) return simd<float,N,Arch>(_mm512_mask_scalef_ps(prior.value,native_mask(),value.value,exponent.value));
4367#if NATIVE_HAS_AVX512VL
4368 else if constexpr(N==4) return simd<float,N,Arch>(_mm_mask_scalef_ps(prior.value,native_mask(),value.value,exponent.value));
4369 else if constexpr(N==8) return simd<float,N,Arch>(_mm256_mask_scalef_ps(prior.value,native_mask(),value.value,exponent.value));
4370#endif
4371 }
4372#endif
4373 }
4374
4376 template <std::size_t N,class M, ::native::isa<> Arch>
4377 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::native_scaleb_shape<N> &&(std::same_as<M,typename simd<float,N,Arch>::mask_type> ||
4378 std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
4380 simd<float,N,Arch> value,simd<float,N,Arch> exponent) noexcept {
4381 if consteval { return masked_scaleb(mask,simd<float,N,Arch>(0.f),value,exponent); }
4382#if NATIVE_HAS_AVX512F
4383 auto native_mask=[&] {if constexpr(M::compact) return mask.to_native();else return to_predicate(mask).to_native();};
4384 if constexpr(N==1) return simd<float,N,Arch>(_mm_cvtss_f32(_mm_maskz_scalef_ss(
4385 __mmask8(native_mask()),_mm_set_ss(value.value),_mm_set_ss(exponent.value))));
4386 else if constexpr(N==16) return simd<float,N,Arch>(_mm512_maskz_scalef_ps(native_mask(),value.value,exponent.value));
4387#if NATIVE_HAS_AVX512VL
4388 else if constexpr(N==2 || N==3) return __builtin_bit_cast(simd<float,N,Arch>,_mm_maskz_scalef_ps(
4389 __mmask8(mask.to_bitset()),__builtin_bit_cast(__m128,value.to_native()),__builtin_bit_cast(__m128,exponent.to_native())));
4390 else if constexpr(N==4) return simd<float,N,Arch>(_mm_maskz_scalef_ps(native_mask(),value.value,exponent.value));
4391 else if constexpr(N==8) return simd<float,N,Arch>(_mm256_maskz_scalef_ps(native_mask(),value.value,exponent.value));
4392#endif
4393#endif
4394 }
4395
4398 template <std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::native_scaleb_shape<N>
4400 simd<float,N,Arch> exponent) noexcept {
4401 return masked_scaleb_zero(typename simd<float,N,Arch>::mask_type(true),value,exponent);
4402 }
4403
4404 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4405 native_nodiscard native_inline constexpr auto operator+(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a+simd<float,N,Arch>(b); }
4407 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4408 native_nodiscard native_inline constexpr auto operator+(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)+b; }
4410 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4411 native_nodiscard native_inline constexpr auto operator-(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a-simd<float,N,Arch>(b); }
4413 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4414 native_nodiscard native_inline constexpr auto operator-(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)-b; }
4416 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4417 native_nodiscard native_inline constexpr auto operator*(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a*simd<float,N,Arch>(b); }
4419 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4420 native_nodiscard native_inline constexpr auto operator*(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)*b; }
4422 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4423 native_nodiscard native_inline constexpr auto operator/(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a/simd<float,N,Arch>(b); }
4425 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4426 native_nodiscard native_inline constexpr auto operator/(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)/b; }
4428 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4429 native_nodiscard native_inline constexpr auto operator<(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a<simd<float,N,Arch>(b); }
4431 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4432 native_nodiscard native_inline constexpr auto operator<(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)<b; }
4434 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4435 native_nodiscard native_inline constexpr auto operator>(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a>simd<float,N,Arch>(b); }
4437 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4438 native_nodiscard native_inline constexpr auto operator>(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)>b; }
4440 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4441 native_nodiscard native_inline constexpr auto operator<=(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a<=simd<float,N,Arch>(b); }
4443 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4444 native_nodiscard native_inline constexpr auto operator<=(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)<=b; }
4446 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4447 native_nodiscard native_inline constexpr auto operator>=(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a>=simd<float,N,Arch>(b); }
4449 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4450 native_nodiscard native_inline constexpr auto operator>=(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)>=b; }
4452 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4453 native_nodiscard native_inline constexpr auto operator==(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a==simd<float,N,Arch>(b); }
4455 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4456 native_nodiscard native_inline constexpr auto operator==(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)==b; }
4458 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4459 native_nodiscard native_inline constexpr auto operator!=(simd<float,N,Arch> a, U b) noexcept(noexcept(simd<float,N,Arch>(b))) { return a!=simd<float,N,Arch>(b); }
4461 template <std::size_t N,class U, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<U>) && (!std::same_as<U,simd<float,N,Arch>>) && std::convertible_to<U,simd<float,N,Arch>>
4462 native_nodiscard native_inline constexpr auto operator!=(U a, simd<float,N,Arch> b) noexcept(noexcept(simd<float,N,Arch>(a))) { return simd<float,N,Arch>(a)!=b; }
4464 template <std::size_t N,class A,class B, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<A> && !::NATIVE_BACKEND_NAMESPACE::custom_argument<B>) && std::convertible_to<A,simd<float,N,Arch>> && std::convertible_to<B,simd<float,N,Arch>>
4465 native_nodiscard native_inline constexpr auto fma(simd<float,N,Arch> a,A b,B c) noexcept(noexcept(simd<float,N,Arch>(b)) && noexcept(simd<float,N,Arch>(c))) { return fma(a,simd<float,N,Arch>(b),simd<float,N,Arch>(c)); }
4467 template <std::size_t N,class A,class B, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<A> && !::NATIVE_BACKEND_NAMESPACE::custom_argument<B>) && (!std::same_as<A,simd<float,N,Arch>>) && std::convertible_to<A,simd<float,N,Arch>> && std::convertible_to<B,simd<float,N,Arch>>
4468 native_nodiscard native_inline constexpr auto fma(A a,simd<float,N,Arch> b,B c) noexcept(noexcept(simd<float,N,Arch>(a)) && noexcept(simd<float,N,Arch>(c))) { return fma(simd<float,N,Arch>(a),b,simd<float,N,Arch>(c)); }
4470 template <std::size_t N,class A,class B, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && (!::NATIVE_BACKEND_NAMESPACE::custom_argument<A> && !::NATIVE_BACKEND_NAMESPACE::custom_argument<B>) && (!std::same_as<A,simd<float,N,Arch>>) && (!std::same_as<B,simd<float,N,Arch>>) && std::convertible_to<A,simd<float,N,Arch>> && std::convertible_to<B,simd<float,N,Arch>>
4471 native_nodiscard native_inline constexpr auto fma(A a,B b,simd<float,N,Arch> c) noexcept(noexcept(simd<float,N,Arch>(a)) && noexcept(simd<float,N,Arch>(b))) { return fma(simd<float,N,Arch>(a),simd<float,N,Arch>(b),c); }
4474 template <std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
4476 return simd<float,N,Arch>::from_bits(a.bits() & typename simd<float,N,Arch>::bits_type(0x7fffffffu));
4477 }
4478
4479#if NATIVE_HOST_NEON
4484 template<std::size_t N, ::native::isa<> Arch>
4485 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
4486 native_nodiscard native_inline constexpr simd<std::int32_t,N,Arch>
4487 fcvtzs(simd<float,N,Arch> x) noexcept {
4488 using I = simd<std::int32_t,N,Arch>;
4489 if consteval {
4490 std::array<float,N> values{};
4491 std::array<std::int32_t,N> result{};
4492 x.store(values.data());
4493 for (std::size_t i = 0; i < N; ++i)
4494 result[i] = detail::float_constant::fcvtzs(std::bit_cast<std::uint32_t>(values[i]));
4495 return I::load(result.data());
4496 } else {
4497 if constexpr (N == 1) return I(vcvts_s32_f32(x.value));
4498 else if constexpr (N == 2 || N == 3) {
4499 // Short float padding is zero, and conversion preserves that invariant.
4500 I result;
4501 result.value = __builtin_bit_cast(typename I::native_type,
4502 fcvtzs(x.to_storage()).to_native());
4503 return result;
4504 }
4505#if NATIVE_HAS_ARM_NEON
4506 else if constexpr (N == 4)
4507 return I::from_native(vreinterpretq_u8_s32(vcvtq_s32_f32(x.value)));
4508#endif
4509 }
4510 }
4511
4513 template<::native::isa<> Arch = NATIVE_BASELINE, class T>
4514 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,float>
4515 native_nodiscard native_inline constexpr std::int32_t fcvtzs(T x) noexcept {
4516 return fcvtzs(simd<float,1,Arch>(x)).value;
4517 }
4518
4522 template<std::size_t N, ::native::isa<> Arch>
4523 requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
4524 native_nodiscard native_inline constexpr simd<std::uint32_t,N,Arch>
4525 fcvtzu(simd<float,N,Arch> x) noexcept {
4526 using I = simd<std::uint32_t,N,Arch>;
4527 if consteval {
4528 std::array<float,N> values{};
4529 std::array<std::uint32_t,N> result{};
4530 x.store(values.data());
4531 for (std::size_t i = 0; i < N; ++i)
4532 result[i] = detail::float_constant::fcvtzu(std::bit_cast<std::uint32_t>(values[i]));
4533 return I::load(result.data());
4534 } else {
4535 if constexpr (N == 1) return I(vcvts_u32_f32(x.value));
4536 else if constexpr (N == 2 || N == 3) {
4537 I result;
4538 result.value = __builtin_bit_cast(typename I::native_type,
4539 fcvtzu(x.to_storage()).to_native());
4540 return result;
4541 }
4542#if NATIVE_HAS_ARM_NEON
4543 else if constexpr (N == 4)
4544 return I::from_native(vreinterpretq_u8_u32(vcvtq_u32_f32(x.value)));
4545#endif
4546 }
4547 }
4548
4550 template<::native::isa<> Arch = NATIVE_BASELINE, class T>
4551 requires NATIVE_ARCH_REQUIRES(Arch) && std::same_as<T,float>
4552 native_nodiscard native_inline constexpr std::uint32_t fcvtzu(T x) noexcept {
4553 return fcvtzu(simd<float,1,Arch>(x)).value;
4554 }
4555#endif
4556
4557 // Truncating float-to-integer conversion requires a representable result.
4561 template <class To, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && std::same_as<To,std::int32_t>
4563 if consteval {
4564 std::array<float,N> a{}; std::array<To,N> b{}; x.store(a.data());
4565 for(std::size_t i=0;i<N;++i) b[i]=static_cast<To>(a[i]);
4566 return simd<To,N,Arch>::load(b.data());
4567 }
4568 if constexpr(N==2 || N==3) return simd<To,N,Arch>::from_storage(convert<To>(x.to_storage()));
4569 else if constexpr (N==1) return simd<To,N,Arch>(static_cast<std::int32_t>(x.value));
4570#if NATIVE_HAS_AVX2
4571 else if constexpr (N==4) return simd<To,N,Arch>::from_native(_mm_cvttps_epi32(x.value));
4572 else if constexpr (N==8) return simd<To,N,Arch>::from_native(_mm256_cvttps_epi32(x.value));
4573#endif
4574#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4575 else if constexpr (N==16) return simd<To,N,Arch>::from_native(_mm512_cvttps_epi32(x.value));
4576#endif
4577#if NATIVE_HAS_ARM_NEON
4578 else if constexpr (N==4) return simd<To,N,Arch>::from_native(vreinterpretq_u8_s32(vcvtq_s32_f32(x.value)));
4579#endif
4580 }
4581
4584 template <class To, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N> && std::same_as<To,float>
4586 if consteval {
4587 std::array<std::int32_t,N> a{}; std::array<To,N> b{}; x.store(a.data());
4588 for(std::size_t i=0;i<N;++i) b[i]=static_cast<To>(a[i]);
4589 return simd<To,N,Arch>::load(b.data());
4590 }
4591 if constexpr(N==2 || N==3) return simd<To,N,Arch>::from_storage(convert<To>(x.to_storage()));
4592 else if constexpr (N==1) return simd<To,N,Arch>(static_cast<float>(x.value));
4593#if NATIVE_HAS_AVX2
4594 else if constexpr (N==4) return simd<To,N,Arch>(_mm_cvtepi32_ps(x.value));
4595 else if constexpr (N==8) return simd<To,N,Arch>(_mm256_cvtepi32_ps(x.value));
4596#endif
4597#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4598 else if constexpr (N==16) return simd<To,N,Arch>(_mm512_cvtepi32_ps(x.value));
4599#endif
4600#if NATIVE_HAS_ARM_NEON
4601 else if constexpr (N==4) return simd<To,N,Arch>(vcvtq_f32_s32(vreinterpretq_s32_u8(x.value)));
4602#endif
4603 }
4604
4605 namespace detail::NATIVE_BACKEND {
4606 template <class V,std::size_t Alignment> native_nodiscard native_artificial native_inline constexpr native_pure V simd_load_native(native_noescape float const * p) noexcept {
4607 if consteval {
4608 std::array<float,V::lanes> values{};
4609 for (std::size_t i=0;i<V::lanes;++i) values[i]=p[i];
4610 return V::from_native(__builtin_bit_cast(typename V::native_type,values));
4611 }
4612 if constexpr(V::lanes==1) return V(*p);
4613#if NATIVE_HAS_AVX2
4614 else if constexpr(V::lanes==4) { if constexpr(Alignment>=16)return V(_mm_load_ps(p));else return V(_mm_loadu_ps(p)); }
4615 else if constexpr(V::lanes==8) { if constexpr(Alignment>=32)return V(_mm256_load_ps(p));else return V(_mm256_loadu_ps(p)); }
4616#endif
4617#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4618 else if constexpr(V::lanes==16) { if constexpr(Alignment>=64)return V(_mm512_load_ps(p));else return V(_mm512_loadu_ps(p)); }
4619#endif
4620#if NATIVE_HAS_ARM_NEON
4621 else if constexpr(V::lanes==4)return V(vld1q_f32(p));
4622#endif
4623 }
4624 template <class V,std::size_t Alignment> native_artificial native_inline constexpr void simd_store_native(native_noescape float * p,V value) noexcept {
4625 if consteval {
4626 auto values=__builtin_bit_cast(std::array<float,V::lanes>,value.to_native());
4627 for (std::size_t i=0;i<V::lanes;++i) p[i]=values[i];
4628 return;
4629 }
4630 if constexpr(V::lanes==1)*p=value.value;
4631#if NATIVE_HAS_AVX2
4632 else if constexpr(V::lanes==4) {if constexpr(Alignment>=16)_mm_store_ps(p,value.value);else _mm_storeu_ps(p,value.value);}
4633 else if constexpr(V::lanes==8) {if constexpr(Alignment>=32)_mm256_store_ps(p,value.value);else _mm256_storeu_ps(p,value.value);}
4634#endif
4635#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4636 else if constexpr(V::lanes==16) {if constexpr(Alignment>=64)_mm512_store_ps(p,value.value);else _mm512_storeu_ps(p,value.value);}
4637#endif
4638#if NATIVE_HAS_ARM_NEON
4639 else if constexpr(V::lanes==4)vst1q_f32(p,value.value);
4640#endif
4641 }
4642 }
4643 namespace detail::NATIVE_BACKEND {
4644 // Alignment is a caller promise. Access is a hint; streaming currently falls
4645 // back to ordinary moves on every backend and needs no completion fence.
4646 template <::native::isa<> Arch, class T,std::size_t N,class U,std::size_t A=1,simd_access Access=simd_access::ordinary>
4647 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4648 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4649 native_nodiscard native_inline constexpr native_pure simd<T,N,Arch> load_simd(native_noescape U const * p,simd_memory<A,Access> = {}) noexcept {
4650 if constexpr (::native::simd_custom_element<T>) return simd<T,N,Arch>::template load_memory<A>(p);
4651 else if constexpr (::native::simd_custom_element<U>) return simd<U,N,Arch>::template load_memory<A>(p).to_native();
4652 else return ::NATIVE_BACKEND_NAMESPACE::simd_load_native<simd<float,N,Arch>,A>(p);
4653 }
4654 template <class U,class T,std::size_t N,std::size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
4655 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4656 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4657 native_inline constexpr void store_simd(native_noescape U * p,simd<T,N,Arch> value,simd_memory<A,Access> = {}) noexcept {
4658 if constexpr (::native::simd_custom_element<U>) simd<U,N,Arch>(value).template store_memory<A>(p);
4660 else if constexpr (::native::simd_custom_element<T>) value.to_native().template store_memory<A>(p);
4661 else ::NATIVE_BACKEND_NAMESPACE::simd_store_native<simd<float,N,Arch>,A>(p,value);
4662 }
4663 template <::native::isa<> Arch, class T,std::size_t N,class U,std::size_t A=1,simd_access Access=simd_access::ordinary>
4664 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4665 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4666 native_nodiscard native_inline constexpr native_pure simd<T,N,Arch> load_simd_partial(native_noescape U const * p,std::size_t count,
4667 T fill=T{},simd_memory<A,Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") {
4668 std::array<U,N> temporary;temporary.fill(U(fill));
4669 for(std::size_t i=0;i<count;++i)temporary[i]=p[i];
4670 return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T,N>(temporary.data());
4671 }
4672 template <class U,class T,std::size_t N,std::size_t A=1,simd_access Access=simd_access::ordinary, ::native::isa<> Arch>
4673 requires NATIVE_ARCH_REQUIRES(Arch) && (std::same_as<U,float> || ::native::simd_custom_element<U>) &&
4674 (std::same_as<T,float> || ::native::simd_custom_element<T>)
4675 native_inline constexpr void store_simd_partial(native_noescape U * p,simd<T,N,Arch> value,std::size_t count,simd_memory<A,Access> = {}) noexcept native_diagnose_if(count > N,"partial SIMD count exceeds the lane count") {
4676 std::array<U,N> temporary;store_simd(temporary.data(),value);
4677 for(std::size_t i=0;i<count;++i)p[i]=temporary[i];
4678 }
4679 template <::native::isa<> Arch,class T,std::size_t N> requires NATIVE_ARCH_REQUIRES(Arch)
4680 native_nodiscard native_inline constexpr native_pure auto load_simd(std::array<T,N> const & values) noexcept { return ::NATIVE_BACKEND_NAMESPACE::load_simd<Arch,T,N>(values.data()); }
4681 template <::native::isa<> Arch,class T,std::size_t N> requires NATIVE_ARCH_REQUIRES(Arch) && (N!=std::dynamic_extent)
4682 native_nodiscard native_inline constexpr native_pure auto load_simd(std::span<T,N> values) noexcept { return load_simd<Arch,std::remove_cv_t<T>,N>(values.data()); }
4683 }
4684
4685}
4686namespace NATIVE_BACKEND_NAMESPACE::native {
4687 template <class V,std::size_t N> using register_memory = ::NATIVE_BACKEND_NAMESPACE::register_memory<V,N>;
4688 using fp32x1 = ::native::simd<float,1,NATIVE_DEFAULT_ARCH>;
4689#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
4690 using fp32x4 = ::native::simd<float,4,NATIVE_DEFAULT_ARCH>;
4691#endif
4692#if NATIVE_HAS_AVX2
4693 using fp32x8 = ::native::simd<float,8,NATIVE_DEFAULT_ARCH>;
4694#endif
4695#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4696 using fp32x16 = ::native::simd<float,16,NATIVE_DEFAULT_ARCH>;
4697#endif
4698 template<class V> concept float_register = requires { V::lanes; V::architecture; typename V::value_type; } &&
4699 std::same_as<typename V::value_type,float> && ::NATIVE_BACKEND_NAMESPACE::float_shape<V::lanes> && (::native::abi_lookup<V::architecture,::native::detail::raw_kernel_policies>::index == NATIVE_RAW_TARGET);
4700 // Ordered comparison: second operand wins on equality or unordered, including
4701 // signed zero and NaNs, identically on each architecture.
4702 template<float_register V> native_nodiscard native_inline constexpr V min(V a, V b) { return select(a < b, a, b); }
4703 template<float_register V> native_nodiscard native_inline constexpr V max(V a, V b) { return select(a > b, a, b); }
4704 template<float_register V> native_nodiscard native_inline native_pure float reduce_add(V v) {
4705 std::array<float, V::lanes> a; v.storeu(a.data()); float sum = 0;
4706 for (float x : a) sum += x; // Increasing lane order, FP32.
4707 return sum;
4708 }
4709 template<float_register V> native_nodiscard native_inline V acos_scalar_lanes(V v) {
4710 std::array<float, V::lanes> a; v.storeu(a.data());
4711 for (float & x : a) x = std::acos(x);
4712 return V::loadu(a.data());
4713 }
4714 struct native_target {
4715#if NATIVE_HAS_AVX512F && NATIVE_HAS_AVX512DQ
4716 using float_type = fp32x16;
4717 static constexpr std::size_t preferred_registers = 6;
4718#elif NATIVE_HAS_AVX2
4719 using float_type = fp32x8;
4720 static constexpr std::size_t preferred_registers = 6;
4721#elif NATIVE_HAS_ARM_NEON
4722 using float_type = fp32x4;
4723 // Prior Mac wide-kernel tuning; remeasure when changing the kernel.
4724 static constexpr std::size_t preferred_registers = 15;
4725#else
4726 using float_type = fp32x1;
4727 static constexpr std::size_t preferred_registers = 1;
4728#endif
4729 // K is derived from the selected register, never a second ISA decision.
4730 static constexpr std::size_t lanes = float_type::lanes;
4731 };
4732}
4733
4734
4735
4736#if NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON
4737#if defined(__x86_64__) || defined(_M_X64)
4738#elif defined(__aarch64__) || defined(_M_ARM64)
4739#endif
4740
4741namespace native {
4742 namespace detail::NATIVE_BACKEND {
4743 template<class T> concept short_element = std::same_as<T,float> ||
4744 std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t> || std::same_as<T,mask32>;
4745 template<class T> using short_lane = std::conditional_t<simd_mask_element<T>,std::uint32_t,T>;
4746 template<class T> using short_native = T __attribute__((ext_vector_type(4)));
4747 template<class T> constexpr short_lane<T> short_word(T value) noexcept {
4748 if constexpr(simd_mask_element<T>) return value.to_bits();
4749 else return value;
4750 }
4751 }
4752
4753 // Logical short vectors use one physical four-lane register. Floating padding
4754 // stays zero; division supplies one in the unused denominator lanes.
4760 template<::NATIVE_BACKEND_NAMESPACE::short_element T,std::size_t N,::native::isa<> Arch>
4761 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && requires { typename simd<T,4,Arch>::native_type; }
4762 // Make the register alignment explicit: MSVC's packed standard-library
4763 // aggregates can otherwise cap an ext_vector_type member's implicit alignment.
4764 struct alignas(typename simd<T,4,Arch>::native_type) simd<T,N,Arch> : detail::swizzle_access<T,N,Arch> {
4765 using value_type=T;
4766 static constexpr isa<> architecture=Arch;
4767 using storage_type=simd<T,4,Arch>;
4768 using native_type=::NATIVE_BACKEND_NAMESPACE::short_native<::NATIVE_BACKEND_NAMESPACE::short_lane<T>>;
4769 using register_type=simd;
4770 using unsigned_register_tag=void;
4771 using bits_type=simd<std::uint32_t,N,Arch>;
4772 using vector_mask_type=simd<mask32,N,Arch>;
4773 using mask_type=std::conditional_t<simd_mask_element<T>,simd,
4774 std::conditional_t<bool(NATIVE_HAS_AVX512VL),predicate<N,Arch>,vector_mask_type>>;
4775 using mask=mask_type;
4776 using predicate_type=predicate<N,Arch>;
4777 template<class U> using rebind=simd<U,N,Arch>;
4778 static constexpr std::size_t lanes=N;
4779 static constexpr std::size_t storage_lanes=4;
4780 static constexpr bool compact=false;
4781 static constexpr std::uint64_t lane_mask=(std::uint64_t(1)<<N)-1;
4782 native_type value;
4783
4785 native_inline constexpr simd() = default;
4787 native_inline constexpr simd(simd const &) = default;
4789 native_inline constexpr simd & operator=(simd const &) = default;
4791 native_inline constexpr simd(T x) noexcept : value{::NATIVE_BACKEND_NAMESPACE::short_word(x),::NATIVE_BACKEND_NAMESPACE::short_word(x),
4792 N==3?::NATIVE_BACKEND_NAMESPACE::short_word(x):NATIVE_BACKEND_NAMESPACE::short_lane<T>(0),0} {}
4793
4794 explicit native_inline constexpr simd(bool x) noexcept requires simd_mask_element<T> : simd(T(x)) {}
4796 native_inline constexpr simd(native_type x) noexcept
4797 : value(__builtin_shufflevector(x,native_type{},0,1,N==3?2:4,4)) {}
4798
4799 template<class... X> requires(sizeof...(X)==N) && (std::convertible_to<X,T> && ...)
4800 native_inline constexpr simd(X... x) noexcept((noexcept(static_cast<T>(x)) && ...))
4801 : value{::NATIVE_BACKEND_NAMESPACE::short_word(static_cast<T>(x))...} {}
4802
4803 native_inline constexpr simd(std::array<T,N> const & values) noexcept : simd(load(values.data())) {}
4804
4806 native_nodiscard native_inline constexpr native_type to_native() const noexcept { return value; }
4808 native_nodiscard native_inline constexpr operator native_type() const noexcept requires(!simd_mask_element<T>) { return value; }
4810 native_nodiscard static native_inline constexpr simd from_native(native_type x) noexcept {
4811 if constexpr(simd_mask_element<T>) return from_storage(storage_type::from_native(__builtin_bit_cast(typename storage_type::native_type,x)));
4812 else return simd(x);
4813 }
4814
4815 native_nodiscard static native_inline constexpr simd unsafe_from_native(native_type x) noexcept { return simd(x); }
4817 native_nodiscard native_inline constexpr storage_type to_storage() const noexcept {
4818 if constexpr(simd_mask_element<T>) return storage_type::unsafe_from_native(__builtin_bit_cast(typename storage_type::native_type,value));
4819 else return storage_type::from_native(__builtin_bit_cast(typename storage_type::native_type,value));
4820 }
4821
4822 native_nodiscard static native_inline constexpr simd from_storage(storage_type x) noexcept {
4823 return simd(__builtin_bit_cast(native_type,x.to_native()));
4824 }
4825 // The caller supplies exactly N logical lanes; alignment never grants a
4826 // readable fourth lane. Three-lane x86 transfers use native masked memory.
4828 template<std::size_t Alignment=1>
4829 native_nodiscard static native_inline constexpr simd load_memory(T const * p) noexcept {
4830 if consteval {
4831 native_type words{};
4832 for (std::size_t i=0;i<N;++i) words[i]=::NATIVE_BACKEND_NAMESPACE::short_word(p[i]);
4833 return simd(unchecked{},words);
4834 }
4835#if defined(__x86_64__) || defined(_M_X64)
4836 if constexpr(bool(NATIVE_HAS_AVX2)) {
4837 if constexpr(N==2) return simd(unchecked{},__builtin_bit_cast(native_type,_mm_loadl_epi64(reinterpret_cast<__m128i const *>(p))));
4838 else if constexpr(bool(NATIVE_HAS_AVX512VL)) {
4839 if constexpr(std::same_as<T,float>) return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskz_loadu_ps(7,p)));
4840 else return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskz_loadu_epi32(7,p)));
4841 } else {
4842 auto active=_mm_set_epi32(0,-1,-1,-1);
4843 if constexpr(std::same_as<T,float>) return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskload_ps(p,active)));
4844 else return simd(unchecked{},__builtin_bit_cast(native_type,_mm_maskload_epi32(reinterpret_cast<int const *>(p),active)));
4845 }
4846 }
4847#elif defined(__aarch64__) || defined(_M_ARM64)
4848 if constexpr((::native::arm_feature::neon <= Arch)) {
4849 if constexpr(std::same_as<T,float>) {
4850 auto x=vcombine_f32(vld1_f32(p),vdup_n_f32(0.f));
4851 if constexpr(N==3) x=vld1q_lane_f32(p+2,x,2);
4852 return simd(unchecked{},__builtin_bit_cast(native_type,x));
4853 } else {
4854 auto q=reinterpret_cast<std::uint32_t const *>(p);
4855 auto x=vcombine_u32(vld1_u32(q),vdup_n_u32(0));
4856 if constexpr(N==3) x=vld1q_lane_u32(q+2,x,2);
4857 return simd(unchecked{},__builtin_bit_cast(native_type,x));
4858 }
4859 }
4860#endif
4861 }
4862
4863 template<std::size_t Alignment=1>
4864 native_inline constexpr void store_memory(T * p) const noexcept {
4865 if consteval {
4866 for (std::size_t i=0;i<N;++i) {
4867 if constexpr(simd_mask_element<T>) p[i]=T::from_bits(value[i]);
4868 else p[i]=value[i];
4869 }
4870 return;
4871 }
4872#if defined(__x86_64__) || defined(_M_X64)
4873 if constexpr(bool(NATIVE_HAS_AVX2)) {
4874 if constexpr(N==2) _mm_storel_epi64(reinterpret_cast<__m128i *>(p),__builtin_bit_cast(__m128i,value));
4875 else if constexpr(bool(NATIVE_HAS_AVX512VL)) {
4876 if constexpr(std::same_as<T,float>) _mm_mask_storeu_ps(p,7,__builtin_bit_cast(__m128,value));
4877 else _mm_mask_storeu_epi32(p,7,__builtin_bit_cast(__m128i,value));
4878 } else {
4879 auto active=_mm_set_epi32(0,-1,-1,-1);
4880 if constexpr(std::same_as<T,float>) _mm_maskstore_ps(p,active,__builtin_bit_cast(__m128,value));
4881 else _mm_maskstore_epi32(reinterpret_cast<int *>(p),active,__builtin_bit_cast(__m128i,value));
4882 }
4883 }
4884#elif defined(__aarch64__) || defined(_M_ARM64)
4885 if constexpr((::native::arm_feature::neon <= Arch)) {
4886 if constexpr(std::same_as<T,float>) {
4887 auto x=__builtin_bit_cast(float32x4_t,value);vst1_f32(p,vget_low_f32(x));
4888 if constexpr(N==3) vst1q_lane_f32(p+2,x,2);
4889 } else {
4890 auto q=reinterpret_cast<std::uint32_t *>(p);auto x=__builtin_bit_cast(uint32x4_t,value);
4891 vst1_u32(q,vget_low_u32(x));
4892 if constexpr(N==3) vst1q_lane_u32(q+2,x,2);
4893 }
4894 }
4895#endif
4896 }
4897
4898 native_nodiscard static native_inline constexpr simd load(T const * p) noexcept { return load_memory(p); }
4900 native_nodiscard static native_inline constexpr simd loadu(T const * p) noexcept { return load_memory(p); }
4902 native_inline constexpr void store(T * p) const noexcept { store_memory(p); }
4904 native_inline constexpr void storeu(T * p) const noexcept { store_memory(p); }
4906 native_nodiscard static native_inline constexpr simd load_partial(T const * p,std::size_t n,T fill={}) noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") {
4907 std::array<T,N> values;values.fill(fill);
4908 if consteval { for (std::size_t i=0;i<n;++i) values[i]=p[i]; }
4909 else { if(n) std::memcpy(values.data(),p,n*sizeof(T)); }
4910 return load(values.data());
4911 }
4912
4913 native_inline constexpr void store_partial(T * p,std::size_t n) const noexcept native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") {
4914 std::array<T,N> values;store(values.data());
4915 if consteval { for (std::size_t i=0;i<n;++i) p[i]=values[i]; }
4916 else { if(n) std::memcpy(p,values.data(),n*sizeof(T)); }
4917 }
4918
4919 native_nodiscard native_inline constexpr bits_type bits() const noexcept requires std::same_as<T,float> {
4920 return bits_type::from_storage(to_storage().bits());
4921 }
4922
4923 native_nodiscard native_inline constexpr bits_type to_bits() const noexcept requires std::same_as<T,float> { return bits(); }
4925 native_nodiscard static native_inline constexpr simd from_bits(bits_type x) noexcept requires std::same_as<T,float> {
4926 return from_storage(storage_type::from_bits(x.to_storage()));
4927 }
4928
4929 native_nodiscard static native_inline constexpr simd from_bits(std::uint32_t x) noexcept requires std::same_as<T,float> { return from_bits(bits_type(x)); }
4931 native_nodiscard static native_inline constexpr simd from_float(float x) noexcept requires std::same_as<T,float> { return simd(x); }
4933 native_nodiscard static native_inline constexpr simd unsafe_from_float32(native_type x) noexcept requires std::same_as<T,float> { return simd(x); }
4935 native_nodiscard static native_inline constexpr simd load_bits(std::uint32_t const * p) noexcept requires std::same_as<T,float> { return from_bits(bits_type::load(p)); }
4937 native_inline constexpr void store_bits(std::uint32_t * p) const noexcept requires std::same_as<T,float> { bits().store(p); }
4939 native_nodiscard static native_inline constexpr simd load_bits_partial(std::uint32_t const * p,std::size_t n,std::uint32_t fill=0) noexcept requires std::same_as<T,float> native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { return from_bits(bits_type::load_partial(p,n,fill)); }
4941 native_inline constexpr void store_bits_partial(std::uint32_t * p,std::size_t n) const noexcept requires std::same_as<T,float> native_diagnose_if(n > simd::lanes,"partial SIMD count exceeds the lane count") { bits().store_partial(p,n); }
4943 native_nodiscard static native_inline constexpr simd from_bitset(std::uint64_t bits) noexcept requires simd_mask_element<T> { return from_storage(storage_type::from_bitset(bits&lane_mask)); }
4945 native_nodiscard native_inline constexpr std::uint64_t to_bitset() const noexcept requires simd_mask_element<T> { return to_storage().to_bitset()&lane_mask; }
4947 native_nodiscard friend native_inline constexpr bool any(simd x) noexcept requires simd_mask_element<T> { return x.to_bitset()!=0; }
4949 native_nodiscard friend native_inline constexpr bool all(simd x) noexcept requires simd_mask_element<T> { return x.to_bitset()==lane_mask; }
4951 native_nodiscard friend native_inline constexpr bool none(simd x) noexcept requires simd_mask_element<T> { return !any(x); }
4952
4955 native_nodiscard friend native_inline constexpr simd operator+(simd a,simd b) noexcept requires(!simd_mask_element<T>) { return clean(a.to_storage()+b.to_storage()); }
4958 native_nodiscard friend native_inline constexpr simd operator-(simd a,simd b) noexcept requires(!simd_mask_element<T>) { return clean(a.to_storage()-b.to_storage()); }
4961 native_nodiscard friend native_inline constexpr simd operator*(simd a,simd b) noexcept requires(!simd_mask_element<T>) { return clean(a.to_storage()*b.to_storage()); }
4964 native_nodiscard friend native_inline constexpr simd operator/(simd a,simd b) noexcept requires std::same_as<T,float> {
4965 auto padded=__builtin_shufflevector(b.value,native_type{1.f,1.f,1.f,1.f},0,1,N==3?2:4,4);
4966 auto divisor=storage_type::from_native(__builtin_bit_cast(typename storage_type::native_type,padded));
4967 return clean(a.to_storage()/divisor);
4968 }
4969
4970 native_nodiscard friend native_inline constexpr simd operator-(simd a) noexcept requires(!simd_mask_element<T>) { return from_storage(-a.to_storage()); }
4972 native_nodiscard friend native_inline constexpr simd operator&(simd a,simd b) noexcept requires(!std::same_as<T,float>) { return from_storage(a.to_storage()&b.to_storage()); }
4974 native_nodiscard friend native_inline constexpr simd operator|(simd a,simd b) noexcept requires(!std::same_as<T,float>) { return from_storage(a.to_storage()|b.to_storage()); }
4976 native_nodiscard friend native_inline constexpr simd operator^(simd a,simd b) noexcept requires(!std::same_as<T,float>) { return from_storage(a.to_storage()^b.to_storage()); }
4978 native_nodiscard friend native_inline constexpr simd operator~(simd a) noexcept requires(!std::same_as<T,float>) { return from_storage(~a.to_storage()); }
4980 native_nodiscard friend native_inline constexpr simd operator!(simd a) noexcept requires simd_mask_element<T> { return ~a; }
4982 native_nodiscard friend native_inline constexpr mask_type operator==(simd a,simd b) noexcept { return comparison(a.to_storage()==b.to_storage()); }
4984 native_nodiscard friend native_inline constexpr mask_type operator!=(simd a,simd b) noexcept { return ~(a==b); }
4986 native_nodiscard friend native_inline constexpr mask_type operator<(simd a,simd b) noexcept requires(!simd_mask_element<T>) { return comparison(a.to_storage()<b.to_storage()); }
4988 native_nodiscard friend native_inline constexpr mask_type operator>(simd a,simd b) noexcept requires(!simd_mask_element<T>) { return comparison(a.to_storage()>b.to_storage()); }
4990 native_nodiscard friend native_inline constexpr mask_type operator<=(simd a,simd b) noexcept requires(!simd_mask_element<T>) { return comparison(a.to_storage()<=b.to_storage()); }
4992 native_nodiscard friend native_inline constexpr mask_type operator>=(simd a,simd b) noexcept requires(!simd_mask_element<T>) { return comparison(a.to_storage()>=b.to_storage()); }
4994 template<class M> requires(std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
4995 native_nodiscard friend native_inline constexpr simd select(M mask,simd a,simd b) noexcept {
4996 using FM=typename storage_type::mask_type;
4997 return clean(select(FM::from_bitset(mask.to_bitset()),a.to_storage(),b.to_storage()));
4998 }
4999
5000 native_nodiscard friend native_inline constexpr simd fma(simd a,simd b,simd c) noexcept requires std::same_as<T,float> { return clean(fma(a.to_storage(),b.to_storage(),c.to_storage())); }
5002 native_nodiscard friend native_inline constexpr simd sqrt(simd a) noexcept requires std::same_as<T,float> { return clean(sqrt(a.to_storage())); }
5004 native_nodiscard friend native_inline constexpr simd round_even(simd a) noexcept requires std::same_as<T,float> { return clean(round_even(a.to_storage())); }
5006 native_nodiscard friend native_inline constexpr simd normal_pow2(simd a) noexcept requires std::same_as<T,float> { return from_storage(normal_pow2(a.to_storage())); }
5008 template<std::size_t K> requires(K<32)
5009 native_nodiscard friend native_inline constexpr simd operator<<(simd a,imm_t<K>) noexcept requires simd_integer_element<T> { return from_storage(a.to_storage()<<imm<K>); }
5011 template<std::size_t K> requires(K<32)
5012 native_nodiscard friend native_inline constexpr simd operator>>(simd a,imm_t<K>) noexcept requires simd_integer_element<T> { return from_storage(a.to_storage()>>imm<K>); }
5014 template<unsigned K> requires(K<32) && simd_integer_element<T>
5015 native_nodiscard native_inline constexpr simd left() const noexcept { return *this<<imm<K>; }
5017 template<unsigned K> requires(K<32) && simd_integer_element<T>
5018 native_nodiscard native_inline constexpr simd right() const noexcept { return *this>>imm<K>; }
5019 // Match native integer registers: division and run-time shifts are absent.
5021 friend simd operator<<(simd,simd) requires simd_integer_element<T> = delete;
5023 friend simd operator>>(simd,simd) requires simd_integer_element<T> = delete;
5025 friend simd operator/(simd,simd) requires simd_integer_element<T> = delete;
5027 friend simd operator%(simd,simd) requires simd_integer_element<T> = delete;
5029 template<simd_integer_element U> friend simd operator<<(simd,U) requires simd_integer_element<T> = delete;
5031 template<simd_integer_element U> friend simd operator>>(simd,U) requires simd_integer_element<T> = delete;
5033 template<simd_integer_element U> friend simd operator/(simd,U) requires simd_integer_element<T> = delete;
5035 template<simd_integer_element U> friend simd operator/(U,simd) requires simd_integer_element<T> = delete;
5037 template<simd_integer_element U> friend simd operator%(simd,U) requires simd_integer_element<T> = delete;
5039 template<simd_integer_element U> friend simd operator%(U,simd) requires simd_integer_element<T> = delete;
5041 native_inline constexpr simd & operator+=(simd b) noexcept requires(!simd_mask_element<T>) { return *this=*this+b; }
5043 native_inline constexpr simd & operator-=(simd b) noexcept requires(!simd_mask_element<T>) { return *this=*this-b; }
5045 native_inline constexpr simd & operator*=(simd b) noexcept requires(!simd_mask_element<T>) { return *this=*this*b; }
5047 native_inline constexpr simd & operator/=(simd b) noexcept requires std::same_as<T,float> { return *this=*this/b; }
5049 native_inline constexpr simd & operator&=(simd b) noexcept requires(!std::same_as<T,float>) { return *this=*this&b; }
5051 native_inline constexpr simd & operator|=(simd b) noexcept requires(!std::same_as<T,float>) { return *this=*this|b; }
5053 native_inline constexpr simd & operator^=(simd b) noexcept requires(!std::same_as<T,float>) { return *this=*this^b; }
5054 private:
5055 struct unchecked {};
5056 native_inline constexpr simd(unchecked,native_type x) noexcept : value(x) {}
5057 native_nodiscard static native_inline constexpr simd clean(storage_type x) noexcept { return simd(unchecked{},__builtin_bit_cast(native_type,x.to_native())); }
5058 template<class M> native_nodiscard static native_inline constexpr mask_type comparison(M x) noexcept {
5059 if constexpr(mask_type::compact) return mask_type::from_native(x.to_native());
5060 else return mask_type::from_storage(x);
5061 }
5062 };
5063
5065 template<simd_integer_element T,std::size_t N,::native::isa<> Arch>
5066 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (sizeof(T)==4)
5067 native_nodiscard native_inline constexpr simd<T,N,Arch> bit_select(simd<T,N,Arch> bits,simd<T,N,Arch> a,simd<T,N,Arch> b) noexcept { return (bits&a)|(~bits&b); }
5069 template<simd_integer_element T,std::size_t N,::native::isa<> Arch,class M>
5070 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (sizeof(T)==4) &&
5071 (std::same_as<M,typename simd<T,N,Arch>::mask> || std::same_as<M,simd<mask32,N,Arch>>)
5074 template<simd_integer_element T,std::size_t N,::native::isa<> Arch,class M>
5075 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (sizeof(T)==4) &&
5076 (std::same_as<M,typename simd<T,N,Arch>::mask> || std::same_as<M,simd<mask32,N,Arch>>)
5079 template<simd_integer_element T,std::size_t N,::native::isa<> Arch,class M>
5080 requires NATIVE_ARCH_REQUIRES(Arch) &&(N==2 || N==3) && (sizeof(T)==4) &&
5081 (std::same_as<M,typename simd<T,N,Arch>::mask> || std::same_as<M,simd<mask32,N,Arch>>)
5083}
5084
5085// SPDX-FileCopyrightText: 2026 Edward Kmett <ekmett@gmail.com>
5086// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
5087#endif
5088namespace native {
5089 namespace detail::NATIVE_BACKEND {
5090 enum class rounding_direction { down, up, zero };
5091
5092 template<rounding_direction Direction,std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch)
5093 native_inline constexpr simd<float,N,Arch> round_integral(simd<float,N,Arch> x) noexcept {
5094 using V = simd<float,N,Arch>;
5095 if consteval {
5096 if constexpr(Direction==rounding_direction::down) return ::native::detail::float_constant::map(::native::detail::float_constant::floor,x);
5097 else if constexpr(Direction==rounding_direction::up) return ::native::detail::float_constant::map(::native::detail::float_constant::ceil,x);
5098 else return ::native::detail::float_constant::map(::native::detail::float_constant::trunc,x);
5099 }
5100 if constexpr (N == 2 || N == 3) {
5101 return V::from_storage(round_integral<Direction>(x.to_storage()));
5102 } else {
5103#if NATIVE_HAS_AVX2
5104 constexpr int mode = (Direction == rounding_direction::down ? _MM_FROUND_TO_NEG_INF :
5105 Direction == rounding_direction::up ? _MM_FROUND_TO_POS_INF : _MM_FROUND_TO_ZERO) | _MM_FROUND_NO_EXC;
5106 if constexpr (N == 1)
5107 return V::from_native(_mm_cvtss_f32(_mm_round_ss(_mm_setzero_ps(),_mm_set_ss(x.to_native()),mode)));
5108 else if constexpr (N == 4) return V::from_native(_mm_round_ps(x.to_native(),mode));
5109 else if constexpr (N == 8) return V::from_native(_mm256_round_ps(x.to_native(),mode));
5110#if NATIVE_HAS_AVX512F
5111 else if constexpr (N == 16) return V::from_native(_mm512_roundscale_ps(x.to_native(),mode));
5112#endif
5113#elif NATIVE_HAS_ARM_NEON
5114 if constexpr (N == 1) {
5115 auto a = vdup_n_f32(x.to_native());
5116 if constexpr (Direction == rounding_direction::down) return V::from_native(vget_lane_f32(vrndm_f32(a),0));
5117 else if constexpr (Direction == rounding_direction::up) return V::from_native(vget_lane_f32(vrndp_f32(a),0));
5118 else return V::from_native(vget_lane_f32(vrnd_f32(a),0));
5119 } else {
5120 if constexpr (Direction == rounding_direction::down) return V::from_native(__builtin_elementwise_floor(x.to_native()));
5121 else if constexpr (Direction == rounding_direction::up) return V::from_native(vrndpq_f32(x.to_native()));
5122 else return V::from_native(vrndq_f32(x.to_native()));
5123 }
5124#else
5125 if constexpr (Direction == rounding_direction::down) return V::from_native(std::floor(x.to_native()));
5126 else if constexpr (Direction == rounding_direction::up) return V::from_native(std::ceil(x.to_native()));
5127 else return V::from_native(std::trunc(x.to_native()));
5128#endif
5129 }
5130 }
5131 }
5132
5133 // Directions are encoded in the instruction, never taken from the ambient
5134 // rounding mode. As for other raw arithmetic, DAZ/FZ input handling follows
5135 // the configured floating-point environment.
5141 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5143 return detail::NATIVE_BACKEND::round_integral<detail::NATIVE_BACKEND::rounding_direction::down>(x);
5144 }
5145
5147 template<std::size_t N,std::size_t M, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5148 native_nodiscard native_inline constexpr std::array<simd<float,N,Arch>,M> floor(std::array<simd<float,N,Arch>,M> const & input) noexcept {
5149 auto const & [...x] = input;
5150 return {{floor(x)...}};
5151 }
5152
5157 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5159 return detail::NATIVE_BACKEND::round_integral<detail::NATIVE_BACKEND::rounding_direction::up>(x);
5160 }
5161
5163 template<std::size_t N,std::size_t M, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5164 native_nodiscard native_inline constexpr std::array<simd<float,N,Arch>,M> ceil(std::array<simd<float,N,Arch>,M> const & input) noexcept {
5165 auto const & [...x] = input;
5166 return {{ceil(x)...}};
5167 }
5168
5173 template<std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5175 return detail::NATIVE_BACKEND::round_integral<detail::NATIVE_BACKEND::rounding_direction::zero>(x);
5176 }
5177
5179 template<std::size_t N,std::size_t M, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) && ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5180 native_nodiscard native_inline constexpr std::array<simd<float,N,Arch>,M> trunc(std::array<simd<float,N,Arch>,M> const & input) noexcept {
5181 auto const & [...x] = input;
5182 return {{trunc(x)...}};
5183 }
5184}
5185
5186// SPDX-FileCopyrightText: 2026 Edward Kmett <ekmett@gmail.com>
5187// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
5188
5189
5190namespace native::detail::NATIVE_BACKEND {
5191 template<std::size_t N,class M>
5192 native_inline std::uint32_t compaction_mask_bits(M mask) noexcept {
5193#if NATIVE_HAS_AVX2
5194 if constexpr (N > 1 && !M::compact) {
5195 if constexpr (sizeof(M) == 32)
5196 return std::uint32_t(_mm256_movemask_ps(__builtin_bit_cast(__m256,mask.to_native())));
5197 else return std::uint32_t(_mm_movemask_ps(__builtin_bit_cast(__m128,mask.to_native()))) & ((1u << N) - 1);
5198 } else
5199#elif NATIVE_HAS_ARM_NEON
5200 if constexpr (N > 1) {
5201 constexpr std::array<std::uint32_t,4> weights{1,2,4,8};
5202 return vaddvq_u32(vandq_u32(__builtin_bit_cast(uint32x4_t,mask.to_native()),vld1q_u32(weights.data()))) & ((1u << N) - 1);
5203 } else
5204#endif
5205 return std::uint32_t(mask.to_bitset()) & ((1u << N) - 1);
5206 }
5207
5208 // Four-lane shuffles use byte indices on both SSSE3 and NEON. The eight-lane
5209 // AVX2 permutation expands eight byte indices to dwords (2 KiB per table).
5210 template<bool Expand, std::size_t Width> inline constexpr auto compaction_indices = [] {
5211 constexpr auto bytes = Width == 4 ? 16 : 8;
5212 std::array<std::array<std::uint8_t,bytes>,std::size_t(1) << Width> table{};
5213 for (std::size_t mask = 0; mask != table.size(); ++mask) {
5214 std::size_t packed = 0;
5215 for (std::size_t lane = 0; lane != Width; ++lane) if (mask & (std::size_t(1) << lane)) {
5216 auto destination = Expand ? lane : packed;
5217 auto source = Expand ? packed : lane;
5218 if constexpr (Width == 4)
5219 for (std::size_t byte = 0; byte != 4; ++byte)
5220 table[mask][4 * destination + byte] = std::uint8_t(4 * source + byte);
5221 else table[mask][destination] = std::uint8_t(source);
5222 ++packed;
5223 }
5224 }
5225 return table;
5226 }();
5227
5228 template<bool Expand, class V>
5229 native_inline V compact_register(std::uint32_t mask, V input, V prior) noexcept {
5230 if constexpr (V::lanes == 1) return mask ? input : prior;
5231 else {
5232#if NATIVE_HAS_AVX512F
5233 if constexpr (sizeof(V) == 64) {
5234 auto x = __builtin_bit_cast(__m512i,input), merge = __builtin_bit_cast(__m512i,prior);
5235 if constexpr (Expand) return __builtin_bit_cast(V,_mm512_mask_expand_epi32(merge, __mmask16(mask), x));
5236 else return __builtin_bit_cast(V,_mm512_mask_compress_epi32(merge, __mmask16(mask), x));
5237 }
5238#if NATIVE_HAS_AVX512VL
5239 else if constexpr (sizeof(V) == 32) {
5240 auto x = __builtin_bit_cast(__m256i,input), merge = __builtin_bit_cast(__m256i,prior);
5241 if constexpr (Expand) return __builtin_bit_cast(V,_mm256_mask_expand_epi32(merge, __mmask8(mask), x));
5242 else return __builtin_bit_cast(V,_mm256_mask_compress_epi32(merge, __mmask8(mask), x));
5243 } else {
5244 auto x = __builtin_bit_cast(__m128i,input), merge = __builtin_bit_cast(__m128i,prior);
5245 if constexpr (Expand) return __builtin_bit_cast(V,_mm_mask_expand_epi32(merge, __mmask8(mask), x));
5246 else return __builtin_bit_cast(V,_mm_mask_compress_epi32(merge, __mmask8(mask), x));
5247 }
5248#else
5249 else
5250#endif
5251#endif
5252#if (!NATIVE_HAS_AVX512F || !NATIVE_HAS_AVX512VL) && (NATIVE_HAS_AVX2 || NATIVE_HAS_ARM_NEON)
5253 {
5254 V permuted;
5255#if NATIVE_HAS_AVX2
5256 if constexpr (sizeof(V) == 32) {
5257 auto const & row = compaction_indices<Expand,8>[mask];
5258 auto indices = _mm256_cvtepu8_epi32(_mm_loadl_epi64(reinterpret_cast<__m128i const *>(row.data())));
5259 permuted = __builtin_bit_cast(V,_mm256_permutevar8x32_epi32(__builtin_bit_cast(__m256i,input), indices));
5260 } else {
5261 auto const & row = compaction_indices<Expand,4>[mask];
5262 auto indices = _mm_loadu_si128(reinterpret_cast<__m128i const *>(row.data()));
5263 permuted = __builtin_bit_cast(V,_mm_shuffle_epi8(__builtin_bit_cast(__m128i,input), indices));
5264 }
5265#else
5266 auto const & row = compaction_indices<Expand,4>[mask];
5267 permuted = __builtin_bit_cast(V,vqtbl1q_u8(__builtin_bit_cast(uint8x16_t,input), vld1q_u8(row.data())));
5268#endif
5269 // Build the destination predicate in registers instead of round-tripping
5270 // a packed mask through the generic byte-oriented mask representation.
5271#if NATIVE_HAS_AVX2
5272 if constexpr (sizeof(V) == 32) {
5273 __m256i live;
5274 if constexpr (Expand) {
5275 auto bits = _mm256_setr_epi32(1,2,4,8,16,32,64,128);
5276 live = _mm256_cmpeq_epi32(_mm256_and_si256(_mm256_set1_epi32(int(mask)),bits),bits);
5277 } else live = _mm256_cmpgt_epi32(_mm256_set1_epi32(std::popcount(mask)),_mm256_setr_epi32(0,1,2,3,4,5,6,7));
5278 return __builtin_bit_cast(V,_mm256_blendv_epi8(__builtin_bit_cast(__m256i,prior),__builtin_bit_cast(__m256i,permuted),live));
5279 } else {
5280 __m128i live;
5281 if constexpr (Expand) {
5282 auto bits = _mm_setr_epi32(1,2,4,8);
5283 live = _mm_cmpeq_epi32(_mm_and_si128(_mm_set1_epi32(int(mask)),bits),bits);
5284 } else live = _mm_cmpgt_epi32(_mm_set1_epi32(std::popcount(mask)),_mm_setr_epi32(0,1,2,3));
5285 return __builtin_bit_cast(V,_mm_blendv_epi8(__builtin_bit_cast(__m128i,prior),__builtin_bit_cast(__m128i,permuted),live));
5286 }
5287#else
5288 uint32x4_t live;
5289 if constexpr (Expand) {
5290 constexpr std::array<std::uint32_t,4> weights{1,2,4,8};
5291 live = vtstq_u32(vdupq_n_u32(mask),vld1q_u32(weights.data()));
5292 } else {
5293 constexpr std::array<std::uint32_t,4> lanes{0,1,2,3};
5294 live = vcltq_u32(vld1q_u32(lanes.data()),vdupq_n_u32(std::uint32_t(std::popcount(mask))));
5295 }
5296 return __builtin_bit_cast(V,vbslq_u8(vreinterpretq_u8_u32(live),__builtin_bit_cast(uint8x16_t,permuted),__builtin_bit_cast(uint8x16_t,prior)));
5297#endif
5298 }
5299#endif
5300 }
5301 }
5302}
5303
5304namespace native {
5310 template<class T, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) &&
5311 (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
5312 ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5314 typename simd<T,N,Arch>::mask mask, simd<T,N,Arch> value, T fill = T{}) noexcept {
5315 if consteval {
5316 std::array<T,N> input{},output{}; value.store(input.data()); output.fill(fill);
5317 auto bits=mask.to_bitset(); std::size_t count=0;
5318 for(std::size_t i=0;i<N;++i) if((bits>>i)&1) output[count++]=input[i];
5319 return {simd<T,N,Arch>::load(output.data()),count};
5320 }
5321 using V = simd<T,N,Arch>;
5322 auto bits = detail::NATIVE_BACKEND::compaction_mask_bits<N>(mask);
5323 // Construct fill through integer object representation, without FP arithmetic.
5325 V prior = __builtin_bit_cast(V,U(std::bit_cast<std::uint32_t>(fill)));
5326 return {detail::NATIVE_BACKEND::compact_register<false>(bits, value, prior), std::size_t(std::popcount(bits))};
5327 }
5328
5332 template<class T, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) &&
5333 (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
5334 ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5336 typename simd<T,N,Arch>::mask mask, simd<T,N,Arch> packed,
5337 simd<T,N,Arch> prior) noexcept {
5338 if consteval {
5339 std::array<T,N> input{},output{}; packed.store(input.data()); prior.store(output.data());
5340 auto bits=mask.to_bitset(); std::size_t count=0;
5341 for(std::size_t i=0;i<N;++i) if((bits>>i)&1) output[i]=input[count++];
5342 return simd<T,N,Arch>::load(output.data());
5343 }
5344 auto bits = detail::NATIVE_BACKEND::compaction_mask_bits<N>(mask);
5345 auto result = detail::NATIVE_BACKEND::compact_register<true>(bits, packed, prior);
5346 if constexpr (N == 2 || N == 3) return simd<T,N,Arch>::from_storage(result.to_storage());
5347 else return result;
5348 }
5349
5354 template<class T, std::size_t N, ::native::isa<> Arch> requires NATIVE_ARCH_REQUIRES(Arch) &&
5355 (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
5356 ::NATIVE_BACKEND_NAMESPACE::float_shape<N>
5357 native_inline constexpr std::size_t compress_store(T * destination, std::size_t capacity,
5358 typename simd<T,N,Arch>::mask mask, simd<T,N,Arch> value) noexcept {
5359 if consteval {
5360 std::array<T,N> input{}; value.store(input.data());
5361 auto bits=mask.to_bitset(); std::size_t written=0;
5362 for(std::size_t i=0;i<N && written<capacity;++i)
5363 if((bits>>i)&1) destination[written++]=input[i];
5364 return written;
5365 }
5366 auto bits = detail::NATIVE_BACKEND::compaction_mask_bits<N>(mask);
5367 auto selected = std::size_t(std::popcount(bits));
5368 auto written = capacity < selected ? capacity : selected;
5369 if (!written) return 0;
5370#if NATIVE_HAS_AVX512F
5371 if constexpr (N > 1 && (sizeof(value) == 64 || NATIVE_HAS_AVX512VL)) {
5372 // Compress in registers, then store only the bounded contiguous prefix.
5373 // This also avoids trimming the source predicate when capacity is small.
5374 auto prefix = (std::uint32_t(1) << written) - 1;
5375 if constexpr (sizeof(value) == 64) {
5376 auto packed = _mm512_maskz_compress_epi32(__mmask16(bits),__builtin_bit_cast(__m512i,value));
5377 _mm512_mask_storeu_epi32(destination,__mmask16(prefix),packed);
5378 } else if constexpr (sizeof(value) == 32) {
5379 auto packed = _mm256_maskz_compress_epi32(__mmask8(bits),__builtin_bit_cast(__m256i,value));
5380 _mm256_mask_storeu_epi32(destination,__mmask8(prefix),packed);
5381 } else {
5382 auto packed = _mm_maskz_compress_epi32(__mmask8(bits),__builtin_bit_cast(__m128i,value));
5383 _mm_mask_storeu_epi32(destination,__mmask8(prefix),packed);
5384 }
5385 return written;
5386 }
5387#endif
5388 std::array<T,N> packed;
5389 compress(mask, value).value.store(packed.data());
5390 std::memcpy(destination, packed.data(), written * sizeof(T));
5391 return written;
5392 }
5393}
5394
5395#if NATIVE_HAS_WASM_SIMD128 || defined(NATIVE_DOXYGEN)
5396// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
5397// Included only in the SIMD128 backend target scope.
5398namespace native {
5399 namespace detail {
5400 template<class T>
5401 concept wasm_number =
5402 simd_integer_element<T> || std::same_as<T, float> || std::same_as<T, double>;
5403 template<class T>
5404 using wasm_word = std::conditional_t<
5405 sizeof(T) == 1, std::uint8_t,
5406 std::conditional_t<sizeof(T) == 2, std::uint16_t,
5407 std::conditional_t<sizeof(T) == 4, std::uint32_t, std::uint64_t>>>;
5408 template<class T>
5409 using wasm_format =
5410 std::conditional_t<sizeof(T) == 4, constexpr_float::binary32, constexpr_float::binary64>;
5411
5412 template<class V>
5413 constexpr auto wasm_lanes(V v) noexcept {
5414 std::array<typename V::value_type, V::lanes> a{};
5415 v.store(a.data());
5416 return a;
5417 }
5418
5419 template<class V, class F, class... W>
5420 constexpr V wasm_map(F f, V v, W... w) noexcept {
5421 auto inputs = std::tuple{wasm_lanes(v), wasm_lanes(w)...};
5422 std::array<typename V::value_type, V::lanes> r{};
5423 for (std::size_t i = 0; i < V::lanes; ++i) {
5424 r[i] = std::apply([&](auto const &... a) { return f(a[i]...); }, inputs);
5425 }
5426 return V::load(r.data());
5427 }
5428
5429 template<class T, std::size_t N, isa<> A>
5430 requires ordinary_simd_element<T> && (A.has(wasm_feature::simd128)) &&
5431 (wasm_number<T> || simd_mask_element<T>) && (sizeof(T) * N == 16)
5432 struct value_traits<simd<T, N, A>> {
5433 static constexpr isa<> value = A;
5434 static constexpr bool known = true;
5435 static constexpr bool aggregate_default = false;
5436 };
5437 } // namespace detail
5438
5440 template<class U, std::size_t N, isa<> A>
5441 requires(A.has(wasm_feature::simd128)) && (sizeof(U) * N == 16)
5442 struct alignas(16) simd<mask_lane<U>, N, A> {
5443 using value_type = mask_lane<U>;
5444 using native_type = v128_t;
5445 using mask_type = simd;
5446 using mask = simd;
5447 template<class T>
5448 using rebind = simd<T, N, A>;
5449 static constexpr auto architecture = A;
5450 static constexpr std::size_t lanes = N;
5451 static constexpr bool compact = false;
5452
5453 private:
5454 native_type value_{};
5455
5456 public:
5458 constexpr simd() noexcept = default;
5459
5461 explicit native_inline constexpr simd(bool x) noexcept
5462 : simd(from_bitset(x ? ~std::uint64_t{} : 0)) {}
5463
5465 native_inline constexpr native_type to_native() const noexcept {
5466 return value_;
5467 }
5468
5470 static native_inline constexpr simd unsafe_from_native(native_type x) noexcept {
5471 simd r;
5472 r.value_ = x;
5473 return r;
5474 }
5475
5477 static native_inline constexpr simd from_native(native_type x) noexcept {
5478 if consteval {
5479 auto words = __builtin_bit_cast(std::array<U, N>, x);
5480 for (auto & w : words) {
5481 w = w ? U(~U(0)) : U(0);
5482 }
5483 return unsafe_from_native(__builtin_bit_cast(native_type, words));
5484 } else {
5485 if constexpr (sizeof(U) == 1) {
5486 return unsafe_from_native(wasm_i8x16_ne(x, wasm_i32x4_splat(0)));
5487 } else if constexpr (sizeof(U) == 2) {
5488 return unsafe_from_native(wasm_i16x8_ne(x, wasm_i32x4_splat(0)));
5489 } else if constexpr (sizeof(U) == 4) {
5490 return unsafe_from_native(wasm_i32x4_ne(x, wasm_i32x4_splat(0)));
5491 } else if constexpr (sizeof(U) == 8) {
5492 return unsafe_from_native(wasm_i64x2_ne(x, wasm_i32x4_splat(0)));
5493 }
5494 }
5495 }
5496
5498 static native_inline constexpr simd from_bitset(std::uint64_t bits) noexcept {
5499 std::array<U, N> a{};
5500 for (std::size_t i = 0; i < N; ++i) {
5501 a[i] = ((bits >> i) & 1) ? U(~U(0)) : U(0);
5502 }
5503 return unsafe_from_native(__builtin_bit_cast(native_type, a));
5504 }
5505
5507 native_inline constexpr std::uint64_t to_bitset() const noexcept {
5508 if consteval {
5509 auto a = __builtin_bit_cast(std::array<U, N>, value_);
5510 std::uint64_t r = 0;
5511 for (std::size_t i = 0; i < N; ++i) {
5512 r |= std::uint64_t(a[i] != 0) << i;
5513 }
5514 return r;
5515 } else {
5516 if constexpr (sizeof(U) == 1) {
5517 return wasm_i8x16_bitmask(value_);
5518 } else if constexpr (sizeof(U) == 2) {
5519 return wasm_i16x8_bitmask(value_);
5520 } else if constexpr (sizeof(U) == 4) {
5521 return wasm_i32x4_bitmask(value_);
5522 } else if constexpr (sizeof(U) == 8) {
5523 return wasm_i64x2_bitmask(value_);
5524 }
5525 }
5526 }
5527
5529 native_inline constexpr std::uint64_t bits() const noexcept {
5530 return to_bitset();
5531 }
5532
5534 static native_inline constexpr simd load(value_type const * p) noexcept {
5535 std::array<U, N> a{};
5536 for (std::size_t i = 0; i < N; ++i) {
5537 a[i] = p[i].to_bits();
5538 }
5539 return unsafe_from_native(__builtin_bit_cast(native_type, a));
5540 }
5541
5543 native_inline constexpr void store(value_type * p) const noexcept {
5544 auto a = __builtin_bit_cast(std::array<U, N>, value_);
5545 for (std::size_t i = 0; i < N; ++i) {
5546 p[i] = value_type::from_bits(a[i]);
5547 }
5548 }
5549
5551 template<std::size_t Align>
5552 static native_inline constexpr simd load_memory(value_type const * p) noexcept {
5553 return load(p);
5554 }
5555
5557 template<std::size_t Align>
5558 native_inline constexpr void store_memory(value_type * p) const noexcept {
5559 store(p);
5560 }
5561
5563 friend native_inline constexpr bool any(simd v) noexcept {
5564 if consteval {
5565 return v.to_bitset() != 0;
5566 } else {
5567 return wasm_v128_any_true(v.value_);
5568 }
5569 }
5570
5572 friend native_inline constexpr bool all(simd v) noexcept {
5573 if consteval {
5574 return v.to_bitset() == ((std::uint64_t{1} << N) - 1);
5575 } else {
5576 if constexpr (sizeof(U) == 1) {
5577 return wasm_i8x16_all_true(v.value_);
5578 } else if constexpr (sizeof(U) == 2) {
5579 return wasm_i16x8_all_true(v.value_);
5580 } else if constexpr (sizeof(U) == 4) {
5581 return wasm_i32x4_all_true(v.value_);
5582 } else {
5583 return wasm_i64x2_all_true(v.value_);
5584 }
5585 }
5586 }
5587
5589 friend native_inline constexpr bool none(simd v) noexcept {
5590 return !any(v);
5591 }
5592
5594 friend native_inline constexpr simd operator&(simd a, simd b) noexcept {
5595 if consteval {
5596 return from_bitset(a.to_bitset() & b.to_bitset());
5597 } else {
5598 return unsafe_from_native(wasm_v128_and(a.value_, b.value_));
5599 }
5600 }
5601
5603 native_inline constexpr simd & operator&=(simd b) noexcept {
5604 return *this = *this & b;
5605 }
5606
5608 friend native_inline constexpr simd operator|(simd a, simd b) noexcept {
5609 if consteval {
5610 return from_bitset(a.to_bitset() | b.to_bitset());
5611 } else {
5612 return unsafe_from_native(wasm_v128_or(a.value_, b.value_));
5613 }
5614 }
5615
5617 native_inline constexpr simd & operator|=(simd b) noexcept {
5618 return *this = *this | b;
5619 }
5620
5622 friend native_inline constexpr simd operator^(simd a, simd b) noexcept {
5623 if consteval {
5624 return from_bitset(a.to_bitset() ^ b.to_bitset());
5625 } else {
5626 return unsafe_from_native(wasm_v128_xor(a.value_, b.value_));
5627 }
5628 }
5629
5631 native_inline constexpr simd & operator^=(simd b) noexcept {
5632 return *this = *this ^ b;
5633 }
5634
5636 friend native_inline constexpr simd operator~(simd a) noexcept {
5637 if consteval {
5638 return from_bitset(~a.to_bitset());
5639 } else {
5640 return unsafe_from_native(wasm_v128_not(a.value_));
5641 }
5642 }
5643
5645 friend native_inline constexpr simd operator!(simd a) noexcept {
5646 return ~a;
5647 }
5648
5650 friend native_inline constexpr simd operator==(simd a, simd b) noexcept {
5651 return ~(a ^ b);
5652 }
5653
5655 friend native_inline constexpr simd operator!=(simd a, simd b) noexcept {
5656 return a ^ b;
5657 }
5658
5660 friend native_inline constexpr simd select(simd m, simd a, simd b) noexcept {
5661 return (m & a) | (~m & b);
5662 }
5663 };
5664
5666 template<detail::wasm_number T, std::size_t N, isa<> A>
5667 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
5668 struct alignas(16) simd<T, N, A> {
5669 using value_type = T;
5670 using native_type = v128_t;
5671 using word_type = detail::wasm_word<T>;
5672 using mask_type = simd<mask_lane<word_type>, N, A>;
5673 using mask = mask_type;
5674 using bits_type = simd<word_type, N, A>;
5675 using register_type = simd;
5676 template<class U>
5677 using rebind = simd<U, N, A>;
5678 static constexpr auto architecture = A;
5679 static constexpr std::size_t lanes = N;
5680
5681 private:
5682 native_type value_{};
5683
5684 public:
5686 constexpr simd() noexcept = default;
5687
5689 native_inline constexpr simd(T x) noexcept
5690 requires std::is_floating_point_v<T>
5691 {
5692 if consteval {
5693 std::array<T, N> a{};
5694 a.fill(x);
5695 value_ = __builtin_bit_cast(native_type, a);
5696 } else {
5697 if constexpr (std::same_as<T, float>) {
5698 value_ = wasm_f32x4_splat(x);
5699 } else {
5700 value_ = wasm_f64x2_splat(x);
5701 }
5702 }
5703 }
5704
5706 template<simd_integer_element U>
5707 requires simd_integer_element<T>
5708 native_inline constexpr simd(U input) noexcept {
5709 auto x = static_cast<T>(input);
5710 if consteval {
5711 std::array<T, N> a{};
5712 a.fill(x);
5713 value_ = __builtin_bit_cast(native_type, a);
5714 } else {
5715 if constexpr (std::same_as<T, std::int8_t>) {
5716 value_ = wasm_i8x16_splat(x);
5717 } else if constexpr (std::same_as<T, std::uint8_t>) {
5718 value_ = wasm_u8x16_splat(x);
5719 } else if constexpr (std::same_as<T, std::int16_t>) {
5720 value_ = wasm_i16x8_splat(x);
5721 } else if constexpr (std::same_as<T, std::uint16_t>) {
5722 value_ = wasm_u16x8_splat(x);
5723 } else if constexpr (std::same_as<T, std::int32_t>) {
5724 value_ = wasm_i32x4_splat(x);
5725 } else if constexpr (std::same_as<T, std::uint32_t>) {
5726 value_ = wasm_u32x4_splat(x);
5727 } else if constexpr (std::same_as<T, std::int64_t>) {
5728 value_ = wasm_i64x2_splat(x);
5729 } else if constexpr (std::same_as<T, std::uint64_t>) {
5730 value_ = wasm_u64x2_splat(x);
5731 }
5732 }
5733 }
5734
5736 native_inline constexpr explicit simd(std::array<T, N> const & a) noexcept
5737 : simd(load(a.data())) {}
5738
5740 template<class... U>
5741 requires(sizeof...(U) == N && ((std::same_as<U, T> && ...) ||
5742 (simd_integer_element<T> && (simd_integer_element<U> && ...))))
5743 native_inline constexpr simd(U... x) noexcept : simd(std::array<T, N>{static_cast<T>(x)...}) {}
5744
5746 native_inline constexpr native_type to_native() const noexcept {
5747 return value_;
5748 }
5749
5751 static native_inline constexpr simd from_native(native_type x) noexcept {
5752 simd r;
5753 r.value_ = x;
5754 return r;
5755 }
5756
5758 static native_inline constexpr simd unsafe_from_native(native_type x) noexcept {
5759 return from_native(x);
5760 }
5761
5763 static native_inline constexpr simd load(T const * p) noexcept {
5764 if consteval {
5765 std::array<T, N> a{};
5766 for (std::size_t i = 0; i < N; ++i) {
5767 a[i] = p[i];
5768 }
5769 return from_native(__builtin_bit_cast(native_type, a));
5770 } else {
5771 return from_native(wasm_v128_load(p));
5772 }
5773 }
5774
5776 native_inline constexpr void store(T * p) const noexcept {
5777 if consteval {
5778 auto a = __builtin_bit_cast(std::array<T, N>, value_);
5779 for (std::size_t i = 0; i < N; ++i) {
5780 p[i] = a[i];
5781 }
5782 } else {
5783 wasm_v128_store(p, value_);
5784 }
5785 }
5786
5788 template<std::size_t Align>
5789 static native_inline constexpr simd load_memory(T const * p) noexcept {
5790 if consteval {
5791 return load(p);
5792 } else {
5793 return load(static_cast<T const *>(__builtin_assume_aligned(p, Align)));
5794 }
5795 }
5796
5798 template<std::size_t Align>
5799 native_inline constexpr void store_memory(T * p) const noexcept {
5800 if consteval {
5801 store(p);
5802 } else {
5803 store(static_cast<T *>(__builtin_assume_aligned(p, Align)));
5804 }
5805 }
5806
5808 static native_inline constexpr simd load_partial(T const * p, std::size_t n,
5809 T fill = {}) noexcept {
5810 std::array<T, N> a{};
5811 a.fill(fill);
5812 for (std::size_t i = 0; i < n; ++i) {
5813 a[i] = p[i];
5814 }
5815 return load(a.data());
5816 }
5817
5819 native_inline constexpr void store_partial(T * p, std::size_t n) const noexcept {
5820 auto a = detail::wasm_lanes(*this);
5821 for (std::size_t i = 0; i < n; ++i) {
5822 p[i] = a[i];
5823 }
5824 }
5825
5827 static native_inline constexpr simd load_bits(word_type const * p) noexcept {
5828 return from_native(bits_type::load(p).to_native());
5829 }
5830
5832 native_inline constexpr void store_bits(word_type * p) const noexcept {
5833 bits().store(p);
5834 }
5835
5837 native_inline constexpr bits_type bits() const noexcept {
5838 return bits_type::from_native(value_);
5839 }
5840
5842 static native_inline constexpr simd from_bits(bits_type x) noexcept {
5843 return from_native(x.to_native());
5844 }
5845
5847 template<std::size_t I>
5848 requires(I < N)
5849 native_inline constexpr T get() const noexcept {
5850 if consteval {
5851 return __builtin_bit_cast(std::array<T, N>, value_)[I];
5852 } else {
5853 if constexpr (std::same_as<T, float>) {
5854 return wasm_f32x4_extract_lane(value_, I);
5855 } else if constexpr (std::same_as<T, double>) {
5856 return wasm_f64x2_extract_lane(value_, I);
5857 } else if constexpr (std::same_as<T, std::int8_t>) {
5858 return wasm_i8x16_extract_lane(value_, I);
5859 } else if constexpr (std::same_as<T, std::uint8_t>) {
5860 return wasm_u8x16_extract_lane(value_, I);
5861 } else if constexpr (std::same_as<T, std::int16_t>) {
5862 return wasm_i16x8_extract_lane(value_, I);
5863 } else if constexpr (std::same_as<T, std::uint16_t>) {
5864 return wasm_u16x8_extract_lane(value_, I);
5865 } else if constexpr (std::same_as<T, std::int32_t>) {
5866 return wasm_i32x4_extract_lane(value_, I);
5867 } else if constexpr (std::same_as<T, std::uint32_t>) {
5868 return wasm_u32x4_extract_lane(value_, I);
5869 } else if constexpr (std::same_as<T, std::int64_t>) {
5870 return wasm_i64x2_extract_lane(value_, I);
5871 } else if constexpr (std::same_as<T, std::uint64_t>) {
5872 return wasm_u64x2_extract_lane(value_, I);
5873 }
5874 }
5875 }
5876
5878 template<std::size_t I>
5879 requires(I < N)
5880 native_inline constexpr simd replace(T x) const noexcept {
5881 if consteval {
5882 auto a = __builtin_bit_cast(std::array<T, N>, value_);
5883 a[I] = x;
5884 return load(a.data());
5885 } else {
5886 if constexpr (std::same_as<T, float>) {
5887 return from_native(wasm_f32x4_replace_lane(value_, I, x));
5888 } else if constexpr (std::same_as<T, double>) {
5889 return from_native(wasm_f64x2_replace_lane(value_, I, x));
5890 } else if constexpr (std::same_as<T, std::int8_t>) {
5891 return from_native(wasm_i8x16_replace_lane(value_, I, x));
5892 } else if constexpr (std::same_as<T, std::uint8_t>) {
5893 return from_native(wasm_u8x16_replace_lane(value_, I, x));
5894 } else if constexpr (std::same_as<T, std::int16_t>) {
5895 return from_native(wasm_i16x8_replace_lane(value_, I, x));
5896 } else if constexpr (std::same_as<T, std::uint16_t>) {
5897 return from_native(wasm_u16x8_replace_lane(value_, I, x));
5898 } else if constexpr (std::same_as<T, std::int32_t>) {
5899 return from_native(wasm_i32x4_replace_lane(value_, I, x));
5900 } else if constexpr (std::same_as<T, std::uint32_t>) {
5901 return from_native(wasm_u32x4_replace_lane(value_, I, x));
5902 } else if constexpr (std::same_as<T, std::int64_t>) {
5903 return from_native(wasm_i64x2_replace_lane(value_, I, x));
5904 } else if constexpr (std::same_as<T, std::uint64_t>) {
5905 return from_native(wasm_u64x2_replace_lane(value_, I, x));
5906 }
5907 }
5908 }
5909
5911 friend native_inline constexpr simd operator+(simd a, simd b) noexcept {
5912 if consteval {
5913 return detail::wasm_map(
5914 [](T x, T y) {
5915 if constexpr (std::is_floating_point_v<T>) {
5916 using format_type = detail::wasm_format<T>;
5917 return std::bit_cast<T>(detail::constexpr_float::add_bits<format_type>(
5918 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
5919 } else {
5920 return std::bit_cast<T>(
5921 word_type(std::uint64_t(word_type(x)) + std::uint64_t(word_type(y))));
5922 }
5923 },
5924 a, b);
5925 } else {
5926 if constexpr (std::same_as<T, float>) {
5927 return from_native(wasm_f32x4_add(a.value_, b.value_));
5928 } else if constexpr (std::same_as<T, double>) {
5929 return from_native(wasm_f64x2_add(a.value_, b.value_));
5930 } else if constexpr (std::same_as<T, std::int8_t>) {
5931 return from_native(wasm_i8x16_add(a.value_, b.value_));
5932 } else if constexpr (std::same_as<T, std::uint8_t>) {
5933 return from_native(wasm_i8x16_add(a.value_, b.value_));
5934 } else if constexpr (std::same_as<T, std::int16_t>) {
5935 return from_native(wasm_i16x8_add(a.value_, b.value_));
5936 } else if constexpr (std::same_as<T, std::uint16_t>) {
5937 return from_native(wasm_i16x8_add(a.value_, b.value_));
5938 } else if constexpr (std::same_as<T, std::int32_t>) {
5939 return from_native(wasm_i32x4_add(a.value_, b.value_));
5940 } else if constexpr (std::same_as<T, std::uint32_t>) {
5941 return from_native(wasm_i32x4_add(a.value_, b.value_));
5942 } else if constexpr (std::same_as<T, std::int64_t>) {
5943 return from_native(wasm_i64x2_add(a.value_, b.value_));
5944 } else if constexpr (std::same_as<T, std::uint64_t>) {
5945 return from_native(wasm_i64x2_add(a.value_, b.value_));
5946 }
5947 }
5948 }
5949
5951 native_inline constexpr simd & operator+=(simd b) noexcept {
5952 return *this = *this + b;
5953 }
5954
5956 friend native_inline constexpr simd operator-(simd a, simd b) noexcept {
5957 if consteval {
5958 return detail::wasm_map(
5959 [](T x, T y) {
5960 if constexpr (std::is_floating_point_v<T>) {
5961 using format_type = detail::wasm_format<T>;
5962 return std::bit_cast<T>(detail::constexpr_float::sub_bits<format_type>(
5963 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
5964 } else {
5965 return std::bit_cast<T>(
5966 word_type(std::uint64_t(word_type(x)) - std::uint64_t(word_type(y))));
5967 }
5968 },
5969 a, b);
5970 } else {
5971 if constexpr (std::same_as<T, float>) {
5972 return from_native(wasm_f32x4_sub(a.value_, b.value_));
5973 } else if constexpr (std::same_as<T, double>) {
5974 return from_native(wasm_f64x2_sub(a.value_, b.value_));
5975 } else if constexpr (std::same_as<T, std::int8_t>) {
5976 return from_native(wasm_i8x16_sub(a.value_, b.value_));
5977 } else if constexpr (std::same_as<T, std::uint8_t>) {
5978 return from_native(wasm_i8x16_sub(a.value_, b.value_));
5979 } else if constexpr (std::same_as<T, std::int16_t>) {
5980 return from_native(wasm_i16x8_sub(a.value_, b.value_));
5981 } else if constexpr (std::same_as<T, std::uint16_t>) {
5982 return from_native(wasm_i16x8_sub(a.value_, b.value_));
5983 } else if constexpr (std::same_as<T, std::int32_t>) {
5984 return from_native(wasm_i32x4_sub(a.value_, b.value_));
5985 } else if constexpr (std::same_as<T, std::uint32_t>) {
5986 return from_native(wasm_i32x4_sub(a.value_, b.value_));
5987 } else if constexpr (std::same_as<T, std::int64_t>) {
5988 return from_native(wasm_i64x2_sub(a.value_, b.value_));
5989 } else if constexpr (std::same_as<T, std::uint64_t>) {
5990 return from_native(wasm_i64x2_sub(a.value_, b.value_));
5991 }
5992 }
5993 }
5994
5996 native_inline constexpr simd & operator-=(simd b) noexcept {
5997 return *this = *this - b;
5998 }
5999
6001 friend native_inline constexpr simd operator*(simd a, simd b) noexcept
6002 requires(sizeof(T) > 1)
6003 {
6004 if consteval {
6005 return detail::wasm_map(
6006 [](T x, T y) {
6007 if constexpr (std::is_floating_point_v<T>) {
6008 using format_type = detail::wasm_format<T>;
6009 return std::bit_cast<T>(detail::constexpr_float::mul_bits<format_type>(
6010 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
6011 } else {
6012 return std::bit_cast<T>(
6013 word_type(std::uint64_t(word_type(x)) * std::uint64_t(word_type(y))));
6014 }
6015 },
6016 a, b);
6017 } else {
6018 if constexpr (std::same_as<T, float>) {
6019 return from_native(wasm_f32x4_mul(a.value_, b.value_));
6020 } else if constexpr (std::same_as<T, double>) {
6021 return from_native(wasm_f64x2_mul(a.value_, b.value_));
6022 } else if constexpr (std::same_as<T, std::int16_t>) {
6023 return from_native(wasm_i16x8_mul(a.value_, b.value_));
6024 } else if constexpr (std::same_as<T, std::uint16_t>) {
6025 return from_native(wasm_i16x8_mul(a.value_, b.value_));
6026 } else if constexpr (std::same_as<T, std::int32_t>) {
6027 return from_native(wasm_i32x4_mul(a.value_, b.value_));
6028 } else if constexpr (std::same_as<T, std::uint32_t>) {
6029 return from_native(wasm_i32x4_mul(a.value_, b.value_));
6030 } else if constexpr (std::same_as<T, std::int64_t>) {
6031 return from_native(wasm_i64x2_mul(a.value_, b.value_));
6032 } else if constexpr (std::same_as<T, std::uint64_t>) {
6033 return from_native(wasm_i64x2_mul(a.value_, b.value_));
6034 }
6035 }
6036 }
6037
6039 native_inline constexpr simd & operator*=(simd b) noexcept
6040 requires(sizeof(T) > 1)
6041 {
6042 return *this = *this * b;
6043 }
6044
6046 friend native_inline constexpr simd operator/(simd a, simd b) noexcept
6047 requires std::is_floating_point_v<T>
6048 {
6049 if consteval {
6050 return detail::wasm_map(
6051 [](T x, T y) {
6052 using format_type = detail::wasm_format<T>;
6053 return std::bit_cast<T>(detail::constexpr_float::div_bits<format_type>(
6054 std::bit_cast<word_type>(x), std::bit_cast<word_type>(y)));
6055 },
6056 a, b);
6057 } else {
6058 if constexpr (std::same_as<T, float>) {
6059 return from_native(wasm_f32x4_div(a.value_, b.value_));
6060 } else if constexpr (std::same_as<T, double>) {
6061 return from_native(wasm_f64x2_div(a.value_, b.value_));
6062 }
6063 }
6064 }
6065
6067 native_inline constexpr simd & operator/=(simd b) noexcept
6068 requires std::is_floating_point_v<T>
6069 {
6070 return *this = *this / b;
6071 }
6072
6074 friend native_inline constexpr simd operator&(simd a, simd b) noexcept {
6075 if consteval {
6076 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6077 auto y = __builtin_bit_cast(std::array<word_type, N>, b.value_);
6078 for (std::size_t i = 0; i < N; ++i) {
6079 x[i] &= y[i];
6080 }
6081 return from_native(__builtin_bit_cast(native_type, x));
6082 } else {
6083 return from_native(wasm_v128_and(a.value_, b.value_));
6084 }
6085 }
6086
6088 native_inline constexpr simd & operator&=(simd b) noexcept {
6089 return *this = *this & b;
6090 }
6091
6093 friend native_inline constexpr simd operator|(simd a, simd b) noexcept {
6094 if consteval {
6095 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6096 auto y = __builtin_bit_cast(std::array<word_type, N>, b.value_);
6097 for (std::size_t i = 0; i < N; ++i) {
6098 x[i] |= y[i];
6099 }
6100 return from_native(__builtin_bit_cast(native_type, x));
6101 } else {
6102 return from_native(wasm_v128_or(a.value_, b.value_));
6103 }
6104 }
6105
6107 native_inline constexpr simd & operator|=(simd b) noexcept {
6108 return *this = *this | b;
6109 }
6110
6112 friend native_inline constexpr simd operator^(simd a, simd b) noexcept {
6113 if consteval {
6114 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6115 auto y = __builtin_bit_cast(std::array<word_type, N>, b.value_);
6116 for (std::size_t i = 0; i < N; ++i) {
6117 x[i] ^= y[i];
6118 }
6119 return from_native(__builtin_bit_cast(native_type, x));
6120 } else {
6121 return from_native(wasm_v128_xor(a.value_, b.value_));
6122 }
6123 }
6124
6126 native_inline constexpr simd & operator^=(simd b) noexcept {
6127 return *this = *this ^ b;
6128 }
6129
6131 friend native_inline constexpr simd operator~(simd a) noexcept {
6132 if consteval {
6133 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6134 for (auto & w : x) {
6135 w = word_type(~w);
6136 }
6137 return from_native(__builtin_bit_cast(native_type, x));
6138 } else {
6139 return from_native(wasm_v128_not(a.value_));
6140 }
6141 }
6142
6144 friend native_inline constexpr simd operator-(simd a) noexcept {
6145 if consteval {
6146 auto x = __builtin_bit_cast(std::array<word_type, N>, a.value_);
6147 for (auto & w : x) {
6148 if constexpr (std::is_floating_point_v<T>) {
6149 w ^= word_type{1} << (sizeof(T) * 8 - 1);
6150 } else {
6151 w = word_type(0 - w);
6152 }
6153 }
6154 return from_native(__builtin_bit_cast(native_type, x));
6155 } else {
6156 if constexpr (std::same_as<T, float>) {
6157 return from_native(wasm_f32x4_neg(a.value_));
6158 } else if constexpr (std::same_as<T, double>) {
6159 return from_native(wasm_f64x2_neg(a.value_));
6160 } else if constexpr (std::same_as<T, std::int8_t>) {
6161 return from_native(wasm_i8x16_neg(a.value_));
6162 } else if constexpr (std::same_as<T, std::uint8_t>) {
6163 return from_native(wasm_i8x16_neg(a.value_));
6164 } else if constexpr (std::same_as<T, std::int16_t>) {
6165 return from_native(wasm_i16x8_neg(a.value_));
6166 } else if constexpr (std::same_as<T, std::uint16_t>) {
6167 return from_native(wasm_i16x8_neg(a.value_));
6168 } else if constexpr (std::same_as<T, std::int32_t>) {
6169 return from_native(wasm_i32x4_neg(a.value_));
6170 } else if constexpr (std::same_as<T, std::uint32_t>) {
6171 return from_native(wasm_i32x4_neg(a.value_));
6172 } else if constexpr (std::same_as<T, std::int64_t>) {
6173 return from_native(wasm_i64x2_neg(a.value_));
6174 } else if constexpr (std::same_as<T, std::uint64_t>) {
6175 return from_native(wasm_i64x2_neg(a.value_));
6176 }
6177 }
6178 }
6179
6181 friend native_inline constexpr mask_type operator==(simd a, simd b) noexcept {
6182 if consteval {
6183 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6184 std::uint64_t bits = 0;
6185 for (std::size_t i = 0; i < N; ++i) {
6186 bits |= std::uint64_t(x[i] == y[i]) << i;
6187 }
6189 } else {
6190 if constexpr (std::same_as<T, float>) {
6191 return mask_type::unsafe_from_native(wasm_f32x4_eq(a.value_, b.value_));
6192 } else if constexpr (std::same_as<T, double>) {
6193 return mask_type::unsafe_from_native(wasm_f64x2_eq(a.value_, b.value_));
6194 } else if constexpr (std::same_as<T, std::int8_t>) {
6195 return mask_type::unsafe_from_native(wasm_i8x16_eq(a.value_, b.value_));
6196 } else if constexpr (std::same_as<T, std::uint8_t>) {
6197 return mask_type::unsafe_from_native(wasm_i8x16_eq(a.value_, b.value_));
6198 } else if constexpr (std::same_as<T, std::int16_t>) {
6199 return mask_type::unsafe_from_native(wasm_i16x8_eq(a.value_, b.value_));
6200 } else if constexpr (std::same_as<T, std::uint16_t>) {
6201 return mask_type::unsafe_from_native(wasm_i16x8_eq(a.value_, b.value_));
6202 } else if constexpr (std::same_as<T, std::int32_t>) {
6203 return mask_type::unsafe_from_native(wasm_i32x4_eq(a.value_, b.value_));
6204 } else if constexpr (std::same_as<T, std::uint32_t>) {
6205 return mask_type::unsafe_from_native(wasm_i32x4_eq(a.value_, b.value_));
6206 } else if constexpr (std::same_as<T, std::int64_t>) {
6207 return mask_type::unsafe_from_native(wasm_i64x2_eq(a.value_, b.value_));
6208 } else if constexpr (std::same_as<T, std::uint64_t>) {
6209 return mask_type::unsafe_from_native(wasm_i64x2_eq(a.value_, b.value_));
6210 }
6211 }
6212 }
6213
6215 friend native_inline constexpr mask_type operator!=(simd a, simd b) noexcept {
6216 if consteval {
6217 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6218 std::uint64_t bits = 0;
6219 for (std::size_t i = 0; i < N; ++i) {
6220 bits |= std::uint64_t(x[i] != y[i]) << i;
6221 }
6223 } else {
6224 if constexpr (std::same_as<T, float>) {
6225 return mask_type::unsafe_from_native(wasm_f32x4_ne(a.value_, b.value_));
6226 } else if constexpr (std::same_as<T, double>) {
6227 return mask_type::unsafe_from_native(wasm_f64x2_ne(a.value_, b.value_));
6228 } else if constexpr (std::same_as<T, std::int8_t>) {
6229 return mask_type::unsafe_from_native(wasm_i8x16_ne(a.value_, b.value_));
6230 } else if constexpr (std::same_as<T, std::uint8_t>) {
6231 return mask_type::unsafe_from_native(wasm_i8x16_ne(a.value_, b.value_));
6232 } else if constexpr (std::same_as<T, std::int16_t>) {
6233 return mask_type::unsafe_from_native(wasm_i16x8_ne(a.value_, b.value_));
6234 } else if constexpr (std::same_as<T, std::uint16_t>) {
6235 return mask_type::unsafe_from_native(wasm_i16x8_ne(a.value_, b.value_));
6236 } else if constexpr (std::same_as<T, std::int32_t>) {
6237 return mask_type::unsafe_from_native(wasm_i32x4_ne(a.value_, b.value_));
6238 } else if constexpr (std::same_as<T, std::uint32_t>) {
6239 return mask_type::unsafe_from_native(wasm_i32x4_ne(a.value_, b.value_));
6240 } else if constexpr (std::same_as<T, std::int64_t>) {
6241 return mask_type::unsafe_from_native(wasm_i64x2_ne(a.value_, b.value_));
6242 } else if constexpr (std::same_as<T, std::uint64_t>) {
6243 return mask_type::unsafe_from_native(wasm_i64x2_ne(a.value_, b.value_));
6244 }
6245 }
6246 }
6247
6249 friend native_inline constexpr mask_type operator<(simd a, simd b) noexcept {
6250 if consteval {
6251 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6252 std::uint64_t bits = 0;
6253 for (std::size_t i = 0; i < N; ++i) {
6254 bits |= std::uint64_t(x[i] < y[i]) << i;
6255 }
6257 } else {
6258 if constexpr (std::same_as<T, float>) {
6259 return mask_type::unsafe_from_native(wasm_f32x4_lt(a.value_, b.value_));
6260 } else if constexpr (std::same_as<T, double>) {
6261 return mask_type::unsafe_from_native(wasm_f64x2_lt(a.value_, b.value_));
6262 } else if constexpr (std::same_as<T, std::int8_t>) {
6263 return mask_type::unsafe_from_native(wasm_i8x16_lt(a.value_, b.value_));
6264 } else if constexpr (std::same_as<T, std::uint8_t>) {
6265 return mask_type::unsafe_from_native(wasm_u8x16_lt(a.value_, b.value_));
6266 } else if constexpr (std::same_as<T, std::int16_t>) {
6267 return mask_type::unsafe_from_native(wasm_i16x8_lt(a.value_, b.value_));
6268 } else if constexpr (std::same_as<T, std::uint16_t>) {
6269 return mask_type::unsafe_from_native(wasm_u16x8_lt(a.value_, b.value_));
6270 } else if constexpr (std::same_as<T, std::int32_t>) {
6271 return mask_type::unsafe_from_native(wasm_i32x4_lt(a.value_, b.value_));
6272 } else if constexpr (std::same_as<T, std::uint32_t>) {
6273 return mask_type::unsafe_from_native(wasm_u32x4_lt(a.value_, b.value_));
6274 } else if constexpr (std::same_as<T, std::int64_t>) {
6275 return mask_type::unsafe_from_native(wasm_i64x2_lt(a.value_, b.value_));
6276 } else if constexpr (std::same_as<T, std::uint64_t>) {
6277 return mask_type::unsafe_from_native(
6278 wasm_i64x2_lt(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6279 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6280 }
6281 }
6282 }
6283
6285 friend native_inline constexpr mask_type operator<=(simd a, simd b) noexcept {
6286 if consteval {
6287 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6288 std::uint64_t bits = 0;
6289 for (std::size_t i = 0; i < N; ++i) {
6290 bits |= std::uint64_t(x[i] <= y[i]) << i;
6291 }
6293 } else {
6294 if constexpr (std::same_as<T, float>) {
6295 return mask_type::unsafe_from_native(wasm_f32x4_le(a.value_, b.value_));
6296 } else if constexpr (std::same_as<T, double>) {
6297 return mask_type::unsafe_from_native(wasm_f64x2_le(a.value_, b.value_));
6298 } else if constexpr (std::same_as<T, std::int8_t>) {
6299 return mask_type::unsafe_from_native(wasm_i8x16_le(a.value_, b.value_));
6300 } else if constexpr (std::same_as<T, std::uint8_t>) {
6301 return mask_type::unsafe_from_native(wasm_u8x16_le(a.value_, b.value_));
6302 } else if constexpr (std::same_as<T, std::int16_t>) {
6303 return mask_type::unsafe_from_native(wasm_i16x8_le(a.value_, b.value_));
6304 } else if constexpr (std::same_as<T, std::uint16_t>) {
6305 return mask_type::unsafe_from_native(wasm_u16x8_le(a.value_, b.value_));
6306 } else if constexpr (std::same_as<T, std::int32_t>) {
6307 return mask_type::unsafe_from_native(wasm_i32x4_le(a.value_, b.value_));
6308 } else if constexpr (std::same_as<T, std::uint32_t>) {
6309 return mask_type::unsafe_from_native(wasm_u32x4_le(a.value_, b.value_));
6310 } else if constexpr (std::same_as<T, std::int64_t>) {
6311 return mask_type::unsafe_from_native(wasm_i64x2_le(a.value_, b.value_));
6312 } else if constexpr (std::same_as<T, std::uint64_t>) {
6313 return mask_type::unsafe_from_native(
6314 wasm_i64x2_le(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6315 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6316 }
6317 }
6318 }
6319
6321 friend native_inline constexpr mask_type operator>(simd a, simd b) noexcept {
6322 if consteval {
6323 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6324 std::uint64_t bits = 0;
6325 for (std::size_t i = 0; i < N; ++i) {
6326 bits |= std::uint64_t(x[i] > y[i]) << i;
6327 }
6329 } else {
6330 if constexpr (std::same_as<T, float>) {
6331 return mask_type::unsafe_from_native(wasm_f32x4_gt(a.value_, b.value_));
6332 } else if constexpr (std::same_as<T, double>) {
6333 return mask_type::unsafe_from_native(wasm_f64x2_gt(a.value_, b.value_));
6334 } else if constexpr (std::same_as<T, std::int8_t>) {
6335 return mask_type::unsafe_from_native(wasm_i8x16_gt(a.value_, b.value_));
6336 } else if constexpr (std::same_as<T, std::uint8_t>) {
6337 return mask_type::unsafe_from_native(wasm_u8x16_gt(a.value_, b.value_));
6338 } else if constexpr (std::same_as<T, std::int16_t>) {
6339 return mask_type::unsafe_from_native(wasm_i16x8_gt(a.value_, b.value_));
6340 } else if constexpr (std::same_as<T, std::uint16_t>) {
6341 return mask_type::unsafe_from_native(wasm_u16x8_gt(a.value_, b.value_));
6342 } else if constexpr (std::same_as<T, std::int32_t>) {
6343 return mask_type::unsafe_from_native(wasm_i32x4_gt(a.value_, b.value_));
6344 } else if constexpr (std::same_as<T, std::uint32_t>) {
6345 return mask_type::unsafe_from_native(wasm_u32x4_gt(a.value_, b.value_));
6346 } else if constexpr (std::same_as<T, std::int64_t>) {
6347 return mask_type::unsafe_from_native(wasm_i64x2_gt(a.value_, b.value_));
6348 } else if constexpr (std::same_as<T, std::uint64_t>) {
6349 return mask_type::unsafe_from_native(
6350 wasm_i64x2_gt(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6351 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6352 }
6353 }
6354 }
6355
6357 friend native_inline constexpr mask_type operator>=(simd a, simd b) noexcept {
6358 if consteval {
6359 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6360 std::uint64_t bits = 0;
6361 for (std::size_t i = 0; i < N; ++i) {
6362 bits |= std::uint64_t(x[i] >= y[i]) << i;
6363 }
6365 } else {
6366 if constexpr (std::same_as<T, float>) {
6367 return mask_type::unsafe_from_native(wasm_f32x4_ge(a.value_, b.value_));
6368 } else if constexpr (std::same_as<T, double>) {
6369 return mask_type::unsafe_from_native(wasm_f64x2_ge(a.value_, b.value_));
6370 } else if constexpr (std::same_as<T, std::int8_t>) {
6371 return mask_type::unsafe_from_native(wasm_i8x16_ge(a.value_, b.value_));
6372 } else if constexpr (std::same_as<T, std::uint8_t>) {
6373 return mask_type::unsafe_from_native(wasm_u8x16_ge(a.value_, b.value_));
6374 } else if constexpr (std::same_as<T, std::int16_t>) {
6375 return mask_type::unsafe_from_native(wasm_i16x8_ge(a.value_, b.value_));
6376 } else if constexpr (std::same_as<T, std::uint16_t>) {
6377 return mask_type::unsafe_from_native(wasm_u16x8_ge(a.value_, b.value_));
6378 } else if constexpr (std::same_as<T, std::int32_t>) {
6379 return mask_type::unsafe_from_native(wasm_i32x4_ge(a.value_, b.value_));
6380 } else if constexpr (std::same_as<T, std::uint32_t>) {
6381 return mask_type::unsafe_from_native(wasm_u32x4_ge(a.value_, b.value_));
6382 } else if constexpr (std::same_as<T, std::int64_t>) {
6383 return mask_type::unsafe_from_native(wasm_i64x2_ge(a.value_, b.value_));
6384 } else if constexpr (std::same_as<T, std::uint64_t>) {
6385 return mask_type::unsafe_from_native(
6386 wasm_i64x2_ge(wasm_v128_xor(a.value_, wasm_i64x2_splat(INT64_MIN)),
6387 wasm_v128_xor(b.value_, wasm_i64x2_splat(INT64_MIN))));
6388 }
6389 }
6390 }
6391
6393 native_inline constexpr simd left(unsigned count) const noexcept
6394 requires simd_integer_element<T>
6395 {
6396 count %= sizeof(T) * 8;
6397 if consteval {
6398 return detail::wasm_map(
6399 [&](T x) { return std::bit_cast<T>(word_type(std::uint64_t(word_type(x)) << count)); },
6400 *this);
6401 } else {
6402 if constexpr (std::same_as<T, std::int8_t>) {
6403 return from_native(wasm_i8x16_shl(value_, count));
6404 } else if constexpr (std::same_as<T, std::uint8_t>) {
6405 return from_native(wasm_i8x16_shl(value_, count));
6406 } else if constexpr (std::same_as<T, std::int16_t>) {
6407 return from_native(wasm_i16x8_shl(value_, count));
6408 } else if constexpr (std::same_as<T, std::uint16_t>) {
6409 return from_native(wasm_i16x8_shl(value_, count));
6410 } else if constexpr (std::same_as<T, std::int32_t>) {
6411 return from_native(wasm_i32x4_shl(value_, count));
6412 } else if constexpr (std::same_as<T, std::uint32_t>) {
6413 return from_native(wasm_i32x4_shl(value_, count));
6414 } else if constexpr (std::same_as<T, std::int64_t>) {
6415 return from_native(wasm_i64x2_shl(value_, count));
6416 } else if constexpr (std::same_as<T, std::uint64_t>) {
6417 return from_native(wasm_i64x2_shl(value_, count));
6418 }
6419 }
6420 }
6421
6423 template<std::size_t K>
6424 native_inline constexpr simd left() const noexcept
6425 requires simd_integer_element<T>
6426 {
6427 return left(K);
6428 }
6429
6431 friend native_inline constexpr simd operator<<(simd a, unsigned n) noexcept
6432 requires simd_integer_element<T>
6433 {
6434 return a.left(n);
6435 }
6436
6438 template<std::size_t K>
6439 friend native_inline constexpr simd operator<<(simd a, imm_t<K>) noexcept
6440 requires simd_integer_element<T>
6441 {
6442 return a.left(K);
6443 }
6444
6446 native_inline constexpr simd right(unsigned count) const noexcept
6447 requires simd_integer_element<T>
6448 {
6449 count %= sizeof(T) * 8;
6450 if consteval {
6451 return detail::wasm_map([&](T x) { return T(x >> count); }, *this);
6452 } else {
6453 if constexpr (std::same_as<T, std::int8_t>) {
6454 return from_native(wasm_i8x16_shr(value_, count));
6455 } else if constexpr (std::same_as<T, std::uint8_t>) {
6456 return from_native(wasm_u8x16_shr(value_, count));
6457 } else if constexpr (std::same_as<T, std::int16_t>) {
6458 return from_native(wasm_i16x8_shr(value_, count));
6459 } else if constexpr (std::same_as<T, std::uint16_t>) {
6460 return from_native(wasm_u16x8_shr(value_, count));
6461 } else if constexpr (std::same_as<T, std::int32_t>) {
6462 return from_native(wasm_i32x4_shr(value_, count));
6463 } else if constexpr (std::same_as<T, std::uint32_t>) {
6464 return from_native(wasm_u32x4_shr(value_, count));
6465 } else if constexpr (std::same_as<T, std::int64_t>) {
6466 return from_native(wasm_i64x2_shr(value_, count));
6467 } else if constexpr (std::same_as<T, std::uint64_t>) {
6468 return from_native(wasm_u64x2_shr(value_, count));
6469 }
6470 }
6471 }
6472
6474 template<std::size_t K>
6475 native_inline constexpr simd right() const noexcept
6476 requires simd_integer_element<T>
6477 {
6478 return right(K);
6479 }
6480
6482 friend native_inline constexpr simd operator>>(simd a, unsigned n) noexcept
6483 requires simd_integer_element<T>
6484 {
6485 return a.right(n);
6486 }
6487
6489 template<std::size_t K>
6490 friend native_inline constexpr simd operator>>(simd a, imm_t<K>) noexcept
6491 requires simd_integer_element<T>
6492 {
6493 return a.right(K);
6494 }
6495 };
6496
6498 template<detail::wasm_number T, std::size_t N, isa<> A>
6499 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6501 simd<T, N, A> b) noexcept {
6502 if consteval {
6503 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6504 auto bits = m.to_bitset();
6505 for (std::size_t i = 0; i < N; ++i) {
6506 if (!((bits >> i) & 1)) {
6507 x[i] = y[i];
6508 }
6509 }
6510 return simd<T, N, A>::load(x.data());
6511 } else {
6513 wasm_v128_bitselect(a.to_native(), b.to_native(), m.to_native()));
6514 }
6515 }
6516
6518 template<detail::wasm_number T, std::size_t N, isa<> A>
6519 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6520 native_inline constexpr simd<T, N, A> bit_select(simd<detail::wasm_word<T>, N, A> m,
6521 simd<T, N, A> a, simd<T, N, A> b) noexcept {
6522 return simd<T, N, A>::from_bits((m & a.bits()) | (~m & b.bits()));
6523 }
6524
6526 template<class To, class U, std::size_t N, isa<> A>
6527 requires(A.has(wasm_feature::simd128)) && (sizeof(U) * N == 16) && (sizeof(To) == sizeof(U)) &&
6528 simd_integer_element<To>
6530 return simd<To, N, A>::from_native(m.to_native());
6531 }
6532
6534 template<std::size_t I, detail::wasm_number T, std::size_t N, isa<> A>
6535 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16) && (I < N)
6537 return simd<T, N, A>(v.template get<I>());
6538 }
6539
6541 template<std::size_t... I, detail::wasm_number T, std::size_t N, isa<> A>
6542 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16) && (sizeof...(I) == N) &&
6543 ((I < 2 * N) && ...)
6545 if consteval {
6546 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
6547 std::array<T, N> r{(I < N ? x[I] : y[I - N])...};
6548 return simd<T, N, A>::load(r.data());
6549 } else {
6550 using vector_type = T __attribute__((ext_vector_type(N)));
6551 return simd<T, N, A>::from_native(__builtin_bit_cast(
6552 v128_t, __builtin_shufflevector(__builtin_bit_cast(vector_type, a.to_native()),
6553 __builtin_bit_cast(vector_type, b.to_native()), I...)));
6554 }
6555 }
6556
6558 template<isa<> A>
6559 requires(A.has(wasm_feature::simd128))
6560 native_inline constexpr simd<std::uint8_t, 16, A>
6562 if consteval {
6563 auto x = detail::wasm_lanes(v), i = detail::wasm_lanes(indices);
6564 for (auto & n : i) {
6565 n = n < 16 ? x[n] : 0;
6566 }
6567 return simd<std::uint8_t, 16, A>::load(i.data());
6568 } else {
6570 wasm_i8x16_swizzle(v.to_native(), indices.to_native()));
6571 }
6572 }
6573
6575 template<std::floating_point T, std::size_t N, isa<> A>
6576 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6578 if consteval {
6579 return detail::wasm_map(
6580 [](T x) {
6581 using format_type = detail::wasm_format<T>;
6582 using bits_type = typename format_type::bits_type;
6583 auto w = std::bit_cast<bits_type>(x);
6584 return std::bit_cast<T>(detail::constexpr_float::sqrt_bits<format_type>(w));
6585 },
6586 v);
6587 } else {
6588 if constexpr (sizeof(T) == 4) {
6589 return simd<T, N, A>::from_native(wasm_f32x4_sqrt(v.to_native()));
6590 } else {
6591 return simd<T, N, A>::from_native(wasm_f64x2_sqrt(v.to_native()));
6592 }
6593 }
6594 }
6595
6597 template<std::floating_point T, std::size_t N, isa<> A>
6598 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6600 if consteval {
6601 return detail::wasm_map(
6602 [](T x) {
6603 using format_type = detail::wasm_format<T>;
6604 using bits_type = typename format_type::bits_type;
6605 auto w = std::bit_cast<bits_type>(x);
6606 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(
6607 w, detail::constexpr_float::rounding::downward));
6608 },
6609 v);
6610 } else {
6611 if constexpr (sizeof(T) == 4) {
6612 return simd<T, N, A>::from_native(wasm_f32x4_floor(v.to_native()));
6613 } else {
6614 return simd<T, N, A>::from_native(wasm_f64x2_floor(v.to_native()));
6615 }
6616 }
6617 }
6618
6620 template<std::floating_point T, std::size_t N, isa<> A>
6621 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6623 if consteval {
6624 return detail::wasm_map(
6625 [](T x) {
6626 using format_type = detail::wasm_format<T>;
6627 using bits_type = typename format_type::bits_type;
6628 auto w = std::bit_cast<bits_type>(x);
6629 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(
6630 w, detail::constexpr_float::rounding::upward));
6631 },
6632 v);
6633 } else {
6634 if constexpr (sizeof(T) == 4) {
6635 return simd<T, N, A>::from_native(wasm_f32x4_ceil(v.to_native()));
6636 } else {
6637 return simd<T, N, A>::from_native(wasm_f64x2_ceil(v.to_native()));
6638 }
6639 }
6640 }
6641
6643 template<std::floating_point T, std::size_t N, isa<> A>
6644 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6646 if consteval {
6647 return detail::wasm_map(
6648 [](T x) {
6649 using format_type = detail::wasm_format<T>;
6650 using bits_type = typename format_type::bits_type;
6651 auto w = std::bit_cast<bits_type>(x);
6652 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(
6653 w, detail::constexpr_float::rounding::toward_zero));
6654 },
6655 v);
6656 } else {
6657 if constexpr (sizeof(T) == 4) {
6658 return simd<T, N, A>::from_native(wasm_f32x4_trunc(v.to_native()));
6659 } else {
6660 return simd<T, N, A>::from_native(wasm_f64x2_trunc(v.to_native()));
6661 }
6662 }
6663 }
6664
6666 template<std::floating_point T, std::size_t N, isa<> A>
6667 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6669 if consteval {
6670 return detail::wasm_map(
6671 [](T x) {
6672 using format_type = detail::wasm_format<T>;
6673 using bits_type = typename format_type::bits_type;
6674 auto w = std::bit_cast<bits_type>(x);
6675 return std::bit_cast<T>(detail::constexpr_float::round_integral_bits<format_type>(w));
6676 },
6677 v);
6678 } else {
6679 if constexpr (sizeof(T) == 4) {
6680 return simd<T, N, A>::from_native(wasm_f32x4_nearest(v.to_native()));
6681 } else {
6682 return simd<T, N, A>::from_native(wasm_f64x2_nearest(v.to_native()));
6683 }
6684 }
6685 }
6686
6688 template<std::floating_point T, std::size_t N, isa<> A>
6689 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6691 if consteval {
6692 return detail::wasm_map(
6693 [](T x) {
6694 using format_type = detail::wasm_format<T>;
6695 using bits_type = typename format_type::bits_type;
6696 auto w = std::bit_cast<bits_type>(x);
6697 return std::bit_cast<T>(bits_type(w & ~format_type::sign_mask));
6698 },
6699 v);
6700 } else {
6701 if constexpr (sizeof(T) == 4) {
6702 return simd<T, N, A>::from_native(wasm_f32x4_abs(v.to_native()));
6703 } else {
6704 return simd<T, N, A>::from_native(wasm_f64x2_abs(v.to_native()));
6705 }
6706 }
6707 }
6708} // namespace native
6709// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
6710// SIMD128 instruction families, included in the SIMD128 target scope.
6711namespace native {
6712 namespace detail {
6713 template<class T>
6714 constexpr T wasm_saturate(std::int64_t x) noexcept {
6715 if (x < std::int64_t(std::numeric_limits<T>::min())) {
6716 return std::numeric_limits<T>::min();
6717 }
6718 if (x > std::int64_t(std::numeric_limits<T>::max())) {
6719 return std::numeric_limits<T>::max();
6720 }
6721 return T(x);
6722 }
6723
6724 template<class To, class From>
6725 constexpr To wasm_trunc_sat(From x) noexcept {
6726 if (x != x) {
6727 return 0;
6728 }
6729 if (x <= double(std::numeric_limits<To>::min())) {
6730 return std::numeric_limits<To>::min();
6731 }
6732 if (x >= double(std::numeric_limits<To>::max())) {
6733 return std::numeric_limits<To>::max();
6734 }
6735 return To(x);
6736 }
6737 } // namespace detail
6738
6740 template<simd_integer_element T, std::size_t N, isa<> A>
6741 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && sizeof(T) <= 2)
6743 using vector_type = simd<T, N, A>;
6744 if consteval {
6745 return detail::wasm_map(
6746 [](T x, T y) { return detail::wasm_saturate<T>(std::int64_t(x) + std::int64_t(y)); }, a, b);
6747 } else {
6748 if constexpr (std::same_as<T, std::int8_t>) {
6749 return vector_type::from_native(wasm_i8x16_add_sat(a.to_native(), b.to_native()));
6750 } else if constexpr (std::same_as<T, std::uint8_t>) {
6751 return vector_type::from_native(wasm_u8x16_add_sat(a.to_native(), b.to_native()));
6752 } else if constexpr (std::same_as<T, std::int16_t>) {
6753 return vector_type::from_native(wasm_i16x8_add_sat(a.to_native(), b.to_native()));
6754 } else if constexpr (std::same_as<T, std::uint16_t>) {
6755 return vector_type::from_native(wasm_u16x8_add_sat(a.to_native(), b.to_native()));
6756 }
6757 }
6758 }
6759
6761 template<simd_integer_element T, std::size_t N, isa<> A>
6762 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && sizeof(T) <= 2)
6764 using vector_type = simd<T, N, A>;
6765 if consteval {
6766 return detail::wasm_map(
6767 [](T x, T y) { return detail::wasm_saturate<T>(std::int64_t(x) - std::int64_t(y)); }, a, b);
6768 } else {
6769 if constexpr (std::same_as<T, std::int8_t>) {
6770 return vector_type::from_native(wasm_i8x16_sub_sat(a.to_native(), b.to_native()));
6771 } else if constexpr (std::same_as<T, std::uint8_t>) {
6772 return vector_type::from_native(wasm_u8x16_sub_sat(a.to_native(), b.to_native()));
6773 } else if constexpr (std::same_as<T, std::int16_t>) {
6774 return vector_type::from_native(wasm_i16x8_sub_sat(a.to_native(), b.to_native()));
6775 } else if constexpr (std::same_as<T, std::uint16_t>) {
6776 return vector_type::from_native(wasm_u16x8_sub_sat(a.to_native(), b.to_native()));
6777 }
6778 }
6779 }
6780
6782 template<simd_integer_element T, std::size_t N, isa<> A>
6783 requires(A.has(wasm_feature::simd128)) &&
6784 (std::is_unsigned_v<T> && sizeof(T) * N == 16 && sizeof(T) <= 2)
6786 if consteval {
6787 return detail::wasm_map([](T x, T y) { return T((unsigned(x) + y + 1) / 2); }, a, b);
6788 } else {
6789 if constexpr (sizeof(T) == 1) {
6790 return simd<T, N, A>::from_native(wasm_u8x16_avgr(a.to_native(), b.to_native()));
6791 } else {
6792 return simd<T, N, A>::from_native(wasm_u16x8_avgr(a.to_native(), b.to_native()));
6793 }
6794 }
6795 }
6796
6798 template<simd_integer_element T, std::size_t N, isa<> A>
6799 requires(A.has(wasm_feature::simd128)) && (std::is_signed_v<T> && sizeof(T) * N == 16)
6801 using vector_type = simd<T, N, A>;
6802 using unsigned_type = std::make_unsigned_t<T>;
6803 if consteval {
6804 return detail::wasm_map(
6805 [](T x) { return x < 0 ? std::bit_cast<T>(unsigned_type(0 - unsigned_type(x))) : x; }, a);
6806 } else {
6807 if constexpr (sizeof(T) == 1) {
6808 return vector_type::from_native(wasm_i8x16_abs(a.to_native()));
6809 } else if constexpr (sizeof(T) == 2) {
6810 return vector_type::from_native(wasm_i16x8_abs(a.to_native()));
6811 } else if constexpr (sizeof(T) == 4) {
6812 return vector_type::from_native(wasm_i32x4_abs(a.to_native()));
6813 } else if constexpr (sizeof(T) == 8) {
6814 return vector_type::from_native(wasm_i64x2_abs(a.to_native()));
6815 }
6816 }
6817 }
6818
6820 template<detail::wasm_number T, std::size_t N, isa<> A>
6821 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6823 using vector_type = simd<T, N, A>;
6824 if consteval {
6825 return detail::wasm_map(
6826 [](T x, T y) {
6827 if constexpr (std::is_floating_point_v<T>) {
6828 using format_type = detail::wasm_format<T>;
6829 using bits_type = typename format_type::bits_type;
6830 auto xx = std::bit_cast<bits_type>(x), yy = std::bit_cast<bits_type>(y);
6831 if (detail::constexpr_float::is_nan<format_type>(xx) ||
6832 detail::constexpr_float::is_nan<format_type>(yy)) {
6833 return std::bit_cast<T>(detail::constexpr_float::default_nan<format_type>({}));
6834 }
6835 if (detail::constexpr_float::is_zero<format_type>(xx) &&
6836 detail::constexpr_float::is_zero<format_type>(yy)) {
6837 return std::bit_cast<T>(bits_type(xx | yy));
6838 }
6839 }
6840 return y < x ? y : x;
6841 },
6842 a, b);
6843 } else {
6844 if constexpr (std::same_as<T, float>) {
6845 return vector_type::from_native(wasm_f32x4_min(a.to_native(), b.to_native()));
6846 } else if constexpr (std::same_as<T, double>) {
6847 return vector_type::from_native(wasm_f64x2_min(a.to_native(), b.to_native()));
6848 } else if constexpr (std::same_as<T, std::int8_t>) {
6849 return vector_type::from_native(wasm_i8x16_min(a.to_native(), b.to_native()));
6850 } else if constexpr (std::same_as<T, std::uint8_t>) {
6851 return vector_type::from_native(wasm_u8x16_min(a.to_native(), b.to_native()));
6852 } else if constexpr (std::same_as<T, std::int16_t>) {
6853 return vector_type::from_native(wasm_i16x8_min(a.to_native(), b.to_native()));
6854 } else if constexpr (std::same_as<T, std::uint16_t>) {
6855 return vector_type::from_native(wasm_u16x8_min(a.to_native(), b.to_native()));
6856 } else if constexpr (std::same_as<T, std::int32_t>) {
6857 return vector_type::from_native(wasm_i32x4_min(a.to_native(), b.to_native()));
6858 } else if constexpr (std::same_as<T, std::uint32_t>) {
6859 return vector_type::from_native(wasm_u32x4_min(a.to_native(), b.to_native()));
6860 } else {
6861 return select(a < b, a, b);
6862 }
6863 }
6864 }
6865
6867 template<detail::wasm_number T, std::size_t N, isa<> A>
6868 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
6870 using vector_type = simd<T, N, A>;
6871 if consteval {
6872 return detail::wasm_map(
6873 [](T x, T y) {
6874 if constexpr (std::is_floating_point_v<T>) {
6875 using format_type = detail::wasm_format<T>;
6876 using bits_type = typename format_type::bits_type;
6877 auto xx = std::bit_cast<bits_type>(x), yy = std::bit_cast<bits_type>(y);
6878 if (detail::constexpr_float::is_nan<format_type>(xx) ||
6879 detail::constexpr_float::is_nan<format_type>(yy)) {
6880 return std::bit_cast<T>(detail::constexpr_float::default_nan<format_type>({}));
6881 }
6882 if (detail::constexpr_float::is_zero<format_type>(xx) &&
6883 detail::constexpr_float::is_zero<format_type>(yy)) {
6884 return std::bit_cast<T>(bits_type(xx & yy));
6885 }
6886 }
6887 return y > x ? y : x;
6888 },
6889 a, b);
6890 } else {
6891 if constexpr (std::same_as<T, float>) {
6892 return vector_type::from_native(wasm_f32x4_max(a.to_native(), b.to_native()));
6893 } else if constexpr (std::same_as<T, double>) {
6894 return vector_type::from_native(wasm_f64x2_max(a.to_native(), b.to_native()));
6895 } else if constexpr (std::same_as<T, std::int8_t>) {
6896 return vector_type::from_native(wasm_i8x16_max(a.to_native(), b.to_native()));
6897 } else if constexpr (std::same_as<T, std::uint8_t>) {
6898 return vector_type::from_native(wasm_u8x16_max(a.to_native(), b.to_native()));
6899 } else if constexpr (std::same_as<T, std::int16_t>) {
6900 return vector_type::from_native(wasm_i16x8_max(a.to_native(), b.to_native()));
6901 } else if constexpr (std::same_as<T, std::uint16_t>) {
6902 return vector_type::from_native(wasm_u16x8_max(a.to_native(), b.to_native()));
6903 } else if constexpr (std::same_as<T, std::int32_t>) {
6904 return vector_type::from_native(wasm_i32x4_max(a.to_native(), b.to_native()));
6905 } else if constexpr (std::same_as<T, std::uint32_t>) {
6906 return vector_type::from_native(wasm_u32x4_max(a.to_native(), b.to_native()));
6907 } else {
6908 return select(a > b, a, b);
6909 }
6910 }
6911 }
6912
6914 template<detail::wasm_number T, std::size_t N, isa<> A>
6915 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && std::is_floating_point_v<T>)
6917 using vector_type = simd<T, N, A>;
6918 if consteval {
6919 return detail::wasm_map([](T x, T y) { return y < x ? y : x; }, a, b);
6920 } else {
6921 if constexpr (std::same_as<T, float>) {
6922 return vector_type::from_native(wasm_f32x4_pmin(a.to_native(), b.to_native()));
6923 } else if constexpr (std::same_as<T, double>) {
6924 return vector_type::from_native(wasm_f64x2_pmin(a.to_native(), b.to_native()));
6925 }
6926 }
6927 }
6928
6930 template<detail::wasm_number T, std::size_t N, isa<> A>
6931 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && std::is_floating_point_v<T>)
6933 using vector_type = simd<T, N, A>;
6934 if consteval {
6935 return detail::wasm_map([](T x, T y) { return y > x ? y : x; }, a, b);
6936 } else {
6937 if constexpr (std::same_as<T, float>) {
6938 return vector_type::from_native(wasm_f32x4_pmax(a.to_native(), b.to_native()));
6939 } else if constexpr (std::same_as<T, double>) {
6940 return vector_type::from_native(wasm_f64x2_pmax(a.to_native(), b.to_native()));
6941 }
6942 }
6943 }
6944
6946 template<simd_integer_element T, std::size_t N, isa<> A>
6947 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && sizeof(T) <= 4)
6948 native_inline constexpr auto extend_low(simd<T, N, A> a) noexcept {
6949 using wide_unsigned_type =
6950 std::conditional_t<sizeof(T) == 1, std::uint16_t,
6951 std::conditional_t<sizeof(T) == 2, std::uint32_t, std::uint64_t>>;
6952 using wide_type =
6953 std::conditional_t<std::is_signed_v<T>, std::make_signed_t<wide_unsigned_type>,
6954 wide_unsigned_type>;
6955 using vector_type = simd<wide_type, N / 2, A>;
6956 if consteval {
6957 auto x = detail::wasm_lanes(a);
6958 std::array<wide_type, N / 2> r{};
6959 for (std::size_t i = 0; i < N / 2; ++i) {
6960 r[i] = x[i];
6961 }
6962 return vector_type::load(r.data());
6963 } else {
6964 if constexpr (std::same_as<T, std::int8_t>) {
6965 return vector_type::from_native(wasm_i16x8_extend_low_i8x16(a.to_native()));
6966 } else if constexpr (std::same_as<T, std::uint8_t>) {
6967 return vector_type::from_native(wasm_u16x8_extend_low_u8x16(a.to_native()));
6968 } else if constexpr (std::same_as<T, std::int16_t>) {
6969 return vector_type::from_native(wasm_i32x4_extend_low_i16x8(a.to_native()));
6970 } else if constexpr (std::same_as<T, std::uint16_t>) {
6971 return vector_type::from_native(wasm_u32x4_extend_low_u16x8(a.to_native()));
6972 } else if constexpr (std::same_as<T, std::int32_t>) {
6973 return vector_type::from_native(wasm_i64x2_extend_low_i32x4(a.to_native()));
6974 } else if constexpr (std::same_as<T, std::uint32_t>) {
6975 return vector_type::from_native(wasm_u64x2_extend_low_u32x4(a.to_native()));
6976 }
6977 }
6978 }
6979
6981 template<simd_integer_element T, std::size_t N, isa<> A>
6982 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && sizeof(T) <= 4)
6984 return extend_low(a) * extend_low(b);
6985 }
6986
6988 template<simd_integer_element T, std::size_t N, isa<> A>
6989 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && sizeof(T) <= 4)
6990 native_inline constexpr auto extend_high(simd<T, N, A> a) noexcept {
6991 using wide_unsigned_type =
6992 std::conditional_t<sizeof(T) == 1, std::uint16_t,
6993 std::conditional_t<sizeof(T) == 2, std::uint32_t, std::uint64_t>>;
6994 using wide_type =
6995 std::conditional_t<std::is_signed_v<T>, std::make_signed_t<wide_unsigned_type>,
6996 wide_unsigned_type>;
6997 using vector_type = simd<wide_type, N / 2, A>;
6998 if consteval {
6999 auto x = detail::wasm_lanes(a);
7000 std::array<wide_type, N / 2> r{};
7001 for (std::size_t i = 0; i < N / 2; ++i) {
7002 r[i] = x[i + N / 2];
7003 }
7004 return vector_type::load(r.data());
7005 } else {
7006 if constexpr (std::same_as<T, std::int8_t>) {
7007 return vector_type::from_native(wasm_i16x8_extend_high_i8x16(a.to_native()));
7008 } else if constexpr (std::same_as<T, std::uint8_t>) {
7009 return vector_type::from_native(wasm_u16x8_extend_high_u8x16(a.to_native()));
7010 } else if constexpr (std::same_as<T, std::int16_t>) {
7011 return vector_type::from_native(wasm_i32x4_extend_high_i16x8(a.to_native()));
7012 } else if constexpr (std::same_as<T, std::uint16_t>) {
7013 return vector_type::from_native(wasm_u32x4_extend_high_u16x8(a.to_native()));
7014 } else if constexpr (std::same_as<T, std::int32_t>) {
7015 return vector_type::from_native(wasm_i64x2_extend_high_i32x4(a.to_native()));
7016 } else if constexpr (std::same_as<T, std::uint32_t>) {
7017 return vector_type::from_native(wasm_u64x2_extend_high_u32x4(a.to_native()));
7018 }
7019 }
7020 }
7021
7023 template<simd_integer_element T, std::size_t N, isa<> A>
7024 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && sizeof(T) <= 4)
7026 return extend_high(a) * extend_high(b);
7027 }
7028
7030 template<simd_integer_element To, simd_integer_element From, std::size_t N, isa<> A>
7031 requires(A.has(wasm_feature::simd128)) &&
7032 (sizeof(From) * N == 16 && sizeof(From) == 2 * sizeof(To) && sizeof(To) <= 2 &&
7033 std::is_signed_v<From>)
7035 simd<From, N, A> b) noexcept {
7036 using vector_type = simd<To, N * 2, A>;
7037 if consteval {
7038 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
7039 std::array<To, N * 2> r{};
7040 for (std::size_t i = 0; i < N; ++i) {
7041 r[i] = detail::wasm_saturate<To>(x[i]);
7042 r[i + N] = detail::wasm_saturate<To>(y[i]);
7043 }
7044 return vector_type::load(r.data());
7045 } else {
7046 if constexpr (std::same_as<To, std::int8_t>) {
7047 return vector_type::from_native(wasm_i8x16_narrow_i16x8(a.to_native(), b.to_native()));
7048 } else if constexpr (std::same_as<To, std::uint8_t>) {
7049 return vector_type::from_native(wasm_u8x16_narrow_i16x8(a.to_native(), b.to_native()));
7050 } else if constexpr (std::same_as<To, std::int16_t>) {
7051 return vector_type::from_native(wasm_i16x8_narrow_i32x4(a.to_native(), b.to_native()));
7052 } else if constexpr (std::same_as<To, std::uint16_t>) {
7053 return vector_type::from_native(wasm_u16x8_narrow_i32x4(a.to_native(), b.to_native()));
7054 }
7055 }
7056 }
7057
7059 template<isa<> A>
7060 requires(A.has(wasm_feature::simd128))
7061 native_inline constexpr simd<std::int16_t, 8, A>
7063 if consteval {
7064 return detail::wasm_map(
7065 [](std::int16_t x, std::int16_t y) {
7066 return detail::wasm_saturate<std::int16_t>((std::int64_t(x) * y + 16384) >> 15);
7067 },
7068 a, b);
7069 } else {
7071 wasm_i16x8_q15mulr_sat(a.to_native(), b.to_native()));
7072 }
7073 }
7074
7076 template<isa<> A>
7077 requires(A.has(wasm_feature::simd128))
7079 simd<std::int16_t, 8, A> b) noexcept {
7080 using vector_type = simd<std::int32_t, 4, A>;
7081 if consteval {
7082 auto x = detail::wasm_lanes(a), y = detail::wasm_lanes(b);
7083 std::array<std::int32_t, 4> r{};
7084 for (std::size_t i = 0; i < 4; ++i) {
7085 r[i] = std::bit_cast<std::int32_t>(std::uint32_t(
7086 std::int64_t(x[2 * i]) * y[2 * i] + std::int64_t(x[2 * i + 1]) * y[2 * i + 1]));
7087 }
7088 return vector_type::load(r.data());
7089 } else {
7090 return vector_type::from_native(wasm_i32x4_dot_i16x8(a.to_native(), b.to_native()));
7091 }
7092 }
7093
7096 template<simd_integer_element To, std::floating_point From, std::size_t N, isa<> A>
7097 requires(A.has(wasm_feature::simd128)) && (sizeof(From) * N == 16 && sizeof(To) == 4)
7099 using vector_type = simd<To, 4, A>;
7100 if consteval {
7101 auto x = detail::wasm_lanes(a);
7102 std::array<To, 4> r{};
7103 for (std::size_t i = 0; i < N; ++i) {
7104 r[i] = detail::wasm_trunc_sat<To>(x[i]);
7105 }
7106 return vector_type::load(r.data());
7107 } else {
7108 if constexpr (sizeof(From) == 4 && std::is_signed_v<To>) {
7109 return vector_type::from_native(wasm_i32x4_trunc_sat_f32x4(a.to_native()));
7110 } else if constexpr (sizeof(From) == 4) {
7111 return vector_type::from_native(wasm_u32x4_trunc_sat_f32x4(a.to_native()));
7112 } else if constexpr (std::is_signed_v<To>) {
7113 return vector_type::from_native(wasm_i32x4_trunc_sat_f64x2_zero(a.to_native()));
7114 } else {
7115 return vector_type::from_native(wasm_u32x4_trunc_sat_f64x2_zero(a.to_native()));
7116 }
7117 }
7118 }
7119
7122 template<std::floating_point To, detail::wasm_number From, std::size_t N, isa<> A>
7123 requires(A.has(wasm_feature::simd128)) &&
7124 (sizeof(From) * N == 16 && (sizeof(From) == 4 || std::is_floating_point_v<From>) &&
7125 !std::same_as<To, From>)
7126 native_inline constexpr simd<To, 16 / sizeof(To), A> convert(simd<From, N, A> a) noexcept {
7127 using vector_type = simd<To, 16 / sizeof(To), A>;
7128 if consteval {
7129 auto x = detail::wasm_lanes(a);
7130 std::array<To, vector_type::lanes> r{};
7131 for (std::size_t i = 0; i < std::min(N, vector_type::lanes); ++i) {
7132 if constexpr (std::is_floating_point_v<From>) {
7133 using format_type = detail::wasm_format<From>;
7134 using result_format = detail::wasm_format<To>;
7135 r[i] =
7136 std::bit_cast<To>(detail::constexpr_float::convert_bits<result_format, format_type>(
7137 std::bit_cast<typename format_type::bits_type>(x[i])));
7138 } else {
7139 r[i] = To(x[i]);
7140 }
7141 }
7142 return vector_type::load(r.data());
7143 } else {
7144 if constexpr (std::same_as<From, float>) {
7145 return vector_type::from_native(wasm_f64x2_promote_low_f32x4(a.to_native()));
7146 } else if constexpr (std::same_as<From, double>) {
7147 return vector_type::from_native(wasm_f32x4_demote_f64x2_zero(a.to_native()));
7148 } else if constexpr (sizeof(To) == 4 && std::is_signed_v<From>) {
7149 return vector_type::from_native(wasm_f32x4_convert_i32x4(a.to_native()));
7150 } else if constexpr (sizeof(To) == 4) {
7151 return vector_type::from_native(wasm_f32x4_convert_u32x4(a.to_native()));
7152 } else if constexpr (std::is_signed_v<From>) {
7153 return vector_type::from_native(wasm_f64x2_convert_low_i32x4(a.to_native()));
7154 } else {
7155 return vector_type::from_native(wasm_f64x2_convert_low_u32x4(a.to_native()));
7156 }
7157 }
7158 }
7159
7161 template<class V>
7162 requires detail::wasm_number<typename V::value_type> &&
7163 (V::architecture.has(wasm_feature::simd128)) &&
7164 (V::lanes * sizeof(typename V::value_type) == 16)
7165 native_inline constexpr V load_splat(typename V::value_type const * p) noexcept {
7166 return V(*p);
7167 }
7168
7170 template<std::size_t I, detail::wasm_number T, std::size_t N, isa<> A>
7171 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && I < N)
7172 native_inline constexpr simd<T, N, A> load_lane(T const * p, simd<T, N, A> a) noexcept {
7173 return a.template replace<I>(*p);
7174 }
7175
7177 template<std::size_t I, detail::wasm_number T, std::size_t N, isa<> A>
7178 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16 && I < N)
7179 native_inline constexpr void store_lane(T * p, simd<T, N, A> a) noexcept {
7180 *p = a.template get<I>();
7181 }
7182
7184 template<class V>
7185 requires(V::architecture.has(wasm_feature::simd128)) &&
7186 (V::lanes * sizeof(typename V::value_type) == 16 && sizeof(typename V::value_type) >= 4)
7187 native_inline constexpr V load_zero(typename V::value_type const * p) noexcept {
7188 if consteval {
7189 std::array<typename V::value_type, V::lanes> a{};
7190 a[0] = *p;
7191 return V::load(a.data());
7192 } else {
7193 if constexpr (sizeof(typename V::value_type) == 4) {
7194 return V::from_native(wasm_v128_load32_zero(p));
7195 } else {
7196 return V::from_native(wasm_v128_load64_zero(p));
7197 }
7198 }
7199 }
7200
7202 template<simd_integer_element To, simd_integer_element From, isa<> A>
7203 requires(A.has(wasm_feature::simd128)) &&
7204 (sizeof(To) == 2 * sizeof(From) && std::is_signed_v<To> == std::is_signed_v<From>)
7205 native_inline constexpr simd<To, 16 / sizeof(To), A> load_widened(From const * p) noexcept {
7206 using vector_type = simd<To, 16 / sizeof(To), A>;
7207 if consteval {
7208 std::array<To, vector_type::lanes> a{};
7209 for (std::size_t i = 0; i < vector_type::lanes; ++i) {
7210 a[i] = p[i];
7211 }
7212 return vector_type::load(a.data());
7213 } else {
7214 if constexpr (std::same_as<To, std::int16_t>) {
7215 return vector_type::from_native(wasm_i16x8_load8x8(p));
7216 } else if constexpr (std::same_as<To, std::uint16_t>) {
7217 return vector_type::from_native(wasm_u16x8_load8x8(p));
7218 } else if constexpr (std::same_as<To, std::int32_t>) {
7219 return vector_type::from_native(wasm_i32x4_load16x4(p));
7220 } else if constexpr (std::same_as<To, std::uint32_t>) {
7221 return vector_type::from_native(wasm_u32x4_load16x4(p));
7222 } else if constexpr (std::same_as<To, std::int64_t>) {
7223 return vector_type::from_native(wasm_i64x2_load32x2(p));
7224 } else if constexpr (std::same_as<To, std::uint64_t>) {
7225 return vector_type::from_native(wasm_u64x2_load32x2(p));
7226 }
7227 }
7228 }
7229
7231 template<simd_integer_element T, std::size_t N, isa<> A>
7232 requires(A.has(wasm_feature::simd128)) &&
7233 (std::is_signed_v<T> && sizeof(T) * N == 16 && sizeof(T) <= 2)
7235 using wide_type = std::conditional_t<sizeof(T) == 1, std::int16_t, std::int32_t>;
7236 using vector_type = simd<wide_type, N / 2, A>;
7237 if consteval {
7238 auto x = detail::wasm_lanes(a);
7239 std::array<wide_type, N / 2> r{};
7240 for (std::size_t i = 0; i < N / 2; ++i) {
7241 r[i] = wide_type(x[2 * i]) + x[2 * i + 1];
7242 }
7243 return vector_type::load(r.data());
7244 } else {
7245 if constexpr (sizeof(T) == 1) {
7246 return vector_type::from_native(wasm_i16x8_extadd_pairwise_i8x16(a.to_native()));
7247 } else {
7248 return vector_type::from_native(wasm_i32x4_extadd_pairwise_i16x8(a.to_native()));
7249 }
7250 }
7251 }
7252
7254 template<simd_integer_element T, std::size_t N, isa<> A>
7255 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
7256 native_inline constexpr std::uint32_t bitmask(simd<T, N, A> a) noexcept {
7257 if consteval {
7258 auto x = detail::wasm_lanes(a);
7259 std::uint32_t r = 0;
7260 for (std::size_t i = 0; i < N; ++i) {
7261 r |= std::uint32_t(std::make_unsigned_t<T>(x[i]) >> (sizeof(T) * 8 - 1)) << i;
7262 }
7263 return r;
7264 } else {
7265 if constexpr (sizeof(T) == 1) {
7266 return wasm_i8x16_bitmask(a.to_native());
7267 } else if constexpr (sizeof(T) == 2) {
7268 return wasm_i16x8_bitmask(a.to_native());
7269 } else if constexpr (sizeof(T) == 4) {
7270 return wasm_i32x4_bitmask(a.to_native());
7271 } else {
7272 return wasm_i64x2_bitmask(a.to_native());
7273 }
7274 }
7275 }
7276
7278 template<simd_integer_element T, std::size_t N, isa<> A>
7279 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
7280 native_inline constexpr bool any(simd<T, N, A> a) noexcept {
7281 if consteval {
7282 for (auto x : detail::wasm_lanes(a)) {
7283 if (x) {
7284 return true;
7285 }
7286 }
7287 return false;
7288 } else {
7289 return wasm_v128_any_true(a.to_native());
7290 }
7291 }
7292
7294 template<simd_integer_element T, std::size_t N, isa<> A>
7295 requires(A.has(wasm_feature::simd128)) && (sizeof(T) * N == 16)
7296 native_inline constexpr bool all(simd<T, N, A> a) noexcept {
7297 if consteval {
7298 for (auto x : detail::wasm_lanes(a)) {
7299 if (!x) {
7300 return false;
7301 }
7302 }
7303 return true;
7304 } else {
7305 if constexpr (sizeof(T) == 1) {
7306 return wasm_i8x16_all_true(a.to_native());
7307 } else if constexpr (sizeof(T) == 2) {
7308 return wasm_i16x8_all_true(a.to_native());
7309 } else if constexpr (sizeof(T) == 4) {
7310 return wasm_i32x4_all_true(a.to_native());
7311 } else {
7312 return wasm_i64x2_all_true(a.to_native());
7313 }
7314 }
7315 }
7316
7317} // namespace native
7318#endif
#define native_diagnose_if(condition, message)
Reject a call when Clang can prove that its arguments violate a precondition.
Definition attributes.h:50
#define native_reinitializes
[[clang::reinitializes]]
Definition attributes.h:443
#define native_artificial
[[artificial]].
Definition attributes.h:245
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_noescape
portable __attribute__((noescape))
Definition attributes.h:175
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
constexpr auto mask_bits(simd< M, N, Arch > m) noexcept
constexpr predicate< N, Arch > to_predicate(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > select(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > to_vector_mask(predicate< N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_mul(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_mul_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< U, N, Arch > mask_cast(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > bit_select(simd< T, N, Arch > bits, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
#define native_const
[[const]] is not const
Definition attributes.h:108
#define native_pure
[[pure]]
Definition attributes.h:126
#define native_target(x)
this indicates a required feature set for the current multiversioned function.
Definition attributes.h:476
std::string_view const type
typename mask_traits< std::remove_cvref_t< T > >::type mask
Definition mask_traits.h:22
constexpr simd< float, N, Arch > masked_scaleb_zero(M mask, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > scaleb(simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > ceil(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > masked_scaleb(M mask, simd< float, N, Arch > prior, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< To, N, Arch > convert(simd< float, N, Arch > x) noexcept
constexpr auto abs(simd< float, N, Arch > a) noexcept
constexpr simd< float, N, Arch > trunc(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > floor(simd< float, N, Arch > x) noexcept
constexpr void store_simd(U *p, V value, simd_memory< A, Access >={}) noexcept(noexcept(value.template store_memory< A >(p)))
Definition common_body.h:58
simd_access
Definition common.h:280
constexpr std::size_t compress_store(T *destination, std::size_t capacity, typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value) noexcept
constexpr V load_simd(U const *p, simd_memory< A, Access >={}) noexcept(noexcept(V::template load_memory< A >(p)))
Definition common_body.h:40
constexpr V load_simd_partial(U const *p, std::size_t count, typename V::value_type fill={}, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_constructible_v< U, typename V::value_type & > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::load_simd< V >(p)))
Definition common_body.h:96
constexpr compaction_result< simd< T, N, Arch > > compress(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value, T fill=T{}) noexcept
constexpr simd< T, N, Arch > expand(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > packed, simd< T, N, Arch > prior) noexcept
constexpr void store_simd_partial(U *p, V value, std::size_t count, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::store_simd(p, value)))
constexpr imm_t< K > imm
Definition common.h:82
constexpr auto operator!(wide< T, N > const &a)
Definition wide.h:610
constexpr auto operator~(wide< T, N > const &a)
Definition wide.h:603
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr simd< std::int16_t, 8, A > q15mulr_sat(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Multiply signed Q15 lanes, round by adding 2^14, shift by 15 and saturate.
constexpr simd< T, N, A > average_round(simd< T, N, A > a, simd< T, N, A > b) noexcept
Rounded unsigned average: (a+b+1)/2 without intermediate overflow.
void operator-(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto multiply_widened_high(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened upper integer lanes; the complete product fits.
constexpr simd< T, N, A > sqrt(simd< T, N, A > v) noexcept
Compute correctly rounded square roots; negative finite lanes produce NaN.
void operator/(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr bool all(simd< T, N, A > a) noexcept
Test whether every integer lane is nonzero.
void operator>>(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< To, N *2, A > narrow_sat(simd< From, N, A > a, simd< From, N, A > b) noexcept
Saturating concatenate from signed source lanes, including unsigned destinations.
void operator*=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< bool, N, Arch > to_bool(simd< T, N, Arch > value) noexcept
Convert lane truth into Boolean data lanes represented as zero or one, preserving the lane count.
void operator==(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator%(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > pmax(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
constexpr simd< T, N, A > shuffle(simd< T, N, A > a, simd< T, N, A > b) noexcept
Select N lanes from the concatenation of two inputs, using constant indices.
void operator>=(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator>(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator&=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > round_even(simd< T, N, A > v) noexcept
Round floating lanes to nearest integers, choosing even at ties.
constexpr simd< T, N, A > add_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
architecture
Instruction-set families; an ISA value belongs to exactly one family.
Definition config.h:24
void operator/=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto extend_high(simd< T, N, A > a) noexcept
Widen the upper half of integer lanes, preserving signedness.
void operator|(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > pmin(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
void operator+=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator*(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator-=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto multiply_widened_low(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened lower integer lanes; the complete product fits.
constexpr simd< fp16, 32, Arch > fma(simd< fp16, 32, Arch > a, simd< fp16, 32, Arch > b, simd< fp16, 32, Arch > c) noexcept
constexpr simd< To, 4, A > trunc_sat(simd< From, N, A > a) noexcept
void operator<=(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< std::uint8_t, 16, A > swizzle(simd< std::uint8_t, 16, A > v, simd< std::uint8_t, 16, A > indices) noexcept
Look up byte indices 0..15; every other index produces zero.
constexpr simd< T, N, A > max(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
void operator^=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > broadcast(simd< T, N, A > v, imm_t< I >) noexcept
Broadcast one compile-time-selected lane.
constexpr std::uint32_t bitmask(simd< T, N, A > a) noexcept
Gather each integer lane's sign bit into bit i of the scalar result.
constexpr auto pairwise_add_widened(simd< T, N, Arch > value) noexcept
constexpr V load_splat(typename V::value_type const *p) noexcept
Read one scalar and broadcast it; the access is exactly sizeof(T) bytes.
constexpr simd< std::int32_t, 4, A > dot(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Sum adjacent signed halfword products modulo 2^32.
void operator<<(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr auto extend_low(simd< T, N, A > a) noexcept
Widen the lower half of integer lanes, preserving signedness.
void operator&(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator^(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
void operator+(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< To, 16/sizeof(To), A > load_widened(From const *p) noexcept
Read exactly eight bytes of source lanes and widen them with their signedness.
void operator!=(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr void store_lane(T *p, simd< T, N, A > a) noexcept
Write only lane I to one scalar object.
constexpr V load_zero(typename V::value_type const *p) noexcept
Read one 32- or 64-bit lane and zero the other lanes.
void operator<(simd< T, N, A >, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr simd< T, N, A > sub_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
void operator|=(simd< T, N, A > &, simd< U, M, B >)=delete
Reject mixed architectures and mismatched short-vector widths before native conversions can participa...
constexpr bool any(simd< T, N, A > a) noexcept
Test whether at least one integer lane is nonzero.
constexpr simd< T, N, A > load_lane(T const *p, simd< T, N, A > a) noexcept
Read one scalar into lane I, preserving the other lanes.
constexpr simd< T, N, A > min(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
Standard-library adaptations documented here for SIMD value types.
static constexpr mask_lane from_bits(U value) noexcept
Normalize nonzero bits to a true, all-one lane.
Definition common.h:96
static constexpr predicate from_bitset(std::uint64_t bits) noexcept
Construct from the logical lane bitset.
constexpr predicate & operator&=(predicate b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
constexpr predicate & operator^=(predicate b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
friend constexpr predicate operator==(predicate a, predicate b) noexcept
Return a mask whose lanes are true where a == b holds.
constexpr std::uint64_t to_bitset() const noexcept
Pack lane truth into low bits, with lane zero in bit zero.
friend constexpr bool none(predicate a) noexcept
Return true exactly when no logical lane is true.
constexpr predicate & operator|=(predicate b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr predicate operator~(predicate a) noexcept
Invert each lane truth value, preserving the mask representation.
static constexpr predicate from_native(native_type value) noexcept
Import compact mask bits and clear bits above the lane count.
friend constexpr bool all(predicate a) noexcept
Return whether every logical lane is true.
friend constexpr predicate operator!(predicate a) noexcept
Return the lane-wise logical complement, retaining this mask type.
friend constexpr predicate operator&(predicate a, predicate b) noexcept
Bitwise AND of corresponding lane representations.
static constexpr predicate from_bitset(std::uint64_t value) noexcept
Import lane truth from the low logical-lane bits.
constexpr native_type to_native() const noexcept
Return the native storage representation.
constexpr predicate() noexcept=default
Initialize every logical lane to false.
friend constexpr predicate operator^(predicate a, predicate b) noexcept
Bitwise XOR of corresponding lane representations.
friend constexpr predicate select(predicate p, predicate a, predicate b) noexcept
Choose a in true lanes and b in false lanes; both values are already evaluated.
friend constexpr predicate operator!=(predicate a, predicate b) noexcept
Return a mask whose lanes are true where a != b holds.
static constexpr predicate unsafe_from_native(native_type value) noexcept
Import compact bits and clear bits above the lane count, just like from_native.
friend constexpr predicate operator|(predicate a, predicate b) noexcept
Bitwise OR of corresponding lane representations.
friend constexpr bool any(predicate a) noexcept
Return whether at least one logical lane is true.
friend constexpr simd operator>>(simd a, imm_t< K >) noexcept
Shift integer lanes right, extending signed lanes and reducing the count modulo their width.
friend constexpr simd operator<<(simd a, unsigned n) noexcept
Shift integer lanes left, reducing the count modulo the lane width.
constexpr simd replace(T x) const noexcept
Replace one compile-time-selected lane and preserve every other lane.
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
static constexpr simd load_bits(word_type const *p) noexcept
Load lane representations from corresponding unsigned words.
static constexpr simd load(T const *p) noexcept
Read exactly N elements with their natural alignment.
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator|(simd a, simd b) noexcept
Unite representation bits.
constexpr void store_memory(T *p) const noexcept
Store all lanes; Align is the caller-provided pointer alignment.
constexpr simd(std::array< T, N > const &a) noexcept
Copy the array in lane order.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane operation and update this value.
constexpr simd left() const noexcept
Shift lanes left by K modulo the lane width.
constexpr T get() const noexcept
Extract the compile-time-selected lane as its scalar element type.
constexpr simd() noexcept=default
Initialize every lane to zero.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
constexpr native_type to_native() const noexcept
Return the implementation register without changing its bits.
friend constexpr simd operator&(simd a, simd b) noexcept
Intersect representation bits.
constexpr void store_partial(T *p, std::size_t n) const noexcept
Write only the first n lanes; require n <= N. Null is valid when n is zero.
static constexpr simd from_native(native_type x) noexcept
Adopt implementation storage; mask specializations normalize nonzero lanes.
constexpr native_type to_native() const noexcept
Bridge to the implementation register without numerical conversion.
constexpr simd left(unsigned count) const noexcept
WebAssembly shifts reduce the count modulo the lane width.
friend constexpr simd operator>>(simd a, unsigned n) noexcept
Shift integer lanes right, extending signed lanes and reducing the count modulo their width.
static constexpr simd load_partial(T const *p, std::size_t n, T fill={}) noexcept
Read n lanes and fill the rest; require n <= N. Null is valid when n is zero.
friend constexpr simd operator-(simd a) noexcept
Subtract or negate lanes; integer results wrap and floating signs are preserved.
constexpr bits_type bits() const noexcept
Return each half lane as its unchanged unsigned representation.
friend constexpr mask_type operator>(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
friend constexpr simd operator~(simd a) noexcept
Complement every representation bit (or lane truth for masks).
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator<<(simd a, imm_t< K >) noexcept
Shift integer lanes left, reducing the count modulo the lane width.
static constexpr simd from_native(native_type value) noexcept
Adopt register bits unchanged; unused physical bytes are unspecified.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Compare lane inequality and return canonical mask lanes.
static constexpr simd load_memory(T const *p) noexcept
Load all lanes; Align is the caller-provided pointer alignment.
static constexpr simd unsafe_from_native(native_type x) noexcept
Adopt implementation bits; mask callers must supply canonical lanes.
constexpr simd right(unsigned count) const noexcept
WebAssembly shifts reduce the count modulo the lane width.
constexpr void store(T *p) const noexcept
Store exactly 16 unaligned bytes in lane order.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator+(simd a, simd b) noexcept
Add lanes; integer results wrap and floating results round to nearest-even.
friend constexpr simd operator-(simd a, simd b) noexcept
Subtract or negate lanes; integer results wrap and floating signs are preserved.
constexpr void store_bits(word_type *p) const noexcept
Store lane representations as corresponding unsigned words.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane operation and update this value.
constexpr simd right() const noexcept
Shift lanes right by K modulo the lane width.
constexpr simd(U input) noexcept
Broadcast an integer, retaining its low lane-width bits in every lane.
friend constexpr mask_type operator<(simd a, simd b) noexcept
Compare lanes using their signedness; unordered floating lanes are false.
constexpr bits_type bits() const noexcept
Reinterpret each lane as an unsigned word of the same width.
friend constexpr simd operator/(simd a, simd b) noexcept
Divide floating lanes with WebAssembly IEEE semantics.
friend constexpr mask_type operator==(simd a, simd b) noexcept
Compare lane equality and return canonical mask lanes.
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator^(simd a, simd b) noexcept
Exclusive-or representation bits.
static constexpr simd from_bits(bits_type x) noexcept
Adopt unsigned lane representations without numerical conversion.
friend constexpr simd operator*(simd a, simd b) noexcept
Multiply lanes; integer results wrap and floating results round to nearest-even.
friend constexpr simd operator!(simd a) noexcept
Return the lane-wise logical complement, retaining this mask type.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
friend constexpr mask_type operator<(simd a, U b) noexcept
Return a mask whose lanes are true where a < b holds. Ordering follows the signedness of T....
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds.
friend simd operator%(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend simd operator%(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr simd operator^(U a, simd b) noexcept
Bitwise XOR of corresponding lane representations. Scalar operands are reduced to the lane width and ...
constexpr simd & operator^=(U b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this. Scalar operands are reduce...
friend constexpr simd operator+(simd a, U b) noexcept
Add corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width and bro...
friend constexpr mask_type operator<=(U a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. Ordering follows the signedness of T....
constexpr simd & operator+=(U b) noexcept
Apply the corresponding lane-wise add operation in place and return *this. Scalar operands are reduce...
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. Ordering follows the signedness of T.
friend constexpr simd operator+(simd a, simd b) noexcept
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane-wise divide operation in place and return *this.
static constexpr simd loadu(T const *p) noexcept
Read all logical lanes without an extra alignment promise.
friend simd operator<<(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
static constexpr simd unsafe_from_native(native_type x) noexcept
Adopt native bits and clear physical padding. If T is a mask element, every logical lane must already...
friend simd operator/(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
static constexpr simd load_memory(T const *p) noexcept
Load logical lanes, assuming the template alignment in bytes.
friend simd operator%(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator|(simd a, simd b) noexcept
Bitwise OR of corresponding lane representations.
friend simd operator>>(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
friend constexpr simd operator|(simd a, U b) noexcept
Bitwise OR of corresponding lane representations. Scalar operands are reduced to the lane width and b...
friend constexpr simd round_even(simd a) noexcept
Round each logical lane to an integral value, ties to even, independently of the ambient rounding dir...
friend constexpr mask_type operator!=(simd a, U b) noexcept
Return a mask whose lanes are true where a != b holds. Ordering follows the signedness of T....
friend constexpr simd fma(simd a, simd b, simd c) noexcept
Compute a*b+c with one fused rounding per logical lane in the caller's floating-point environment.
static constexpr simd load_partial(T const *p, std::size_t n, T fill=T(0)) noexcept
Read exactly n logical lanes and fill the remainder; require n <= lanes. For n == 0,...
static constexpr simd unsafe_from_float32(native_type x) noexcept
Adopt raw float storage without numerical conversion or normalization.
friend constexpr simd operator*(U a, simd b) noexcept
Multiply corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
friend constexpr simd operator&(U a, simd b) noexcept
Bitwise AND of corresponding lane representations. Scalar operands are reduced to the lane width and ...
constexpr simd left() const noexcept
Shift each lane left by compile-time K, discarding high bits; require K below the lane bit width.
friend simd operator/(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator/(simd a, simd b) noexcept
constexpr std::uint64_t to_bitset() const noexcept
Pack each logical lane truth value into bit i; higher bits are zero.
constexpr native_type to_native() const noexcept
Return the native storage representation under its minimal register ABI.
constexpr simd & operator-=(U b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this. Scalar operands are r...
friend constexpr simd operator+(simd a) noexcept
Return the unchanged vector value.
static constexpr simd load_bits_partial(std::uint32_t const *p, std::size_t n, std::uint32_t fill=0) noexcept
Read n representation words and fill the remaining logical lanes; require n <= lanes....
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
friend constexpr simd operator^(simd a, simd b) noexcept
Bitwise XOR of corresponding lane representations.
friend constexpr mask_type operator<=(simd a, U b) noexcept
Return a mask whose lanes are true where a <= b holds. Ordering follows the signedness of T....
constexpr simd(bool x) noexcept
Broadcast the supplied value to each logical lane. Unused physical lanes are zero.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. Ordering follows the signedness of T.
friend constexpr simd normal_pow2(simd a) noexcept
Construct normal powers of two; require each input to be an integral exponent in [-126,...
static constexpr simd from_bitset(std::uint64_t bits) noexcept
Import lane i from bit i, clearing bits above the logical lane count.
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exactly the logical count of binary32 words without normalizing their representations.
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
constexpr simd()=default
Default-construct the element customization; its initialization contract is retained.
friend constexpr simd operator&(simd a, U b) noexcept
Bitwise AND of corresponding lane representations. Scalar operands are reduced to the lane width and ...
friend simd operator/(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
static constexpr simd from_native(native_type x) noexcept
Adopt native storage without numerical conversion.
constexpr simd(std::array< T, N > const &values) noexcept
Copy one value per logical lane in array order.
constexpr simd & operator&=(U b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this. Scalar operands are reduce...
friend constexpr mask_type operator>(simd a, U b) noexcept
Return a mask whose lanes are true where a > b holds. Ordering follows the signedness of T....
friend simd operator<<(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
friend simd operator%(simd, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
constexpr simd(std::array< T, N > const &data) noexcept
Copy one value per logical lane in array order.
friend constexpr simd operator&(simd a, simd b) noexcept
Bitwise AND of corresponding lane representations.
constexpr simd & operator*=(U b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this. Scalar operands are r...
friend constexpr simd operator+(U a, simd b) noexcept
Add corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width and bro...
constexpr void store_bits(std::uint32_t *p) const noexcept
Store the exact binary32 words for every logical lane.
friend simd operator>>(simd, U)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
friend simd operator%(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr mask_type operator!=(U a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. Ordering follows the signedness of T....
constexpr void store_partial(T *p, std::size_t n) const noexcept
Write exactly n logical lanes; require n <= lanes. For n == 0, p may be null.
friend simd operator>>(simd, simd)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
constexpr simd & operator=(simd const &)=default
Copy the stored value and return *this; no numerical conversion is performed.
constexpr simd left() const noexcept
Shift each lane left by compile-time K, discarding high bits; require K below the lane bit width.
friend constexpr mask_type operator<(simd a, simd b) noexcept
Return a mask whose lanes are true where a < b holds.
constexpr void storeu(T *p) const noexcept
Write all logical lanes without an extra alignment promise.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
friend simd operator<<(simd, simd)=delete
Reject runtime shift counts; use a compile-time imm<K> within the lane width.
friend constexpr simd operator-(simd a) noexcept
Negate each lane modulo 2^(sizeof(T)*8).
static constexpr simd from_float(float x) noexcept
Broadcast the float value without adding an FTZ or other normalization policy.
friend constexpr mask_type operator>(simd a, simd b) noexcept
Return a mask whose lanes are true where a > b holds. Ordering follows the signedness of T.
friend constexpr simd operator~(simd a) noexcept
Complement every bit in every lane.
friend constexpr simd operator~(simd a) noexcept
Complement every bit in every lane.
friend constexpr mask_type operator==(U a, simd b) noexcept
Return a mask whose lanes are true where a == b holds. Ordering follows the signedness of T....
static constexpr simd load(T const *p) noexcept
Read all logical lanes from an unaligned element pointer.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
friend constexpr mask_type operator>(U a, simd b) noexcept
Return a mask whose lanes are true where a > b holds. Ordering follows the signedness of T....
constexpr void store_memory(T *p) const noexcept
Store logical lanes, assuming the template alignment in bytes.
constexpr simd right() const noexcept
Shift each lane right by compile-time K; signed lanes extend their sign. Require K below the lane bit...
friend constexpr simd operator*(simd a, simd b) noexcept
friend constexpr simd operator^(simd a, U b) noexcept
Bitwise XOR of corresponding lane representations. Scalar operands are reduced to the lane width and ...
friend constexpr simd operator*(simd a, simd b) noexcept
Multiply corresponding lanes modulo 2^(sizeof(T)*8).
constexpr void store_bits_partial(std::uint32_t *p, std::size_t n) const noexcept
Write n exact representation words; require n <= lanes. A zero count permits null.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds.
friend constexpr simd operator|(simd a, simd b) noexcept
Bitwise OR of corresponding lane representations.
friend constexpr bool any(simd x) noexcept
Return true when at least one logical lane is true.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. Ordering follows the signedness of T.
constexpr simd() noexcept=default
Initialize the stored lane values to zero.
constexpr simd & operator|=(U b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this. Scalar operands are reduced...
friend constexpr mask_type operator<(U a, simd b) noexcept
Return a mask whose lanes are true where a < b holds. Ordering follows the signedness of T....
constexpr bits_type bits() const noexcept
Return the exact binary32 lane representations in the unsigned vector.
friend constexpr simd operator|(U a, simd b) noexcept
Bitwise OR of corresponding lane representations. Scalar operands are reduced to the lane width and b...
constexpr simd(simd const &)=default
Copy the stored value without arithmetic or normalization.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
friend constexpr mask_type operator==(simd a, U b) noexcept
Return a mask whose lanes are true where a == b holds. Ordering follows the signedness of T....
friend constexpr simd sqrt(simd a) noexcept
Compute the native square root of each logical lane in the caller's floating-point environment.
friend simd operator%(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator-(simd a, U b) noexcept
Subtract corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
static constexpr simd from_bits(std::uint32_t x) noexcept
Reinterpret binary32 words as lane values without normalization; a scalar word is broadcast.
friend constexpr simd operator+(simd a, simd b) noexcept
Add corresponding lanes modulo 2^(sizeof(T)*8).
friend constexpr simd operator<<(simd a, simd counts) noexcept
Shift each unsigned 32-bit lane by its corresponding count, which must be less than 32.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr simd operator-(simd a, simd b) noexcept
Subtract corresponding lanes modulo 2^(sizeof(T)*8).
friend constexpr mask_type operator>=(U a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. Ordering follows the signedness of T....
constexpr simd(T x) noexcept
Broadcast the supplied value to each logical lane. Unused physical lanes are zero.
friend constexpr simd operator-(simd a) noexcept
Negate every logical lane; floating-point lanes change sign.
static constexpr simd from_bits(bits_type x) noexcept
Reinterpret binary32 words as lane values without normalization; a scalar word is broadcast.
friend constexpr mask_type operator>(simd a, simd b) noexcept
Return a mask whose lanes are true where a > b holds.
friend constexpr mask_type operator<(simd a, simd b) noexcept
Return a mask whose lanes are true where a < b holds. Ordering follows the signedness of T.
friend simd operator/(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
friend constexpr simd operator-(U a, simd b) noexcept
Subtract corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
friend constexpr simd select(M mask, simd a, simd b) noexcept
Choose a in true mask lanes and b in false lanes; both values are already evaluated.
constexpr bits_type to_bits() const noexcept
Return the exact binary32 lane representations in the unsigned vector.
static constexpr simd from_native(native_type x) noexcept
Import native lanes, normalizing mask elements; other elements retain their bits. Clear physical padd...
constexpr simd(native_type x) noexcept
Adopt native lane storage and clear physical padding. Mask elements must already be canonical.
friend constexpr simd operator*(simd a, U b) noexcept
Multiply corresponding lanes modulo 2^(sizeof(T)*8). Scalar operands are reduced to the lane width an...
friend constexpr mask_type operator>=(simd a, U b) noexcept
Return a mask whose lanes are true where a >= b holds. Ordering follows the signedness of T....
friend constexpr mask_type operator==(simd a, simd b) noexcept
Return a mask whose lanes are true where a == b holds. Ordering follows the signedness of T.
constexpr storage_type to_storage() const noexcept
Return the corresponding four-lane storage vector without changing logical lane bits.
friend simd operator/(U, simd)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
static constexpr simd from_storage(storage_type x) noexcept
Copy the first N lanes from a four-lane storage value and clear physical padding.
static constexpr simd load_partial(T const *p, std::size_t n, T fill={}) noexcept
Read exactly n logical lanes and fill the remainder; require n <= lanes. For n == 0,...
friend constexpr simd operator^(simd a, simd b) noexcept
Bitwise XOR of corresponding lane representations.
friend constexpr simd operator-(simd a, simd b) noexcept
constexpr simd right() const noexcept
Shift each lane right by compile-time K; signed lanes extend their sign. Require K below the lane bit...
friend constexpr simd operator&(simd a, simd b) noexcept
Bitwise AND of corresponding lane representations.
friend constexpr bool none(simd x) noexcept
Return true exactly when no logical lane is true.
constexpr native_type to_native() const noexcept
Return the native storage representation, including physical padding when present.
friend constexpr bool all(simd x) noexcept
Return true exactly when every logical lane is true.
constexpr void store(T *p) const noexcept
Write all logical lanes to an unaligned element pointer.
friend simd operator/(simd, U)=delete
Integer division and remainder are not provided; do not fall back to native-register conversions.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
friend constexpr bool any(simd a) noexcept
Return whether at least one logical lane is true.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane-wise OR operation in place and return *this.
friend constexpr simd operator|(simd a, simd b) noexcept
Bitwise OR of corresponding lane representations.
friend constexpr simd operator!(simd a) noexcept
Return the lane-wise logical complement, retaining this mask type.
constexpr native_type to_native() const noexcept
Return the native storage representation.
static constexpr simd load_partial(bool const *p, std::size_t count, bool fill=false) noexcept
Read exactly n logical lanes and fill the remainder; require n <= lanes. For n == 0,...
friend constexpr bool none(simd a) noexcept
Return true exactly when no logical lane is true.
constexpr void store_partial(bool *p, std::size_t count) const noexcept
Write exactly n logical lanes; require n <= lanes. For n == 0, p may be null.
friend constexpr simd operator&(simd a, simd b) noexcept
Bitwise AND of corresponding lane representations.
static constexpr simd load_memory(bool const *p) noexcept
Load exactly the logical lanes; the template alignment is a caller promise, never permission to read ...
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane-wise AND operation in place and return *this.
constexpr simd(std::array< bool, N > const &value) noexcept
Read all logical lanes from an unaligned element pointer.
friend constexpr simd select(M m, simd a, simd b) noexcept
Choose a in true lanes and b in false lanes; both values are already evaluated.
static constexpr simd unsafe_from_native(native_type value) noexcept
Adopt storage with the precondition that every Boolean byte is zero or one.
friend constexpr bool all(simd a) noexcept
Return whether every logical lane is true.
friend constexpr simd operator~(simd a) noexcept
Invert each lane truth value, preserving the mask representation.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds.
constexpr simd() noexcept
Initialize every logical lane to false.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane-wise XOR operation in place and return *this.
static constexpr simd from_native(native_type value) noexcept
Import native byte lanes, converting each nonzero byte to Boolean one.
constexpr void store_memory(bool *p) const noexcept
Store exactly the logical lanes; the template alignment is a caller promise, never permission to writ...
friend constexpr mask_type operator==(simd a, simd b) noexcept
Return a mask whose lanes are true where a == b holds.
friend constexpr simd operator^(simd a, simd b) noexcept
Bitwise XOR of corresponding lane representations.
constexpr simd(bool value) noexcept
Broadcast the supplied truth value to every logical lane.
static constexpr simd load(bool const *p) noexcept
Read all logical lanes from an unaligned element pointer.
constexpr void store(bool *p) const noexcept
Write all logical lanes to an unaligned element pointer.
friend constexpr simd normal_pow2(simd n)
Construct normal powers of two; require integral exponents in [-126,127].
static constexpr simd load(float const *p)
Load every logical lane; no extra alignment is required.
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. NaN lanes yield false.
friend constexpr simd operator-(simd a)
Negate every logical lane; floating-point lanes change sign.
static constexpr simd loadu(float const *p)
Synonym for an unaligned full-vector load.
static constexpr simd from_bits(bits_type bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
friend constexpr mask_type operator<(simd a, simd b)
Return a mask whose lanes are true where a < b holds. NaN lanes yield false.
constexpr void store_bits(std::uint32_t *p) const noexcept
Store exact binary32 representations as uint32_t words.
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exact binary32 representations from uint32_t words.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
static constexpr simd from_float(float x) noexcept
Broadcast one binary32 value to every lane.
friend constexpr mask_type operator==(simd a, simd b)
Return a mask whose lanes are true where a == b holds. NaN lanes yield false.
static constexpr simd load_memory(float const *p) noexcept
Load full lanes, assuming Alignment-byte pointer alignment.
friend constexpr simd sqrt(simd a)
Compute the native square root in every lane.
constexpr void storeu(float *p) const
Synonym for an unaligned full-vector store.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. NaN lanes yield false.
constexpr bits_type to_bits() const noexcept
Synonym for bits(): preserve all binary32 representation bits.
friend constexpr simd operator+(simd a, simd b)
Add corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr simd & operator=(simd const &)=default
Copy the stored value and return *this; no numerical conversion is performed.
friend constexpr simd round_even(simd a)
Round to an integral value, ties to even, independent of ambient direction.
friend constexpr simd operator*(simd a, simd b)
Multiply corresponding floating-point lanes using the caller's rounding and denormal environment.
friend constexpr simd select(M m, simd a, simd b)
Choose a where the canonical mask is true, otherwise b; both operands are evaluated.
static constexpr simd from_bits(std::uint32_t bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
friend constexpr simd operator/(simd a, simd b)
Divide corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr void store_bits_partial(std::uint32_t *p, std::size_t n) const noexcept
Store exactly n representation words; require n <= lanes.
constexpr simd(simd const &)=default
Copy the stored value without arithmetic or normalization.
static constexpr simd load_bits_partial(std::uint32_t const *p, std::size_t n, std::uint32_t fill=0) noexcept
Load n words and fill the remaining lanes; require n <= lanes.
constexpr simd(float x)
Broadcast x to all lanes.
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane-wise divide operation in place and return *this.
constexpr bits_type bits() const noexcept
Project exact binary32 lane words into the unsigned vector.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. NaN lanes compare unequal.
static constexpr simd unsafe_from_float32(native_type x) noexcept
Adopt native raw float storage; this raw type adds no normalization.
constexpr void store_memory(float *p) const noexcept
Store full lanes, assuming Alignment-byte pointer alignment.
friend constexpr mask_type operator>(simd a, simd b)
Return a mask whose lanes are true where a > b holds. NaN lanes yield false.
static constexpr simd from_native(native_type x) noexcept
Adopt native register storage without changing its bits.
constexpr native_type to_native() const noexcept
Project native register storage without a numerical conversion.
constexpr simd(std::array< float, 4 > const &values) noexcept
Load one lane from each array element, in array order.
constexpr void store(float *p) const
Store every logical lane; no extra alignment is required.
simd()=default
Default initialization leaves storage unspecified; value initialization with braces zero-initializes ...
friend constexpr simd fma(simd a, simd b, simd c)
Compute a*b+c with one fused rounding per lane.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
constexpr simd(X... x) noexcept((noexcept(static_cast< float >(x)) &&...))
Convert one argument per lane; exceptions follow those named-lvalue conversions.
friend constexpr simd operator-(simd a, simd b)
Subtract corresponding floating-point lanes using the caller's rounding and denormal environment.
friend constexpr simd normal_pow2(simd n)
Construct normal powers of two; require integral exponents in [-126,127].
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Return a mask whose lanes are true where a >= b holds. NaN lanes yield false.
friend constexpr simd operator-(simd a)
Negate every logical lane; floating-point lanes change sign.
simd()=default
Default initialization leaves storage unspecified; value initialization with braces zero-initializes ...
friend constexpr mask_type operator<(simd a, simd b)
Return a mask whose lanes are true where a < b holds. NaN lanes yield false.
static constexpr simd load_memory(float const *p) noexcept
Load full lanes, assuming Alignment-byte pointer alignment.
constexpr void store(float *p) const
Store every logical lane; no extra alignment is required.
static constexpr simd load_bits_partial(std::uint32_t const *p, std::size_t n, std::uint32_t fill=0) noexcept
Load n words and fill the remaining lanes; require n <= lanes.
constexpr simd & operator/=(simd b) noexcept
Apply the corresponding lane-wise divide operation in place and return *this.
constexpr void store_memory(float *p) const noexcept
Store full lanes, assuming Alignment-byte pointer alignment.
friend constexpr mask_type operator==(simd a, simd b)
Return a mask whose lanes are true where a == b holds. NaN lanes yield false.
constexpr native_type to_native() const noexcept
Project native register storage without a numerical conversion.
friend constexpr simd sqrt(simd a)
Compute the native square root in every lane.
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Return a mask whose lanes are true where a <= b holds. NaN lanes yield false.
friend constexpr simd operator+(simd a, simd b)
Add corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr simd(float x)
Broadcast x to all lanes.
static constexpr simd load(float const *p)
Load every logical lane; no extra alignment is required.
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exact binary32 representations from uint32_t words.
constexpr simd & operator-=(simd b) noexcept
Apply the corresponding lane-wise subtract operation in place and return *this.
friend constexpr simd round_even(simd a)
Round to an integral value, ties to even, independent of ambient direction.
static constexpr simd from_float(float x) noexcept
Broadcast one binary32 value to every lane.
friend constexpr simd operator*(simd a, simd b)
Multiply corresponding floating-point lanes using the caller's rounding and denormal environment.
static constexpr simd loadu(float const *p)
Synonym for an unaligned full-vector load.
friend constexpr simd select(M m, simd a, simd b)
Choose a where the canonical mask is true, otherwise b; both operands are evaluated.
friend constexpr simd operator/(simd a, simd b)
Divide corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr simd & operator*=(simd b) noexcept
Apply the corresponding lane-wise multiply operation in place and return *this.
constexpr simd & operator+=(simd b) noexcept
Apply the corresponding lane-wise add operation in place and return *this.
constexpr void storeu(float *p) const
Synonym for an unaligned full-vector store.
static constexpr simd unsafe_from_float32(native_type x) noexcept
Adopt native raw float storage; this raw type adds no normalization.
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Return a mask whose lanes are true where a != b holds. NaN lanes compare unequal.
constexpr simd(simd const &)=default
Copy the stored value without arithmetic or normalization.
constexpr bits_type bits() const noexcept
Project exact binary32 lane words into the unsigned vector.
constexpr void store_bits_partial(std::uint32_t *p, std::size_t n) const noexcept
Store exactly n representation words; require n <= lanes.
constexpr simd(X... x) noexcept((noexcept(static_cast< float >(x)) &&...))
Convert one argument per lane; exceptions follow those named-lvalue conversions.
static constexpr simd from_native(native_type x) noexcept
Adopt native register storage without changing its bits.
constexpr bits_type to_bits() const noexcept
Synonym for bits(): preserve all binary32 representation bits.
friend constexpr mask_type operator>(simd a, simd b)
Return a mask whose lanes are true where a > b holds. NaN lanes yield false.
constexpr simd(std::array< float, 8 > const &values) noexcept
Load one lane from each array element, in array order.
static constexpr simd from_bits(bits_type bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
constexpr simd & operator=(simd const &)=default
Copy the stored value and return *this; no numerical conversion is performed.
constexpr void store_bits(std::uint32_t *p) const noexcept
Store exact binary32 representations as uint32_t words.
static constexpr simd from_bits(std::uint32_t bits) noexcept
Reinterpret unsigned lane words as binary32, without normalization.
friend constexpr simd fma(simd a, simd b, simd c)
Compute a*b+c with one fused rounding per lane.
friend constexpr simd operator-(simd a, simd b)
Subtract corresponding floating-point lanes using the caller's rounding and denormal environment.
constexpr std::uint64_t to_bitset() const noexcept
Gather lane truth values into low scalar bits.
friend constexpr simd operator|(simd a, simd b) noexcept
Unite representation bits.
friend constexpr simd operator!(simd a) noexcept
Complement every mask lane.
friend constexpr bool any(simd v) noexcept
Test whether at least one lane is true.
constexpr simd & operator^=(simd b) noexcept
Apply the corresponding lane operation and update this value.
constexpr simd & operator&=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator!=(simd a, simd b) noexcept
Compare lane inequality and return canonical mask lanes.
friend constexpr bool all(simd v) noexcept
Test whether every lane is true.
static constexpr simd load_memory(value_type const *p) noexcept
Load all lanes; Align is the caller-provided pointer alignment.
constexpr native_type to_native() const noexcept
Return the implementation register without changing its bits.
static constexpr simd from_native(native_type x) noexcept
Adopt implementation storage; mask specializations normalize nonzero lanes.
friend constexpr simd operator&(simd a, simd b) noexcept
Intersect representation bits.
constexpr void store(value_type *p) const noexcept
Write exactly N canonical mask lane objects.
friend constexpr bool none(simd v) noexcept
Test whether no lane is true.
constexpr std::uint64_t bits() const noexcept
Return the compact lane truth bitset.
static constexpr simd unsafe_from_native(native_type x) noexcept
Adopt implementation bits; mask callers must supply canonical lanes.
constexpr simd() noexcept=default
Initialize every lane to zero.
friend constexpr simd operator==(simd a, simd b) noexcept
Compare lane equality and return canonical mask lanes.
constexpr simd & operator|=(simd b) noexcept
Apply the corresponding lane operation and update this value.
friend constexpr simd operator~(simd a) noexcept
Complement every representation bit (or lane truth for masks).
constexpr void store_memory(value_type *p) const noexcept
Store all lanes; Align is the caller-provided pointer alignment.
static constexpr simd load(value_type const *p) noexcept
Read exactly N canonical mask lane objects.
static constexpr simd from_bitset(std::uint64_t bits) noexcept
Expand low scalar bits to canonical mask lanes, ignoring excess bits.
friend constexpr simd select(simd m, simd a, simd b) noexcept
Choose mask lanes from a where m is true, otherwise from b.
friend constexpr simd operator^(simd a, simd b) noexcept
Exclusive-or representation bits.
Omitted architecture arguments use the native.simd provider's baseline.