16 using ::native::simd_traits;
17 using ::native::simd_customization;
18 using ::native::simd_custom_element;
19 using ::native::simd_integer_element;
20 using ::native::imm_t;
23 using ::native::simd_memory;
28 using ::native::compaction_result;
36 using ::native::operator+;
38 using ::native::operator-;
40 using ::native::operator*;
42 using ::native::operator/;
44 using ::native::operator%;
46 using ::native::operator&;
48 using ::native::operator|;
50 using ::native::operator^;
51 using ::native::operator<<;
53 using ::native::operator>>;
55 using ::native::operator!;
57 using ::native::operator~;
59 using ::native::operator==;
61 using ::native::operator!=;
62 using ::native::operator<;
63 using ::native::operator<=;
65 using ::native::operator>;
67 using ::native::operator>=;
69 using ::native::operator+=;
71 using ::native::operator-=;
73 using ::native::operator*=;
75 using ::native::operator/=;
77 using ::native::operator%=;
79 using ::native::operator&=;
81 using ::native::operator|=;
83 using ::native::operator^=;
84 using ::native::operator<<=;
86 using ::native::operator>>=;
90 using ::native::predicate;
91 using ::native::scalar;
93 using ::native::avx512;
94 using ::native::avx512_bf16;
95 using ::native::avx512_fp16;
97 using ::native::neon_fp16;
98 using ::native::neon_bf16;
100 using ::native::arch;
109 using ::native::simd_integer_element;
110 using ::native::imm_t;
113 using ::native::simd_memory;
114 using ::native::mask_lane;
115 using ::native::mask8;
116 using ::native::mask16;
117 using ::native::mask32;
118 using ::native::mask64;
119 using ::native::simd_mask_element;
123 using ::native::fcvtzs;
124 using ::native::fcvtzu;
161#if (NATIVE_HOST_X86 || NATIVE_HOST_NEON || NATIVE_HOST_WASM) && (!defined(NATIVE_PROFILE) || NATIVE_PROFILE != 0)
166#if NATIVE_HOST_WASM && (!defined(NATIVE_PROFILE) || NATIVE_PROFILE != 0)
constexpr auto mask_bits(simd< M, N, Arch > m) noexcept
constexpr predicate< N, Arch > to_predicate(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > select(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > to_vector_mask(predicate< N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_mul(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_mul_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< U, N, Arch > mask_cast(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > bit_select(simd< T, N, Arch > bits, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< float, N, Arch > masked_scaleb_zero(M mask, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > scaleb(simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > ceil(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > masked_scaleb(M mask, simd< float, N, Arch > prior, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< To, N, Arch > convert(simd< float, N, Arch > x) noexcept
constexpr auto abs(simd< float, N, Arch > a) noexcept
constexpr simd< float, N, Arch > trunc(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > floor(simd< float, N, Arch > x) noexcept
constexpr void store_simd(U *p, V value, simd_memory< A, Access >={}) noexcept(noexcept(value.template store_memory< A >(p)))
constexpr std::size_t compress_store(T *destination, std::size_t capacity, typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value) noexcept
constexpr V load_simd(U const *p, simd_memory< A, Access >={}) noexcept(noexcept(V::template load_memory< A >(p)))
constexpr V load_simd_partial(U const *p, std::size_t count, typename V::value_type fill={}, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_constructible_v< U, typename V::value_type & > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::load_simd< V >(p)))
constexpr compaction_result< simd< T, N, Arch > > compress(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value, T fill=T{}) noexcept
constexpr simd< T, N, Arch > expand(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > packed, simd< T, N, Arch > prior) noexcept
constexpr void store_simd_partial(U *p, V value, std::size_t count, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::store_simd(p, value)))
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr simd< std::int16_t, 8, A > q15mulr_sat(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Multiply signed Q15 lanes, round by adding 2^14, shift by 15 and saturate.
constexpr simd< T, N, A > average_round(simd< T, N, A > a, simd< T, N, A > b) noexcept
Rounded unsigned average: (a+b+1)/2 without intermediate overflow.
constexpr auto multiply_widened_high(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened upper integer lanes; the complete product fits.
constexpr simd< T, N, A > sqrt(simd< T, N, A > v) noexcept
Compute correctly rounded square roots; negative finite lanes produce NaN.
constexpr bool all(simd< T, N, A > a) noexcept
Test whether every integer lane is nonzero.
wide(T, U...) -> wide< T, 1+sizeof...(U)>
Deduce the common element type and the number of homogeneous constructor arguments.
constexpr simd< To, N *2, A > narrow_sat(simd< From, N, A > a, simd< From, N, A > b) noexcept
Saturating concatenate from signed source lanes, including unsigned destinations.
constexpr simd< bool, N, Arch > to_bool(simd< T, N, Arch > value) noexcept
Convert lane truth into Boolean data lanes represented as zero or one, preserving the lane count.
constexpr simd< T, N, A > pmax(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
constexpr simd< T, N, A > shuffle(simd< T, N, A > a, simd< T, N, A > b) noexcept
Select N lanes from the concatenation of two inputs, using constant indices.
constexpr simd< T, N, A > round_even(simd< T, N, A > v) noexcept
Round floating lanes to nearest integers, choosing even at ties.
constexpr simd< T, N, A > add_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
constexpr auto extend_high(simd< T, N, A > a) noexcept
Widen the upper half of integer lanes, preserving signedness.
constexpr simd< T, N, A > pmin(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
constexpr auto reinterpret_bits(simd< From, N, Arch > value) noexcept -> simd< To, sizeof(From) *N/sizeof(To), Arch >
constexpr simd< To, 2 *N, Arch > narrow_concat(simd< From, N, Arch > a, simd< From, N, Arch > b) noexcept
constexpr auto multiply_widened_low(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened lower integer lanes; the complete product fits.
constexpr simd< T, N, Arch > popcount(simd< T, N, Arch > value) noexcept
constexpr simd< fp16, 32, Arch > fma(simd< fp16, 32, Arch > a, simd< fp16, 32, Arch > b, simd< fp16, 32, Arch > c) noexcept
constexpr simd< To, 4, A > trunc_sat(simd< From, N, A > a) noexcept
constexpr simd< std::uint8_t, 16, A > swizzle(simd< std::uint8_t, 16, A > v, simd< std::uint8_t, 16, A > indices) noexcept
Look up byte indices 0..15; every other index produces zero.
constexpr simd< T, N, A > max(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
constexpr simd< T, N, A > broadcast(simd< T, N, A > v, imm_t< I >) noexcept
Broadcast one compile-time-selected lane.
constexpr std::uint32_t bitmask(simd< T, N, A > a) noexcept
Gather each integer lane's sign bit into bit i of the scalar result.
constexpr std::array< V, N > flush_to_zero(std::array< V, N > const &x) noexcept
Replace binary32 subnormal lanes with signed zero. Both zero signs, normal values,...
constexpr auto pairwise_add_widened(simd< T, N, Arch > value) noexcept
constexpr V load_splat(typename V::value_type const *p) noexcept
Read one scalar and broadcast it; the access is exactly sizeof(T) bytes.
constexpr simd< std::int32_t, 4, A > dot(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Sum adjacent signed halfword products modulo 2^32.
constexpr auto extend_low(simd< T, N, A > a) noexcept
Widen the lower half of integer lanes, preserving signedness.
constexpr simd< To, 16/sizeof(To), A > load_widened(From const *p) noexcept
Read exactly eight bytes of source lanes and widen them with their signedness.
constexpr void store_lane(T *p, simd< T, N, A > a) noexcept
Write only lane I to one scalar object.
constexpr V load_zero(typename V::value_type const *p) noexcept
Read one 32- or 64-bit lane and zero the other lanes.
constexpr simd< T, N, A > sub_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
constexpr std::uint64_t reduce_add_widened(simd< T, N, Arch > value) noexcept
constexpr bool any(simd< T, N, A > a) noexcept
Test whether at least one integer lane is nonzero.
constexpr simd< T, N, A > load_lane(T const *p, simd< T, N, A > a) noexcept
Read one scalar into lane I, preserving the other lanes.
constexpr simd< T, N, A > min(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
isa(E) -> isa< detail::family_of< E > >
Deduce the family from a feature enum; explicit isa<> always names the host family.