native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
exports.h
1#pragma once
2// Shared export declarations, included only below a named-module declaration.
3// Primitive and array-kernel definitions are in the GMF; wide is imported.
4// SPDX-FileCopyrightText: 2026 Edward Kmett <ekmett@gmail.com>
5// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
6
7export namespace native {
8 using std::int8_t;
9 using std::int16_t;
10 using std::int32_t;
11 using std::int64_t;
12 using std::uint8_t;
13 using std::uint16_t;
14 using std::uint32_t;
15 using std::uint64_t;
16 using ::native::simd_traits;
17 using ::native::simd_customization;
18 using ::native::simd_custom_element;
19 using ::native::simd_integer_element;
20 using ::native::imm_t;
21 using ::native::imm;
22 using ::native::simd_access;
23 using ::native::simd_memory;
24 using ::native::load_simd;
26 using ::native::store_simd;
28 using ::native::compaction_result;
29 using ::native::compress;
30 using ::native::expand;
32}
33
34export namespace native {
36 using ::native::operator+;
38 using ::native::operator-;
40 using ::native::operator*;
42 using ::native::operator/;
44 using ::native::operator%;
46 using ::native::operator&;
48 using ::native::operator|;
50 using ::native::operator^;
51 using ::native::operator<<;
53 using ::native::operator>>;
55 using ::native::operator!;
57 using ::native::operator~;
59 using ::native::operator==;
61 using ::native::operator!=;
62 using ::native::operator<;
63 using ::native::operator<=;
65 using ::native::operator>;
67 using ::native::operator>=;
69 using ::native::operator+=;
71 using ::native::operator-=;
73 using ::native::operator*=;
75 using ::native::operator/=;
77 using ::native::operator%=;
79 using ::native::operator&=;
81 using ::native::operator|=;
83 using ::native::operator^=;
84 using ::native::operator<<=;
86 using ::native::operator>>=;
87
88 using ::native::wide;
89 using ::native::simd;
90 using ::native::predicate;
91 using ::native::scalar;
92 using ::native::avx2;
93 using ::native::avx512;
94 using ::native::avx512_bf16;
95 using ::native::avx512_fp16;
96 using ::native::neon;
97 using ::native::neon_fp16;
98 using ::native::neon_bf16;
99 using ::native::isa;
100 using ::native::arch;
101 using std::int8_t;
102 using std::int16_t;
103 using std::int32_t;
104 using std::int64_t;
105 using std::uint8_t;
106 using std::uint16_t;
107 using std::uint32_t;
108 using std::uint64_t;
109 using ::native::simd_integer_element;
110 using ::native::imm_t;
111 using ::native::imm;
112 using ::native::simd_access;
113 using ::native::simd_memory;
114 using ::native::mask_lane;
115 using ::native::mask8;
116 using ::native::mask16;
117 using ::native::mask32;
118 using ::native::mask64;
119 using ::native::simd_mask_element;
120 using ::native::flush_to_zero;
121 using ::native::convert;
122#if NATIVE_HOST_NEON
123 using ::native::fcvtzs;
124 using ::native::fcvtzu;
125#endif
127 using ::native::popcount;
130 using ::native::mask_bits;
131 using ::native::mask_cast;
132 using ::native::to_bool;
133 using ::native::to_predicate;
136 using ::native::bit_select;
137 using ::native::select;
139 using ::native::masked_add;
142 using ::native::masked_sub;
145 using ::native::masked_mul;
147 using ::native::masked_scaleb;
149 using ::native::scaleb;
150 using ::native::fma;
151 using ::native::broadcast;
152 using ::native::abs;
153 using ::native::sqrt;
154 using ::native::floor;
155 using ::native::ceil;
156 using ::native::trunc;
157}
158
159
160export namespace native {
161#if (NATIVE_HOST_X86 || NATIVE_HOST_NEON || NATIVE_HOST_WASM) && (!defined(NATIVE_PROFILE) || NATIVE_PROFILE != 0)
162 using ::native::narrow_concat;
163#endif
164}
165
166#if NATIVE_HOST_WASM && (!defined(NATIVE_PROFILE) || NATIVE_PROFILE != 0)
167export namespace native {
168 using ::native::shuffle;
169 using ::native::swizzle;
170 using ::native::round_even;
171 using ::native::add_sat;
172 using ::native::sub_sat;
173 using ::native::average_round;
174 using ::native::min;
175 using ::native::max;
176 using ::native::pmin;
177 using ::native::pmax;
178 using ::native::extend_low;
179 using ::native::extend_high;
182 using ::native::bitmask;
183 using ::native::any;
184 using ::native::all;
185 using ::native::narrow_sat;
186 using ::native::q15mulr_sat;
187 using ::native::dot;
188 using ::native::trunc_sat;
189 using ::native::load_splat;
190 using ::native::load_zero;
191 using ::native::load_lane;
192 using ::native::store_lane;
193 using ::native::load_widened;
194}
195#endif
constexpr auto mask_bits(simd< M, N, Arch > m) noexcept
constexpr predicate< N, Arch > to_predicate(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > select(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > to_vector_mask(predicate< N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_mul(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_mul_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< U, N, Arch > mask_cast(simd< T, N, Arch > value) noexcept
constexpr simd< T, N, Arch > masked_add(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > bit_select(simd< T, N, Arch > bits, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub_zero(M m, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< T, N, Arch > masked_sub(M m, simd< T, N, Arch > prior, simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
constexpr simd< float, N, Arch > masked_scaleb_zero(M mask, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > scaleb(simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< float, N, Arch > ceil(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > masked_scaleb(M mask, simd< float, N, Arch > prior, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
constexpr simd< To, N, Arch > convert(simd< float, N, Arch > x) noexcept
constexpr auto abs(simd< float, N, Arch > a) noexcept
constexpr simd< float, N, Arch > trunc(simd< float, N, Arch > x) noexcept
constexpr simd< float, N, Arch > floor(simd< float, N, Arch > x) noexcept
constexpr void store_simd(U *p, V value, simd_memory< A, Access >={}) noexcept(noexcept(value.template store_memory< A >(p)))
Definition common_body.h:58
simd_access
Definition common.h:280
constexpr std::size_t compress_store(T *destination, std::size_t capacity, typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value) noexcept
constexpr V load_simd(U const *p, simd_memory< A, Access >={}) noexcept(noexcept(V::template load_memory< A >(p)))
Definition common_body.h:40
constexpr V load_simd_partial(U const *p, std::size_t count, typename V::value_type fill={}, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_constructible_v< U, typename V::value_type & > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::load_simd< V >(p)))
Definition common_body.h:96
constexpr compaction_result< simd< T, N, Arch > > compress(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > value, T fill=T{}) noexcept
constexpr simd< T, N, Arch > expand(typename simd< T, N, Arch >::mask mask, simd< T, N, Arch > packed, simd< T, N, Arch > prior) noexcept
constexpr void store_simd_partial(U *p, V value, std::size_t count, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::store_simd(p, value)))
constexpr imm_t< K > imm
Definition common.h:82
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr simd< std::int16_t, 8, A > q15mulr_sat(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Multiply signed Q15 lanes, round by adding 2^14, shift by 15 and saturate.
constexpr simd< T, N, A > average_round(simd< T, N, A > a, simd< T, N, A > b) noexcept
Rounded unsigned average: (a+b+1)/2 without intermediate overflow.
constexpr auto multiply_widened_high(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened upper integer lanes; the complete product fits.
constexpr simd< T, N, A > sqrt(simd< T, N, A > v) noexcept
Compute correctly rounded square roots; negative finite lanes produce NaN.
constexpr bool all(simd< T, N, A > a) noexcept
Test whether every integer lane is nonzero.
wide(T, U...) -> wide< T, 1+sizeof...(U)>
Deduce the common element type and the number of homogeneous constructor arguments.
constexpr simd< To, N *2, A > narrow_sat(simd< From, N, A > a, simd< From, N, A > b) noexcept
Saturating concatenate from signed source lanes, including unsigned destinations.
constexpr simd< bool, N, Arch > to_bool(simd< T, N, Arch > value) noexcept
Convert lane truth into Boolean data lanes represented as zero or one, preserving the lane count.
constexpr simd< T, N, A > pmax(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
constexpr simd< T, N, A > shuffle(simd< T, N, A > a, simd< T, N, A > b) noexcept
Select N lanes from the concatenation of two inputs, using constant indices.
constexpr simd< T, N, A > round_even(simd< T, N, A > v) noexcept
Round floating lanes to nearest integers, choosing even at ties.
constexpr simd< T, N, A > add_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
constexpr auto extend_high(simd< T, N, A > a) noexcept
Widen the upper half of integer lanes, preserving signedness.
constexpr simd< T, N, A > pmin(simd< T, N, A > a, simd< T, N, A > b) noexcept
Pseudo minimum/maximum selects the first operand for unordered or equal lanes.
constexpr auto reinterpret_bits(simd< From, N, Arch > value) noexcept -> simd< To, sizeof(From) *N/sizeof(To), Arch >
constexpr simd< To, 2 *N, Arch > narrow_concat(simd< From, N, Arch > a, simd< From, N, Arch > b) noexcept
constexpr auto multiply_widened_low(simd< T, N, A > a, simd< T, N, A > b) noexcept
Multiply widened lower integer lanes; the complete product fits.
constexpr simd< T, N, Arch > popcount(simd< T, N, Arch > value) noexcept
constexpr simd< fp16, 32, Arch > fma(simd< fp16, 32, Arch > a, simd< fp16, 32, Arch > b, simd< fp16, 32, Arch > c) noexcept
constexpr simd< To, 4, A > trunc_sat(simd< From, N, A > a) noexcept
constexpr simd< std::uint8_t, 16, A > swizzle(simd< std::uint8_t, 16, A > v, simd< std::uint8_t, 16, A > indices) noexcept
Look up byte indices 0..15; every other index produces zero.
constexpr simd< T, N, A > max(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
constexpr simd< T, N, A > broadcast(simd< T, N, A > v, imm_t< I >) noexcept
Broadcast one compile-time-selected lane.
constexpr std::uint32_t bitmask(simd< T, N, A > a) noexcept
Gather each integer lane's sign bit into bit i of the scalar result.
constexpr std::array< V, N > flush_to_zero(std::array< V, N > const &x) noexcept
Replace binary32 subnormal lanes with signed zero. Both zero signs, normal values,...
Definition bits_body.h:52
constexpr auto pairwise_add_widened(simd< T, N, Arch > value) noexcept
constexpr V load_splat(typename V::value_type const *p) noexcept
Read one scalar and broadcast it; the access is exactly sizeof(T) bytes.
constexpr simd< std::int32_t, 4, A > dot(simd< std::int16_t, 8, A > a, simd< std::int16_t, 8, A > b) noexcept
Sum adjacent signed halfword products modulo 2^32.
constexpr auto extend_low(simd< T, N, A > a) noexcept
Widen the lower half of integer lanes, preserving signedness.
constexpr simd< To, 16/sizeof(To), A > load_widened(From const *p) noexcept
Read exactly eight bytes of source lanes and widen them with their signedness.
constexpr void store_lane(T *p, simd< T, N, A > a) noexcept
Write only lane I to one scalar object.
constexpr V load_zero(typename V::value_type const *p) noexcept
Read one 32- or 64-bit lane and zero the other lanes.
constexpr simd< T, N, A > sub_sat(simd< T, N, A > a, simd< T, N, A > b) noexcept
Saturate signed or unsigned byte/halfword arithmetic at the lane limits.
constexpr std::uint64_t reduce_add_widened(simd< T, N, Arch > value) noexcept
constexpr bool any(simd< T, N, A > a) noexcept
Test whether at least one integer lane is nonzero.
constexpr simd< T, N, A > load_lane(T const *p, simd< T, N, A > a) noexcept
Read one scalar into lane I, preserving the other lanes.
constexpr simd< T, N, A > min(simd< T, N, A > a, simd< T, N, A > b) noexcept
Minimum/maximum; floating NaNs propagate and signed zeros follow WebAssembly rules.
isa(E) -> isa< detail::family_of< E > >
Deduce the family from a feature enum; explicit isa<> always names the host family.