native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
Raw vector math

Functions

template<bool Flush = false, std::size_t L, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr simd< float, L, Arch > native::exp2 (simd< float, L, Arch > input) noexcept
template<bool Flush = false, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr std::array< simd< float, L, Arch >, N > native::exp2 (std::array< simd< float, L, Arch >, N > const &input) noexcept
template<bool Flush, std::size_t L, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr simd< float, L, Arch > native::exp2 (simd< float, L, Arch > input, std::bool_constant< Flush >) noexcept
template<bool Flush, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr std::array< simd< float, L, Arch >, N > native::exp2 (std::array< simd< float, L, Arch >, N > const &input, std::bool_constant< Flush >) noexcept
template<std::size_t L, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr simd< float, L, Arch > native::atan2 (simd< float, L, Arch > y, simd< float, L, Arch > x) noexcept
template<std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr std::array< simd< float, L, Arch >, N > native::atan2 (std::array< simd< float, L, Arch >, N > const &y, std::array< simd< float, L, Arch >, N > const &x) noexcept
template<bool Flush = false, unsigned Degree = 6, std::size_t L, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr simd< float, L, Arch > native::exp (simd< float, L, Arch > input) noexcept
 Evaluate the binary32 range-reduced exponential approximation. Degree selects a polynomial from one through seven, defaulting to six. Every degree shares range reduction and exponent scaling; none promises correctly rounded exp for every input. Flush selects the early underflow cutoff at compile time; it does not change CPU controls or turn a raw vector into a policy-bearing FTZ type.
template<bool Flush = false, unsigned Degree = 6, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr std::array< simd< float, L, Arch >, N > native::exp (std::array< simd< float, L, Arch >, N > const &input) noexcept
template<bool Flush, unsigned Degree = 6, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr std::array< simd< float, L, Arch >, N > native::exp (std::array< simd< float, L, Arch >, N > const &input, std::bool_constant< Flush >, std::integral_constant< unsigned, Degree >={}) noexcept
template<bool Flush, unsigned Degree = 6, std::size_t L, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
constexpr simd< float, L, Arch > native::exp (simd< float, L, Arch > input, std::bool_constant< Flush >, std::integral_constant< unsigned, Degree >={}) noexcept
template<std::size_t N, class M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::native_scaleb_shape<N> &&(std::same_as<M,typename simd
<float,N,Arch>::mask_type> || std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
constexpr simd< float, N, Arch > native::masked_scaleb (M maskmask, simd< float, N, Arch > prior, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
template<std::size_t N, class M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::native_scaleb_shape<N> &&(std::same_as<M,typename simd
<float,N,Arch>::mask_type> || std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
constexpr simd< float, N, Arch > native::masked_scaleb_zero (M maskmask, simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::native_scaleb_shape<N>
constexpr simd< float, N, Arch > native::scaleb (simd< float, N, Arch > value, simd< float, N, Arch > exponent) noexcept
template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
constexpr auto native::abs (simd< float, N, Arch > a) noexcept
template<class To, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N> && std::same_as<To,std::int32_t>
constexpr simd< To, N, Arch > native::convert (simd< float, N, Arch > x) noexcept
template<class To, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N> && std::same_as<To,float>
constexpr simd< To, N, Arch > native::convert (simd< std::int32_t, N, Arch > x) noexcept
template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
constexpr simd< float, N, Arch > native::floor (simd< float, N, Arch > x) noexcept
template<std::size_t N, std::size_t M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
constexpr std::array< simd< float, N, Arch >, M > native::floor (std::array< simd< float, N, Arch >, M > const &input) noexcept
template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
constexpr simd< float, N, Arch > native::ceil (simd< float, N, Arch > x) noexcept
template<std::size_t N, std::size_t M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
constexpr std::array< simd< float, N, Arch >, M > native::ceil (std::array< simd< float, N, Arch >, M > const &input) noexcept
template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
constexpr simd< float, N, Arch > native::trunc (simd< float, N, Arch > x) noexcept
template<std::size_t N, std::size_t M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
constexpr std::array< simd< float, N, Arch >, M > native::trunc (std::array< simd< float, N, Arch >, M > const &input) noexcept

Detailed Description

At runtime, raw float operations inherit the caller's floating-point environment. Constant evaluation uses nearest-even rounding and gradual underflow, without accessing floating-point controls or status flags. They do not establish FTZ policy or a reproducible scalar type. Use unqualified calls in generic code so the element library can supply its own operations.

template<native::isa<> Arch>
void arithmetic() {
auto y = fma(V(2.f), V(3.f), V(1.f));
check(all(y == V(7.f)));
check(all(sqrt(V(4.f)) == V(2.f)));
check(all(native::abs(V(-2.f)) == V(2.f)));
if constexpr(requires(V v) { native::scaleb(v,v); })
check(all(native::scaleb(V(1.f), V(3.5f)) == V(8.f)));
auto integers = native::convert<std::int32_t>(V(3.75f));
check(all(native::convert<float>(integers) == V(3.f)));
}

Function Documentation

◆ abs()

template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
auto native::abs ( simd< float, N, Arch > a)
inlineconstexprnoexcept

Clear each binary32 sign bit, preserving the remaining payload bits.

Definition at line 4475 of file simd_family.h.

Here is the caller graph for this function:

◆ atan2() [1/2]

template<std::size_t L, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
simd< float, L, Arch > native::atan2 ( simd< float, L, Arch > y,
simd< float, L, Arch > x )
inlineconstexprnoexcept

Evaluate atan2(y,x), retaining the common SIMD architecture and lane count.

Definition at line 91 of file exp_body.h.

Here is the caller graph for this function:

◆ atan2() [2/2]

template<std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
std::array< simd< float, L, Arch >, N > native::atan2 ( std::array< simd< float, L, Arch >, N > const & y,
std::array< simd< float, L, Arch >, N > const & x )
inlineconstexprnoexcept

Advance each atan2 stage across matching arrays of independent registers.

Definition at line 97 of file exp_body.h.

◆ ceil() [1/2]

template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
simd< float, N, Arch > native::ceil ( simd< float, N, Arch > x)
inlineconstexprnoexcept

Round each lane toward positive infinity, independently of ambient rounding. Signed zeros and infinities are preserved; NaNs remain NaNs. Raw denormal handling still follows the CPU environment. No NaN payload or FP-status promise is added.

template<native::isa<> Arch>
void rounding() {
V x{-1.75f, -0.25f, 0.25f, 1.75f};
check(all(native::floor(x) == V{-2.f, -1.f, 0.f, 1.f}));
check(all(native::ceil(x) == V{-1.f, -0.f, 1.f, 2.f}));
check(all(native::trunc(x) == V{-1.f, -0.f, 0.f, 1.f}));
}

Definition at line 5158 of file simd_family.h.

Here is the caller graph for this function:

◆ ceil() [2/2]

template<std::size_t N, std::size_t M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
std::array< simd< float, N, Arch >, M > native::ceil ( std::array< simd< float, N, Arch >, M > const & input)
inlineconstexprnoexcept

Apply ceil to each register in an array; an empty array performs no lane work.

Definition at line 5164 of file simd_family.h.

◆ convert() [1/2]

template<class To, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N> && std::same_as<To,std::int32_t>
simd< To, N, Arch > native::convert ( simd< float, N, Arch > x)
inlineconstexprnoexcept

Convert floats to signed 32-bit integers by truncation toward zero.

Precondition
Every truncated result is representable; NaNs and infinities are outside the contract.

Definition at line 4562 of file simd_family.h.

Here is the caller graph for this function:

◆ convert() [2/2]

template<class To, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N> && std::same_as<To,float>
simd< To, N, Arch > native::convert ( simd< std::int32_t, N, Arch > x)
inlineconstexprnoexcept

Numerically convert signed 32-bit integers to binary32. This is not a bit cast. Values outside binary32's exact integer range round.

Definition at line 4585 of file simd_family.h.

◆ exp() [1/4]

template<bool Flush = false, unsigned Degree = 6, std::size_t L, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
simd< float, L, Arch > native::exp ( simd< float, L, Arch > input)
inlineconstexprnoexcept

Evaluate the binary32 range-reduced exponential approximation. Degree selects a polynomial from one through seven, defaulting to six. Every degree shares range reduction and exponent scaling; none promises correctly rounded exp for every input. Flush selects the early underflow cutoff at compile time; it does not change CPU controls or turn a raw vector into a policy-bearing FTZ type.

template<native::isa<> Arch>
void exponential() {
std::array<V, 2> registers{V(0.f), V(1.f)};
auto result = native::exp(registers);
auto powers = native::exp2(registers);
check(all(powers[0] == V(1.f)) && all(powers[1] == V(2.f)));
auto exponents = native::log2(powers);
check(all(exponents[0] == V(0.f)) && all(exponents[1] == V(1.f)));
check(all(result[0] == V(1.f)));
check(all(result[1] > V(2.718f)) && all(result[1] < V(2.719f)));
}

Definition at line 109 of file exp_body.h.

Here is the caller graph for this function:

◆ exp() [2/4]

template<bool Flush, unsigned Degree = 6, std::size_t L, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
simd< float, L, Arch > native::exp ( simd< float, L, Arch > input,
std::bool_constant< Flush > ,
std::integral_constant< unsigned, Degree > = {} )
inlineconstexprnoexcept

Pass cutoff and degree through constant tags for dependent calls.

Definition at line 130 of file exp_body.h.

◆ exp() [3/4]

template<bool Flush = false, unsigned Degree = 6, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
std::array< simd< float, L, Arch >, N > native::exp ( std::array< simd< float, L, Arch >, N > const & input)
inlineconstexprnoexcept

Evaluate exp stage by stage across independent registers; N may be zero.

Definition at line 115 of file exp_body.h.

◆ exp() [4/4]

template<bool Flush, unsigned Degree = 6, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (Degree >= 1 && Degree <= 7) && (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
std::array< simd< float, L, Arch >, N > native::exp ( std::array< simd< float, L, Arch >, N > const & input,
std::bool_constant< Flush > ,
std::integral_constant< unsigned, Degree > = {} )
inlineconstexprnoexcept

Pass cutoff and degree through constant tags for dependent calls.

Definition at line 123 of file exp_body.h.

◆ exp2() [1/4]

template<bool Flush = false, std::size_t L, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
simd< float, L, Arch > native::exp2 ( simd< float, L, Arch > input)
inlineconstexprnoexcept

Base-two exponential with the same compile-time underflow policy as exp.

Definition at line 65 of file exp_body.h.

Here is the caller graph for this function:

◆ exp2() [2/4]

template<bool Flush, std::size_t L, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
simd< float, L, Arch > native::exp2 ( simd< float, L, Arch > input,
std::bool_constant< Flush >  )
inlineconstexprnoexcept

Select exp2's underflow policy through a tag for dependent calls.

Definition at line 77 of file exp_body.h.

◆ exp2() [3/4]

template<bool Flush = false, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
std::array< simd< float, L, Arch >, N > native::exp2 ( std::array< simd< float, L, Arch >, N > const & input)
inlineconstexprnoexcept

Evaluate exp2 stage by stage across independent registers.

Definition at line 71 of file exp_body.h.

◆ exp2() [4/4]

template<bool Flush, std::size_t L, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<L>
std::array< simd< float, L, Arch >, N > native::exp2 ( std::array< simd< float, L, Arch >, N > const & input,
std::bool_constant< Flush >  )
inlineconstexprnoexcept

Select exp2's underflow policy for a register batch through a tag.

Definition at line 83 of file exp_body.h.

◆ floor() [1/2]

template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
simd< float, N, Arch > native::floor ( simd< float, N, Arch > x)
inlineconstexprnoexcept

Round each lane toward negative infinity, independently of ambient rounding. Signed zeros and infinities are preserved; NaNs remain NaNs. Raw denormal handling still follows the CPU environment. No NaN payload or FP-status promise is added.

template<native::isa<> Arch>
void rounding() {
V x{-1.75f, -0.25f, 0.25f, 1.75f};
check(all(native::floor(x) == V{-2.f, -1.f, 0.f, 1.f}));
check(all(native::ceil(x) == V{-1.f, -0.f, 1.f, 2.f}));
check(all(native::trunc(x) == V{-1.f, -0.f, 0.f, 1.f}));
}

Definition at line 5142 of file simd_family.h.

Here is the caller graph for this function:

◆ floor() [2/2]

template<std::size_t N, std::size_t M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
std::array< simd< float, N, Arch >, M > native::floor ( std::array< simd< float, N, Arch >, M > const & input)
inlineconstexprnoexcept

Apply floor to each register in an array; an empty array performs no lane work.

Definition at line 5148 of file simd_family.h.

◆ masked_scaleb()

template<std::size_t N, class M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::native_scaleb_shape<N> &&(std::same_as<M,typename simd
<float,N,Arch>::mask_type> || std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
simd< float, N, Arch > native::masked_scaleb ( M mask,
simd< float, N, Arch > prior,
simd< float, N, Arch > value,
simd< float, N, Arch > exponent )
inlineconstexprnoexcept

Scale active lanes by 2^floor(exponent), retaining prior in other lanes. Inactive lanes are excluded from the scaling operation. The result follows the caller's floating-point environment, including denormal controls. Available only for native AVX512F scaling shapes; packed widths below 16 require AVX512VL. No scalar, AVX2, NEON or Wasm software fallback exists.

Definition at line 4345 of file simd_family.h.

Here is the caller graph for this function:

◆ masked_scaleb_zero()

template<std::size_t N, class M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::native_scaleb_shape<N> &&(std::same_as<M,typename simd
<float,N,Arch>::mask_type> || std::same_as<M,typename simd<float,N,Arch>::vector_mask_type>)
simd< float, N, Arch > native::masked_scaleb_zero ( M mask,
simd< float, N, Arch > value,
simd< float, N, Arch > exponent )
inlineconstexprnoexcept

Scale active lanes by 2^floor(exponent), writing positive zero elsewhere.

Definition at line 4379 of file simd_family.h.

Here is the caller graph for this function:

◆ scaleb()

template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::native_scaleb_shape<N>
simd< float, N, Arch > native::scaleb ( simd< float, N, Arch > value,
simd< float, N, Arch > exponent )
inlineconstexprnoexcept

Scale every lane by 2^floor(exponent), including fractional exponents. Unlike an integer ldexp exponent, the exponent argument is itself a vector.

Definition at line 4399 of file simd_family.h.

Here is the caller graph for this function:

◆ trunc() [1/2]

template<std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
simd< float, N, Arch > native::trunc ( simd< float, N, Arch > x)
inlineconstexprnoexcept

Round each lane toward zero, independently of ambient rounding. Signed zeros and infinities are preserved; NaNs remain NaNs. Raw denormal handling still follows the CPU environment. No NaN payload or FP-status promise is added.

template<native::isa<> Arch>
void rounding() {
V x{-1.75f, -0.25f, 0.25f, 1.75f};
check(all(native::floor(x) == V{-2.f, -1.f, 0.f, 1.f}));
check(all(native::ceil(x) == V{-1.f, -0.f, 1.f, 2.f}));
check(all(native::trunc(x) == V{-1.f, -0.f, 0.f, 1.f}));
}

Definition at line 5174 of file simd_family.h.

Here is the caller graph for this function:

◆ trunc() [2/2]

template<std::size_t N, std::size_t M, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && ::native::detail::avx512_backend::float_shape<N>
std::array< simd< float, N, Arch >, M > native::trunc ( std::array< simd< float, N, Arch >, M > const & input)
inlineconstexprnoexcept

Apply trunc to each register in an array; an empty array performs no lane work.

Definition at line 5180 of file simd_family.h.