native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
NEON
Collaboration diagram for NEON:

Functions

template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::sqadd (simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
 Signed saturating lane addition.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::uqadd (simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
 Unsigned saturating lane addition.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::sqsub (simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
 Signed saturating lane subtraction.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::uqsub (simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
 Unsigned saturating lane subtraction.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) == 2 || sizeof(T) == 4) && (sizeof(T) * N == 8 ||
sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::sqdmulh (simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
 Signed doubled multiply-high, saturating the minimum-times-minimum case.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) == 2 || sizeof(T) == 4) && (sizeof(T) * N == 8 ||
sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::sqrdmulh (simd< T, N, Arch > a, simd< T, N, Arch > b) noexcept
 Signed doubled multiply-high with rounding before final saturation.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::sshl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
 Shift by each count lane's signed low byte; negative counts shift right.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::srshl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
 Shift by each count lane's signed low byte; negative counts shift right with rounding.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::sqshl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::sqrshl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::ushl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
 Shift by each count lane's signed low byte; negative counts shift right.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::urshl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
 Shift by each count lane's signed low byte; negative counts shift right with rounding.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::uqshl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::uqrshl (simd< T, N, Arch > a, simd< std::make_signed_t< T >, N, Arch > b) noexcept
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && sizeof(T) * N == 16 && (sizeof(T) == 2 || sizeof(T) == 4 ||
sizeof(T) == 8))
constexpr auto native::sqxtn (simd< T, N, Arch > a) noexcept
 Narrow 128 bits with saturation to a 64-bit result.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && sizeof(T) * N == 16 && (sizeof(T) == 2 || sizeof(T) == 4 ||
sizeof(T) == 8))
constexpr auto native::sqxtn_high (simd< std::conditional_t< sizeof(T)==2, std::int8_t, std::conditional_t< sizeof(T)==4, std::int16_t, std::int32_t > >, N, Arch > low, simd< T, N, Arch > a) noexcept
 Narrow with saturation and append above the preserved low half.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && sizeof(T) * N == 16 && (sizeof(T) == 2 || sizeof(T) == 4 ||
sizeof(T) == 8))
constexpr auto native::uqxtn (simd< T, N, Arch > a) noexcept
 Narrow 128 bits with saturation to a 64-bit result.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && sizeof(T) * N == 16 && (sizeof(T) == 2 || sizeof(T) == 4 ||
sizeof(T) == 8))
constexpr auto native::uqxtn_high (simd< std::conditional_t< sizeof(T)==2, std::uint8_t, std::conditional_t< sizeof(T)==4, std::uint16_t, std::uint32_t > >, N, Arch > low, simd< T, N, Arch > a) noexcept
 Narrow with saturation and append above the preserved low half.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && sizeof(T) * N == 16 && (sizeof(T) == 2 || sizeof(T) == 4 ||
sizeof(T) == 8))
constexpr auto native::sqxtun (simd< T, N, Arch > a) noexcept
 Narrow 128 bits with saturation to a 64-bit result.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && sizeof(T) * N == 16 && (sizeof(T) == 2 || sizeof(T) == 4 ||
sizeof(T) == 8))
constexpr auto native::sqxtun_high (simd< std::conditional_t< sizeof(T)==2, std::uint8_t, std::conditional_t< sizeof(T)==4, std::uint16_t, std::uint32_t > >, N, Arch > low, simd< T, N, Arch > a) noexcept
 Narrow with saturation and append above the preserved low half.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && (sizeof(T) <= 4) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::clz (simd< T, N, Arch > a) noexcept
 Count leading zero bits in each lane, including the full width for zero.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 4) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::cls (simd< T, N, Arch > a) noexcept
 Count leading sign bits in each lane, excluding the sign bit itself.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && (sizeof(T) == 1) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::rbit (simd< T, N, Arch > a) noexcept
 Reverse the eight bits within each byte lane.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && (sizeof(T) == 1) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::rev16 (simd< T, N, Arch > a) noexcept
 Reverse byte lanes within each 16-bit block.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && (sizeof(T) <= 2) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::rev32 (simd< T, N, Arch > a) noexcept
 Reverse byte or halfword lanes within each 32-bit block.
template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && (sizeof(T) <= 4) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
constexpr simd< T, N, Arch > native::rev64 (simd< T, N, Arch > a) noexcept
 Reverse byte, halfword or word lanes within each 64-bit block.

Detailed Description

Typed instruction operations require arm_feature::neon and a matching caller target. Saturation may set sticky FPSR.QC at runtime; constant evaluation computes values only. Big-endian two-lane 64-bit add/subtract and variable shifts carry a known zero-overhead exception: Clang adds register permutations for these shapes.

Function Documentation

◆ sqrshl()

template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
simd< T, N, Arch > native::sqrshl ( simd< T, N, Arch > a,
simd< std::make_signed_t< T >, N, Arch > b )
inlineconstexprexportnoexcept

Shift by each count lane's signed low byte; negative counts shift right with rounding. Left shifts saturate and can set FPSR.QC.

Definition at line 528 of file native.arm.neon.ccm.

Here is the caller graph for this function:

◆ sqshl()

template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_signed_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
simd< T, N, Arch > native::sqshl ( simd< T, N, Arch > a,
simd< std::make_signed_t< T >, N, Arch > b )
inlineconstexprexportnoexcept

Shift by each count lane's signed low byte; negative counts shift right. Left shifts saturate and can set FPSR.QC.

Definition at line 470 of file native.arm.neon.ccm.

Here is the caller graph for this function:

◆ uqrshl()

template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
simd< T, N, Arch > native::uqrshl ( simd< T, N, Arch > a,
simd< std::make_signed_t< T >, N, Arch > b )
inlineconstexprexportnoexcept

Shift by each count lane's signed low byte; negative counts shift right with rounding. Left shifts saturate and can set FPSR.QC.

Definition at line 759 of file native.arm.neon.ccm.

Here is the caller graph for this function:

◆ uqshl()

template<isa< arm > Arch, simd_integer_element T, std::size_t N>
requires (Arch.has(arm_feature::neon) && std::is_unsigned_v<T> && (sizeof(T) <= 8) && (sizeof(T) * N == 8 || sizeof(T) * N == 16))
simd< T, N, Arch > native::uqshl ( simd< T, N, Arch > a,
simd< std::make_signed_t< T >, N, Arch > b )
inlineconstexprexportnoexcept

Shift by each count lane's signed low byte; negative counts shift right. Left shifts saturate and can set FPSR.QC.

Definition at line 701 of file native.arm.neon.ccm.

Here is the caller graph for this function: