native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
Vector memory

Classes

struct  native::compaction_result< V >
struct  native::simd_memory< Alignment, Access >

Enumerations

enum class  native::simd_access

Functions

template<class V, class U, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) && requires(U const
* p) { V::template load_memory<A>(p); }
constexpr V native::load_simd (U const *p, simd_memory< A, Access >={}) noexcept(noexcept(V::template load_memory< A >(p)))
template<class U, class V, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) && requires(V value,U
* p) { value.template store_memory<A>(p); }
constexpr void native::store_simd (U *p, V value, simd_memory< A, Access >={}) noexcept(noexcept(value.template store_memory< A >(p)))
template<class V, class U, std::size_t N>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&(N==V::lanes) &&
requires(U const * p) { ::native::load_simd<V>(p); }
constexpr V native::load_simd (std::array< U, N > const &values) noexcept(noexcept(::native::load_simd< V >(values.data())))
template<class V, class U, std::size_t N>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&(N==V::lanes) &&
requires(U * p) { ::native::load_simd<V>(p); }
constexpr V native::load_simd (std::span< U, N > values) noexcept(noexcept(::native::load_simd< V >(values.data())))
template<class V, class U, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&
std::default_initializable<U> && std::constructible_from<U,typename V::value_type &> && std::is_copy_assignable_v<U> &&
requires(U const * p) { ::native::load_simd<V>(p); }
constexpr V native::load_simd_partial (U const *p, std::size_t count, typename V::value_type fill={}, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_constructible_v< U, typename V::value_type & > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::load_simd< V >(p)))
template<class U, class V, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&
std::default_initializable<U> && std::is_copy_assignable_v<U> && requires(U * p,V value) { ::native::store_simd(p,value); }
constexpr void native::store_simd_partial (U *p, V value, std::size_t count, simd_memory< A, Access >={}) noexcept(std::is_nothrow_default_constructible_v< U > &&std::is_nothrow_copy_assignable_v< U > &&noexcept(::native::store_simd(p, value)))
template<class T, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
::native::detail::avx512_backend::float_shape<N>
constexpr compaction_result< simd< T, N, Arch > > native::compress (typename simd< T, N, Arch >::mask maskmask, simd< T, N, Arch > value, T fill=T{}) noexcept
template<class T, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
::native::detail::avx512_backend::float_shape<N>
constexpr simd< T, N, Arch > native::expand (typename simd< T, N, Arch >::mask maskmask, simd< T, N, Arch > packed, simd< T, N, Arch > prior) noexcept
template<class T, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
::native::detail::avx512_backend::float_shape<N>
constexpr std::size_t native::compress_store (T *destination, std::size_t capacity, typename simd< T, N, Arch >::mask maskmask, simd< T, N, Arch > value) noexcept

Detailed Description

Loads select a concrete vector type. Full operations require storage for all logical lanes; partial operations touch exactly the requested prefix. Alignment is a caller promise, not a runtime check.

template<native::isa<> Arch>
void memory() {
std::array<float, 4> input{1.f, 2.f, 3.f, 4.f};
auto full = native::load_simd<V>(input);
auto tail = native::load_simd_partial<V>(input.data(), 3, -1.f);
std::array<float, 4> output{9.f, 9.f, 9.f, 9.f};
native::store_simd_partial(output.data(), tail, 3);
check(output == std::array<float, 4>{1.f, 2.f, 3.f, 9.f});
native::store_simd(output.data(), full);
check(output == input);
}

Enumeration Type Documentation

◆ simd_access

enum class native::simd_access
strong

Requested access policy. Streaming currently uses ordinary memory access.

Definition at line 280 of file common.h.

Function Documentation

◆ compress()

template<class T, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
::native::detail::avx512_backend::float_shape<N>
compaction_result< simd< T, N, Arch > > native::compress ( typename simd< T, N, Arch >::mask mask,
simd< T, N, Arch > value,
T fill = T{} )
inlineconstexprnoexcept

Pack selected logical lanes in increasing order; fill all remaining lanes. This rearranges bits, including NaN payloads and signed zeros. Short padding never contributes to count or output and is zero in the result.

template<native::isa<> Arch>
void compaction() {
auto active = V::mask::from_bitset(0b1010);
V input{10u, 20u, 30u, 40u};
auto packed = native::compress(active, input, 99u);
check(packed.count == 2 && all(packed.value == V{20u, 40u, 99u, 99u}));
auto restored = native::expand(active, packed.value, V(77u));
check(all(restored == V{77u, 20u, 77u, 40u}));
std::array<std::uint32_t, 2> output{0u, 123u};
auto written = native::compress_store(output.data(), 1, active, input);
check(written == 1 && output[0] == 20u && output[1] == 123u);
}

Definition at line 5313 of file simd_family.h.

Here is the caller graph for this function:

◆ compress_store()

template<class T, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
::native::detail::avx512_backend::float_shape<N>
std::size_t native::compress_store ( T * destination,
std::size_t capacity,
typename simd< T, N, Arch >::mask mask,
simd< T, N, Arch > value )
inlineconstexprnoexcept

Write the first min(capacity,popcount(mask)) selected lanes, in input order. Return the number WRITTEN, not the total selected count. No element beyond that prefix is accessed. Null is valid when capacity or the mask is zero.

Definition at line 5357 of file simd_family.h.

Here is the caller graph for this function:

◆ expand()

template<class T, std::size_t N, ::native::isa<> Arch>
requires (::native::avx512 <= Arch ) && (std::same_as<T,float> || std::same_as<T,std::int32_t> || std::same_as<T,std::uint32_t>) &&
::native::detail::avx512_backend::float_shape<N>
simd< T, N, Arch > native::expand ( typename simd< T, N, Arch >::mask mask,
simd< T, N, Arch > packed,
simd< T, N, Arch > prior )
inlineconstexprnoexcept

Consume packed's first popcount(mask) lanes in order at selected positions. Unselected lanes retain prior bitwise. Short physical padding is zeroed.

Definition at line 5335 of file simd_family.h.

Here is the caller graph for this function:

◆ load_simd() [1/3]

template<class V, class U, std::size_t N>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&(N==V::lanes) &&
requires(U const * p) { ::native::load_simd<V>(p); }
V native::load_simd ( std::array< U, N > const & values)
inlineconstexprnoexcept

Load an array or fixed-extent span whose extent equals the lane count.

Definition at line 75 of file common_body.h.

◆ load_simd() [2/3]

template<class V, class U, std::size_t N>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&(N==V::lanes) &&
requires(U * p) { ::native::load_simd<V>(p); }
V native::load_simd ( std::span< U, N > values)
inlineconstexprnoexcept

Load an array or fixed-extent span whose extent equals the lane count.

Definition at line 83 of file common_body.h.

◆ load_simd() [3/3]

template<class V, class U, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) && requires(U const
* p) { V::template load_memory<A>(p); }
V native::load_simd ( U const * p,
simd_memory< A, Access > = {} )
inlineconstexprnoexcept

Load all V::lanes elements using the element's memory customization.

Precondition
p addresses that many readable elements and meets alignment A. Exceptions propagate from the selected V::load_memory<A> operation.

Definition at line 40 of file common_body.h.

Here is the caller graph for this function:

◆ load_simd_partial()

template<class V, class U, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&
std::default_initializable<U> && std::constructible_from<U,typename V::value_type &> && std::is_copy_assignable_v<U> &&
requires(U const * p) { ::native::load_simd<V>(p); }
V native::load_simd_partial ( U const * p,
std::size_t count,
typename V::value_type fill = {},
simd_memory< A, Access > = {} )
inlineconstexprnoexcept

Load count elements and fill the remaining lanes with fill.

Precondition
count <= V::lanes; p may be null only when count == 0. Uses an element temporary; construction, assignment and loading may throw. The alignment hint does not extend the readable prefix.

Definition at line 96 of file common_body.h.

Here is the caller graph for this function:

◆ store_simd()

template<class U, class V, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) && requires(V value,U
* p) { value.template store_memory<A>(p); }
void native::store_simd ( U * p,
V value,
simd_memory< A, Access > = {} )
inlineconstexprnoexcept

Store all logical lanes using the element's memory customization.

Precondition
p addresses that many writable elements and meets alignment A.

Definition at line 58 of file common_body.h.

Here is the caller graph for this function:

◆ store_simd_partial()

template<class U, class V, std::size_t A = 1, simd_access Access = simd_access::ordinary>
requires detail::memory_architecture<V,U>::known && NATIVE_COMMON_ARCH(detail::memory_architecture_v<V,U>) &&
std::default_initializable<U> && std::is_copy_assignable_v<U> && requires(U * p,V value) { ::native::store_simd(p,value); }
void native::store_simd_partial ( U * p,
V value,
std::size_t count,
simd_memory< A, Access > = {} )
inlineconstexprnoexcept

Store the first count logical lanes, leaving the following memory alone.

Precondition
count <= V::lanes; p may be null only when count == 0. Element construction, assignment and storage determine the exception guarantee.

Definition at line 112 of file common_body.h.

Here is the caller graph for this function: