|
native 0.0.1
Vectors, masks and wide register packs for C++26
|
Functions | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpbusd (simd< std::int32_t, 4, Arch > accumulator, simd< std::uint8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and signed bytes from b, modulo 2^32. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::mask_dpbusd (simd< std::int32_t, 4, Arch > accumulator, predicate< 4, Arch > maskmask, simd< std::uint8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| dpbusd in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::maskz_dpbusd (predicate< 4, Arch > maskmask, simd< std::int32_t, 4, Arch > accumulator, simd< std::uint8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| dpbusd in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpbusds (simd< std::int32_t, 4, Arch > accumulator, simd< std::uint8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and signed bytes from b, with signed 32-bit saturation. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::mask_dpbusds (simd< std::int32_t, 4, Arch > accumulator, predicate< 4, Arch > maskmask, simd< std::uint8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| dpbusds in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::maskz_dpbusds (predicate< 4, Arch > maskmask, simd< std::int32_t, 4, Arch > accumulator, simd< std::uint8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| dpbusds in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpwssd (simd< std::int32_t, 4, Arch > accumulator, simd< std::int16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| Accumulate products of two signed words from a and b, modulo 2^32. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::mask_dpwssd (simd< std::int32_t, 4, Arch > accumulator, predicate< 4, Arch > maskmask, simd< std::int16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| dpwssd in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::maskz_dpwssd (predicate< 4, Arch > maskmask, simd< std::int32_t, 4, Arch > accumulator, simd< std::int16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| dpwssd in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpwssds (simd< std::int32_t, 4, Arch > accumulator, simd< std::int16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| Accumulate products of two signed words from a and b, with signed 32-bit saturation. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::mask_dpwssds (simd< std::int32_t, 4, Arch > accumulator, predicate< 4, Arch > maskmask, simd< std::int16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| dpwssds in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::maskz_dpwssds (predicate< 4, Arch > maskmask, simd< std::int32_t, 4, Arch > accumulator, simd< std::int16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| dpwssds in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpbusd (simd< std::int32_t, 8, Arch > accumulator, simd< std::uint8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and signed bytes from b, modulo 2^32. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::mask_dpbusd (simd< std::int32_t, 8, Arch > accumulator, predicate< 8, Arch > maskmask, simd< std::uint8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| dpbusd in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::maskz_dpbusd (predicate< 8, Arch > maskmask, simd< std::int32_t, 8, Arch > accumulator, simd< std::uint8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| dpbusd in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpbusds (simd< std::int32_t, 8, Arch > accumulator, simd< std::uint8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and signed bytes from b, with signed 32-bit saturation. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::mask_dpbusds (simd< std::int32_t, 8, Arch > accumulator, predicate< 8, Arch > maskmask, simd< std::uint8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| dpbusds in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::maskz_dpbusds (predicate< 8, Arch > maskmask, simd< std::int32_t, 8, Arch > accumulator, simd< std::uint8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| dpbusds in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpwssd (simd< std::int32_t, 8, Arch > accumulator, simd< std::int16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| Accumulate products of two signed words from a and b, modulo 2^32. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::mask_dpwssd (simd< std::int32_t, 8, Arch > accumulator, predicate< 8, Arch > maskmask, simd< std::int16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| dpwssd in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::maskz_dpwssd (predicate< 8, Arch > maskmask, simd< std::int32_t, 8, Arch > accumulator, simd< std::int16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| dpwssd in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnni)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpwssds (simd< std::int32_t, 8, Arch > accumulator, simd< std::int16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| Accumulate products of two signed words from a and b, with signed 32-bit saturation. Uses AVX-VNNI. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::mask_dpwssds (simd< std::int32_t, 8, Arch > accumulator, predicate< 8, Arch > maskmask, simd< std::int16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| dpwssds in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::maskz_dpwssds (predicate< 8, Arch > maskmask, simd< std::int32_t, 8, Arch > accumulator, simd< std::int16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| dpwssds in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::dpbusd (simd< std::int32_t, 16, Arch > accumulator, simd< std::uint8_t, 64, Arch > a, simd< std::int8_t, 64, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and signed bytes from b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::mask_dpbusd (simd< std::int32_t, 16, Arch > accumulator, predicate< 16, Arch > maskmask, simd< std::uint8_t, 64, Arch > a, simd< std::int8_t, 64, Arch > b) noexcept |
| dpbusd in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::maskz_dpbusd (predicate< 16, Arch > maskmask, simd< std::int32_t, 16, Arch > accumulator, simd< std::uint8_t, 64, Arch > a, simd< std::int8_t, 64, Arch > b) noexcept |
| dpbusd in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::dpbusds (simd< std::int32_t, 16, Arch > accumulator, simd< std::uint8_t, 64, Arch > a, simd< std::int8_t, 64, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and signed bytes from b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::mask_dpbusds (simd< std::int32_t, 16, Arch > accumulator, predicate< 16, Arch > maskmask, simd< std::uint8_t, 64, Arch > a, simd< std::int8_t, 64, Arch > b) noexcept |
| dpbusds in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::maskz_dpbusds (predicate< 16, Arch > maskmask, simd< std::int32_t, 16, Arch > accumulator, simd< std::uint8_t, 64, Arch > a, simd< std::int8_t, 64, Arch > b) noexcept |
| dpbusds in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::dpwssd (simd< std::int32_t, 16, Arch > accumulator, simd< std::int16_t, 32, Arch > a, simd< std::int16_t, 32, Arch > b) noexcept |
| Accumulate products of two signed words from a and b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::mask_dpwssd (simd< std::int32_t, 16, Arch > accumulator, predicate< 16, Arch > maskmask, simd< std::int16_t, 32, Arch > a, simd< std::int16_t, 32, Arch > b) noexcept |
| dpwssd in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::maskz_dpwssd (predicate< 16, Arch > maskmask, simd< std::int32_t, 16, Arch > accumulator, simd< std::int16_t, 32, Arch > a, simd< std::int16_t, 32, Arch > b) noexcept |
| dpwssd in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::dpwssds (simd< std::int32_t, 16, Arch > accumulator, simd< std::int16_t, 32, Arch > a, simd< std::int16_t, 32, Arch > b) noexcept |
| Accumulate products of two signed words from a and b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::mask_dpwssds (simd< std::int32_t, 16, Arch > accumulator, predicate< 16, Arch > maskmask, simd< std::int16_t, 32, Arch > a, simd< std::int16_t, 32, Arch > b) noexcept |
| dpwssds in active lanes; inactive lanes retain accumulator. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512vnni)) | |
| constexpr simd< std::int32_t, 16, Arch > | native::maskz_dpwssds (predicate< 16, Arch > maskmask, simd< std::int32_t, 16, Arch > accumulator, simd< std::int16_t, 32, Arch > a, simd< std::int16_t, 32, Arch > b) noexcept |
| dpwssds in active lanes; inactive lanes become zero. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpbssd (simd< std::int32_t, 4, Arch > accumulator, simd< std::int8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpbssd (simd< std::int32_t, 8, Arch > accumulator, simd< std::int8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpbssds (simd< std::int32_t, 4, Arch > accumulator, simd< std::int8_t, 16, Arch > a, simd< std::int8_t, 16, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpbssds (simd< std::int32_t, 8, Arch > accumulator, simd< std::int8_t, 32, Arch > a, simd< std::int8_t, 32, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpbsud (simd< std::int32_t, 4, Arch > accumulator, simd< std::int8_t, 16, Arch > a, simd< std::uint8_t, 16, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and unsigned bytes from b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpbsud (simd< std::int32_t, 8, Arch > accumulator, simd< std::int8_t, 32, Arch > a, simd< std::uint8_t, 32, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and unsigned bytes from b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpbsuds (simd< std::int32_t, 4, Arch > accumulator, simd< std::int8_t, 16, Arch > a, simd< std::uint8_t, 16, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and unsigned bytes from b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpbsuds (simd< std::int32_t, 8, Arch > accumulator, simd< std::int8_t, 32, Arch > a, simd< std::uint8_t, 32, Arch > b) noexcept |
| Accumulate products of four signed bytes from a and unsigned bytes from b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::uint32_t, 4, Arch > | native::dpbuud (simd< std::uint32_t, 4, Arch > accumulator, simd< std::uint8_t, 16, Arch > a, simd< std::uint8_t, 16, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::uint32_t, 8, Arch > | native::dpbuud (simd< std::uint32_t, 8, Arch > accumulator, simd< std::uint8_t, 32, Arch > a, simd< std::uint8_t, 32, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::uint32_t, 4, Arch > | native::dpbuuds (simd< std::uint32_t, 4, Arch > accumulator, simd< std::uint8_t, 16, Arch > a, simd< std::uint8_t, 16, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and b, with unsigned 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint8)) | |
| constexpr simd< std::uint32_t, 8, Arch > | native::dpbuuds (simd< std::uint32_t, 8, Arch > accumulator, simd< std::uint8_t, 32, Arch > a, simd< std::uint8_t, 32, Arch > b) noexcept |
| Accumulate products of four unsigned bytes from a and b, with unsigned 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpwsud (simd< std::int32_t, 4, Arch > accumulator, simd< std::int16_t, 8, Arch > a, simd< std::uint16_t, 8, Arch > b) noexcept |
| Accumulate products of two signed words from a and unsigned words from b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpwsud (simd< std::int32_t, 8, Arch > accumulator, simd< std::int16_t, 16, Arch > a, simd< std::uint16_t, 16, Arch > b) noexcept |
| Accumulate products of two signed words from a and unsigned words from b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpwsuds (simd< std::int32_t, 4, Arch > accumulator, simd< std::int16_t, 8, Arch > a, simd< std::uint16_t, 8, Arch > b) noexcept |
| Accumulate products of two signed words from a and unsigned words from b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpwsuds (simd< std::int32_t, 8, Arch > accumulator, simd< std::int16_t, 16, Arch > a, simd< std::uint16_t, 16, Arch > b) noexcept |
| Accumulate products of two signed words from a and unsigned words from b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpwusd (simd< std::int32_t, 4, Arch > accumulator, simd< std::uint16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and signed words from b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpwusd (simd< std::int32_t, 8, Arch > accumulator, simd< std::uint16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and signed words from b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 4, Arch > | native::dpwusds (simd< std::int32_t, 4, Arch > accumulator, simd< std::uint16_t, 8, Arch > a, simd< std::int16_t, 8, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and signed words from b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::int32_t, 8, Arch > | native::dpwusds (simd< std::int32_t, 8, Arch > accumulator, simd< std::uint16_t, 16, Arch > a, simd< std::int16_t, 16, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and signed words from b, with signed 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::uint32_t, 4, Arch > | native::dpwuud (simd< std::uint32_t, 4, Arch > accumulator, simd< std::uint16_t, 8, Arch > a, simd< std::uint16_t, 8, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::uint32_t, 8, Arch > | native::dpwuud (simd< std::uint32_t, 8, Arch > accumulator, simd< std::uint16_t, 16, Arch > a, simd< std::uint16_t, 16, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and b, modulo 2^32. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::uint32_t, 4, Arch > | native::dpwuuds (simd< std::uint32_t, 4, Arch > accumulator, simd< std::uint16_t, 8, Arch > a, simd< std::uint16_t, 8, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and b, with unsigned 32-bit saturation. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::avxvnniint16)) | |
| constexpr simd< std::uint32_t, 8, Arch > | native::dpwuuds (simd< std::uint32_t, 8, Arch > accumulator, simd< std::uint16_t, 16, Arch > a, simd< std::uint16_t, 16, Arch > b) noexcept |
| Accumulate products of two unsigned words from a and b, with unsigned 32-bit saturation. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbusd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::mask_dpbusd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::maskz_dpbusd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbusds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::mask_dpbusds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::maskz_dpbusds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwssd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::mask_dpwssd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::maskz_dpwssd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwssds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::mask_dpwssds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::maskz_dpwssds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbssd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbssds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbsud (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbsuds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbuud (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpbuuds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwsud (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwsuds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwusd (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwusds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwuud (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::dpwuuds (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
Each 32-bit accumulator lane receives the sum of four byte products or two word products from the corresponding adjacent input group. Non-saturating forms wrap modulo 2^32. Saturating forms clamp the complete sum, including the accumulator, without first wrapping or clamping the product sum. All saturation is signed except dpbuuds and dpwuuds, which use unsigned accumulators and unsigned saturation. These register operations do not change integer flags or floating-point status.
Unmasked 128/256-bit core operations use AVX-VNNI when Arch contains avxvnni; otherwise they require AVX512F, AVX512VL and AVX512VNNI. 512-bit core operations require AVX512F and AVX512VNNI. Masked forms always require those EVEX features, with AVX512VL for 128/256 bits. INT8/INT16 extensions have only unmasked 128/256-bit forms and require their own independent feature. Compiler prerequisite closure and CPU/OS admission are separate from the exact instruction constraints below.
Mask bit i selects 32-bit result lane i. Merge forms retain accumulator lanes; zero forms clear inactive lanes. Bits above the lane count are ignored. Callers must enable a compatible target and admit its CPU and OS state requirements before execution. There is no runtime dispatch. Constant evaluation uses exact integer semantics. Tags without the instruction features are accepted only at compile time and require complete SIMD storage.