|
native 0.0.1
Vectors, masks and wide register packs for C++26
|
Functions | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni)) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8mulb (simd< std::uint8_t, 16, Arch > a, simd< std::uint8_t, 16, Arch > b) noexcept |
| Multiply corresponding bytes in GF(2^8). | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8affineqb (simd< std::uint8_t, 16, Arch > a, simd< std::uint64_t, 2, Arch > matrix) noexcept |
| Apply the binary matrix in each 64-bit lane and XOR Imm8. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8affineinvqb (simd< std::uint8_t, 16, Arch > a, simd< std::uint64_t, 2, Arch > matrix) noexcept |
| Invert each field byte, apply its lane's binary matrix, and XOR Imm8. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8mulb_mask (simd< std::uint8_t, 16, Arch > src, predicate< 16, Arch > k, simd< std::uint8_t, 16, Arch > a, simd< std::uint8_t, 16, Arch > b) noexcept |
| Merge inactive bytes from src after gf2p8mulb. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8mulb_maskz (predicate< 16, Arch > k, simd< std::uint8_t, 16, Arch > a, simd< std::uint8_t, 16, Arch > b) noexcept |
| Zero inactive bytes after gf2p8mulb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8affineqb_mask (simd< std::uint8_t, 16, Arch > src, predicate< 16, Arch > k, simd< std::uint8_t, 16, Arch > a, simd< std::uint64_t, 2, Arch > matrix) noexcept |
| Merge inactive bytes from src after gf2p8affineqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8affineqb_maskz (predicate< 16, Arch > k, simd< std::uint8_t, 16, Arch > a, simd< std::uint64_t, 2, Arch > matrix) noexcept |
| Zero inactive bytes after gf2p8affineqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8affineinvqb_mask (simd< std::uint8_t, 16, Arch > src, predicate< 16, Arch > k, simd< std::uint8_t, 16, Arch > a, simd< std::uint64_t, 2, Arch > matrix) noexcept |
| Merge inactive bytes from src after gf2p8affineinvqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 16, Arch > | native::gf2p8affineinvqb_maskz (predicate< 16, Arch > k, simd< std::uint8_t, 16, Arch > a, simd< std::uint64_t, 2, Arch > matrix) noexcept |
| Zero inactive bytes after gf2p8affineinvqb. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx)) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8mulb (simd< std::uint8_t, 32, Arch > a, simd< std::uint8_t, 32, Arch > b) noexcept |
| Multiply corresponding bytes in GF(2^8). | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8affineqb (simd< std::uint8_t, 32, Arch > a, simd< std::uint64_t, 4, Arch > matrix) noexcept |
| Apply the binary matrix in each 64-bit lane and XOR Imm8. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8affineinvqb (simd< std::uint8_t, 32, Arch > a, simd< std::uint64_t, 4, Arch > matrix) noexcept |
| Invert each field byte, apply its lane's binary matrix, and XOR Imm8. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8mulb_mask (simd< std::uint8_t, 32, Arch > src, predicate< 32, Arch > k, simd< std::uint8_t, 32, Arch > a, simd< std::uint8_t, 32, Arch > b) noexcept |
| Merge inactive bytes from src after gf2p8mulb. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl)) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8mulb_maskz (predicate< 32, Arch > k, simd< std::uint8_t, 32, Arch > a, simd< std::uint8_t, 32, Arch > b) noexcept |
| Zero inactive bytes after gf2p8mulb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8affineqb_mask (simd< std::uint8_t, 32, Arch > src, predicate< 32, Arch > k, simd< std::uint8_t, 32, Arch > a, simd< std::uint64_t, 4, Arch > matrix) noexcept |
| Merge inactive bytes from src after gf2p8affineqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8affineqb_maskz (predicate< 32, Arch > k, simd< std::uint8_t, 32, Arch > a, simd< std::uint64_t, 4, Arch > matrix) noexcept |
| Zero inactive bytes after gf2p8affineqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8affineinvqb_mask (simd< std::uint8_t, 32, Arch > src, predicate< 32, Arch > k, simd< std::uint8_t, 32, Arch > a, simd< std::uint64_t, 4, Arch > matrix) noexcept |
| Merge inactive bytes from src after gf2p8affineinvqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Arch.has(x86_feature::avx512vl) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 32, Arch > | native::gf2p8affineinvqb_maskz (predicate< 32, Arch > k, simd< std::uint8_t, 32, Arch > a, simd< std::uint64_t, 4, Arch > matrix) noexcept |
| Zero inactive bytes after gf2p8affineinvqb. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f)) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8mulb (simd< std::uint8_t, 64, Arch > a, simd< std::uint8_t, 64, Arch > b) noexcept |
| Multiply corresponding bytes in GF(2^8). | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8affineqb (simd< std::uint8_t, 64, Arch > a, simd< std::uint64_t, 8, Arch > matrix) noexcept |
| Apply the binary matrix in each 64-bit lane and XOR Imm8. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8affineinvqb (simd< std::uint8_t, 64, Arch > a, simd< std::uint64_t, 8, Arch > matrix) noexcept |
| Invert each field byte, apply its lane's binary matrix, and XOR Imm8. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw)) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8mulb_mask (simd< std::uint8_t, 64, Arch > src, predicate< 64, Arch > k, simd< std::uint8_t, 64, Arch > a, simd< std::uint8_t, 64, Arch > b) noexcept |
| Merge inactive bytes from src after gf2p8mulb. | |
|
template<isa< x86 > Arch> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw)) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8mulb_maskz (predicate< 64, Arch > k, simd< std::uint8_t, 64, Arch > a, simd< std::uint8_t, 64, Arch > b) noexcept |
| Zero inactive bytes after gf2p8mulb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8affineqb_mask (simd< std::uint8_t, 64, Arch > src, predicate< 64, Arch > k, simd< std::uint8_t, 64, Arch > a, simd< std::uint64_t, 8, Arch > matrix) noexcept |
| Merge inactive bytes from src after gf2p8affineqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8affineqb_maskz (predicate< 64, Arch > k, simd< std::uint8_t, 64, Arch > a, simd< std::uint64_t, 8, Arch > matrix) noexcept |
| Zero inactive bytes after gf2p8affineqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8affineinvqb_mask (simd< std::uint8_t, 64, Arch > src, predicate< 64, Arch > k, simd< std::uint8_t, 64, Arch > a, simd< std::uint64_t, 8, Arch > matrix) noexcept |
| Merge inactive bytes from src after gf2p8affineinvqb. | |
|
template<isa< x86 > Arch, unsigned Imm8> requires (Arch.has(x86_feature::gfni) && Arch.has(x86_feature::avx512f) && Arch.has(x86_feature::avx512bw) && Imm8 <= 255) | |
| constexpr simd< std::uint8_t, 64, Arch > | native::gf2p8affineinvqb_maskz (predicate< 64, Arch > k, simd< std::uint8_t, 64, Arch > a, simd< std::uint64_t, 8, Arch > matrix) noexcept |
| Zero inactive bytes after gf2p8affineinvqb. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::gf2p8mulb (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, unsigned Imm8, class... Args> | |
| void | native::gf2p8affineqb (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, unsigned Imm8, class... Args> | |
| void | native::gf2p8affineinvqb (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::gf2p8mulb_mask (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, class... Args> | |
| void | native::gf2p8mulb_maskz (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, unsigned Imm8, class... Args> | |
| void | native::gf2p8affineqb_mask (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, unsigned Imm8, class... Args> | |
| void | native::gf2p8affineqb_maskz (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, unsigned Imm8, class... Args> | |
| void | native::gf2p8affineinvqb_mask (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
| template<isa< x86 > Arch, unsigned Imm8, class... Args> | |
| void | native::gf2p8affineinvqb_maskz (Args...)=delete |
| Reject unsupported signatures, including implicit raw-register conversions. | |
Byte arithmetic modulo x^8 + x^4 + x^3 + x + 1 and binary affine maps. Arch records instruction requirements; callers must separately enable and admit a matching target. The vector lane count selects the overload and minimum target requirements.
Each matrix operand contains one 8 by 8 binary matrix per 64-bit lane. Output bit i uses matrix byte 7-i within that lane, with input bit j multiplying bit j of that byte. Imm8 is XORed into every result byte. Inverse-affine first takes the field inverse of each input byte (zero maps to zero), then applies the matrix and Imm8; it does not invert the matrix.
Mask bit i selects byte i. Merge forms retain src in inactive bytes; zero forms clear them. LLVM's byte-mask intrinsics require AVX512BW; narrower masked forms also require AVX512VL. All operations depend only on their register arguments and have no side effects. Constant evaluation uses exact integer semantics. Tags without the instruction features are accepted only at compile time and require complete SIMD storage.