|
native 0.0.1
Vectors, masks and wide register packs for C++26
|
Functions | |
|
template<isa< arm > Arch> requires (Arch.has(arm_feature::neon_bf16)) | |
| constexpr simd< float, 2, Arch > | native::bfdot (simd< float, 2, Arch > acc, simd< bf16, 4, Arch > a, simd< bf16, 4, Arch > b) noexcept |
| Accumulate each adjacent pair of BF16 products into the corresponding FP32 lane. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 2) | |
| constexpr simd< float, 2, Arch > | native::bfdot_lane (simd< float, 2, Arch > acc, simd< bf16, 4, Arch > a, simd< bf16, 4, Arch > b) noexcept |
| BFDOT with the BF16 pair b[2*Lane], b[2*Lane+1] shared by all output lanes. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 4) | |
| constexpr simd< float, 2, Arch > | native::bfdot_lane (simd< float, 2, Arch > acc, simd< bf16, 4, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| BFDOT with the BF16 pair b[2*Lane], b[2*Lane+1] shared by all output lanes. | |
|
template<isa< arm > Arch> requires (Arch.has(arm_feature::neon_bf16)) | |
| constexpr simd< float, 4, Arch > | native::bfdot (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| Accumulate each adjacent pair of BF16 products into the corresponding FP32 lane. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 2) | |
| constexpr simd< float, 4, Arch > | native::bfdot_lane (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 4, Arch > b) noexcept |
| BFDOT with the BF16 pair b[2*Lane], b[2*Lane+1] shared by all output lanes. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 4) | |
| constexpr simd< float, 4, Arch > | native::bfdot_lane (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| BFDOT with the BF16 pair b[2*Lane], b[2*Lane+1] shared by all output lanes. | |
|
template<isa< arm > Arch> requires (Arch.has(arm_feature::neon_bf16)) | |
| constexpr simd< float, 4, Arch > | native::bfmmla (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| Accumulate a row-major 2x4 matrix times a column-major 4x2 matrix, two BFDOT steps per result. | |
|
template<isa< arm > Arch> requires (Arch.has(arm_feature::neon_bf16)) | |
| constexpr simd< float, 4, Arch > | native::bfmlalb (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| Fused multiply-add of the even BF16 lanes into the corresponding FP32 lanes. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 4) | |
| constexpr simd< float, 4, Arch > | native::bfmlalb_lane (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 4, Arch > b) noexcept |
| Fused multiply-add of the even lanes of a with b[Lane] shared by all output lanes. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 8) | |
| constexpr simd< float, 4, Arch > | native::bfmlalb_lane (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| Fused multiply-add of the even lanes of a with b[Lane] shared by all output lanes. | |
|
template<isa< arm > Arch> requires (Arch.has(arm_feature::neon_bf16)) | |
| constexpr simd< float, 4, Arch > | native::bfmlalt (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| Fused multiply-add of the odd BF16 lanes into the corresponding FP32 lanes. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 4) | |
| constexpr simd< float, 4, Arch > | native::bfmlalt_lane (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 4, Arch > b) noexcept |
| Fused multiply-add of the odd lanes of a with b[Lane] shared by all output lanes. | |
|
template<isa< arm > Arch, unsigned Lane> requires (Arch.has(arm_feature::neon_bf16) && Lane < 8) | |
| constexpr simd< float, 4, Arch > | native::bfmlalt_lane (simd< float, 4, Arch > acc, simd< bf16, 8, Arch > a, simd< bf16, 8, Arch > b) noexcept |
| Fused multiply-add of the odd lanes of a with b[Lane] shared by all output lanes. | |
Runtime calls require arm_feature::neon_bf16 and a compatible bf16 compiler target. BFDOT/BFMMLA use round-to-odd steps by default, or fused pairs followed by separate accumulation when FEAT_EBF16 and FPCR.EBF enable enhanced behavior. They return default NaNs and leave FPSR unchanged in either mode. BFMLALB/T instead perform single fused binary32 multiply-adds. With AH=0 they use ordinary FP32 controls and accumulate exception flags. With AFP and AH=1 they force RNE and input/output flushing, suppressing exceptions. All operations read the caller's FPCR without changing it. Constant evaluation uses RNE, gradual inputs/results, DN=AH=AHP=FZ=FZ16=FIZ=EBF=0, with masked exceptions and no status effects; fixed instruction rules still apply. Missing-feature overloads are consteval-only and require complete storage types.