native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
bf16_constexpr.h
1// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
2#pragma once
3#include "native/detail/constexpr_float.h"
4
5namespace native::detail {
6 // BFDOT scalar encoding semantics with FPCR.EBF=AH=0. Each product, pair
7 // sum and accumulator sum is a separate round-to-odd operation. Instruction
8 // rules force signed subnormal flushing, positive default NaN and infinity
9 // on overflow; the caller's rounding and flush controls do not participate.
10 constexpr std::uint32_t arm_bfdot_bits(std::uint32_t accumulator,
11 std::uint16_t a0,std::uint16_t a1,std::uint16_t b0,std::uint16_t b1) noexcept {
12 namespace cf=constexpr_float;
13 constexpr cf::policy p{.nan=cf::nan_propagation::default_nan,
14 .flush_inputs=true,.flush_outputs=true,.odd_overflow_infinity=true};
15 constexpr auto r=cf::rounding::to_odd;
16 auto x=cf::mul_bits<cf::binary32>(std::uint32_t(a0)<<16,std::uint32_t(b0)<<16,r,p);
17 auto y=cf::mul_bits<cf::binary32>(std::uint32_t(a1)<<16,std::uint32_t(b1)<<16,r,p);
18 return cf::add_bits<cf::binary32>(accumulator,cf::add_bits<cf::binary32>(x,y,r,p),r,p);
19 }
20}