9namespace ftz {
namespace detail {
namespace math {
10 struct atan2_result {
unsigned int bits, valid; };
11 struct atan2_value {
float radians;
unsigned int valid; };
13 template <
bool Hardware = FTZ_FP32_HARDWARE_FTZ != 0>
14 native_constexpr
inline atan2_result atan2_words(
unsigned int y,
unsigned int x) {
15 atan2_result result; result.bits=result.valid=0u;
16 if (!approx_finite(y) || !approx_finite(x))
return result;
17 y=approx_flush(y); x=approx_flush(x);
18 unsigned int ay=y&0x7fffffffu, ax=x&0x7fffffffu;
19 if ((ay|ax)==0u)
return result;
20 bool swap=ay>ax, negative_x=(x&0x80000000u)!=0u;
21 if (ay==0u || ax==0u) {
22 unsigned int magnitude=ax==0u ? 0x3fc90fdbu : (negative_x ? 0x40490fdbu : 0u);
23 result.bits=magnitude|(y&0x80000000u); result.valid=1u;
26 approx_result quotient=approx_divide(swap?ax:ay,swap?ay:ax);
27 if (quotient.valid==0u)
return result;
29 unsigned int ratio_bits=quotient.bits>0x3f800000u ? 0x3f800000u : quotient.bits;
30 bool tiny=ratio_bits<=0x39800000u;
32 unsigned int masked=Hardware ? ratio_bits : (tiny?0u:ratio_bits);
33 float ratio=fp32_decode(ratio_bits), t=fp32_decode(masked);
34 float z=fp32_mul<false>(t,t);
35 float h=fp32_decode(0x3b390ccdu);
36 h=fp32_fma<false>(h,z,fp32_decode(0xbc82b80du));
37 h=fp32_fma<false>(h,z,fp32_decode(0x3d2e19b6u));
38 h=fp32_fma<false>(h,z,fp32_decode(0xbd995ffau));
39 h=fp32_fma<false>(h,z,fp32_decode(0x3dd9ccf2u));
40 h=fp32_fma<false>(h,z,fp32_decode(0xbe116f9fu));
41 h=fp32_fma<false>(h,z,fp32_decode(0x3e4cb9a7u));
42 h=fp32_fma<false>(h,z,fp32_decode(0xbeaaaa5du));
43 float q=fp32_mul<false>(z,h);
44 float angle=fp32_fma<false>(q,t,ratio);
45 angle=fp32_decode(tiny?ratio_bits:fp32_encode(angle));
46 if (swap) angle=fp32_fma<false>(-1.0f,angle,fp32_decode(0x3fc90fdbu));
47 if (negative_x) angle=fp32_fma<false>(-1.0f,angle,fp32_decode(0x40490fdbu));
48 result.bits=fp32_encode(angle)|(y&0x80000000u); result.valid=1u;
51 template <
bool Hardware = FTZ_FP32_HARDWARE_FTZ != 0>
52 native_constexpr
inline atan2_value atan2_checked(
float y,
float x) {
53 atan2_result r=atan2_words<Hardware>(fp32_encode(y),fp32_encode(x));
54 atan2_value result;result.radians=fp32_decode(r.bits);result.valid=r.valid;
return result;
Reciprocal-refined division, square root and reciprocal square root.
FP32 bit casts, precise arithmetic and signed flush-to-zero normalization.