3#include "ftz/math/float.h"
4#include "ftz/math/exp_coefficients.h"
6namespace ftz {
namespace detail {
namespace math {
11 template <
unsigned int Degree = 6>
12 uint2 exp_value(
unsigned int bits) {
13 if ((bits & 0x7f800000u) == 0x7f800000u)
return uint2(0u, 0u);
14 precise
float x = asfloat(bits);
15 bool active = !(x < asfloat(0xc2aeac4fu));
16 bool overflow = x > asfloat(0x42b17217u);
17 precise
float product = x * asfloat(0x3fb8aa3bu);
18 precise
float n = round(product);
19 precise
float first = fp32_fma<false>(n, asfloat(0xbf317200u), x);
20 precise
float r = fp32_fma<false>(n, asfloat(0xb5bfbe8eu), first);
22 precise
float y = fp32_fma<false>(r, asfloat(exp_coefficients<Degree>::leading), asfloat(exp_coefficients<Degree>::next));
23 if (Degree >= 7) y = fp32_fma<false>(r, y, asfloat(exp_coefficients<Degree>::c5));
24 if (Degree >= 6) y = fp32_fma<false>(r, y, asfloat(exp_coefficients<Degree>::c4));
25 if (Degree >= 5) y = fp32_fma<false>(r, y, asfloat(exp_coefficients<Degree>::c3));
26 if (Degree >= 4) y = fp32_fma<false>(r, y, asfloat(exp_coefficients<Degree>::c2));
27 if (Degree >= 3) y = fp32_fma<false>(r, y, asfloat(exp_coefficients<Degree>::c1));
28 if (Degree >= 2) y = fp32_fma<false>(r, y, asfloat(exp_coefficients<Degree>::c0));
31 bool high = n > 127.0f;
32 precise
float biased = active && !overflow ? n + (high ? 126.0f : 127.0f) : 0.0f;
33 float factor = asfloat((
unsigned int)biased << 23);
34 precise
float first_scaled = y * factor;
35 precise
float high_scaled = (high ? first_scaled : 0.0f) * 2.0f;
36 precise
float scaled = high ? high_scaled : first_scaled;
37 precise
float value = overflow ? asfloat(0x7f800000u) : active ? scaled : 0.0f;
38 return uint2(asuint(value), 1u);
43namespace ftz {
namespace detail {
namespace math {
49 float expm1_core(
float input) {
50 precise
float x = max(input, -18.0f);
51 precise
float product = x * asfloat(0x3fb8aa3bu);
52 precise
float n = round(product);
53 precise
float r0 = fp32_fma<false>(n, asfloat(0xbf317200u), x);
54 precise
float r1 = fp32_fma<false>(n, asfloat(0xb5bfbe8eu), r0);
55 precise
float r = n == 0.0f ? x : r1;
56 bool tiny = (asuint(r) & 0x7fffffffu) <= 0x33000000u;
57#if FTZ_FP32_HARDWARE_FTZ
60 precise
float t = tiny ? 0.0f : r;
62 precise
float z = t * t;
63 precise
float h0 = asfloat(0x3493f27eu);
64 precise
float h1 = fp32_fma<false>(t, h0, asfloat(0x3638ef1du));
65 precise
float h2 = fp32_fma<false>(t, h1, asfloat(0x37d00d01u));
66 precise
float h3 = fp32_fma<false>(t, h2, asfloat(0x39500d01u));
67 precise
float h4 = fp32_fma<false>(t, h3, asfloat(0x3ab60b61u));
68 precise
float h5 = fp32_fma<false>(t, h4, asfloat(0x3c088889u));
69 precise
float h6 = fp32_fma<false>(t, h5, asfloat(0x3d2aaaabu));
70 precise
float h7 = fp32_fma<false>(t, h6, asfloat(0x3e2aaaabu));
71 precise
float h8 = fp32_fma<false>(t, h7, asfloat(0x3f000000u));
72 precise
float fused = fp32_fma<false>(z, h8, r);
73 precise
float p = tiny ? r : fused;
74 int index = (int)max(n, -24.0f);
75 precise
float scale = asfloat((
unsigned int)(index + 127) << 23u);
76 precise
float offset = scale - 1.0f;
77 precise
float scaled = fp32_fma<false>(scale, p, offset);
78 precise
float result = n == 0.0f ? p : scaled;
79 result = n == -25.0f ? (p > 0.0f ? asfloat(0xbf7fffffu) : -1.0f) : result;
80 result = n < -25.0f ? -1.0f : result;
82 return fp32_ftz(result);
84 expm1_result expm1_checked(
float input) {
85 unsigned int word = asuint(input), magnitude = word & 0x7fffffffu;
86 bool valid = magnitude < 0x7f800000u &&
87 ((word & 0x80000000u) != 0u || magnitude <= 0x3f800000u);
88 word = valid ? word : 0u;
89 word = (word & 0x7f800000u) == 0u ? (word & 0x80000000u) : word;
90 precise
float evaluated = expm1_core(asfloat(word));
92 result.value = valid ? evaluated : 0.0f;
93 result.valid = valid ? 1u : 0u;
96 expm1_result damping_gain_checked(
float input) {
97 unsigned int word = asuint(input), magnitude = word & 0x7fffffffu;
98 bool valid = magnitude < 0x7f800000u &&
99 ((word & 0x80000000u) == 0u || magnitude == 0u);
100 word = valid ? word : 0u;
101 word = (word & 0x7f800000u) == 0u ? (word & 0x80000000u) : word;
102 precise
float evaluated = expm1_core(asfloat(word ^ 0x80000000u));
104 result.value = asfloat(valid ? (asuint(evaluated) ^ 0x80000000u) : 0u);
105 result.valid = valid ? 1u : 0u;
116namespace ftz {
namespace detail {
namespace math {
117 struct sincos_result {
float sine, cosine; };
119 sincos_result trig_ftz_graph(
float v,
bool bounded) {
120 uint bits = asuint(v);
121 uint canonical = (bits & 0x7f800000u) == 0 ? bits & 0x80000000u : bits;
122 uint magnitude = canonical & 0x7fffffffu;
126 precise
float product = asfloat(magnitude) * asfloat(0x3fa2f983u);
127 index = ((uint)product + 1u) & 0xfffffffeu;
128 precise
float multiple = (float)index;
129 precise
float r0 = fp32_fma<false>(multiple, asfloat(0xbf490000u), asfloat(magnitude));
130 precise
float r1 = fp32_fma<false>(multiple, asfloat(0xb97da000u), r0);
131 x = fp32_fma<false>(multiple, asfloat(0xb3222169u), r1);
133 x = asfloat(canonical);
135 bool active = (asuint(x) & 0x7fffffffu) > 0x39800000u;
136#if FTZ_FP32_HARDWARE_FTZ
139 precise
float t = asfloat(active ? asuint(x) : 0u);
141 precise
float z = t * t;
142 precise
float s0 = fp32_fma<false>(asfloat(0xb94ca1f9u), z, asfloat(0x3c08839eu));
143 precise
float c0 = fp32_fma<false>(asfloat(0x37ccf5ceu), z, asfloat(0xbab6061au));
144 precise
float s1 = fp32_fma<false>(s0, z, asfloat(0xbe2aaaa3u));
145 precise
float c1 = fp32_fma<false>(c0, z, asfloat(0x3d2aaaa5u));
146 precise
float s2 = s1 * z;
147 precise
float c2 = c1 * z;
148 precise
float c3 = c2 * z;
149 precise
float half_z = z * 0.5f;
150 precise
float c4 = c3 - half_z;
151 precise
float s3 = fp32_fma<false>(s2, t, x);
152 precise
float c5 = c4 + 1.0f;
153 uint sine = active ? asuint(s3) : asuint(x);
154 uint cosine = asuint(c5);
155 sincos_result result;
157 bool swap_pair = (index & 2u) != 0;
158 result.sine = asfloat((swap_pair ? cosine : sine) ^
159 (canonical & 0x80000000u) ^ ((index & 4u) << 29));
160 result.cosine = asfloat((swap_pair ? sine : cosine) ^
161 ((~(index - 2u) & 4u) << 29));
163 result.sine = asfloat(sine);
164 result.cosine = asfloat(cosine);
170 sincos_result sincos_reduced_ftz(
float x) {
return trig_ftz_graph(x,
false); }
172 sincos_result sincos_ftz(
float x) {
return trig_ftz_graph(x,
true); }
173 float sin_reduced_ftz(
float x) {
return sincos_reduced_ftz(x).sine; }
174 float cos_reduced_ftz(
float x) {
return sincos_reduced_ftz(x).cosine; }
175 float sin_ftz(
float x) {
return sincos_ftz(x).sine; }
176 float cos_ftz(
float x) {
return sincos_ftz(x).cosine; }