ftz 0.0.1
Fast, reproducible floating-point arithmetic
Loading...
Searching...
No Matches
float.h
Go to the documentation of this file.
1#pragma once
2#include "ftz/config.h"
3#include "native/attributes.h"
4#ifdef __cplusplus
5#include <bit>
6#include <cmath>
7#include <native/detail/constexpr_float.h>
8#endif
9
10namespace ftz { namespace detail { namespace math {
11 [[nodiscard]] native_constexpr native_inline native_const float fp32_decode(unsigned int bits) {
12#ifdef __cplusplus
13 return std::bit_cast<float>(bits);
14#else
15 return asfloat(bits);
16#endif
17 }
18 [[nodiscard]] native_constexpr native_inline native_const unsigned int fp32_encode(float value) {
19#ifdef __cplusplus
20 return std::bit_cast<unsigned int>(value);
21#else
22 return asuint(value);
23#endif
24 }
25 // The hardware policy applies to the whole graph. It skips explicit flushing
26 // only after the caller qualifies the FP environment; it sets no FP controls.
27 template <bool Hardware = FTZ_FP32_HARDWARE_FTZ != 0>
28 native_constexpr inline float fp32_ftz(float value) {
29#ifdef __cplusplus
30 if !consteval { if (Hardware) return value; }
31#else
32 if (Hardware) return value;
33#endif
34 unsigned int bits = fp32_encode(value);
35 return fp32_decode((bits & 0x7f800000u) == 0u ? bits & 0x80000000u : bits);
36 }
37 // Flush=false is for graphs that already bound their intermediates, or
38 // perform their own final repair. It must not add per-operation FTZ work.
39 template <bool Flush = true, bool Hardware = FTZ_FP32_HARDWARE_FTZ != 0>
40 [[nodiscard]] native_constexpr native_inline float fp32_add(float a, float b) {
41#ifdef __cplusplus
42 if consteval {
43 namespace fp = ::native::detail::constexpr_float;
44 auto const bits = fp::add_bits<fp::binary32>(fp32_encode(a), fp32_encode(b));
45 float result = fp32_decode(bits);
46 return Flush ? fp32_ftz<false>(result) : result;
47 }
48 float result = a + b;
49#else
50 precise float result = a + b;
51#endif
52 return Flush ? fp32_ftz<Hardware>(result) : result;
53 }
54 template <bool Flush = true, bool Hardware = FTZ_FP32_HARDWARE_FTZ != 0>
55 [[nodiscard]] native_constexpr native_inline float fp32_mul(float a, float b) {
56#ifdef __cplusplus
57 if consteval {
58 namespace fp = ::native::detail::constexpr_float;
59 auto const bits = fp::mul_bits<fp::binary32>(fp32_encode(a), fp32_encode(b));
60 float result = fp32_decode(bits);
61 return Flush ? fp32_ftz<false>(result) : result;
62 }
63 float result = a * b;
64#else
65 precise float result = a * b;
66#endif
67 return Flush ? fp32_ftz<Hardware>(result) : result;
68 }
69 template <bool Flush = true, bool Hardware = FTZ_FP32_HARDWARE_FTZ != 0>
70 [[nodiscard]] native_constexpr native_inline float fp32_fma(float a, float b, float c) {
71#ifdef __cplusplus
72 if consteval {
73 namespace fp = ::native::detail::constexpr_float;
74 auto const bits = fp::fma_bits<fp::binary32>(fp32_encode(a), fp32_encode(b), fp32_encode(c));
75 float result = fp32_decode(bits);
76 return Flush ? fp32_ftz<false>(result) : result;
77 }
78 float result = std::fma(a, b, c);
79#else
80 precise float result = mad(a, b, c);
81#endif
82 return Flush ? fp32_ftz<Hardware>(result) : result;
83 }
84#ifndef __cplusplus
85 native_constexpr inline float2 fp32_ftz(float2 value) {
86#if FTZ_FP32_HARDWARE_FTZ
87 return value;
88#else
89 uint2 bits = asuint(value);
90 uint2 nonzero = uint2((bits & 0x7f800000u) != 0u);
91 return asfloat(bits & ((0u - nonzero) | 0x80000000u));
92#endif
93 }
94 native_constexpr inline float3 fp32_ftz(float3 value) {
95#if FTZ_FP32_HARDWARE_FTZ
96 return value;
97#else
98 uint3 bits = asuint(value);
99 uint3 nonzero = uint3((bits & 0x7f800000u) != 0u);
100 return asfloat(bits & ((0u - nonzero) | 0x80000000u));
101#endif
102 }
103 native_constexpr inline float4 fp32_ftz(float4 value) {
104#if FTZ_FP32_HARDWARE_FTZ
105 return value;
106#else
107 uint4 bits = asuint(value);
108 uint4 nonzero = uint4((bits & 0x7f800000u) != 0u);
109 return asfloat(bits & ((0u - nonzero) | 0x80000000u));
110#endif
111 }
112#endif
113}}}
114