ftz 0.0.1
Fast, reproducible floating-point arithmetic
Loading...
Searching...
No Matches
simd.h
1// SPDX-FileCopyrightText: 2026 Edward Kmett <ekmett@gmail.com>
2// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
3#pragma once
4// Included below the FTZ module declaration; all SIMD operations are dependent.
5
6namespace ftz::detail {
7 // An emulation-only ISA has no register operations to prefer. Reuse the
8 // scalar semantic graph; adding polyfill to a native ISA keeps its fast path.
9 template<class V> inline constexpr bool ftz32_emulated = [] {
10 if constexpr (requires { V::architecture; })
11 return V::architecture == ::native::isa<>(::native::polyfill);
12 else return false;
13 }();
14 using ::native::mask_bits;
15 template <class V> using ftz32_bridge = native::detail::fp32_bit_bridge<V>;
16 template <class V> using ftz32_words = typename V::bits_type;
17 template <auto Operation, class V, class... X>
18 [[nodiscard]] constexpr V ftz32_constant(V input, X... rest) noexcept {
19 using B = ftz32_bridge<V>; using U = ftz32_words<V>;
20 std::array<std::array<std::uint32_t,V::lanes>,1+sizeof...(X)> words{};
21 B::encode(input).store(words[0].data());
22 std::size_t index=1;
23 (B::encode(rest).store(words[index++].data()),...);
24 std::array<std::uint32_t,V::lanes> result{};
25 for(std::size_t lane=0;lane<V::lanes;++lane)
26 result[lane]=[&]<std::size_t... I>(std::index_sequence<I...>) {
27 return Operation(words[I][lane]...);
28 }(std::make_index_sequence<1+sizeof...(X)>{});
29 return B::decode(U::load(result.data()));
30 }
31 template <class U> [[nodiscard]] native_inline constexpr bool ftz32_any(U mask) noexcept {
32 return any(mask != U(0));
33 }
34 template <class U> [[nodiscard]] native_inline constexpr native_const U ftz32_import_words(U bits) noexcept {
35 auto magnitude = bits & U(0x7fffffffu);
36 auto tiny = U(0x00800000u) > magnitude;
37 return select(tiny, bits & U(0x80000000u), bits);
38 }
39 template <class V> [[nodiscard]] native_inline constexpr native_const ftz32_words<V> ftz32_nonfinite(V v) noexcept {
40 using U = ftz32_words<V>;
41 return mask_bits<std::uint32_t>(U(ftz32_infinity) > (ftz32_bridge<V>::encode(v) & U(0x7fffffffu))) ^ U(0xffffffffu);
42 }
43 template <class V>
44 [[nodiscard]] native_inline constexpr native_const ftz32_words<V> ftz32_repair_mask(V result) noexcept {
45 using U = ftz32_words<V>;
46 auto boundary_mask = mask_bits<std::uint32_t>((ftz32_bridge<V>::encode(result) & U(0x7fffffffu)) > U(0x00800000u)) ^ U(0xffffffffu);
47 return boundary_mask;
48 }
49 template <auto Repair, class A, std::size_t... I>
50 [[nodiscard]] native_inline constexpr native_pure std::uint32_t ftz32_repair_lane(A const & inputs, std::size_t lane, std::index_sequence<I...>) noexcept {
51 return Repair(inputs[I][lane]...);
52 }
53 // No lane extraction on the normal path. Only flagged lanes call the shared
54 // scalar contract; the operation and backend are compile-time selections.
55 template <auto Repair, class V, class... X>
56 [[nodiscard]] native_inline constexpr native_const V ftz32_repair(V result, ftz32_words<V> mask, X... operands) noexcept {
57 if (!ftz32_any(mask)) return result;
58 using B = ftz32_bridge<V>;
59 using U = ftz32_words<V>;
60 std::array<std::uint32_t, V::lanes> flags, output;
61 std::array<std::array<std::uint32_t, V::lanes>, sizeof...(X)> inputs;
62 mask.store(flags.data()); B::encode(result).store(output.data());
63 std::size_t index = 0;
64 (B::encode(operands).store(inputs[index++].data()), ...);
65 for (std::size_t lane = 0; lane < V::lanes; ++lane)
66 if (flags[lane] != 0)
67 output[lane] = ftz32_repair_lane<Repair>(inputs, lane, std::index_sequence_for<X...>{});
68 return B::decode(U::load(output.data()));
69 }
70 // General FTZ scaling owns its semantic graph: raw native scaling exists
71 // only where the target has an instruction. Normal bases need no arithmetic
72 // on their significands, including the sole tiny result that rounds to normal.
73 template <class V>
74 [[nodiscard]] native_inline constexpr native_const V ftz32_vector_scaleb(V value, V exponent) noexcept {
75 using B = ftz32_bridge<V>; using U = ftz32_words<V>;
76 using I = typename V::template rebind<std::int32_t>;
77 U x = B::encode(value), y = ftz32_import_words(B::encode(exponent));
78 U magnitude = x & U(0x7fffffffu), shift_magnitude = y & U(0x7fffffffu);
79 U sign = x & U(0x80000000u), fraction = x & U(0x007fffffu);
80 // +/-256 settles every finite FTZ base. Clamp by word magnitude before
81 // floor/conversion, so even infinities and NaNs have a representable integer
82 // intermediate. Their value semantics are selected independently below.
83 U bounded = select(shift_magnitude < U(0x43800000u), y,
84 (y & U(0x80000000u)) | U(0x43800000u));
85 I shift = convert<std::int32_t>(floor(B::decode(bounded)));
86 U biased = magnitude.template right<23>();
87 I adjusted = I::from_native(__builtin_bit_cast(typename I::native_type, biased.to_native())) + shift;
88 U fields = U::from_native(__builtin_bit_cast(typename U::native_type, adjusted.to_native()));
89 U result = select((adjusted > I(0)) & (adjusted < I(255)),
90 fields.template left<23>() | fraction,
91 select(adjusted > I(254), U(0x7f800000u), U(0)));
92 // A maximum significand at adjusted exponent zero is exactly halfway
93 // below minimum normal; nearest-even rounds upward before signed flushing.
94 result = select((adjusted == I(0)) & (fraction == U(0x007fffffu)), U(0x00800000u), result) | sign;
95 result = select((magnitude == U(0)) | (magnitude == U(0x7f800000u)), x, result);
96 result = select(magnitude > U(0x7f800000u), x | U(0x00400000u), result);
97 U upward = select(magnitude == U(0), U(0x7fc00000u), sign | U(0x7f800000u));
98 U downward = select(magnitude == U(0x7f800000u), U(0x7fc00000u), sign);
99 // Infinite exponents treat every NaN base as quiet: +infinity gives
100 // +infinity, -infinity gives +zero, independent of NaN sign/payload.
101 upward = select(magnitude > U(0x7f800000u), U(0x7f800000u), upward);
102 downward = select(magnitude > U(0x7f800000u), U(0), downward);
103 result = select(y == U(0x7f800000u), upward, result);
104 result = select(y == U(0xff800000u), downward, result);
105 return B::decode(select(shift_magnitude > U(0x7f800000u), y | U(0x00400000u), result));
106 }
107 template <bool Hardware, class V> [[nodiscard]] native_inline constexpr native_const V ftz32_vector_add(V a, V b) noexcept {
108 if consteval { return ftz32_constant<ftz32_add<Hardware>>(a, b); }
109 if constexpr (ftz32_emulated<V>) return ftz32_constant<ftz32_add<Hardware>>(a, b);
110 V r = a + b;
111 if constexpr(Hardware) return r;
112 else return ftz32_repair<ftz32_add<Hardware>>(r, ftz32_repair_mask(r), a, b);
113 }
114 template <bool Hardware, class V> [[nodiscard]] native_inline constexpr native_const V ftz32_vector_sub(V a, V b) noexcept {
115 if consteval { return ftz32_constant<ftz32_sub<Hardware>>(a, b); }
116 if constexpr (ftz32_emulated<V>) return ftz32_constant<ftz32_sub<Hardware>>(a, b);
117 V r = a - b;
118 if constexpr(Hardware) return r;
119 else return ftz32_repair<ftz32_sub<Hardware>>(r, ftz32_repair_mask(r), a, b);
120 }
121 template <class V> [[nodiscard]] native_inline constexpr native_const V ftz32_vector_mul(V a, V b) noexcept {
122 if consteval { return ftz32_constant<ftz32_mul>(a, b); }
123 if constexpr (ftz32_emulated<V>) return ftz32_constant<ftz32_mul>(a, b);
124 V r = a * b; return ftz32_repair<ftz32_mul>(r, ftz32_repair_mask(r), a, b);
125 }
126 template <class V> [[nodiscard]] native_inline constexpr native_const V ftz32_vector_fma(V a, V b, V c) noexcept {
127 if consteval { return ftz32_constant<ftz32_fma>(a, b, c); }
128 if constexpr (ftz32_emulated<V>) return ftz32_constant<ftz32_fma>(a, b, c);
129 V r = fma(a, b, c); return ftz32_repair<ftz32_fma>(r, ftz32_repair_mask(r), a, b, c);
130 }
131 // Same normalized three-refinement graph as ftz/math/approx.h.
132 // Every mantissa operation stays normal. Integer exponent scaling handles
133 // ordinary results; unusual scales and IEEE-like special values use the core.
134 template <bool Hardware, class V> [[nodiscard]] native_inline constexpr native_const V ftz32_vector_div(V a, V b) noexcept {
135 if consteval { return ftz32_constant<ftz32_div<Hardware>>(a, b); }
136 if constexpr (ftz32_emulated<V>) return ftz32_constant<ftz32_div<Hardware>>(a, b);
137 using B = ftz32_bridge<V>; using U = ftz32_words<V>;
138 U aw = B::encode(a), bw = B::encode(b);
139 U aa = aw & U(0x7fffffffu), bb = bw & U(0x7fffffffu);
140 U ma = U(0x3f800000u) | (aa & U(0x007fffffu));
141 U mb = U(0x3f800000u) | (bb & U(0x007fffffu));
142 V m = B::decode(mb), r = B::decode(U(0x7ef311c3u) - mb);
143 V e = fma(-m, r, V(1)); r = fma(r, e, r);
144 e = fma(-m, r, V(1)); r = fma(r, e, r);
145 e = fma(-m, r, V(1)); r = fma(r, e, r);
146 r = B::decode(ma) * r;
147 U rw = B::encode(r);
148 U exponent = rw.template right<23>() + aa.template right<23>() - bb.template right<23>();
149 U normal = mask_bits<std::uint32_t>((exponent > U(0)) & (U(255) > exponent));
150 U bits = ((aw ^ bw) & U(0x80000000u)) | exponent.template left<23>() | (rw & U(0x007fffffu));
151 U exceptional = (normal ^ U(0xffffffffu)) | mask_bits<std::uint32_t>((aa == U(0)) | (bb == U(0))) |
152 ftz32_nonfinite(a) | ftz32_nonfinite(b);
153 return ftz32_repair<ftz32_div<Hardware>>(B::decode(bits), exceptional, a, b);
154 }
155 template <class V> [[nodiscard]] native_inline constexpr native_const V ftz32_vector_sqrt(V a) noexcept {
156 if consteval { return ftz32_constant<ftz32_sqrt>(a); }
157 if constexpr (ftz32_emulated<V>) return ftz32_constant<ftz32_sqrt>(a);
158 using B = ftz32_bridge<V>; using U = ftz32_words<V>;
159 U aw = B::encode(a), aa = aw & U(0x7fffffffu), exponent_a = aa.template right<23>();
160 U parity = U(1) - (exponent_a & U(1));
161 U mword = (U(0x3f800000u) | (aa & U(0x007fffffu))) + parity.template left<23>();
162 V m = B::decode(mword), r = B::decode(U(0x5f375a86u) - mword.template right<1>());
163 V p = m * r, e = fma(-p, r, V(1)), h = V(0.5f) * r; r = fma(h, e, r);
164 p = m * r; e = fma(-p, r, V(1)); h = V(0.5f) * r; r = fma(h, e, r);
165 p = m * r; e = fma(-p, r, V(1)); h = V(0.5f) * r; r = fma(h, e, r);
166 r = m * r;
167 U rw = B::encode(r);
168 U exponent = rw.template right<23>() + (exponent_a + U(127) - parity).template right<1>() - U(127);
169 U bits = exponent.template left<23>() | (rw & U(0x007fffffu));
170 U exceptional = mask_bits<std::uint32_t>((aa == U(0)) | ((aw & U(0x80000000u)) > U(0))) | ftz32_nonfinite(a);
171 return ftz32_repair<ftz32_sqrt>(B::decode(bits), exceptional, a);
172 }
173
174 template <class V> concept raw_register = requires { typename V::value_type; typename V::bits_type; V::lanes; } && std::same_as<typename V::value_type,float>;
175 template <class R> concept ftz32_vector = requires { typename R::value_type; typename R::register_type; } && ftz32_type<typename R::value_type>;
176 template <class V, class F = ftz32> using ftz32_simd = typename V::template rebind<F>;
177}
178
203export namespace native {
206 template<bool Hardware> struct mask_traits<::ftz::basic_ftz32<Hardware>> { using type = bool; };
209 template <bool Hardware> struct simd_traits<::ftz::basic_ftz32<Hardware>> { using storage_type = float; };
212 template <bool Hardware, class V, class Self>
213 struct simd_customization<::ftz::basic_ftz32<Hardware>,V,Self> {
214 using simd = Self;
215 using ftz32 = ::ftz::basic_ftz32<Hardware>;
216 static constexpr bool hardware = Hardware;
217 static constexpr std::size_t N = V::lanes;
218 using register_type = V;
219 using value_type = ftz32;
220 using native_type = typename V::native_type;
221 using bits_type = ::ftz::detail::ftz32_words<V>;
222 using mask_type = typename V::mask_type;
223 using mask = mask_type;
224 using vector_mask_type = typename V::vector_mask_type;
225 static constexpr std::size_t lanes = V::lanes;
227 native_inline constexpr simd_customization() noexcept : value_(0.0f) {}
229 native_inline constexpr simd_customization(float value) noexcept : simd_customization(V(value)) {}
231 native_inline constexpr simd_customization(ftz32 value) noexcept : value_(value.to_float()) {}
233 native_inline constexpr simd_customization(V value) noexcept
234 : value_(import(value)) {}
235
236 [[nodiscard]] native_inline constexpr operator V() const noexcept { return value_; }
238 native_inline constexpr simd_customization(native_type value) noexcept requires (N>1) : simd_customization(V(value)) {}
240 [[nodiscard]] native_inline constexpr operator native_type() const noexcept { return value_.value; }
242 template <class... X> requires (sizeof...(X)==N && N>1) && (std::convertible_to<X,ftz32> && ...)
243 native_inline constexpr simd_customization(X... x) noexcept((noexcept(static_cast<float>(x)) && ...))
244 : simd_customization(V(static_cast<float>(x)...)) {}
245
246 template <class T> requires (std::same_as<T,float> || std::same_as<T,ftz32>)
247 native_inline constexpr simd_customization(std::array<T,N> const & values) noexcept : value_(load_memory(values.data()).to_native()) {}
248
250 [[nodiscard]] native_inline constexpr native_pure V to_native() const noexcept { return value_; }
252 [[nodiscard]] native_inline constexpr native_pure bits_type to_bits() const noexcept { return bits(); }
254 [[nodiscard]] static native_inline constexpr native_const simd from_float(float value) noexcept { return value; }
256 [[nodiscard]] static native_inline constexpr native_const simd from_native(V value) noexcept { return value; }
257 // Caller promises normal, signed zero, infinity or any NaN lanes.
258 // No classification or normalization; this is an explicit invariant escape.
260 [[nodiscard]] static native_inline constexpr native_const simd unsafe_from_float32(V value) noexcept {
261 return {canonical{}, value};
262 }
263
264 [[nodiscard]] static native_inline constexpr native_const simd unsafe_from_float32(float value) noexcept {
265 return {canonical{}, V(value)};
266 }
267
268 [[nodiscard]] native_inline constexpr native_pure bits_type bits() const noexcept { return bridge::encode(value_); }
270 [[nodiscard]] static native_inline constexpr native_const simd from_bits(bits_type value) noexcept {
271 return simd(bridge::decode(value));
272 }
273
274 [[nodiscard]] static native_inline constexpr native_const simd from_bits(std::uint32_t value) noexcept {
275 return from_bits(bits_type(value));
276 }
277
279 template <std::size_t Alignment = 1, class T> requires (std::same_as<T,float> || std::same_as<T,ftz32>)
280 [[nodiscard]] static native_inline constexpr simd load_memory(T const * p) noexcept {
281 if constexpr (std::same_as<T,float>) return simd(V::template load_memory<Alignment>(p));
282 else {
283 std::array<std::uint32_t,N> words;
284 if consteval {
285 for(std::size_t lane=0;lane<N;++lane) words[lane]=p[lane].to_bits();
286 } else { std::memcpy(words.data(),p,sizeof(words)); }
287 return unsafe_from_float32(V::from_bits(bits_type::load(words.data())));
288 }
289 }
290
291 template <std::size_t Alignment = 1, class T> requires (std::same_as<T,float> || std::same_as<T,ftz32>)
292 native_inline constexpr void store_memory(T * p) const noexcept {
293 if constexpr (std::same_as<T,float>) value_.template store_memory<Alignment>(p);
294 else {
295 std::array<std::uint32_t,N> words; bits().store(words.data());
296 if consteval {
297 for(std::size_t lane=0;lane<N;++lane)
298 p[lane]=ftz32::unsafe_from_float32(std::bit_cast<float>(words[lane]));
299 } else { std::memcpy(p,words.data(),sizeof(words)); }
300 }
301 }
302 // Legacy pointer spellings forward to the safe unaligned memory helpers.
304 [[nodiscard]] static native_inline constexpr native_pure simd load(float const * p) noexcept {
305 return load_memory(p);
306 }
307
308 [[nodiscard]] static native_inline constexpr native_pure simd loadu(float const * p) noexcept { return load_memory(p); }
310 native_inline constexpr void store(float * p) const noexcept { store_memory(p); }
312 native_inline constexpr void storeu(float * p) const noexcept { store_memory(p); }
315 [[nodiscard]] static native_inline constexpr native_pure simd load_partial(float const * p, std::size_t n, float fill = 0) noexcept {
316 std::array<float, lanes> temporary; temporary.fill(fill);
317 for (std::size_t i=0;i<n;++i) temporary[i]=p[i];
318 return loadu(temporary.data());
319 }
320
321 native_inline constexpr void store_partial(float * p, std::size_t n) const noexcept {
322 std::array<float, lanes> temporary; storeu(temporary.data());
323 for (std::size_t i=0;i<n;++i) p[i]=temporary[i];
324 }
325
326 [[nodiscard]] static native_inline constexpr native_pure simd load_bits(std::uint32_t const * p) noexcept {
327 return from_bits(bits_type::load(p));
328 }
329
330 native_inline constexpr void store_bits(std::uint32_t * p) const noexcept { bits().store(p); }
333 [[nodiscard]] static native_inline constexpr native_pure simd load_bits_partial(
334 std::uint32_t const * p, std::size_t n, std::uint32_t fill = 0) noexcept {
335 return from_bits(bits_type::load_partial(p, n, fill));
336 }
337
338 native_inline constexpr void store_bits_partial(std::uint32_t * p, std::size_t n) const noexcept {
339 bits().store_partial(p, n);
340 }
341
343 [[nodiscard]] friend native_inline constexpr native_const simd operator+(simd a, simd b) noexcept {
344 return {canonical{}, ::ftz::detail::ftz32_vector_add<Hardware>(a.value_, b.value_)};
345 }
346
348 [[nodiscard]] friend native_inline constexpr native_const simd operator-(simd a, simd b) noexcept {
349 return {canonical{}, ::ftz::detail::ftz32_vector_sub<Hardware>(a.value_, b.value_)};
350 }
351
353 [[nodiscard]] friend native_inline constexpr native_const simd operator*(simd a, simd b) noexcept {
354 return {canonical{}, ::ftz::detail::ftz32_vector_mul(a.value_, b.value_)};
355 }
356
358 [[nodiscard]] friend native_inline constexpr native_const simd operator/(simd a, simd b) noexcept {
359 return {canonical{}, ::ftz::detail::ftz32_vector_div<Hardware>(a.value_, b.value_)};
360 }
361
362 native_inline constexpr simd & operator+=(simd b) noexcept { return static_cast<Self &>(*this) = static_cast<Self &>(*this) + b; }
364 native_inline constexpr simd & operator-=(simd b) noexcept { return static_cast<Self &>(*this) = static_cast<Self &>(*this) - b; }
366 native_inline constexpr simd & operator*=(simd b) noexcept { return static_cast<Self &>(*this) = static_cast<Self &>(*this) * b; }
368 native_inline constexpr simd & operator/=(simd b) noexcept { return static_cast<Self &>(*this) = static_cast<Self &>(*this) / b; }
370 [[nodiscard]] friend native_inline constexpr native_const simd operator-(simd a) noexcept {
371 return {canonical{}, bridge::decode(a.bits() ^ bits_type(0x80000000u))};
372 }
373
374 [[nodiscard]] friend native_inline constexpr native_const simd operator+(simd a) noexcept { return a; }
376 [[nodiscard]] friend native_inline constexpr native_const simd abs(simd a) noexcept {
377 return {canonical{}, bridge::decode(a.bits() & bits_type(0x7fffffffu))};
378 }
379
381 [[nodiscard]] friend native_inline constexpr native_const mask_type operator<(simd a, simd b) noexcept { return a.value_ < b.value_; }
384 [[nodiscard]] friend native_inline constexpr native_const mask_type operator>(simd a, simd b) noexcept { return a.value_ > b.value_; }
387 [[nodiscard]] friend native_inline constexpr native_const mask_type operator==(simd a, simd b) noexcept { return a.value_ == b.value_; }
390 [[nodiscard]] friend native_inline constexpr native_const mask_type operator!=(simd a, simd b) noexcept { return ~(a == b); }
393 [[nodiscard]] friend native_inline constexpr native_const mask_type operator<=(simd a, simd b) noexcept { return (a < b) | (a == b); }
396 [[nodiscard]] friend native_inline constexpr native_const mask_type operator>=(simd a, simd b) noexcept { return (a > b) | (a == b); }
398 template<class M> requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>)
399 [[nodiscard]] friend native_inline constexpr native_const simd select(M mask,simd a,simd b) noexcept {
400 // Both mask domains select whole lanes, preserving chosen NaN payloads.
401 return {canonical{},select(mask,a.value_,b.value_)};
402 }
403
405 template <class M, class A, class B, class E>
406 requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>) &&
407 (std::same_as<A,simd> || std::same_as<B,simd>) &&
408 std::convertible_to<A,simd> && std::convertible_to<B,simd> &&
409 (std::same_as<E,simd> || std::convertible_to<E,V>)
410 [[nodiscard]] friend native_inline constexpr simd masked_scaleb(
411 M active, A prior, B value, E exponent)
412 noexcept(noexcept(simd(prior)) && noexcept(simd(value)) &&
413 noexcept(scaling_exponent(exponent))) {
414 simd imported_prior(prior), imported_value(value);
415 V shift = scaling_exponent(exponent);
416 V result = ::ftz::detail::ftz32_vector_scaleb(imported_value.value_, shift);
417 return {canonical{},select(active,result,imported_prior.value_)};
418 }
419
420 template <class M, class A, class E>
421 requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>) &&
422 std::same_as<A,simd> && (std::same_as<E,simd> || std::convertible_to<E,V>)
423 [[nodiscard]] friend native_inline constexpr simd masked_scaleb_zero(
424 M active, A value, E exponent) noexcept(noexcept(scaling_exponent(exponent))) {
425 V shift = scaling_exponent(exponent);
426 V result = ::ftz::detail::ftz32_vector_scaleb(value.value_, shift);
427 return {canonical{},select(active,result,V(0.f))};
428 }
429
430 template <class A, class E> requires std::same_as<A,simd> &&
431 (std::same_as<E,simd> || std::convertible_to<E,V>)
432 [[nodiscard]] friend native_inline constexpr simd scaleb(A value, E exponent)
433 noexcept(noexcept(scaling_exponent(exponent))) {
434 V shift = scaling_exponent(exponent);
435 return {canonical{},::ftz::detail::ftz32_vector_scaleb(value.value_, shift)};
436 }
437 // An FTZ exponent alone does not change a raw base's arithmetic contract.
438 // Explicit forwarding also prevents competing implicit native conversions.
441 template <class M, class A, class B>
442 requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>) &&
443 std::same_as<A,V> && std::same_as<B,V> &&
444 requires(M mask, V raw) { masked_scaleb(mask,raw,raw,raw); }
445 [[nodiscard]] friend native_inline constexpr native_const V masked_scaleb(
446 M active, A prior, B value, simd exponent) noexcept {
447 return masked_scaleb(active,prior,value,exponent.value_);
448 }
449
450 template <class M, class A>
451 requires (std::same_as<M,mask_type> || std::same_as<M,vector_mask_type>) && std::same_as<A,V> &&
452 requires(M mask, V raw) { masked_scaleb_zero(mask,raw,raw); }
453 [[nodiscard]] friend native_inline constexpr native_const V masked_scaleb_zero(
454 M active, A value, simd exponent) noexcept {
455 return masked_scaleb_zero(active,value,exponent.value_);
456 }
457
458 template <class A> requires std::same_as<A,V> && requires(V raw) { scaleb(raw,raw); }
459 [[nodiscard]] friend native_inline constexpr native_const V scaleb(A value, simd exponent) noexcept {
460 return scaleb(value,exponent.value_);
461 }
462
464 [[nodiscard]] friend native_inline constexpr native_const simd sqrt(simd a) noexcept {
465 return {canonical{}, ::ftz::detail::ftz32_vector_sqrt(a.value_)};
466 }
467
469 template <class A, class B> requires std::convertible_to<A, simd> && std::convertible_to<B, simd>
470 [[nodiscard]] friend native_inline constexpr simd fma(simd a, A b, B c)
471 noexcept(noexcept(simd(b)) && noexcept(simd(c))) {
472 return fused(a, simd(b), simd(c));
473 }
474
476 template <class A, class B> requires (!std::same_as<A, simd>) && std::convertible_to<A, simd> && std::convertible_to<B, simd>
477 [[nodiscard]] friend native_inline constexpr simd fma(A a, simd b, B c)
478 noexcept(noexcept(simd(a)) && noexcept(simd(c))) {
479 return fused(simd(a), b, simd(c));
480 }
481
483 template <class A, class B> requires (!std::same_as<A, simd>) && (!std::same_as<B, simd>) &&
484 std::convertible_to<A, simd> && std::convertible_to<B, simd>
485 [[nodiscard]] friend native_inline constexpr simd fma(A a, B b, simd c)
486 noexcept(noexcept(simd(a)) && noexcept(simd(b))) {
487 return fused(simd(a), simd(b), c);
488 }
489
490 // Mixed operands select this value contract instead of escaping via the
491 // deliberately implicit native-register conversion. No runtime dispatch.
494 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
495 [[nodiscard]] friend native_inline constexpr simd operator+(simd a, T b)
496 noexcept(noexcept(a + simd(b))) { return a + simd(b); }
497
499 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
500 [[nodiscard]] friend native_inline constexpr simd operator+(T a, simd b)
501 noexcept(noexcept(simd(a) + b)) { return simd(a) + b; }
502
504 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
505 [[nodiscard]] friend native_inline constexpr simd operator-(simd a, T b)
506 noexcept(noexcept(a - simd(b))) { return a - simd(b); }
507
509 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
510 [[nodiscard]] friend native_inline constexpr simd operator-(T a, simd b)
511 noexcept(noexcept(simd(a) - b)) { return simd(a) - b; }
512
514 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
515 [[nodiscard]] friend native_inline constexpr simd operator*(simd a, T b)
516 noexcept(noexcept(a * simd(b))) { return a * simd(b); }
517
519 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
520 [[nodiscard]] friend native_inline constexpr simd operator*(T a, simd b)
521 noexcept(noexcept(simd(a) * b)) { return simd(a) * b; }
522
524 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
525 [[nodiscard]] friend native_inline constexpr simd operator/(simd a, T b)
526 noexcept(noexcept(a / simd(b))) { return a / simd(b); }
527
529 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
530 [[nodiscard]] friend native_inline constexpr simd operator/(T a, simd b)
531 noexcept(noexcept(simd(a) / b)) { return simd(a) / b; }
532
534 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
535 [[nodiscard]] friend native_inline constexpr mask_type operator<(simd a, T b)
536 noexcept(noexcept(a < simd(b))) { return a < simd(b); }
537
539 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
540 [[nodiscard]] friend native_inline constexpr mask_type operator<(T a, simd b)
541 noexcept(noexcept(simd(a) < b)) { return simd(a) < b; }
542
544 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
545 [[nodiscard]] friend native_inline constexpr mask_type operator>(simd a, T b)
546 noexcept(noexcept(a > simd(b))) { return a > simd(b); }
547
549 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
550 [[nodiscard]] friend native_inline constexpr mask_type operator>(T a, simd b)
551 noexcept(noexcept(simd(a) > b)) { return simd(a) > b; }
552
554 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
555 [[nodiscard]] friend native_inline constexpr mask_type operator==(simd a, T b)
556 noexcept(noexcept(a == simd(b))) { return a == simd(b); }
557
559 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
560 [[nodiscard]] friend native_inline constexpr mask_type operator==(T a, simd b)
561 noexcept(noexcept(simd(a) == b)) { return simd(a) == b; }
562
564 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
565 [[nodiscard]] friend native_inline constexpr mask_type operator!=(simd a, T b)
566 noexcept(noexcept(a != simd(b))) { return a != simd(b); }
567
569 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
570 [[nodiscard]] friend native_inline constexpr mask_type operator!=(T a, simd b)
571 noexcept(noexcept(simd(a) != b)) { return simd(a) != b; }
572
574 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
575 [[nodiscard]] friend native_inline constexpr mask_type operator<=(simd a, T b)
576 noexcept(noexcept(a <= simd(b))) { return a <= simd(b); }
577
579 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
580 [[nodiscard]] friend native_inline constexpr mask_type operator<=(T a, simd b)
581 noexcept(noexcept(simd(a) <= b)) { return simd(a) <= b; }
582
584 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
585 [[nodiscard]] friend native_inline constexpr mask_type operator>=(simd a, T b)
586 noexcept(noexcept(a >= simd(b))) { return a >= simd(b); }
587
589 template <class T> requires (!std::same_as<T, simd>) && std::convertible_to<T, simd>
590 [[nodiscard]] friend native_inline constexpr mask_type operator>=(T a, simd b)
591 noexcept(noexcept(simd(a) >= b)) { return simd(a) >= b; }
592
593 private:
594
595 using bridge = ::ftz::detail::ftz32_bridge<V>;
596 template <class E>
597 [[nodiscard]] static native_inline constexpr V scaling_exponent(E exponent)
598 noexcept(noexcept(V(exponent))) {
599 if constexpr (std::same_as<E,simd>) return exponent.value_;
600 else {
601 V raw(exponent);
602 if constexpr(Hardware) return raw;
603 else return bridge::decode(::ftz::detail::ftz32_import_words(bridge::encode(raw)));
604 }
605 }
606 [[nodiscard]] static native_inline constexpr native_const V import(V value) noexcept {
607 return bridge::decode(::ftz::detail::ftz32_import_words(bridge::encode(value)));
608
609 }
610 struct canonical {};
611 V value_;
612 native_inline constexpr simd_customization(canonical, V value) noexcept : value_(value) {}
613 [[nodiscard]] static native_inline constexpr native_const simd fused(simd a, simd b, simd c) noexcept {
614 return {canonical{}, ::ftz::detail::ftz32_vector_fma(a.value_, b.value_, c.value_)};
615 }
616 };
617}
618
619export namespace ftz {
620 // A scalar FTZ operand promotes a raw register even when no FTZ SIMD
621 // argument exists for hidden-friend lookup. Never fall back to built-in float.
625 template <bool Hardware, detail::raw_register V>
626 [[nodiscard]] native_inline constexpr native_const auto operator+(V a,basic_ftz32<Hardware> b) noexcept {
627 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) + detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
628 }
629
632 template <bool Hardware, detail::raw_register V>
633 [[nodiscard]] native_inline constexpr native_const auto operator+(basic_ftz32<Hardware> a,V b) noexcept {
634 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) + detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
635 }
636
639 template <bool Hardware, detail::raw_register V>
640 [[nodiscard]] native_inline constexpr native_const auto operator-(V a,basic_ftz32<Hardware> b) noexcept {
641 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) - detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
642 }
643
646 template <bool Hardware, detail::raw_register V>
647 [[nodiscard]] native_inline constexpr native_const auto operator-(basic_ftz32<Hardware> a,V b) noexcept {
648 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) - detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
649 }
650
653 template <bool Hardware, detail::raw_register V>
654 [[nodiscard]] native_inline constexpr native_const auto operator*(V a,basic_ftz32<Hardware> b) noexcept {
655 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) * detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
656 }
657
660 template <bool Hardware, detail::raw_register V>
661 [[nodiscard]] native_inline constexpr native_const auto operator*(basic_ftz32<Hardware> a,V b) noexcept {
662 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) * detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
663 }
664
667 template <bool Hardware, detail::raw_register V>
668 [[nodiscard]] native_inline constexpr native_const auto operator/(V a,basic_ftz32<Hardware> b) noexcept {
669 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) / detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
670 }
671
674 template <bool Hardware, detail::raw_register V>
675 [[nodiscard]] native_inline constexpr native_const auto operator/(basic_ftz32<Hardware> a,V b) noexcept {
676 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) / detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
677 }
678
681 template <bool Hardware, detail::raw_register V>
682 [[nodiscard]] native_inline constexpr native_const auto operator<(V a,basic_ftz32<Hardware> b) noexcept {
683 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) < detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
684 }
685
688 template <bool Hardware, detail::raw_register V>
689 [[nodiscard]] native_inline constexpr native_const auto operator<(basic_ftz32<Hardware> a,V b) noexcept {
690 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) < detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
691 }
692
695 template <bool Hardware, detail::raw_register V>
696 [[nodiscard]] native_inline constexpr native_const auto operator>(V a,basic_ftz32<Hardware> b) noexcept {
697 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) > detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
698 }
699
702 template <bool Hardware, detail::raw_register V>
703 [[nodiscard]] native_inline constexpr native_const auto operator>(basic_ftz32<Hardware> a,V b) noexcept {
704 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) > detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
705 }
706
709 template <bool Hardware, detail::raw_register V>
710 [[nodiscard]] native_inline constexpr native_const auto operator==(V a,basic_ftz32<Hardware> b) noexcept {
711 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) == detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
712 }
713
716 template <bool Hardware, detail::raw_register V>
717 [[nodiscard]] native_inline constexpr native_const auto operator==(basic_ftz32<Hardware> a,V b) noexcept {
718 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) == detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
719 }
720
723 template <bool Hardware, detail::raw_register V>
724 [[nodiscard]] native_inline constexpr native_const auto operator!=(V a,basic_ftz32<Hardware> b) noexcept {
725 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) != detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
726 }
727
730 template <bool Hardware, detail::raw_register V>
731 [[nodiscard]] native_inline constexpr native_const auto operator!=(basic_ftz32<Hardware> a,V b) noexcept {
732 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) != detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
733 }
734
737 template <bool Hardware, detail::raw_register V>
738 [[nodiscard]] native_inline constexpr native_const auto operator<=(V a,basic_ftz32<Hardware> b) noexcept {
739 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) <= detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
740 }
741
744 template <bool Hardware, detail::raw_register V>
745 [[nodiscard]] native_inline constexpr native_const auto operator<=(basic_ftz32<Hardware> a,V b) noexcept {
746 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) <= detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
747 }
748
751 template <bool Hardware, detail::raw_register V>
752 [[nodiscard]] native_inline constexpr native_const auto operator>=(V a,basic_ftz32<Hardware> b) noexcept {
753 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) >= detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
754 }
755
758 template <bool Hardware, detail::raw_register V>
759 [[nodiscard]] native_inline constexpr native_const auto operator>=(basic_ftz32<Hardware> a,V b) noexcept {
760 return detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) >= detail::ftz32_simd<V,basic_ftz32<Hardware>>(b);
761 }
762 // Compound assignment keeps the destination type, while the FTZ operand
763 // still chooses the arithmetic contract before publication to raw storage.
766 template <bool Hardware, detail::raw_register V>
767 native_inline constexpr V & operator+=(V & a,basic_ftz32<Hardware> b) noexcept {
768 a=(detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) + detail::ftz32_simd<V,basic_ftz32<Hardware>>(b)).to_native();
769 return a;
770 }
771
773 template <detail::raw_register V, detail::ftz32_vector R> requires std::same_as<typename R::register_type,V>
774 native_inline constexpr V & operator+=(V & a,R b) noexcept {
775 a=(R(a) + R(b)).to_native();
776 return a;
777 }
778
780 template <bool Hardware, detail::raw_register V>
781 native_inline constexpr V & operator-=(V & a,basic_ftz32<Hardware> b) noexcept {
782 a=(detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) - detail::ftz32_simd<V,basic_ftz32<Hardware>>(b)).to_native();
783 return a;
784 }
785
787 template <detail::raw_register V, detail::ftz32_vector R> requires std::same_as<typename R::register_type,V>
788 native_inline constexpr V & operator-=(V & a,R b) noexcept {
789 a=(R(a) - R(b)).to_native();
790 return a;
791 }
792
794 template <bool Hardware, detail::raw_register V>
795 native_inline constexpr V & operator*=(V & a,basic_ftz32<Hardware> b) noexcept {
796 a=(detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) * detail::ftz32_simd<V,basic_ftz32<Hardware>>(b)).to_native();
797 return a;
798 }
799
801 template <detail::raw_register V, detail::ftz32_vector R> requires std::same_as<typename R::register_type,V>
802 native_inline constexpr V & operator*=(V & a,R b) noexcept {
803 a=(R(a) * R(b)).to_native();
804 return a;
805 }
806
808 template <bool Hardware, detail::raw_register V>
809 native_inline constexpr V & operator/=(V & a,basic_ftz32<Hardware> b) noexcept {
810 a=(detail::ftz32_simd<V,basic_ftz32<Hardware>>(a) / detail::ftz32_simd<V,basic_ftz32<Hardware>>(b)).to_native();
811 return a;
812 }
813
815 template <detail::raw_register V, detail::ftz32_vector R> requires std::same_as<typename R::register_type,V>
816 native_inline constexpr V & operator/=(V & a,R b) noexcept {
817 a=(R(a) / R(b)).to_native();
818 return a;
819 }
820 namespace detail {
821 template <class... T> struct raw_family { using type = void; };
822 template <class T,class... Rest> struct raw_family<T,Rest...> {
823 using type = std::conditional_t<raw_register<T>,T,typename raw_family<Rest...>::type>;
824 };
825 template <class... X> using raw_family_t = typename raw_family<X...>::type;
826 template <class... T> struct scalar_family { using type = void; };
827 template <class T,class... Rest> struct scalar_family<T,Rest...> {
828 using type = std::conditional_t<ftz32_type<T>,T,typename scalar_family<Rest...>::type>;
829 };
830 template <class... X> using scalar_family_t = typename scalar_family<X...>::type;
831 template <class... X> concept ftz32_raw_scalar_fma =
832 (ftz32_type<std::remove_cvref_t<X>> || ...) &&
833 raw_register<raw_family_t<X...>> &&
834 (std::convertible_to<X &,ftz32_simd<raw_family_t<X...>,scalar_family_t<X...>>> && ...);
835 }
839 template <class A,class B,class C> requires detail::ftz32_raw_scalar_fma<A,B,C>
840 [[nodiscard]] native_inline constexpr auto fma(A a,B b,C c)
841 noexcept(std::is_nothrow_constructible_v<detail::ftz32_simd<detail::raw_family_t<A,B,C>,detail::scalar_family_t<A,B,C>>,A &> &&
842 std::is_nothrow_constructible_v<detail::ftz32_simd<detail::raw_family_t<A,B,C>,detail::scalar_family_t<A,B,C>>,B &> &&
843 std::is_nothrow_constructible_v<detail::ftz32_simd<detail::raw_family_t<A,B,C>,detail::scalar_family_t<A,B,C>>,C &>) {
844 using result_type=detail::ftz32_simd<detail::raw_family_t<A,B,C>,detail::scalar_family_t<A,B,C>>;
845 return fma(result_type(a),result_type(b),result_type(c));
846 }
847}
848
849#if !defined(__cpp_structured_bindings) || __cpp_structured_bindings < 202411L
850#error "FTZ array math requires C++26 structured-binding packs (Clang 21+ with -std=c++2c or clang-cl /std:c++latest)."
851#endif
852
853export namespace ftz {
854 namespace detail {
855 template <class R> struct ftz32_native_for;
856 template <ftz32_vector R> struct ftz32_native_for<R> { using type = typename R::register_type; };
857 template <bool H> struct ftz32_native_for<basic_ftz32<H>> { using type = ::ftz::detail::native::fp32x1; };
858 template <class R> using ftz32_native = typename ftz32_native_for<R>::type;
859 template <class R> concept ftz32_value = requires { typename ftz32_native_for<R>::type; };
860 template <ftz32_value R> [[nodiscard]] native_inline constexpr native_const ftz32_native<R> ftz32_unwrap(R value) noexcept {
861 if constexpr (ftz32_type<R>) return ::ftz::detail::native::fp32x1(value.to_float());
862 else return value.to_native();
863 }
864 template <ftz32_value R> [[nodiscard]] native_inline constexpr native_const R ftz32_wrap(ftz32_native<R> value) noexcept {
865 if constexpr (ftz32_type<R>) return R::unsafe_from_float32(value.value);
866 else return R::unsafe_from_float32(value);
867 }
868
869 }
870
871 namespace detail::ftz32_math {
872 template<auto Operation, detail::ftz32_value R, class... X>
873 [[nodiscard]] constexpr R constant(R input, X... rest) noexcept {
874 if constexpr(ftz32_type<R>)
875 return R::unsafe_from_float32(std::bit_cast<float>(Operation(input.to_bits(),rest.to_bits()...)));
876 else return R::unsafe_from_float32(detail::ftz32_constant<Operation>(input.to_native(),rest.to_native()...));
877 }
878 // The exceptional pair shares extraction and full-range reduction. Keep
879 // bounded registers untouched and standalone sin/cos on their own graphs.
880 template <bool Hardware, class V>
881 [[nodiscard]] native_inline constexpr native_pure std::pair<V, V> repair_sincos(
882 V sine, V cosine, detail::ftz32_words<V> mask, V original) noexcept {
883 if (!detail::ftz32_any(mask)) return {sine, cosine};
884 using B = detail::ftz32_bridge<V>; using U = detail::ftz32_words<V>;
885 std::array<std::uint32_t, V::lanes> flags, input, sine_words, cosine_words;
886 mask.store(flags.data()); B::encode(original).store(input.data());
887 B::encode(sine).store(sine_words.data()); B::encode(cosine).store(cosine_words.data());
888 for (std::size_t lane = 0; lane < V::lanes; ++lane) {
889 if (flags[lane] != 0) {
890 auto pair = detail::ftz32_sincos<Hardware>(input[lane]);
891 sine_words[lane] = pair.sine; cosine_words[lane] = pair.cosine;
892 }
893 }
894 return {B::decode(U::load(sine_words.data())), B::decode(U::load(cosine_words.data()))};
895 }
896 // All register chains enter the existing stage-interleaved polynomial in one
897 // call. Only out-of-domain lanes use the scalar full-range/special-value path.
898 template <detail::ftz32_value R, std::size_t N>
899 [[nodiscard]] native_inline constexpr native_pure auto sincos(std::array<R, N> const & input) noexcept {
900 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
901 std::array<R,N> sine{},cosine{};
902 constexpr auto sine_word=[](std::uint32_t word) { return detail::ftz32_sincos<R::hardware>(word).sine; };
903 constexpr auto cosine_word=[](std::uint32_t word) { return detail::ftz32_sincos<R::hardware>(word).cosine; };
904 for(std::size_t i=0;i<N;++i) {
905 sine[i]=constant<sine_word>(input[i]);
906 cosine[i]=constant<cosine_word>(input[i]);
907 }
908 return std::pair{sine,cosine};
909 }
910 if constexpr (N == 0) return std::pair{std::array<R,0>{}, std::array<R,0>{}};
911 else {
912 using V = detail::ftz32_native<R>; using B = detail::ftz32_bridge<V>; using U = detail::ftz32_words<V>;
913 auto const & [...input_register] = input;
914 auto const [...original] = std::array{detail::ftz32_unwrap(input_register)...};
915 auto bounded = [](V value) {
916 return mask_bits<std::uint32_t>(U(0x46000000u) > (B::encode(value) & U(0x7fffffffu)));
917 };
918 auto const [...allowed] = std::array{bounded(original)...};
919 auto [sine_values, cosine_values] = ::ftz::detail::native::sincos_ftz<R::hardware>(
920 std::array{B::decode(B::encode(original) & allowed)...});
921 auto const & [...sine] = sine_values;
922 auto const & [...cosine] = cosine_values;
923 auto const [...repaired] = std::array{repair_sincos<R::hardware>(
924 sine, cosine, allowed ^ U(0xffffffffu), original)...};
925 return std::pair{std::array{detail::ftz32_wrap<R>(repaired.first)...},
926 std::array{detail::ftz32_wrap<R>(repaired.second)...}};
927 }
928 }
929 template <bool Cosine, detail::ftz32_value R, std::size_t N>
930 [[nodiscard]] native_inline constexpr native_pure std::array<R, N> trig_single(std::array<R, N> const & input) noexcept {
931 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
932 std::array<R,N> result{};
933 for(std::size_t i=0;i<N;++i) {
934 if constexpr(Cosine) result[i]=constant<detail::ftz32_cos<R::hardware>>(input[i]);
935 else result[i]=constant<detail::ftz32_sin<R::hardware>>(input[i]);
936 }
937 return result;
938 }
939 if constexpr (N == 0) return {};
940 else {
941 using V = detail::ftz32_native<R>; using B = detail::ftz32_bridge<V>; using U = detail::ftz32_words<V>;
942 auto const & [...input_register] = input;
943 auto const [...original] = std::array{detail::ftz32_unwrap(input_register)...};
944 auto const [...allowed] = std::array{mask_bits<std::uint32_t>(
945 U(0x46000000u) > (B::encode(original) & U(0x7fffffffu)))...};
946 auto const safe = std::array{B::decode(B::encode(original) & allowed)...};
947 auto const [...value] = [&] {
948 if constexpr (Cosine) return native::cos_ftz<R::hardware>(safe);
949 else return native::sin_ftz<R::hardware>(safe);
950 }();
951 if constexpr (Cosine)
952 return std::array{detail::ftz32_wrap<R>(detail::ftz32_repair<detail::ftz32_cos<R::hardware>>(
953 value, allowed ^ U(0xffffffffu), original))...};
954 else
955 return std::array{detail::ftz32_wrap<R>(detail::ftz32_repair<detail::ftz32_sin<R::hardware>>(
956 value, allowed ^ U(0xffffffffu), original))...};
957 }
958 }
959 template <detail::ftz32_value R, std::size_t N>
960 [[nodiscard]] native_inline constexpr native_pure std::array<R, N> sin(std::array<R, N> const & input) noexcept { return trig_single<false>(input); }
961 template <detail::ftz32_value R, std::size_t N>
962 [[nodiscard]] native_inline constexpr native_pure std::array<R, N> cos(std::array<R, N> const & input) noexcept { return trig_single<true>(input); }
963 template <unsigned int Degree = 6, detail::ftz32_value R, std::size_t N> requires (Degree >= 1 && Degree <= 7)
964 [[nodiscard]] native_inline constexpr native_pure std::array<R, N> exp(std::array<R, N> const & input) noexcept {
965 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
966 std::array<R,N> result{};
967 for(std::size_t i=0;i<N;++i) result[i]=constant<detail::ftz32_exp<Degree>>(input[i]);
968 return result;
969 }
970 if constexpr (N == 0) return {};
971 else {
972 auto const & [...input_register] = input;
973 auto const [...value] = native::exp_ftz<Degree>(
974 std::array{detail::ftz32_unwrap(input_register)...});
975 return std::array{detail::ftz32_wrap<R>(value)...};
976 }
977 }
978 template <detail::ftz32_value R, std::size_t N>
979 [[nodiscard]] native_inline constexpr native_pure std::array<R, N> expm1(std::array<R, N> const & input) noexcept {
980 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
981 std::array<R,N> result{};
982 for(std::size_t i=0;i<N;++i) result[i]=constant<detail::ftz32_expm1<R::hardware>>(input[i]);
983 return result;
984 }
985 if constexpr (N == 0) return {};
986 else {
987 using U = detail::ftz32_words<detail::ftz32_native<R>>;
988 auto const & [...input_register] = input;
989 auto original = std::array{detail::ftz32_unwrap(input_register)...};
990 auto [values, validity] = ::ftz::detail::native::expm1_checked<R::hardware>(original);
991 auto const & [...x] = original;
992 auto const & [...value] = values;
993 auto const & [...valid] = validity;
994 // The native graph classifies and safely evaluates once. Its validity is
995 // 0/1 per lane; subtraction maps invalid lanes to the full repair mask.
996 return std::array{detail::ftz32_wrap<R>(detail::ftz32_repair<detail::ftz32_expm1<R::hardware>>(
997 value, valid - U(1), x))...};
998 }
999 }
1000 }
1004 template <detail::ftz32_vector R>
1005 [[nodiscard]] native_inline constexpr native_pure auto sincos(R input) noexcept {
1006 auto [sine_values, cosine_values] = detail::ftz32_math::sincos(std::array{input});
1007 auto [sine] = sine_values;
1008 auto [cosine] = cosine_values;
1009 return std::pair{sine, cosine};
1010 }
1011
1014 template <detail::ftz32_vector R>
1015 [[nodiscard]] native_inline constexpr native_pure R sin(R input) noexcept { return detail::ftz32_math::sin(std::array{input})[0]; }
1019 template <detail::ftz32_vector R>
1020 [[nodiscard]] native_inline constexpr native_pure R cos(R input) noexcept { return detail::ftz32_math::cos(std::array{input})[0]; }
1025 template <unsigned int Degree = 6, detail::ftz32_vector R> requires (Degree >= 1 && Degree <= 7)
1026 [[nodiscard]] native_inline constexpr native_pure R exp(R input) noexcept {
1027 auto [value] = detail::ftz32_math::exp<Degree>(std::array{input});
1028 return value;
1029 }
1030
1034 template <detail::ftz32_vector R, bool Flush, unsigned int Degree> requires (!Flush && Degree >= 1 && Degree <= 7)
1035 [[nodiscard]] native_inline constexpr native_pure R exp(R input, std::bool_constant<Flush>, std::integral_constant<unsigned int, Degree>) noexcept { return exp<Degree>(input); }
1038 template <detail::ftz32_vector R>
1039 [[nodiscard]] native_inline constexpr native_pure R expm1(R input) noexcept {
1040 auto [value] = detail::ftz32_math::expm1(std::array{input});
1041 return value;
1042 }
1043
1044 namespace detail {
1045 template <char Op, class V> [[nodiscard]] native_inline constexpr native_const V ftz32_native_binary(V a,V b) noexcept {
1046 if constexpr(Op=='+') return a+b;
1047 else if constexpr(Op=='-') return a-b;
1048 else return a*b;
1049 }
1050 template <char Op, class R, std::size_t N>
1051 [[nodiscard]] native_inline constexpr native_pure std::array<R,N> ftz32_array_binary(
1052 std::array<R,N> const & a, std::array<R,N> const & b) noexcept {
1053 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
1054 std::array<R,N> result{};
1055 for(std::size_t i=0;i<N;++i) {
1056 if constexpr(Op=='+') result[i]=a[i]+b[i];
1057 else if constexpr(Op=='-') result[i]=a[i]-b[i];
1058 else result[i]=a[i]*b[i];
1059 }
1060 return result;
1061 }
1062 if constexpr (N == 0) return {};
1063 else {
1064 auto const & [...x] = a;
1065 auto const & [...y] = b;
1066 // Issue every native operation before classifying or repairing lanes.
1067 auto const [...native] = std::array{ftz32_native_binary<Op>(x.to_native(), y.to_native())...};
1068 if constexpr (R::hardware && (Op == '+' || Op == '-'))
1069 return std::array{R::unsafe_from_float32(native)...};
1070 auto const [...mask] = std::array{ftz32_repair_mask(native)...};
1071 constexpr auto repair = [] {
1072 if constexpr (Op == '+') return ftz32_add<R::hardware>;
1073 else if constexpr (Op == '-') return ftz32_sub<R::hardware>;
1074 else return ftz32_mul;
1075 }();
1076 return std::array{R::unsafe_from_float32(ftz32_repair<repair>(
1077 native, mask, x.to_native(), y.to_native()))...};
1078 }
1079 }
1080 template <class R, std::size_t N>
1081 [[nodiscard]] native_inline constexpr native_pure std::array<R,N> ftz32_array_fma(std::array<R,N> const & a,
1082 std::array<R,N> const & b, std::array<R,N> const & c) noexcept {
1083 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
1084 std::array<R,N> result{};
1085 for(std::size_t i=0;i<N;++i) result[i]=fma(a[i],b[i],c[i]);
1086 return result;
1087 }
1088 if constexpr (N == 0) return {};
1089 else {
1090 auto const & [...x] = a;
1091 auto const & [...y] = b;
1092 auto const & [...z] = c;
1093 auto const [...native] = std::array{fma(x.to_native(), y.to_native(), z.to_native())...};
1094 auto const [...mask] = std::array{ftz32_repair_mask(native)...};
1095 return std::array{R::unsafe_from_float32(ftz32_repair<ftz32_fma>(
1096 native, mask, x.to_native(), y.to_native(), z.to_native()))...};
1097 }
1098 }
1099 }
1100 namespace detail {
1101 template<class T, std::size_t N> native_inline constexpr std::array<T,N> array_broadcast(T const & value) noexcept {
1102 std::array<T,N> result;
1103 auto & [...element] = result;
1104 ((element = value), ...);
1105 return result;
1106 }
1107 }
1111 template<class R, std::size_t N> requires detail::ftz32_vector<R>
1112 [[nodiscard]] native_inline constexpr std::array<R,N> add(std::array<R,N> const & a, std::array<R,N> const & b) noexcept {
1113 return detail::ftz32_array_binary<'+'>(a,b);
1114 }
1115
1118 template<class R, std::size_t N, class B> requires detail::ftz32_vector<R> && std::convertible_to<B const &,R>
1119 [[nodiscard]] native_inline constexpr std::array<R,N> add(std::array<R,N> const & a, B const & b)
1120 noexcept(std::is_nothrow_constructible_v<R,B const &>) {
1121 return add(a, detail::array_broadcast<R,N>(R(b)));
1122 }
1123
1126 template<class R, std::size_t N, class A> requires detail::ftz32_vector<R> && std::convertible_to<A const &,R>
1127 [[nodiscard]] native_inline constexpr std::array<R,N> add(A const & a, std::array<R,N> const & b)
1128 noexcept(std::is_nothrow_constructible_v<R,A const &>) {
1129 return add(detail::array_broadcast<R,N>(R(a)), b);
1130 }
1131
1134 template<class R, std::size_t N> requires detail::ftz32_vector<R>
1135 [[nodiscard]] native_inline constexpr std::array<R,N> sub(std::array<R,N> const & a, std::array<R,N> const & b) noexcept {
1136 return detail::ftz32_array_binary<'-'>(a,b);
1137 }
1138
1141 template<class R, std::size_t N, class B> requires detail::ftz32_vector<R> && std::convertible_to<B const &,R>
1142 [[nodiscard]] native_inline constexpr std::array<R,N> sub(std::array<R,N> const & a, B const & b)
1143 noexcept(std::is_nothrow_constructible_v<R,B const &>) {
1144 return sub(a, detail::array_broadcast<R,N>(R(b)));
1145 }
1146
1149 template<class R, std::size_t N, class A> requires detail::ftz32_vector<R> && std::convertible_to<A const &,R>
1150 [[nodiscard]] native_inline constexpr std::array<R,N> sub(A const & a, std::array<R,N> const & b)
1151 noexcept(std::is_nothrow_constructible_v<R,A const &>) {
1152 return sub(detail::array_broadcast<R,N>(R(a)), b);
1153 }
1154
1157 template<class R, std::size_t N> requires detail::ftz32_vector<R>
1158 [[nodiscard]] native_inline constexpr std::array<R,N> mul(std::array<R,N> const & a, std::array<R,N> const & b) noexcept {
1159 return detail::ftz32_array_binary<'*'>(a,b);
1160 }
1161
1164 template<class R, std::size_t N, class B> requires detail::ftz32_vector<R> && std::convertible_to<B const &,R>
1165 [[nodiscard]] native_inline constexpr std::array<R,N> mul(std::array<R,N> const & a, B const & b)
1166 noexcept(std::is_nothrow_constructible_v<R,B const &>) {
1167 return mul(a, detail::array_broadcast<R,N>(R(b)));
1168 }
1169
1172 template<class R, std::size_t N, class A> requires detail::ftz32_vector<R> && std::convertible_to<A const &,R>
1173 [[nodiscard]] native_inline constexpr std::array<R,N> mul(A const & a, std::array<R,N> const & b)
1174 noexcept(std::is_nothrow_constructible_v<R,A const &>) {
1175 return mul(detail::array_broadcast<R,N>(R(a)), b);
1176 }
1177
1180 template<class R, std::size_t N> requires detail::ftz32_vector<R>
1181 [[nodiscard]] native_inline constexpr std::array<R,N> fma(std::array<R,N> const & a,
1182 std::array<R,N> const & b, std::array<R,N> const & c) noexcept {
1183 return detail::ftz32_array_fma(a,b,c);
1184 }
1185
1188 template<detail::ftz32_value R, std::size_t N>
1189 [[nodiscard]] native_inline constexpr auto sin(std::array<R,N> const & input) noexcept {
1190 return detail::ftz32_math::sin(input);
1191 }
1192
1195 template<detail::ftz32_value R, std::size_t N>
1196 [[nodiscard]] native_inline constexpr auto cos(std::array<R,N> const & input) noexcept {
1197 return detail::ftz32_math::cos(input);
1198 }
1199
1203 template<unsigned int Degree = 6, detail::ftz32_value R, std::size_t N> requires (Degree >= 1 && Degree <= 7)
1204 [[nodiscard]] native_inline constexpr auto exp(std::array<R,N> const & input) noexcept {
1205 return detail::ftz32_math::exp<Degree>(input);
1206 }
1207
1211 template<detail::ftz32_value R, std::size_t N, bool Flush, unsigned int Degree> requires (!Flush && Degree >= 1 && Degree <= 7)
1212 [[nodiscard]] native_inline constexpr auto exp(std::array<R,N> const & input, std::bool_constant<Flush>, std::integral_constant<unsigned int, Degree>) noexcept { return exp<Degree>(input); }
1215 template<detail::ftz32_value R, std::size_t N>
1216 [[nodiscard]] native_inline constexpr auto expm1(std::array<R,N> const & input) noexcept {
1217 return detail::ftz32_math::expm1(input);
1218 }
1219
1222 template<detail::ftz32_value R, std::size_t N>
1223 [[nodiscard]] native_inline constexpr auto sincos(std::array<R,N> const & input) noexcept {
1224 return detail::ftz32_math::sincos(input);
1225 }
1226
1230 template<detail::ftz32_value R,std::size_t N>
1231 [[nodiscard]] native_inline constexpr std::array<R,N> tanh(std::array<R,N> const & input) noexcept {
1232 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
1233 std::array<R,N> result{};
1234 for(std::size_t i=0;i<N;++i)
1235 result[i]=detail::ftz32_math::constant<detail::ftz32_tanh<R::hardware>>(input[i]);
1236 return result;
1237 }
1238 if constexpr(N==0) return {};
1239 else {
1240 auto const & [...value]=input;
1241 auto const [...result]=detail::native::tanh_ftz<R::hardware>(
1242 std::array{detail::ftz32_unwrap(value)...});
1243 return {{detail::ftz32_wrap<R>(result)...}};
1244 }
1245 }
1246
1249 template<detail::ftz32_vector R>
1250 [[nodiscard]] native_inline constexpr R tanh(R input) noexcept {
1251 auto [result]=::ftz::tanh(std::array{input});
1252 return result;
1253 }
1254
1258 template<detail::ftz32_value R,std::size_t N>
1259 [[nodiscard]] native_inline constexpr std::array<R,N> log(std::array<R,N> const & input) noexcept {
1260 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
1261 std::array<R,N> result{};
1262 for(std::size_t i=0;i<N;++i)
1263 result[i]=detail::ftz32_math::constant<detail::ftz32_log<R::hardware>>(input[i]);
1264 return result;
1265 }
1266 if constexpr(N==0) return {};
1267 else {
1268 auto const & [...value]=input;
1269 auto const [...result]=detail::native::log_ftz<false,R::hardware>(
1270 std::array{detail::ftz32_unwrap(value)...});
1271 return {{detail::ftz32_wrap<R>(result)...}};
1272 }
1273 }
1274
1277 template<detail::ftz32_vector R>
1278 [[nodiscard]] native_inline constexpr R log(R input) noexcept {
1279 auto [result]=::ftz::log(std::array{input});
1280 return result;
1281 }
1282
1285 template<detail::ftz32_value R,std::size_t N>
1286 [[nodiscard]] native_inline constexpr std::array<R,N> log1p(std::array<R,N> const & input) noexcept {
1287 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
1288 std::array<R,N> result{};
1289 for(std::size_t i=0;i<N;++i)
1290 result[i]=detail::ftz32_math::constant<detail::ftz32_log1p<R::hardware>>(input[i]);
1291 return result;
1292 }
1293 if constexpr(N==0) return {};
1294 else {
1295 auto const & [...value]=input;
1296 auto const [...result]=detail::native::log_ftz<true,R::hardware>(
1297 std::array{detail::ftz32_unwrap(value)...});
1298 return {{detail::ftz32_wrap<R>(result)...}};
1299 }
1300 }
1301
1304 template<detail::ftz32_vector R>
1305 [[nodiscard]] native_inline constexpr R log1p(R input) noexcept {
1306 auto [result]=::ftz::log1p(std::array{input});
1307 return result;
1308 }
1309
1313 template<detail::ftz32_value R,std::size_t N>
1314 [[nodiscard]] native_inline constexpr std::array<R,N> atan2(
1315 std::array<R,N> const & y,std::array<R,N> const & x) noexcept {
1316 if (std::is_constant_evaluated() || ::ftz::detail::ftz32_emulated<R>) {
1317 std::array<R,N> result{};
1318 for(std::size_t i=0;i<N;++i)
1319 result[i]=detail::ftz32_math::constant<detail::ftz32_atan2<R::hardware>>(y[i],x[i]);
1320 return result;
1321 }
1322 if constexpr(N==0) return {};
1323 else {
1324 auto const & [...a]=y;
1325 auto const & [...b]=x;
1326 auto const [...result]=detail::native::atan2_ftz<R::hardware>(
1327 std::array{detail::ftz32_unwrap(a)...},std::array{detail::ftz32_unwrap(b)...});
1328 return {{detail::ftz32_wrap<R>(result)...}};
1329 }
1330 }
1331
1333 template<detail::ftz32_vector R>
1334 [[nodiscard]] native_inline constexpr R atan2(R y,R x) noexcept {
1335 auto [result]=::ftz::atan2(std::array{y},std::array{x});
1336 return result;
1337 }
1338
1341 template<detail::ftz32_vector R>
1342 [[nodiscard]] native_inline constexpr typename R::mask isnan(R value) noexcept {
1343 using U=typename R::bits_type;
1344 return (value.to_bits() & U(0x7fffffffu)) > U(0x7f800000u);
1345 }
1346
1348 template<detail::ftz32_vector R>
1349 [[nodiscard]] native_inline constexpr typename R::mask isinf(R value) noexcept {
1350 using U=typename R::bits_type;
1351 return (value.to_bits() & U(0x7fffffffu)) == U(0x7f800000u);
1352 }
1353
1355 template<detail::ftz32_vector R>
1356 [[nodiscard]] native_inline constexpr typename R::mask isfinite(R value) noexcept {
1357 using U=typename R::bits_type;
1358 return (value.to_bits() & U(0x7fffffffu)) < U(0x7f800000u);
1359 }
1360
1362 template<detail::ftz32_vector R>
1363 [[nodiscard]] native_inline constexpr typename R::mask signbit(R value) noexcept {
1364 using U=typename R::bits_type;
1365 return (value.to_bits() & U(0x80000000u)) != U(0);
1366 }
1367
1370 template<detail::ftz32_vector R>
1371 [[nodiscard]] native_inline constexpr R copysign(R magnitude,R sign) noexcept {
1372 using U=typename R::bits_type;
1373 using V=typename R::register_type;
1374 return R::unsafe_from_float32(V::from_bits(
1375 (magnitude.to_bits() & U(0x7fffffffu)) | (sign.to_bits() & U(0x80000000u))));
1376 }
1377
1378 // Integral-valued results cannot be subnormal; preserve the raw rounding result.
1382 template<detail::ftz32_vector R>
1383 [[nodiscard]] native_inline constexpr R floor(R value) noexcept {
1384 using ::native::floor;
1385 return R::unsafe_from_float32(floor(value.to_native()));
1386 }
1387
1390 template<detail::ftz32_value R,std::size_t N>
1391 [[nodiscard]] native_inline constexpr std::array<R,N> floor(std::array<R,N> const & input) noexcept {
1392 auto const & [...value]=input;
1393 return {{floor(value)...}};
1394 }
1395
1398 template<detail::ftz32_vector R>
1399 [[nodiscard]] native_inline constexpr R ceil(R value) noexcept {
1400 using ::native::ceil;
1401 return R::unsafe_from_float32(ceil(value.to_native()));
1402 }
1403
1406 template<detail::ftz32_value R,std::size_t N>
1407 [[nodiscard]] native_inline constexpr std::array<R,N> ceil(std::array<R,N> const & input) noexcept {
1408 auto const & [...value]=input;
1409 return {{ceil(value)...}};
1410 }
1411
1414 template<detail::ftz32_vector R>
1415 [[nodiscard]] native_inline constexpr R trunc(R value) noexcept {
1416 using ::native::trunc;
1417 return R::unsafe_from_float32(trunc(value.to_native()));
1418 }
1419
1422 template<detail::ftz32_value R,std::size_t N>
1423 [[nodiscard]] native_inline constexpr std::array<R,N> trunc(std::array<R,N> const & input) noexcept {
1424 auto const & [...value]=input;
1425 return {{trunc(value)...}};
1426 }
1427}
constexpr std::array< R, N > mul(std::array< R, N > const &a, std::array< R, N > const &b) noexcept
Multiplies matching register arrays; a non-array operand is converted once and broadcast,...
Definition simd.h:1158
constexpr std::array< R, N > log1p(std::array< R, N > const &input) noexcept
Computes log(1+x), preserving signed zero; -1 gives negative infinity and x < -1 gives NaN....
Definition simd.h:1286
constexpr std::array< R, N > add(std::array< R, N > const &a, std::array< R, N > const &b) noexcept
Adds matching register arrays; a non-array operand is converted once and broadcast,...
Definition simd.h:1112
constexpr std::array< R, N > log(std::array< R, N > const &input) noexcept
Computes natural logarithms; either zero gives negative infinity, negative nonzero values give NaN....
Definition simd.h:1259
constexpr std::array< R, N > sub(std::array< R, N > const &a, std::array< R, N > const &b) noexcept
Subtracts matching register arrays; a non-array operand is converted once and broadcast,...
Definition simd.h:1135
constexpr std::array< R, N > atan2(std::array< R, N > const &y, std::array< R, N > const &x) noexcept
Computes atan2(y,x) in radians with the scalar FTZ signed-axis and infinity rules....
Definition simd.h:1314
constexpr std::array< R, N > tanh(std::array< R, N > const &input) noexcept
Computes hyperbolic tangent with the scalar FTZ graph, preserving signed zero and saturating infiniti...
Definition simd.h:1231
constexpr R trunc(R value) noexcept
Rounds toward zero, independently of ambient rounding mode; signed zero and infinities survive....
Definition simd.h:1415
constexpr R::mask isfinite(R value) noexcept
Returns R::mask for finite lanes by inspecting words, without changing FP status.
Definition simd.h:1356
constexpr R::mask isinf(R value) noexcept
Returns R::mask for either infinity by inspecting lane words.
Definition simd.h:1349
constexpr auto operator!=(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns the native comparison mask.
Definition simd.h:724
constexpr R floor(R value) noexcept
Rounds toward negative infinity, independently of ambient rounding mode; signed zero and infinities s...
Definition simd.h:1383
constexpr R cos(R input) noexcept
Computes cosine in radians with dedicated output reconstruction and the scalar FTZ special-value rule...
Definition simd.h:1020
constexpr V & operator+=(V &a, basic_ftz32< Hardware > b) noexcept
Evaluates FTZ arithmetic before assigning to the existing raw-register destination.
Definition simd.h:767
constexpr auto operator>=(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns the native comparison mask.
Definition simd.h:752
constexpr R::mask signbit(R value) noexcept
Returns R::mask for set sign bits, including negative zero and signed NaNs.
Definition simd.h:1363
constexpr auto operator==(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns the native comparison mask.
Definition simd.h:710
constexpr V & operator-=(V &a, basic_ftz32< Hardware > b) noexcept
Evaluates FTZ arithmetic before assigning to the existing raw-register destination.
Definition simd.h:781
constexpr auto operator>(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns the native comparison mask.
Definition simd.h:696
constexpr auto operator/(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns its FTZ vector rebind.
Definition simd.h:668
constexpr R expm1(R input) noexcept
Computes exp(x)-1 with the scalar FTZ graph, preserving signed zero. Returns R.
Definition simd.h:1039
constexpr auto operator*(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns its FTZ vector rebind.
Definition simd.h:654
constexpr auto sincos(R input) noexcept
Computes sine and cosine in radians with the scalar special-value rules; returns a pair in that order...
Definition simd.h:1005
constexpr R exp(R input) noexcept
Computes the exponential with the scalar FTZ underflow, overflow and special-value rules....
Definition simd.h:1026
constexpr R ceil(R value) noexcept
Rounds toward positive infinity, independently of ambient rounding mode; signed zero and infinities s...
Definition simd.h:1399
constexpr V & operator*=(V &a, basic_ftz32< Hardware > b) noexcept
Evaluates FTZ arithmetic before assigning to the existing raw-register destination.
Definition simd.h:795
constexpr R::mask isnan(R value) noexcept
Returns R::mask for NaN lanes by inspecting words; no FP evaluation or NaN quieting occurs.
Definition simd.h:1342
constexpr auto fma(A a, B b, C c) noexcept(std::is_nothrow_constructible_v< detail::ftz32_simd< detail::raw_family_t< A, B, C >, detail::scalar_family_t< A, B, C > >, A & > &&std::is_nothrow_constructible_v< detail::ftz32_simd< detail::raw_family_t< A, B, C >, detail::scalar_family_t< A, B, C > >, B & > &&std::is_nothrow_constructible_v< detail::ftz32_simd< detail::raw_family_t< A, B, C >, detail::scalar_family_t< A, B, C > >, C & >)
Imports raw SIMD operands into the FTZ scalar operand's policy and evaluates fused a*b+c....
Definition simd.h:840
constexpr R sin(R input) noexcept
Computes sine in radians with dedicated output reconstruction and the scalar FTZ special-value rules....
Definition simd.h:1015
constexpr auto operator<(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns the native comparison mask.
Definition simd.h:682
constexpr auto operator-(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns its FTZ vector rebind.
Definition simd.h:640
constexpr auto operator<=(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns the native comparison mask.
Definition simd.h:738
constexpr auto operator+(V a, basic_ftz32< Hardware > b) noexcept
Imports the raw register into the scalar operand's FTZ policy and returns its FTZ vector rebind.
Definition simd.h:626
constexpr V & operator/=(V &a, basic_ftz32< Hardware > b) noexcept
Evaluates FTZ arithmetic before assigning to the existing raw-register destination.
Definition simd.h:809
constexpr R copysign(R magnitude, R sign) noexcept
Returns magnitude with the sign bits of sign; all other words, including NaN payloads,...
Definition simd.h:1371
Binary32 value with a compile-time signed-FTZ policy.
Definition ftz.ccm:64
static constexpr basic_ftz32 unsafe_from_float32(float value) noexcept
Wraps float bits without classification or normalization.
Definition ftz.ccm:78
friend constexpr V masked_scaleb(M active, A prior, B value, simd exponent) noexcept
Forwards an FTZ exponent to raw scaling; the raw base and result retain their raw arithmetic contract...
Definition simd.h:445
constexpr simd_customization(float value) noexcept
Imports and broadcasts a float, replacing signed subnormals with signed zero.
Definition simd.h:229
friend constexpr simd fma(A a, simd b, B c) noexcept(noexcept(simd(a)) &&noexcept(simd(c)))
Evaluates fused a*b+c in this FTZ policy; converting operands can throw as specified by noexcept.
Definition simd.h:477
constexpr simd_customization(ftz32 value) noexcept
Broadcasts an already canonical FTZ scalar without normalization.
Definition simd.h:231
static constexpr simd unsafe_from_float32(float value) noexcept
Wraps raw float lanes unchanged; the caller must supply canonical FTZ values.
Definition simd.h:264
friend constexpr mask_type operator>=(simd a, simd b) noexcept
Compares lanes after conversion to this policy, returning the architecture-native mask type.
Definition simd.h:396
friend constexpr simd sqrt(simd a) noexcept
Evaluates the scalar FTZ square-root graph per lane, retaining signed zero and special values.
Definition simd.h:464
constexpr simd & operator/=(simd b) noexcept
Assigns the corresponding FTZ arithmetic result and returns this vector by reference.
Definition simd.h:368
constexpr void storeu(float *p) const noexcept
Stores N unaligned float elements without normalization.
Definition simd.h:312
constexpr void store_partial(float *p, std::size_t n) const noexcept
Stores the first n lanes; requires n <= lanes, and n == 0 does not access p.
Definition simd.h:321
static constexpr simd load_partial(float const *p, std::size_t n, float fill=0) noexcept
Imports n float elements and fills remaining lanes; requires n <= lanes, and n == 0 does not access p...
Definition simd.h:315
static constexpr simd load(float const *p) noexcept
Loads N unaligned float elements and normalizes signed subnormals.
Definition simd.h:304
static constexpr simd from_bits(bits_type value) noexcept
Imports float words, replacing signed subnormal words with signed zero.
Definition simd.h:270
constexpr V to_native() const noexcept
Returns the raw float register with every stored bit unchanged.
Definition simd.h:250
static constexpr simd unsafe_from_float32(V value) noexcept
Wraps raw float lanes unchanged; the caller must supply canonical FTZ values.
Definition simd.h:260
friend constexpr simd operator/(simd a, simd b) noexcept
Evaluates lane arithmetic in this FTZ policy; converting operands retain their conditional noexcept.
Definition simd.h:358
constexpr void store_bits(std::uint32_t *p) const noexcept
Stores N exact lane words without floating-point evaluation.
Definition simd.h:330
friend constexpr simd operator+(simd a) noexcept
Returns the input vector unchanged.
Definition simd.h:374
constexpr simd_customization(X... x) noexcept((noexcept(static_cast< float >(x)) &&...))
Imports one value per lane; conversion exceptions determine noexcept.
Definition simd.h:243
constexpr void store_bits_partial(std::uint32_t *p, std::size_t n) const noexcept
Stores n exact lane words; requires n <= lanes, with no access to p when n == 0.
Definition simd.h:338
friend constexpr mask_type operator<=(simd a, simd b) noexcept
Compares lanes after conversion to this policy, returning the architecture-native mask type.
Definition simd.h:393
static constexpr simd from_float(float value) noexcept
Imports and broadcasts a float, normalizing signed subnormals.
Definition simd.h:254
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Imports N float words and normalizes signed subnormal words.
Definition simd.h:326
friend constexpr V scaleb(A value, simd exponent) noexcept
Forwards an FTZ exponent to raw scaling without changing the raw base or result contract.
Definition simd.h:459
friend constexpr V masked_scaleb_zero(M active, A value, simd exponent) noexcept
Forwards an FTZ exponent to raw masked scaling with inactive lanes zeroed.
Definition simd.h:453
constexpr bits_type to_bits() const noexcept
Returns each lane as an unsigned 32-bit word, without floating-point evaluation.
Definition simd.h:252
friend constexpr simd masked_scaleb_zero(M active, A value, E exponent) noexcept(noexcept(scaling_exponent(exponent)))
Scales active lanes by 2^floor(exponent) and returns positive zero in inactive lanes.
Definition simd.h:423
constexpr bits_type bits() const noexcept
Returns the stored lane words without classification or normalization.
Definition simd.h:268
friend constexpr simd fma(A a, B b, simd c) noexcept(noexcept(simd(a)) &&noexcept(simd(b)))
Evaluates fused a*b+c in this FTZ policy; converting operands can throw as specified by noexcept.
Definition simd.h:485
constexpr void store(float *p) const noexcept
Stores N unaligned float elements without normalization.
Definition simd.h:310
constexpr simd & operator-=(simd b) noexcept
Assigns the corresponding FTZ arithmetic result and returns this vector by reference.
Definition simd.h:364
static constexpr simd load_bits_partial(std::uint32_t const *p, std::size_t n, std::uint32_t fill=0) noexcept
Imports n words and fills remaining lanes; requires n <= lanes, with no access to p when n == 0.
Definition simd.h:333
friend constexpr simd operator-(simd a) noexcept
Flips lane sign bits exactly, including signed zero and NaN payloads.
Definition simd.h:370
static constexpr simd from_native(V value) noexcept
Imports raw float lanes and normalizes signed subnormals.
Definition simd.h:256
friend constexpr simd scaleb(A value, E exponent) noexcept(noexcept(scaling_exponent(exponent)))
Scales each lane by 2^floor(exponent), retaining the FTZ result and boundary repair.
Definition simd.h:432
friend constexpr mask_type operator>(simd a, simd b) noexcept
Compares lanes after conversion to this policy, returning the architecture-native mask type.
Definition simd.h:384
friend constexpr simd operator*(simd a, simd b) noexcept
Evaluates lane arithmetic in this FTZ policy; converting operands retain their conditional noexcept.
Definition simd.h:353
friend constexpr mask_type operator!=(simd a, simd b) noexcept
Compares lanes after conversion to this policy, returning the architecture-native mask type.
Definition simd.h:390
static constexpr simd loadu(float const *p) noexcept
Loads N unaligned float elements and normalizes signed subnormals.
Definition simd.h:308
friend constexpr simd abs(simd a) noexcept
Clears each sign bit, preserving magnitude words including NaN payloads.
Definition simd.h:376
friend constexpr simd operator+(simd a, simd b) noexcept
Evaluates lane arithmetic in this FTZ policy; converting operands retain their conditional noexcept.
Definition simd.h:343
friend constexpr simd operator-(simd a, simd b) noexcept
Evaluates lane arithmetic in this FTZ policy; converting operands retain their conditional noexcept.
Definition simd.h:348
constexpr simd & operator*=(simd b) noexcept
Assigns the corresponding FTZ arithmetic result and returns this vector by reference.
Definition simd.h:366
constexpr simd_customization(std::array< T, N > const &values) noexcept
Loads exactly N float or same-policy FTZ elements.
Definition simd.h:247
constexpr simd & operator+=(simd b) noexcept
Assigns the corresponding FTZ arithmetic result and returns this vector by reference.
Definition simd.h:362
friend constexpr mask_type operator<(simd a, simd b) noexcept
Compares lanes after conversion to this policy, returning the architecture-native mask type.
Definition simd.h:381
constexpr void store_memory(T *p) const noexcept
Stores exactly N float or same-policy FTZ elements, preserving stored bits.
Definition simd.h:292
constexpr simd_customization() noexcept
Constructs positive zero in every lane.
Definition simd.h:227
constexpr simd_customization(V value) noexcept
Imports a raw float register, normalizing signed subnormal lanes.
Definition simd.h:233
constexpr simd_customization(native_type value) noexcept
Imports native storage through the normalizing raw-register constructor.
Definition simd.h:238
static constexpr simd load_memory(T const *p) noexcept
Loads exactly N elements; float memory is normalized, same-policy FTZ memory is copied exactly.
Definition simd.h:280
friend constexpr mask_type operator==(simd a, simd b) noexcept
Compares lanes after conversion to this policy, returning the architecture-native mask type.
Definition simd.h:387
friend constexpr simd masked_scaleb(M active, A prior, B value, E exponent) noexcept(noexcept(simd(prior)) &&noexcept(simd(value)) &&noexcept(scaling_exponent(exponent)))
Scales active lanes by 2^floor(exponent), preserving prior lanes elsewhere under the FTZ contract.
Definition simd.h:410
friend constexpr simd select(M mask, simd a, simd b) noexcept
Selects complete lanes from a when the mask is true, otherwise b, preserving exact words.
Definition simd.h:399
friend constexpr simd fma(simd a, A b, B c) noexcept(noexcept(simd(b)) &&noexcept(simd(c)))
Evaluates fused a*b+c in this FTZ policy; converting operands can throw as specified by noexcept.
Definition simd.h:470
static constexpr simd from_bits(std::uint32_t value) noexcept
Imports float words, replacing signed subnormal words with signed zero.
Definition simd.h:274