native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
native.numerics.ccm
1// SPDX-FileCopyrightText: 2024 Edward Kmett <ekmett@gmail.com>
2// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
4module;
5#include "native/isa_import.h"
6#include <bit>
7#include <cmath>
8#include <compare>
9#include <concepts>
10#include <cstddef>
11#include <cstdint>
12#include <limits>
13#include <type_traits>
14#include <utility>
15#include "native/attributes.h"
16#include "native/simd/common.h"
17
18#if !defined(NATIVE_HAS_NATIVE_FP16) || !defined(NATIVE_HAS_NATIVE_BF16)
19#error Configure native half capabilities before compiling native.numerics.
20#endif
21
22export module native.numerics;
23export import native.types;
24
25export namespace native {
26
34struct fp16 {
35private:
36 // Round binary32 to binary16, ties to even, independently of the FP controls.
37 // Raw native-half projections below remain representation copies.
38 static constexpr uint16_t encode(float value) noexcept {
39 auto word = std::bit_cast<uint32_t>(value);
40 auto sign = (word >> 16) & 0x8000u;
41 auto a = word & 0x7fffffffu;
42 if (a >= 0x7f800000u)
43 return uint16_t(sign | 0x7c00u | (a == 0x7f800000u ? 0u : 0x200u | ((a >> 13) & 0x3ffu)));
44 if (a >= 0x47800000u) return uint16_t(sign | 0x7c00u);
45 if (a >= 0x38800000u)
46 return uint16_t(sign | (((a + 0xfffu + ((a >> 13) & 1u)) >> 13) - 0x1c000u));
47 if (a < 0x33000000u) return uint16_t(sign);
48 auto shift = 126u - (a >> 23);
49 auto significand = (a & 0x7fffffu) | 0x800000u;
50 auto result = significand >> shift;
51 auto remainder = significand & ((1u << shift) - 1u);
52 auto half = 1u << (shift - 1u);
53 result += remainder > half || (remainder == half && (result & 1u));
54 return uint16_t(sign | result);
55 }
56 static constexpr float decode(uint16_t word) noexcept {
57 auto sign = uint32_t(word & 0x8000u) << 16;
58 auto exponent = (word >> 10) & 31u;
59 auto fraction = uint32_t(word & 0x3ffu);
60 uint32_t result;
61 if (exponent == 0) {
62 if (fraction == 0) return std::bit_cast<float>(sign);
63 auto shift = unsigned(std::countl_zero(fraction)) - 21u;
64 result = ((113u - shift) << 23) | (((fraction << shift) & 0x3ffu) << 13);
65 } else if (exponent == 31) {
66 result = 0x7f800000u | (fraction << 13) | (fraction ? 0x400000u : 0u);
67 } else result = ((exponent + 112u) << 23) | (fraction << 13);
68 return std::bit_cast<float>(sign | result);
69 }
70public:
72#if NATIVE_HAS_NATIVE_FP16
73 using underlying_type = _Float16;
74#else
75 using underlying_type = uint16_t;
76#endif
77
80
83 constexpr fp16() noexcept = default;
84
87 constexpr fp16(native_noescape fp16 const &) noexcept = default;
88
91 constexpr fp16(native_noescape fp16 &&) noexcept = default;
92
95 constexpr fp16(float value) noexcept : content(std::bit_cast<underlying_type>(encode(value))) {}
96
97#if NATIVE_HAS_NATIVE_FP16
100 constexpr fp16(_Float16 content) noexcept : content(content) {}
101#endif
102
105 constexpr operator float (this fp16 self) noexcept { return decode(self.to_bits()); }
106
107#if NATIVE_HAS_NATIVE_FP16
110 constexpr operator _Float16 (this fp16 self) noexcept { return self.content; }
111#endif
112
115 constexpr fp16 & operator = (native_noescape fp16 const &) noexcept
116 native_lifetimebound = default;
117
120 constexpr fp16 & operator = (native_noescape fp16 &&) noexcept
121 native_lifetimebound = default;
122
125 friend constexpr bool operator == (fp16 x, fp16 y) noexcept {
126 return static_cast<float>(x) == static_cast<float>(y);
127 }
128
131 friend constexpr bool operator != (fp16 x, fp16 y) noexcept {
132 return static_cast<float>(x) != static_cast<float>(y);
133 }
134
137 friend constexpr std::partial_ordering operator <=> (fp16 x, fp16 y) noexcept {
138 return static_cast<float>(x) <=> static_cast<float>(y);
139 }
140
143 friend constexpr bool operator < (fp16 x, fp16 y) noexcept {
144 return static_cast<float>(x) < static_cast<float>(y);
145 }
146
149 friend constexpr bool operator <= (fp16 x, fp16 y) noexcept {
150 return static_cast<float>(x) <= static_cast<float>(y);
151 }
152
154 friend constexpr bool operator > (fp16 x, fp16 y) noexcept {
155 return static_cast<float>(x) > static_cast<float>(y);
156 }
157
160 friend constexpr bool operator >= (fp16 x, fp16 y) noexcept {
161 return static_cast<float>(x) >= static_cast<float>(y);
162 }
163
166 friend constexpr void swap(
169 ) noexcept {
170 using std::swap;
171 swap(x.content,y.content);
172 }
173
176 static constexpr fp16 from_bits(uint16_t data) noexcept {
177 return std::bit_cast<fp16>(data);
178 }
179
182 constexpr uint16_t to_bits(this fp16 self) noexcept {
183 return std::bit_cast<uint16_t>(self.content);
184 }
185
186};
187
190template<> struct mask_traits<fp16> { using type = bool; };
191
194constexpr fp16 operator""_fp16(long double v) noexcept {
195 return fp16(static_cast<float>(v));
196}
197
198} // namespace native
199
202export namespace std {
203 using ::std::numeric_limits;
206 constexpr bool isnan(::native::fp16 x) noexcept {
207 // Sign doesn't matter, frac not zero (infinity)
208 return (x.to_bits() & 0x7FFF) > 0x7c00;
209 }
210
213 template <>
214 class numeric_limits<::native::fp16> {
215 public:
216 static constexpr bool is_specialized = true;
219 static constexpr ::native::fp16 min() noexcept {
220 return ::native::fp16::from_bits(0x0400);
221 }
222
224 static constexpr ::native::fp16 max() noexcept {
225 return ::native::fp16::from_bits(0x7BFF);
226 }
227
229 static constexpr ::native::fp16 lowest() noexcept {
230 return ::native::fp16::from_bits(0xFBFF);
231 }
232 static constexpr int digits = 11;
233 static constexpr int digits10 = 3;
234 static constexpr int max_digits10 = 5;
235 static constexpr bool is_signed = true;
236 static constexpr bool is_integer = false;
237 static constexpr bool is_exact = false;
238 static constexpr int radix = 2;
241 static constexpr ::native::fp16 epsilon() noexcept {
242 return ::native::fp16::from_bits(0x1400);
243 }
244
246 static constexpr ::native::fp16 round_error() noexcept {
247 return ::native::fp16::from_bits(0x3800);
248 }
249 static constexpr int min_exponent = -13;
250 static constexpr int min_exponent10 = -4;
251 static constexpr int max_exponent = 16;
252 static constexpr int max_exponent10 = 4;
253 static constexpr bool has_infinity = true;
254 static constexpr bool has_quiet_NaN = true;
255 static constexpr bool has_signaling_NaN = true;
256 [[deprecated("std::float_denorm_style is deprecated since C++23")]]
257 static constexpr float_denorm_style has_denorm = denorm_present;
258 [[deprecated("numeric_limits::has_denorm_loss is deprecated since C++23")]]
259 static constexpr bool has_denorm_loss = false;
262 static constexpr ::native::fp16 infinity() noexcept {
263 return ::native::fp16::from_bits(0x7C00);
264 }
265
267 static constexpr ::native::fp16 quiet_NaN() noexcept {
268 return ::native::fp16::from_bits(0x7FFF);
269 }
270
272 static constexpr ::native::fp16 signaling_NaN() noexcept {
273 return ::native::fp16::from_bits(0x7DFF);
274 }
275
277 static constexpr ::native::fp16 denorm_min() noexcept {
278 return ::native::fp16::from_bits(1);
279 }
280 static constexpr bool is_iec559 = false;
281 static constexpr bool is_bounded = true;
282 static constexpr bool is_modulo = false;
283 static constexpr bool traps = false;
284 static constexpr bool tinyness_before = false;
285 static constexpr float_round_style round_style = round_to_nearest;
286 };
287
288
289
290} // namespace std
291
292
293export namespace native {
294
300struct bf16 {
301private:
302 // Conversion is integer RNE, including subnormals, under every FP environment.
303 static constexpr uint16_t encode(float value) noexcept {
304 auto word = std::bit_cast<uint32_t>(value);
305 if ((word & 0x7fffffffu) > 0x7f800000u)
306 return uint16_t((word >> 16) | 0x40u);
307 return uint16_t((word + 0x7fffu + ((word >> 16) & 1u)) >> 16);
308 }
309 static constexpr float decode(uint16_t word) noexcept {
310 auto result = uint32_t(word) << 16;
311 if ((word & 0x7fffu) > 0x7f80u) result |= 0x400000u;
312 return std::bit_cast<float>(result);
313 }
314public:
316#if NATIVE_HAS_NATIVE_BF16
317 using underlying_type = __bf16;
318#else
319 using underlying_type = uint16_t;
320#endif
321
324
327 bf16() noexcept = default;
328
331 bf16(float f) noexcept : content(std::bit_cast<underlying_type>(encode(f))) {}
332
333#if NATIVE_HAS_NATIVE_BF16
336 bf16(__bf16 f) noexcept : content(f) {}
337#endif
338
341 bf16(native_noescape bf16 const &) noexcept = default;
342
345 bf16(native_noescape bf16 &&) noexcept = default;
346
347#if NATIVE_HAS_NATIVE_BF16
350 operator __bf16 (this bf16 self) noexcept { return self.content; }
351#endif
352
355 operator float (this bf16 self) noexcept { return decode(self.to_bits()); }
356
360 native_lifetimebound = default;
361
365 native_lifetimebound = default;
366
369 friend constexpr bool operator == (bf16 x, bf16 y) noexcept {
370 return static_cast<float>(x) == static_cast<float>(y);
371 }
372
375 friend constexpr std::partial_ordering operator <=> (bf16 x, bf16 y) noexcept {
376 return static_cast<float>(x) <=> static_cast<float>(y);
377 }
378
381 friend constexpr bool operator != (bf16 x, bf16 y) noexcept {
382 return static_cast<float>(x) != static_cast<float>(y);
383 }
384
387 friend constexpr bool operator < (bf16 x, bf16 y) noexcept {
388 return static_cast<float>(x) < static_cast<float>(y);
389 }
390
393 friend constexpr bool operator > (bf16 x, bf16 y) noexcept {
394 return static_cast<float>(x) > static_cast<float>(y);
395 }
396
399 friend constexpr bool operator <= (bf16 x, bf16 y) noexcept {
400 return static_cast<float>(x) <= static_cast<float>(y);
401 }
402
405 friend constexpr bool operator >= (bf16 x, bf16 y) noexcept {
406 return static_cast<float>(x) >= static_cast<float>(y);
407 }
408
411 friend constexpr void swap(
414 ) noexcept {
415 using std::swap;
416 swap(x.content,y.content);
417 }
418
421 static constexpr bf16 from_bits(uint16_t data) noexcept {
422 return std::bit_cast<bf16>(data);
423 }
424
427 constexpr uint16_t to_bits(this bf16 self) noexcept {
428 return std::bit_cast<uint16_t>(self.content);
429 }
430};
431
434template<> struct mask_traits<bf16> { using type = bool; };
435
438constexpr bf16 operator""_bf16(long double v) noexcept {
439 return bf16(static_cast<float>(v));
440}
441
446constexpr bf16 fast_to_bf16(float f) noexcept {
447 if consteval {
448 return f;
449 } else {
450 return std::bit_cast<bf16>(static_cast<uint16_t>(std::bit_cast<uint32_t>(f)>>16));
451 }
452}
453
456constexpr bf16 fast_to_bf16(bf16 f) noexcept {
457 return f;
458}
459
460} // namespace native
461
462export namespace std {
463 using ::std::numeric_limits;
466 constexpr bool isnan(::native::bf16 x) noexcept {
467 // Exponent maxed out and fraction non-zero indicates NaN
468 return (x.to_bits() & 0x7FFF) > 0x7F80;
469 }
470
473 template <> class numeric_limits<::native::bf16> {
474 public:
475 static constexpr bool is_specialized = true;
478 static constexpr ::native::bf16 min() noexcept {
479 return ::native::bf16::from_bits(0x0080);
480 }
481
483 static constexpr ::native::bf16 max() noexcept {
484 return ::native::bf16::from_bits(0x7F7F);
485 }
486
488 static constexpr ::native::bf16 lowest() noexcept {
489 return ::native::bf16::from_bits(0xFF7F);
490 }
491 static constexpr int digits = 8;
492 static constexpr int digits10 = 2;
493 static constexpr int max_digits10 = 4;
494 static constexpr bool is_signed = true;
495 static constexpr bool is_integer = false;
496 static constexpr bool is_exact = false;
497 static constexpr int radix = 2;
500 static constexpr ::native::bf16 epsilon() noexcept {
501 return ::native::bf16::from_bits(0x3C00);
502 }
503
505 static constexpr ::native::bf16 round_error() noexcept {
506 return ::native::bf16::from_bits(0x3F00);
507 }
508 static constexpr int min_exponent = -125;
509 static constexpr int min_exponent10 = -37;
510 static constexpr int max_exponent = 128;
511 static constexpr int max_exponent10 = 38;
512 static constexpr bool has_infinity = true;
513 static constexpr bool has_quiet_NaN = true;
514 static constexpr bool has_signaling_NaN = true;
515 [[deprecated("std::float_denorm_style is deprecated since C++23")]]
516 static constexpr float_denorm_style has_denorm = denorm_present;
517 [[deprecated("numeric_limits::has_denorm_loss is deprecated since C++23")]]
518 static constexpr bool has_denorm_loss = false;
521 static constexpr ::native::bf16 infinity() noexcept {
522 return ::native::bf16::from_bits(0x7F80);
523 }
524
526 static constexpr ::native::bf16 quiet_NaN() noexcept {
527 return ::native::bf16::from_bits(0x7FC0);
528 }
529
531 static constexpr ::native::bf16 signaling_NaN() noexcept {
532 return ::native::bf16::from_bits(0x7F81);
533 }
534
536 static constexpr ::native::bf16 denorm_min() noexcept {
537 return ::native::bf16::from_bits(1);
538 }
539 static constexpr bool is_iec559 = false;
540 static constexpr bool is_bounded = true;
541 static constexpr bool is_modulo = false;
542 static constexpr bool traps = false;
543 static constexpr bool tinyness_before = false;
544 static constexpr float_round_style round_style = round_to_nearest;
545 };
546
547
548
549} // namespace std
550
551
552namespace native { using std::isnan; }
553
554export namespace native {
555 using ::native::imm_t;
556 using ::native::imm;
559
561template <auto N, auto ... candidates>
562concept one_of = ((N==candidates) || ... || false);
563
565template <auto N, auto ... candidates>
566concept not_one_of = (!one_of<N,candidates...>);
567
569template <std::size_t N> struct integer_traits {};
570template <> struct integer_traits<8> { using signed_t = int8_t; using unsigned_t = uint8_t; };
571template <> struct integer_traits<16> { using signed_t = int16_t; using unsigned_t = uint16_t; };
572template <> struct integer_traits<32> { using signed_t = int32_t; using unsigned_t = uint32_t; };
573template <> struct integer_traits<64> { using signed_t = int64_t; using unsigned_t = uint64_t; };
575
578template <typename T>
579requires one_of<sizeof(T),1,2,4,8>
580using int_t = typename integer_traits<sizeof(T)*8>::signed_t;
581
584template <typename T>
585requires one_of<sizeof(T),1,2,4,8>
586using uint_t = typename integer_traits<sizeof(T)*8>::unsigned_t;
587
588// The shared immediate tag is defined by native/common.h.
589
596template <typename T>
598constexpr bool cmp_unord(T a, T b) noexcept(noexcept(isnan(a) || isnan(b))) {
599 return isnan(a) || isnan(b);
600}
601
603
604
606
613template <typename T>
615constexpr bool cmp_ord(T a, T b) noexcept(noexcept(!isnan(a) && !isnan(b))) {
616 return !isnan(a) && !isnan(b);
617}
618
620
621
623
632template <one_of_t<float, double> T>
634constexpr T scalef(T x, T y) noexcept {
635 if consteval {
636 using U = uint_t<T>;
637 constexpr int fraction_bits = std::numeric_limits<T>::digits - 1;
638 constexpr int bias = std::numeric_limits<T>::max_exponent - 1;
639 constexpr U hidden = U(1) << fraction_bits;
640 constexpr U sign_mask = U(1) << (sizeof(T) * 8 - 1);
641 constexpr U exponent_mask = U(2 * bias + 1) << fraction_bits;
642 auto const word = std::bit_cast<U>(x);
643 auto const sign = word & sign_mask;
644 auto const field = (word & exponent_mask) >> fraction_bits;
645 auto significand = word & (hidden - 1);
646 if (field == U(2 * bias + 1) || (field == 0 && significand == 0)) return x;
647 int const scale = static_cast<int>(y);
648
649 // Probe behavior, not just a library version: some constexpr math facilities
650 // reject range errors during constant evaluation, including signed underflow.
651 if constexpr (requires {
652 typename std::enable_if_t<
653 std::scalbn(T(1), 1 - bias) == std::numeric_limits<T>::min() &&
654 std::scalbn(T(1), 1 - bias - fraction_bits) == std::numeric_limits<T>::denorm_min() &&
655 std::scalbn(T(-1), bias + 1) == -std::numeric_limits<T>::infinity() &&
656 std::bit_cast<U>(std::scalbn(T(-1), -bias - fraction_bits)) == sign_mask &&
657 std::scalbn(T(1), std::numeric_limits<int>::min()) == T(0) &&
658 std::scalbn(T(1), std::numeric_limits<int>::max()) == std::numeric_limits<T>::infinity()>;
659 }) {
660 return std::scalbn(x, scale);
661 } else {
662 // Normalize even the smallest subnormal before changing its exponent.
663 std::int64_t exponent = field ? std::int64_t(field) - bias : 1 - bias;
664 if (field) significand |= hidden;
665 else while (significand < hidden) { significand <<= 1; --exponent; }
666 exponent += scale; // The full int exponent cannot overflow this wider sum.
667 if (exponent > bias) return std::bit_cast<T>(sign | exponent_mask);
668 if (exponent >= 1 - bias) {
669 return std::bit_cast<T>(sign | (U(exponent + bias) << fraction_bits) |
670 (significand & (hidden - 1)));
671 }
672 auto const shift = (1 - bias) - exponent;
673 if (shift > fraction_bits + 1) return std::bit_cast<T>(sign);
674 U const remainder = significand & ((U(1) << shift) - 1);
675 U const halfway = U(1) << (shift - 1);
676 U rounded = significand >> shift;
677 rounded += remainder > halfway || (remainder == halfway && (rounded & 1));
678 // Rounding can carry into the minimum normal; its encoding is hidden.
679 return std::bit_cast<T>(sign | rounded);
680 }
681 } else {
682 return std::scalbn(x, static_cast<int>(y));
683 }
684}
685
689enum class native_nodiscard CMPINT : std::size_t {
690 EQ = 0x0uz
691, LT = 0x1uz
692, LE = 0x2uz
693, FALSE = 0x3uz
694, NE = 0x4uz
695, NLT = 0x5uz
696, NLE = 0x6uz
697, TRUE = 0x7uz
698};
699
702template <CMPINT imm8, typename T>
703requires (one_of_t<T, uint8_t, int8_t, uint16_t, int16_t, uint32_t, int32_t, uint64_t, int64_t> && (std::size_t(imm8) < 8uz))
705constexpr bool cmpint(T a, T b) noexcept {
706 using enum CMPINT;
707 if constexpr (imm8 == TRUE) return -1;
708 else if constexpr (imm8 == FALSE) return 0;
709 else if constexpr (imm8 == LT) return a < b;
710 else if constexpr (imm8 == NLT) return a >= b;
711 else if constexpr (imm8 == LE) return a <= b;
712 else if constexpr (imm8 == NLE) return a > b;
713 else if constexpr (imm8 == EQ) return a == b;
714 else if constexpr (imm8 == NE) return a != b;
715 else static_assert(false);
716}
717
721inline constexpr std::size_t max_fp_comparison_predicate = [] {
722#ifdef __AVX512F__
723 return 32uz;
724#else
725 return 8uz;
726#endif
727}();
728
734enum class native_nodiscard CMP : std::size_t {
735 EQ_OQ = 0x00uz
736, LT_OS = 0x01uz
737, LE_OS = 0x02uz
738, UNORD_Q = 0x03uz
739, NEQ_UQ = 0x04uz
740, NLT_US = 0x05uz
741, NLE_US = 0x06uz
742, ORD_Q = 0x07uz
743, EQ_UQ = 0x08uz
744, NGE_US = 0x09uz
745, NGT_US = 0x0Auz
746, FALSE_OQ = 0x0Buz
747, NEQ_OQ = 0x0Cuz
748, GE_OS = 0x0Duz
749, GT_OS = 0x0Euz
750, TRUE_UQ = 0x0Fuz
751, EQ_OS = 0x10uz
752, LT_OQ = 0x11uz
753, LE_OQ = 0x12uz
754, UNORD_S = 0x13uz
755, NEQ_US = 0x14uz
756, NLT_UQ = 0x15uz
757, NLE_UQ = 0x16uz
758, ORD_S = 0x17uz
759, EQ_US = 0x18uz
760, NGE_UQ = 0x19uz
761, NGT_UQ = 0x1Auz
762, FALSE_OS = 0x1Buz
763, NEQ_OS = 0x1Cuz
764, GE_OQ = 0x1Duz
765, GT_OQ = 0x1Euz
766, TRUE_US = 0x1Fuz
767};
768
772template <CMP imm8, typename T>
773requires (one_of_t<T, float, double> && (std::size_t(imm8) < 32uz))
775constexpr bool cmp(T a, T b) noexcept {
776 using enum CMP;
777 if constexpr (imm8 == EQ_OQ) return cmp_ord(a, b) && (a == b);
778 else if constexpr (imm8 == LT_OS) return cmp_ord(a, b) && (a < b);
779 else if constexpr (imm8 == LE_OS) return cmp_ord(a, b) && (a <= b);
780 else if constexpr (imm8 == UNORD_Q) return cmp_unord(a, b);
781 else if constexpr (imm8 == NEQ_UQ) return cmp_unord(a, b) || (a != b);
782 else if constexpr (imm8 == NLT_US) return cmp_unord(a, b) || !(a < b);
783 else if constexpr (imm8 == NLE_US) return cmp_unord(a, b) || !(a <= b);
784 else if constexpr (imm8 == ORD_Q) return cmp_ord(a, b);
785 else if constexpr (imm8 == EQ_UQ) return cmp_unord(a, b) || (a == b);
786 else if constexpr (imm8 == NGE_US) return cmp_unord(a, b) || !(a >= b);
787 else if constexpr (imm8 == NGT_US) return cmp_unord(a, b) || !(a > b);
788 else if constexpr (imm8 == FALSE_OQ) return 0;
789 else if constexpr (imm8 == NEQ_OQ) return cmp_ord(a, b) && (a != b);
790 else if constexpr (imm8 == GE_OS) return cmp_ord(a, b) && (a >= b);
791 else if constexpr (imm8 == GT_OS) return cmp_ord(a, b) && (a > b);
792 else if constexpr (imm8 == TRUE_UQ) return -1;
793 else if constexpr (imm8 == EQ_OS) return cmp_ord(a, b) && (a == b);
794 else if constexpr (imm8 == LT_OQ) return cmp_ord(a, b) && (a < b);
795 else if constexpr (imm8 == LE_OQ) return cmp_ord(a, b) && (a <= b);
796 else if constexpr (imm8 == UNORD_S) return cmp_unord(a, b);
797 else if constexpr (imm8 == NEQ_US) return cmp_unord(a, b) || (a != b);
798 else if constexpr (imm8 == NLT_UQ) return cmp_unord(a, b) || !(a < b);
799 else if constexpr (imm8 == NLE_UQ) return cmp_unord(a, b) || !(a <= b);
800 else if constexpr (imm8 == ORD_S) return cmp_ord(a, b);
801 else if constexpr (imm8 == EQ_US) return cmp_unord(a, b) || (a == b);
802 else if constexpr (imm8 == NGE_UQ) return cmp_unord(a, b) || !(a >= b);
803 else if constexpr (imm8 == NGT_UQ) return cmp_unord(a, b) || !(a > b);
804 else if constexpr (imm8 == FALSE_OS) return 0;
805 else if constexpr (imm8 == NEQ_OS) return cmp_ord(a, b) && (a != b);
806 else if constexpr (imm8 == GE_OQ) return cmp_ord(a, b) && (a >= b);
807 else if constexpr (imm8 == GT_OQ) return cmp_ord(a, b) && (a > b);
808 else if constexpr (imm8 == TRUE_US) return -1;
809 else static_assert(false);
810}
811
812
813
814
816}
817
818
819namespace native {
821 template bool cmp_unord(float, float);
823 template bool cmp_unord(double, double);
825 template bool cmp_ord(float, float);
827 template bool cmp_ord(double, double);
830 template float scalef(float, float) noexcept;
833 template double scalef(double, double) noexcept;
834}
Compiler attributes for host code, with shader-safe shared modifiers.
static constexpr ::native::bf16 epsilon() noexcept
Return the distance from one to its successor, 2^-7 (bits 0x3c00).
static constexpr float_denorm_style has_denorm
static constexpr ::native::bf16 min() noexcept
Return the least positive normal value, 2^-126 (bits 0x0080).
static constexpr ::native::bf16 denorm_min() noexcept
Return the least positive subnormal value, 2^-133 (bits 0x0001).
static constexpr ::native::bf16 max() noexcept
Return the greatest finite value, (2 - 2^-7) * 2^127 (bits 0x7f7f).
static constexpr ::native::bf16 quiet_NaN() noexcept
Return a quiet NaN with the fixed encoding 0x7fc0.
static constexpr ::native::bf16 round_error() noexcept
Return the round-to-nearest error bound, one half (bits 0x3f00).
static constexpr ::native::bf16 signaling_NaN() noexcept
Return a signaling NaN encoding (bits 0x7f81); decoding to float quiets it.
static constexpr ::native::bf16 infinity() noexcept
Return positive infinity (bits 0x7f80).
static constexpr ::native::bf16 lowest() noexcept
Return the negative of max() (bits 0xff7f).
static constexpr ::native::fp16 round_error() noexcept
Return the round-to-nearest error bound, one half (bits 0x3800).
static constexpr ::native::fp16 min() noexcept
Return the least positive normal value, 2^-14 (bits 0x0400).
static constexpr ::native::fp16 max() noexcept
Return the greatest finite value, 65504 (bits 0x7bff).
static constexpr ::native::fp16 infinity() noexcept
Return positive infinity (bits 0x7c00).
static constexpr ::native::fp16 denorm_min() noexcept
Return the least positive subnormal value, 2^-24 (bits 0x0001).
static constexpr ::native::fp16 quiet_NaN() noexcept
Return a quiet NaN with the fixed encoding 0x7fff.
static constexpr ::native::fp16 signaling_NaN() noexcept
Return a signaling NaN encoding (bits 0x7dff); decoding to float quiets it.
static constexpr float_denorm_style has_denorm
static constexpr ::native::fp16 epsilon() noexcept
Return the distance from one to its successor, 2^-10 (bits 0x1400).
static constexpr ::native::fp16 lowest() noexcept
Return the most negative finite value, -65504 (bits 0xfbff).
Declares SIMD element types and pointer access policies.
N is not one of the candidates
N is one of the candidates
#define native_reinitializes
[[clang::reinitializes]]
Definition attributes.h:443
#define native_artificial
[[artificial]].
Definition attributes.h:245
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_noescape
portable __attribute__((noescape))
Definition attributes.h:175
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
#define native_lifetimebound
[[lifetimebound]]
Definition attributes.h:161
constexpr bool cmp_ord(T a, T b) noexcept(noexcept(!isnan(a) &&!isnan(b)))
Return true if neither argument is NaN.
constexpr std::size_t max_fp_comparison_predicate
constexpr T scalef(T x, T y) noexcept
typename integer_traits< sizeof(T) *8 >::unsigned_t uint_t
constexpr bool cmpint(T a, T b) noexcept
typename integer_traits< sizeof(T) *8 >::signed_t int_t
constexpr bool cmp(T a, T b) noexcept
CMP
Scalar floating-comparison predicates, including ordered and unordered truth tables.
CMPINT
Scalar integer-comparison predicates, encoded like the corresponding x86 immediates.
constexpr bool cmp_unord(T a, T b) noexcept(noexcept(isnan(a)||isnan(b)))
Return true if either argument is NaN.
@ LE_OQ
Less-than-or-equal (ordered, nonsignaling).
@ FALSE_OQ
False (ordered, nonsignaling).
@ NEQ_OS
Not-equal (ordered, signaling).
@ NLE_US
Not-less-than-or-equal (unordered, signaling).
@ TRUE_US
True (unordered, signaling).
@ NGE_UQ
Not-greater-than-or-equal (unordered, nonsignaling).
@ GE_OS
Greater-than-or-equal (ordered, signaling).
@ NEQ_OQ
Not-equal (ordered, nonsignaling).
@ ORD_Q
Ordered (nonsignaling).
@ LT_OS
Less-than (ordered, signaling).
@ GE_OQ
Greater-than-or-equal (ordered, nonsignaling).
@ EQ_US
Equal (unordered, signaling).
@ TRUE_UQ
True (unordered, nonsignaling).
@ EQ_OS
Equal (ordered, signaling).
@ FALSE_OS
False (ordered, signaling).
@ GT_OS
Greater-than (ordered, signaling).
@ ORD_S
Ordered (signaling).
@ NLE_UQ
Not-less-than-or-equal (unordered, nonsignaling).
@ GT_OQ
Greater-than (ordered, nonsignaling).
@ UNORD_S
Unordered (signaling).
@ LE_OS
Less-than-or-equal (ordered, signaling).
@ EQ_OQ
Equal (ordered, nonsignaling).
@ NEQ_UQ
Not-equal (unordered, nonsignaling).
@ EQ_UQ
Equal (unordered, nonsignaling).
@ NLT_US
Not-less-than (unordered, signaling).
@ UNORD_Q
Unordered (nonsignaling).
@ LT_OQ
Less-than (ordered, nonsignaling).
@ NLT_UQ
Not-less-than (unordered, nonsignaling).
@ NGT_US
Not-greater-than (unordered, signaling).
@ NGE_US
Not-greater-than-or-equal (unordered, signaling).
@ NGT_UQ
Not-greater-than (unordered, nonsignaling).
@ NEQ_US
Not-equal (unordered, signaling).
@ FALSE
always false
#define native_const
[[const]] is not const
Definition attributes.h:108
#define native_pure
[[pure]]
Definition attributes.h:126
constexpr imm_t< K > imm
Definition common.h:82
constexpr auto isnan(wide< T, N > const &input) noexcept(noexcept(detail::wide_generic_detail::wide_map< detail::wide_generic_detail::wide_isnan >(input)))
Definition wide.h:907
Scalar half storage, numeric limits and legacy scalar predicates.
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr bf16 fast_to_bf16(float f) noexcept
Standard-library adaptations documented here for SIMD value types.
constexpr bool isnan(::native::fp16 x) noexcept
Classify the stored bits as NaN without performing a floating-point conversion.
friend constexpr bool operator<(bf16 x, bf16 y) noexcept
Return whether the decoded left value is less; false for NaN operands.
constexpr uint16_t to_bits(this bf16 self) noexcept
Return the exact sixteen representation bits.
friend constexpr std::partial_ordering operator<=>(bf16 x, bf16 y) noexcept
Compare decoded values; NaNs are unordered and signed zeros equivalent.
constexpr bf16(bf16 &&) noexcept=default
Copy the native representation; the source retains its bits.
friend constexpr bool operator<=(bf16 x, bf16 y) noexcept
Return whether the decoded left value is at most the right; false for NaN operands.
constexpr bf16(__bf16 f) noexcept
Copy native storage without a numerical conversion.
constexpr bf16(bf16 const &) noexcept=default
Copy the native sixteen-bit representation.
friend constexpr bool operator>=(bf16 x, bf16 y) noexcept
Return whether the decoded left value is at least the right; false for NaN operands.
underlying_type content
Target-dependent storage. Use to_bits()/from_bits() for portable representation access.
friend constexpr bool operator!=(bf16 x, bf16 y) noexcept
Compare decoded values for inequality; a NaN operand makes the result true.
friend constexpr bool operator>(bf16 x, bf16 y) noexcept
Return whether the decoded left value is greater; false for NaN operands.
static constexpr bf16 from_bits(uint16_t data) noexcept
Adopt the exact sixteen representation bits without normalization.
friend constexpr void swap(bf16 &x, bf16 &y) noexcept
Exchange native representations, including signed zeros and NaN payloads.
constexpr bf16 & operator=(bf16 const &) noexcept=default
Copy the native representation, replacing all previous bits.
__bf16 underlying_type
Native half storage when supported by the target; otherwise sixteen raw bits.
constexpr bf16() noexcept=default
Default-initialize storage without setting its bits; value-initialization zeroes it.
friend constexpr std::partial_ordering operator<=>(fp16 x, fp16 y) noexcept
Compare decoded values; NaNs are unordered and signed zeros equivalent.
_Float16 underlying_type
Native half storage when supported by the target; otherwise sixteen raw bits.
constexpr fp16 & operator=(fp16 const &) noexcept=default
Copy the native representation, replacing all previous bits.
friend constexpr bool operator<(fp16 x, fp16 y) noexcept
Return whether the decoded left value is less; false for NaN operands.
constexpr uint16_t to_bits(this fp16 self) noexcept
Return the exact sixteen representation bits.
constexpr fp16(_Float16 content) noexcept
Copy native storage without a numerical conversion.
friend constexpr bool operator<=(fp16 x, fp16 y) noexcept
Return whether the decoded left value is at most the right; false for NaN operands.
static constexpr fp16 from_bits(uint16_t data) noexcept
Adopt the exact sixteen representation bits without normalization.
friend constexpr bool operator>=(fp16 x, fp16 y) noexcept
Return whether the decoded left value is at least the right; false for NaN operands.
underlying_type content
Target-dependent storage. Use to_bits()/from_bits() for portable representation access.
friend constexpr bool operator!=(fp16 x, fp16 y) noexcept
Compare decoded values for inequality; a NaN operand makes the result true.
friend constexpr bool operator>(fp16 x, fp16 y) noexcept
Return whether the decoded left value is greater; false for NaN operands.
constexpr fp16() noexcept=default
Default-initialize storage without setting its bits; value-initialization zeroes it.
friend constexpr void swap(fp16 &x, fp16 &y) noexcept
Exchange native representations, including signed zeros and NaN payloads.