3#include "native/isa_import.h"
4#include "native/arm/fp16fml.h"
5#include "native/detail/constexpr_float.h"
9#if NATIVE_HOST_NEON || defined(NATIVE_DOXYGEN)
10namespace native::detail {
11 template<isa<arm> Arch,
bool Subtract,
bool High,
int Lane,std::
size_t N,std::
size_t M>
12 consteval simd<float,N,Arch> fp16fml_value(simd<float,N,Arch> acc,
13 simd<fp16,2*N,Arch> a,simd<fp16,M,Arch> b)
noexcept {
14 namespace cf=constexpr_float;
15 std::array<std::uint32_t,N> c{};
16 std::array<std::uint16_t,2*N> av{};
17 std::array<std::uint16_t,M> bv{};
18 acc.store_bits(c.data());a.store_bits(av.data());b.store_bits(bv.data());
19 auto widen=[](std::uint16_t x) {
20 return cf::is_nan<cf::binary16>(x)?cf::resize_nan<cf::binary32,cf::binary16>(x,
false)
21 :cf::convert_bits<cf::binary32,cf::binary16>(x);
23 for(std::size_t i=0;i<N;++i) {
24 auto index=i+(High?N:0);
26 if constexpr(Subtract) x^=0x8000;
27 c[i]=cf::fma_bits<cf::binary32>(widen(x),widen(bv[Lane<0?index:Lane]),c[i]);
29 return simd<float,N,Arch>::load_bits(c.data());
46 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
49 simd<
float, 2, Arch> acc,
52 if consteval {
return detail::fp16fml_value<Arch,
false,
false,-1>(acc,a,b); }
else {
53 auto result = detail::fmlal<Arch>(
54 vget_low_f32(acc.to_native()),
55 __builtin_bit_cast(float16x4_t, a.to_native()),
56 __builtin_bit_cast(float16x4_t, b.to_native()));
62 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
68 return detail::fp16fml_value<Arch,
false,
false,-1>(acc,a,b);
72 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
78 if consteval {
return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); }
else {
79 auto result = detail::fmlal_lane<Arch, Lane>(
80 vget_low_f32(acc.to_native()),
81 __builtin_bit_cast(float16x4_t, a.to_native()),
82 __builtin_bit_cast(float16x4_t, b.to_native()));
88 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
94 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
98 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
104 if consteval {
return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); }
else {
105 auto result = detail::fmlal_lane<Arch, Lane>(
106 vget_low_f32(acc.to_native()),
107 __builtin_bit_cast(float16x4_t, a.to_native()),
108 __builtin_bit_cast(float16x8_t, b.to_native()));
114 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
120 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
124 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
130 if consteval {
return detail::fp16fml_value<Arch,
false,
false,-1>(acc,a,b); }
else {
131 auto result = detail::fmlal<Arch>(
132 __builtin_bit_cast(float32x4_t, acc.to_native()),
133 __builtin_bit_cast(float16x8_t, a.to_native()),
134 __builtin_bit_cast(float16x8_t, b.to_native()));
140 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
146 return detail::fp16fml_value<Arch,
false,
false,-1>(acc,a,b);
150 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
156 if consteval {
return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); }
else {
157 auto result = detail::fmlal_lane<Arch, Lane>(
158 __builtin_bit_cast(float32x4_t, acc.to_native()),
159 __builtin_bit_cast(float16x8_t, a.to_native()),
160 __builtin_bit_cast(float16x4_t, b.to_native()));
166 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
172 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
176 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
182 if consteval {
return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); }
else {
183 auto result = detail::fmlal_lane<Arch, Lane>(
184 __builtin_bit_cast(float32x4_t, acc.to_native()),
185 __builtin_bit_cast(float16x8_t, a.to_native()),
186 __builtin_bit_cast(float16x8_t, b.to_native()));
192 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
198 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
202 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
208 if consteval {
return detail::fp16fml_value<Arch,
false,
true,-1>(acc,a,b); }
else {
209 auto result = detail::fmlal2<Arch>(
210 vget_low_f32(acc.to_native()),
211 __builtin_bit_cast(float16x4_t, a.to_native()),
212 __builtin_bit_cast(float16x4_t, b.to_native()));
218 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
224 return detail::fp16fml_value<Arch,
false,
true,-1>(acc,a,b);
228 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
234 if consteval {
return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); }
else {
235 auto result = detail::fmlal2_lane<Arch, Lane>(
236 vget_low_f32(acc.to_native()),
237 __builtin_bit_cast(float16x4_t, a.to_native()),
238 __builtin_bit_cast(float16x4_t, b.to_native()));
244 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
250 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
254 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
260 if consteval {
return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); }
else {
261 auto result = detail::fmlal2_lane<Arch, Lane>(
262 vget_low_f32(acc.to_native()),
263 __builtin_bit_cast(float16x4_t, a.to_native()),
264 __builtin_bit_cast(float16x8_t, b.to_native()));
270 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
276 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
280 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
286 if consteval {
return detail::fp16fml_value<Arch,
false,
true,-1>(acc,a,b); }
else {
287 auto result = detail::fmlal2<Arch>(
288 __builtin_bit_cast(float32x4_t, acc.to_native()),
289 __builtin_bit_cast(float16x8_t, a.to_native()),
290 __builtin_bit_cast(float16x8_t, b.to_native()));
296 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
302 return detail::fp16fml_value<Arch,
false,
true,-1>(acc,a,b);
306 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
312 if consteval {
return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); }
else {
313 auto result = detail::fmlal2_lane<Arch, Lane>(
314 __builtin_bit_cast(float32x4_t, acc.to_native()),
315 __builtin_bit_cast(float16x8_t, a.to_native()),
316 __builtin_bit_cast(float16x4_t, b.to_native()));
322 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
328 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
332 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
338 if consteval {
return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); }
else {
339 auto result = detail::fmlal2_lane<Arch, Lane>(
340 __builtin_bit_cast(float32x4_t, acc.to_native()),
341 __builtin_bit_cast(float16x8_t, a.to_native()),
342 __builtin_bit_cast(float16x8_t, b.to_native()));
348 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
354 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
358 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
364 if consteval {
return detail::fp16fml_value<Arch,
true,
false,-1>(acc,a,b); }
else {
365 auto result = detail::fmlsl<Arch>(
366 vget_low_f32(acc.to_native()),
367 __builtin_bit_cast(float16x4_t, a.to_native()),
368 __builtin_bit_cast(float16x4_t, b.to_native()));
374 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
380 return detail::fp16fml_value<Arch,
true,
false,-1>(acc,a,b);
384 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
390 if consteval {
return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); }
else {
391 auto result = detail::fmlsl_lane<Arch, Lane>(
392 vget_low_f32(acc.to_native()),
393 __builtin_bit_cast(float16x4_t, a.to_native()),
394 __builtin_bit_cast(float16x4_t, b.to_native()));
400 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
406 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
410 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
416 if consteval {
return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); }
else {
417 auto result = detail::fmlsl_lane<Arch, Lane>(
418 vget_low_f32(acc.to_native()),
419 __builtin_bit_cast(float16x4_t, a.to_native()),
420 __builtin_bit_cast(float16x8_t, b.to_native()));
426 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
432 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
436 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
442 if consteval {
return detail::fp16fml_value<Arch,
true,
false,-1>(acc,a,b); }
else {
443 auto result = detail::fmlsl<Arch>(
444 __builtin_bit_cast(float32x4_t, acc.to_native()),
445 __builtin_bit_cast(float16x8_t, a.to_native()),
446 __builtin_bit_cast(float16x8_t, b.to_native()));
452 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
458 return detail::fp16fml_value<Arch,
true,
false,-1>(acc,a,b);
462 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
468 if consteval {
return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); }
else {
469 auto result = detail::fmlsl_lane<Arch, Lane>(
470 __builtin_bit_cast(float32x4_t, acc.to_native()),
471 __builtin_bit_cast(float16x8_t, a.to_native()),
472 __builtin_bit_cast(float16x4_t, b.to_native()));
478 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
484 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
488 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
494 if consteval {
return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); }
else {
495 auto result = detail::fmlsl_lane<Arch, Lane>(
496 __builtin_bit_cast(float32x4_t, acc.to_native()),
497 __builtin_bit_cast(float16x8_t, a.to_native()),
498 __builtin_bit_cast(float16x8_t, b.to_native()));
504 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
510 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
514 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
520 if consteval {
return detail::fp16fml_value<Arch,
true,
true,-1>(acc,a,b); }
else {
521 auto result = detail::fmlsl2<Arch>(
522 vget_low_f32(acc.to_native()),
523 __builtin_bit_cast(float16x4_t, a.to_native()),
524 __builtin_bit_cast(float16x4_t, b.to_native()));
530 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
536 return detail::fp16fml_value<Arch,
true,
true,-1>(acc,a,b);
540 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
546 if consteval {
return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); }
else {
547 auto result = detail::fmlsl2_lane<Arch, Lane>(
548 vget_low_f32(acc.to_native()),
549 __builtin_bit_cast(float16x4_t, a.to_native()),
550 __builtin_bit_cast(float16x4_t, b.to_native()));
556 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
562 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
566 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
572 if consteval {
return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); }
else {
573 auto result = detail::fmlsl2_lane<Arch, Lane>(
574 vget_low_f32(acc.to_native()),
575 __builtin_bit_cast(float16x4_t, a.to_native()),
576 __builtin_bit_cast(float16x8_t, b.to_native()));
582 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
588 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
592 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
598 if consteval {
return detail::fp16fml_value<Arch,
true,
true,-1>(acc,a,b); }
else {
599 auto result = detail::fmlsl2<Arch>(
600 __builtin_bit_cast(float32x4_t, acc.to_native()),
601 __builtin_bit_cast(float16x8_t, a.to_native()),
602 __builtin_bit_cast(float16x8_t, b.to_native()));
608 template<isa<arm> Arch>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
614 return detail::fp16fml_value<Arch,
true,
true,-1>(acc,a,b);
618 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
624 if consteval {
return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); }
else {
625 auto result = detail::fmlsl2_lane<Arch, Lane>(
626 __builtin_bit_cast(float32x4_t, acc.to_native()),
627 __builtin_bit_cast(float16x8_t, a.to_native()),
628 __builtin_bit_cast(float16x4_t, b.to_native()));
634 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
640 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
644 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
650 if consteval {
return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); }
else {
651 auto result = detail::fmlsl2_lane<Arch, Lane>(
652 __builtin_bit_cast(float32x4_t, acc.to_native()),
653 __builtin_bit_cast(float16x8_t, a.to_native()),
654 __builtin_bit_cast(float16x8_t, b.to_native()));
660 template<isa<arm> Arch,
unsigned Lane>
requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
666 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
670 template<isa<arm> Arch,
class A,
class B,
class C>
671 void fmlal(A, B, C) =
delete;
672 template<isa<arm> Arch,
unsigned Lane,
class A,
class B,
class C>
673 void fmlal_lane(A, B, C) =
delete;
674 template<isa<arm> Arch,
class A,
class B,
class C>
675 void fmlal2(A, B, C) =
delete;
676 template<isa<arm> Arch,
unsigned Lane,
class A,
class B,
class C>
677 void fmlal2_lane(A, B, C) =
delete;
678 template<isa<arm> Arch,
class A,
class B,
class C>
679 void fmlsl(A, B, C) =
delete;
680 template<isa<arm> Arch,
unsigned Lane,
class A,
class B,
class C>
681 void fmlsl_lane(A, B, C) =
delete;
682 template<isa<arm> Arch,
class A,
class B,
class C>
683 void fmlsl2(A, B, C) =
delete;
684 template<isa<arm> Arch,
unsigned Lane,
class A,
class B,
class C>
685 void fmlsl2_lane(A, B, C) =
delete;
constexpr simd< float, 2, Arch > fmlal(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Add products from the low 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Subtract products from the low 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl2_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLSL2 with b[Lane] broadcast; selects the high 2 lanes of a.
constexpr simd< float, 2, Arch > fmlal_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLAL with b[Lane] broadcast; selects the low 2 lanes of a.
constexpr simd< float, 2, Arch > fmlal2(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Add products from the high 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl2(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Subtract products from the high 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlal2_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLAL2 with b[Lane] broadcast; selects the high 2 lanes of a.
constexpr simd< float, 2, Arch > fmlsl_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLSL with b[Lane] broadcast; selects the low 2 lanes of a.
#define native_inline
inline [[always_inline]]
#define native_nodiscard
C++17 [[nodiscard]].
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr int target
First matching requirement, with every later choice checked for shadowing.
Omitted architecture arguments use the native.simd provider's baseline.