3#include "native/isa_import.h"
4#include "native/arm/fcma.h"
5#include "native/detail/constexpr_float.h"
9#if NATIVE_HOST_NEON || defined(NATIVE_DOXYGEN)
10namespace native::detail {
11 template<
class T>
using fcma_format=std::conditional_t<std::same_as<T,fp16>,
12 constexpr_float::binary16,std::conditional_t<std::same_as<T,float>,
13 constexpr_float::binary32,constexpr_float::binary64>>;
15 template<
class T,std::
size_t N,isa<arm> Arch>
16 consteval auto fcma_bits(simd<T,N,Arch> value)
noexcept {
17 using word=
typename fcma_format<T>::bits_type;
18 std::array<word,N> result{};
19 if constexpr(std::same_as<T,double>) {
20 std::array<double,N> values{};value.store(values.data());
21 result=std::bit_cast<std::array<word,N>>(values);
22 }
else value.store_bits(result.data());
25 template<
class T,std::
size_t N,isa<arm> Arch>
26 consteval simd<T,N,Arch> fcma_from_bits(std::array<
typename fcma_format<T>::bits_type,N> bits)
noexcept {
27 if constexpr(std::same_as<T,double>) {
28 auto values=std::bit_cast<std::array<double,N>>(bits);
32 template<
unsigned Rotation,
class T,std::
size_t N,isa<arm> Arch>
33 consteval simd<T,N,Arch> fcadd_value(simd<T,N,Arch> a,simd<T,N,Arch> b)
noexcept {
34 using F=fcma_format<T>;
35 auto av=fcma_bits(a),bv=fcma_bits(b);
36 for(std::size_t i=0;i<N;i+=2) {
37 av[i]=constexpr_float::add_bits<F>(av[i],bv[i+1]^(Rotation==90?F::sign_mask:0));
38 av[i+1]=constexpr_float::add_bits<F>(av[i+1],bv[i]^(Rotation==270?F::sign_mask:0));
40 return fcma_from_bits<T,N,Arch>(av);
42 template<
unsigned Rotation,
int Lane,
class T,std::
size_t N,std::
size_t M,isa<arm> Arch>
43 consteval simd<T,N,Arch> fcmla_value(simd<T,N,Arch> acc,
44 simd<T,N,Arch> a,simd<T,M,Arch> b)
noexcept {
45 using F=fcma_format<T>;
46 auto cv=fcma_bits(acc);
auto av=fcma_bits(a);
auto bv=fcma_bits(b);
47 for(std::size_t i=0;i<N;i+=2) {
48 auto index=Lane<0?i:2*Lane;
49 auto factor=av[i+(Rotation==90 || Rotation==270)];
50 auto real=bv[index+(Rotation==90 || Rotation==270)];
51 auto imag=bv[index+(Rotation==0 || Rotation==180)];
52 if constexpr(Rotation==90 || Rotation==180) real^=F::sign_mask;
53 if constexpr(Rotation==180 || Rotation==270) imag^=F::sign_mask;
54 cv[i]=constexpr_float::fma_bits<F>(factor,real,cv[i]);
55 cv[i+1]=constexpr_float::fma_bits<F>(factor,imag,cv[i+1]);
57 return fcma_from_bits<T,N,Arch>(cv);
76 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum)
77 && (Rotation == 90 || Rotation == 270))
79 constexpr
simd<
float, 2, Arch>
fcadd(
simd<
float, 2, Arch> a,
simd<
float, 2, Arch> b) noexcept {
80 if consteval {
return detail::fcadd_value<Rotation>(a,b); }
else {
81 auto result = detail::fcadd<Arch, Rotation>(
82 vget_low_f32(a.to_native()),
83 vget_low_f32(b.to_native()));
89 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
90 && (Rotation == 90 || Rotation == 270))
93 return detail::fcadd_value<Rotation>(a,b);
97 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum)
98 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
104 if consteval {
return detail::fcmla_value<Rotation,-1>(acc,a,b); }
else {
105 auto result = detail::fcmla<Arch, Rotation>(
106 vget_low_f32(acc.to_native()),
107 vget_low_f32(a.to_native()),
108 vget_low_f32(b.to_native()));
114 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
115 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
121 return detail::fcmla_value<Rotation,-1>(acc,a,b);
125 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
126 requires(Arch.has(arm_feature::complxnum)
127 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
134 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
135 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
136 vget_low_f32(acc.to_native()),
137 vget_low_f32(a.to_native()),
138 vget_low_f32(b.to_native()));
144 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
145 requires(
requires {
sizeof(simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
146 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
153 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
157 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
158 requires(Arch.has(arm_feature::complxnum)
159 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
166 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
167 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
168 vget_low_f32(acc.to_native()),
169 vget_low_f32(a.to_native()),
170 __builtin_bit_cast(float32x4_t, b.to_native()));
176 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
177 requires(
requires {
sizeof(simd<float, 2, Arch>); } &&
requires {
sizeof(
simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
178 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
185 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
189 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum)
190 && (Rotation == 90 || Rotation == 270))
193 if consteval {
return detail::fcadd_value<Rotation>(a,b); }
else {
194 auto result = detail::fcadd<Arch, Rotation>(
195 __builtin_bit_cast(float32x4_t, a.to_native()),
196 __builtin_bit_cast(float32x4_t, b.to_native()));
202 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
203 && (Rotation == 90 || Rotation == 270))
206 return detail::fcadd_value<Rotation>(a,b);
210 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum)
211 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
217 if consteval {
return detail::fcmla_value<Rotation,-1>(acc,a,b); }
else {
218 auto result = detail::fcmla<Arch, Rotation>(
219 __builtin_bit_cast(float32x4_t, acc.to_native()),
220 __builtin_bit_cast(float32x4_t, a.to_native()),
221 __builtin_bit_cast(float32x4_t, b.to_native()));
227 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
228 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
234 return detail::fcmla_value<Rotation,-1>(acc,a,b);
238 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
239 requires(Arch.has(arm_feature::complxnum)
240 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
247 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
248 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
249 __builtin_bit_cast(float32x4_t, acc.to_native()),
250 __builtin_bit_cast(float32x4_t, a.to_native()),
251 vget_low_f32(b.to_native()));
257 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
258 requires(
requires {
sizeof(simd<float, 4, Arch>); } &&
requires {
sizeof(
simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
259 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
266 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
270 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
271 requires(Arch.has(arm_feature::complxnum)
272 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
279 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
280 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
281 __builtin_bit_cast(float32x4_t, acc.to_native()),
282 __builtin_bit_cast(float32x4_t, a.to_native()),
283 __builtin_bit_cast(float32x4_t, b.to_native()));
289 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
290 requires(
requires {
sizeof(simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
291 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
298 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
302 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum)
303 && (Rotation == 90 || Rotation == 270))
306 if consteval {
return detail::fcadd_value<Rotation>(a,b); }
else {
307 auto result = detail::fcadd<Arch, Rotation>(
308 __builtin_bit_cast(float64x2_t, a.to_native()),
309 __builtin_bit_cast(float64x2_t, b.to_native()));
315 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<double, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
316 && (Rotation == 90 || Rotation == 270))
319 return detail::fcadd_value<Rotation>(a,b);
323 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum)
324 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
330 if consteval {
return detail::fcmla_value<Rotation,-1>(acc,a,b); }
else {
331 auto result = detail::fcmla<Arch, Rotation>(
332 __builtin_bit_cast(float64x2_t, acc.to_native()),
333 __builtin_bit_cast(float64x2_t, a.to_native()),
334 __builtin_bit_cast(float64x2_t, b.to_native()));
340 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<double, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
341 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
347 return detail::fcmla_value<Rotation,-1>(acc,a,b);
351 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
352 && (Rotation == 90 || Rotation == 270))
355 if consteval {
return detail::fcadd_value<Rotation>(a,b); }
else {
356 auto result = detail::fcadd<Arch, Rotation>(
357 __builtin_bit_cast(float16x4_t, a.to_native()),
358 __builtin_bit_cast(float16x4_t, b.to_native()));
364 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
365 && (Rotation == 90 || Rotation == 270))
368 return detail::fcadd_value<Rotation>(a,b);
372 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
373 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
379 if consteval {
return detail::fcmla_value<Rotation,-1>(acc,a,b); }
else {
380 auto result = detail::fcmla<Arch, Rotation>(
381 __builtin_bit_cast(float16x4_t, acc.to_native()),
382 __builtin_bit_cast(float16x4_t, a.to_native()),
383 __builtin_bit_cast(float16x4_t, b.to_native()));
389 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
390 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
396 return detail::fcmla_value<Rotation,-1>(acc,a,b);
400 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
401 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
402 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
409 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
410 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
411 __builtin_bit_cast(float16x4_t, acc.to_native()),
412 __builtin_bit_cast(float16x4_t, a.to_native()),
413 __builtin_bit_cast(float16x4_t, b.to_native()));
419 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
420 requires(
requires {
sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
421 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
428 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
432 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
433 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
434 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
441 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
442 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
443 __builtin_bit_cast(float16x4_t, acc.to_native()),
444 __builtin_bit_cast(float16x4_t, a.to_native()),
445 __builtin_bit_cast(float16x8_t, b.to_native()));
451 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
452 requires(
requires {
sizeof(simd<fp16, 4, Arch>); } &&
requires {
sizeof(
simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
453 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
460 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
464 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
465 && (Rotation == 90 || Rotation == 270))
468 if consteval {
return detail::fcadd_value<Rotation>(a,b); }
else {
469 auto result = detail::fcadd<Arch, Rotation>(
470 __builtin_bit_cast(float16x8_t, a.to_native()),
471 __builtin_bit_cast(float16x8_t, b.to_native()));
477 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
478 && (Rotation == 90 || Rotation == 270))
481 return detail::fcadd_value<Rotation>(a,b);
485 template<isa<arm> Arch,
unsigned Rotation>
requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
486 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
492 if consteval {
return detail::fcmla_value<Rotation,-1>(acc,a,b); }
else {
493 auto result = detail::fcmla<Arch, Rotation>(
494 __builtin_bit_cast(float16x8_t, acc.to_native()),
495 __builtin_bit_cast(float16x8_t, a.to_native()),
496 __builtin_bit_cast(float16x8_t, b.to_native()));
502 template<isa<arm> Arch,
unsigned Rotation>
requires(
requires {
sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
503 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
509 return detail::fcmla_value<Rotation,-1>(acc,a,b);
513 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
514 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
515 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
522 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
523 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
524 __builtin_bit_cast(float16x8_t, acc.to_native()),
525 __builtin_bit_cast(float16x8_t, a.to_native()),
526 __builtin_bit_cast(float16x4_t, b.to_native()));
532 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
533 requires(
requires {
sizeof(simd<fp16, 8, Arch>); } &&
requires {
sizeof(
simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
534 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
541 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
545 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
546 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
547 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
554 if consteval {
return detail::fcmla_value<Rotation,Lane>(acc,a,b); }
else {
555 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
556 __builtin_bit_cast(float16x8_t, acc.to_native()),
557 __builtin_bit_cast(float16x8_t, a.to_native()),
558 __builtin_bit_cast(float16x8_t, b.to_native()));
564 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane>
565 requires(
requires {
sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
566 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
573 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
577 template<isa<arm> Arch,
unsigned Rotation,
class A,
class B>
578 void fcadd(A, B) =
delete;
579 template<isa<arm> Arch,
unsigned Rotation,
class A,
class B,
class C>
580 void fcmla(A, B, C) =
delete;
581 template<isa<arm> Arch,
unsigned Rotation,
unsigned Lane,
class A,
class B,
class C>
582 void fcmla_lane(A, B, C) =
delete;
constexpr simd< float, 2, Arch > fcmla_lane(simd< float, 2, Arch > acc, simd< float, 2, Arch > a, simd< float, 2, Arch > b) noexcept
FCMLA using complex pair Lane of b (the lane indexes pairs, not scalars).
constexpr simd< float, 2, Arch > fcadd(simd< float, 2, Arch > a, simd< float, 2, Arch > b) noexcept
FCADD on 1 binary32 complex pair(s), rotation in degrees.
constexpr simd< float, 2, Arch > fcmla(simd< float, 2, Arch > acc, simd< float, 2, Arch > a, simd< float, 2, Arch > b) noexcept
FCMLA on 1 binary32 complex pair(s), rotation in degrees.
#define native_inline
inline [[always_inline]]
#define native_nodiscard
C++17 [[nodiscard]].
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr int target
First matching requirement, with every later choice checked for shadowing.
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exactly the logical count of binary32 words without normalizing their representations.
static constexpr simd load(T const *p) noexcept
Read all logical lanes from an unaligned element pointer.
Omitted architecture arguments use the native.simd provider's baseline.