native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
native.arm.fcma.ccm
1// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
2module;
3#include "native/isa_import.h"
4#include "native/arm/fcma.h"
5#include "native/detail/constexpr_float.h"
6export module native.arm.fcma;
7export import native.arm.features;
8export import native.simd;
9#if NATIVE_HOST_NEON || defined(NATIVE_DOXYGEN)
10namespace native::detail {
11 template<class T> using fcma_format=std::conditional_t<std::same_as<T,fp16>,
12 constexpr_float::binary16,std::conditional_t<std::same_as<T,float>,
13 constexpr_float::binary32,constexpr_float::binary64>>;
14
15 template<class T,std::size_t N,isa<arm> Arch>
16 consteval auto fcma_bits(simd<T,N,Arch> value) noexcept {
17 using word=typename fcma_format<T>::bits_type;
18 std::array<word,N> result{};
19 if constexpr(std::same_as<T,double>) {
20 std::array<double,N> values{};value.store(values.data());
21 result=std::bit_cast<std::array<word,N>>(values);
22 } else value.store_bits(result.data());
23 return result;
24 }
25 template<class T,std::size_t N,isa<arm> Arch>
26 consteval simd<T,N,Arch> fcma_from_bits(std::array<typename fcma_format<T>::bits_type,N> bits) noexcept {
27 if constexpr(std::same_as<T,double>) {
28 auto values=std::bit_cast<std::array<double,N>>(bits);
29 return simd<T,N,Arch>::load(values.data());
30 } else return simd<T,N,Arch>::load_bits(bits.data());
31 }
32 template<unsigned Rotation,class T,std::size_t N,isa<arm> Arch>
33 consteval simd<T,N,Arch> fcadd_value(simd<T,N,Arch> a,simd<T,N,Arch> b) noexcept {
34 using F=fcma_format<T>;
35 auto av=fcma_bits(a),bv=fcma_bits(b);
36 for(std::size_t i=0;i<N;i+=2) {
37 av[i]=constexpr_float::add_bits<F>(av[i],bv[i+1]^(Rotation==90?F::sign_mask:0));
38 av[i+1]=constexpr_float::add_bits<F>(av[i+1],bv[i]^(Rotation==270?F::sign_mask:0));
39 }
40 return fcma_from_bits<T,N,Arch>(av);
41 }
42 template<unsigned Rotation,int Lane,class T,std::size_t N,std::size_t M,isa<arm> Arch>
43 consteval simd<T,N,Arch> fcmla_value(simd<T,N,Arch> acc,
44 simd<T,N,Arch> a,simd<T,M,Arch> b) noexcept {
45 using F=fcma_format<T>;
46 auto cv=fcma_bits(acc);auto av=fcma_bits(a);auto bv=fcma_bits(b);
47 for(std::size_t i=0;i<N;i+=2) {
48 auto index=Lane<0?i:2*Lane;
49 auto factor=av[i+(Rotation==90 || Rotation==270)];
50 auto real=bv[index+(Rotation==90 || Rotation==270)];
51 auto imag=bv[index+(Rotation==0 || Rotation==180)];
52 if constexpr(Rotation==90 || Rotation==180) real^=F::sign_mask;
53 if constexpr(Rotation==180 || Rotation==270) imag^=F::sign_mask;
54 cv[i]=constexpr_float::fma_bits<F>(factor,real,cv[i]);
55 cv[i+1]=constexpr_float::fma_bits<F>(factor,imag,cv[i+1]);
56 }
57 return fcma_from_bits<T,N,Arch>(cv);
58 }
59}
60export namespace native {
73
74 // All vector operands share Arch; native registers remain implementation details.
76 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum)
77 && (Rotation == 90 || Rotation == 270))
78 native_nodiscard native_inline __attribute__((target("complxnum")))
79 constexpr simd<float, 2, Arch> fcadd(simd<float, 2, Arch> a, simd<float, 2, Arch> b) noexcept {
80 if consteval { return detail::fcadd_value<Rotation>(a,b); } else {
81 auto result = detail::fcadd<Arch, Rotation>(
82 vget_low_f32(a.to_native()),
83 vget_low_f32(b.to_native()));
84 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
85 }
86 }
87
89 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
90 && (Rotation == 90 || Rotation == 270))
91 native_nodiscard consteval
93 return detail::fcadd_value<Rotation>(a,b);
94 }
95
97 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum)
98 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
99 native_nodiscard native_inline __attribute__((target("complxnum")))
103 simd<float, 2, Arch> b) noexcept {
104 if consteval { return detail::fcmla_value<Rotation,-1>(acc,a,b); } else {
105 auto result = detail::fcmla<Arch, Rotation>(
106 vget_low_f32(acc.to_native()),
107 vget_low_f32(a.to_native()),
108 vget_low_f32(b.to_native()));
109 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
110 }
111 }
112
114 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
115 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
116 native_nodiscard consteval
120 simd<float, 2, Arch> b) noexcept {
121 return detail::fcmla_value<Rotation,-1>(acc,a,b);
122 }
123
125 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
126 requires(Arch.has(arm_feature::complxnum)
127 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
128 && Lane < 1)
129 native_nodiscard native_inline __attribute__((target("complxnum")))
133 simd<float, 2, Arch> b) noexcept {
134 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
135 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
136 vget_low_f32(acc.to_native()),
137 vget_low_f32(a.to_native()),
138 vget_low_f32(b.to_native()));
139 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
140 }
141 }
142
144 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
145 requires(requires { sizeof(simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
146 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
147 && Lane < 1)
148 native_nodiscard consteval
152 simd<float, 2, Arch> b) noexcept {
153 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
154 }
155
157 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
158 requires(Arch.has(arm_feature::complxnum)
159 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
160 && Lane < 2)
161 native_nodiscard native_inline __attribute__((target("complxnum")))
165 simd<float, 4, Arch> b) noexcept {
166 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
167 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
168 vget_low_f32(acc.to_native()),
169 vget_low_f32(a.to_native()),
170 __builtin_bit_cast(float32x4_t, b.to_native()));
171 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
172 }
173 }
174
176 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
177 requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
178 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
179 && Lane < 2)
180 native_nodiscard consteval
184 simd<float, 4, Arch> b) noexcept {
185 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
186 }
187
189 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum)
190 && (Rotation == 90 || Rotation == 270))
191 native_nodiscard native_inline __attribute__((target("complxnum")))
193 if consteval { return detail::fcadd_value<Rotation>(a,b); } else {
194 auto result = detail::fcadd<Arch, Rotation>(
195 __builtin_bit_cast(float32x4_t, a.to_native()),
196 __builtin_bit_cast(float32x4_t, b.to_native()));
197 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
198 }
199 }
200
202 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
203 && (Rotation == 90 || Rotation == 270))
204 native_nodiscard consteval
206 return detail::fcadd_value<Rotation>(a,b);
207 }
208
210 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum)
211 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
212 native_nodiscard native_inline __attribute__((target("complxnum")))
216 simd<float, 4, Arch> b) noexcept {
217 if consteval { return detail::fcmla_value<Rotation,-1>(acc,a,b); } else {
218 auto result = detail::fcmla<Arch, Rotation>(
219 __builtin_bit_cast(float32x4_t, acc.to_native()),
220 __builtin_bit_cast(float32x4_t, a.to_native()),
221 __builtin_bit_cast(float32x4_t, b.to_native()));
222 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
223 }
224 }
225
227 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
228 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
229 native_nodiscard consteval
233 simd<float, 4, Arch> b) noexcept {
234 return detail::fcmla_value<Rotation,-1>(acc,a,b);
235 }
236
238 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
239 requires(Arch.has(arm_feature::complxnum)
240 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
241 && Lane < 1)
242 native_nodiscard native_inline __attribute__((target("complxnum")))
246 simd<float, 2, Arch> b) noexcept {
247 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
248 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
249 __builtin_bit_cast(float32x4_t, acc.to_native()),
250 __builtin_bit_cast(float32x4_t, a.to_native()),
251 vget_low_f32(b.to_native()));
252 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
253 }
254 }
255
257 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
258 requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<float, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
259 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
260 && Lane < 1)
261 native_nodiscard consteval
265 simd<float, 2, Arch> b) noexcept {
266 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
267 }
268
270 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
271 requires(Arch.has(arm_feature::complxnum)
272 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
273 && Lane < 2)
274 native_nodiscard native_inline __attribute__((target("complxnum")))
278 simd<float, 4, Arch> b) noexcept {
279 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
280 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
281 __builtin_bit_cast(float32x4_t, acc.to_native()),
282 __builtin_bit_cast(float32x4_t, a.to_native()),
283 __builtin_bit_cast(float32x4_t, b.to_native()));
284 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
285 }
286 }
287
289 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
290 requires(requires { sizeof(simd<float, 4, Arch>); } && !(Arch.has(arm_feature::complxnum))
291 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
292 && Lane < 2)
293 native_nodiscard consteval
297 simd<float, 4, Arch> b) noexcept {
298 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
299 }
300
302 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum)
303 && (Rotation == 90 || Rotation == 270))
304 native_nodiscard native_inline __attribute__((target("complxnum")))
306 if consteval { return detail::fcadd_value<Rotation>(a,b); } else {
307 auto result = detail::fcadd<Arch, Rotation>(
308 __builtin_bit_cast(float64x2_t, a.to_native()),
309 __builtin_bit_cast(float64x2_t, b.to_native()));
310 return simd<double, 2, Arch>::from_native(__builtin_bit_cast(typename simd<double, 2, Arch>::native_type, result));
311 }
312 }
313
315 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<double, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
316 && (Rotation == 90 || Rotation == 270))
317 native_nodiscard consteval
319 return detail::fcadd_value<Rotation>(a,b);
320 }
321
323 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum)
324 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
325 native_nodiscard native_inline __attribute__((target("complxnum")))
329 simd<double, 2, Arch> b) noexcept {
330 if consteval { return detail::fcmla_value<Rotation,-1>(acc,a,b); } else {
331 auto result = detail::fcmla<Arch, Rotation>(
332 __builtin_bit_cast(float64x2_t, acc.to_native()),
333 __builtin_bit_cast(float64x2_t, a.to_native()),
334 __builtin_bit_cast(float64x2_t, b.to_native()));
335 return simd<double, 2, Arch>::from_native(__builtin_bit_cast(typename simd<double, 2, Arch>::native_type, result));
336 }
337 }
338
340 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<double, 2, Arch>); } && !(Arch.has(arm_feature::complxnum))
341 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
342 native_nodiscard consteval
346 simd<double, 2, Arch> b) noexcept {
347 return detail::fcmla_value<Rotation,-1>(acc,a,b);
348 }
349
351 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
352 && (Rotation == 90 || Rotation == 270))
353 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
355 if consteval { return detail::fcadd_value<Rotation>(a,b); } else {
356 auto result = detail::fcadd<Arch, Rotation>(
357 __builtin_bit_cast(float16x4_t, a.to_native()),
358 __builtin_bit_cast(float16x4_t, b.to_native()));
359 return simd<fp16, 4, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 4, Arch>::native_type, result));
360 }
361 }
362
364 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
365 && (Rotation == 90 || Rotation == 270))
366 native_nodiscard consteval
368 return detail::fcadd_value<Rotation>(a,b);
369 }
370
372 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
373 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
374 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
378 simd<fp16, 4, Arch> b) noexcept {
379 if consteval { return detail::fcmla_value<Rotation,-1>(acc,a,b); } else {
380 auto result = detail::fcmla<Arch, Rotation>(
381 __builtin_bit_cast(float16x4_t, acc.to_native()),
382 __builtin_bit_cast(float16x4_t, a.to_native()),
383 __builtin_bit_cast(float16x4_t, b.to_native()));
384 return simd<fp16, 4, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 4, Arch>::native_type, result));
385 }
386 }
387
389 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
390 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
391 native_nodiscard consteval
395 simd<fp16, 4, Arch> b) noexcept {
396 return detail::fcmla_value<Rotation,-1>(acc,a,b);
397 }
398
400 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
401 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
402 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
403 && Lane < 2)
404 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
408 simd<fp16, 4, Arch> b) noexcept {
409 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
410 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
411 __builtin_bit_cast(float16x4_t, acc.to_native()),
412 __builtin_bit_cast(float16x4_t, a.to_native()),
413 __builtin_bit_cast(float16x4_t, b.to_native()));
414 return simd<fp16, 4, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 4, Arch>::native_type, result));
415 }
416 }
417
419 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
420 requires(requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
421 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
422 && Lane < 2)
423 native_nodiscard consteval
427 simd<fp16, 4, Arch> b) noexcept {
428 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
429 }
430
432 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
433 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
434 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
435 && Lane < 4)
436 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
440 simd<fp16, 8, Arch> b) noexcept {
441 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
442 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
443 __builtin_bit_cast(float16x4_t, acc.to_native()),
444 __builtin_bit_cast(float16x4_t, a.to_native()),
445 __builtin_bit_cast(float16x8_t, b.to_native()));
446 return simd<fp16, 4, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 4, Arch>::native_type, result));
447 }
448 }
449
451 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
452 requires(requires { sizeof(simd<fp16, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
453 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
454 && Lane < 4)
455 native_nodiscard consteval
459 simd<fp16, 8, Arch> b) noexcept {
460 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
461 }
462
464 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
465 && (Rotation == 90 || Rotation == 270))
466 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
468 if consteval { return detail::fcadd_value<Rotation>(a,b); } else {
469 auto result = detail::fcadd<Arch, Rotation>(
470 __builtin_bit_cast(float16x8_t, a.to_native()),
471 __builtin_bit_cast(float16x8_t, b.to_native()));
472 return simd<fp16, 8, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 8, Arch>::native_type, result));
473 }
474 }
475
477 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
478 && (Rotation == 90 || Rotation == 270))
479 native_nodiscard consteval
481 return detail::fcadd_value<Rotation>(a,b);
482 }
483
485 template<isa<arm> Arch, unsigned Rotation> requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
486 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
487 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
491 simd<fp16, 8, Arch> b) noexcept {
492 if consteval { return detail::fcmla_value<Rotation,-1>(acc,a,b); } else {
493 auto result = detail::fcmla<Arch, Rotation>(
494 __builtin_bit_cast(float16x8_t, acc.to_native()),
495 __builtin_bit_cast(float16x8_t, a.to_native()),
496 __builtin_bit_cast(float16x8_t, b.to_native()));
497 return simd<fp16, 8, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 8, Arch>::native_type, result));
498 }
499 }
500
502 template<isa<arm> Arch, unsigned Rotation> requires(requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
503 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270))
504 native_nodiscard consteval
508 simd<fp16, 8, Arch> b) noexcept {
509 return detail::fcmla_value<Rotation,-1>(acc,a,b);
510 }
511
513 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
514 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
515 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
516 && Lane < 2)
517 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
521 simd<fp16, 4, Arch> b) noexcept {
522 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
523 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
524 __builtin_bit_cast(float16x8_t, acc.to_native()),
525 __builtin_bit_cast(float16x8_t, a.to_native()),
526 __builtin_bit_cast(float16x4_t, b.to_native()));
527 return simd<fp16, 8, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 8, Arch>::native_type, result));
528 }
529 }
530
532 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
533 requires(requires { sizeof(simd<fp16, 8, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
534 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
535 && Lane < 2)
536 native_nodiscard consteval
540 simd<fp16, 4, Arch> b) noexcept {
541 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
542 }
543
545 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
546 requires(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16)
547 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
548 && Lane < 4)
549 native_nodiscard native_inline __attribute__((target("complxnum,fullfp16")))
553 simd<fp16, 8, Arch> b) noexcept {
554 if consteval { return detail::fcmla_value<Rotation,Lane>(acc,a,b); } else {
555 auto result = detail::fcmla_lane<Arch, Rotation, Lane>(
556 __builtin_bit_cast(float16x8_t, acc.to_native()),
557 __builtin_bit_cast(float16x8_t, a.to_native()),
558 __builtin_bit_cast(float16x8_t, b.to_native()));
559 return simd<fp16, 8, Arch>::from_native(__builtin_bit_cast(typename simd<fp16, 8, Arch>::native_type, result));
560 }
561 }
562
564 template<isa<arm> Arch, unsigned Rotation, unsigned Lane>
565 requires(requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::complxnum) && Arch.has(arm_feature::neon_fp16))
566 && (Rotation == 0 || Rotation == 90 || Rotation == 180 || Rotation == 270)
567 && Lane < 4)
568 native_nodiscard consteval
572 simd<fp16, 8, Arch> b) noexcept {
573 return detail::fcmla_value<Rotation,Lane>(acc,a,b);
574 }
575
577 template<isa<arm> Arch, unsigned Rotation, class A, class B>
578 void fcadd(A, B) = delete;
579 template<isa<arm> Arch, unsigned Rotation, class A, class B, class C>
580 void fcmla(A, B, C) = delete;
581 template<isa<arm> Arch, unsigned Rotation, unsigned Lane, class A, class B, class C>
582 void fcmla_lane(A, B, C) = delete;
585}
586#endif
constexpr simd< float, 2, Arch > fcmla_lane(simd< float, 2, Arch > acc, simd< float, 2, Arch > a, simd< float, 2, Arch > b) noexcept
FCMLA using complex pair Lane of b (the lane indexes pairs, not scalars).
constexpr simd< float, 2, Arch > fcadd(simd< float, 2, Arch > a, simd< float, 2, Arch > b) noexcept
FCADD on 1 binary32 complex pair(s), rotation in degrees.
constexpr simd< float, 2, Arch > fcmla(simd< float, 2, Arch > acc, simd< float, 2, Arch > a, simd< float, 2, Arch > b) noexcept
FCMLA on 1 binary32 complex pair(s), rotation in degrees.
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr int target
First matching requirement, with every later choice checked for shadowing.
Definition isa.h:396
static constexpr simd load_bits(std::uint32_t const *p) noexcept
Load exactly the logical count of binary32 words without normalizing their representations.
static constexpr simd load(T const *p) noexcept
Read all logical lanes from an unaligned element pointer.
Omitted architecture arguments use the native.simd provider's baseline.