native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
native.arm.fp16fml.ccm
1// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
2module;
3#include "native/isa_import.h"
4#include "native/arm/fp16fml.h"
5#include "native/detail/constexpr_float.h"
6export module native.arm.fp16fml;
7export import native.arm.features;
8export import native.simd;
9#if NATIVE_HOST_NEON || defined(NATIVE_DOXYGEN)
10namespace native::detail {
11 template<isa<arm> Arch,bool Subtract,bool High,int Lane,std::size_t N,std::size_t M>
12 consteval simd<float,N,Arch> fp16fml_value(simd<float,N,Arch> acc,
13 simd<fp16,2*N,Arch> a,simd<fp16,M,Arch> b) noexcept {
14 namespace cf=constexpr_float;
15 std::array<std::uint32_t,N> c{};
16 std::array<std::uint16_t,2*N> av{};
17 std::array<std::uint16_t,M> bv{};
18 acc.store_bits(c.data());a.store_bits(av.data());b.store_bits(bv.data());
19 auto widen=[](std::uint16_t x) {
20 return cf::is_nan<cf::binary16>(x)?cf::resize_nan<cf::binary32,cf::binary16>(x,false)
21 :cf::convert_bits<cf::binary32,cf::binary16>(x);
22 };
23 for(std::size_t i=0;i<N;++i) {
24 auto index=i+(High?N:0);
25 auto x=av[index];
26 if constexpr(Subtract) x^=0x8000;
27 c[i]=cf::fma_bits<cf::binary32>(widen(x),widen(bv[Lane<0?index:Lane]),c[i]);
28 }
29 return simd<float,N,Arch>::load_bits(c.data());
30 }
31}
32export namespace native {
43
44 // All vector operands share Arch; native registers remain implementation details.
46 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
47 native_nodiscard native_inline __attribute__((target("fp16fml")))
48 constexpr simd<float, 2, Arch> fmlal(
49 simd<float, 2, Arch> acc,
50 simd<fp16, 4, Arch> a,
51 simd<fp16, 4, Arch> b) noexcept {
52 if consteval { return detail::fp16fml_value<Arch,false,false,-1>(acc,a,b); } else {
53 auto result = detail::fmlal<Arch>(
54 vget_low_f32(acc.to_native()),
55 __builtin_bit_cast(float16x4_t, a.to_native()),
56 __builtin_bit_cast(float16x4_t, b.to_native()));
57 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
58 }
59 }
60
62 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
63 native_nodiscard consteval
67 simd<fp16, 4, Arch> b) noexcept {
68 return detail::fp16fml_value<Arch,false,false,-1>(acc,a,b);
69 }
70
72 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
73 native_nodiscard native_inline __attribute__((target("fp16fml")))
77 simd<fp16, 4, Arch> b) noexcept {
78 if consteval { return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); } else {
79 auto result = detail::fmlal_lane<Arch, Lane>(
80 vget_low_f32(acc.to_native()),
81 __builtin_bit_cast(float16x4_t, a.to_native()),
82 __builtin_bit_cast(float16x4_t, b.to_native()));
83 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
84 }
85 }
86
88 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
89 native_nodiscard consteval
93 simd<fp16, 4, Arch> b) noexcept {
94 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
95 }
96
98 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
99 native_nodiscard native_inline __attribute__((target("fp16fml")))
103 simd<fp16, 8, Arch> b) noexcept {
104 if consteval { return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); } else {
105 auto result = detail::fmlal_lane<Arch, Lane>(
106 vget_low_f32(acc.to_native()),
107 __builtin_bit_cast(float16x4_t, a.to_native()),
108 __builtin_bit_cast(float16x8_t, b.to_native()));
109 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
110 }
111 }
112
114 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
115 native_nodiscard consteval
119 simd<fp16, 8, Arch> b) noexcept {
120 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
121 }
122
124 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
125 native_nodiscard native_inline __attribute__((target("fp16fml")))
129 simd<fp16, 8, Arch> b) noexcept {
130 if consteval { return detail::fp16fml_value<Arch,false,false,-1>(acc,a,b); } else {
131 auto result = detail::fmlal<Arch>(
132 __builtin_bit_cast(float32x4_t, acc.to_native()),
133 __builtin_bit_cast(float16x8_t, a.to_native()),
134 __builtin_bit_cast(float16x8_t, b.to_native()));
135 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
136 }
137 }
138
140 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
141 native_nodiscard consteval
145 simd<fp16, 8, Arch> b) noexcept {
146 return detail::fp16fml_value<Arch,false,false,-1>(acc,a,b);
147 }
148
150 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
151 native_nodiscard native_inline __attribute__((target("fp16fml")))
155 simd<fp16, 4, Arch> b) noexcept {
156 if consteval { return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); } else {
157 auto result = detail::fmlal_lane<Arch, Lane>(
158 __builtin_bit_cast(float32x4_t, acc.to_native()),
159 __builtin_bit_cast(float16x8_t, a.to_native()),
160 __builtin_bit_cast(float16x4_t, b.to_native()));
161 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
162 }
163 }
164
166 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
167 native_nodiscard consteval
171 simd<fp16, 4, Arch> b) noexcept {
172 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
173 }
174
176 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
177 native_nodiscard native_inline __attribute__((target("fp16fml")))
181 simd<fp16, 8, Arch> b) noexcept {
182 if consteval { return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b); } else {
183 auto result = detail::fmlal_lane<Arch, Lane>(
184 __builtin_bit_cast(float32x4_t, acc.to_native()),
185 __builtin_bit_cast(float16x8_t, a.to_native()),
186 __builtin_bit_cast(float16x8_t, b.to_native()));
187 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
188 }
189 }
190
192 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
193 native_nodiscard consteval
197 simd<fp16, 8, Arch> b) noexcept {
198 return detail::fp16fml_value<Arch,false,false,Lane>(acc,a,b);
199 }
200
202 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
203 native_nodiscard native_inline __attribute__((target("fp16fml")))
207 simd<fp16, 4, Arch> b) noexcept {
208 if consteval { return detail::fp16fml_value<Arch,false,true,-1>(acc,a,b); } else {
209 auto result = detail::fmlal2<Arch>(
210 vget_low_f32(acc.to_native()),
211 __builtin_bit_cast(float16x4_t, a.to_native()),
212 __builtin_bit_cast(float16x4_t, b.to_native()));
213 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
214 }
215 }
216
218 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
219 native_nodiscard consteval
223 simd<fp16, 4, Arch> b) noexcept {
224 return detail::fp16fml_value<Arch,false,true,-1>(acc,a,b);
225 }
226
228 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
229 native_nodiscard native_inline __attribute__((target("fp16fml")))
233 simd<fp16, 4, Arch> b) noexcept {
234 if consteval { return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); } else {
235 auto result = detail::fmlal2_lane<Arch, Lane>(
236 vget_low_f32(acc.to_native()),
237 __builtin_bit_cast(float16x4_t, a.to_native()),
238 __builtin_bit_cast(float16x4_t, b.to_native()));
239 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
240 }
241 }
242
244 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
245 native_nodiscard consteval
249 simd<fp16, 4, Arch> b) noexcept {
250 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
251 }
252
254 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
255 native_nodiscard native_inline __attribute__((target("fp16fml")))
259 simd<fp16, 8, Arch> b) noexcept {
260 if consteval { return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); } else {
261 auto result = detail::fmlal2_lane<Arch, Lane>(
262 vget_low_f32(acc.to_native()),
263 __builtin_bit_cast(float16x4_t, a.to_native()),
264 __builtin_bit_cast(float16x8_t, b.to_native()));
265 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
266 }
267 }
268
270 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
271 native_nodiscard consteval
275 simd<fp16, 8, Arch> b) noexcept {
276 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
277 }
278
280 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
281 native_nodiscard native_inline __attribute__((target("fp16fml")))
285 simd<fp16, 8, Arch> b) noexcept {
286 if consteval { return detail::fp16fml_value<Arch,false,true,-1>(acc,a,b); } else {
287 auto result = detail::fmlal2<Arch>(
288 __builtin_bit_cast(float32x4_t, acc.to_native()),
289 __builtin_bit_cast(float16x8_t, a.to_native()),
290 __builtin_bit_cast(float16x8_t, b.to_native()));
291 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
292 }
293 }
294
296 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
297 native_nodiscard consteval
301 simd<fp16, 8, Arch> b) noexcept {
302 return detail::fp16fml_value<Arch,false,true,-1>(acc,a,b);
303 }
304
306 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
307 native_nodiscard native_inline __attribute__((target("fp16fml")))
311 simd<fp16, 4, Arch> b) noexcept {
312 if consteval { return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); } else {
313 auto result = detail::fmlal2_lane<Arch, Lane>(
314 __builtin_bit_cast(float32x4_t, acc.to_native()),
315 __builtin_bit_cast(float16x8_t, a.to_native()),
316 __builtin_bit_cast(float16x4_t, b.to_native()));
317 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
318 }
319 }
320
322 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
323 native_nodiscard consteval
327 simd<fp16, 4, Arch> b) noexcept {
328 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
329 }
330
332 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
333 native_nodiscard native_inline __attribute__((target("fp16fml")))
337 simd<fp16, 8, Arch> b) noexcept {
338 if consteval { return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b); } else {
339 auto result = detail::fmlal2_lane<Arch, Lane>(
340 __builtin_bit_cast(float32x4_t, acc.to_native()),
341 __builtin_bit_cast(float16x8_t, a.to_native()),
342 __builtin_bit_cast(float16x8_t, b.to_native()));
343 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
344 }
345 }
346
348 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
349 native_nodiscard consteval
353 simd<fp16, 8, Arch> b) noexcept {
354 return detail::fp16fml_value<Arch,false,true,Lane>(acc,a,b);
355 }
356
358 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
359 native_nodiscard native_inline __attribute__((target("fp16fml")))
363 simd<fp16, 4, Arch> b) noexcept {
364 if consteval { return detail::fp16fml_value<Arch,true,false,-1>(acc,a,b); } else {
365 auto result = detail::fmlsl<Arch>(
366 vget_low_f32(acc.to_native()),
367 __builtin_bit_cast(float16x4_t, a.to_native()),
368 __builtin_bit_cast(float16x4_t, b.to_native()));
369 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
370 }
371 }
372
374 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
375 native_nodiscard consteval
379 simd<fp16, 4, Arch> b) noexcept {
380 return detail::fp16fml_value<Arch,true,false,-1>(acc,a,b);
381 }
382
384 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
385 native_nodiscard native_inline __attribute__((target("fp16fml")))
389 simd<fp16, 4, Arch> b) noexcept {
390 if consteval { return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); } else {
391 auto result = detail::fmlsl_lane<Arch, Lane>(
392 vget_low_f32(acc.to_native()),
393 __builtin_bit_cast(float16x4_t, a.to_native()),
394 __builtin_bit_cast(float16x4_t, b.to_native()));
395 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
396 }
397 }
398
400 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
401 native_nodiscard consteval
405 simd<fp16, 4, Arch> b) noexcept {
406 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
407 }
408
410 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
411 native_nodiscard native_inline __attribute__((target("fp16fml")))
415 simd<fp16, 8, Arch> b) noexcept {
416 if consteval { return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); } else {
417 auto result = detail::fmlsl_lane<Arch, Lane>(
418 vget_low_f32(acc.to_native()),
419 __builtin_bit_cast(float16x4_t, a.to_native()),
420 __builtin_bit_cast(float16x8_t, b.to_native()));
421 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
422 }
423 }
424
426 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
427 native_nodiscard consteval
431 simd<fp16, 8, Arch> b) noexcept {
432 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
433 }
434
436 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
437 native_nodiscard native_inline __attribute__((target("fp16fml")))
441 simd<fp16, 8, Arch> b) noexcept {
442 if consteval { return detail::fp16fml_value<Arch,true,false,-1>(acc,a,b); } else {
443 auto result = detail::fmlsl<Arch>(
444 __builtin_bit_cast(float32x4_t, acc.to_native()),
445 __builtin_bit_cast(float16x8_t, a.to_native()),
446 __builtin_bit_cast(float16x8_t, b.to_native()));
447 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
448 }
449 }
450
452 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
453 native_nodiscard consteval
457 simd<fp16, 8, Arch> b) noexcept {
458 return detail::fp16fml_value<Arch,true,false,-1>(acc,a,b);
459 }
460
462 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
463 native_nodiscard native_inline __attribute__((target("fp16fml")))
467 simd<fp16, 4, Arch> b) noexcept {
468 if consteval { return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); } else {
469 auto result = detail::fmlsl_lane<Arch, Lane>(
470 __builtin_bit_cast(float32x4_t, acc.to_native()),
471 __builtin_bit_cast(float16x8_t, a.to_native()),
472 __builtin_bit_cast(float16x4_t, b.to_native()));
473 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
474 }
475 }
476
478 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
479 native_nodiscard consteval
483 simd<fp16, 4, Arch> b) noexcept {
484 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
485 }
486
488 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
489 native_nodiscard native_inline __attribute__((target("fp16fml")))
493 simd<fp16, 8, Arch> b) noexcept {
494 if consteval { return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b); } else {
495 auto result = detail::fmlsl_lane<Arch, Lane>(
496 __builtin_bit_cast(float32x4_t, acc.to_native()),
497 __builtin_bit_cast(float16x8_t, a.to_native()),
498 __builtin_bit_cast(float16x8_t, b.to_native()));
499 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
500 }
501 }
502
504 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
505 native_nodiscard consteval
509 simd<fp16, 8, Arch> b) noexcept {
510 return detail::fp16fml_value<Arch,true,false,Lane>(acc,a,b);
511 }
512
514 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
515 native_nodiscard native_inline __attribute__((target("fp16fml")))
519 simd<fp16, 4, Arch> b) noexcept {
520 if consteval { return detail::fp16fml_value<Arch,true,true,-1>(acc,a,b); } else {
521 auto result = detail::fmlsl2<Arch>(
522 vget_low_f32(acc.to_native()),
523 __builtin_bit_cast(float16x4_t, a.to_native()),
524 __builtin_bit_cast(float16x4_t, b.to_native()));
525 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
526 }
527 }
528
530 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
531 native_nodiscard consteval
535 simd<fp16, 4, Arch> b) noexcept {
536 return detail::fp16fml_value<Arch,true,true,-1>(acc,a,b);
537 }
538
540 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
541 native_nodiscard native_inline __attribute__((target("fp16fml")))
545 simd<fp16, 4, Arch> b) noexcept {
546 if consteval { return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); } else {
547 auto result = detail::fmlsl2_lane<Arch, Lane>(
548 vget_low_f32(acc.to_native()),
549 __builtin_bit_cast(float16x4_t, a.to_native()),
550 __builtin_bit_cast(float16x4_t, b.to_native()));
551 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
552 }
553 }
554
556 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
557 native_nodiscard consteval
561 simd<fp16, 4, Arch> b) noexcept {
562 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
563 }
564
566 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
567 native_nodiscard native_inline __attribute__((target("fp16fml")))
571 simd<fp16, 8, Arch> b) noexcept {
572 if consteval { return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); } else {
573 auto result = detail::fmlsl2_lane<Arch, Lane>(
574 vget_low_f32(acc.to_native()),
575 __builtin_bit_cast(float16x4_t, a.to_native()),
576 __builtin_bit_cast(float16x8_t, b.to_native()));
577 return simd<float, 2, Arch>::from_native(vcombine_f32(result, vdup_n_f32(0.f)));
578 }
579 }
580
582 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 2, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
583 native_nodiscard consteval
587 simd<fp16, 8, Arch> b) noexcept {
588 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
589 }
590
592 template<isa<arm> Arch> requires(Arch.has(arm_feature::fp16fml))
593 native_nodiscard native_inline __attribute__((target("fp16fml")))
597 simd<fp16, 8, Arch> b) noexcept {
598 if consteval { return detail::fp16fml_value<Arch,true,true,-1>(acc,a,b); } else {
599 auto result = detail::fmlsl2<Arch>(
600 __builtin_bit_cast(float32x4_t, acc.to_native()),
601 __builtin_bit_cast(float16x8_t, a.to_native()),
602 __builtin_bit_cast(float16x8_t, b.to_native()));
603 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
604 }
605 }
606
608 template<isa<arm> Arch> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)))
609 native_nodiscard consteval
613 simd<fp16, 8, Arch> b) noexcept {
614 return detail::fp16fml_value<Arch,true,true,-1>(acc,a,b);
615 }
616
618 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
619 native_nodiscard native_inline __attribute__((target("fp16fml")))
623 simd<fp16, 4, Arch> b) noexcept {
624 if consteval { return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); } else {
625 auto result = detail::fmlsl2_lane<Arch, Lane>(
626 __builtin_bit_cast(float32x4_t, acc.to_native()),
627 __builtin_bit_cast(float16x8_t, a.to_native()),
628 __builtin_bit_cast(float16x4_t, b.to_native()));
629 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
630 }
631 }
632
634 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && requires { sizeof(simd<fp16, 4, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 4)
635 native_nodiscard consteval
639 simd<fp16, 4, Arch> b) noexcept {
640 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
641 }
642
644 template<isa<arm> Arch, unsigned Lane> requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
645 native_nodiscard native_inline __attribute__((target("fp16fml")))
649 simd<fp16, 8, Arch> b) noexcept {
650 if consteval { return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b); } else {
651 auto result = detail::fmlsl2_lane<Arch, Lane>(
652 __builtin_bit_cast(float32x4_t, acc.to_native()),
653 __builtin_bit_cast(float16x8_t, a.to_native()),
654 __builtin_bit_cast(float16x8_t, b.to_native()));
655 return simd<float, 4, Arch>::from_native(__builtin_bit_cast(typename simd<float, 4, Arch>::native_type, result));
656 }
657 }
658
660 template<isa<arm> Arch, unsigned Lane> requires(requires { sizeof(simd<float, 4, Arch>); } && requires { sizeof(simd<fp16, 8, Arch>); } && !(Arch.has(arm_feature::fp16fml)) && Lane < 8)
661 native_nodiscard consteval
665 simd<fp16, 8, Arch> b) noexcept {
666 return detail::fp16fml_value<Arch,true,true,Lane>(acc,a,b);
667 }
668
670 template<isa<arm> Arch, class A, class B, class C>
671 void fmlal(A, B, C) = delete;
672 template<isa<arm> Arch, unsigned Lane, class A, class B, class C>
673 void fmlal_lane(A, B, C) = delete;
674 template<isa<arm> Arch, class A, class B, class C>
675 void fmlal2(A, B, C) = delete;
676 template<isa<arm> Arch, unsigned Lane, class A, class B, class C>
677 void fmlal2_lane(A, B, C) = delete;
678 template<isa<arm> Arch, class A, class B, class C>
679 void fmlsl(A, B, C) = delete;
680 template<isa<arm> Arch, unsigned Lane, class A, class B, class C>
681 void fmlsl_lane(A, B, C) = delete;
682 template<isa<arm> Arch, class A, class B, class C>
683 void fmlsl2(A, B, C) = delete;
684 template<isa<arm> Arch, unsigned Lane, class A, class B, class C>
685 void fmlsl2_lane(A, B, C) = delete;
688}
689#endif
constexpr simd< float, 2, Arch > fmlal(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Add products from the low 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Subtract products from the low 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl2_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLSL2 with b[Lane] broadcast; selects the high 2 lanes of a.
constexpr simd< float, 2, Arch > fmlal_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLAL with b[Lane] broadcast; selects the low 2 lanes of a.
constexpr simd< float, 2, Arch > fmlal2(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Add products from the high 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl2(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Subtract products from the high 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlal2_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLAL2 with b[Lane] broadcast; selects the high 2 lanes of a.
constexpr simd< float, 2, Arch > fmlsl_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLSL with b[Lane] broadcast; selects the low 2 lanes of a.
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
constexpr int target
First matching requirement, with every later choice checked for shadowing.
Definition isa.h:396
Omitted architecture arguments use the native.simd provider's baseline.