4#include "native/config.h"
7#include "native/arm/detail/register_order.h"
11#if NATIVE_HOST_NEON || defined(NATIVE_DOXYGEN)
12namespace native::detail {
20 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
22 float32x2_t
fmlal(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
23 acc = detail::arm_register_order(acc);
24 a = detail::arm_register_order(a);
25 b = detail::arm_register_order(b);
26 asm volatile(
"fmlal %0.2s, %1.2h, %2.2h"
27 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
28 return detail::arm_register_order(acc);
31 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
33 float32x2_t
fmlal_lane(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
34 acc = detail::arm_register_order(acc);
35 a = detail::arm_register_order(a);
36 auto source = detail::arm_register_order(vcombine_f16(b, b));
37 asm volatile(
"fmlal %0.2s, %1.2h, %2.h[%3]"
38 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
39 return detail::arm_register_order(acc);
42 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
44 float32x2_t
fmlal_lane(float32x2_t acc, float16x4_t a, float16x8_t b) noexcept {
45 acc = detail::arm_register_order(acc);
46 a = detail::arm_register_order(a);
47 auto source = detail::arm_register_order(b);
48 asm volatile(
"fmlal %0.2s, %1.2h, %2.h[%3]"
49 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
50 return detail::arm_register_order(acc);
53 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
55 float32x4_t
fmlal(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
56 acc = detail::arm_register_order(acc);
57 a = detail::arm_register_order(a);
58 b = detail::arm_register_order(b);
59 asm volatile(
"fmlal %0.4s, %1.4h, %2.4h"
60 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
61 return detail::arm_register_order(acc);
64 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
66 float32x4_t
fmlal_lane(float32x4_t acc, float16x8_t a, float16x4_t b) noexcept {
67 acc = detail::arm_register_order(acc);
68 a = detail::arm_register_order(a);
69 auto source = detail::arm_register_order(vcombine_f16(b, b));
70 asm volatile(
"fmlal %0.4s, %1.4h, %2.h[%3]"
71 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
72 return detail::arm_register_order(acc);
75 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
77 float32x4_t
fmlal_lane(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
78 acc = detail::arm_register_order(acc);
79 a = detail::arm_register_order(a);
80 auto source = detail::arm_register_order(b);
81 asm volatile(
"fmlal %0.4s, %1.4h, %2.h[%3]"
82 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
83 return detail::arm_register_order(acc);
86 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
88 float32x2_t
fmlal2(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
89 acc = detail::arm_register_order(acc);
90 a = detail::arm_register_order(a);
91 b = detail::arm_register_order(b);
92 asm volatile(
"fmlal2 %0.2s, %1.2h, %2.2h"
93 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
94 return detail::arm_register_order(acc);
97 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
99 float32x2_t
fmlal2_lane(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
100 acc = detail::arm_register_order(acc);
101 a = detail::arm_register_order(a);
102 auto source = detail::arm_register_order(vcombine_f16(b, b));
103 asm volatile(
"fmlal2 %0.2s, %1.2h, %2.h[%3]"
104 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
105 return detail::arm_register_order(acc);
108 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
110 float32x2_t
fmlal2_lane(float32x2_t acc, float16x4_t a, float16x8_t b) noexcept {
111 acc = detail::arm_register_order(acc);
112 a = detail::arm_register_order(a);
113 auto source = detail::arm_register_order(b);
114 asm volatile(
"fmlal2 %0.2s, %1.2h, %2.h[%3]"
115 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
116 return detail::arm_register_order(acc);
119 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
121 float32x4_t
fmlal2(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
122 acc = detail::arm_register_order(acc);
123 a = detail::arm_register_order(a);
124 b = detail::arm_register_order(b);
125 asm volatile(
"fmlal2 %0.4s, %1.4h, %2.4h"
126 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
127 return detail::arm_register_order(acc);
130 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
132 float32x4_t
fmlal2_lane(float32x4_t acc, float16x8_t a, float16x4_t b) noexcept {
133 acc = detail::arm_register_order(acc);
134 a = detail::arm_register_order(a);
135 auto source = detail::arm_register_order(vcombine_f16(b, b));
136 asm volatile(
"fmlal2 %0.4s, %1.4h, %2.h[%3]"
137 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
138 return detail::arm_register_order(acc);
141 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
143 float32x4_t
fmlal2_lane(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
144 acc = detail::arm_register_order(acc);
145 a = detail::arm_register_order(a);
146 auto source = detail::arm_register_order(b);
147 asm volatile(
"fmlal2 %0.4s, %1.4h, %2.h[%3]"
148 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
149 return detail::arm_register_order(acc);
152 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
154 float32x2_t
fmlsl(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
155 acc = detail::arm_register_order(acc);
156 a = detail::arm_register_order(a);
157 b = detail::arm_register_order(b);
158 asm volatile(
"fmlsl %0.2s, %1.2h, %2.2h"
159 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
160 return detail::arm_register_order(acc);
163 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
165 float32x2_t
fmlsl_lane(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
166 acc = detail::arm_register_order(acc);
167 a = detail::arm_register_order(a);
168 auto source = detail::arm_register_order(vcombine_f16(b, b));
169 asm volatile(
"fmlsl %0.2s, %1.2h, %2.h[%3]"
170 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
171 return detail::arm_register_order(acc);
174 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
176 float32x2_t
fmlsl_lane(float32x2_t acc, float16x4_t a, float16x8_t b) noexcept {
177 acc = detail::arm_register_order(acc);
178 a = detail::arm_register_order(a);
179 auto source = detail::arm_register_order(b);
180 asm volatile(
"fmlsl %0.2s, %1.2h, %2.h[%3]"
181 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
182 return detail::arm_register_order(acc);
185 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
187 float32x4_t
fmlsl(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
188 acc = detail::arm_register_order(acc);
189 a = detail::arm_register_order(a);
190 b = detail::arm_register_order(b);
191 asm volatile(
"fmlsl %0.4s, %1.4h, %2.4h"
192 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
193 return detail::arm_register_order(acc);
196 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
198 float32x4_t
fmlsl_lane(float32x4_t acc, float16x8_t a, float16x4_t b) noexcept {
199 acc = detail::arm_register_order(acc);
200 a = detail::arm_register_order(a);
201 auto source = detail::arm_register_order(vcombine_f16(b, b));
202 asm volatile(
"fmlsl %0.4s, %1.4h, %2.h[%3]"
203 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
204 return detail::arm_register_order(acc);
207 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
209 float32x4_t
fmlsl_lane(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
210 acc = detail::arm_register_order(acc);
211 a = detail::arm_register_order(a);
212 auto source = detail::arm_register_order(b);
213 asm volatile(
"fmlsl %0.4s, %1.4h, %2.h[%3]"
214 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
215 return detail::arm_register_order(acc);
218 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
220 float32x2_t
fmlsl2(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
221 acc = detail::arm_register_order(acc);
222 a = detail::arm_register_order(a);
223 b = detail::arm_register_order(b);
224 asm volatile(
"fmlsl2 %0.2s, %1.2h, %2.2h"
225 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
226 return detail::arm_register_order(acc);
229 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
231 float32x2_t
fmlsl2_lane(float32x2_t acc, float16x4_t a, float16x4_t b) noexcept {
232 acc = detail::arm_register_order(acc);
233 a = detail::arm_register_order(a);
234 auto source = detail::arm_register_order(vcombine_f16(b, b));
235 asm volatile(
"fmlsl2 %0.2s, %1.2h, %2.h[%3]"
236 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
237 return detail::arm_register_order(acc);
240 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
242 float32x2_t
fmlsl2_lane(float32x2_t acc, float16x4_t a, float16x8_t b) noexcept {
243 acc = detail::arm_register_order(acc);
244 a = detail::arm_register_order(a);
245 auto source = detail::arm_register_order(b);
246 asm volatile(
"fmlsl2 %0.2s, %1.2h, %2.h[%3]"
247 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
248 return detail::arm_register_order(acc);
251 template<isa<arm> Arch>
requires(Arch.has(arm_feature::fp16fml))
253 float32x4_t
fmlsl2(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
254 acc = detail::arm_register_order(acc);
255 a = detail::arm_register_order(a);
256 b = detail::arm_register_order(b);
257 asm volatile(
"fmlsl2 %0.4s, %1.4h, %2.4h"
258 :
"+w"(acc) :
"w"(a),
"w"(b) :
"memory");
259 return detail::arm_register_order(acc);
262 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 4)
264 float32x4_t
fmlsl2_lane(float32x4_t acc, float16x8_t a, float16x4_t b) noexcept {
265 acc = detail::arm_register_order(acc);
266 a = detail::arm_register_order(a);
267 auto source = detail::arm_register_order(vcombine_f16(b, b));
268 asm volatile(
"fmlsl2 %0.4s, %1.4h, %2.h[%3]"
269 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
270 return detail::arm_register_order(acc);
273 template<isa<arm> Arch,
unsigned Lane>
requires(Arch.has(arm_feature::fp16fml) && Lane < 8)
275 float32x4_t
fmlsl2_lane(float32x4_t acc, float16x8_t a, float16x8_t b) noexcept {
276 acc = detail::arm_register_order(acc);
277 a = detail::arm_register_order(a);
278 auto source = detail::arm_register_order(b);
279 asm volatile(
"fmlsl2 %0.4s, %1.4h, %2.h[%3]"
280 :
"+w"(acc) :
"w"(a),
"x"(source),
"i"(Lane) :
"memory");
281 return detail::arm_register_order(acc);
Compiler attributes for host code, with shader-safe shared modifiers.
constexpr simd< float, 2, Arch > fmlal(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Add products from the low 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Subtract products from the low 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl2_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLSL2 with b[Lane] broadcast; selects the high 2 lanes of a.
constexpr simd< float, 2, Arch > fmlal_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLAL with b[Lane] broadcast; selects the low 2 lanes of a.
constexpr simd< float, 2, Arch > fmlal2(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Add products from the high 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlsl2(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
Subtract products from the high 2 half lanes of a and b.
constexpr simd< float, 2, Arch > fmlal2_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLAL2 with b[Lane] broadcast; selects the high 2 lanes of a.
constexpr simd< float, 2, Arch > fmlsl_lane(simd< float, 2, Arch > acc, simd< fp16, 4, Arch > a, simd< fp16, 4, Arch > b) noexcept
FMLSL with b[Lane] broadcast; selects the low 2 lanes of a.
#define native_inline
inline [[always_inline]]
#define native_nodiscard
C++17 [[nodiscard]].
constexpr int target
First matching requirement, with every later choice checked for shadowing.