native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
f16c.h
1// SPDX-FileCopyrightText: 2026 Edward Kmett <ekmett@gmail.com>
2// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
3#pragma once
4#include "native/config.h"
5#include "native/attributes.h"
6#include "native/isa.h"
7#include <concepts>
8#include <cstdint>
9#if NATIVE_HOST_X86
10#include <immintrin.h>
11#endif
12
13#if NATIVE_HOST_X86 || defined(NATIVE_DOXYGEN)
14namespace native::detail::x86_f16c {
15// Internal register helpers for the native.x86.f16c module.
16
17
19 template<isa<x86> Arch, unsigned Imm8, class V>
20 requires(Arch.has(x86_feature::f16c) && Imm8 <= 255 && std::same_as<V, __m128>)
22 __m128i cvtps_ph(V a) noexcept {
23 __m128i result;
24 // LLVM's conversion intrinsics can disappear when their output is unused.
25 // Volatile asm retains the instruction; the memory clobber orders it with
26 // MXCSR loads/stores. Dialect alternatives support both assembler syntaxes.
27 __asm__ volatile("vcvtps2ph {%2, %1, %0|%0, %1, %2}"
28 : "=x"(result) : "x"(a), "i"(Imm8) : "memory");
29 return result;
30 }
31
33 template<isa<x86> Arch, unsigned Imm8, class V>
34 requires(Arch.has(x86_feature::f16c) && Imm8 <= 255 && std::same_as<V, __m256>)
36 __m128i cvtps_ph(V a) noexcept {
37 __m128i result;
38 __asm__ volatile("vcvtps2ph {%2, %1, %0|%0, %1, %2}"
39 : "=x"(result) : "x"(a), "i"(Imm8) : "memory");
40 return result;
41 }
42
44 template<isa<x86> Arch, unsigned Lanes, class V>
45 requires(Arch.has(x86_feature::f16c) && Lanes == 4 && std::same_as<V, __m128i>)
47 __m128 cvtph_ps(V a) noexcept {
48 __m128 result;
49 __asm__ volatile("vcvtph2ps {%1, %0|%0, %1}"
50 : "=x"(result) : "x"(a) : "memory");
51 return result;
52 }
53
55 template<isa<x86> Arch, unsigned Lanes, class V>
56 requires(Arch.has(x86_feature::f16c) && Lanes == 8 && std::same_as<V, __m128i>)
58 __m256 cvtph_ps(V a) noexcept {
59 __m256 result;
60 __asm__ volatile("vcvtph2ps {%1, %0|%0, %1}"
61 : "=x"(result) : "x"(a) : "memory");
62 return result;
63 }
64
66 template<isa<x86> Arch, unsigned Imm8>
67 requires(Arch.has(x86_feature::f16c) && Imm8 <= 255)
69 std::uint16_t cvtss_sh(float a) noexcept {
70 return static_cast<std::uint16_t>(_mm_cvtsi128_si32(cvtps_ph<Arch, Imm8>(_mm_set_ss(a))));
71 }
72
74 template<isa<x86> Arch> requires(Arch.has(x86_feature::f16c))
76 float cvtsh_ss(std::uint16_t a) noexcept {
77 return _mm_cvtss_f32(cvtph_ps<Arch, 4>(_mm_cvtsi32_si128(a)));
78 }
79
80}
81#endif
Compiler attributes for host code, with shader-safe shared modifiers.
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
#define native_target(x)
this indicates a required feature set for the current multiversioned function.
Definition attributes.h:476