native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
native.x86.sm4.ccm
1// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
2module;
3#include "native/isa_import.h"
4#include "native/config.h"
5#include "native/attributes.h"
6#include <array>
7#include <bit>
8#include <cstdint>
9#include <immintrin.h>
10#include "native/x86/detail/constant_lanes.h"
11#include "native/detail/sm4_sbox.h"
12export module native.x86.sm4;
13export import native.x86.features;
14export import native.simd;
15
16namespace native::detail::x86_sm4_constant {
17 template<bool Key, class V> constexpr V rounds(V state, V keys) noexcept {
18 auto x = x86_constant::lanes(state), k = x86_constant::lanes(keys);
19 for (unsigned base = 0; base < V::lanes; base += 4) {
20 for (unsigned i = 0; i < 4; ++i) {
21 auto mixed = x[base + 1] ^ x[base + 2] ^ x[base + 3] ^ k[base + i];
22 std::uint32_t t = 0;
23 for (unsigned byte = 0; byte < 4; ++byte)
24 t |= std::uint32_t(sbox[(mixed >> (byte * 8)) & 255]) << (byte * 8);
25 auto next = x[base] ^ t;
26 if constexpr (Key) next ^= std::rotl(t, 13) ^ std::rotl(t, 23);
27 else next ^= std::rotl(t, 2) ^ std::rotl(t, 10) ^ std::rotl(t, 18) ^ std::rotl(t, 24);
28 for (unsigned j = 0; j < 3; ++j) x[base + j] = x[base + j + 1];
29 x[base + 3] = next;
30 }
31 }
32 return x86_constant::pack<V>(x);
33 }
34 template<class V> constexpr V sm4rnds4(V a, V b) noexcept { return rounds<false>(a, b); }
35 template<class V> constexpr V sm4key4(V a, V b) noexcept { return rounds<true>(a, b); }
36}
37
38export namespace native {
45
47 template<isa<x86> Arch>
48 requires(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx))
50 constexpr simd<std::uint32_t, 4, Arch> sm4rnds4(simd<std::uint32_t, 4, Arch> a, simd<std::uint32_t, 4, Arch> b) noexcept {
51 if consteval { return detail::x86_sm4_constant::sm4rnds4(a, b); }
52 else { return simd<std::uint32_t, 4, Arch>::from_native(_mm_sm4rnds4_epi32(a.to_native(), b.to_native())); }
53 }
54
56 template<isa<x86> Arch>
57 requires(!(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx)) &&
58 requires { sizeof(simd<std::uint32_t, 4, Arch>); })
60 return detail::x86_sm4_constant::sm4rnds4(a, b);
61 }
62
64 template<isa<x86> Arch>
65 requires(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx))
68 if consteval { return detail::x86_sm4_constant::sm4key4(a, b); }
69 else { return simd<std::uint32_t, 4, Arch>::from_native(_mm_sm4key4_epi32(a.to_native(), b.to_native())); }
70 }
71
73 template<isa<x86> Arch>
74 requires(!(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx)) &&
75 requires { sizeof(simd<std::uint32_t, 4, Arch>); })
77 return detail::x86_sm4_constant::sm4key4(a, b);
78 }
79
81 template<isa<x86> Arch>
82 requires(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx))
85 if consteval { return detail::x86_sm4_constant::sm4rnds4(a, b); }
86 else { return simd<std::uint32_t, 8, Arch>::from_native(_mm256_sm4rnds4_epi32(a.to_native(), b.to_native())); }
87 }
88
90 template<isa<x86> Arch>
91 requires(!(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx)) &&
92 requires { sizeof(simd<std::uint32_t, 8, Arch>); })
94 return detail::x86_sm4_constant::sm4rnds4(a, b);
95 }
96
98 template<isa<x86> Arch>
99 requires(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx))
102 if consteval { return detail::x86_sm4_constant::sm4key4(a, b); }
103 else { return simd<std::uint32_t, 8, Arch>::from_native(_mm256_sm4key4_epi32(a.to_native(), b.to_native())); }
104 }
105
107 template<isa<x86> Arch>
108 requires(!(Arch.has(x86_feature::sm4) && Arch.has(x86_feature::avx)) &&
109 requires { sizeof(simd<std::uint32_t, 8, Arch>); })
111 return detail::x86_sm4_constant::sm4key4(a, b);
112 }
113
115 template<isa<x86> Arch, class... Args> void sm4rnds4(Args...) = delete;
116
118 template<isa<x86> Arch, class... Args> void sm4key4(Args...) = delete;
120}
Compiler attributes for host code, with shader-safe shared modifiers.
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
#define native_const
[[const]] is not const
Definition attributes.h:108
#define native_target(x)
this indicates a required feature set for the current multiversioned function.
Definition attributes.h:476
constexpr simd< std::uint32_t, 4, Arch > sm4key4(simd< std::uint32_t, 4, Arch > a, simd< std::uint32_t, 4, Arch > b) noexcept
Four SM4 key-schedule rounds independently within each 128-bit block; a is state, b is constants.
constexpr simd< std::uint32_t, 4, Arch > sm4rnds4(simd< std::uint32_t, 4, Arch > a, simd< std::uint32_t, 4, Arch > b) noexcept
Four SM4 data rounds independently within each 128-bit block; a is state, b is keys.
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
Standard-library adaptations documented here for SIMD value types.
Omitted architecture arguments use the native.simd provider's baseline.