3#include "native/isa_import.h"
4#include "native/detail/sm4_sbox.h"
5#include "native/arm/detail/constant_lanes.h"
6#include "native/config.h"
13#include "native/targets.h"
17namespace native::detail::arm_sm4_constant {
20 template<
bool Key,
class V>
constexpr V rounds(V state, V keys)
noexcept {
21 auto x = arm_constant::lanes(state);
22 auto k = arm_constant::lanes(keys);
23 for (
unsigned i = 0; i < 4; ++i) {
24 auto mixed = x[1] ^ x[2] ^ x[3] ^ k[i];
25 std::uint32_t substituted = 0;
26 for (
unsigned byte = 0;
byte < 4; ++byte)
27 substituted |= std::uint32_t(sbox[(mixed >> (
byte * 8)) & 255]) << (
byte * 8);
28 auto next = substituted;
30 next ^= std::rotl(substituted, 13) ^ std::rotl(substituted, 23);
32 next ^= std::rotl(substituted, 2) ^ std::rotl(substituted, 10) ^
33 std::rotl(substituted, 18) ^ std::rotl(substituted, 24);
35 x = {x[1], x[2], x[3], next};
37 return arm_constant::pack<V>(x);
40 template<
class V>
constexpr V sm4e(V state, V keys)
noexcept {
41 return rounds<false>(state, keys);
44 template<
class V>
constexpr V sm4ekey(V state, V constants)
noexcept {
45 return rounds<true>(state, constants);
52namespace native::detail::arm_sm4 {
53 template<isa<arm> Arch>
54 requires(Arch.has(arm_feature::sm4))
56 uint32x4_t a, uint32x4_t b) noexcept {
57 return vsm4eq_u32(a, b);
60 template<isa<arm> Arch>
61 requires(Arch.has(arm_feature::sm4))
63 uint32x4_t a, uint32x4_t b) noexcept {
64 return vsm4ekeyq_u32(a, b);
71#if NATIVE_HOST_NEON || defined(NATIVE_DOXYGEN)
83 template<isa<arm> Arch>
84 requires(Arch.has(arm_feature::sm4))
89 return detail::arm_sm4_constant::sm4e(a, b);
91 auto result = detail::arm_sm4::sm4e<Arch>(__builtin_bit_cast(uint32x4_t, a.to_native()),
92 __builtin_bit_cast(uint32x4_t, b.to_native()));
99 template<isa<arm> Arch>
100 requires(!Arch.has(arm_feature::sm4) &&
requires {
sizeof(simd<std::uint32_t, 4, Arch>); })
103 return detail::arm_sm4_constant::sm4e(a, b);
108 template<isa<arm> Arch>
109 requires(Arch.has(arm_feature::sm4))
114 return detail::arm_sm4_constant::sm4ekey(a, b);
116 auto result = detail::arm_sm4::sm4ekey<Arch>(__builtin_bit_cast(uint32x4_t, a.to_native()),
117 __builtin_bit_cast(uint32x4_t, b.to_native()));
124 template<isa<arm> Arch>
125 requires(!Arch.has(arm_feature::sm4) &&
requires {
sizeof(simd<std::uint32_t, 4, Arch>); })
128 return detail::arm_sm4_constant::sm4ekey(a, b);
Compiler attributes for host code, with shader-safe shared modifiers.
constexpr simd< std::uint32_t, 4, Arch > sm4e(simd< std::uint32_t, 4, Arch > a, simd< std::uint32_t, 4, Arch > b) noexcept
Perform four SM4 data rounds with state words and successive round keys in ascending lanes.
constexpr simd< std::uint32_t, 4, Arch > sm4ekey(simd< std::uint32_t, 4, Arch > a, simd< std::uint32_t, 4, Arch > b) noexcept
#define native_inline
inline [[always_inline]]
#define native_nodiscard
C++17 [[nodiscard]].
#define native_const
[[const]] is not const
#define native_target(x)
this indicates a required feature set for the current multiversioned function.
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
Standard-library adaptations documented here for SIMD value types.
Omitted architecture arguments use the native.simd provider's baseline.