native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
native.x86.sha512.ccm
1// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
2module;
3#include "native/isa_import.h"
4#include "native/config.h"
5#include "native/attributes.h"
6#include <array>
7#include <bit>
8#include <cstdint>
9#include <immintrin.h>
10#include "native/x86/detail/constant_lanes.h"
11export module native.x86.sha512;
12export import native.x86.features;
13export import native.simd;
14
15namespace native::detail::x86_sha512_constant {
16 constexpr std::uint64_t sigma0(std::uint64_t x) noexcept {
17 return std::rotr(x, 1) ^ std::rotr(x, 8) ^ (x >> 7);
18 }
19 constexpr std::uint64_t sigma1(std::uint64_t x) noexcept {
20 return std::rotr(x, 19) ^ std::rotr(x, 61) ^ (x >> 6);
21 }
22 template<class V, class W> constexpr V sha512msg1(V a, W b) noexcept {
23 auto x = x86_constant::lanes(a);
24 auto y = x86_constant::lanes(b);
25 for (unsigned i = 0; i < 3; ++i) x[i] += sigma0(x[i + 1]);
26 x[3] += sigma0(y[0]);
27 return x86_constant::pack<V>(x);
28 }
29 template<class V> constexpr V sha512msg2(V a, V b) noexcept {
30 auto x = x86_constant::lanes(a);
31 auto y = x86_constant::lanes(b);
32 x[0] += sigma1(y[2]);
33 x[1] += sigma1(y[3]);
34 x[2] += sigma1(x[0]);
35 x[3] += sigma1(x[1]);
36 return x86_constant::pack<V>(x);
37 }
38 template<class V, class W> constexpr V sha512rnds2(V cdgh, V abef, W message) noexcept {
39 auto x = x86_constant::lanes(cdgh), y = x86_constant::lanes(abef);
40 auto words = x86_constant::lanes(message);
41 auto a = y[3], b = y[2], c = x[3], d = x[2];
42 auto e = y[1], f = y[0], g = x[1], h = x[0];
43 for (unsigned i = 0; i < 2; ++i) {
44 auto t1 = h + (std::rotr(e, 14) ^ std::rotr(e, 18) ^ std::rotr(e, 41)) +
45 ((e & f) ^ (~e & g)) + words[i];
46 auto t2 = (std::rotr(a, 28) ^ std::rotr(a, 34) ^ std::rotr(a, 39)) +
47 ((a & b) ^ (a & c) ^ (b & c));
48 h = g; g = f; f = e; e = d + t1;
49 d = c; c = b; b = a; a = t1 + t2;
50 }
51 return x86_constant::pack<V>(std::array{f, e, b, a});
52 }
53}
54
55export namespace native {
62
64 template<isa<x86> Arch>
65 requires(Arch.has(x86_feature::sha512) && Arch.has(x86_feature::avx))
67 constexpr simd<std::uint64_t, 4, Arch> sha512msg1(simd<std::uint64_t, 4, Arch> a, simd<std::uint64_t, 2, Arch> b) noexcept {
68 if consteval { return detail::x86_sha512_constant::sha512msg1(a, b); }
69 else { return simd<std::uint64_t, 4, Arch>::from_native(_mm256_sha512msg1_epi64(a.to_native(), b.to_native())); }
70 }
71
73 template<isa<x86> Arch>
74 requires(!(Arch.has(x86_feature::sha512) && Arch.has(x86_feature::avx)) &&
75 requires { sizeof(simd<std::uint64_t, 2, Arch>); sizeof(simd<std::uint64_t, 4, Arch>); })
77 return detail::x86_sha512_constant::sha512msg1(a, b);
78 }
79
81 template<isa<x86> Arch>
82 requires(Arch.has(x86_feature::sha512) && Arch.has(x86_feature::avx))
85 if consteval { return detail::x86_sha512_constant::sha512msg2(a, b); }
86 else { return simd<std::uint64_t, 4, Arch>::from_native(_mm256_sha512msg2_epi64(a.to_native(), b.to_native())); }
87 }
88
90 template<isa<x86> Arch>
91 requires(!(Arch.has(x86_feature::sha512) && Arch.has(x86_feature::avx)) &&
92 requires { sizeof(simd<std::uint64_t, 4, Arch>); })
94 return detail::x86_sha512_constant::sha512msg2(a, b);
95 }
96
98 template<isa<x86> Arch>
99 requires(Arch.has(x86_feature::sha512) && Arch.has(x86_feature::avx))
102 if consteval { return detail::x86_sha512_constant::sha512rnds2(a, b, c); }
103 else { return simd<std::uint64_t, 4, Arch>::from_native(_mm256_sha512rnds2_epi64(a.to_native(), b.to_native(), c.to_native())); }
104 }
105
107 template<isa<x86> Arch>
108 requires(!(Arch.has(x86_feature::sha512) && Arch.has(x86_feature::avx)) &&
109 requires { sizeof(simd<std::uint64_t, 2, Arch>); sizeof(simd<std::uint64_t, 4, Arch>); })
111 return detail::x86_sha512_constant::sha512rnds2(a, b, c);
112 }
113
115 template<isa<x86> Arch, class... Args> void sha512msg1(Args...) = delete;
116
118 template<isa<x86> Arch, class... Args> void sha512msg2(Args...) = delete;
119
121 template<isa<x86> Arch, class... Args> void sha512rnds2(Args...) = delete;
123}
Compiler attributes for host code, with shader-safe shared modifiers.
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
#define native_const
[[const]] is not const
Definition attributes.h:108
#define native_target(x)
this indicates a required feature set for the current multiversioned function.
Definition attributes.h:476
constexpr simd< std::uint64_t, 4, Arch > sha512rnds2(simd< std::uint64_t, 4, Arch > a, simd< std::uint64_t, 4, Arch > b, simd< std::uint64_t, 2, Arch > c) noexcept
Two SHA-512 rounds: a is CDGH, b is ABEF, c contains two message-plus-constant words.
constexpr simd< std::uint64_t, 4, Arch > sha512msg2(simd< std::uint64_t, 4, Arch > a, simd< std::uint64_t, 4, Arch > b) noexcept
Final SHA-512 schedule step with recurrence through the low output lanes.
constexpr simd< std::uint64_t, 4, Arch > sha512msg1(simd< std::uint64_t, 4, Arch > a, simd< std::uint64_t, 2, Arch > b) noexcept
First SHA-512 schedule step; only the low lane of b is used.
Architecture-tagged vectors, register packs and supporting value types. Native arithmetic follows its...
Standard-library adaptations documented here for SIMD value types.
Omitted architecture arguments use the native.simd provider's baseline.