native 0.0.1
Vectors, masks and wide register packs for C++26
Loading...
Searching...
No Matches
sha.h
1// SPDX-License-Identifier: BSD-2-Clause OR Apache-2.0
2#pragma once
3
4#include "native/config.h"
5#include "native/attributes.h"
6#include "native/isa.h"
7#if NATIVE_HOST_X86
8#include <immintrin.h>
9
10namespace native::detail::x86_sha {
11 template<isa<x86> Arch, unsigned Selector> requires(Arch.has(x86_feature::sha) && Selector < 4)
13 __m128i sha1rnds4(__m128i a, __m128i b) noexcept {
14 return _mm_sha1rnds4_epu32(a, b, Selector);
15 }
16
17 template<isa<x86> Arch, unsigned Selector, class... Args>
18 void sha1rnds4(Args...) = delete;
19
20 template<isa<x86> Arch> requires(Arch.has(x86_feature::sha))
22 __m128i sha1nexte(__m128i a, __m128i b) noexcept {
23 return _mm_sha1nexte_epu32(a, b);
24 }
25
26 template<isa<x86> Arch, class... Args>
27 void sha1nexte(Args...) = delete;
28
29 template<isa<x86> Arch> requires(Arch.has(x86_feature::sha))
31 __m128i sha1msg1(__m128i a, __m128i b) noexcept {
32 return _mm_sha1msg1_epu32(a, b);
33 }
34
35 template<isa<x86> Arch, class... Args>
36 void sha1msg1(Args...) = delete;
37
38 template<isa<x86> Arch> requires(Arch.has(x86_feature::sha))
40 __m128i sha1msg2(__m128i a, __m128i b) noexcept {
41 return _mm_sha1msg2_epu32(a, b);
42 }
43
44 template<isa<x86> Arch, class... Args>
45 void sha1msg2(Args...) = delete;
46
47 template<isa<x86> Arch> requires(Arch.has(x86_feature::sha))
49 __m128i sha256rnds2(__m128i a, __m128i b, __m128i c) noexcept {
50 return _mm_sha256rnds2_epu32(a, b, c);
51 }
52
53 template<isa<x86> Arch, class... Args>
54 void sha256rnds2(Args...) = delete;
55
56 template<isa<x86> Arch> requires(Arch.has(x86_feature::sha))
58 __m128i sha256msg1(__m128i a, __m128i b) noexcept {
59 return _mm_sha256msg1_epu32(a, b);
60 }
61
62 template<isa<x86> Arch, class... Args>
63 void sha256msg1(Args...) = delete;
64
65 template<isa<x86> Arch> requires(Arch.has(x86_feature::sha))
67 __m128i sha256msg2(__m128i a, __m128i b) noexcept {
68 return _mm_sha256msg2_epu32(a, b);
69 }
70
71 template<isa<x86> Arch, class... Args>
72 void sha256msg2(Args...) = delete;
73
74}
75#endif
76
77#include <array>
78#include <bit>
79#include <cstdint>
80
81namespace native::detail::x86_sha_constant {
82 template<class V>
83 constexpr auto lanes(V value) noexcept {
84 std::array<std::uint32_t, 4> result{};
85 value.store(result.data());
86 return result;
87 }
88
89 constexpr std::uint32_t choose(std::uint32_t x, std::uint32_t y, std::uint32_t z) noexcept {
90 return (x & y) ^ (~x & z);
91 }
92
93 constexpr std::uint32_t majority(std::uint32_t x, std::uint32_t y, std::uint32_t z) noexcept {
94 return (x & y) ^ (x & z) ^ (y & z);
95 }
96
97 constexpr std::uint32_t small_sigma0(std::uint32_t x) noexcept {
98 return std::rotr(x, 7) ^ std::rotr(x, 18) ^ (x >> 3);
99 }
100
101 constexpr std::uint32_t small_sigma1(std::uint32_t x) noexcept {
102 return std::rotr(x, 17) ^ std::rotr(x, 19) ^ (x >> 10);
103 }
104
105 // SHA-1 packs A,B,C,D from the most significant dword down. The first
106 // message dword already contains E; later rounds obtain E from the old D.
107 template<unsigned Selector, class V>
108 constexpr V rounds1(V state, V message) noexcept {
109 auto x = lanes(state);
110 auto words = lanes(message);
111 auto a = x[3];
112 auto b = x[2];
113 auto c = x[1];
114 auto d = x[0];
115 std::uint32_t e = 0;
116 constexpr std::uint32_t constants[]{0x5a827999, 0x6ed9eba1, 0x8f1bbcdc, 0xca62c1d6};
117 for (unsigned round = 0; round < 4; ++round) {
118 std::uint32_t function;
119 if constexpr (Selector == 0) {
120 function = choose(b, c, d);
121 } else if constexpr (Selector == 2) {
122 function = majority(b, c, d);
123 } else {
124 function = b ^ c ^ d;
125 }
126 auto next = std::rotl(a, 5) + function + e + words[3 - round] + constants[Selector];
127 e = d;
128 d = c;
129 c = std::rotl(b, 30);
130 b = a;
131 a = next;
132 }
133 std::array<std::uint32_t, 4> result{d, c, b, a};
134 return V::load(result.data());
135 }
136
137 template<class V>
138 constexpr V next_e(V state, V message) noexcept {
139 auto x = lanes(state);
140 auto result = lanes(message);
141 result[3] += std::rotl(x[3], 30);
142 return V::load(result.data());
143 }
144
145 template<class V>
146 constexpr V message1_first(V a, V b) noexcept {
147 auto x = lanes(a);
148 auto y = lanes(b);
149 std::array<std::uint32_t, 4> result{x[0] ^ y[2], x[1] ^ y[3], x[2] ^ x[0], x[3] ^ x[1]};
150 return V::load(result.data());
151 }
152
153 template<class V>
154 constexpr V message1_last(V a, V b) noexcept {
155 auto x = lanes(a);
156 auto y = lanes(b);
157 std::array<std::uint32_t, 4> result{};
158 result[3] = std::rotl(x[3] ^ y[2], 1);
159 result[2] = std::rotl(x[2] ^ y[1], 1);
160 result[1] = std::rotl(x[1] ^ y[0], 1);
161 result[0] = std::rotl(x[0] ^ result[3], 1);
162 return V::load(result.data());
163 }
164
165 // CDGH and ABEF are packed from the most significant dword down. Only
166 // message dwords 0/1 participate; both already contain the round constants.
167 template<class V>
168 constexpr V rounds256(V cdgh, V abef, V message) noexcept {
169 auto x = lanes(cdgh);
170 auto y = lanes(abef);
171 auto words = lanes(message);
172 auto a = y[3];
173 auto b = y[2];
174 auto c = x[3];
175 auto d = x[2];
176 auto e = y[1];
177 auto f = y[0];
178 auto g = x[1];
179 auto h = x[0];
180 for (unsigned round = 0; round < 2; ++round) {
181 auto sigma1 = std::rotr(e, 6) ^ std::rotr(e, 11) ^ std::rotr(e, 25);
182 auto sigma0 = std::rotr(a, 2) ^ std::rotr(a, 13) ^ std::rotr(a, 22);
183 auto first = h + sigma1 + choose(e, f, g) + words[round];
184 auto second = sigma0 + majority(a, b, c);
185 h = g;
186 g = f;
187 f = e;
188 e = d + first;
189 d = c;
190 c = b;
191 b = a;
192 a = first + second;
193 }
194 std::array<std::uint32_t, 4> result{f, e, b, a};
195 return V::load(result.data());
196 }
197
198 template<class V>
199 constexpr V message256_first(V a, V b) noexcept {
200 auto x = lanes(a);
201 auto y = lanes(b);
202 std::array<std::uint32_t, 4> result{x[0] + small_sigma0(x[1]), x[1] + small_sigma0(x[2]),
203 x[2] + small_sigma0(x[3]), x[3] + small_sigma0(y[0])};
204 return V::load(result.data());
205 }
206
207 template<class V>
208 constexpr V message256_last(V a, V b) noexcept {
209 auto x = lanes(a);
210 auto y = lanes(b);
211 std::array<std::uint32_t, 4> result{};
212 result[0] = x[0] + small_sigma1(y[2]);
213 result[1] = x[1] + small_sigma1(y[3]);
214 result[2] = x[2] + small_sigma1(result[0]);
215 result[3] = x[3] + small_sigma1(result[1]);
216 return V::load(result.data());
217 }
218}
Compiler attributes for host code, with shader-safe shared modifiers.
#define native_inline
inline [[always_inline]]
Definition attributes.h:212
#define native_nodiscard
C++17 [[nodiscard]].
Definition attributes.h:189
#define native_const
[[const]] is not const
Definition attributes.h:108
#define native_target(x)
this indicates a required feature set for the current multiversioned function.
Definition attributes.h:476