/bitcoin/src/crypto/sha256.cpp
Line | Count | Source |
1 | | // Copyright (c) 2014-2022 The Bitcoin Core developers |
2 | | // Distributed under the MIT software license, see the accompanying |
3 | | // file COPYING or http://www.opensource.org/licenses/mit-license.php. |
4 | | |
5 | | #include <crypto/sha256.h> |
6 | | #include <crypto/common.h> |
7 | | |
8 | | #include <algorithm> |
9 | | #include <cassert> |
10 | | #include <cstring> |
11 | | |
12 | | #if !defined(DISABLE_OPTIMIZED_SHA256) |
13 | | #include <compat/cpuid.h> |
14 | | |
15 | | #if defined(__linux__) && defined(ENABLE_ARM_SHANI) |
16 | | #include <sys/auxv.h> |
17 | | #include <asm/hwcap.h> |
18 | | #endif |
19 | | |
20 | | #if defined(__APPLE__) && defined(ENABLE_ARM_SHANI) |
21 | | #include <sys/types.h> |
22 | | #include <sys/sysctl.h> |
23 | | #endif |
24 | | |
25 | | #if defined(__x86_64__) || defined(__amd64__) || defined(__i386__) |
26 | | namespace sha256_sse4 |
27 | | { |
28 | | void Transform(uint32_t* s, const unsigned char* chunk, size_t blocks); |
29 | | } |
30 | | #endif |
31 | | |
32 | | namespace sha256d64_sse41 |
33 | | { |
34 | | void Transform_4way(unsigned char* out, const unsigned char* in); |
35 | | } |
36 | | |
37 | | namespace sha256d64_avx2 |
38 | | { |
39 | | void Transform_8way(unsigned char* out, const unsigned char* in); |
40 | | } |
41 | | |
42 | | namespace sha256d64_x86_shani |
43 | | { |
44 | | void Transform_2way(unsigned char* out, const unsigned char* in); |
45 | | } |
46 | | |
47 | | namespace sha256_x86_shani |
48 | | { |
49 | | void Transform(uint32_t* s, const unsigned char* chunk, size_t blocks); |
50 | | } |
51 | | |
52 | | namespace sha256_arm_shani |
53 | | { |
54 | | void Transform(uint32_t* s, const unsigned char* chunk, size_t blocks); |
55 | | } |
56 | | |
57 | | namespace sha256d64_arm_shani |
58 | | { |
59 | | void Transform_2way(unsigned char* out, const unsigned char* in); |
60 | | } |
61 | | #endif // DISABLE_OPTIMIZED_SHA256 |
62 | | |
63 | | // Internal implementation code. |
64 | | namespace |
65 | | { |
66 | | /// Internal SHA-256 implementation. |
67 | | namespace sha256 |
68 | | { |
69 | 42.5M | uint32_t inline Ch(uint32_t x, uint32_t y, uint32_t z) { return z ^ (x & (y ^ z)); } |
70 | 42.5M | uint32_t inline Maj(uint32_t x, uint32_t y, uint32_t z) { return (x & y) | (z & (x | y)); } |
71 | 42.5M | uint32_t inline Sigma0(uint32_t x) { return (x >> 2 | x << 30) ^ (x >> 13 | x << 19) ^ (x >> 22 | x << 10); } |
72 | 42.5M | uint32_t inline Sigma1(uint32_t x) { return (x >> 6 | x << 26) ^ (x >> 11 | x << 21) ^ (x >> 25 | x << 7); } |
73 | 31.9M | uint32_t inline sigma0(uint32_t x) { return (x >> 7 | x << 25) ^ (x >> 18 | x << 14) ^ (x >> 3); } |
74 | 31.9M | uint32_t inline sigma1(uint32_t x) { return (x >> 17 | x << 15) ^ (x >> 19 | x << 13) ^ (x >> 10); } |
75 | | |
76 | | /** One round of SHA-256. */ |
77 | | void inline Round(uint32_t a, uint32_t b, uint32_t c, uint32_t& d, uint32_t e, uint32_t f, uint32_t g, uint32_t& h, uint32_t k) |
78 | 42.5M | { |
79 | 42.5M | uint32_t t1 = h + Sigma1(e) + Ch(e, f, g) + k; |
80 | 42.5M | uint32_t t2 = Sigma0(a) + Maj(a, b, c); |
81 | 42.5M | d += t1; |
82 | 42.5M | h = t1 + t2; |
83 | 42.5M | } |
84 | | |
85 | | /** Initialize SHA-256 state. */ |
86 | | void inline Initialize(uint32_t* s) |
87 | 162M | { |
88 | 162M | s[0] = 0x6a09e667ul; |
89 | 162M | s[1] = 0xbb67ae85ul; |
90 | 162M | s[2] = 0x3c6ef372ul; |
91 | 162M | s[3] = 0xa54ff53aul; |
92 | 162M | s[4] = 0x510e527ful; |
93 | 162M | s[5] = 0x9b05688cul; |
94 | 162M | s[6] = 0x1f83d9abul; |
95 | 162M | s[7] = 0x5be0cd19ul; |
96 | 162M | } |
97 | | |
98 | | /** Perform a number of SHA-256 transformations, processing 64-byte chunks. */ |
99 | | void Transform(uint32_t* s, const unsigned char* chunk, size_t blocks) |
100 | 665k | { |
101 | 1.33M | while (blocks--) { Branch (101:12): [True: 665k, False: 665k]
|
102 | 665k | uint32_t a = s[0], b = s[1], c = s[2], d = s[3], e = s[4], f = s[5], g = s[6], h = s[7]; |
103 | 665k | uint32_t w0, w1, w2, w3, w4, w5, w6, w7, w8, w9, w10, w11, w12, w13, w14, w15; |
104 | | |
105 | 665k | Round(a, b, c, d, e, f, g, h, 0x428a2f98 + (w0 = ReadBE32(chunk + 0))); |
106 | 665k | Round(h, a, b, c, d, e, f, g, 0x71374491 + (w1 = ReadBE32(chunk + 4))); |
107 | 665k | Round(g, h, a, b, c, d, e, f, 0xb5c0fbcf + (w2 = ReadBE32(chunk + 8))); |
108 | 665k | Round(f, g, h, a, b, c, d, e, 0xe9b5dba5 + (w3 = ReadBE32(chunk + 12))); |
109 | 665k | Round(e, f, g, h, a, b, c, d, 0x3956c25b + (w4 = ReadBE32(chunk + 16))); |
110 | 665k | Round(d, e, f, g, h, a, b, c, 0x59f111f1 + (w5 = ReadBE32(chunk + 20))); |
111 | 665k | Round(c, d, e, f, g, h, a, b, 0x923f82a4 + (w6 = ReadBE32(chunk + 24))); |
112 | 665k | Round(b, c, d, e, f, g, h, a, 0xab1c5ed5 + (w7 = ReadBE32(chunk + 28))); |
113 | 665k | Round(a, b, c, d, e, f, g, h, 0xd807aa98 + (w8 = ReadBE32(chunk + 32))); |
114 | 665k | Round(h, a, b, c, d, e, f, g, 0x12835b01 + (w9 = ReadBE32(chunk + 36))); |
115 | 665k | Round(g, h, a, b, c, d, e, f, 0x243185be + (w10 = ReadBE32(chunk + 40))); |
116 | 665k | Round(f, g, h, a, b, c, d, e, 0x550c7dc3 + (w11 = ReadBE32(chunk + 44))); |
117 | 665k | Round(e, f, g, h, a, b, c, d, 0x72be5d74 + (w12 = ReadBE32(chunk + 48))); |
118 | 665k | Round(d, e, f, g, h, a, b, c, 0x80deb1fe + (w13 = ReadBE32(chunk + 52))); |
119 | 665k | Round(c, d, e, f, g, h, a, b, 0x9bdc06a7 + (w14 = ReadBE32(chunk + 56))); |
120 | 665k | Round(b, c, d, e, f, g, h, a, 0xc19bf174 + (w15 = ReadBE32(chunk + 60))); |
121 | | |
122 | 665k | Round(a, b, c, d, e, f, g, h, 0xe49b69c1 + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
123 | 665k | Round(h, a, b, c, d, e, f, g, 0xefbe4786 + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
124 | 665k | Round(g, h, a, b, c, d, e, f, 0x0fc19dc6 + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
125 | 665k | Round(f, g, h, a, b, c, d, e, 0x240ca1cc + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
126 | 665k | Round(e, f, g, h, a, b, c, d, 0x2de92c6f + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
127 | 665k | Round(d, e, f, g, h, a, b, c, 0x4a7484aa + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
128 | 665k | Round(c, d, e, f, g, h, a, b, 0x5cb0a9dc + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
129 | 665k | Round(b, c, d, e, f, g, h, a, 0x76f988da + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
130 | 665k | Round(a, b, c, d, e, f, g, h, 0x983e5152 + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
131 | 665k | Round(h, a, b, c, d, e, f, g, 0xa831c66d + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
132 | 665k | Round(g, h, a, b, c, d, e, f, 0xb00327c8 + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
133 | 665k | Round(f, g, h, a, b, c, d, e, 0xbf597fc7 + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
134 | 665k | Round(e, f, g, h, a, b, c, d, 0xc6e00bf3 + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
135 | 665k | Round(d, e, f, g, h, a, b, c, 0xd5a79147 + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
136 | 665k | Round(c, d, e, f, g, h, a, b, 0x06ca6351 + (w14 += sigma1(w12) + w7 + sigma0(w15))); |
137 | 665k | Round(b, c, d, e, f, g, h, a, 0x14292967 + (w15 += sigma1(w13) + w8 + sigma0(w0))); |
138 | | |
139 | 665k | Round(a, b, c, d, e, f, g, h, 0x27b70a85 + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
140 | 665k | Round(h, a, b, c, d, e, f, g, 0x2e1b2138 + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
141 | 665k | Round(g, h, a, b, c, d, e, f, 0x4d2c6dfc + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
142 | 665k | Round(f, g, h, a, b, c, d, e, 0x53380d13 + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
143 | 665k | Round(e, f, g, h, a, b, c, d, 0x650a7354 + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
144 | 665k | Round(d, e, f, g, h, a, b, c, 0x766a0abb + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
145 | 665k | Round(c, d, e, f, g, h, a, b, 0x81c2c92e + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
146 | 665k | Round(b, c, d, e, f, g, h, a, 0x92722c85 + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
147 | 665k | Round(a, b, c, d, e, f, g, h, 0xa2bfe8a1 + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
148 | 665k | Round(h, a, b, c, d, e, f, g, 0xa81a664b + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
149 | 665k | Round(g, h, a, b, c, d, e, f, 0xc24b8b70 + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
150 | 665k | Round(f, g, h, a, b, c, d, e, 0xc76c51a3 + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
151 | 665k | Round(e, f, g, h, a, b, c, d, 0xd192e819 + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
152 | 665k | Round(d, e, f, g, h, a, b, c, 0xd6990624 + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
153 | 665k | Round(c, d, e, f, g, h, a, b, 0xf40e3585 + (w14 += sigma1(w12) + w7 + sigma0(w15))); |
154 | 665k | Round(b, c, d, e, f, g, h, a, 0x106aa070 + (w15 += sigma1(w13) + w8 + sigma0(w0))); |
155 | | |
156 | 665k | Round(a, b, c, d, e, f, g, h, 0x19a4c116 + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
157 | 665k | Round(h, a, b, c, d, e, f, g, 0x1e376c08 + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
158 | 665k | Round(g, h, a, b, c, d, e, f, 0x2748774c + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
159 | 665k | Round(f, g, h, a, b, c, d, e, 0x34b0bcb5 + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
160 | 665k | Round(e, f, g, h, a, b, c, d, 0x391c0cb3 + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
161 | 665k | Round(d, e, f, g, h, a, b, c, 0x4ed8aa4a + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
162 | 665k | Round(c, d, e, f, g, h, a, b, 0x5b9cca4f + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
163 | 665k | Round(b, c, d, e, f, g, h, a, 0x682e6ff3 + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
164 | 665k | Round(a, b, c, d, e, f, g, h, 0x748f82ee + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
165 | 665k | Round(h, a, b, c, d, e, f, g, 0x78a5636f + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
166 | 665k | Round(g, h, a, b, c, d, e, f, 0x84c87814 + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
167 | 665k | Round(f, g, h, a, b, c, d, e, 0x8cc70208 + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
168 | 665k | Round(e, f, g, h, a, b, c, d, 0x90befffa + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
169 | 665k | Round(d, e, f, g, h, a, b, c, 0xa4506ceb + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
170 | 665k | Round(c, d, e, f, g, h, a, b, 0xbef9a3f7 + (w14 + sigma1(w12) + w7 + sigma0(w15))); |
171 | 665k | Round(b, c, d, e, f, g, h, a, 0xc67178f2 + (w15 + sigma1(w13) + w8 + sigma0(w0))); |
172 | | |
173 | 665k | s[0] += a; |
174 | 665k | s[1] += b; |
175 | 665k | s[2] += c; |
176 | 665k | s[3] += d; |
177 | 665k | s[4] += e; |
178 | 665k | s[5] += f; |
179 | 665k | s[6] += g; |
180 | 665k | s[7] += h; |
181 | 665k | chunk += 64; |
182 | 665k | } |
183 | 665k | } |
184 | | |
185 | | void TransformD64(unsigned char* out, const unsigned char* in) |
186 | 0 | { |
187 | | // Transform 1 |
188 | 0 | uint32_t a = 0x6a09e667ul; |
189 | 0 | uint32_t b = 0xbb67ae85ul; |
190 | 0 | uint32_t c = 0x3c6ef372ul; |
191 | 0 | uint32_t d = 0xa54ff53aul; |
192 | 0 | uint32_t e = 0x510e527ful; |
193 | 0 | uint32_t f = 0x9b05688cul; |
194 | 0 | uint32_t g = 0x1f83d9abul; |
195 | 0 | uint32_t h = 0x5be0cd19ul; |
196 | |
|
197 | 0 | uint32_t w0, w1, w2, w3, w4, w5, w6, w7, w8, w9, w10, w11, w12, w13, w14, w15; |
198 | |
|
199 | 0 | Round(a, b, c, d, e, f, g, h, 0x428a2f98ul + (w0 = ReadBE32(in + 0))); |
200 | 0 | Round(h, a, b, c, d, e, f, g, 0x71374491ul + (w1 = ReadBE32(in + 4))); |
201 | 0 | Round(g, h, a, b, c, d, e, f, 0xb5c0fbcful + (w2 = ReadBE32(in + 8))); |
202 | 0 | Round(f, g, h, a, b, c, d, e, 0xe9b5dba5ul + (w3 = ReadBE32(in + 12))); |
203 | 0 | Round(e, f, g, h, a, b, c, d, 0x3956c25bul + (w4 = ReadBE32(in + 16))); |
204 | 0 | Round(d, e, f, g, h, a, b, c, 0x59f111f1ul + (w5 = ReadBE32(in + 20))); |
205 | 0 | Round(c, d, e, f, g, h, a, b, 0x923f82a4ul + (w6 = ReadBE32(in + 24))); |
206 | 0 | Round(b, c, d, e, f, g, h, a, 0xab1c5ed5ul + (w7 = ReadBE32(in + 28))); |
207 | 0 | Round(a, b, c, d, e, f, g, h, 0xd807aa98ul + (w8 = ReadBE32(in + 32))); |
208 | 0 | Round(h, a, b, c, d, e, f, g, 0x12835b01ul + (w9 = ReadBE32(in + 36))); |
209 | 0 | Round(g, h, a, b, c, d, e, f, 0x243185beul + (w10 = ReadBE32(in + 40))); |
210 | 0 | Round(f, g, h, a, b, c, d, e, 0x550c7dc3ul + (w11 = ReadBE32(in + 44))); |
211 | 0 | Round(e, f, g, h, a, b, c, d, 0x72be5d74ul + (w12 = ReadBE32(in + 48))); |
212 | 0 | Round(d, e, f, g, h, a, b, c, 0x80deb1feul + (w13 = ReadBE32(in + 52))); |
213 | 0 | Round(c, d, e, f, g, h, a, b, 0x9bdc06a7ul + (w14 = ReadBE32(in + 56))); |
214 | 0 | Round(b, c, d, e, f, g, h, a, 0xc19bf174ul + (w15 = ReadBE32(in + 60))); |
215 | 0 | Round(a, b, c, d, e, f, g, h, 0xe49b69c1ul + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
216 | 0 | Round(h, a, b, c, d, e, f, g, 0xefbe4786ul + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
217 | 0 | Round(g, h, a, b, c, d, e, f, 0x0fc19dc6ul + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
218 | 0 | Round(f, g, h, a, b, c, d, e, 0x240ca1ccul + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
219 | 0 | Round(e, f, g, h, a, b, c, d, 0x2de92c6ful + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
220 | 0 | Round(d, e, f, g, h, a, b, c, 0x4a7484aaul + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
221 | 0 | Round(c, d, e, f, g, h, a, b, 0x5cb0a9dcul + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
222 | 0 | Round(b, c, d, e, f, g, h, a, 0x76f988daul + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
223 | 0 | Round(a, b, c, d, e, f, g, h, 0x983e5152ul + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
224 | 0 | Round(h, a, b, c, d, e, f, g, 0xa831c66dul + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
225 | 0 | Round(g, h, a, b, c, d, e, f, 0xb00327c8ul + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
226 | 0 | Round(f, g, h, a, b, c, d, e, 0xbf597fc7ul + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
227 | 0 | Round(e, f, g, h, a, b, c, d, 0xc6e00bf3ul + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
228 | 0 | Round(d, e, f, g, h, a, b, c, 0xd5a79147ul + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
229 | 0 | Round(c, d, e, f, g, h, a, b, 0x06ca6351ul + (w14 += sigma1(w12) + w7 + sigma0(w15))); |
230 | 0 | Round(b, c, d, e, f, g, h, a, 0x14292967ul + (w15 += sigma1(w13) + w8 + sigma0(w0))); |
231 | 0 | Round(a, b, c, d, e, f, g, h, 0x27b70a85ul + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
232 | 0 | Round(h, a, b, c, d, e, f, g, 0x2e1b2138ul + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
233 | 0 | Round(g, h, a, b, c, d, e, f, 0x4d2c6dfcul + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
234 | 0 | Round(f, g, h, a, b, c, d, e, 0x53380d13ul + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
235 | 0 | Round(e, f, g, h, a, b, c, d, 0x650a7354ul + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
236 | 0 | Round(d, e, f, g, h, a, b, c, 0x766a0abbul + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
237 | 0 | Round(c, d, e, f, g, h, a, b, 0x81c2c92eul + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
238 | 0 | Round(b, c, d, e, f, g, h, a, 0x92722c85ul + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
239 | 0 | Round(a, b, c, d, e, f, g, h, 0xa2bfe8a1ul + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
240 | 0 | Round(h, a, b, c, d, e, f, g, 0xa81a664bul + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
241 | 0 | Round(g, h, a, b, c, d, e, f, 0xc24b8b70ul + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
242 | 0 | Round(f, g, h, a, b, c, d, e, 0xc76c51a3ul + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
243 | 0 | Round(e, f, g, h, a, b, c, d, 0xd192e819ul + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
244 | 0 | Round(d, e, f, g, h, a, b, c, 0xd6990624ul + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
245 | 0 | Round(c, d, e, f, g, h, a, b, 0xf40e3585ul + (w14 += sigma1(w12) + w7 + sigma0(w15))); |
246 | 0 | Round(b, c, d, e, f, g, h, a, 0x106aa070ul + (w15 += sigma1(w13) + w8 + sigma0(w0))); |
247 | 0 | Round(a, b, c, d, e, f, g, h, 0x19a4c116ul + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
248 | 0 | Round(h, a, b, c, d, e, f, g, 0x1e376c08ul + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
249 | 0 | Round(g, h, a, b, c, d, e, f, 0x2748774cul + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
250 | 0 | Round(f, g, h, a, b, c, d, e, 0x34b0bcb5ul + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
251 | 0 | Round(e, f, g, h, a, b, c, d, 0x391c0cb3ul + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
252 | 0 | Round(d, e, f, g, h, a, b, c, 0x4ed8aa4aul + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
253 | 0 | Round(c, d, e, f, g, h, a, b, 0x5b9cca4ful + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
254 | 0 | Round(b, c, d, e, f, g, h, a, 0x682e6ff3ul + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
255 | 0 | Round(a, b, c, d, e, f, g, h, 0x748f82eeul + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
256 | 0 | Round(h, a, b, c, d, e, f, g, 0x78a5636ful + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
257 | 0 | Round(g, h, a, b, c, d, e, f, 0x84c87814ul + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
258 | 0 | Round(f, g, h, a, b, c, d, e, 0x8cc70208ul + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
259 | 0 | Round(e, f, g, h, a, b, c, d, 0x90befffaul + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
260 | 0 | Round(d, e, f, g, h, a, b, c, 0xa4506cebul + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
261 | 0 | Round(c, d, e, f, g, h, a, b, 0xbef9a3f7ul + (w14 + sigma1(w12) + w7 + sigma0(w15))); |
262 | 0 | Round(b, c, d, e, f, g, h, a, 0xc67178f2ul + (w15 + sigma1(w13) + w8 + sigma0(w0))); |
263 | |
|
264 | 0 | a += 0x6a09e667ul; |
265 | 0 | b += 0xbb67ae85ul; |
266 | 0 | c += 0x3c6ef372ul; |
267 | 0 | d += 0xa54ff53aul; |
268 | 0 | e += 0x510e527ful; |
269 | 0 | f += 0x9b05688cul; |
270 | 0 | g += 0x1f83d9abul; |
271 | 0 | h += 0x5be0cd19ul; |
272 | |
|
273 | 0 | uint32_t t0 = a, t1 = b, t2 = c, t3 = d, t4 = e, t5 = f, t6 = g, t7 = h; |
274 | | |
275 | | // Transform 2 |
276 | 0 | Round(a, b, c, d, e, f, g, h, 0xc28a2f98ul); |
277 | 0 | Round(h, a, b, c, d, e, f, g, 0x71374491ul); |
278 | 0 | Round(g, h, a, b, c, d, e, f, 0xb5c0fbcful); |
279 | 0 | Round(f, g, h, a, b, c, d, e, 0xe9b5dba5ul); |
280 | 0 | Round(e, f, g, h, a, b, c, d, 0x3956c25bul); |
281 | 0 | Round(d, e, f, g, h, a, b, c, 0x59f111f1ul); |
282 | 0 | Round(c, d, e, f, g, h, a, b, 0x923f82a4ul); |
283 | 0 | Round(b, c, d, e, f, g, h, a, 0xab1c5ed5ul); |
284 | 0 | Round(a, b, c, d, e, f, g, h, 0xd807aa98ul); |
285 | 0 | Round(h, a, b, c, d, e, f, g, 0x12835b01ul); |
286 | 0 | Round(g, h, a, b, c, d, e, f, 0x243185beul); |
287 | 0 | Round(f, g, h, a, b, c, d, e, 0x550c7dc3ul); |
288 | 0 | Round(e, f, g, h, a, b, c, d, 0x72be5d74ul); |
289 | 0 | Round(d, e, f, g, h, a, b, c, 0x80deb1feul); |
290 | 0 | Round(c, d, e, f, g, h, a, b, 0x9bdc06a7ul); |
291 | 0 | Round(b, c, d, e, f, g, h, a, 0xc19bf374ul); |
292 | 0 | Round(a, b, c, d, e, f, g, h, 0x649b69c1ul); |
293 | 0 | Round(h, a, b, c, d, e, f, g, 0xf0fe4786ul); |
294 | 0 | Round(g, h, a, b, c, d, e, f, 0x0fe1edc6ul); |
295 | 0 | Round(f, g, h, a, b, c, d, e, 0x240cf254ul); |
296 | 0 | Round(e, f, g, h, a, b, c, d, 0x4fe9346ful); |
297 | 0 | Round(d, e, f, g, h, a, b, c, 0x6cc984beul); |
298 | 0 | Round(c, d, e, f, g, h, a, b, 0x61b9411eul); |
299 | 0 | Round(b, c, d, e, f, g, h, a, 0x16f988faul); |
300 | 0 | Round(a, b, c, d, e, f, g, h, 0xf2c65152ul); |
301 | 0 | Round(h, a, b, c, d, e, f, g, 0xa88e5a6dul); |
302 | 0 | Round(g, h, a, b, c, d, e, f, 0xb019fc65ul); |
303 | 0 | Round(f, g, h, a, b, c, d, e, 0xb9d99ec7ul); |
304 | 0 | Round(e, f, g, h, a, b, c, d, 0x9a1231c3ul); |
305 | 0 | Round(d, e, f, g, h, a, b, c, 0xe70eeaa0ul); |
306 | 0 | Round(c, d, e, f, g, h, a, b, 0xfdb1232bul); |
307 | 0 | Round(b, c, d, e, f, g, h, a, 0xc7353eb0ul); |
308 | 0 | Round(a, b, c, d, e, f, g, h, 0x3069bad5ul); |
309 | 0 | Round(h, a, b, c, d, e, f, g, 0xcb976d5ful); |
310 | 0 | Round(g, h, a, b, c, d, e, f, 0x5a0f118ful); |
311 | 0 | Round(f, g, h, a, b, c, d, e, 0xdc1eeefdul); |
312 | 0 | Round(e, f, g, h, a, b, c, d, 0x0a35b689ul); |
313 | 0 | Round(d, e, f, g, h, a, b, c, 0xde0b7a04ul); |
314 | 0 | Round(c, d, e, f, g, h, a, b, 0x58f4ca9dul); |
315 | 0 | Round(b, c, d, e, f, g, h, a, 0xe15d5b16ul); |
316 | 0 | Round(a, b, c, d, e, f, g, h, 0x007f3e86ul); |
317 | 0 | Round(h, a, b, c, d, e, f, g, 0x37088980ul); |
318 | 0 | Round(g, h, a, b, c, d, e, f, 0xa507ea32ul); |
319 | 0 | Round(f, g, h, a, b, c, d, e, 0x6fab9537ul); |
320 | 0 | Round(e, f, g, h, a, b, c, d, 0x17406110ul); |
321 | 0 | Round(d, e, f, g, h, a, b, c, 0x0d8cd6f1ul); |
322 | 0 | Round(c, d, e, f, g, h, a, b, 0xcdaa3b6dul); |
323 | 0 | Round(b, c, d, e, f, g, h, a, 0xc0bbbe37ul); |
324 | 0 | Round(a, b, c, d, e, f, g, h, 0x83613bdaul); |
325 | 0 | Round(h, a, b, c, d, e, f, g, 0xdb48a363ul); |
326 | 0 | Round(g, h, a, b, c, d, e, f, 0x0b02e931ul); |
327 | 0 | Round(f, g, h, a, b, c, d, e, 0x6fd15ca7ul); |
328 | 0 | Round(e, f, g, h, a, b, c, d, 0x521afacaul); |
329 | 0 | Round(d, e, f, g, h, a, b, c, 0x31338431ul); |
330 | 0 | Round(c, d, e, f, g, h, a, b, 0x6ed41a95ul); |
331 | 0 | Round(b, c, d, e, f, g, h, a, 0x6d437890ul); |
332 | 0 | Round(a, b, c, d, e, f, g, h, 0xc39c91f2ul); |
333 | 0 | Round(h, a, b, c, d, e, f, g, 0x9eccabbdul); |
334 | 0 | Round(g, h, a, b, c, d, e, f, 0xb5c9a0e6ul); |
335 | 0 | Round(f, g, h, a, b, c, d, e, 0x532fb63cul); |
336 | 0 | Round(e, f, g, h, a, b, c, d, 0xd2c741c6ul); |
337 | 0 | Round(d, e, f, g, h, a, b, c, 0x07237ea3ul); |
338 | 0 | Round(c, d, e, f, g, h, a, b, 0xa4954b68ul); |
339 | 0 | Round(b, c, d, e, f, g, h, a, 0x4c191d76ul); |
340 | |
|
341 | 0 | w0 = t0 + a; |
342 | 0 | w1 = t1 + b; |
343 | 0 | w2 = t2 + c; |
344 | 0 | w3 = t3 + d; |
345 | 0 | w4 = t4 + e; |
346 | 0 | w5 = t5 + f; |
347 | 0 | w6 = t6 + g; |
348 | 0 | w7 = t7 + h; |
349 | | |
350 | | // Transform 3 |
351 | 0 | a = 0x6a09e667ul; |
352 | 0 | b = 0xbb67ae85ul; |
353 | 0 | c = 0x3c6ef372ul; |
354 | 0 | d = 0xa54ff53aul; |
355 | 0 | e = 0x510e527ful; |
356 | 0 | f = 0x9b05688cul; |
357 | 0 | g = 0x1f83d9abul; |
358 | 0 | h = 0x5be0cd19ul; |
359 | |
|
360 | 0 | Round(a, b, c, d, e, f, g, h, 0x428a2f98ul + w0); |
361 | 0 | Round(h, a, b, c, d, e, f, g, 0x71374491ul + w1); |
362 | 0 | Round(g, h, a, b, c, d, e, f, 0xb5c0fbcful + w2); |
363 | 0 | Round(f, g, h, a, b, c, d, e, 0xe9b5dba5ul + w3); |
364 | 0 | Round(e, f, g, h, a, b, c, d, 0x3956c25bul + w4); |
365 | 0 | Round(d, e, f, g, h, a, b, c, 0x59f111f1ul + w5); |
366 | 0 | Round(c, d, e, f, g, h, a, b, 0x923f82a4ul + w6); |
367 | 0 | Round(b, c, d, e, f, g, h, a, 0xab1c5ed5ul + w7); |
368 | 0 | Round(a, b, c, d, e, f, g, h, 0x5807aa98ul); |
369 | 0 | Round(h, a, b, c, d, e, f, g, 0x12835b01ul); |
370 | 0 | Round(g, h, a, b, c, d, e, f, 0x243185beul); |
371 | 0 | Round(f, g, h, a, b, c, d, e, 0x550c7dc3ul); |
372 | 0 | Round(e, f, g, h, a, b, c, d, 0x72be5d74ul); |
373 | 0 | Round(d, e, f, g, h, a, b, c, 0x80deb1feul); |
374 | 0 | Round(c, d, e, f, g, h, a, b, 0x9bdc06a7ul); |
375 | 0 | Round(b, c, d, e, f, g, h, a, 0xc19bf274ul); |
376 | 0 | Round(a, b, c, d, e, f, g, h, 0xe49b69c1ul + (w0 += sigma0(w1))); |
377 | 0 | Round(h, a, b, c, d, e, f, g, 0xefbe4786ul + (w1 += 0xa00000ul + sigma0(w2))); |
378 | 0 | Round(g, h, a, b, c, d, e, f, 0x0fc19dc6ul + (w2 += sigma1(w0) + sigma0(w3))); |
379 | 0 | Round(f, g, h, a, b, c, d, e, 0x240ca1ccul + (w3 += sigma1(w1) + sigma0(w4))); |
380 | 0 | Round(e, f, g, h, a, b, c, d, 0x2de92c6ful + (w4 += sigma1(w2) + sigma0(w5))); |
381 | 0 | Round(d, e, f, g, h, a, b, c, 0x4a7484aaul + (w5 += sigma1(w3) + sigma0(w6))); |
382 | 0 | Round(c, d, e, f, g, h, a, b, 0x5cb0a9dcul + (w6 += sigma1(w4) + 0x100ul + sigma0(w7))); |
383 | 0 | Round(b, c, d, e, f, g, h, a, 0x76f988daul + (w7 += sigma1(w5) + w0 + 0x11002000ul)); |
384 | 0 | Round(a, b, c, d, e, f, g, h, 0x983e5152ul + (w8 = 0x80000000ul + sigma1(w6) + w1)); |
385 | 0 | Round(h, a, b, c, d, e, f, g, 0xa831c66dul + (w9 = sigma1(w7) + w2)); |
386 | 0 | Round(g, h, a, b, c, d, e, f, 0xb00327c8ul + (w10 = sigma1(w8) + w3)); |
387 | 0 | Round(f, g, h, a, b, c, d, e, 0xbf597fc7ul + (w11 = sigma1(w9) + w4)); |
388 | 0 | Round(e, f, g, h, a, b, c, d, 0xc6e00bf3ul + (w12 = sigma1(w10) + w5)); |
389 | 0 | Round(d, e, f, g, h, a, b, c, 0xd5a79147ul + (w13 = sigma1(w11) + w6)); |
390 | 0 | Round(c, d, e, f, g, h, a, b, 0x06ca6351ul + (w14 = sigma1(w12) + w7 + 0x400022ul)); |
391 | 0 | Round(b, c, d, e, f, g, h, a, 0x14292967ul + (w15 = 0x100ul + sigma1(w13) + w8 + sigma0(w0))); |
392 | 0 | Round(a, b, c, d, e, f, g, h, 0x27b70a85ul + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
393 | 0 | Round(h, a, b, c, d, e, f, g, 0x2e1b2138ul + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
394 | 0 | Round(g, h, a, b, c, d, e, f, 0x4d2c6dfcul + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
395 | 0 | Round(f, g, h, a, b, c, d, e, 0x53380d13ul + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
396 | 0 | Round(e, f, g, h, a, b, c, d, 0x650a7354ul + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
397 | 0 | Round(d, e, f, g, h, a, b, c, 0x766a0abbul + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
398 | 0 | Round(c, d, e, f, g, h, a, b, 0x81c2c92eul + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
399 | 0 | Round(b, c, d, e, f, g, h, a, 0x92722c85ul + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
400 | 0 | Round(a, b, c, d, e, f, g, h, 0xa2bfe8a1ul + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
401 | 0 | Round(h, a, b, c, d, e, f, g, 0xa81a664bul + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
402 | 0 | Round(g, h, a, b, c, d, e, f, 0xc24b8b70ul + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
403 | 0 | Round(f, g, h, a, b, c, d, e, 0xc76c51a3ul + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
404 | 0 | Round(e, f, g, h, a, b, c, d, 0xd192e819ul + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
405 | 0 | Round(d, e, f, g, h, a, b, c, 0xd6990624ul + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
406 | 0 | Round(c, d, e, f, g, h, a, b, 0xf40e3585ul + (w14 += sigma1(w12) + w7 + sigma0(w15))); |
407 | 0 | Round(b, c, d, e, f, g, h, a, 0x106aa070ul + (w15 += sigma1(w13) + w8 + sigma0(w0))); |
408 | 0 | Round(a, b, c, d, e, f, g, h, 0x19a4c116ul + (w0 += sigma1(w14) + w9 + sigma0(w1))); |
409 | 0 | Round(h, a, b, c, d, e, f, g, 0x1e376c08ul + (w1 += sigma1(w15) + w10 + sigma0(w2))); |
410 | 0 | Round(g, h, a, b, c, d, e, f, 0x2748774cul + (w2 += sigma1(w0) + w11 + sigma0(w3))); |
411 | 0 | Round(f, g, h, a, b, c, d, e, 0x34b0bcb5ul + (w3 += sigma1(w1) + w12 + sigma0(w4))); |
412 | 0 | Round(e, f, g, h, a, b, c, d, 0x391c0cb3ul + (w4 += sigma1(w2) + w13 + sigma0(w5))); |
413 | 0 | Round(d, e, f, g, h, a, b, c, 0x4ed8aa4aul + (w5 += sigma1(w3) + w14 + sigma0(w6))); |
414 | 0 | Round(c, d, e, f, g, h, a, b, 0x5b9cca4ful + (w6 += sigma1(w4) + w15 + sigma0(w7))); |
415 | 0 | Round(b, c, d, e, f, g, h, a, 0x682e6ff3ul + (w7 += sigma1(w5) + w0 + sigma0(w8))); |
416 | 0 | Round(a, b, c, d, e, f, g, h, 0x748f82eeul + (w8 += sigma1(w6) + w1 + sigma0(w9))); |
417 | 0 | Round(h, a, b, c, d, e, f, g, 0x78a5636ful + (w9 += sigma1(w7) + w2 + sigma0(w10))); |
418 | 0 | Round(g, h, a, b, c, d, e, f, 0x84c87814ul + (w10 += sigma1(w8) + w3 + sigma0(w11))); |
419 | 0 | Round(f, g, h, a, b, c, d, e, 0x8cc70208ul + (w11 += sigma1(w9) + w4 + sigma0(w12))); |
420 | 0 | Round(e, f, g, h, a, b, c, d, 0x90befffaul + (w12 += sigma1(w10) + w5 + sigma0(w13))); |
421 | 0 | Round(d, e, f, g, h, a, b, c, 0xa4506cebul + (w13 += sigma1(w11) + w6 + sigma0(w14))); |
422 | 0 | Round(c, d, e, f, g, h, a, b, 0xbef9a3f7ul + (w14 + sigma1(w12) + w7 + sigma0(w15))); |
423 | 0 | Round(b, c, d, e, f, g, h, a, 0xc67178f2ul + (w15 + sigma1(w13) + w8 + sigma0(w0))); |
424 | | |
425 | | // Output |
426 | 0 | WriteBE32(out + 0, a + 0x6a09e667ul); |
427 | 0 | WriteBE32(out + 4, b + 0xbb67ae85ul); |
428 | 0 | WriteBE32(out + 8, c + 0x3c6ef372ul); |
429 | 0 | WriteBE32(out + 12, d + 0xa54ff53aul); |
430 | 0 | WriteBE32(out + 16, e + 0x510e527ful); |
431 | 0 | WriteBE32(out + 20, f + 0x9b05688cul); |
432 | 0 | WriteBE32(out + 24, g + 0x1f83d9abul); |
433 | 0 | WriteBE32(out + 28, h + 0x5be0cd19ul); |
434 | 0 | } |
435 | | |
436 | | } // namespace sha256 |
437 | | |
438 | | typedef void (*TransformType)(uint32_t*, const unsigned char*, size_t); |
439 | | typedef void (*TransformD64Type)(unsigned char*, const unsigned char*); |
440 | | |
441 | | template<TransformType tr> |
442 | | void TransformD64Wrapper(unsigned char* out, const unsigned char* in) |
443 | 56.8k | { |
444 | 56.8k | uint32_t s[8]; |
445 | 56.8k | static const unsigned char padding1[64] = { |
446 | 56.8k | 0x80, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, |
447 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, |
448 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, |
449 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2, 0 |
450 | 56.8k | }; |
451 | 56.8k | unsigned char buffer2[64] = { |
452 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, |
453 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, |
454 | 56.8k | 0x80, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, |
455 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0 |
456 | 56.8k | }; |
457 | 56.8k | sha256::Initialize(s); |
458 | 56.8k | tr(s, in, 1); |
459 | 56.8k | tr(s, padding1, 1); |
460 | 56.8k | WriteBE32(buffer2 + 0, s[0]); |
461 | 56.8k | WriteBE32(buffer2 + 4, s[1]); |
462 | 56.8k | WriteBE32(buffer2 + 8, s[2]); |
463 | 56.8k | WriteBE32(buffer2 + 12, s[3]); |
464 | 56.8k | WriteBE32(buffer2 + 16, s[4]); |
465 | 56.8k | WriteBE32(buffer2 + 20, s[5]); |
466 | 56.8k | WriteBE32(buffer2 + 24, s[6]); |
467 | 56.8k | WriteBE32(buffer2 + 28, s[7]); |
468 | 56.8k | sha256::Initialize(s); |
469 | 56.8k | tr(s, buffer2, 1); |
470 | 56.8k | WriteBE32(out + 0, s[0]); |
471 | 56.8k | WriteBE32(out + 4, s[1]); |
472 | 56.8k | WriteBE32(out + 8, s[2]); |
473 | 56.8k | WriteBE32(out + 12, s[3]); |
474 | 56.8k | WriteBE32(out + 16, s[4]); |
475 | 56.8k | WriteBE32(out + 20, s[5]); |
476 | 56.8k | WriteBE32(out + 24, s[6]); |
477 | 56.8k | WriteBE32(out + 28, s[7]); |
478 | 56.8k | } sha256.cpp:void (anonymous namespace)::TransformD64Wrapper<&sha256_x86_shani::Transform>(unsigned char*, unsigned char const*) Line | Count | Source | 443 | 56.8k | { | 444 | 56.8k | uint32_t s[8]; | 445 | 56.8k | static const unsigned char padding1[64] = { | 446 | 56.8k | 0x80, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, | 447 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, | 448 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, | 449 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2, 0 | 450 | 56.8k | }; | 451 | 56.8k | unsigned char buffer2[64] = { | 452 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, | 453 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, | 454 | 56.8k | 0x80, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, | 455 | 56.8k | 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0 | 456 | 56.8k | }; | 457 | 56.8k | sha256::Initialize(s); | 458 | 56.8k | tr(s, in, 1); | 459 | 56.8k | tr(s, padding1, 1); | 460 | 56.8k | WriteBE32(buffer2 + 0, s[0]); | 461 | 56.8k | WriteBE32(buffer2 + 4, s[1]); | 462 | 56.8k | WriteBE32(buffer2 + 8, s[2]); | 463 | 56.8k | WriteBE32(buffer2 + 12, s[3]); | 464 | 56.8k | WriteBE32(buffer2 + 16, s[4]); | 465 | 56.8k | WriteBE32(buffer2 + 20, s[5]); | 466 | 56.8k | WriteBE32(buffer2 + 24, s[6]); | 467 | 56.8k | WriteBE32(buffer2 + 28, s[7]); | 468 | 56.8k | sha256::Initialize(s); | 469 | 56.8k | tr(s, buffer2, 1); | 470 | 56.8k | WriteBE32(out + 0, s[0]); | 471 | 56.8k | WriteBE32(out + 4, s[1]); | 472 | 56.8k | WriteBE32(out + 8, s[2]); | 473 | 56.8k | WriteBE32(out + 12, s[3]); | 474 | 56.8k | WriteBE32(out + 16, s[4]); | 475 | 56.8k | WriteBE32(out + 20, s[5]); | 476 | 56.8k | WriteBE32(out + 24, s[6]); | 477 | 56.8k | WriteBE32(out + 28, s[7]); | 478 | 56.8k | } |
Unexecuted instantiation: sha256.cpp:void (anonymous namespace)::TransformD64Wrapper<&sha256_sse4::Transform>(unsigned char*, unsigned char const*) |
479 | | |
480 | | TransformType Transform = sha256::Transform; |
481 | | TransformD64Type TransformD64 = sha256::TransformD64; |
482 | | TransformD64Type TransformD64_2way = nullptr; |
483 | | TransformD64Type TransformD64_4way = nullptr; |
484 | | TransformD64Type TransformD64_8way = nullptr; |
485 | | |
486 | 11.0k | bool SelfTest() { |
487 | | // Input state (equal to the initial SHA256 state) |
488 | 11.0k | static const uint32_t init[8] = { |
489 | 11.0k | 0x6a09e667ul, 0xbb67ae85ul, 0x3c6ef372ul, 0xa54ff53aul, 0x510e527ful, 0x9b05688cul, 0x1f83d9abul, 0x5be0cd19ul |
490 | 11.0k | }; |
491 | | // Some random input data to test with |
492 | 11.0k | static const unsigned char data[641] = "-" // Intentionally not aligned |
493 | 11.0k | "Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do " |
494 | 11.0k | "eiusmod tempor incididunt ut labore et dolore magna aliqua. Et m" |
495 | 11.0k | "olestie ac feugiat sed lectus vestibulum mattis ullamcorper. Mor" |
496 | 11.0k | "bi blandit cursus risus at ultrices mi tempus imperdiet nulla. N" |
497 | 11.0k | "unc congue nisi vita suscipit tellus mauris. Imperdiet proin fer" |
498 | 11.0k | "mentum leo vel orci. Massa tempor nec feugiat nisl pretium fusce" |
499 | 11.0k | " id velit. Telus in metus vulputate eu scelerisque felis. Mi tem" |
500 | 11.0k | "pus imperdiet nulla malesuada pellentesque. Tristique magna sit."; |
501 | | // Expected output state for hashing the i*64 first input bytes above (excluding SHA256 padding). |
502 | 11.0k | static const uint32_t result[9][8] = { |
503 | 11.0k | {0x6a09e667ul, 0xbb67ae85ul, 0x3c6ef372ul, 0xa54ff53aul, 0x510e527ful, 0x9b05688cul, 0x1f83d9abul, 0x5be0cd19ul}, |
504 | 11.0k | {0x91f8ec6bul, 0x4da10fe3ul, 0x1c9c292cul, 0x45e18185ul, 0x435cc111ul, 0x3ca26f09ul, 0xeb954caeul, 0x402a7069ul}, |
505 | 11.0k | {0xcabea5acul, 0x374fb97cul, 0x182ad996ul, 0x7bd69cbful, 0x450ff900ul, 0xc1d2be8aul, 0x6a41d505ul, 0xe6212dc3ul}, |
506 | 11.0k | {0xbcff09d6ul, 0x3e76f36eul, 0x3ecb2501ul, 0x78866e97ul, 0xe1c1e2fdul, 0x32f4eafful, 0x8aa6c4e5ul, 0xdfc024bcul}, |
507 | 11.0k | {0xa08c5d94ul, 0x0a862f93ul, 0x6b7f2f40ul, 0x8f9fae76ul, 0x6d40439ful, 0x79dcee0cul, 0x3e39ff3aul, 0xdc3bdbb1ul}, |
508 | 11.0k | {0x216a0895ul, 0x9f1a3662ul, 0xe99946f9ul, 0x87ba4364ul, 0x0fb5db2cul, 0x12bed3d3ul, 0x6689c0c7ul, 0x292f1b04ul}, |
509 | 11.0k | {0xca3067f8ul, 0xbc8c2656ul, 0x37cb7e0dul, 0x9b6b8b0ful, 0x46dc380bul, 0xf1287f57ul, 0xc42e4b23ul, 0x3fefe94dul}, |
510 | 11.0k | {0x3e4c4039ul, 0xbb6fca8cul, 0x6f27d2f7ul, 0x301e44a4ul, 0x8352ba14ul, 0x5769ce37ul, 0x48a1155ful, 0xc0e1c4c6ul}, |
511 | 11.0k | {0xfe2fa9ddul, 0x69d0862bul, 0x1ae0db23ul, 0x471f9244ul, 0xf55c0145ul, 0xc30f9c3bul, 0x40a84ea0ul, 0x5b8a266cul}, |
512 | 11.0k | }; |
513 | | // Expected output for each of the individual 8 64-byte messages under full double SHA256 (including padding). |
514 | 11.0k | static const unsigned char result_d64[256] = { |
515 | 11.0k | 0x09, 0x3a, 0xc4, 0xd0, 0x0f, 0xf7, 0x57, 0xe1, 0x72, 0x85, 0x79, 0x42, 0xfe, 0xe7, 0xe0, 0xa0, |
516 | 11.0k | 0xfc, 0x52, 0xd7, 0xdb, 0x07, 0x63, 0x45, 0xfb, 0x53, 0x14, 0x7d, 0x17, 0x22, 0x86, 0xf0, 0x52, |
517 | 11.0k | 0x48, 0xb6, 0x11, 0x9e, 0x6e, 0x48, 0x81, 0x6d, 0xcc, 0x57, 0x1f, 0xb2, 0x97, 0xa8, 0xd5, 0x25, |
518 | 11.0k | 0x9b, 0x82, 0xaa, 0x89, 0xe2, 0xfd, 0x2d, 0x56, 0xe8, 0x28, 0x83, 0x0b, 0xe2, 0xfa, 0x53, 0xb7, |
519 | 11.0k | 0xd6, 0x6b, 0x07, 0x85, 0x83, 0xb0, 0x10, 0xa2, 0xf5, 0x51, 0x3c, 0xf9, 0x60, 0x03, 0xab, 0x45, |
520 | 11.0k | 0x6c, 0x15, 0x6e, 0xef, 0xb5, 0xac, 0x3e, 0x6c, 0xdf, 0xb4, 0x92, 0x22, 0x2d, 0xce, 0xbf, 0x3e, |
521 | 11.0k | 0xe9, 0xe5, 0xf6, 0x29, 0x0e, 0x01, 0x4f, 0xd2, 0xd4, 0x45, 0x65, 0xb3, 0xbb, 0xf2, 0x4c, 0x16, |
522 | 11.0k | 0x37, 0x50, 0x3c, 0x6e, 0x49, 0x8c, 0x5a, 0x89, 0x2b, 0x1b, 0xab, 0xc4, 0x37, 0xd1, 0x46, 0xe9, |
523 | 11.0k | 0x3d, 0x0e, 0x85, 0xa2, 0x50, 0x73, 0xa1, 0x5e, 0x54, 0x37, 0xd7, 0x94, 0x17, 0x56, 0xc2, 0xd8, |
524 | 11.0k | 0xe5, 0x9f, 0xed, 0x4e, 0xae, 0x15, 0x42, 0x06, 0x0d, 0x74, 0x74, 0x5e, 0x24, 0x30, 0xce, 0xd1, |
525 | 11.0k | 0x9e, 0x50, 0xa3, 0x9a, 0xb8, 0xf0, 0x4a, 0x57, 0x69, 0x78, 0x67, 0x12, 0x84, 0x58, 0xbe, 0xc7, |
526 | 11.0k | 0x36, 0xaa, 0xee, 0x7c, 0x64, 0xa3, 0x76, 0xec, 0xff, 0x55, 0x41, 0x00, 0x2a, 0x44, 0x68, 0x4d, |
527 | 11.0k | 0xb6, 0x53, 0x9e, 0x1c, 0x95, 0xb7, 0xca, 0xdc, 0x7f, 0x7d, 0x74, 0x27, 0x5c, 0x8e, 0xa6, 0x84, |
528 | 11.0k | 0xb5, 0xac, 0x87, 0xa9, 0xf3, 0xff, 0x75, 0xf2, 0x34, 0xcd, 0x1a, 0x3b, 0x82, 0x2c, 0x2b, 0x4e, |
529 | 11.0k | 0x6a, 0x46, 0x30, 0xa6, 0x89, 0x86, 0x23, 0xac, 0xf8, 0xa5, 0x15, 0xe9, 0x0a, 0xaa, 0x1e, 0x9a, |
530 | 11.0k | 0xd7, 0x93, 0x6b, 0x28, 0xe4, 0x3b, 0xfd, 0x59, 0xc6, 0xed, 0x7c, 0x5f, 0xa5, 0x41, 0xcb, 0x51 |
531 | 11.0k | }; |
532 | | |
533 | | |
534 | | // Test Transform() for 0 through 8 transformations. |
535 | 110k | for (size_t i = 0; i <= 8; ++i) { Branch (535:24): [True: 99.8k, False: 11.0k]
|
536 | 99.8k | uint32_t state[8]; |
537 | 99.8k | std::copy(init, init + 8, state); |
538 | 99.8k | Transform(state, data + 1, i); |
539 | 99.8k | if (!std::equal(state, state + 8, result[i])) return false; Branch (539:13): [True: 0, False: 99.8k]
|
540 | 99.8k | } |
541 | | |
542 | | // Test TransformD64 |
543 | 11.0k | unsigned char out[32]; |
544 | 11.0k | TransformD64(out, data + 1); |
545 | 11.0k | if (!std::equal(out, out + 32, result_d64)) return false; Branch (545:9): [True: 0, False: 11.0k]
|
546 | | |
547 | | // Test TransformD64_2way, if available. |
548 | 11.0k | if (TransformD64_2way) { Branch (548:9): [True: 11.0k, False: 0]
|
549 | 11.0k | unsigned char out[64]; |
550 | 11.0k | TransformD64_2way(out, data + 1); |
551 | 11.0k | if (!std::equal(out, out + 64, result_d64)) return false; Branch (551:13): [True: 0, False: 11.0k]
|
552 | 11.0k | } |
553 | | |
554 | | // Test TransformD64_4way, if available. |
555 | 11.0k | if (TransformD64_4way) { Branch (555:9): [True: 0, False: 11.0k]
|
556 | 0 | unsigned char out[128]; |
557 | 0 | TransformD64_4way(out, data + 1); |
558 | 0 | if (!std::equal(out, out + 128, result_d64)) return false; Branch (558:13): [True: 0, False: 0]
|
559 | 0 | } |
560 | | |
561 | | // Test TransformD64_8way, if available. |
562 | 11.0k | if (TransformD64_8way) { Branch (562:9): [True: 0, False: 11.0k]
|
563 | 0 | unsigned char out[256]; |
564 | 0 | TransformD64_8way(out, data + 1); |
565 | 0 | if (!std::equal(out, out + 256, result_d64)) return false; Branch (565:13): [True: 0, False: 0]
|
566 | 0 | } |
567 | | |
568 | 11.0k | return true; |
569 | 11.0k | } |
570 | | |
571 | | #if !defined(DISABLE_OPTIMIZED_SHA256) |
572 | | #if (defined(__x86_64__) || defined(__amd64__) || defined(__i386__)) |
573 | | /** Check whether the OS has enabled AVX registers. */ |
574 | | bool AVXEnabled() |
575 | 11.0k | { |
576 | 11.0k | uint32_t a, d; |
577 | 11.0k | __asm__("xgetbv" : "=a"(a), "=d"(d) : "c"(0)); |
578 | 11.0k | return (a & 6) == 6; |
579 | 11.0k | } |
580 | | #endif |
581 | | #endif // DISABLE_OPTIMIZED_SHA256 |
582 | | } // namespace |
583 | | |
584 | | |
585 | | std::string SHA256AutoDetect(sha256_implementation::UseImplementation use_implementation) |
586 | 11.0k | { |
587 | 11.0k | std::string ret = "standard"; |
588 | 11.0k | Transform = sha256::Transform; |
589 | 11.0k | TransformD64 = sha256::TransformD64; |
590 | 11.0k | TransformD64_2way = nullptr; |
591 | 11.0k | TransformD64_4way = nullptr; |
592 | 11.0k | TransformD64_8way = nullptr; |
593 | | |
594 | 11.0k | #if !defined(DISABLE_OPTIMIZED_SHA256) |
595 | 11.0k | #if defined(HAVE_GETCPUID) |
596 | 11.0k | bool have_sse4 = false; |
597 | 11.0k | bool have_xsave = false; |
598 | 11.0k | bool have_avx = false; |
599 | 11.0k | [[maybe_unused]] bool have_avx2 = false; |
600 | 11.0k | [[maybe_unused]] bool have_x86_shani = false; |
601 | 11.0k | [[maybe_unused]] bool enabled_avx = false; |
602 | | |
603 | 11.0k | uint32_t eax, ebx, ecx, edx; |
604 | 11.0k | GetCPUID(1, 0, eax, ebx, ecx, edx); |
605 | 11.0k | if (use_implementation & sha256_implementation::USE_SSE4) { Branch (605:9): [True: 11.0k, False: 0]
|
606 | 11.0k | have_sse4 = (ecx >> 19) & 1; |
607 | 11.0k | } |
608 | 11.0k | have_xsave = (ecx >> 27) & 1; |
609 | 11.0k | have_avx = (ecx >> 28) & 1; |
610 | 11.0k | if (have_xsave && have_avx) { Branch (610:9): [True: 11.0k, False: 0]
Branch (610:23): [True: 11.0k, False: 0]
|
611 | 11.0k | enabled_avx = AVXEnabled(); |
612 | 11.0k | } |
613 | 11.0k | if (have_sse4) { Branch (613:9): [True: 11.0k, False: 0]
|
614 | 11.0k | GetCPUID(7, 0, eax, ebx, ecx, edx); |
615 | 11.0k | if (use_implementation & sha256_implementation::USE_AVX2) { Branch (615:13): [True: 11.0k, False: 0]
|
616 | 11.0k | have_avx2 = (ebx >> 5) & 1; |
617 | 11.0k | } |
618 | 11.0k | if (use_implementation & sha256_implementation::USE_SHANI) { Branch (618:13): [True: 11.0k, False: 0]
|
619 | 11.0k | have_x86_shani = (ebx >> 29) & 1; |
620 | 11.0k | } |
621 | 11.0k | } |
622 | | |
623 | 11.0k | #if defined(ENABLE_SSE41) && defined(ENABLE_X86_SHANI) |
624 | 11.0k | if (have_x86_shani) { Branch (624:9): [True: 11.0k, False: 0]
|
625 | 11.0k | Transform = sha256_x86_shani::Transform; |
626 | 11.0k | TransformD64 = TransformD64Wrapper<sha256_x86_shani::Transform>; |
627 | 11.0k | TransformD64_2way = sha256d64_x86_shani::Transform_2way; |
628 | 11.0k | ret = "x86_shani(1way,2way)"; |
629 | 11.0k | have_sse4 = false; // Disable SSE4/AVX2; |
630 | 11.0k | have_avx2 = false; |
631 | 11.0k | } |
632 | 11.0k | #endif |
633 | | |
634 | 11.0k | if (have_sse4) { Branch (634:9): [True: 0, False: 11.0k]
|
635 | 0 | #if defined(__x86_64__) || defined(__amd64__) |
636 | 0 | Transform = sha256_sse4::Transform; |
637 | 0 | TransformD64 = TransformD64Wrapper<sha256_sse4::Transform>; |
638 | 0 | ret = "sse4(1way)"; |
639 | 0 | #endif |
640 | 0 | #if defined(ENABLE_SSE41) |
641 | 0 | TransformD64_4way = sha256d64_sse41::Transform_4way; |
642 | 0 | ret += ",sse41(4way)"; |
643 | 0 | #endif |
644 | 0 | } |
645 | | |
646 | 11.0k | #if defined(ENABLE_AVX2) |
647 | 11.0k | if (have_avx2 && have_avx && enabled_avx) { Branch (647:9): [True: 0, False: 11.0k]
Branch (647:22): [True: 0, False: 0]
Branch (647:34): [True: 0, False: 0]
|
648 | 0 | TransformD64_8way = sha256d64_avx2::Transform_8way; |
649 | 0 | ret += ",avx2(8way)"; |
650 | 0 | } |
651 | 11.0k | #endif |
652 | 11.0k | #endif // defined(HAVE_GETCPUID) |
653 | | |
654 | | #if defined(ENABLE_ARM_SHANI) |
655 | | bool have_arm_shani = false; |
656 | | if (use_implementation & sha256_implementation::USE_SHANI) { |
657 | | #if defined(__linux__) |
658 | | #if defined(__arm__) // 32-bit |
659 | | if (getauxval(AT_HWCAP2) & HWCAP2_SHA2) { |
660 | | have_arm_shani = true; |
661 | | } |
662 | | #endif |
663 | | #if defined(__aarch64__) // 64-bit |
664 | | if (getauxval(AT_HWCAP) & HWCAP_SHA2) { |
665 | | have_arm_shani = true; |
666 | | } |
667 | | #endif |
668 | | #endif |
669 | | |
670 | | #if defined(__APPLE__) |
671 | | int val = 0; |
672 | | size_t len = sizeof(val); |
673 | | if (sysctlbyname("hw.optional.arm.FEAT_SHA256", &val, &len, nullptr, 0) == 0) { |
674 | | have_arm_shani = val != 0; |
675 | | } |
676 | | #endif |
677 | | } |
678 | | |
679 | | if (have_arm_shani) { |
680 | | Transform = sha256_arm_shani::Transform; |
681 | | TransformD64 = TransformD64Wrapper<sha256_arm_shani::Transform>; |
682 | | TransformD64_2way = sha256d64_arm_shani::Transform_2way; |
683 | | ret = "arm_shani(1way,2way)"; |
684 | | } |
685 | | #endif |
686 | 11.0k | #endif // DISABLE_OPTIMIZED_SHA256 |
687 | | |
688 | 11.0k | assert(SelfTest()); Branch (688:5): [True: 11.0k, False: 0]
|
689 | 11.0k | return ret; |
690 | 11.0k | } |
691 | | |
692 | | ////// SHA-256 |
693 | | |
694 | | CSHA256::CSHA256() |
695 | 82.2M | { |
696 | 82.2M | sha256::Initialize(s); |
697 | 82.2M | } |
698 | | |
699 | | CSHA256& CSHA256::Write(const unsigned char* data, size_t len) |
700 | 808M | { |
701 | 808M | const unsigned char* end = data + len; |
702 | 808M | size_t bufsize = bytes % 64; |
703 | 808M | if (bufsize && bufsize + len >= 64) { Branch (703:9): [True: 630M, False: 177M]
Branch (703:20): [True: 218M, False: 412M]
|
704 | | // Fill the buffer, and process it. |
705 | 218M | memcpy(buf + bufsize, data, 64 - bufsize); |
706 | 218M | bytes += 64 - bufsize; |
707 | 218M | data += 64 - bufsize; |
708 | 218M | Transform(s, buf, 1); |
709 | 218M | bufsize = 0; |
710 | 218M | } |
711 | 808M | if (end - data >= 64) { Branch (711:9): [True: 13.2M, False: 795M]
|
712 | 13.2M | size_t blocks = (end - data) / 64; |
713 | 13.2M | Transform(s, data, blocks); |
714 | 13.2M | data += 64 * blocks; |
715 | 13.2M | bytes += 64 * blocks; |
716 | 13.2M | } |
717 | 808M | if (end > data) { Branch (717:9): [True: 630M, False: 177M]
|
718 | | // Fill the buffer with what remains. |
719 | 630M | memcpy(buf + bufsize, data, end - data); |
720 | 630M | bytes += end - data; |
721 | 630M | } |
722 | 808M | return *this; |
723 | 808M | } |
724 | | |
725 | | void CSHA256::Finalize(unsigned char hash[OUTPUT_SIZE]) |
726 | 162M | { |
727 | 162M | static const unsigned char pad[64] = {0x80}; |
728 | 162M | unsigned char sizedesc[8]; |
729 | 162M | WriteBE64(sizedesc, bytes << 3); |
730 | 162M | Write(pad, 1 + ((119 - (bytes % 64)) % 64)); |
731 | 162M | Write(sizedesc, 8); |
732 | 162M | WriteBE32(hash, s[0]); |
733 | 162M | WriteBE32(hash + 4, s[1]); |
734 | 162M | WriteBE32(hash + 8, s[2]); |
735 | 162M | WriteBE32(hash + 12, s[3]); |
736 | 162M | WriteBE32(hash + 16, s[4]); |
737 | 162M | WriteBE32(hash + 20, s[5]); |
738 | 162M | WriteBE32(hash + 24, s[6]); |
739 | 162M | WriteBE32(hash + 28, s[7]); |
740 | 162M | } |
741 | | |
742 | | CSHA256& CSHA256::Reset() |
743 | 80.2M | { |
744 | 80.2M | bytes = 0; |
745 | 80.2M | sha256::Initialize(s); |
746 | 80.2M | return *this; |
747 | 80.2M | } |
748 | | |
749 | | void SHA256D64(unsigned char* out, const unsigned char* in, size_t blocks) |
750 | 80.1k | { |
751 | 80.1k | if (TransformD64_8way) { Branch (751:9): [True: 0, False: 80.1k]
|
752 | 0 | while (blocks >= 8) { Branch (752:16): [True: 0, False: 0]
|
753 | 0 | TransformD64_8way(out, in); |
754 | 0 | out += 256; |
755 | 0 | in += 512; |
756 | 0 | blocks -= 8; |
757 | 0 | } |
758 | 0 | } |
759 | 80.1k | if (TransformD64_4way) { Branch (759:9): [True: 0, False: 80.1k]
|
760 | 0 | while (blocks >= 4) { Branch (760:16): [True: 0, False: 0]
|
761 | 0 | TransformD64_4way(out, in); |
762 | 0 | out += 128; |
763 | 0 | in += 256; |
764 | 0 | blocks -= 4; |
765 | 0 | } |
766 | 0 | } |
767 | 80.1k | if (TransformD64_2way) { Branch (767:9): [True: 80.1k, False: 0]
|
768 | 293k | while (blocks >= 2) { Branch (768:16): [True: 213k, False: 80.1k]
|
769 | 213k | TransformD64_2way(out, in); |
770 | 213k | out += 64; |
771 | 213k | in += 128; |
772 | 213k | blocks -= 2; |
773 | 213k | } |
774 | 80.1k | } |
775 | 125k | while (blocks) { Branch (775:12): [True: 45.7k, False: 80.1k]
|
776 | 45.7k | TransformD64(out, in); |
777 | 45.7k | out += 32; |
778 | 45.7k | in += 64; |
779 | 45.7k | --blocks; |
780 | 45.7k | } |
781 | 80.1k | } |