ReactOS 0.4.17-dev-1005-g171e1de
sha256-xmm.c
Go to the documentation of this file.
1#include "precomp.h"
2
3#if SYMCRYPT_CPU_X86 | SYMCRYPT_CPU_AMD64
4
5#ifdef __clang__
6#pragma clang attribute push (__attribute__((target("ssse3"))), apply_to=function)
7#else
8#pragma GCC push_options
9#pragma GCC target("ssse3")
10#endif
11
12extern SYMCRYPT_ALIGN_AT(256) const UINT32 SymCryptSha256K[64];
13
14
15// Endianness transformation for 4 32-bit values in an XMM register
16const SYMCRYPT_ALIGN_AT(16) UINT32 BYTE_REVERSE_32[4] = {
17 0x00010203, 0x04050607, 0x08090a0b, 0x0c0d0e0f,
18};
19
20// Shuffle 32-bit words in an XMM register: W3 W2 W1 W0 -> 0 0 W2 W0
21// Used by the SSSE3 assembly implementation
22const SYMCRYPT_ALIGN_AT(16) UINT32 XMM_PACKLOW[4] = {
23 0x03020100, 0x0b0a0908, 0x80808080, 0x80808080,
24};
25
26// Shuffle 32-bit words in an XMM register: W3 W2 W1 W0 -> W2 W0 0 0
27// Used by the SSSE3 assembly implementation
28const SYMCRYPT_ALIGN_AT(16) UINT32 XMM_PACKHIGH[4] = {
29 0x80808080, 0x80808080, 0x03020100, 0x0b0a0908,
30};
31
32
33#if SYMCRYPT_MS_VC && !defined(__clang__)
34#define RORX_U32 _rorx_u32
35#define RORX_U64 _rorx_u64
36#else
37// TODO: implement _rorx functions for clang
38#define RORX_U32 ROR32
39#define RORX_U64 ROR64
40#endif // SYMCRYPT_MS_VC
41
42
43//
44// For documentation on these function see FIPS 180-2
45//
46// MAJ and CH are the functions Maj and Ch from the standard.
47// CSIGMA0 and CSIGMA1 are the capital sigma functions.
48// LSIGMA0 and LSIGMA1 are the lowercase sigma functions.
49//
50// The canonical definitions of the MAJ and CH functions are:
51//#define MAJ( x, y, z ) (((x) & (y)) ^ ((x) & (z)) ^ ((y) & (z)))
52//#define CH( x, y, z ) (((x) & (y)) ^ ((~(x)) & (z)))
53// We use optimized versions defined below
54//
55#define MAJ( x, y, z ) ((((z) | (y)) & (x) ) | ((z) & (y)))
56#define CH( x, y, z ) ((((z) ^ (y)) & (x)) ^ (z))
57
58#define LSIGMA0( x ) (ROR32((x), 7) ^ ROR32((x), 18) ^ ((x)>> 3))
59#define LSIGMA1( x ) (ROR32((x), 17) ^ ROR32((x), 19) ^ ((x)>>10))
60
61#define CSIGMA0(x) (RORX_U32(x, 2) ^ RORX_U32(x, 13) ^ RORX_U32(x, 22))
62#define CSIGMA1(x) (RORX_U32(x, 6) ^ RORX_U32(x, 11) ^ RORX_U32(x, 25))
63
64
65#define LSIGMA0XMM( x ) \
66 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
67 _mm_slli_epi32(x,25) , _mm_srli_epi32(x, 7) ),\
68 _mm_slli_epi32(x,14) ), _mm_srli_epi32(x, 18) ),\
69 _mm_srli_epi32(x, 3) )
70#define LSIGMA1XMM( x ) \
71 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
72 _mm_slli_epi32(x,15) , _mm_srli_epi32(x, 17) ),\
73 _mm_slli_epi32(x,13) ), _mm_srli_epi32(x, 19) ),\
74 _mm_srli_epi32(x,10) )
75
76
77
78// Initial loading of message words and endianness transformation.
79// bl : The number of blocks to load, 1 <= bl <= 4.
80//
81// When bl < 4, the high order lanes of the XMM registers corresponding to the missing blocks are unused.
82//
83#define SHA256_MSG_LOAD_4BLOCKS(bl) { \
84 for(SIZE_T i = 0; i < bl; i++) \
85 { \
86 Wx.xmm[i + 0] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 0]), kBYTE_REVERSE_32); \
87 Wx.xmm[i + 4] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 16]), kBYTE_REVERSE_32); \
88 Wx.xmm[i + 8] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 32]), kBYTE_REVERSE_32); \
89 Wx.xmm[i + 12] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 48]), kBYTE_REVERSE_32); \
90 } \
91}
92
93// Shuffles the initially loaded message words from multiple blocks
94// so that each XMM register contains message words with the same index
95// within a block (e.g. Wx.xmm[0] contains the first words of each block).
96//
97// We have to use this macro four times to transform the message blocks of 64-bytes.
98// ind=0 processes the first quarter (16-bytes), ind=1 does the second quarter and so on.
99//
100#define SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(ind) { \
101 __m128i t1, t2, t3, t4; \
102 t1 = _mm_unpacklo_epi32(Wx.xmm[4 * (ind) + 0], Wx.xmm[4 * (ind) + 1]); \
103 t2 = _mm_unpacklo_epi32(Wx.xmm[4 * (ind) + 2], Wx.xmm[4 * (ind) + 3]); \
104 t3 = _mm_unpackhi_epi32(Wx.xmm[4 * (ind) + 0], Wx.xmm[4 * (ind) + 1]); \
105 t4 = _mm_unpackhi_epi32(Wx.xmm[4 * (ind) + 2], Wx.xmm[4 * (ind) + 3]); \
106 Wx.xmm[4 * (ind) + 0] = _mm_unpacklo_epi64(t1, t2); \
107 Wx.xmm[4 * (ind) + 1] = _mm_unpackhi_epi64(t1, t2); \
108 Wx.xmm[4 * (ind) + 2] = _mm_unpacklo_epi64(t3, t4); \
109 Wx.xmm[4 * (ind) + 3] = _mm_unpackhi_epi64(t3, t4); \
110}
111
112#define SHA256_MSG_TRANSPOSE_4BLOCKS() { \
113 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(0); \
114 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(1); \
115 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(2); \
116 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(3); \
117}
118
119// One round message schedule, updates the rth message word. ( 16 <= r < 64 )
120// Also adds the constants for round (r-16).
121#define SHA256_MSG_EXPAND_4BLOCKS_1ROUND(r) { \
122 Wx.xmm[r] = _mm_add_epi32(_mm_add_epi32(_mm_add_epi32(Wx.xmm[r - 16], Wx.xmm[r - 7]), \
123 LSIGMA0XMM(Wx.xmm[r - 15])), LSIGMA1XMM(Wx.xmm[r - 2])); \
124 Wx.xmm[r - 16] = _mm_add_epi32(Wx.xmm[r - 16], _mm_set1_epi32(SymCryptSha256K[r - 16])); \
125}
126
127// Four rounds of message schedule. Generates message words for rounds r, r+1, r+2, r+3.
128#define SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS(r) { \
129 SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 0); SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 1); \
130 SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 2); SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 3); \
131}
132// Sixteen rounds of message schedule. Generates message words for rounds r, ..., r+15.
133#define SHA256_MSG_EXPAND_4BLOCKS_16ROUNDS(r) { \
134 SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 0); SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 4); \
135 SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 8); SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 12); \
136}
137
138// Core round function using message words from Wx array.
139// Wx contains -interleaved- expanded message words from b blocks.
140// i.e. Message words for round r for each block, followed by the message words for the (r+1)^th block.
141//
142// r16 : round number mod 16
143// rb : base round number so that (rb+r16) gives the actual round number
144// b : message block index, b = 0..3
145#define CROUND_4BLOCKS(r16, rb, b) { \
146 Wt = Wx.ul4[(rb)+(r16)][b]; \
147 ah[ r16 &7] += CSIGMA1(ah[(r16+3)&7]) + CH(ah[(r16+3)&7], ah[(r16+2)&7], ah[(r16+1)&7]) + Wt;\
148 ah[(r16+4)&7] += ah[r16 &7];\
149 ah[ r16 &7] += CSIGMA0(ah[(r16+7)&7]) + MAJ(ah[(r16+7)&7], ah[(r16+6)&7], ah[(r16+5)&7]);\
150}
151
152//
153// Core round function
154//
155// r16 : round number mod 16
156// r : round number, r = 0..63
157//
158#define CROUND( r16, r ) {;\
159 ah[ r16 &7] += CSIGMA1(ah[(r16+3)&7]) + CH(ah[(r16+3)&7], ah[(r16+2)&7], ah[(r16+1)&7]) + SymCryptSha256K[r] + Wt;\
160 ah[(r16+4)&7] += ah[r16 &7];\
161 ah[ r16 &7] += CSIGMA0(ah[(r16+7)&7]) + MAJ(ah[(r16+7)&7], ah[(r16+6)&7], ah[(r16+5)&7]);\
162}
163
164//
165// Initial round that reads the message.
166// r is the round number 0..15
167//
168#define IROUND( r ) {\
169 Wt = SYMCRYPT_LOAD_MSBFIRST32( &pbData[ 4*r ] );\
170 Wx.ul[r] = Wt; \
171 CROUND(r,r);\
172}
173
174//
175// Subsequent rounds.
176// r16 is the round number mod 16. rb is the round number minus r16.
177//
178#define FROUND(r16, rb) { \
179 Wt = LSIGMA1( Wx.ul[(r16-2) & 15] ) + Wx.ul[(r16-7) & 15] + \
180 LSIGMA0( Wx.ul[(r16-15) & 15]) + Wx.ul[r16 & 15]; \
181 Wx.ul[r16] = Wt; \
182 CROUND( r16, r16+rb ); \
183}
184
185
186
187VOID
193 _Out_ SIZE_T* pcbRemaining)
194{
195
196 SYMCRYPT_ALIGN union { UINT32 ul[16]; UINT32 ul4[64][4]; __m128i xmm[64]; } Wx;
198 UINT32 Wt;
199 SIZE_T uWipeSize = (cbData >= (3 * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE)) ? (64 * 4 * sizeof(UINT32)) : (16 * sizeof(UINT32));
200
201 const __m128i kBYTE_REVERSE_32 = _mm_load_si128((const __m128i*)BYTE_REVERSE_32);
202
204 {
205 // If we have 4 or more blocks then process 4, else process whatever is left.
207
208 SHA256_MSG_LOAD_4BLOCKS(numBlocks);
209 SHA256_MSG_TRANSPOSE_4BLOCKS();
210
211 for (int j = 16; j < 64; j += 16)
212 {
213 SHA256_MSG_EXPAND_4BLOCKS_16ROUNDS(j);
214 }
215
216 // Constants up to r=48 were added during message expansion. Add the remaining ones here.
217 for (int i = 48; i < 64; i++)
218 {
219 Wx.xmm[i] = _mm_add_epi32(Wx.xmm[i], _mm_set1_epi32(SymCryptSha256K[i]));
220 }
221
222 for (SIZE_T bl = 0; bl < numBlocks; bl++)
223 {
224 ah[7] = pChain->H[0];
225 ah[6] = pChain->H[1];
226 ah[5] = pChain->H[2];
227 ah[4] = pChain->H[3];
228 ah[3] = pChain->H[4];
229 ah[2] = pChain->H[5];
230 ah[1] = pChain->H[6];
231 ah[0] = pChain->H[7];
232
233 for (int iterCount = 0; iterCount < (64/8); iterCount++)
234 {
235 const int roundBase = iterCount*8;
236 CROUND_4BLOCKS( 0, roundBase, bl);
237 CROUND_4BLOCKS( 1, roundBase, bl);
238 CROUND_4BLOCKS( 2, roundBase, bl);
239 CROUND_4BLOCKS( 3, roundBase, bl);
240 CROUND_4BLOCKS( 4, roundBase, bl);
241 CROUND_4BLOCKS( 5, roundBase, bl);
242 CROUND_4BLOCKS( 6, roundBase, bl);
243 CROUND_4BLOCKS( 7, roundBase, bl);
244 //CROUND_4BLOCKS( 8, roundBase, bl);
245 //CROUND_4BLOCKS( 9, roundBase, bl);
246 //CROUND_4BLOCKS(10, roundBase, bl);
247 //CROUND_4BLOCKS(11, roundBase, bl);
248 //CROUND_4BLOCKS(12, roundBase, bl);
249 //CROUND_4BLOCKS(13, roundBase, bl);
250 //CROUND_4BLOCKS(14, roundBase, bl);
251 //CROUND_4BLOCKS(15, roundBase, bl);
252 }
253
254 pChain->H[0] = ah[7] + pChain->H[0];
255 pChain->H[1] = ah[6] + pChain->H[1];
256 pChain->H[2] = ah[5] + pChain->H[2];
257 pChain->H[3] = ah[4] + pChain->H[3];
258 pChain->H[4] = ah[3] + pChain->H[4];
259 pChain->H[5] = ah[2] + pChain->H[5];
260 pChain->H[6] = ah[1] + pChain->H[6];
261 pChain->H[7] = ah[0] + pChain->H[7];
262 }
263
266 }
267
268
270 {
271 ah[7] = pChain->H[0];
272 ah[6] = pChain->H[1];
273 ah[5] = pChain->H[2];
274 ah[4] = pChain->H[3];
275 ah[3] = pChain->H[4];
276 ah[2] = pChain->H[5];
277 ah[1] = pChain->H[6];
278 ah[0] = pChain->H[7];
279
280 //
281 // initial rounds 1 to 16
282 //
283
284 IROUND(0);
285 IROUND(1);
286 IROUND(2);
287 IROUND(3);
288 IROUND(4);
289 IROUND(5);
290 IROUND(6);
291 IROUND(7);
292 IROUND(8);
293 IROUND(9);
294 IROUND(10);
295 IROUND(11);
296 IROUND(12);
297 IROUND(13);
298 IROUND(14);
299 IROUND(15);
300
301
302 //
303 // rounds 16 to 64.
304 //
305 for (int iterCount = 1; iterCount < (64/16); iterCount++)
306 {
307 const int roundBase = iterCount*16;
308 FROUND(0, roundBase);
309 FROUND(1, roundBase);
310 FROUND(2, roundBase);
311 FROUND(3, roundBase);
312 FROUND(4, roundBase);
313 FROUND(5, roundBase);
314 FROUND(6, roundBase);
315 FROUND(7, roundBase);
316 FROUND(8, roundBase);
317 FROUND(9, roundBase);
318 FROUND(10, roundBase);
319 FROUND(11, roundBase);
320 FROUND(12, roundBase);
321 FROUND(13, roundBase);
322 FROUND(14, roundBase);
323 FROUND(15, roundBase);
324 }
325
326 pChain->H[0] = ah[7] + pChain->H[0];
327 pChain->H[1] = ah[6] + pChain->H[1];
328 pChain->H[2] = ah[5] + pChain->H[2];
329 pChain->H[3] = ah[4] + pChain->H[3];
330 pChain->H[4] = ah[3] + pChain->H[4];
331 pChain->H[5] = ah[2] + pChain->H[5];
332 pChain->H[6] = ah[1] + pChain->H[6];
333 pChain->H[7] = ah[0] + pChain->H[7];
334
337 }
338
339 *pcbRemaining = cbData;
340
341 //
342 // Wipe the variables;
343 //
344 SymCryptWipe(&Wx, uWipeSize);
345 SymCryptWipeKnownSize(ah, sizeof(ah));
346}
347
348#ifdef __clang__
349#pragma clang attribute pop
350#else
351#pragma GCC pop_options
352#endif
353
354#endif // SYMCRYPT_CPU_X86 | SYMCRYPT_CPU_AMD64
__m128i _mm_load_si128(__m128i const *p)
Definition: emmintrin.h:1556
__m128i _mm_add_epi32(__m128i a, __m128i b)
Definition: emmintrin.h:1137
__m128i _mm_set1_epi32(int i)
Definition: emmintrin.h:1631
GLsizei GLenum const GLvoid GLsizei GLenum GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLint GLint GLint GLshort GLshort GLshort GLubyte GLubyte GLubyte GLuint GLuint GLuint GLushort GLushort GLushort GLbyte GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLfloat GLint GLint GLint GLint GLshort GLshort GLshort GLshort GLubyte GLubyte GLubyte GLubyte GLuint GLuint GLuint GLuint GLushort GLushort GLushort GLushort GLboolean const GLdouble const GLfloat const GLint const GLshort const GLbyte const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLdouble const GLfloat const GLfloat const GLint const GLint const GLshort const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort GLenum GLenum GLenum GLfloat GLenum GLint GLenum GLenum GLenum GLfloat GLenum GLenum GLint GLenum GLfloat GLenum GLint GLint GLushort GLenum GLenum GLfloat GLenum GLenum GLint GLfloat const GLubyte GLenum GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLint GLint GLsizei GLsizei GLint GLenum GLenum const GLvoid GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLenum const GLdouble GLenum GLenum const GLfloat GLenum GLenum const GLint GLsizei GLuint GLfloat GLuint GLbitfield GLfloat GLint GLuint GLboolean GLenum GLfloat GLenum GLbitfield GLenum GLfloat GLfloat GLint GLint const GLfloat GLenum GLfloat GLfloat GLint GLint GLfloat GLfloat GLint GLint const GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat const GLdouble const GLfloat const GLdouble const GLfloat GLint i
Definition: glfuncs.h:248
GLsizei GLenum const GLvoid GLsizei GLenum GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLint GLint GLint GLshort GLshort GLshort GLubyte GLubyte GLubyte GLuint GLuint GLuint GLushort GLushort GLushort GLbyte GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLfloat GLint GLint GLint GLint GLshort GLshort GLshort GLshort GLubyte GLubyte GLubyte GLubyte GLuint GLuint GLuint GLuint GLushort GLushort GLushort GLushort GLboolean const GLdouble const GLfloat const GLint const GLshort const GLbyte const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLdouble const GLfloat const GLfloat const GLint const GLint const GLshort const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort GLenum GLenum GLenum GLfloat GLenum GLint GLenum GLenum GLenum GLfloat GLenum GLenum GLint GLenum GLfloat GLenum GLint GLint GLushort GLenum GLenum GLfloat GLenum GLenum GLint GLfloat const GLubyte GLenum GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLint GLint GLsizei GLsizei GLint GLenum GLenum const GLvoid GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLenum const GLdouble GLenum GLenum const GLfloat GLenum GLenum const GLint GLsizei GLuint GLfloat GLuint GLbitfield GLfloat GLint GLuint GLboolean GLenum GLfloat GLenum GLbitfield GLenum GLfloat GLfloat GLint GLint const GLfloat GLenum GLfloat GLfloat GLint GLint GLfloat GLfloat GLint GLint const GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat const GLdouble const GLfloat const GLdouble const GLfloat GLint GLint GLint j
Definition: glfuncs.h:250
#define _In_reads_(s)
Definition: no_sal2.h:168
#define _Inout_
Definition: no_sal2.h:162
#define _Out_
Definition: no_sal2.h:160
VOID SYMCRYPT_CALL SymCryptSha256AppendBlocks_xmm_4blocks(_Inout_ SYMCRYPT_SHA256_CHAINING_STATE *pChain, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_ SIZE_T *pcbRemaining)
#define FROUND(r, Func)
Definition: md4.c:227
#define IROUND(r, Func)
Definition: md4.c:218
#define SYMCRYPT_SHA256_INPUT_BLOCK_SIZE
Definition: symcrypt.h:1223
FORCEINLINE VOID SYMCRYPT_CALL SymCryptWipeKnownSize(_Out_writes_bytes_(cbData) PVOID pbData, SIZE_T cbData)
VOID SYMCRYPT_CALL SymCryptWipe(_Out_writes_bytes_(cbData) PVOID pbData, SIZE_T cbData)
Definition: libmain.c:137
#define SYMCRYPT_ALIGN
#define SYMCRYPT_CALL
SYMCRYPT_SHA256_CHAINING_STATE
#define SYMCRYPT_ALIGN_AT(alignment)
PCBYTE PBYTE SIZE_T cbData
const BYTE * PCBYTE
PCBYTE pbData
ULONG_PTR SIZE_T
Definition: typedefs.h:80
uint32_t UINT32
Definition: typedefs.h:59
#define const
Definition: zconf.h:233