3#if SYMCRYPT_CPU_X86 | SYMCRYPT_CPU_AMD64
6#pragma clang attribute push (__attribute__((target("ssse3"))), apply_to=function)
8#pragma GCC push_options
9#pragma GCC target("ssse3")
17 0x00010203, 0x04050607, 0x08090a0b, 0x0c0d0e0f,
23 0x03020100, 0x0b0a0908, 0x80808080, 0x80808080,
29 0x80808080, 0x80808080, 0x03020100, 0x0b0a0908,
33#if SYMCRYPT_MS_VC && !defined(__clang__)
34#define RORX_U32 _rorx_u32
35#define RORX_U64 _rorx_u64
55#define MAJ( x, y, z ) ((((z) | (y)) & (x) ) | ((z) & (y)))
56#define CH( x, y, z ) ((((z) ^ (y)) & (x)) ^ (z))
58#define LSIGMA0( x ) (ROR32((x), 7) ^ ROR32((x), 18) ^ ((x)>> 3))
59#define LSIGMA1( x ) (ROR32((x), 17) ^ ROR32((x), 19) ^ ((x)>>10))
61#define CSIGMA0(x) (RORX_U32(x, 2) ^ RORX_U32(x, 13) ^ RORX_U32(x, 22))
62#define CSIGMA1(x) (RORX_U32(x, 6) ^ RORX_U32(x, 11) ^ RORX_U32(x, 25))
65#define LSIGMA0XMM( x ) \
66 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
67 _mm_slli_epi32(x,25) , _mm_srli_epi32(x, 7) ),\
68 _mm_slli_epi32(x,14) ), _mm_srli_epi32(x, 18) ),\
69 _mm_srli_epi32(x, 3) )
70#define LSIGMA1XMM( x ) \
71 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
72 _mm_slli_epi32(x,15) , _mm_srli_epi32(x, 17) ),\
73 _mm_slli_epi32(x,13) ), _mm_srli_epi32(x, 19) ),\
74 _mm_srli_epi32(x,10) )
83#define SHA256_MSG_LOAD_4BLOCKS(bl) { \
84 for(SIZE_T i = 0; i < bl; i++) \
86 Wx.xmm[i + 0] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 0]), kBYTE_REVERSE_32); \
87 Wx.xmm[i + 4] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 16]), kBYTE_REVERSE_32); \
88 Wx.xmm[i + 8] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 32]), kBYTE_REVERSE_32); \
89 Wx.xmm[i + 12] = _mm_shuffle_epi8(_mm_loadu_si128((__m128i*) &pbData[i * SYMCRYPT_SHA256_INPUT_BLOCK_SIZE + 48]), kBYTE_REVERSE_32); \
100#define SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(ind) { \
101 __m128i t1, t2, t3, t4; \
102 t1 = _mm_unpacklo_epi32(Wx.xmm[4 * (ind) + 0], Wx.xmm[4 * (ind) + 1]); \
103 t2 = _mm_unpacklo_epi32(Wx.xmm[4 * (ind) + 2], Wx.xmm[4 * (ind) + 3]); \
104 t3 = _mm_unpackhi_epi32(Wx.xmm[4 * (ind) + 0], Wx.xmm[4 * (ind) + 1]); \
105 t4 = _mm_unpackhi_epi32(Wx.xmm[4 * (ind) + 2], Wx.xmm[4 * (ind) + 3]); \
106 Wx.xmm[4 * (ind) + 0] = _mm_unpacklo_epi64(t1, t2); \
107 Wx.xmm[4 * (ind) + 1] = _mm_unpackhi_epi64(t1, t2); \
108 Wx.xmm[4 * (ind) + 2] = _mm_unpacklo_epi64(t3, t4); \
109 Wx.xmm[4 * (ind) + 3] = _mm_unpackhi_epi64(t3, t4); \
112#define SHA256_MSG_TRANSPOSE_4BLOCKS() { \
113 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(0); \
114 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(1); \
115 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(2); \
116 SHA256_MSG_TRANSPOSE_QUARTER_4BLOCKS(3); \
121#define SHA256_MSG_EXPAND_4BLOCKS_1ROUND(r) { \
122 Wx.xmm[r] = _mm_add_epi32(_mm_add_epi32(_mm_add_epi32(Wx.xmm[r - 16], Wx.xmm[r - 7]), \
123 LSIGMA0XMM(Wx.xmm[r - 15])), LSIGMA1XMM(Wx.xmm[r - 2])); \
124 Wx.xmm[r - 16] = _mm_add_epi32(Wx.xmm[r - 16], _mm_set1_epi32(SymCryptSha256K[r - 16])); \
128#define SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS(r) { \
129 SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 0); SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 1); \
130 SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 2); SHA256_MSG_EXPAND_4BLOCKS_1ROUND((r) + 3); \
133#define SHA256_MSG_EXPAND_4BLOCKS_16ROUNDS(r) { \
134 SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 0); SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 4); \
135 SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 8); SHA256_MSG_EXPAND_4BLOCKS_4ROUNDS((r) + 12); \
145#define CROUND_4BLOCKS(r16, rb, b) { \
146 Wt = Wx.ul4[(rb)+(r16)][b]; \
147 ah[ r16 &7] += CSIGMA1(ah[(r16+3)&7]) + CH(ah[(r16+3)&7], ah[(r16+2)&7], ah[(r16+1)&7]) + Wt;\
148 ah[(r16+4)&7] += ah[r16 &7];\
149 ah[ r16 &7] += CSIGMA0(ah[(r16+7)&7]) + MAJ(ah[(r16+7)&7], ah[(r16+6)&7], ah[(r16+5)&7]);\
158#define CROUND( r16, r ) {;\
159 ah[ r16 &7] += CSIGMA1(ah[(r16+3)&7]) + CH(ah[(r16+3)&7], ah[(r16+2)&7], ah[(r16+1)&7]) + SymCryptSha256K[r] + Wt;\
160 ah[(r16+4)&7] += ah[r16 &7];\
161 ah[ r16 &7] += CSIGMA0(ah[(r16+7)&7]) + MAJ(ah[(r16+7)&7], ah[(r16+6)&7], ah[(r16+5)&7]);\
168#define IROUND( r ) {\
169 Wt = SYMCRYPT_LOAD_MSBFIRST32( &pbData[ 4*r ] );\
178#define FROUND(r16, rb) { \
179 Wt = LSIGMA1( Wx.ul[(r16-2) & 15] ) + Wx.ul[(r16-7) & 15] + \
180 LSIGMA0( Wx.ul[(r16-15) & 15]) + Wx.ul[r16 & 15]; \
182 CROUND( r16, r16+rb ); \
201 const __m128i kBYTE_REVERSE_32 =
_mm_load_si128((
const __m128i*)BYTE_REVERSE_32);
208 SHA256_MSG_LOAD_4BLOCKS(numBlocks);
209 SHA256_MSG_TRANSPOSE_4BLOCKS();
211 for (
int j = 16;
j < 64;
j += 16)
213 SHA256_MSG_EXPAND_4BLOCKS_16ROUNDS(
j);
217 for (
int i = 48;
i < 64;
i++)
222 for (
SIZE_T bl = 0; bl < numBlocks; bl++)
224 ah[7] = pChain->H[0];
225 ah[6] = pChain->H[1];
226 ah[5] = pChain->H[2];
227 ah[4] = pChain->H[3];
228 ah[3] = pChain->H[4];
229 ah[2] = pChain->H[5];
230 ah[1] = pChain->H[6];
231 ah[0] = pChain->H[7];
233 for (
int iterCount = 0; iterCount < (64/8); iterCount++)
235 const int roundBase = iterCount*8;
236 CROUND_4BLOCKS( 0, roundBase, bl);
237 CROUND_4BLOCKS( 1, roundBase, bl);
238 CROUND_4BLOCKS( 2, roundBase, bl);
239 CROUND_4BLOCKS( 3, roundBase, bl);
240 CROUND_4BLOCKS( 4, roundBase, bl);
241 CROUND_4BLOCKS( 5, roundBase, bl);
242 CROUND_4BLOCKS( 6, roundBase, bl);
243 CROUND_4BLOCKS( 7, roundBase, bl);
254 pChain->H[0] = ah[7] + pChain->H[0];
255 pChain->H[1] = ah[6] + pChain->H[1];
256 pChain->H[2] = ah[5] + pChain->H[2];
257 pChain->H[3] = ah[4] + pChain->H[3];
258 pChain->H[4] = ah[3] + pChain->H[4];
259 pChain->H[5] = ah[2] + pChain->H[5];
260 pChain->H[6] = ah[1] + pChain->H[6];
261 pChain->H[7] = ah[0] + pChain->H[7];
271 ah[7] = pChain->H[0];
272 ah[6] = pChain->H[1];
273 ah[5] = pChain->H[2];
274 ah[4] = pChain->H[3];
275 ah[3] = pChain->H[4];
276 ah[2] = pChain->H[5];
277 ah[1] = pChain->H[6];
278 ah[0] = pChain->H[7];
305 for (
int iterCount = 1; iterCount < (64/16); iterCount++)
307 const int roundBase = iterCount*16;
326 pChain->H[0] = ah[7] + pChain->H[0];
327 pChain->H[1] = ah[6] + pChain->H[1];
328 pChain->H[2] = ah[5] + pChain->H[2];
329 pChain->H[3] = ah[4] + pChain->H[3];
330 pChain->H[4] = ah[3] + pChain->H[4];
331 pChain->H[5] = ah[2] + pChain->H[5];
332 pChain->H[6] = ah[1] + pChain->H[6];
333 pChain->H[7] = ah[0] + pChain->H[7];
349#pragma clang attribute pop
351#pragma GCC pop_options
__m128i _mm_load_si128(__m128i const *p)
__m128i _mm_add_epi32(__m128i a, __m128i b)
__m128i _mm_set1_epi32(int i)
GLsizei GLenum const GLvoid GLsizei GLenum GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLint GLint GLint GLshort GLshort GLshort GLubyte GLubyte GLubyte GLuint GLuint GLuint GLushort GLushort GLushort GLbyte GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLfloat GLint GLint GLint GLint GLshort GLshort GLshort GLshort GLubyte GLubyte GLubyte GLubyte GLuint GLuint GLuint GLuint GLushort GLushort GLushort GLushort GLboolean const GLdouble const GLfloat const GLint const GLshort const GLbyte const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLdouble const GLfloat const GLfloat const GLint const GLint const GLshort const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort GLenum GLenum GLenum GLfloat GLenum GLint GLenum GLenum GLenum GLfloat GLenum GLenum GLint GLenum GLfloat GLenum GLint GLint GLushort GLenum GLenum GLfloat GLenum GLenum GLint GLfloat const GLubyte GLenum GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLint GLint GLsizei GLsizei GLint GLenum GLenum const GLvoid GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLenum const GLdouble GLenum GLenum const GLfloat GLenum GLenum const GLint GLsizei GLuint GLfloat GLuint GLbitfield GLfloat GLint GLuint GLboolean GLenum GLfloat GLenum GLbitfield GLenum GLfloat GLfloat GLint GLint const GLfloat GLenum GLfloat GLfloat GLint GLint GLfloat GLfloat GLint GLint const GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat const GLdouble const GLfloat const GLdouble const GLfloat GLint i
GLsizei GLenum const GLvoid GLsizei GLenum GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLint GLint GLint GLshort GLshort GLshort GLubyte GLubyte GLubyte GLuint GLuint GLuint GLushort GLushort GLushort GLbyte GLbyte GLbyte GLbyte GLdouble GLdouble GLdouble GLdouble GLfloat GLfloat GLfloat GLfloat GLint GLint GLint GLint GLshort GLshort GLshort GLshort GLubyte GLubyte GLubyte GLubyte GLuint GLuint GLuint GLuint GLushort GLushort GLushort GLushort GLboolean const GLdouble const GLfloat const GLint const GLshort const GLbyte const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLdouble const GLfloat const GLfloat const GLint const GLint const GLshort const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort const GLdouble const GLfloat const GLint const GLshort GLenum GLenum GLenum GLfloat GLenum GLint GLenum GLenum GLenum GLfloat GLenum GLenum GLint GLenum GLfloat GLenum GLint GLint GLushort GLenum GLenum GLfloat GLenum GLenum GLint GLfloat const GLubyte GLenum GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLint GLint GLsizei GLsizei GLint GLenum GLenum const GLvoid GLenum GLenum const GLfloat GLenum GLenum const GLint GLenum GLenum const GLdouble GLenum GLenum const GLfloat GLenum GLenum const GLint GLsizei GLuint GLfloat GLuint GLbitfield GLfloat GLint GLuint GLboolean GLenum GLfloat GLenum GLbitfield GLenum GLfloat GLfloat GLint GLint const GLfloat GLenum GLfloat GLfloat GLint GLint GLfloat GLfloat GLint GLint const GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat GLint GLfloat GLfloat const GLdouble const GLfloat const GLdouble const GLfloat GLint GLint GLint j
VOID SYMCRYPT_CALL SymCryptSha256AppendBlocks_xmm_4blocks(_Inout_ SYMCRYPT_SHA256_CHAINING_STATE *pChain, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_ SIZE_T *pcbRemaining)
#define SYMCRYPT_SHA256_INPUT_BLOCK_SIZE
FORCEINLINE VOID SYMCRYPT_CALL SymCryptWipeKnownSize(_Out_writes_bytes_(cbData) PVOID pbData, SIZE_T cbData)
VOID SYMCRYPT_CALL SymCryptWipe(_Out_writes_bytes_(cbData) PVOID pbData, SIZE_T cbData)
SYMCRYPT_SHA256_CHAINING_STATE
#define SYMCRYPT_ALIGN_AT(alignment)
PCBYTE PBYTE SIZE_T cbData