ReactOS 0.4.17-dev-1005-g171e1de
sha256.c
Go to the documentation of this file.
1//
2// Sha256.c
3//
4// Copyright (c) Microsoft Corporation. Licensed under the MIT license.
5//
6
7//
8// This module contains the routines to implement SHA2-256 from FIPS 180-2
9//
10// This revised implementation is based on the older one in RSA32LIB by Scott Field from 2001
11//
12
13#include "precomp.h"
14
15//
16// See the symcrypt.h file for documentation on what the various functions do.
17//
18
25 sizeof( SYMCRYPT_SHA224_STATE ),
30};
31
38 sizeof( SYMCRYPT_SHA256_STATE ),
43};
44
47
48//
49// SHA-256 uses 64 magic constants of 32 bits each. These are
50// referred to as K^{256}_i for i=0...63 by FIPS 180-2.
51// This array is also used by the parallel SHA256 implementation
52// For performance we align to 256 bytes, which gives optimal cache alignment.
53//
54SYMCRYPT_ALIGN_AT( 256 ) const UINT32 SymCryptSha256K[64] = {
55 0x428a2f98UL, 0x71374491UL, 0xb5c0fbcfUL, 0xe9b5dba5UL,
56 0x3956c25bUL, 0x59f111f1UL, 0x923f82a4UL, 0xab1c5ed5UL,
57 0xd807aa98UL, 0x12835b01UL, 0x243185beUL, 0x550c7dc3UL,
58 0x72be5d74UL, 0x80deb1feUL, 0x9bdc06a7UL, 0xc19bf174UL,
59 0xe49b69c1UL, 0xefbe4786UL, 0x0fc19dc6UL, 0x240ca1ccUL,
60 0x2de92c6fUL, 0x4a7484aaUL, 0x5cb0a9dcUL, 0x76f988daUL,
61 0x983e5152UL, 0xa831c66dUL, 0xb00327c8UL, 0xbf597fc7UL,
62 0xc6e00bf3UL, 0xd5a79147UL, 0x06ca6351UL, 0x14292967UL,
63 0x27b70a85UL, 0x2e1b2138UL, 0x4d2c6dfcUL, 0x53380d13UL,
64 0x650a7354UL, 0x766a0abbUL, 0x81c2c92eUL, 0x92722c85UL,
65 0xa2bfe8a1UL, 0xa81a664bUL, 0xc24b8b70UL, 0xc76c51a3UL,
66 0xd192e819UL, 0xd6990624UL, 0xf40e3585UL, 0x106aa070UL,
67 0x19a4c116UL, 0x1e376c08UL, 0x2748774cUL, 0x34b0bcb5UL,
68 0x391c0cb3UL, 0x4ed8aa4aUL, 0x5b9cca4fUL, 0x682e6ff3UL,
69 0x748f82eeUL, 0x78a5636fUL, 0x84c87814UL, 0x8cc70208UL,
70 0x90befffaUL, 0xa4506cebUL, 0xbef9a3f7UL, 0xc67178f2UL
71};
72
73
74//
75// Initial state
76//
77static const UINT32 sha224InitialState[8] = {
78 0xc1059ed8UL,
79 0x367cd507UL,
80 0x3070dd17UL,
81 0xf70e5939UL,
82 0xffc00b31UL,
83 0x68581511UL,
84 0x64f98fa7UL,
85 0xbefa4fa4UL,
86};
87
88static const UINT32 sha256InitialState[8] = {
89 0x6a09e667UL,
90 0xbb67ae85UL,
91 0x3c6ef372UL,
92 0xa54ff53aUL,
93 0x510e527fUL,
94 0x9b05688cUL,
95 0x1f83d9abUL,
96 0x5be0cd19UL,
97};
98
99//
100// SymCryptSha224
101//
102#define ALG SHA224
103#define Alg Sha224
104#include "hash_pattern.c"
105#undef ALG
106#undef Alg
107
108//
109// SymCryptSha256
110//
111#define ALG SHA256
112#define Alg Sha256
113#include "hash_pattern.c"
114#undef ALG
115#undef Alg
116
117
118
119//
120// SymCryptSha256Init
121//
122SYMCRYPT_NOINLINE
123VOID
126{
128
129 pState->dataLengthL = 0;
130 //pState->dataLengthH = 0; // not used
131 pState->bytesInBuffer = 0;
132
133 memcpy( &pState->chain.H[0], &sha256InitialState[0], sizeof( sha256InitialState ) );
134
135 //
136 // There is no need to initialize the buffer part of the state as that will be
137 // filled before it is used.
138 //
139}
140
141
142//
143// SymCryptSha224Init
144//
145SYMCRYPT_NOINLINE
146VOID
149{
151
152 pState->dataLengthL = 0;
153 //pState->dataLengthH = 0; // not used
154 pState->bytesInBuffer = 0;
155
156 memcpy( &pState->chain.H[0], &sha224InitialState[0], sizeof( sha224InitialState ) );
157
158 //
159 // There is no need to initialize the buffer part of the state as that will be
160 // filled before it is used.
161 //
162}
163
164
165//
166// SymCryptSha256Append
167//
168SYMCRYPT_NOINLINE
169VOID
174 SIZE_T cbData )
175{
177 UINT32 freeInBuffer;
178 SIZE_T tmp;
179
181
182 pState->dataLengthL += cbData; // dataLengthH is not used...
183
184 bytesInBuffer = pState->bytesInBuffer;
185
186 //
187 // If previous data in buffer, buffer new input and transform if possible.
188 //
189 if( bytesInBuffer > 0 )
190 {
192
194 if( cbData < freeInBuffer )
195 {
196 //
197 // All the data will fit in the buffer.
198 // We don't do anything here.
199 // As cbData < inputBlockSize the bulk data processing is skipped,
200 // and the data will be copied to the buffer at the end
201 // of this code.
202 } else {
203 //
204 // Enough data to fill the whole buffer & process it
205 //
206 memcpy(&pState->buffer[bytesInBuffer], pbData, freeInBuffer);
207 pbData += freeInBuffer;
208 cbData -= freeInBuffer;
210
211 bytesInBuffer = 0;
212 }
213 }
214
215 //
216 // Internal buffer is empty; process all remaining whole blocks in the input
217 //
219 {
222 pbData += cbData - tmp;
223 cbData = tmp;
224 }
225
227
228 //
229 // buffer remaining input if necessary.
230 //
231 if( cbData > 0 )
232 {
233 memcpy( &pState->buffer[bytesInBuffer], pbData, cbData );
235 }
236
237 pState->bytesInBuffer = bytesInBuffer;
238}
239
240
241//
242// SymCryptSha224Append
243//
244SYMCRYPT_NOINLINE
245VOID
250 SIZE_T cbData )
251{
253}
254
255
256//
257// SymCryptSha256Result
258//
259SYMCRYPT_NOINLINE
260VOID
265{
266 //
267 // We don't use the common padding code as that is slower, and SHA-256 is very frequently used in
268 // performance-sensitive areas.
269 //
271 SIZE_T tmp;
272
274
275 bytesInBuffer = pState->bytesInBuffer;
276
277 //
278 // The buffer is never completely full, so we can always put the first
279 // padding byte in.
280 //
281 pState->buffer[bytesInBuffer++] = 0x80;
282
283 if( bytesInBuffer > 64-8 ) {
284 //
285 // No room for the rest of the padding. Pad with zeroes & process block
286 // bytesInBuffer is at most 64, so we do not have an integer underflow
287 //
289 SymCryptSha256AppendBlocks( &pState->chain, pState->buffer, 64, &tmp );
290 bytesInBuffer = 0;
291 }
292
293 //
294 // Set rest of padding
295 // At this point bytesInBuffer <= 64-8, so we don't have an underflow
296 // We wipe to the end of the buffer as it is 16-aligned,
297 // and it is faster to wipe to an aligned point
298 //
300 SYMCRYPT_STORE_MSBFIRST64( &pState->buffer[64-8], pState->dataLengthL * 8 );
301
302 //
303 // Process the final block
304 //
305 SymCryptSha256AppendBlocks( &pState->chain, pState->buffer, 64, &tmp );
306
307 //
308 // Write the output in the correct byte order
309 //
310 SymCryptUint32ToMsbFirst( &pState->chain.H[0], pbResult, 8 );
311
312 //
313 // Wipe & re-initialize
314 // We have to wipe the whole state because the Init call
315 // might be optimized away by a smart compiler.
316 //
317 SymCryptWipeKnownSize( pState, sizeof( *pState ) );
318
319 memcpy( &pState->chain.H[0], &sha256InitialState[0], sizeof( sha256InitialState ) );
321}
322
323
324//
325// SymCryptSha224Result
326//
327SYMCRYPT_NOINLINE
328VOID
333{
334 SYMCRYPT_ALIGN BYTE sha256Result[SYMCRYPT_SHA256_RESULT_SIZE]; // Buffer for SHA-256 output
335
336 //
337 // The SHA-3224 result is the first 28 bytes of the SHA-256 result of our state
338 //
341
342 //
343 // The buffer was already wiped by the SymCryptSha256Result function, we
344 // just have to re-initialize for SHA-224
345 //
347
348 SymCryptWipeKnownSize( sha256Result, sizeof( sha256Result ) );
349}
350
351
352VOID
358{
359 SYMCRYPT_ALIGN SYMCRYPT_SHA256_STATE_EXPORT_BLOB blob; // local copy to have proper alignment.
361
363
364 SymCryptWipeKnownSize( &blob, sizeof( blob ) ); // wipe to avoid any data leakage
365
366 blob.header.magic = SYMCRYPT_BLOB_MAGIC;
368 blob.header.type = type;
369
370 //
371 // Copy the relevant data. Buffer will be 0-padded.
372 //
373
374 SymCryptUint32ToMsbFirst( &pState->chain.H[0], &blob.chain[0], 8 );
375 blob.dataLength = pState->dataLengthL;
376 memcpy( &blob.buffer[0], &pState->buffer[0], blob.dataLength & 0x3f );
377
378 SYMCRYPT_ASSERT( (PCBYTE) &blob + sizeof( blob ) - sizeof( SYMCRYPT_BLOB_TRAILER ) == (PCBYTE) &blob.trailer );
379 SymCryptMarvin32( SymCryptMarvin32DefaultSeed, (PCBYTE) &blob, sizeof( blob ) - sizeof( SYMCRYPT_BLOB_TRAILER ), &blob.trailer.checksum[0] );
380
381 memcpy( pbBlob, &blob, sizeof( blob ) );
382
383//cleanup:
384 SymCryptWipeKnownSize( &blob, sizeof( blob ) );
385 return;
386}
387
388
389VOID
394{
396}
397
398
399VOID
404{
406}
407
408
415{
416 SYMCRYPT_ERROR scError = SYMCRYPT_NO_ERROR;
417 SYMCRYPT_ALIGN SYMCRYPT_SHA256_STATE_EXPORT_BLOB blob; // local copy to have proper alignment.
418 BYTE checksum[8];
419
421 memcpy( &blob, pbBlob, sizeof( blob ) );
422
423 if( blob.header.magic != SYMCRYPT_BLOB_MAGIC ||
425 blob.header.type != type )
426 {
427 scError = SYMCRYPT_INVALID_BLOB;
428 goto cleanup;
429 }
430
432 if( memcmp( checksum, &blob.trailer.checksum[0], 8 ) != 0 )
433 {
434 scError = SYMCRYPT_INVALID_BLOB;
435 goto cleanup;
436 }
437
438 SymCryptMsbFirstToUint32( &blob.chain[0], &pState->chain.H[0], 8 );
439 pState->dataLengthL = blob.dataLength;
440 pState->bytesInBuffer = blob.dataLength & 0x3f;
441 memcpy( &pState->buffer[0], &blob.buffer[0], pState->bytesInBuffer );
442
444
445cleanup:
446 SymCryptWipeKnownSize( &blob, sizeof(blob) );
447 return scError;
448}
449
450
456{
458}
459
460
466{
468}
469
470
471
472//
473// Simple test vector for FIPS module testing
474//
475
477 0xba, 0x78, 0x16, 0xbf, 0x8f, 0x01, 0xcf, 0xea,
478 0x41, 0x41, 0x40, 0xde, 0x5d, 0xae, 0x22, 0x23,
479 0xb0, 0x03, 0x61, 0xa3, 0x96, 0x17, 0x7a, 0x9c,
480 0xb4, 0x10, 0xff, 0x61, 0xf2, 0x00, 0x15, 0xad,
481 } ;
482
483VOID
486{
488
490
491 SymCryptInjectError( result, sizeof( result ) );
492
493 if( memcmp( result, SymCryptSha256KATAnswer, sizeof( result ) ) != 0 ) {
494 SymCryptFatal( 'SH25' );
495 }
496}
497
498//
499// Simple test vector for FIPS module testing
500//
501
503 0x23, 0x09, 0x7d, 0x22, 0x34, 0x05, 0xd8, 0x22,
504 0x86, 0x42, 0xa4, 0x77, 0xbd, 0xa2, 0x55, 0xb3,
505 0x2a, 0xad, 0xbc, 0xe4, 0xbd, 0xa0, 0xb3, 0xf7,
506 0xe3, 0x6c, 0x9d, 0xa7,
507 } ;
508
509VOID
512{
514
516
517 SymCryptInjectError( result, sizeof( result ) );
518
519 if( memcmp( result, SymCryptSha224KATAnswer, sizeof( result ) ) != 0 ) {
520 SymCryptFatal( 'SH22' );
521 }
522}
523
524
525
526//
527// Below are multiple implementations of the SymCryptSha256AppendBlocks function,
528// with a compile-time switch about which one to use.
529// We keep the multiple implementations here for future reference;
530// as CPU architectures evolve we might want to switch to one of the
531// other implementations.
532// All implementations here have been tested, but some lack production hardening.
533//
534
535//
536// Enable frame pointer omission to free up an extra register on X86.
537//
538#if SYMCRYPT_CPU_X86 && SYMCRYPT_MS_VC && !defined(__clang__)
539#pragma optimize( "y", on )
540#endif
541
542//
543// For documentation on these function see FIPS 180-2
544//
545// MAJ and CH are the functions Maj and Ch from the standard.
546// CSIGMA0 and CSIGMA1 are the capital sigma functions.
547// LSIGMA0 and LSIGMA1 are the lowercase sigma functions.
548//
549// The canonical definitions of the MAJ and CH functions are:
550//#define MAJ( x, y, z ) (((x) & (y)) ^ ((x) & (z)) ^ ((y) & (z)))
551//#define CH( x, y, z ) (((x) & (y)) ^ ((~(x)) & (z)))
552// We use optimized versions defined below
553//
554#define MAJ( x, y, z ) ((((z) | (y)) & (x) ) | ((z) & (y)))
555#define CH( x, y, z ) ((((z) ^ (y)) & (x)) ^ (z))
556
557//
558// The four Sigma functions
559//
560
561//
562// We have two versions of the rotate-and-xor functions.
563// one is just a macro that does the rotations and xors.
564// This works well on ARM
565// For Intel/AMD we have one where we use the rotated value
566// from one intermediate result to derive the next rotated
567// value from. This removes one register copy from the
568// code stream.
569//
570// In practice, our compiler doesn't take advantage of the
571// reduction in the # operations required, and inserts a
572// bunch of extra register copies anyway.
573// It actually hurts on AMD64.
574//
575// This should be re-tuned for every release to get the best overall
576// SHA-256 performance.
577// At the moment we get an improvement from 19.76 c/B to 19.40 c/B on a Core 2 core.
578// We should probably tune this to the Atom CPU.
579//
580#if SYMCRYPT_CPU_X86
581#define USE_CSIGMA0_MULTIROT 1
582#define USE_CSIGMA1_MULTIROT 0
583#define USE_LSIGMA0_MULTIROT 0
584#define USE_LSIGMA1_MULTIROT 0
585
586#else
587//
588// On ARM we have no reason to believe this helps at all.
589// on AMD64 it slows our code down.
590//
591#define USE_CSIGMA0_MULTIROT 0
592#define USE_CSIGMA1_MULTIROT 0
593#define USE_LSIGMA0_MULTIROT 0
594#define USE_LSIGMA1_MULTIROT 0
595#endif
596
597#if USE_CSIGMA0_MULTIROT
599UINT32
601{
602 UINT32 res;
603 x = ROR32( x, 2 );
604 res = x;
605 x = ROR32( x, 11 );
606 res ^= x;
607 x = ROR32( x, 9 );
608 res ^= x;
609 return res;
610}
611#else
612#define CSIGMA0( x ) (ROR32((x), 2) ^ ROR32((x), 13) ^ ROR32((x), 22))
613#endif
614
615#if USE_CSIGMA1_MULTIROT
617UINT32
619{
620 UINT32 res;
621 x = ROR32( x, 6 );
622 res = x;
623 x = ROR32( x, 5 );
624 res ^= x;
625 x = ROR32( x, 14 );
626 res ^= x;
627 return res;
628}
629#else
630#define CSIGMA1( x ) (ROR32((x), 6) ^ ROR32((x), 11) ^ ROR32((x), 25))
631#endif
632
633#if USE_LSIGMA0_MULTIROT
635UINT32
637{
638 UINT32 res;
639 res = x >> 3;
640 x = ROR32( x, 7 );
641 res ^= x;
642 x = ROR32( x, 11 );
643 res ^= x;
644 return res;
645}
646#else
647#define LSIGMA0( x ) (ROR32((x), 7) ^ ROR32((x), 18) ^ ((x)>> 3))
648#endif
649
650#if USE_LSIGMA1_MULTIROT
652UINT32
654{
655 UINT32 res;
656 res = x >> 10;
657 x = ROR32( x, 17 );
658 res ^= x;
659 x = ROR32( x, 2 );
660 res ^= x;
661 return res;
662}
663#else
664#define LSIGMA1( x ) (ROR32((x), 17) ^ ROR32((x), 19) ^ ((x)>>10))
665#endif
666
667
668//
669// The values a-h are stored in an array called ah.
670// We have unrolled the loop 16 times. This makes both the indices into
671// the ah array constant, and it makes the message addressing constant.
672// This provides a significant speed improvement, at the cost of making
673// the main loop about 4 kB in code.
674//
675// The earlier implementation had the loop unrolled 8 times, and is
676// around 10 cycles/byte slower. If loading the code from disk takes
677// 100 cycles/byte, then we break even once you have hashed 20 kB.
678// This is a worthwhile tradeoff as all code is codesigned with SHA-256.
679//
680
681//
682// Core round macro
683//
684// r16 is the round number mod 16, r is the round number.
685// r16 is a separate macro argument because it is always a compile-time constant
686// which allows much better optimizations of the memory accesses.
687//
688// ah[ r16 &7] = h
689// ah[(r16+1)&7] = g;
690// ah[(r16+2)&7] = f;
691// ah[(r16+3)&7] = e;
692// ah[(r16+4)&7] = d;
693// ah[(r16+5)&7] = c;
694// ah[(r16+6)&7] = b;
695// ah[(r16+7)&7] = a;
696//
697// After that incrementing the round number will automatically map a->b, b->c, etc.
698//
699// The core round, after the message word has been computed for this round and put in Wt.
700// r16 is the round number modulo 16. (Static after loop unrolling)
701// r is the round number (dynamic, which is why we don't use (r&0xf) for r16)
702// In more readable form this macro does the following:
703// h += CSIGMA( e ) + CH( e, f, g ) + K[round] + W[round];
704// d += h;
705// h += CSIGMA( a ) + MAJ( a, b, c );
706//
707#define CROUND( r16, r ) {;\
708 ah[ r16 &7] += CSIGMA1(ah[(r16+3)&7]) + CH(ah[(r16+3)&7], ah[(r16+2)&7], ah[(r16+1)&7]) + SymCryptSha256K[r] + Wt;\
709 ah[(r16+4)&7] += ah[r16 &7];\
710 ah[ r16 &7] += CSIGMA0(ah[(r16+7)&7]) + MAJ(ah[(r16+7)&7], ah[(r16+6)&7], ah[(r16+5)&7]);\
711}
712
713//
714// Initial round that reads the message.
715// r is the round number 0..15
716//
717#define IROUND( r ) {\
718 Wt = SYMCRYPT_LOAD_MSBFIRST32( &pbData[ 4*r ] );\
719 W[r] = Wt; \
720 CROUND(r,r);\
721 }
722
723//
724// Subsequent rounds.
725// r16 is the round number mod 16. rb is the round number minus r16.
726//
727#define FROUND(r16, rb) { \
728 Wt = LSIGMA1( W[(r16-2) & 15] ) + W[(r16-7) & 15] + \
729 LSIGMA0( W[(r16-15) & 15]) + W[r16 & 15]; \
730 W[r16] = Wt; \
731 CROUND( r16, r16+rb ); \
732}
733
734//
735// UINT32 implementation 1
736//
737VOID
743 _Out_ SIZE_T * pcbRemaining )
744{
747 int round;
748 UINT32 Wt;
749
750 while( cbData >= 64 )
751 {
752 ah[7] = pChain->H[0];
753 ah[6] = pChain->H[1];
754 ah[5] = pChain->H[2];
755 ah[4] = pChain->H[3];
756 ah[3] = pChain->H[4];
757 ah[2] = pChain->H[5];
758 ah[1] = pChain->H[6];
759 ah[0] = pChain->H[7];
760
761 //
762 // initial rounds 1 to 16
763 //
764
765 IROUND( 0 );
766 IROUND( 1 );
767 IROUND( 2 );
768 IROUND( 3 );
769 IROUND( 4 );
770 IROUND( 5 );
771 IROUND( 6 );
772 IROUND( 7 );
773 IROUND( 8 );
774 IROUND( 9 );
775 IROUND( 10 );
776 IROUND( 11 );
777 IROUND( 12 );
778 IROUND( 13 );
779 IROUND( 14 );
780 IROUND( 15 );
781
782
783 //
784 // rounds 16 to 64.
785 //
786 for( round=16; round<64; round += 16 )
787 {
788 FROUND( 0, round );
789 FROUND( 1, round );
790 FROUND( 2, round );
791 FROUND( 3, round );
792 FROUND( 4, round );
793 FROUND( 5, round );
794 FROUND( 6, round );
795 FROUND( 7, round );
796 FROUND( 8, round );
797 FROUND( 9, round );
798 FROUND( 10, round );
799 FROUND( 11, round );
800 FROUND( 12, round );
801 FROUND( 13, round );
802 FROUND( 14, round );
803 FROUND( 15, round );
804 }
805
806 pChain->H[0] = ah[7] + pChain->H[0];
807 pChain->H[1] = ah[6] + pChain->H[1];
808 pChain->H[2] = ah[5] + pChain->H[2];
809 pChain->H[3] = ah[4] + pChain->H[3];
810 pChain->H[4] = ah[3] + pChain->H[4];
811 pChain->H[5] = ah[2] + pChain->H[5];
812 pChain->H[6] = ah[1] + pChain->H[6];
813 pChain->H[7] = ah[0] + pChain->H[7];
814
815 pbData += 64;
816 cbData -= 64;
817
818 }
819
820 *pcbRemaining = cbData;
821
822 //
823 // Wipe the variables;
824 //
825 SymCryptWipeKnownSize( ah, sizeof( ah ) );
826 SymCryptWipeKnownSize( W, sizeof( W ) );
827 SYMCRYPT_FORCE_WRITE32( &Wt, 0 );
828}
829
830VOID
836 _Out_ SIZE_T * pcbRemaining )
837{
838 //
839 // Different arrangement of the code, currently 25 c/B vs 20 c/b for the version above.
840 // On Atom: 50 c/B vs 41 c/B for the one above.
841 //
842 SYMCRYPT_ALIGN UINT32 buf[4 + 8 + 64]; // chaining state concatenated with the expanded input block
843 UINT32 * W = &buf[4 + 8];
844 UINT32 * ha = &buf[4]; // initial state words, in order h, g, ..., b, a
845 UINT32 A, B, C, D, T;
846 int r;
847
848 ha[7] = pChain->H[0]; buf[3] = ha[7];
849 ha[6] = pChain->H[1]; buf[2] = ha[6];
850 ha[5] = pChain->H[2]; buf[1] = ha[5];
851 ha[4] = pChain->H[3]; buf[0] = ha[4];
852 ha[3] = pChain->H[4];
853 ha[2] = pChain->H[5];
854 ha[1] = pChain->H[6];
855 ha[0] = pChain->H[7];
856
857 while( cbData >= 64 )
858 {
859 //
860 // Capture the input into W[0..15]
861 //
862 for( r=0; r<16; r++ )
863 {
864 W[r] = SYMCRYPT_LOAD_MSBFIRST32( &pbData[ 4*r ] );
865 }
866
867 //
868 // Expand the message
869 //
870 A = W[15];
871 B = W[14];
872 D = W[0];
873 for( r=16; r<64; r+= 2 )
874 {
875 // Loop invariant: A=W[r-1], B = W[r-2], D = W[r-16]
876
877 //
878 // Macro for one word of message expansion.
879 // Invariant:
880 // on entry: a = W[r-1], b = W[r-2], d = W[r-16]
881 // on exit: W[r] computed, a = W[r-1], b = W[r], c = W[r-15]
882 //
883 #define EXPAND( a, b, c, d, r ) \
884 c = W[r-15]; \
885 b = d + LSIGMA1( b ) + W[r-7] + LSIGMA0( c ); \
886 W[r] = b; \
887
888 EXPAND( A, B, C, D, r );
889 EXPAND( B, A, D, C, (r+1));
890
891 #undef EXPAND
892 }
893
894 A = ha[7];
895 B = ha[6];
896 C = ha[5];
897 D = ha[4];
898
899 for( r=0; r<64; r += 4 )
900 {
901 //
902 // Loop invariant:
903 // A, B, C, and D are the a,b,c,d values of the current state.
904 // W[r] is the next expanded message word to be processed.
905 // W[r-8 .. r-5] contain the current state words h, g, f, e.
906 //
907
908 //
909 // Macro to compute one round
910 //
911 #define DO_ROUND( a, b, c, d, t, r ) \
912 t = W[r] + CSIGMA1( W[r-5] ) + W[r-8] + CH( W[r-5], W[r-6], W[r-7] ) + SymCryptSha256K[r]; \
913 W[r-4] = t + d; \
914 d = t + CSIGMA0( a ) + MAJ( c, b, a );
915
916 DO_ROUND( A, B, C, D, T, r );
917 DO_ROUND( D, A, B, C, T, (r+1) );
918 DO_ROUND( C, D, A, B, T, (r+2) );
919 DO_ROUND( B, C, D, A, T, (r+3) );
920 #undef DO_ROUND
921 }
922
923 buf[3] = ha[7] = buf[3] + A;
924 buf[2] = ha[6] = buf[2] + B;
925 buf[1] = ha[5] = buf[1] + C;
926 buf[0] = ha[4] = buf[0] + D;
927 ha[3] += W[r-5];
928 ha[2] += W[r-6];
929 ha[1] += W[r-7];
930 ha[0] += W[r-8];
931
932 pbData += 64;
933 cbData -= 64;
934 }
935
936 pChain->H[0] = ha[7];
937 pChain->H[1] = ha[6];
938 pChain->H[2] = ha[5];
939 pChain->H[3] = ha[4];
940 pChain->H[4] = ha[3];
941 pChain->H[5] = ha[2];
942 pChain->H[6] = ha[1];
943 pChain->H[7] = ha[0];
944
945 *pcbRemaining = cbData;
946
947 SymCryptWipeKnownSize( buf, sizeof( buf ) );
952}
953
954#undef CROUND
955#undef IROUND
956#undef FROUND
957
958#if SYMCRYPT_CPU_X86 | SYMCRYPT_CPU_AMD64
959
960//
961// Don't omit frame pointer for XMM code; it isn't register-starved as much
962//
963#if SYMCRYPT_CPU_X86 && SYMCRYPT_MS_VC && !defined(__clang__)
964#pragma optimize( "y", off )
965#endif
966
967#ifdef __clang__
968#pragma clang attribute push (__attribute__((target("ssse3,sha"))), apply_to=function)
969#else
970#pragma GCC push_options
971#pragma GCC target("ssse3,sha")
972#endif
973
974//
975// Code that uses the XMM registers.
976// This code is currently unused. It was written in case it would provide better performance, but
977// it did not. We are retaining it in case it might be useful in a future CPU generation.
978//
979#if 0
980
981#define MAJXMM( x, y, z ) _mm_or_si128( _mm_and_si128( _mm_or_si128( z, y ), x ), _mm_and_si128( z, y ))
982#define CHXMM( x, y, z ) _mm_xor_si128( _mm_and_si128( _mm_xor_si128( z, y ), x ), z )
983
984#define CSIGMA0XMM( x ) \
985 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
986 _mm_slli_epi32(x,30) , _mm_srli_epi32(x, 2) ),\
987 _mm_slli_epi32(x,19) ), _mm_srli_epi32(x, 13) ),\
988 _mm_slli_epi32(x,10) ), _mm_srli_epi32(x, 22) )
989#define CSIGMA1XMM( x ) \
990 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
991 _mm_slli_epi32(x,26) , _mm_srli_epi32(x, 6) ),\
992 _mm_slli_epi32(x,21) ), _mm_srli_epi32(x, 11) ),\
993 _mm_slli_epi32(x,7) ), _mm_srli_epi32(x, 25) )
994#define LSIGMA0XMM( x ) \
995 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
996 _mm_slli_epi32(x,25) , _mm_srli_epi32(x, 7) ),\
997 _mm_slli_epi32(x,14) ), _mm_srli_epi32(x, 18) ),\
998 _mm_srli_epi32(x, 3) )
999#define LSIGMA1XMM( x ) \
1000 _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( _mm_xor_si128( \
1001 _mm_slli_epi32(x,15) , _mm_srli_epi32(x, 17) ),\
1002 _mm_slli_epi32(x,13) ), _mm_srli_epi32(x, 19) ),\
1003 _mm_srli_epi32(x,10) )
1004
1005VOID
1007SymCryptSha256AppendBlocks_xmm1(
1010 SIZE_T cbData,
1011 _Out_ SIZE_T * pcbRemaining )
1012{
1013 //
1014 // Implementation that has one value in each XMM register.
1015 // This is significantly slower than the _ul1 implementation
1016 // but can be extended to compute 4 hash blocks in parallel.
1017 //
1018 SYMCRYPT_ALIGN __m128i buf[4 + 8 + 64]; // chaining state concatenated with the expanded input block
1019 __m128i * W = &buf[4 + 8];
1020 __m128i * ha = &buf[4]; // initial state words, in order h, g, ..., b, a
1021 __m128i A, B, C, D, T;
1022 int r;
1023
1024 //
1025 // For 1-input only; set the input buffer to zero so that we have known values in every byte
1026 //
1027 //SymCryptWipeKnownSize( buf, sizeof( buf ) );
1028
1029 //
1030 // Copy the chaining state into the start of the buffer, order = h,g,f,e,d,c,b,a
1031 //
1032 ha[7] = _mm_insert_epi32(ha[7], pChain->H[0], 0);
1033 ha[6] = _mm_insert_epi32(ha[6], pChain->H[1], 0);
1034 ha[5] = _mm_insert_epi32(ha[5], pChain->H[2], 0);
1035 ha[4] = _mm_insert_epi32(ha[4], pChain->H[3], 0);
1036 ha[3] = _mm_insert_epi32(ha[3], pChain->H[4], 0);
1037 ha[2] = _mm_insert_epi32(ha[2], pChain->H[5], 0);
1038 ha[1] = _mm_insert_epi32(ha[1], pChain->H[6], 0);
1039 ha[0] = _mm_insert_epi32(ha[0], pChain->H[7], 0);
1040
1041 buf[0] = ha[4];
1042 buf[1] = ha[5];
1043 buf[2] = ha[6];
1044 buf[3] = ha[7];
1045
1046 while( cbData >= 64 )
1047 {
1048
1049 //
1050 // Capture the input into W[0..15]
1051 //
1052 for( r=0; r<16; r++ )
1053 {
1054 W[r] = _mm_insert_epi32(W[r], SYMCRYPT_LOAD_MSBFIRST32( &pbData[ 4*r ] ), 0);
1055 }
1056
1057 //
1058 // Expand the message
1059 //
1060 A = W[15];
1061 B = W[14];
1062 D = W[0];
1063 for( r=16; r<64; r+= 2 )
1064 {
1065 // Loop invariant: A=W[r-1], B = W[r-2], D = W[r-16]
1066
1067 //
1068 // Macro for one word of message expansion.
1069 // Invariant:
1070 // on entry: a = W[r-1], b = W[r-2], d = W[r-16]
1071 // on exit: W[r] computed, a = W[r-1], b = W[r], c = W[r-15]
1072 //
1073 #define EXPAND( a, b, c, d, r ) \
1074 c = W[r-15]; \
1075 b = _mm_add_epi32( _mm_add_epi32( _mm_add_epi32( d, LSIGMA1XMM( b ) ), W[r-7] ), LSIGMA0XMM( c ) ); \
1076 W[r] = b; \
1077
1078 EXPAND( A, B, C, D, r );
1079 EXPAND( B, A, D, C, (r+1));
1080
1081 #undef EXPAND
1082 }
1083
1084 A = ha[7];
1085 B = ha[6];
1086 C = ha[5];
1087 D = ha[4];
1088
1089 for( r=0; r<64; r += 4 )
1090 {
1091 //
1092 // Loop invariant:
1093 // A, B, C, and D are the a,b,c,d values of the current state.
1094 // W[r] is the next expanded message word to be processed.
1095 // W[r-8 .. r-5] contain the current state words h, g, f, e.
1096 //
1097
1098 //
1099 // Macro to compute one round
1100 //
1101 #define DO_ROUND( a, b, c, d, t, r ) \
1102 t = W[r]; \
1103 t = _mm_add_epi32( t, CSIGMA1XMM( W[r-5] ) ); \
1104 t = _mm_add_epi32( t, W[r-8] ); \
1105 t = _mm_add_epi32( t, CHXMM( W[r-5], W[r-6], W[r-7] ) ); \
1106 t = _mm_add_epi32( t, _mm_cvtsi32_si128( SymCryptSha256K[r] ) ); \
1107 W[r-4] = _mm_add_epi32( t, d ); \
1108 d = _mm_add_epi32( t, CSIGMA0XMM( a ) ); \
1109 d = _mm_add_epi32( d, MAJXMM( c, b, a ) );
1110
1111 DO_ROUND( A, B, C, D, T, r );
1112 DO_ROUND( D, A, B, C, T, (r+1) );
1113 DO_ROUND( C, D, A, B, T, (r+2) );
1114 DO_ROUND( B, C, D, A, T, (r+3) );
1115 #undef DO_ROUND
1116 }
1117
1118 buf[3] = ha[7] = _mm_add_epi32( buf[3], A );
1119 buf[2] = ha[6] = _mm_add_epi32( buf[2], B );
1120 buf[1] = ha[5] = _mm_add_epi32( buf[1], C );
1121 buf[0] = ha[4] = _mm_add_epi32( buf[0], D );
1122 ha[3] = _mm_add_epi32( ha[3], W[r-5] );
1123 ha[2] = _mm_add_epi32( ha[2], W[r-6] );
1124 ha[1] = _mm_add_epi32( ha[1], W[r-7] );
1125 ha[0] = _mm_add_epi32( ha[0], W[r-8] );
1126
1127 pbData += 64;
1128 cbData -= 64;
1129 }
1130
1131 //
1132 // Copy the chaining state back into the hash structure
1133 //
1134 pChain->H[0] = _mm_extract_epi32(ha[7], 0);
1135 pChain->H[1] = _mm_extract_epi32(ha[6], 0);
1136 pChain->H[2] = _mm_extract_epi32(ha[5], 0);
1137 pChain->H[3] = _mm_extract_epi32(ha[4], 0);
1138 pChain->H[4] = _mm_extract_epi32(ha[3], 0);
1139 pChain->H[5] = _mm_extract_epi32(ha[2], 0);
1140 pChain->H[6] = _mm_extract_epi32(ha[1], 0);
1141 pChain->H[7] = _mm_extract_epi32(ha[0], 0);
1142
1143 *pcbRemaining = cbData;
1144
1145 SymCryptWipeKnownSize( buf, sizeof( buf ) );
1146 SymCryptWipeKnownSize( &A, sizeof( A ) );
1147 SymCryptWipeKnownSize( &B, sizeof( B ) );
1148 SymCryptWipeKnownSize( &C, sizeof( C ) );
1149 SymCryptWipeKnownSize( &D, sizeof( D ) );
1150 SymCryptWipeKnownSize( &T, sizeof( T ) );
1151}
1152
1153
1154//
1155// XMM implementation 2
1156// We use the XMM registers to compute part of the message schedule.
1157// The load, BSWAP, and part of the message schedule recursion are done in XMM registers.
1158// The rest of the work is done using integers.
1159//
1160// Core2: 0.1 c/B slower than the _ul1
1161// Atom: 1.0 c/B slower than _ul1 (42.34 vs 41.39 c/B)
1162//
1163VOID
1165SymCryptSha256AppendBlocks_xmm2(
1168 SIZE_T cbData,
1169 _Out_ SIZE_T * pcbRemaining )
1170{
1171 SYMCRYPT_ALIGN union { UINT32 ul[16]; __m128i xmm[4]; } W;
1172 SYMCRYPT_ALIGN UINT32 ah[8];
1173 int round;
1174 UINT32 Wt;
1175 const __m128i BYTE_REVERSE_32 = _mm_set_epi8( 12, 13, 14, 15, 8, 9, 10, 11, 4, 5, 6, 7, 0, 1, 2, 3 );
1176
1177 ah[7] = pChain->H[0];
1178 ah[6] = pChain->H[1];
1179 ah[5] = pChain->H[2];
1180 ah[4] = pChain->H[3];
1181 ah[3] = pChain->H[4];
1182 ah[2] = pChain->H[5];
1183 ah[1] = pChain->H[6];
1184 ah[0] = pChain->H[7];
1185
1186#define CROUND( r16, r ) {;\
1187 ah[ r16 &7] += CSIGMA1(ah[(r16+3)&7]) + CH(ah[(r16+3)&7], ah[(r16+2)&7], ah[(r16+1)&7]) + SymCryptSha256K[r] + Wt;\
1188 ah[(r16+4)&7] += ah[r16 &7];\
1189 ah[ r16 &7] += CSIGMA0(ah[(r16+7)&7]) + MAJ(ah[(r16+7)&7], ah[(r16+6)&7], ah[(r16+5)&7]);\
1190}
1191
1192
1193//
1194// Initial round that reads the message.
1195// r is the round number 0..15
1196//
1197// Wt = LOAD_MSBFIRST32( &pbData[ 4*r ] );\
1198// W.ul[r] = Wt; \
1199
1200#define IROUND( r ) {\
1201 Wt = W.ul[r];\
1202 CROUND(r,r);\
1203 }
1204
1205//
1206// Subsequent rounds.
1207// r16 is the round number mod 16. rb is the round number minus r16.
1208//
1209#define FROUND(r16, rb) { \
1210 Wt = W.ul[r16];\
1211 CROUND( r16, r16+rb ); \
1212}
1213
1214
1215 while( cbData >= 64 )
1216 {
1217 //
1218 // The code is faster if we directly access the W.ul array, rather than the W.xmm alias.
1219 // I think the compiler gets more confused if you use the W.xmm values.
1220 // We retain them in the union to ensure alignment
1221 //
1222 _mm_store_si128( (__m128i *)&W.ul[ 0], _mm_shuffle_epi8( _mm_loadu_si128( (__m128i *)&pbData[ 0 ] ), BYTE_REVERSE_32 ));
1223 _mm_store_si128( (__m128i *)&W.ul[ 4], _mm_shuffle_epi8( _mm_loadu_si128( (__m128i *)&pbData[ 16 ] ), BYTE_REVERSE_32 ));
1224 _mm_store_si128( (__m128i *)&W.ul[ 8], _mm_shuffle_epi8( _mm_loadu_si128( (__m128i *)&pbData[ 32 ] ), BYTE_REVERSE_32 ));
1225 _mm_store_si128( (__m128i *)&W.ul[12], _mm_shuffle_epi8( _mm_loadu_si128( (__m128i *)&pbData[ 48 ] ), BYTE_REVERSE_32 ));
1226
1227 //
1228 // initial rounds 1 to 16
1229 //
1230
1231 IROUND( 0 );
1232 IROUND( 1 );
1233 IROUND( 2 );
1234 IROUND( 3 );
1235 IROUND( 4 );
1236 IROUND( 5 );
1237 IROUND( 6 );
1238 IROUND( 7 );
1239 IROUND( 8 );
1240 IROUND( 9 );
1241 IROUND( 10 );
1242 IROUND( 11 );
1243 IROUND( 12 );
1244 IROUND( 13 );
1245 IROUND( 14 );
1246 IROUND( 15 );
1247
1248
1249 //
1250 // rounds 16 to 64.
1251 //
1252 for( round=16; round<64; round += 16 )
1253 {
1254 __m128i Tmp;
1255
1257 LSIGMA0XMM(_mm_loadu_si128( (__m128i *)&W.ul[1] )),
1258 _mm_load_si128( (__m128i *)&W.ul[0] ) ),
1259 _mm_loadu_si128( (__m128i *)&W.ul[9] ) );
1260
1261 //
1262 // The final part of the message schedule can be done in XMM registers, but it isn't worth it.
1263 // The rotates in XMM take two shifts and an OR/XOR, vs one instruction in integer registers.
1264 // As the sigma1( W_{t-2} ) recursion component can only be computed 2 at a time
1265 // (because the result of the first two are the inputs to the second two)
1266 // you lose more than you gain by using XMM registers.
1267 //
1268 //Tmp = _mm_add_epi32( Tmp, LSIGMA1XMM( _mm_srli_si128( _mm_load_si128( (__m128i *)&W.ul[12] ), 8 ) ) );
1269 //Tmp = _mm_add_epi32( Tmp, LSIGMA1XMM( _mm_slli_si128( Tmp, 8 ) ) );
1270 //_mm_store_si128( (__m128i *)&W.ul[0], Tmp );
1271 //
1272
1273 _mm_store_si128( (__m128i *)&W.ul[0], Tmp );
1274 W.ul[0] += LSIGMA1( W.ul[14] );
1275 W.ul[1] += LSIGMA1( W.ul[15] );
1276 W.ul[2] += LSIGMA1( W.ul[0] );
1277 W.ul[3] += LSIGMA1( W.ul[1] );
1278
1279 FROUND( 0, round );
1280 FROUND( 1, round );
1281 FROUND( 2, round );
1282 FROUND( 3, round );
1283
1285 LSIGMA0XMM(_mm_loadu_si128( (__m128i *)&W.ul[5] )),
1286 _mm_load_si128( (__m128i *)&W.ul[4] ) ),
1287 _mm_alignr_epi8( _mm_load_si128( (__m128i *)&W.ul[0] ), _mm_load_si128( (__m128i *)&W.ul[12] ), 4) );
1288
1289 _mm_store_si128( (__m128i *)&W.ul[4], Tmp );
1290
1291 W.ul[4] += LSIGMA1( W.ul[2] );
1292 W.ul[5] += LSIGMA1( W.ul[3] );
1293 W.ul[6] += LSIGMA1( W.ul[4] );
1294 W.ul[7] += LSIGMA1( W.ul[5] );
1295
1296 FROUND( 4, round );
1297 FROUND( 5, round );
1298 FROUND( 6, round );
1299 FROUND( 7, round );
1300
1302 LSIGMA0XMM(_mm_loadu_si128( (__m128i *)&W.ul[9] )),
1303 _mm_load_si128( (__m128i *)&W.ul[8] ) ),
1304 _mm_loadu_si128( (__m128i *)&W.ul[1] ) );
1305
1306 _mm_store_si128( (__m128i *)&W.ul[8], Tmp );
1307 W.ul[ 8] += LSIGMA1( W.ul[6] );
1308 W.ul[ 9] += LSIGMA1( W.ul[7] );
1309 W.ul[10] += LSIGMA1( W.ul[8] );
1310 W.ul[11] += LSIGMA1( W.ul[9] );
1311
1312 FROUND( 8, round );
1313 FROUND( 9, round );
1314 FROUND( 10, round );
1315 FROUND( 11, round );
1316
1317
1319 LSIGMA0XMM( _mm_alignr_epi8( _mm_load_si128( (__m128i *)&W.ul[0] ), _mm_load_si128( (__m128i *)&W.ul[12] ), 4) ),
1320 _mm_load_si128( (__m128i *)&W.ul[12] ) ),
1321 _mm_loadu_si128( (__m128i *)&W.ul[5] ) );
1322
1323 _mm_store_si128( (__m128i *)&W.ul[12], Tmp );
1324 W.ul[12] += LSIGMA1( W.ul[10] );
1325 W.ul[13] += LSIGMA1( W.ul[11] );
1326 W.ul[14] += LSIGMA1( W.ul[12] );
1327 W.ul[15] += LSIGMA1( W.ul[13] );
1328
1329 FROUND( 12, round );
1330 FROUND( 13, round );
1331 FROUND( 14, round );
1332 FROUND( 15, round );
1333 }
1334
1335 pChain->H[0] = ah[7] = ah[7] + pChain->H[0];
1336 pChain->H[1] = ah[6] = ah[6] + pChain->H[1];
1337 pChain->H[2] = ah[5] = ah[5] + pChain->H[2];
1338 pChain->H[3] = ah[4] = ah[4] + pChain->H[3];
1339 pChain->H[4] = ah[3] = ah[3] + pChain->H[4];
1340 pChain->H[5] = ah[2] = ah[2] + pChain->H[5];
1341 pChain->H[6] = ah[1] = ah[1] + pChain->H[6];
1342 pChain->H[7] = ah[0] = ah[0] + pChain->H[7];
1343
1344 pbData += 64;
1345 cbData -= 64;
1346
1347 }
1348
1349 *pcbRemaining = cbData;
1350
1351 //
1352 // Wipe the variables;
1353 //
1354 SymCryptWipeKnownSize( ah, sizeof( ah ) );
1355 SymCryptWipeKnownSize( &W, sizeof( W ) );
1356 SYMCRYPT_FORCE_WRITE32( &Wt, 0 );
1357
1358#undef IROUND
1359#undef FROUND
1360#undef CROUND
1361}
1362
1363#endif
1364
1365//
1366// SHA-NI Implementation
1367//
1368
1369#if SYMCRYPT_MS_VC && !defined(__clang__)
1370// Intrinsic definitions included here
1371// until the header is updated.
1372// *******************************
1373// *******************************
1374// *******************************
1375extern __m128i _mm_sha256rnds2_epu32(__m128i, __m128i, __m128i);
1376extern __m128i _mm_sha256msg1_epu32(__m128i, __m128i);
1377extern __m128i _mm_sha256msg2_epu32(__m128i, __m128i);
1378// *******************************
1379// *******************************
1380// *******************************
1381#endif
1382
1383// For the SHA-NI implementation we will utilize 128-bit XMM registers. Each
1384// XMM state will be denoted as (R_3, R_2, R_1, R_0), where each R_i
1385// is a 32-bit word and R_i refers to bits [32*i : (32*i + 31)] of the
1386// 128-bit XMM state.
1387//
1388// The following macro updates the state variables A,B,C,...,H of the SHA algorithms
1389// for 4 rounds using:
1390// - The current round number t with 0<=t<= 63 and t a multiple of 4.
1391// - A current message XMM state _MSG which consists of 4 32-bit words
1392// ( W_(t+3), W_(t+2), W_(t+1), W_(t+0) ).
1393// - Two XMM states _ABEF and _CDGH which contain the variables
1394// ( A, B, E, F ) and ( C, D, G, H ) respectively.
1395
1396#define SHANI_UPDATE_STATE( _round, _MSG, _ABEF, _CDGH ) \
1397 _MSG = _mm_add_epi32( _MSG, *(__m128i *)&SymCryptSha256K[_round] ); /* Add the K_t constants to the W_t's */ \
1398 _CDGH = _mm_sha256rnds2_epu32( _CDGH, _ABEF, _MSG ); /* 2 rounds using SHA-NI */ \
1399 _MSG = _mm_shuffle_epi32( _MSG, 0x0e ); /* Move words 2 & 3 to positions 0 & 1 */ \
1400 _ABEF = _mm_sha256rnds2_epu32( _ABEF, _CDGH, _MSG ); /* 2 rounds using SHA-NI */
1401
1402// For the SHA message schedule (i.e. to create words W_16 to W_63) we use 4 XMM states / accumulators.
1403// Each accumulator holds 4 words.
1404//
1405// The final result for each word will be of the form W_t = X_t + Y_t, where
1406// X_t = W_(t-16) + \sigma_0(W_(t-15)) and
1407// Y_t = W_(t- 7) + \sigma_1(W_(t- 2))
1408//
1409// The X_t's are calculated by the _mm_sha256msg1_epu32 intrinsic.
1410// The \sigma_1(W_(t-2)) part of the Y_t's by the _mm_sha256msg2_epu32 intrinsic.
1411//
1412// Remarks:
1413// - Calculation of the first four X_t's (i.e. 16<=t<=19) can start from round 4 (since 19-15 = 4).
1414// - Calculation of the first four Y_t's can start from round 12 (since 19-7=12 and W_(19-7) is calculated
1415// in the intrinsic call).
1416// - Due to the W_(t-7) term, producing the Y_t's need special shifting via the _mm_alignr_epi8 intrinsic and
1417// adding the correct accumulator into another variable MTEMP.
1418//
1419// For rounds 16 - 51 we execute the following macro in a loop. For all the other rounds we
1420// use specific code.
1421//
1422// The loop invariant to be satisfied at the beginning of iteration i (corresponding to rounds
1423// (16+4*i) to (19+4*i) ) is the following:
1424// _MSG_0 = ( W_(19 + 4*i), W_(18 + 4*i), W_(17 + 4*i), W_(16 + 4*i) )
1425// _MSG_1 = ( X_(23 + 4*i), X_(22 + 4*i), X_(21 + 4*i), X_(20 + 4*i) )
1426// _MSG_2 = ( X_(27 + 4*i), X_(26 + 4*i), X_(25 + 4*i), X_(24 + 4*i) )
1427// _MSG_3 = ( W_(15 + 4*i), W_(14 + 4*i), W_(13 + 4*i), W_(12 + 4*i) )
1428//
1429#define SHANI_MESSAGE_SCHEDULE( _MSG_0, _MSG_1, _MSG_2, _MSG_3, _MTEMP ) \
1430 _MTEMP = _mm_alignr_epi8( _MSG_0, _MSG_3, 4); /* _MTEMP := ( W_(16 + 4*i), W_(15 + 4*i), W_(14 + 4*i), W_(13 + 4*i) ) */ \
1431 _MSG_1 = _mm_add_epi32( _MSG_1, _MTEMP); /* _MSG_1 := _MSG_1 + ( W_(16 + 4*i), W_(15 + 4*i), W_(14 + 4*i), W_(13 + 4*i) ) */ \
1432 _MSG_1 = _mm_sha256msg2_epu32( _MSG_1, _MSG_0 ); /* _MSG_1 := ( W_(23 + 4*i), W_(22 + 4*i), W_(21 + 4*i), W_(20 + 4*i) ) */ \
1433 _MSG_3 = _mm_sha256msg1_epu32( _MSG_3, _MSG_0 ); /* _MSG_3 := ( X_(31+4*i), X_(30+4*i), X_(29+4*i), X_(28+4*i) ) */
1434//
1435// After each iteration the subsequent call rotates the accumulators so that the loop
1436// invariant is preserved (please verify!):
1437// -- MSG_0 <---- MSG_1 <--- MSG_2 <--- MSG_3 <--
1438// | |
1439// ----------------------------------------------
1440
1441VOID
1443SymCryptSha256AppendBlocks_shani(
1446 SIZE_T cbData,
1447 _Out_ SIZE_T * pcbRemaining )
1448{
1449 const __m128i BYTE_REVERSE_32 = _mm_set_epi8( 12, 13, 14, 15, 8, 9, 10, 11, 4, 5, 6, 7, 0, 1, 2, 3 );
1450
1451 // Our chain state is in order A, B, ..., H.
1452 // First load our chaining state
1453 __m128i DCBA = _mm_loadu_si128( (__m128i *)&(pChain->H[0]) ); // (D, C, B, A)
1454 __m128i HGFE = _mm_loadu_si128( (__m128i *)&(pChain->H[4]) ); // (H, G, F, E)
1455 __m128i FEBA = _mm_unpacklo_epi64( DCBA, HGFE ); // (F, E, B, A)
1456 __m128i HGDC = _mm_unpackhi_epi64( DCBA, HGFE ); // (H, G, D, C)
1457 __m128i ABEF = _mm_shuffle_epi32( FEBA, 0x1b ); // (A, B, E, F)
1458 __m128i CDGH = _mm_shuffle_epi32( HGDC, 0x1b ); // (C, D, G, H)
1459
1460 while( cbData >= 64 )
1461 {
1462 // Save the current state for the feed-forward later
1463 __m128i ABEF_start = ABEF;
1464 __m128i CDGH_start = CDGH;
1465
1466 // Current message and temporary state
1467 __m128i MSG;
1468
1469 // Accumulators
1470 __m128i MSG_0;
1471 __m128i MSG_1;
1472 __m128i MSG_2;
1473 __m128i MSG_3;
1474
1475 // Rounds 0-3
1476 MSG = _mm_loadu_si128( (__m128i *)pbData ); // Reversed word - ( M_3, M_2, M_1, M_0 )
1477 pbData += 16;
1478 MSG = _mm_shuffle_epi8( MSG, BYTE_REVERSE_32 ); // Reverse each word
1479 MSG_0 = MSG; // MSG_0 := ( W_3 = M3, W_2 = M_2, W_1 = M_1, W_0 = M_0 )
1480
1481 SHANI_UPDATE_STATE( 0, MSG, ABEF, CDGH );
1482
1483 // Rounds 4-7
1484 MSG = _mm_loadu_si128( (__m128i *)pbData ); // Reversed word - ( M_7, M_6, M_5, M_4 )
1485 pbData += 16;
1486 MSG = _mm_shuffle_epi8( MSG, BYTE_REVERSE_32 ); // Reverse each word
1487 MSG_1 = MSG; // MSG_1 := ( W_7 = M_7, W_6 = M_6, W_5 = M_5, W_4 = M_4 )
1488
1489 SHANI_UPDATE_STATE( 4, MSG, ABEF, CDGH );
1490
1491 MSG_0 = _mm_sha256msg1_epu32( MSG_0, MSG_1 ); // MSG_0 := ( X_19, X_18, X_17, X_16 ) =
1492 // ( W_3 + \sigma_0(W_4), ..., W_0 + \sigma_0(W_1) )
1493
1494 // Rounds 8-11
1495 MSG = _mm_loadu_si128( (__m128i *)pbData ); // Reversed word - ( M_11, M_10, M_9, M_8 )
1496 pbData += 16;
1497 MSG = _mm_shuffle_epi8( MSG, BYTE_REVERSE_32 ); // Reverse each word
1498 MSG_2 = MSG; // MSG_2 := ( W_11 = M_11, W_10 = M_10, W_9 = M_9, W_8 = M_8 )
1499
1500 SHANI_UPDATE_STATE( 8, MSG, ABEF, CDGH );
1501
1502 MSG_1 = _mm_sha256msg1_epu32( MSG_1, MSG_2 ); // MSG_1 := ( X_23, X_22, X_21, X_20 )
1503
1504 // Rounds 12-15
1505 MSG = _mm_loadu_si128( (__m128i *)pbData ); // Reversed word - ( M_15, M_14, M_13, M_12 )
1506 pbData += 16;
1507 MSG = _mm_shuffle_epi8( MSG, BYTE_REVERSE_32 ); // Reverse each word
1508 MSG_3 = MSG; // MSG_3 := ( W_15 = M_15, W_14 = M_14, W_13 = M_13, W_12 = M_12 )
1509
1510 SHANI_UPDATE_STATE( 12, MSG, ABEF, CDGH );
1511
1512 MSG = _mm_alignr_epi8( MSG_3, MSG_2, 4); // MSG := ( W_12, W_11, W_10, W_9 )
1513 MSG_0 = _mm_add_epi32( MSG_0, MSG); // MSG_0 := MSG_0 + ( W_12, W_11, W_10, W_9 )
1514 MSG_0 = _mm_sha256msg2_epu32( MSG_0, MSG_3 ); // MSG_0 := ( W_19, W_18, W_17, W_16 ) =
1515 // ( X_19 + W_12 + \sigma_1(W_17)], ..., X_16 + W_9 + \sigma_1(W_14)] )
1516
1517 MSG_2 = _mm_sha256msg1_epu32( MSG_2, MSG_3 ); // MSG_2 := ( X_27, X_26, X_25, X_24 )
1518
1519
1520 // Rounds 16 - 19
1521 MSG = MSG_0;
1522 SHANI_UPDATE_STATE( 16, MSG, ABEF, CDGH );
1523 SHANI_MESSAGE_SCHEDULE( MSG_0, MSG_1, MSG_2, MSG_3, MSG );
1524
1525 // Rounds 20 - 23
1526 MSG = MSG_1;
1527 SHANI_UPDATE_STATE( 20, MSG, ABEF, CDGH );
1528 SHANI_MESSAGE_SCHEDULE( MSG_1, MSG_2, MSG_3, MSG_0, MSG );
1529
1530 // Rounds 24 - 27
1531 MSG = MSG_2;
1532 SHANI_UPDATE_STATE( 24, MSG, ABEF, CDGH );
1533 SHANI_MESSAGE_SCHEDULE( MSG_2, MSG_3, MSG_0, MSG_1, MSG );
1534
1535 // Rounds 28 - 31
1536 MSG = MSG_3;
1537 SHANI_UPDATE_STATE( 28, MSG, ABEF, CDGH );
1538 SHANI_MESSAGE_SCHEDULE( MSG_3, MSG_0, MSG_1, MSG_2, MSG );
1539
1540 // Rounds 32 - 35
1541 MSG = MSG_0;
1542 SHANI_UPDATE_STATE( 32, MSG, ABEF, CDGH );
1543 SHANI_MESSAGE_SCHEDULE( MSG_0, MSG_1, MSG_2, MSG_3, MSG );
1544
1545 // Rounds 36 - 39
1546 MSG = MSG_1;
1547 SHANI_UPDATE_STATE( 36, MSG, ABEF, CDGH );
1548 SHANI_MESSAGE_SCHEDULE( MSG_1, MSG_2, MSG_3, MSG_0, MSG );
1549
1550 // Rounds 40 - 43
1551 MSG = MSG_2;
1552 SHANI_UPDATE_STATE( 40, MSG, ABEF, CDGH );
1553 SHANI_MESSAGE_SCHEDULE( MSG_2, MSG_3, MSG_0, MSG_1, MSG );
1554
1555 // Rounds 44 - 47
1556 MSG = MSG_3;
1557 SHANI_UPDATE_STATE( 44, MSG, ABEF, CDGH );
1558 SHANI_MESSAGE_SCHEDULE( MSG_3, MSG_0, MSG_1, MSG_2, MSG );
1559
1560 // Rounds 48 - 51
1561 MSG = MSG_0;
1562 SHANI_UPDATE_STATE( 48, MSG, ABEF, CDGH );
1563 SHANI_MESSAGE_SCHEDULE( MSG_0, MSG_1, MSG_2, MSG_3, MSG );
1564
1565 // Rounds 52 - 55
1566 MSG = MSG_1; // ( W_55, W_54, W_53, W_52 )
1567 SHANI_UPDATE_STATE( 52, MSG, ABEF, CDGH );
1568
1569 MSG = _mm_alignr_epi8( MSG_1, MSG_0, 4); // MSG := ( W_52, W_51, W_50, W_49 )
1570 MSG_2 = _mm_add_epi32( MSG_2, MSG); // MSG_2 := MSG_2 + ( W_52, W_51, W_50, W_49 )
1571 MSG_2 = _mm_sha256msg2_epu32( MSG_2, MSG_1 ); // Calculate ( W_59, W_58, W_57, W_56 )
1572
1573 // Rounds 56 - 59
1574 MSG = MSG_2; // ( W_59, W_58, W_57, W_56 )
1575 SHANI_UPDATE_STATE( 56, MSG, ABEF, CDGH );
1576
1577 MSG = _mm_alignr_epi8( MSG_2, MSG_1, 4); // MSG := ( W_56, W_55, W_54, W_53 )
1578 MSG_3 = _mm_add_epi32( MSG_3, MSG); // MSG_3 := MSG_3 + ( W_56, W_55, W_54, W_53 )
1579 MSG_3 = _mm_sha256msg2_epu32( MSG_3, MSG_2 ); // Calculate ( W_63, W_62, W_61, W_60 )
1580
1581 // Rounds 60 - 63
1582 SHANI_UPDATE_STATE( 60, MSG_3, ABEF, CDGH );
1583
1584 // Add the feed-forward
1585 ABEF = _mm_add_epi32( ABEF, ABEF_start );
1586 CDGH = _mm_add_epi32( CDGH, CDGH_start );
1587
1588 cbData -= 64;
1589 }
1590
1591 // Unpack the state registers and store them in the state
1592 FEBA = _mm_shuffle_epi32( ABEF, 0x1b );
1593 HGDC = _mm_shuffle_epi32( CDGH, 0x1b );
1594 DCBA = _mm_unpacklo_epi64( FEBA, HGDC ); // (D, C, B, A)
1595 HGFE = _mm_unpackhi_epi64( FEBA, HGDC ); // (H, G, F, E)
1596 _mm_storeu_si128 ( (__m128i *)&(pChain->H[0]), DCBA); // (D, C, B, A)
1597 _mm_storeu_si128 ( (__m128i *)&(pChain->H[4]), HGFE); // (H, G, F, E)
1598
1599 *pcbRemaining = cbData;
1600}
1601
1602#undef SHANI_UPDATE_STATE
1603#undef SHANI_MESSAGE_SCHEDULE
1604
1605#ifdef __clang__
1606#pragma clang attribute pop
1607#else
1608#pragma GCC pop_options
1609#endif
1610
1611#endif // SYMCRYPT_CPU_X86 | SYMCRYPT_CPU_AMD64
1612
1613#if SYMCRYPT_CPU_ARM64
1614/*
1615ARM64 has special SHA-256 instructions
1616
1617SHA256H and SHA256H2 implement 4 rounds of SHA-256. The inputs are two registers containing the 256-bit state,
1618and one register containing 128 bits of expanded message plus the round constants.
1619These instructions perform the same computation, but SHA256H returns the first half of the 256-bit result,
1620and SHA256H2 returns the second half of the 256-bit result.
1621
1622SHA256H( ABCDE, FGHIJ, W )
1623Where the least significant word of the ABCDE vector is A. The W vector contains W_i + K_i for the four rounds being computed.
1624
1625SHA256SU0 is the message schedule update function.
1626It takes 2 inputs and produces 1 output.
1627We describe the vectors for i=0,1,2,3
1628Inputs: [W_{t-16+i}], [W_{t-12+i}]
1629Output: [Sigma0(W_{t-15+i}) + W_{t-16+i}]
1630
1631SHA256SU1 is the second message schedule update function
1632Takes 3 inputs and produces 1 output
1633Input 1: Output of SHA256SU0: [Sigma0(W_{t-15+i}) + W_{t-16+i}]
1634Input 2:
1635Input 3: [W_{t-4+i}]
1636
1637*/
1638
1639#ifdef __clang__
1640#pragma clang attribute push (__attribute__((target("sha2"))), apply_to=function)
1641#else
1642#pragma GCC push_options
1643#pragma GCC target("sha2")
1644#endif
1645
1646#define vldq(_p) (*(__n128 *)(_p))
1647#define vstq(_p, _v) (*(__n128 *)(_p) = (_v) )
1648
1649VOID
1651SymCryptSha256AppendBlocks_instr(
1654 SIZE_T cbData,
1655 _Out_ SIZE_T * pcbRemaining )
1656{
1657 //
1658 // Armv8 has 32 Neon registers. We can use a lot of variables.
1659 // 16 for the constants, 4 for the message, 2 for the current state, 2 for the starting state,
1660 // total = 24 which leaves enough for some temp values
1661 //
1662 __n128 ABCD, ABCDstart;
1663 __n128 EFGH, EFGHstart;
1664 __n128 W0, W1, W2, W3;
1665 __n128 K0, K1, K2, K3, K4, K5, K6, K7, K8, K9, K10, K11, K12, K13, K14, K15;
1666
1667 __n128 Wr;
1668 __n128 t;
1669
1670 ABCD = ABCDstart = vldq( &pChain->H[0] );
1671 EFGH = EFGHstart = vldq( &pChain->H[4] );
1672
1673 K0 = vldq( &SymCryptSha256K[ 4 * 0 ] );
1674 K1 = vldq( &SymCryptSha256K[ 4 * 1 ] );
1675 K2 = vldq( &SymCryptSha256K[ 4 * 2 ] );
1676 K3 = vldq( &SymCryptSha256K[ 4 * 3 ] );
1677 K4 = vldq( &SymCryptSha256K[ 4 * 4 ] );
1678 K5 = vldq( &SymCryptSha256K[ 4 * 5 ] );
1679 K6 = vldq( &SymCryptSha256K[ 4 * 6 ] );
1680 K7 = vldq( &SymCryptSha256K[ 4 * 7 ] );
1681 K8 = vldq( &SymCryptSha256K[ 4 * 8 ] );
1682 K9 = vldq( &SymCryptSha256K[ 4 * 9 ] );
1683 K10 = vldq( &SymCryptSha256K[ 4 * 10 ] );
1684 K11 = vldq( &SymCryptSha256K[ 4 * 11 ] );
1685 K12 = vldq( &SymCryptSha256K[ 4 * 12 ] );
1686 K13 = vldq( &SymCryptSha256K[ 4 * 13 ] );
1687 K14 = vldq( &SymCryptSha256K[ 4 * 14 ] );
1688 K15 = vldq( &SymCryptSha256K[ 4 * 15 ] );
1689
1690 while( cbData >= 64 )
1691 {
1692 W0 = vrev32q_u8( vldq( &pbData[ 0] ) );
1693 W1 = vrev32q_u8( vldq( &pbData[16] ) );
1694 W2 = vrev32q_u8( vldq( &pbData[32] ) );
1695 W3 = vrev32q_u8( vldq( &pbData[48] ) );
1696
1697 //
1698 // The sha256h/sha256h2 instructions overwrite one of the two state input registers.
1699 // This implies we have to have a copy made of one of the input states.
1700 //
1701#define ROUNDOP {\
1702 t = ABCD;\
1703 ABCD = vsha256hq_u32 ( ABCD, EFGH, Wr );\
1704 EFGH = vsha256h2q_u32( EFGH, t, Wr );\
1705 }
1706
1707 Wr = vaddq_u32( W0, K0 );
1708 ROUNDOP;
1709 Wr = vaddq_u32( W1, K1 );
1710 ROUNDOP;
1711 Wr = vaddq_u32( W2, K2 );
1712 ROUNDOP;
1713 Wr = vaddq_u32( W3, K3 );
1714 ROUNDOP;
1715
1716 t = vsha256su0q_u32( W0, W1 );
1717 W0 = vsha256su1q_u32( t, W2, W3 );
1718 Wr = vaddq_u32( W0, K4 );
1719 ROUNDOP;
1720
1721 t = vsha256su0q_u32( W1, W2 );
1722 W1 = vsha256su1q_u32( t, W3, W0 );
1723 Wr = vaddq_u32( W1, K5 );
1724 ROUNDOP;
1725
1726 t = vsha256su0q_u32( W2, W3 );
1727 W2 = vsha256su1q_u32( t, W0, W1 );
1728 Wr = vaddq_u32( W2, K6 );
1729 ROUNDOP;
1730
1731 t = vsha256su0q_u32( W3, W0 );
1732 W3 = vsha256su1q_u32( t, W1, W2 );
1733 Wr = vaddq_u32( W3, K7 );
1734 ROUNDOP;
1735
1736
1737 t = vsha256su0q_u32( W0, W1 );
1738 W0 = vsha256su1q_u32( t, W2, W3 );
1739 Wr = vaddq_u32( W0, K8 );
1740 ROUNDOP;
1741
1742 t = vsha256su0q_u32( W1, W2 );
1743 W1 = vsha256su1q_u32( t, W3, W0 );
1744 Wr = vaddq_u32( W1, K9 );
1745 ROUNDOP;
1746
1747 t = vsha256su0q_u32( W2, W3 );
1748 W2 = vsha256su1q_u32( t, W0, W1 );
1749 Wr = vaddq_u32( W2, K10 );
1750 ROUNDOP;
1751
1752 t = vsha256su0q_u32( W3, W0 );
1753 W3 = vsha256su1q_u32( t, W1, W2 );
1754 Wr = vaddq_u32( W3, K11 );
1755 ROUNDOP;
1756
1757
1758 t = vsha256su0q_u32( W0, W1 );
1759 W0 = vsha256su1q_u32( t, W2, W3 );
1760 Wr = vaddq_u32( W0, K12 );
1761 ROUNDOP;
1762
1763 t = vsha256su0q_u32( W1, W2 );
1764 W1 = vsha256su1q_u32( t, W3, W0 );
1765 Wr = vaddq_u32( W1, K13 );
1766 ROUNDOP;
1767
1768 t = vsha256su0q_u32( W2, W3 );
1769 W2 = vsha256su1q_u32( t, W0, W1 );
1770 Wr = vaddq_u32( W2, K14 );
1771 ROUNDOP;
1772
1773 t = vsha256su0q_u32( W3, W0 );
1774 W3 = vsha256su1q_u32( t, W1, W2 );
1775 Wr = vaddq_u32( W3, K15 );
1776 ROUNDOP;
1777
1778 ABCDstart = ABCD = vaddq_u32( ABCDstart, ABCD );
1779 EFGHstart = EFGH = vaddq_u32( EFGHstart, EFGH );
1780
1781 pbData += 64;
1782 cbData -= 64;
1783#undef ROUNDOP
1784
1785 }
1786
1787 *pcbRemaining = cbData;
1788 vstq( &pChain->H[0], ABCD );
1789 vstq( &pChain->H[4], EFGH );
1790
1791 //
1792 // All our local variables should be in registers, so no way to wipe them.
1793 //
1794}
1795
1796#ifdef __clang__
1797#pragma clang attribute pop
1798#else
1799#pragma GCC pop_options
1800#endif
1801
1802#endif
1803
1804
1805
1806//
1807// Easy switch between different implementations
1808//
1809//FORCEINLINE
1810VOID
1815 SIZE_T cbData,
1816 _Out_ SIZE_T* pcbRemaining)
1817{
1818#if SYMCRYPT_CPU_AMD64
1819
1820 SYMCRYPT_EXTENDED_SAVE_DATA SaveData;
1821
1823 SymCryptSaveXmm(&SaveData) == SYMCRYPT_NO_ERROR)
1824 {
1825 SymCryptSha256AppendBlocks_shani(pChain, pbData, cbData, pcbRemaining);
1826
1827 SymCryptRestoreXmm(&SaveData);
1828 }
1829 // Temporarily disabling use of Ymm in SHA2
1830 // else if (SYMCRYPT_CPU_FEATURES_PRESENT(SYMCRYPT_CPU_FEATURE_AVX2 | SYMCRYPT_CPU_FEATURE_BMI2) &&
1831 // SymCryptSaveYmm(&SaveData) == SYMCRYPT_NO_ERROR)
1832 // {
1833 // //SymCryptSha256AppendBlocks_ul1(pChain, pbData, cbData, pcbRemaining);
1834 // //SymCryptSha256AppendBlocks_ymm_8blocks(pChain, pbData, cbData, pcbRemaining);
1835 // SymCryptSha256AppendBlocks_ymm_avx2_asm(pChain, pbData, cbData, pcbRemaining);
1836
1837 // SymCryptRestoreYmm(&SaveData);
1838 // }
1839 else if (SYMCRYPT_CPU_FEATURES_PRESENT(SYMCRYPT_CPU_FEATURE_SSSE3 | SYMCRYPT_CPU_FEATURE_BMI2) &&
1840 SymCryptSaveXmm(&SaveData) == SYMCRYPT_NO_ERROR)
1841 {
1842 //SymCryptSha256AppendBlocks_xmm_4blocks(pChain, pbData, cbData, pcbRemaining);
1844
1845 SymCryptRestoreXmm(&SaveData);
1846 }
1847 else
1848 {
1849 SymCryptSha256AppendBlocks_ul1( pChain, pbData, cbData, pcbRemaining );
1850 //SymCryptSha256AppendBlocks_ul2(pChain, pbData, cbData, pcbRemaining);
1851 }
1852#elif SYMCRYPT_CPU_X86
1853 SYMCRYPT_EXTENDED_SAVE_DATA SaveData;
1854
1855 if( SYMCRYPT_CPU_FEATURES_PRESENT( SYMCRYPT_CPU_FEATURES_FOR_SHANI_CODE | SYMCRYPT_CPU_FEATURE_SAVEXMM_NOFAIL ) &&
1856 SymCryptSaveXmm( &SaveData ) == SYMCRYPT_NO_ERROR )
1857 {
1858 SymCryptSha256AppendBlocks_shani( pChain, pbData, cbData, pcbRemaining );
1859 SymCryptRestoreXmm( &SaveData );
1860 }
1861 else if (SYMCRYPT_CPU_FEATURES_PRESENT(SYMCRYPT_CPU_FEATURE_SSSE3 | SYMCRYPT_CPU_FEATURE_BMI2)
1862 && SymCryptSaveXmm(&SaveData) == SYMCRYPT_NO_ERROR)
1863 {
1865 SymCryptRestoreXmm(&SaveData);
1866 }
1867 else {
1868 SymCryptSha256AppendBlocks_ul1( pChain, pbData, cbData, pcbRemaining );
1869 }
1870#elif SYMCRYPT_CPU_ARM64
1871 if( SYMCRYPT_CPU_FEATURES_PRESENT( SYMCRYPT_CPU_FEATURE_NEON_SHA256 ) )
1872 {
1873 SymCryptSha256AppendBlocks_instr( pChain, pbData, cbData, pcbRemaining );
1874 } else {
1875 SymCryptSha256AppendBlocks_ul1( pChain, pbData, cbData, pcbRemaining );
1876 }
1877#else
1878 SymCryptSha256AppendBlocks_ul1( pChain, pbData, cbData, pcbRemaining );
1879#endif
1880
1881 //SymCryptSha256AppendBlocks_ul2( pChain, pbData, cbData, pcbRemaining );
1882 //SymCryptSha256AppendBlocks_xmm1( pChain, pbData, cbData, pcbRemaining ); !!! Needs Save/restore logic
1883 //SymCryptSha256AppendBlocks_xmm2( pChain, pbData, cbData, pcbRemaining );
1884}
#define MSG
Definition: Mailslot.c:11
#define D(d)
Definition: builtin.c:4557
#define C(c)
Definition: builtin.c:4556
Definition: ehthrow.cxx:93
Definition: ehthrow.cxx:54
Definition: terminate.cpp:24
VOID SaveData(HWND hwndDlg)
Definition: volume.c:368
#define W(I)
#define A(row, col)
#define B(row, col)
static cab_ULONG checksum(const cab_UBYTE *data, cab_UWORD bytes, cab_ULONG csum)
Definition: fdi.c:353
static void cleanup(void)
Definition: main.c:1335
_ACRTIMP int __cdecl memcmp(const void *, const void *, size_t)
Definition: string.c:2807
void _mm_storeu_si128(__m128i_u *p, __m128i b)
Definition: emmintrin.h:1684
__m128i _mm_set_epi8(char b15, char b14, char b13, char b12, char b11, char b10, char b9, char b8, char b7, char b6, char b5, char b4, char b3, char b2, char b1, char b0)
Definition: emmintrin.h:1610
void _mm_store_si128(__m128i *p, __m128i b)
Definition: emmintrin.h:1679
__m128i _mm_load_si128(__m128i const *p)
Definition: emmintrin.h:1556
__m128i _mm_unpackhi_epi64(__m128i a, __m128i b)
Definition: emmintrin.h:1833
__m128i _mm_add_epi32(__m128i a, __m128i b)
Definition: emmintrin.h:1137
#define _mm_shuffle_epi32(a, imm)
Definition: emmintrin.h:1793
__m128i _mm_loadu_si128(__m128i_u const *p)
Definition: emmintrin.h:1561
__m128i _mm_unpacklo_epi64(__m128i a, __m128i b)
Definition: emmintrin.h:1873
GLint GLint GLint GLint GLint x
Definition: gl.h:1548
GLuint GLuint GLsizei GLenum type
Definition: gl.h:1545
GLdouble GLdouble GLdouble r
Definition: gl.h:2055
GLdouble GLdouble t
Definition: gl.h:2047
GLuint res
Definition: glext.h:9613
GLenum GLuint GLenum GLsizei const GLchar * buf
Definition: glext.h:7751
GLuint64EXT * result
Definition: glext.h:11304
#define C_ASSERT(e)
Definition: intsafe.h:73
#define memcpy(s1, s2, n)
Definition: mkisofs.h:878
#define _In_reads_bytes_(s)
Definition: no_sal2.h:170
#define _In_reads_(s)
Definition: no_sal2.h:168
#define _Inout_
Definition: no_sal2.h:162
#define _Out_writes_(s)
Definition: no_sal2.h:176
#define _Out_
Definition: no_sal2.h:160
#define _In_
Definition: no_sal2.h:158
#define _Out_writes_bytes_(s)
Definition: no_sal2.h:178
BYTE * PBYTE
Definition: pedump.c:66
#define T(num)
Definition: thunks.c:311
#define SYMCRYPT_BLOB_MAGIC
Definition: sc_lib.h:1077
VOID SYMCRYPT_CALL SymCryptSha256AppendBlocks_xmm_4blocks(_Inout_ SYMCRYPT_SHA256_CHAINING_STATE *pChain, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_ SIZE_T *pcbRemaining)
#define SYMCRYPT_CPU_FEATURES_FOR_SHANI_CODE
Definition: sc_lib.h:312
@ SymCryptBlobTypeSha224State
Definition: sc_lib.h:1071
@ SymCryptBlobTypeSha256State
Definition: sc_lib.h:1065
VOID SYMCRYPT_CALL SymCryptInjectError(PBYTE pbData, SIZE_T cbData)
FORCEINLINE VOID SYMCRYPT_CALL SymCryptUint32ToMsbFirst(_In_reads_(cuData) PCUINT32 puData, _Out_writes_(4 *cuData) PBYTE pbResult, SIZE_T cuData)
Definition: sc_lib.h:405
const BYTE SymCryptTestMsg3[3]
Definition: selftest.c:8
FORCEINLINE VOID SYMCRYPT_CALL SymCryptMsbFirstToUint32(_In_reads_(4 *cuResult) PCBYTE pbData, _Out_writes_(cuResult) PUINT32 puResult, SIZE_T cuResult)
Definition: sc_lib.h:422
VOID SYMCRYPT_CALL SymCryptSha256AppendBlocks_xmm_ssse3_asm(_Inout_ SYMCRYPT_SHA256_CHAINING_STATE *pChain, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_ SIZE_T *pcbRemaining)
VOID SYMCRYPT_CALL SymCryptSha256AppendBlocks(_Inout_ SYMCRYPT_SHA256_CHAINING_STATE *pChain, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_ SIZE_T *pcbRemaining)
Definition: sha256.c:1812
static const UINT32 sha224InitialState[8]
Definition: sha256.c:77
const SYMCRYPT_HASH SymCryptSha224Algorithm_default
Definition: sha256.c:19
static const UINT32 sha256InitialState[8]
Definition: sha256.c:88
SYMCRYPT_NOINLINE VOID SYMCRYPT_CALL SymCryptSha256Append(_Inout_ PSYMCRYPT_SHA256_STATE pState, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData)
Definition: sha256.c:171
const PCSYMCRYPT_HASH SymCryptSha256Algorithm
Definition: sha256.c:46
const SYMCRYPT_HASH SymCryptSha256Algorithm_default
Definition: sha256.c:32
#define LSIGMA1(x)
Definition: sha256.c:664
#define LSIGMA0(x)
Definition: sha256.c:647
VOID SYMCRYPT_CALL SymCryptSha256Selftest(void)
Definition: sha256.c:485
#define IROUND(r)
Definition: sha256.c:717
SYMCRYPT_ERROR SYMCRYPT_CALL SymCryptSha256StateImport(_Out_ PSYMCRYPT_SHA256_STATE pState, _In_reads_bytes_(SYMCRYPT_SHA256_STATE_EXPORT_SIZE) PCBYTE pbBlob)
Definition: sha256.c:453
SYMCRYPT_NOINLINE VOID SYMCRYPT_CALL SymCryptSha224Append(_Inout_ PSYMCRYPT_SHA224_STATE pState, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData)
Definition: sha256.c:247
const BYTE SymCryptSha224KATAnswer[28]
Definition: sha256.c:502
#define CSIGMA0(x)
Definition: sha256.c:612
VOID SYMCRYPT_CALL SymCryptSha256AppendBlocks_ul1(_Inout_ SYMCRYPT_SHA256_CHAINING_STATE *pChain, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_ SIZE_T *pcbRemaining)
Definition: sha256.c:739
VOID SYMCRYPT_CALL SymCryptSha256AppendBlocks_ul2(_Inout_ SYMCRYPT_SHA256_CHAINING_STATE *pChain, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_ SIZE_T *pcbRemaining)
Definition: sha256.c:832
#define DO_ROUND(a, b, c, d, t, r)
VOID SYMCRYPT_CALL SymCryptSha256StateExportCore(_In_ PCSYMCRYPT_SHA256_STATE pState, _Out_writes_bytes_(SYMCRYPT_SHA256_STATE_EXPORT_SIZE) PBYTE pbBlob, _In_ UINT32 type)
Definition: sha256.c:354
const PCSYMCRYPT_HASH SymCryptSha224Algorithm
Definition: sha256.c:45
SYMCRYPT_ERROR SYMCRYPT_CALL SymCryptSha256StateImportCore(_Out_ PSYMCRYPT_SHA256_STATE pState, _In_reads_bytes_(SYMCRYPT_SHA256_STATE_EXPORT_SIZE) PCBYTE pbBlob, _In_ UINT32 type)
Definition: sha256.c:411
VOID SYMCRYPT_CALL SymCryptSha224StateExport(_In_ PCSYMCRYPT_SHA224_STATE pState, _Out_writes_bytes_(SYMCRYPT_SHA256_STATE_EXPORT_SIZE) PBYTE pbBlob)
Definition: sha256.c:401
SYMCRYPT_NOINLINE VOID SYMCRYPT_CALL SymCryptSha256Result(_Inout_ PSYMCRYPT_SHA256_STATE pState, _Out_writes_(SYMCRYPT_SHA256_RESULT_SIZE) PBYTE pbResult)
Definition: sha256.c:262
SYMCRYPT_NOINLINE VOID SYMCRYPT_CALL SymCryptSha224Result(_Inout_ PSYMCRYPT_SHA224_STATE pState, _Out_writes_(SYMCRYPT_SHA224_RESULT_SIZE) PBYTE pbResult)
Definition: sha256.c:330
#define EXPAND(a, b, c, d, r)
SYMCRYPT_ERROR SYMCRYPT_CALL SymCryptSha224StateImport(_Out_ PSYMCRYPT_SHA224_STATE pState, _In_reads_bytes_(SYMCRYPT_SHA224_STATE_EXPORT_SIZE) PCBYTE pbBlob)
Definition: sha256.c:463
VOID SYMCRYPT_CALL SymCryptSha224Selftest(void)
Definition: sha256.c:511
VOID SYMCRYPT_CALL SymCryptSha256StateExport(_In_ PCSYMCRYPT_SHA256_STATE pState, _Out_writes_bytes_(SYMCRYPT_SHA256_STATE_EXPORT_SIZE) PBYTE pbBlob)
Definition: sha256.c:391
const BYTE SymCryptSha256KATAnswer[32]
Definition: sha256.c:476
#define FROUND(r16, rb)
Definition: sha256.c:727
SYMCRYPT_NOINLINE VOID SYMCRYPT_CALL SymCryptSha224Init(_Out_ PSYMCRYPT_SHA224_STATE pState)
Definition: sha256.c:148
#define CSIGMA1(x)
Definition: sha256.c:630
SYMCRYPT_NOINLINE VOID SYMCRYPT_CALL SymCryptSha256Init(_Out_ PSYMCRYPT_SHA256_STATE pState)
Definition: sha256.c:125
static const BYTE pbResult[]
Definition: polytest.cpp:36
Definition: image.c:229
#define SYMCRYPT_ASSERT(_x)
Definition: symcrypt.h:10807
VOID SYMCRYPT_CALL SymCryptSha224(_In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_writes_(SYMCRYPT_SHA224_RESULT_SIZE) PBYTE pbResult)
VOID SYMCRYPT_CALL SymCryptSha256StateCopy(_In_ PCSYMCRYPT_SHA256_STATE pSrc, _Out_ PSYMCRYPT_SHA256_STATE pDst)
VOID SYMCRYPT_CALL SymCryptMarvin32(_In_ PCSYMCRYPT_MARVIN32_EXPANDED_SEED pExpandedSeed, _In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_writes_(SYMCRYPT_MARVIN32_RESULT_SIZE) PBYTE pbResult)
Definition: marvin32.c:239
#define SYMCRYPT_SHA256_INPUT_BLOCK_SIZE
Definition: symcrypt.h:1223
#define SYMCRYPT_LOAD_MSBFIRST32(p)
Definition: symcrypt.h:303
FORCEINLINE VOID SYMCRYPT_CALL SymCryptWipeKnownSize(_Out_writes_bytes_(cbData) PVOID pbData, SIZE_T cbData)
#define SYMCRYPT_SHA224_INPUT_BLOCK_SIZE
Definition: symcrypt.h:1158
VOID SYMCRYPT_CALL SymCryptWipe(_Out_writes_bytes_(cbData) PVOID pbData, SIZE_T cbData)
Definition: libmain.c:137
VOID SYMCRYPT_CALL SymCryptSha256(_In_reads_(cbData) PCBYTE pbData, SIZE_T cbData, _Out_writes_(SYMCRYPT_SHA256_RESULT_SIZE) PBYTE pbResult)
_Analysis_noreturn_ VOID SYMCRYPT_CALL SymCryptFatal(UINT32 fatalCode)
#define SYMCRYPT_FORCE_WRITE32(_p, _v)
Definition: symcrypt.h:423
PCSYMCRYPT_MARVIN32_EXPANDED_SEED const SymCryptMarvin32DefaultSeed
Definition: marvin32.c:29
#define SYMCRYPT_SHA256_RESULT_SIZE
Definition: symcrypt.h:1222
VOID SYMCRYPT_CALL SymCryptSha224StateCopy(_In_ PCSYMCRYPT_SHA224_STATE pSrc, _Out_ PSYMCRYPT_SHA224_STATE pDst)
#define SYMCRYPT_SHA224_RESULT_SIZE
Definition: symcrypt.h:1157
#define SYMCRYPT_STORE_MSBFIRST64(p, v)
Definition: symcrypt.h:312
SYMCRYPT_ERROR
Definition: symcrypt.h:227
#define SYMCRYPT_ALIGN
#define SYMCRYPT_CALL
SYMCRYPT_SHA256_STATE
struct _SYMCRYPT_HASH SYMCRYPT_HASH
#define SYMCRYPT_FIELD_SIZE(type, field)
SYMCRYPT_SHA256_CHAINING_STATE
#define SYMCRYPT_CPU_FEATURES_PRESENT(x)
#define SYMCRYPT_ALIGN_AT(alignment)
SIZE_T bytesInBuffer
BYTE K2[16]
const SYMCRYPT_HASH * PCSYMCRYPT_HASH
PCBYTE PBYTE SIZE_T cbData
#define SYMCRYPT_SET_MAGIC(p)
SYMCRYPT_SHA224_STATE
#define SYMCRYPT_SHA256_STATE_EXPORT_SIZE
* PSYMCRYPT_SHA256_STATE
#define SYMCRYPT_FIELD_OFFSET(type, field)
PSYMCRYPT_COMMON_HASH_STATE pState
const SYMCRYPT_SHA256_STATE * PCSYMCRYPT_SHA256_STATE
const SYMCRYPT_SHA224_STATE * PCSYMCRYPT_SHA224_STATE
const BYTE * PCBYTE
* PSYMCRYPT_SHA224_STATE
PCBYTE pbData
BYTE K1[16]
#define SYMCRYPT_SHA224_STATE_EXPORT_SIZE
#define SYMCRYPT_CHECK_MAGIC(p)
struct sock * chain
Definition: tcpcore.h:1
TW_UINT32 TW_UINT16 TW_UINT16 MSG
Definition: twain.h:1829
ULONG_PTR SIZE_T
Definition: typedefs.h:80
uint32_t UINT32
Definition: typedefs.h:59
#define FORCEINLINE
Definition: wdftypes.h:67
#define round(x)
Definition: opentype.c:51
unsigned char BYTE
Definition: xxhash.c:193
#define const
Definition: zconf.h:233