Coverage Report

Created: 2026-08-30 07:14

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/openssl/crypto/sha/sha512.c
Line
Count
Source
1
/*
2
 * Copyright 2004-2026 The OpenSSL Project Authors. All Rights Reserved.
3
 *
4
 * Licensed under the Apache License 2.0 (the "License").  You may not use
5
 * this file except in compliance with the License.  You can obtain a copy
6
 * in the file LICENSE in the source distribution or at
7
 * https://www.openssl.org/source/license.html
8
 */
9
10
/*
11
 * SHA512 low level APIs are deprecated for public use, but still ok for
12
 * internal use.
13
 */
14
#include "internal/deprecated.h"
15
16
#include <stdio.h>
17
#include <openssl/opensslconf.h>
18
#include <openssl/byteorder.h>
19
/*-
20
 * IMPLEMENTATION NOTES.
21
 *
22
 * As you might have noticed, 32-bit hash algorithms:
23
 *
24
 * - permit SHA_LONG to be wider than 32-bit
25
 * - optimized versions implement two transform functions: one operating
26
 *   on [aligned] data in host byte order, and one operating on data in input
27
 *   stream byte order;
28
 * - share common byte-order neutral collector and padding function
29
 *   implementations, crypto/md32_common.inc;
30
 *
31
 * Neither of the above applies to this SHA-512 implementation. Reasons
32
 * [in reverse order] are:
33
 *
34
 * - it's the only 64-bit hash algorithm for the moment of this writing,
35
 *   there is no need for common collector/padding implementation [yet];
36
 * - by supporting only one transform function [which operates on
37
 *   *aligned* data in input stream byte order, big-endian in this case]
38
 *   we minimize burden of maintenance in two ways: a) collector/padding
39
 *   function is simpler; b) only one transform function to stare at;
40
 * - SHA_LONG64 is required to be exactly 64-bit in order to be able to
41
 *   apply a number of optimizations to mitigate potential performance
42
 *   penalties caused by previous design decision;
43
 *
44
 * Caveat lector.
45
 *
46
 * Implementation relies on the fact that "long long" is 64-bit on
47
 * both 32- and 64-bit platforms. If some compiler vendor comes up
48
 * with 128-bit long long, adjustment to sha.h would be required.
49
 * As this implementation relies on 64-bit integer type, it's totally
50
 * inappropriate for platforms which don't support it, most notably
51
 * 16-bit platforms.
52
 */
53
#include <stdlib.h>
54
#include <string.h>
55
56
#include <openssl/crypto.h>
57
#include <openssl/sha.h>
58
#include <openssl/opensslv.h>
59
60
#include "internal/cryptlib.h"
61
#include "crypto/sha.h"
62
63
#if defined(__i386) || defined(__i386__) || defined(_M_IX86) || defined(__x86_64) || defined(_M_AMD64) || defined(_M_X64) || defined(__s390__) || defined(__s390x__) || defined(__aarch64__) || defined(__e2k__) || defined(SHA512_ASM)
64
#define SHA512_BLOCK_CAN_MANAGE_UNALIGNED_DATA
65
#endif
66
67
#if (defined(_WIN32) || defined(_WIN64)) && !defined(__MINGW32__)
68
#define U64(C) C##UI64
69
#elif defined(__arch64__)
70
#define U64(C) C##UL
71
#else
72
63.8k
#define U64(C) C##ULL
73
#endif
74
75
int SHA512_Update_thunk(void *cp, const unsigned char *data, size_t len);
76
77
int sha512_224_init(SHA512_CTX *c)
78
2
{
79
2
    c->h[0] = U64(0x8c3d37c819544da2);
80
2
    c->h[1] = U64(0x73e1996689dcd4d6);
81
2
    c->h[2] = U64(0x1dfab7ae32ff9c82);
82
2
    c->h[3] = U64(0x679dd514582f9fcf);
83
2
    c->h[4] = U64(0x0f6d2b697bd44da8);
84
2
    c->h[5] = U64(0x77e36f7304c48942);
85
2
    c->h[6] = U64(0x3f9d85a86a1d36c8);
86
2
    c->h[7] = U64(0x1112e6ad91d692a1);
87
88
2
    c->Nl = 0;
89
2
    c->Nh = 0;
90
2
    c->num = 0;
91
2
    c->md_len = SHA224_DIGEST_LENGTH;
92
2
    return 1;
93
2
}
94
95
int sha512_256_init(SHA512_CTX *c)
96
2
{
97
2
    c->h[0] = U64(0x22312194fc2bf72c);
98
2
    c->h[1] = U64(0x9f555fa3c84c64c2);
99
2
    c->h[2] = U64(0x2393b86b6f53b151);
100
2
    c->h[3] = U64(0x963877195940eabd);
101
2
    c->h[4] = U64(0x96283ee2a88effe3);
102
2
    c->h[5] = U64(0xbe5e1e2553863992);
103
2
    c->h[6] = U64(0x2b0199fc2c85b8aa);
104
2
    c->h[7] = U64(0x0eb72ddc81c52ca2);
105
106
2
    c->Nl = 0;
107
2
    c->Nh = 0;
108
2
    c->num = 0;
109
2
    c->md_len = SHA256_DIGEST_LENGTH;
110
2
    return 1;
111
2
}
112
113
int SHA384_Init(SHA512_CTX *c)
114
42
{
115
42
    c->h[0] = U64(0xcbbb9d5dc1059ed8);
116
42
    c->h[1] = U64(0x629a292a367cd507);
117
42
    c->h[2] = U64(0x9159015a3070dd17);
118
42
    c->h[3] = U64(0x152fecd8f70e5939);
119
42
    c->h[4] = U64(0x67332667ffc00b31);
120
42
    c->h[5] = U64(0x8eb44a8768581511);
121
42
    c->h[6] = U64(0xdb0c2e0d64f98fa7);
122
42
    c->h[7] = U64(0x47b5481dbefa4fa4);
123
124
42
    c->Nl = 0;
125
42
    c->Nh = 0;
126
42
    c->num = 0;
127
42
    c->md_len = SHA384_DIGEST_LENGTH;
128
42
    return 1;
129
42
}
130
131
int SHA512_Init(SHA512_CTX *c)
132
7.04k
{
133
7.04k
    c->h[0] = U64(0x6a09e667f3bcc908);
134
7.04k
    c->h[1] = U64(0xbb67ae8584caa73b);
135
7.04k
    c->h[2] = U64(0x3c6ef372fe94f82b);
136
7.04k
    c->h[3] = U64(0xa54ff53a5f1d36f1);
137
7.04k
    c->h[4] = U64(0x510e527fade682d1);
138
7.04k
    c->h[5] = U64(0x9b05688c2b3e6c1f);
139
7.04k
    c->h[6] = U64(0x1f83d9abfb41bd6b);
140
7.04k
    c->h[7] = U64(0x5be0cd19137e2179);
141
142
7.04k
    c->Nl = 0;
143
7.04k
    c->Nh = 0;
144
7.04k
    c->num = 0;
145
7.04k
    c->md_len = SHA512_DIGEST_LENGTH;
146
7.04k
    return 1;
147
7.04k
}
148
149
#ifndef SHA512_ASM
150
static
151
#else
152
#ifdef INCLUDE_C_SHA512
153
void sha512_block_data_order_c(SHA512_CTX *ctx, const void *in, size_t num);
154
#endif
155
#endif
156
    void sha512_block_data_order(SHA512_CTX *ctx, const void *in, size_t num);
157
158
#define OUTPUT_RESULT(md, len)      \
159
31.8k
    for (n = 0; n < (len / 8); n++) \
160
28.3k
    md = OPENSSL_store_u64_be(md, (uint64_t)c->h[n])
161
162
int SHA512_Final(unsigned char *out, SHA512_CTX *c)
163
3.54k
{
164
3.54k
    unsigned char *p = (unsigned char *)c->u.p;
165
3.54k
    size_t n = c->num;
166
167
3.54k
    p[n] = 0x80; /* There always is a room for one */
168
3.54k
    n++;
169
3.54k
    if (n > (sizeof(c->u) - 16)) {
170
42
        memset(p + n, 0, sizeof(c->u) - n);
171
42
        n = 0;
172
42
        sha512_block_data_order(c, p, 1);
173
42
    }
174
175
3.54k
    memset(p + n, 0, sizeof(c->u) - 16 - n);
176
#ifdef B_ENDIAN
177
    c->u.d[SHA_LBLOCK - 2] = c->Nh;
178
    c->u.d[SHA_LBLOCK - 1] = c->Nl;
179
#else
180
3.54k
    uint8_t *cu = p + sizeof(c->u) - 16;
181
182
3.54k
    cu = OPENSSL_store_u64_be(cu, (uint64_t)c->Nh);
183
3.54k
    cu = OPENSSL_store_u64_be(cu, (uint64_t)c->Nl);
184
3.54k
#endif
185
186
3.54k
    sha512_block_data_order(c, p, 1);
187
188
3.54k
    if (out == NULL)
189
0
        return 0;
190
191
    /* Let compiler decide if it's appropriate to unroll... */
192
3.54k
    switch (c->md_len) {
193
0
    case SHA256_192_DIGEST_LENGTH:
194
0
        OUTPUT_RESULT(out, SHA256_192_DIGEST_LENGTH);
195
0
        break;
196
1
    case SHA256_DIGEST_LENGTH:
197
1
        OUTPUT_RESULT(out, SHA256_DIGEST_LENGTH);
198
1
        break;
199
21
    case SHA384_DIGEST_LENGTH:
200
21
        OUTPUT_RESULT(out, SHA384_DIGEST_LENGTH);
201
21
        break;
202
3.52k
    case SHA512_DIGEST_LENGTH:
203
3.52k
        OUTPUT_RESULT(out, SHA512_DIGEST_LENGTH);
204
3.52k
        break;
205
1
    case SHA224_DIGEST_LENGTH: {
206
1
        OUTPUT_RESULT(out, SHA224_DIGEST_LENGTH);
207
        /*
208
         * For 224 bits, there are four bytes left over that have to be
209
         * processed separately.
210
         */
211
1
        {
212
1
            SHA_LONG64 t = c->h[SHA224_DIGEST_LENGTH / 8];
213
214
1
            *(out++) = (unsigned char)(t >> 56);
215
1
            *(out++) = (unsigned char)(t >> 48);
216
1
            *(out++) = (unsigned char)(t >> 40);
217
1
            *(out++) = (unsigned char)(t >> 32);
218
1
        }
219
1
        break;
220
0
    }
221
    /* ... as well as make sure md_len is not abused. */
222
0
    default:
223
0
        return 0;
224
3.54k
    }
225
226
3.54k
    return 1;
227
3.54k
}
228
229
int SHA384_Final(unsigned char *md, SHA512_CTX *c)
230
21
{
231
21
    return SHA512_Final(md, c);
232
21
}
233
234
int SHA512_Update_thunk(void *cp, const unsigned char *data, size_t len)
235
7.09k
{
236
7.09k
    SHA512_CTX *c = (SHA512_CTX *)cp;
237
7.09k
    SHA_LONG64 l;
238
7.09k
    unsigned char *p = c->u.p;
239
240
7.09k
    if (len == 0)
241
0
        return 1;
242
243
7.09k
    l = (c->Nl + (((SHA_LONG64)len) << 3)) & U64(0xffffffffffffffff);
244
7.09k
    if (l < c->Nl)
245
0
        c->Nh++;
246
7.09k
    if (sizeof(len) >= 8)
247
7.09k
        c->Nh += (((SHA_LONG64)len) >> 61);
248
7.09k
    c->Nl = l;
249
250
7.09k
    if (c->num != 0) {
251
0
        size_t n = sizeof(c->u) - c->num;
252
253
0
        if (len < n) {
254
0
            memcpy(p + c->num, data, len), c->num += (unsigned int)len;
255
0
            return 1;
256
0
        } else {
257
0
            memcpy(p + c->num, data, n), c->num = 0;
258
0
            len -= n, data += n;
259
0
            sha512_block_data_order(c, p, 1);
260
0
        }
261
0
    }
262
263
7.09k
    if (len >= sizeof(c->u)) {
264
#ifndef SHA512_BLOCK_CAN_MANAGE_UNALIGNED_DATA
265
        if ((size_t)data % sizeof(c->u.d[0]) != 0)
266
            while (len >= sizeof(c->u))
267
                memcpy(p, data, sizeof(c->u)),
268
                    sha512_block_data_order(c, p, 1),
269
                    len -= sizeof(c->u), data += sizeof(c->u);
270
        else
271
#endif
272
7.09k
            sha512_block_data_order(c, data, len / sizeof(c->u)),
273
7.09k
                data += len, len %= sizeof(c->u), data -= len;
274
7.09k
    }
275
276
7.09k
    if (len != 0)
277
82
        memcpy(p, data, len), c->num = (int)len;
278
279
7.09k
    return 1;
280
7.09k
}
281
282
int SHA512_Update(SHA512_CTX *c, const void *_data, size_t len)
283
0
{
284
0
    return SHA512_Update_thunk((void *)c, (const unsigned char *)_data, len);
285
0
}
286
287
int SHA384_Update(SHA512_CTX *c, const void *data, size_t len)
288
0
{
289
0
    return SHA512_Update_thunk((void *)c, (const unsigned char *)data, len);
290
0
}
291
292
void SHA512_Transform(SHA512_CTX *c, const unsigned char *data)
293
0
{
294
#ifndef SHA512_BLOCK_CAN_MANAGE_UNALIGNED_DATA
295
    if ((size_t)data % sizeof(c->u.d[0]) != 0)
296
        memcpy(c->u.p, data, sizeof(c->u.p)), data = c->u.p;
297
#endif
298
0
    sha512_block_data_order(c, data, 1);
299
0
}
300
301
#if !defined(SHA512_ASM) || defined(INCLUDE_C_SHA512)
302
static const SHA_LONG64 K512[80] = {
303
    U64(0x428a2f98d728ae22), U64(0x7137449123ef65cd),
304
    U64(0xb5c0fbcfec4d3b2f), U64(0xe9b5dba58189dbbc),
305
    U64(0x3956c25bf348b538), U64(0x59f111f1b605d019),
306
    U64(0x923f82a4af194f9b), U64(0xab1c5ed5da6d8118),
307
    U64(0xd807aa98a3030242), U64(0x12835b0145706fbe),
308
    U64(0x243185be4ee4b28c), U64(0x550c7dc3d5ffb4e2),
309
    U64(0x72be5d74f27b896f), U64(0x80deb1fe3b1696b1),
310
    U64(0x9bdc06a725c71235), U64(0xc19bf174cf692694),
311
    U64(0xe49b69c19ef14ad2), U64(0xefbe4786384f25e3),
312
    U64(0x0fc19dc68b8cd5b5), U64(0x240ca1cc77ac9c65),
313
    U64(0x2de92c6f592b0275), U64(0x4a7484aa6ea6e483),
314
    U64(0x5cb0a9dcbd41fbd4), U64(0x76f988da831153b5),
315
    U64(0x983e5152ee66dfab), U64(0xa831c66d2db43210),
316
    U64(0xb00327c898fb213f), U64(0xbf597fc7beef0ee4),
317
    U64(0xc6e00bf33da88fc2), U64(0xd5a79147930aa725),
318
    U64(0x06ca6351e003826f), U64(0x142929670a0e6e70),
319
    U64(0x27b70a8546d22ffc), U64(0x2e1b21385c26c926),
320
    U64(0x4d2c6dfc5ac42aed), U64(0x53380d139d95b3df),
321
    U64(0x650a73548baf63de), U64(0x766a0abb3c77b2a8),
322
    U64(0x81c2c92e47edaee6), U64(0x92722c851482353b),
323
    U64(0xa2bfe8a14cf10364), U64(0xa81a664bbc423001),
324
    U64(0xc24b8b70d0f89791), U64(0xc76c51a30654be30),
325
    U64(0xd192e819d6ef5218), U64(0xd69906245565a910),
326
    U64(0xf40e35855771202a), U64(0x106aa07032bbd1b8),
327
    U64(0x19a4c116b8d2d0c8), U64(0x1e376c085141ab53),
328
    U64(0x2748774cdf8eeb99), U64(0x34b0bcb5e19b48a8),
329
    U64(0x391c0cb3c5c95a63), U64(0x4ed8aa4ae3418acb),
330
    U64(0x5b9cca4f7763e373), U64(0x682e6ff3d6b2b8a3),
331
    U64(0x748f82ee5defb2fc), U64(0x78a5636f43172f60),
332
    U64(0x84c87814a1f0ab72), U64(0x8cc702081a6439ec),
333
    U64(0x90befffa23631e28), U64(0xa4506cebde82bde9),
334
    U64(0xbef9a3f7b2c67915), U64(0xc67178f2e372532b),
335
    U64(0xca273eceea26619c), U64(0xd186b8c721c0c207),
336
    U64(0xeada7dd6cde0eb1e), U64(0xf57d4f7fee6ed178),
337
    U64(0x06f067aa72176fba), U64(0x0a637dc5a2c898a6),
338
    U64(0x113f9804bef90dae), U64(0x1b710b35131c471b),
339
    U64(0x28db77f523047d84), U64(0x32caab7b40c72493),
340
    U64(0x3c9ebe0a15c9bebc), U64(0x431d67c49c100d4c),
341
    U64(0x4cc5d4becb3e42b6), U64(0x597f299cfc657e2a),
342
    U64(0x5fcb6fab3ad6faec), U64(0x6c44198c4a475817)
343
};
344
345
#ifndef PEDANTIC
346
#if defined(__GNUC__) && !defined(OPENSSL_NO_ASM) && !defined(OPENSSL_NO_INLINE_ASM)
347
#if defined(__x86_64) || defined(__x86_64__)
348
#define ROTR(a, n) ({ SHA_LONG64 ret;             \
349
                                asm ("rorq %1,%0"       \
350
                                : "=r"(ret)             \
351
                                : "J"(n),"0"(a)         \
352
                                : "cc"); ret; })
353
#if !defined(B_ENDIAN)
354
#define PULL64(x) ({ SHA_LONG64 ret=*((const SHA_LONG64 *)(&(x)));  \
355
                                asm ("bswapq    %0"             \
356
                                : "=r"(ret)                     \
357
                                : "0"(ret)); ret; })
358
#endif
359
#elif (defined(__i386) || defined(__i386__)) && !defined(B_ENDIAN)
360
#if defined(I386_ONLY)
361
#define PULL64(x) ({ const unsigned int *p=(const unsigned int *)(&(x));\
362
                          unsigned int hi=p[0],lo=p[1];          \
363
                                asm("xchgb %%ah,%%al;xchgb %%dh,%%dl;"\
364
                                    "roll $16,%%eax; roll $16,%%edx; "\
365
                                    "xchgb %%ah,%%al;xchgb %%dh,%%dl;"\
366
                                : "=a"(lo),"=d"(hi)             \
367
                                : "0"(lo),"1"(hi) : "cc");      \
368
                                ((SHA_LONG64)hi)<<32|lo; })
369
#else
370
#define PULL64(x) ({ const unsigned int *p=(const unsigned int *)(&(x));\
371
                          unsigned int hi=p[0],lo=p[1];         \
372
                                asm ("bswapl %0; bswapl %1;"    \
373
                                : "=r"(lo),"=r"(hi)             \
374
                                : "0"(lo),"1"(hi));             \
375
                                ((SHA_LONG64)hi)<<32|lo; })
376
#endif
377
#elif (defined(_ARCH_PPC) && defined(__64BIT__)) || defined(_ARCH_PPC64)
378
#define ROTR(a, n) ({ SHA_LONG64 ret;             \
379
                                asm ("rotrdi %0,%1,%2"  \
380
                                : "=r"(ret)             \
381
                                : "r"(a),"K"(n)); ret; })
382
#elif defined(__aarch64__)
383
#define ROTR(a, n) ({ SHA_LONG64 ret;             \
384
                                asm ("ror %0,%1,%2"     \
385
                                : "=r"(ret)             \
386
                                : "r"(a),"I"(n)); ret; })
387
#if defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
388
#define PULL64(x) ({ SHA_LONG64 ret;                     \
389
                                asm ("rev       %0,%1"          \
390
                                : "=r"(ret)                     \
391
                                : "r"(*((const SHA_LONG64 *)(&(x))))); ret; })
392
#endif
393
#elif (defined(__riscv_zbkb) || defined(__riscv_zbb)) && __riscv_xlen == 32
394
#define PULL64(x) ({ SHA_LONG64 ret;                                        \
395
                        unsigned int *r = (unsigned int *)(&(ret));             \
396
                        const unsigned int *p = (const unsigned int *)(&(x));   \
397
                        asm ("rev8 %0, %1"                                      \
398
                        : "=r"(r[0])                                            \
399
                        : "r" (p[1]));                                          \
400
                        asm ("rev8 %0, %1"                                      \
401
                        : "=r"(r[1])                                            \
402
                        : "r" (p[0])); ret; })
403
#elif (defined(__riscv_zbkb) || defined(__riscv_zbb)) && __riscv_xlen == 64
404
#define PULL64(x) ({ SHA_LONG64 ret;    \
405
                        asm ("rev8 %0, %1"  \
406
                        : "=r"(ret)         \
407
                        : "r"(x)); ret; })
408
#elif defined(__e2k__)
409
#include <x86gprintrin.h>
410
#define PULL64(x) __builtin_bswap64(x)
411
#define ROTR(x, s) ((s) > 48 ? __rolq((x), 64 - (s)) \
412
                             : __rorq((x), (s)))
413
#endif
414
#if defined(__riscv_zknh) && __riscv_xlen == 32
415
#define Sigma0(x) ({ SHA_LONG64 ret; unsigned int *r = (unsigned int *)(&(ret));    \
416
                        const unsigned int *p = (const unsigned int *)(&(x));           \
417
                        asm ("sha512sum0r %0, %1, %2"                                   \
418
                        : "=r"(r[0])                                                    \
419
                        : "r" (p[0]), "r" (p[1]));                                      \
420
                        asm ("sha512sum0r %0, %2, %1"                                   \
421
                        : "=r"(r[1])                                                    \
422
                        : "r" (p[0]), "r" (p[1])); ret; })
423
#define Sigma1(x) ({ SHA_LONG64 ret; unsigned int *r = (unsigned int *)(&(ret));    \
424
                        const unsigned int *p = (const unsigned int *)(&(x));           \
425
                        asm ("sha512sum1r %0, %1, %2"                                   \
426
                        : "=r"(r[0])                                                    \
427
                        : "r" (p[0]), "r" (p[1]));                                      \
428
                        asm ("sha512sum1r %0, %2, %1"                                   \
429
                        : "=r"(r[1])                                                    \
430
                        : "r" (p[0]), "r" (p[1])); ret; })
431
#define sigma0(x) ({ SHA_LONG64 ret; unsigned int *r = (unsigned int *)(&(ret));    \
432
                        const unsigned int *p = (const unsigned int *)(&(x));           \
433
                        asm ("sha512sig0l %0, %1, %2"                                   \
434
                        : "=r"(r[0])                                                    \
435
                        : "r" (p[0]), "r" (p[1]));                                      \
436
                        asm ("sha512sig0h %0, %2, %1"                                   \
437
                        : "=r"(r[1])                                                    \
438
                        : "r" (p[0]), "r" (p[1])); ret; })
439
#define sigma1(x) ({ SHA_LONG64 ret; unsigned int *r = (unsigned int *)(&(ret));    \
440
                        const unsigned int *p = (const unsigned int *)(&(x));           \
441
                        asm ("sha512sig1l %0, %1, %2"                                   \
442
                        : "=r"(r[0])                                                    \
443
                        : "r" (p[0]), "r" (p[1]));                                      \
444
                        asm ("sha512sig1h %0, %2, %1"                                   \
445
                        : "=r"(r[1])                                                    \
446
                        : "r" (p[0]), "r" (p[1])); ret; })
447
#elif defined(__riscv_zknh) && __riscv_xlen == 64
448
#define Sigma0(x) ({ SHA_LONG64 ret;            \
449
                        asm ("sha512sum0 %0, %1"    \
450
                        : "=r"(ret)                 \
451
                        : "r"(x)); ret; })
452
#define Sigma1(x) ({ SHA_LONG64 ret;            \
453
                        asm ("sha512sum1 %0, %1"    \
454
                        : "=r"(ret)                 \
455
                        : "r"(x)); ret; })
456
#define sigma0(x) ({ SHA_LONG64 ret;            \
457
                        asm ("sha512sig0 %0, %1"    \
458
                        : "=r"(ret)                 \
459
                        : "r"(x)); ret; })
460
#define sigma1(x) ({ SHA_LONG64 ret;            \
461
                        asm ("sha512sig1 %0, %1"    \
462
                        : "=r"(ret)                 \
463
                        : "r"(x)); ret; })
464
#endif
465
#if (defined(__riscv_zbt) || defined(__riscv_zpn)) && __riscv_xlen == 32
466
#define Ch(x, y, z) ({  SHA_LONG64 ret; unsigned int *r = (unsigned int *)(&(ret));   \
467
                        const unsigned int *xp = (const unsigned int *)(&(x));          \
468
                        const unsigned int *yp = (const unsigned int *)(&(y));          \
469
                        const unsigned int *zp = (const unsigned int *)(&(z));          \
470
                        asm (".insn r4 0x33, 1, 0x3, %0, %2, %1, %3\n\t"                \
471
                        : "=r"(r[0])                                                    \
472
                        : "r"(xp[0]), "r"(yp[0]), "r"(zp[0]));                          \
473
                        asm (".insn r4 0x33, 1, 0x3, %0, %2, %1, %3\n\t"                \
474
                        : "=r"(r[1])                                                    \
475
                        : "r"(xp[1]), "r"(yp[1]), "r"(zp[1])); ret; })
476
#define Maj(x, y, z) ({ SHA_LONG64 ret; unsigned int *r = (unsigned int *)(&(ret));   \
477
                        const unsigned int *xp = (const unsigned int *)(&(x));          \
478
                        const unsigned int *yp = (const unsigned int *)(&(y));          \
479
                        const unsigned int *zp = (const unsigned int *)(&(z));          \
480
                        asm (".insn r4 0x33, 1, 0x3, %0, %2, %1, %3\n\t"                \
481
                        : "=r"(r[0])                                                    \
482
                        : "r"(xp[0]^zp[0]), "r"(yp[0]), "r"(zp[0]));                    \
483
                        asm (".insn r4 0x33, 1, 0x3, %0, %2, %1, %3\n\t"                \
484
                        : "=r"(r[1])                                                    \
485
                        : "r"(xp[1]^zp[1]), "r"(yp[1]), "r"(zp[1])); ret; })
486
#elif (defined(__riscv_zbt) || defined(__riscv_zpn)) && __riscv_xlen == 64
487
#define Ch(x, y, z) ({  SHA_LONG64 ret;                           \
488
                        asm (".insn r4 0x33, 1, 0x3, %0, %2, %1, %3"\
489
                        : "=r"(ret)                                 \
490
                        : "r"(x), "r"(y), "r"(z)); ret; })
491
#define Maj(x, y, z) ({ SHA_LONG64 ret;                           \
492
                        asm (".insn r4 0x33, 1, 0x3, %0, %2, %1, %3"\
493
                        : "=r"(ret)                                 \
494
                        : "r"(x^z), "r"(y), "r"(x)); ret; })
495
#elif defined(__e2k__) && __iset__ >= 5
496
#define sigma0(x) (__builtin_e2k_plog(0x96, ROTR((x), 1), ROTR((x), 8), ((x) >> 7)))
497
#define sigma1(x) (__builtin_e2k_plog(0x96, ROTR((x), 19), ROTR((x), 61), ((x) >> 6)))
498
#endif
499
#elif defined(_MSC_VER)
500
#if defined(_WIN64) /* applies to both IA-64 and AMD64 */
501
#pragma intrinsic(_rotr64)
502
#define ROTR(a, n) _rotr64((a), n)
503
#endif
504
#if defined(_M_IX86) && !defined(OPENSSL_NO_ASM) && !defined(OPENSSL_NO_INLINE_ASM)
505
#if defined(I386_ONLY)
506
static SHA_LONG64 __fastcall __pull64be(const void *x)
507
{
508
    _asm mov edx, [ecx + 0] _asm mov eax, [ecx + 4] _asm xchg dh, dl _asm xchg ah, al _asm rol edx, 16 _asm rol eax, 16 _asm xchg dh, dl _asm xchg ah, al
509
}
510
#else
511
static SHA_LONG64 __fastcall __pull64be(const void *x) {
512
    _asm mov edx, [ecx + 0] _asm mov eax, [ecx + 4] _asm bswap edx _asm bswap eax
513
}
514
#endif
515
#define PULL64(x) __pull64be(&(x))
516
#endif
517
#endif
518
#endif
519
#ifndef PULL64
520
427M
#define B(x, j) (((SHA_LONG64)(*(((const unsigned char *)(&x)) + j))) << ((7 - j) * 8))
521
53.4M
#define PULL64(x) (B(x, 0) | B(x, 1) | B(x, 2) | B(x, 3) | B(x, 4) | B(x, 5) | B(x, 6) | B(x, 7))
522
#endif
523
#ifndef ROTR
524
2.45G
#define ROTR(x, s) (((x) >> s) | (x) << (64 - s))
525
#endif
526
#ifndef Sigma0
527
267M
#define Sigma0(x) (ROTR((x), 28) ^ ROTR((x), 34) ^ ROTR((x), 39))
528
#endif
529
#ifndef Sigma1
530
267M
#define Sigma1(x) (ROTR((x), 14) ^ ROTR((x), 18) ^ ROTR((x), 41))
531
#endif
532
#ifndef sigma0
533
213M
#define sigma0(x) (ROTR((x), 1) ^ ROTR((x), 8) ^ ((x) >> 7))
534
#endif
535
#ifndef sigma1
536
213M
#define sigma1(x) (ROTR((x), 19) ^ ROTR((x), 61) ^ ((x) >> 6))
537
#endif
538
#ifndef Ch
539
267M
#define Ch(x, y, z) (((x) & (y)) ^ ((~(x)) & (z)))
540
#endif
541
#ifndef Maj
542
267M
#define Maj(x, y, z) (((x) & (y)) ^ ((x) & (z)) ^ ((y) & (z)))
543
#endif
544
545
#if defined(__i386) || defined(__i386__) || defined(_M_IX86)
546
/*
547
 * This code should give better results on 32-bit CPU with less than
548
 * ~24 registers, both size and performance wise...
549
 */
550
551
static void sha512_block_data_order(SHA512_CTX *ctx, const void *in,
552
    size_t num)
553
{
554
    const SHA_LONG64 *W = in;
555
    SHA_LONG64 A, E, T;
556
    SHA_LONG64 X[9 + 80], *F;
557
    int i;
558
559
    while (num--) {
560
561
        F = X + 80;
562
        A = ctx->h[0];
563
        F[1] = ctx->h[1];
564
        F[2] = ctx->h[2];
565
        F[3] = ctx->h[3];
566
        E = ctx->h[4];
567
        F[5] = ctx->h[5];
568
        F[6] = ctx->h[6];
569
        F[7] = ctx->h[7];
570
571
        for (i = 0; i < 16; i++, F--) {
572
#ifdef B_ENDIAN
573
            T = W[i];
574
#else
575
            T = PULL64(W[i]);
576
#endif
577
            F[0] = A;
578
            F[4] = E;
579
            F[8] = T;
580
            T += F[7] + Sigma1(E) + Ch(E, F[5], F[6]) + K512[i];
581
            E = F[3] + T;
582
            A = T + Sigma0(A) + Maj(A, F[1], F[2]);
583
        }
584
585
        for (; i < 80; i++, F--) {
586
            T = sigma0(F[8 + 16 - 1]);
587
            T += sigma1(F[8 + 16 - 14]);
588
            T += F[8 + 16] + F[8 + 16 - 9];
589
590
            F[0] = A;
591
            F[4] = E;
592
            F[8] = T;
593
            T += F[7] + Sigma1(E) + Ch(E, F[5], F[6]) + K512[i];
594
            E = F[3] + T;
595
            A = T + Sigma0(A) + Maj(A, F[1], F[2]);
596
        }
597
598
        ctx->h[0] += A;
599
        ctx->h[1] += F[1];
600
        ctx->h[2] += F[2];
601
        ctx->h[3] += F[3];
602
        ctx->h[4] += E;
603
        ctx->h[5] += F[5];
604
        ctx->h[6] += F[6];
605
        ctx->h[7] += F[7];
606
607
        W += SHA_LBLOCK;
608
    }
609
}
610
611
#elif defined(OPENSSL_SMALL_FOOTPRINT)
612
613
static void sha512_block_data_order(SHA512_CTX *ctx, const void *in,
614
    size_t num)
615
{
616
    const SHA_LONG64 *W = in;
617
    SHA_LONG64 a, b, c, d, e, f, g, h, s0, s1, T1, T2;
618
    SHA_LONG64 X[16];
619
    int i;
620
621
    while (num--) {
622
623
        a = ctx->h[0];
624
        b = ctx->h[1];
625
        c = ctx->h[2];
626
        d = ctx->h[3];
627
        e = ctx->h[4];
628
        f = ctx->h[5];
629
        g = ctx->h[6];
630
        h = ctx->h[7];
631
632
        for (i = 0; i < 16; i++) {
633
#ifdef B_ENDIAN
634
            T1 = X[i] = W[i];
635
#else
636
            T1 = X[i] = PULL64(W[i]);
637
#endif
638
            T1 += h + Sigma1(e) + Ch(e, f, g) + K512[i];
639
            T2 = Sigma0(a) + Maj(a, b, c);
640
            h = g;
641
            g = f;
642
            f = e;
643
            e = d + T1;
644
            d = c;
645
            c = b;
646
            b = a;
647
            a = T1 + T2;
648
        }
649
650
        for (; i < 80; i++) {
651
            s0 = X[(i + 1) & 0x0f];
652
            s0 = sigma0(s0);
653
            s1 = X[(i + 14) & 0x0f];
654
            s1 = sigma1(s1);
655
656
            T1 = X[i & 0xf] += s0 + s1 + X[(i + 9) & 0xf];
657
            T1 += h + Sigma1(e) + Ch(e, f, g) + K512[i];
658
            T2 = Sigma0(a) + Maj(a, b, c);
659
            h = g;
660
            g = f;
661
            f = e;
662
            e = d + T1;
663
            d = c;
664
            c = b;
665
            b = a;
666
            a = T1 + T2;
667
        }
668
669
        ctx->h[0] += a;
670
        ctx->h[1] += b;
671
        ctx->h[2] += c;
672
        ctx->h[3] += d;
673
        ctx->h[4] += e;
674
        ctx->h[5] += f;
675
        ctx->h[6] += g;
676
        ctx->h[7] += h;
677
678
        W += SHA_LBLOCK;
679
    }
680
}
681
682
#else
683
#define ROUND_00_15(i, a, b, c, d, e, f, g, h)       \
684
267M
    do {                                             \
685
267M
        T1 += h + Sigma1(e) + Ch(e, f, g) + K512[i]; \
686
267M
        h = Sigma0(a) + Maj(a, b, c);                \
687
267M
        d += T1;                                     \
688
267M
        h += T1;                                     \
689
267M
    } while (0)
690
691
#define ROUND_16_80(i, j, a, b, c, d, e, f, g, h, X)       \
692
213M
    do {                                                   \
693
213M
        s0 = X[(j + 1) & 0x0f];                            \
694
213M
        s0 = sigma0(s0);                                   \
695
213M
        s1 = X[(j + 14) & 0x0f];                           \
696
213M
        s1 = sigma1(s1);                                   \
697
213M
        T1 = X[(j) & 0x0f] += s0 + s1 + X[(j + 9) & 0x0f]; \
698
213M
        ROUND_00_15(i + j, a, b, c, d, e, f, g, h);        \
699
213M
    } while (0)
700
701
#ifdef INCLUDE_C_SHA512
702
void sha512_block_data_order_c(SHA512_CTX *ctx, const void *in, size_t num)
703
#else
704
static void sha512_block_data_order(SHA512_CTX *ctx, const void *in,
705
    size_t num)
706
#endif
707
10.6k
{
708
10.6k
    const SHA_LONG64 *W = in;
709
10.6k
    SHA_LONG64 a, b, c, d, e, f, g, h, s0, s1, T1;
710
10.6k
    SHA_LONG64 X[16];
711
10.6k
    int i;
712
713
3.34M
    while (num--) {
714
715
3.33M
        a = ctx->h[0];
716
3.33M
        b = ctx->h[1];
717
3.33M
        c = ctx->h[2];
718
3.33M
        d = ctx->h[3];
719
3.33M
        e = ctx->h[4];
720
3.33M
        f = ctx->h[5];
721
3.33M
        g = ctx->h[6];
722
3.33M
        h = ctx->h[7];
723
724
#ifdef B_ENDIAN
725
        T1 = X[0] = W[0];
726
        ROUND_00_15(0, a, b, c, d, e, f, g, h);
727
        T1 = X[1] = W[1];
728
        ROUND_00_15(1, h, a, b, c, d, e, f, g);
729
        T1 = X[2] = W[2];
730
        ROUND_00_15(2, g, h, a, b, c, d, e, f);
731
        T1 = X[3] = W[3];
732
        ROUND_00_15(3, f, g, h, a, b, c, d, e);
733
        T1 = X[4] = W[4];
734
        ROUND_00_15(4, e, f, g, h, a, b, c, d);
735
        T1 = X[5] = W[5];
736
        ROUND_00_15(5, d, e, f, g, h, a, b, c);
737
        T1 = X[6] = W[6];
738
        ROUND_00_15(6, c, d, e, f, g, h, a, b);
739
        T1 = X[7] = W[7];
740
        ROUND_00_15(7, b, c, d, e, f, g, h, a);
741
        T1 = X[8] = W[8];
742
        ROUND_00_15(8, a, b, c, d, e, f, g, h);
743
        T1 = X[9] = W[9];
744
        ROUND_00_15(9, h, a, b, c, d, e, f, g);
745
        T1 = X[10] = W[10];
746
        ROUND_00_15(10, g, h, a, b, c, d, e, f);
747
        T1 = X[11] = W[11];
748
        ROUND_00_15(11, f, g, h, a, b, c, d, e);
749
        T1 = X[12] = W[12];
750
        ROUND_00_15(12, e, f, g, h, a, b, c, d);
751
        T1 = X[13] = W[13];
752
        ROUND_00_15(13, d, e, f, g, h, a, b, c);
753
        T1 = X[14] = W[14];
754
        ROUND_00_15(14, c, d, e, f, g, h, a, b);
755
        T1 = X[15] = W[15];
756
        ROUND_00_15(15, b, c, d, e, f, g, h, a);
757
#else
758
3.33M
        T1 = X[0] = PULL64(W[0]);
759
3.33M
        ROUND_00_15(0, a, b, c, d, e, f, g, h);
760
3.33M
        T1 = X[1] = PULL64(W[1]);
761
3.33M
        ROUND_00_15(1, h, a, b, c, d, e, f, g);
762
3.33M
        T1 = X[2] = PULL64(W[2]);
763
3.33M
        ROUND_00_15(2, g, h, a, b, c, d, e, f);
764
3.33M
        T1 = X[3] = PULL64(W[3]);
765
3.33M
        ROUND_00_15(3, f, g, h, a, b, c, d, e);
766
3.33M
        T1 = X[4] = PULL64(W[4]);
767
3.33M
        ROUND_00_15(4, e, f, g, h, a, b, c, d);
768
3.33M
        T1 = X[5] = PULL64(W[5]);
769
3.33M
        ROUND_00_15(5, d, e, f, g, h, a, b, c);
770
3.33M
        T1 = X[6] = PULL64(W[6]);
771
3.33M
        ROUND_00_15(6, c, d, e, f, g, h, a, b);
772
3.33M
        T1 = X[7] = PULL64(W[7]);
773
3.33M
        ROUND_00_15(7, b, c, d, e, f, g, h, a);
774
3.33M
        T1 = X[8] = PULL64(W[8]);
775
3.33M
        ROUND_00_15(8, a, b, c, d, e, f, g, h);
776
3.33M
        T1 = X[9] = PULL64(W[9]);
777
3.33M
        ROUND_00_15(9, h, a, b, c, d, e, f, g);
778
3.33M
        T1 = X[10] = PULL64(W[10]);
779
3.33M
        ROUND_00_15(10, g, h, a, b, c, d, e, f);
780
3.33M
        T1 = X[11] = PULL64(W[11]);
781
3.33M
        ROUND_00_15(11, f, g, h, a, b, c, d, e);
782
3.33M
        T1 = X[12] = PULL64(W[12]);
783
3.33M
        ROUND_00_15(12, e, f, g, h, a, b, c, d);
784
3.33M
        T1 = X[13] = PULL64(W[13]);
785
3.33M
        ROUND_00_15(13, d, e, f, g, h, a, b, c);
786
3.33M
        T1 = X[14] = PULL64(W[14]);
787
3.33M
        ROUND_00_15(14, c, d, e, f, g, h, a, b);
788
3.33M
        T1 = X[15] = PULL64(W[15]);
789
3.33M
        ROUND_00_15(15, b, c, d, e, f, g, h, a);
790
3.33M
#endif
791
792
16.6M
        for (i = 16; i < 80; i += 16) {
793
13.3M
            ROUND_16_80(i, 0, a, b, c, d, e, f, g, h, X);
794
13.3M
            ROUND_16_80(i, 1, h, a, b, c, d, e, f, g, X);
795
13.3M
            ROUND_16_80(i, 2, g, h, a, b, c, d, e, f, X);
796
13.3M
            ROUND_16_80(i, 3, f, g, h, a, b, c, d, e, X);
797
13.3M
            ROUND_16_80(i, 4, e, f, g, h, a, b, c, d, X);
798
13.3M
            ROUND_16_80(i, 5, d, e, f, g, h, a, b, c, X);
799
13.3M
            ROUND_16_80(i, 6, c, d, e, f, g, h, a, b, X);
800
13.3M
            ROUND_16_80(i, 7, b, c, d, e, f, g, h, a, X);
801
13.3M
            ROUND_16_80(i, 8, a, b, c, d, e, f, g, h, X);
802
13.3M
            ROUND_16_80(i, 9, h, a, b, c, d, e, f, g, X);
803
13.3M
            ROUND_16_80(i, 10, g, h, a, b, c, d, e, f, X);
804
13.3M
            ROUND_16_80(i, 11, f, g, h, a, b, c, d, e, X);
805
13.3M
            ROUND_16_80(i, 12, e, f, g, h, a, b, c, d, X);
806
13.3M
            ROUND_16_80(i, 13, d, e, f, g, h, a, b, c, X);
807
13.3M
            ROUND_16_80(i, 14, c, d, e, f, g, h, a, b, X);
808
13.3M
            ROUND_16_80(i, 15, b, c, d, e, f, g, h, a, X);
809
13.3M
        }
810
811
3.33M
        ctx->h[0] += a;
812
3.33M
        ctx->h[1] += b;
813
3.33M
        ctx->h[2] += c;
814
3.33M
        ctx->h[3] += d;
815
3.33M
        ctx->h[4] += e;
816
3.33M
        ctx->h[5] += f;
817
3.33M
        ctx->h[6] += g;
818
3.33M
        ctx->h[7] += h;
819
820
3.33M
        W += SHA_LBLOCK;
821
3.33M
    }
822
10.6k
}
823
824
#endif
825
826
#endif /* SHA512_ASM */