Text file src/crypto/sha1/sha1block_loong64.s

     1  // Copyright 2024 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  //go:build !purego
     6  
     7  #include "textflag.h"
     8  
     9  // SHA-1 block routine. See sha1block.go for Go equivalent.
    10  //
    11  // There are 80 rounds of 4 types:
    12  //   - rounds 0-15 are type 1 and load data (ROUND1 macro).
    13  //   - rounds 16-19 are type 1 and do not load data (ROUND1x macro).
    14  //   - rounds 20-39 are type 2 and do not load data (ROUND2 macro).
    15  //   - rounds 40-59 are type 3 and do not load data (ROUND3 macro).
    16  //   - rounds 60-79 are type 4 and do not load data (ROUND4 macro).
    17  //
    18  // Each round loads or shuffles the data, then computes a per-round
    19  // function of b, c, d, and then mixes the result into and rotates the
    20  // five registers a, b, c, d, e holding the intermediate results.
    21  //
    22  // The register rotation is implemented by rotating the arguments to
    23  // the round macros instead of by explicit move instructions.
    24  
    25  // REGTMP1 and REGTMP3 are scratch registers that also carry values
    26  // between macros within a single round: they are not purely local to
    27  // the macro that sets them.
    28  //
    29  //   REGTMP3 is set by LOAD/LOAD1 to w[index] and is consumed by the
    30  //     MIX macro that follows later in the same round, as an extra
    31  //     input (the message word for this round).
    32  //
    33  //   REGTMP1 is set by FUNC1/FUNC2/FUNC3/FUNC4 to f(b,c,d) and is
    34  //     consumed by the MIX macro that immediately follows it, as an
    35  //     extra input (the per-round function result). Note: LOAD also
    36  //     writes REGTMP1, but there it is pure local scratch (dead on
    37  //     exit) consumed only by LOAD itself, not by the macro that
    38  //     follows it (FUNC1/FUNC2/FUNC3/FUNC4 unconditionally overwrite
    39  //     REGTMP1 before it could be read).
    40  //
    41  // REGTMP and REGTMP2 are used only as pure local scratch (by LOAD,
    42  // FUNC3, and MIX) and never carry a value across a macro boundary.
    43  
    44  #define REGTMP	R30
    45  #define REGTMP1	R17
    46  #define REGTMP2	R18
    47  #define REGTMP3	R19
    48  #define KEYREG1	R25
    49  #define KEYREG2	R26
    50  #define KEYREG3	R27
    51  #define KEYREG4	R28
    52  
    53  // LOAD1 loads w[index] from the input block, byte-swaps it, stores it
    54  // into the message schedule ring buffer, and leaves it in REGTMP3 for
    55  // the MIX macro that follows it in the same round (output: REGTMP3).
    56  #define LOAD1(index) \
    57  	MOVW	(index*4)(R5), REGTMP3; \
    58  	REVB2W	REGTMP3, REGTMP3; \
    59  	MOVW	REGTMP3, (index*4)(R3)
    60  
    61  // LOAD computes w[index] from the ring buffer (message schedule
    62  // expansion), using REGTMP/REGTMP1/REGTMP2 as local scratch, and
    63  // leaves the result in REGTMP3 for the MIX macro that follows it in
    64  // the same round (output: REGTMP3; REGTMP/REGTMP1/REGTMP2 are dead
    65  // on exit).
    66  #define LOAD(index) \
    67  	MOVW	(((index)&0xf)*4)(R3), REGTMP3; \
    68  	MOVW	(((index-3)&0xf)*4)(R3), REGTMP; \
    69  	MOVW	(((index-8)&0xf)*4)(R3), REGTMP1; \
    70  	MOVW	(((index-14)&0xf)*4)(R3), REGTMP2; \
    71  	XOR	REGTMP, REGTMP3; \
    72  	XOR	REGTMP1, REGTMP3; \
    73  	XOR	REGTMP2, REGTMP3; \
    74  	ROTR	$31, REGTMP3; \
    75  	MOVW	REGTMP3, (((index)&0xf)*4)(R3)
    76  
    77  // f = d ^ (b & (c ^ d))
    78  // Output: REGTMP1 = f, consumed by the MIX macro that follows it in
    79  // the same round.
    80  #define FUNC1(a, b, c, d, e) \
    81  	XOR	c, d, REGTMP1; \
    82  	AND	b, REGTMP1; \
    83  	XOR	d, REGTMP1
    84  
    85  // f = b ^ c ^ d
    86  // Output: REGTMP1 = f, consumed by the MIX macro that follows it in
    87  // the same round.
    88  #define FUNC2(a, b, c, d, e) \
    89  	XOR	b, c, REGTMP1; \
    90  	XOR	d, REGTMP1
    91  
    92  // f = (b & c) | ((b | c) & d)
    93  // REGTMP, REGTMP2: local scratch only.
    94  // Output: REGTMP1 = f, consumed by the MIX macro that follows it in
    95  // the same round.
    96  #define FUNC3(a, b, c, d, e) \
    97  	OR	b, c, REGTMP2; \
    98  	AND	b, c, REGTMP; \
    99  	AND	d, REGTMP2; \
   100  	OR	REGTMP, REGTMP2, REGTMP1
   101  
   102  #define FUNC4 FUNC2
   103  
   104  // MIX combines the per-round function result and message word computed
   105  // by the preceding FUNC* and LOAD/LOAD1 macros into e, and rotates b.
   106  // Inputs (set by the preceding macros earlier in the same round, not
   107  // local to MIX): REGTMP1 = f(b,c,d), REGTMP3 = w[index].
   108  // REGTMP2 is local scratch (holds a<<<5) and is dead on exit; REGTMP
   109  // is not used by this macro.
   110  #define MIX(a, b, c, d, e, key) \
   111  	ROTR	$2, b; \		// b << 30
   112  	ROTR	$27, a, REGTMP2; \	// a << 5
   113  	ADD	REGTMP3, REGTMP1; \	// t1 = f + w[i]  (independent of e)
   114  	ADDV	key, REGTMP2; \		// t2 = k + a<<5  (independent of e)
   115  	ADD	REGTMP1, e; \		// e += t1
   116  	ADD	REGTMP2, e		// e += t2
   117  
   118  #define ROUND1(a, b, c, d, e, index) \
   119  	LOAD1(index); \
   120  	FUNC1(a, b, c, d, e); \
   121  	MIX(a, b, c, d, e, KEYREG1)
   122  
   123  #define ROUND1x(a, b, c, d, e, index) \
   124  	LOAD(index); \
   125  	FUNC1(a, b, c, d, e); \
   126  	MIX(a, b, c, d, e, KEYREG1)
   127  
   128  #define ROUND2(a, b, c, d, e, index) \
   129  	LOAD(index); \
   130  	FUNC2(a, b, c, d, e); \
   131  	MIX(a, b, c, d, e, KEYREG2)
   132  
   133  #define ROUND3(a, b, c, d, e, index) \
   134  	LOAD(index); \
   135  	FUNC3(a, b, c, d, e); \
   136  	MIX(a, b, c, d, e, KEYREG3)
   137  
   138  #define ROUND4(a, b, c, d, e, index) \
   139  	LOAD(index); \
   140  	FUNC4(a, b, c, d, e); \
   141  	MIX(a, b, c, d, e, KEYREG4)
   142  
   143  // A stack frame size of 64 bytes is required here, because
   144  // the frame size used for data expansion is 64 bytes.
   145  // See the definition of the macro LOAD above, and the definition
   146  // of the local variable w in the general implementation (sha1block.go).
   147  TEXT ·block(SB),NOSPLIT,$64-32
   148  	MOVV	dig+0(FP),	R4
   149  	MOVV	p_base+8(FP),	R5
   150  	MOVV	p_len+16(FP),	R6
   151  	AND	$~63, R6
   152  	BEQ	R6, zero
   153  
   154  	// p_len >= 64
   155  	ADDV	R5, R6, R24
   156  	MOVW	(0*4)(R4), R7
   157  	MOVW	(1*4)(R4), R8
   158  	MOVW	(2*4)(R4), R9
   159  	MOVW	(3*4)(R4), R10
   160  	MOVW	(4*4)(R4), R11
   161  
   162  	MOVV	$·_K(SB), R21
   163  	MOVW	(0*4)(R21), KEYREG1
   164  	MOVW	(1*4)(R21), KEYREG2
   165  	MOVW	(2*4)(R21), KEYREG3
   166  	MOVW	(3*4)(R21), KEYREG4
   167  
   168  loop:
   169  	MOVW	R7,	R12
   170  	MOVW	R8,	R13
   171  	MOVW	R9,	R14
   172  	MOVW	R10,	R15
   173  	MOVW	R11,	R16
   174  
   175  	ROUND1(R7,  R8,  R9,  R10, R11, 0)
   176  	ROUND1(R11, R7,  R8,  R9,  R10, 1)
   177  	ROUND1(R10, R11, R7,  R8,  R9,  2)
   178  	ROUND1(R9,  R10, R11, R7,  R8,  3)
   179  	ROUND1(R8,  R9,  R10, R11, R7,  4)
   180  	ROUND1(R7,  R8,  R9,  R10, R11, 5)
   181  	ROUND1(R11, R7,  R8,  R9,  R10, 6)
   182  	ROUND1(R10, R11, R7,  R8,  R9,  7)
   183  	ROUND1(R9,  R10, R11, R7,  R8,  8)
   184  	ROUND1(R8,  R9,  R10, R11, R7,  9)
   185  	ROUND1(R7,  R8,  R9,  R10, R11, 10)
   186  	ROUND1(R11, R7,  R8,  R9,  R10, 11)
   187  	ROUND1(R10, R11, R7,  R8,  R9,  12)
   188  	ROUND1(R9,  R10, R11, R7,  R8,  13)
   189  	ROUND1(R8,  R9,  R10, R11, R7,  14)
   190  	ROUND1(R7,  R8,  R9,  R10, R11, 15)
   191  
   192  	ROUND1x(R11, R7,  R8,  R9,  R10, 16)
   193  	ROUND1x(R10, R11, R7,  R8,  R9,  17)
   194  	ROUND1x(R9,  R10, R11, R7,  R8,  18)
   195  	ROUND1x(R8,  R9,  R10, R11, R7,  19)
   196  
   197  	ROUND2(R7,  R8,  R9,  R10, R11, 20)
   198  	ROUND2(R11, R7,  R8,  R9,  R10, 21)
   199  	ROUND2(R10, R11, R7,  R8,  R9,  22)
   200  	ROUND2(R9,  R10, R11, R7,  R8,  23)
   201  	ROUND2(R8,  R9,  R10, R11, R7,  24)
   202  	ROUND2(R7,  R8,  R9,  R10, R11, 25)
   203  	ROUND2(R11, R7,  R8,  R9,  R10, 26)
   204  	ROUND2(R10, R11, R7,  R8,  R9,  27)
   205  	ROUND2(R9,  R10, R11, R7,  R8,  28)
   206  	ROUND2(R8,  R9,  R10, R11, R7,  29)
   207  	ROUND2(R7,  R8,  R9,  R10, R11, 30)
   208  	ROUND2(R11, R7,  R8,  R9,  R10, 31)
   209  	ROUND2(R10, R11, R7,  R8,  R9,  32)
   210  	ROUND2(R9,  R10, R11, R7,  R8,  33)
   211  	ROUND2(R8,  R9,  R10, R11, R7,  34)
   212  	ROUND2(R7,  R8,  R9,  R10, R11, 35)
   213  	ROUND2(R11, R7,  R8,  R9,  R10, 36)
   214  	ROUND2(R10, R11, R7,  R8,  R9,  37)
   215  	ROUND2(R9,  R10, R11, R7,  R8,  38)
   216  	ROUND2(R8,  R9,  R10, R11, R7,  39)
   217  
   218  	ROUND3(R7,  R8,  R9,  R10, R11, 40)
   219  	ROUND3(R11, R7,  R8,  R9,  R10, 41)
   220  	ROUND3(R10, R11, R7,  R8,  R9,  42)
   221  	ROUND3(R9,  R10, R11, R7,  R8,  43)
   222  	ROUND3(R8,  R9,  R10, R11, R7,  44)
   223  	ROUND3(R7,  R8,  R9,  R10, R11, 45)
   224  	ROUND3(R11, R7,  R8,  R9,  R10, 46)
   225  	ROUND3(R10, R11, R7,  R8,  R9,  47)
   226  	ROUND3(R9,  R10, R11, R7,  R8,  48)
   227  	ROUND3(R8,  R9,  R10, R11, R7,  49)
   228  	ROUND3(R7,  R8,  R9,  R10, R11, 50)
   229  	ROUND3(R11, R7,  R8,  R9,  R10, 51)
   230  	ROUND3(R10, R11, R7,  R8,  R9,  52)
   231  	ROUND3(R9,  R10, R11, R7,  R8,  53)
   232  	ROUND3(R8,  R9,  R10, R11, R7,  54)
   233  	ROUND3(R7,  R8,  R9,  R10, R11, 55)
   234  	ROUND3(R11, R7,  R8,  R9,  R10, 56)
   235  	ROUND3(R10, R11, R7,  R8,  R9,  57)
   236  	ROUND3(R9,  R10, R11, R7,  R8,  58)
   237  	ROUND3(R8,  R9,  R10, R11, R7,  59)
   238  
   239  	ROUND4(R7,  R8,  R9,  R10, R11, 60)
   240  	ROUND4(R11, R7,  R8,  R9,  R10, 61)
   241  	ROUND4(R10, R11, R7,  R8,  R9,  62)
   242  	ROUND4(R9,  R10, R11, R7,  R8,  63)
   243  	ROUND4(R8,  R9,  R10, R11, R7,  64)
   244  	ROUND4(R7,  R8,  R9,  R10, R11, 65)
   245  	ROUND4(R11, R7,  R8,  R9,  R10, 66)
   246  	ROUND4(R10, R11, R7,  R8,  R9,  67)
   247  	ROUND4(R9,  R10, R11, R7,  R8,  68)
   248  	ROUND4(R8,  R9,  R10, R11, R7,  69)
   249  	ROUND4(R7,  R8,  R9,  R10, R11, 70)
   250  	ROUND4(R11, R7,  R8,  R9,  R10, 71)
   251  	ROUND4(R10, R11, R7,  R8,  R9,  72)
   252  	ROUND4(R9,  R10, R11, R7,  R8,  73)
   253  	ROUND4(R8,  R9,  R10, R11, R7,  74)
   254  	ROUND4(R7,  R8,  R9,  R10, R11, 75)
   255  	ROUND4(R11, R7,  R8,  R9,  R10, 76)
   256  	ROUND4(R10, R11, R7,  R8,  R9,  77)
   257  	ROUND4(R9,  R10, R11, R7,  R8,  78)
   258  	ROUND4(R8,  R9,  R10, R11, R7,  79)
   259  
   260  	ADD	R12, R7
   261  	ADD	R13, R8
   262  	ADD	R14, R9
   263  	ADD	R15, R10
   264  	ADD	R16, R11
   265  
   266  	ADDV	$64, R5
   267  	BNE	R5, R24, loop
   268  
   269  end:
   270  	MOVW	R7, (0*4)(R4)
   271  	MOVW	R8, (1*4)(R4)
   272  	MOVW	R9, (2*4)(R4)
   273  	MOVW	R10, (3*4)(R4)
   274  	MOVW	R11, (4*4)(R4)
   275  zero:
   276  	RET
   277  
   278  GLOBL	·_K(SB),RODATA,$16
   279  DATA	·_K+0(SB)/4, $0x5A827999
   280  DATA	·_K+4(SB)/4, $0x6ED9EBA1
   281  DATA	·_K+8(SB)/4, $0x8F1BBCDC
   282  DATA	·_K+12(SB)/4, $0xCA62C1D6
   283  

View as plain text