Text file src/crypto/internal/fips140/aes/gcm/gcm_arm64.s

     1  // Copyright 2018 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  //go:build !purego
     6  
     7  #include "textflag.h"
     8  
     9  #define B0 V0
    10  #define B1 V1
    11  #define B2 V2
    12  #define B3 V3
    13  #define B4 V4
    14  #define B5 V5
    15  #define B6 V6
    16  #define B7 V7
    17  
    18  #define ACC0 V8
    19  #define ACC1 V9
    20  #define ACCM V10
    21  
    22  #define T0 V11
    23  #define T1 V12
    24  #define T2 V13
    25  #define T3 V14
    26  
    27  #define POLY V15
    28  #define ZERO V16
    29  #define INC V17
    30  #define CTR V18
    31  
    32  #define K0 V19
    33  #define K1 V20
    34  #define K2 V21
    35  #define K3 V22
    36  #define K4 V23
    37  #define K5 V24
    38  #define K6 V25
    39  #define K7 V26
    40  #define K8 V27
    41  #define K9 V28
    42  #define K10 V29
    43  #define K11 V30
    44  #define KLAST V31
    45  
    46  #define reduce() \
    47  	VEOR	ACC0.B16, ACCM.B16, ACCM.B16     \
    48  	VEOR	ACC1.B16, ACCM.B16, ACCM.B16     \
    49  	VEXT	$8, ZERO.B16, ACCM.B16, T0.B16   \
    50  	VEXT	$8, ACCM.B16, ZERO.B16, ACCM.B16 \
    51  	VEOR	ACCM.B16, ACC0.B16, ACC0.B16     \
    52  	VEOR	T0.B16, ACC1.B16, ACC1.B16       \
    53  	VPMULL	POLY.D1, ACC0.D1, T0.Q1          \
    54  	VEXT	$8, ACC0.B16, ACC0.B16, ACC0.B16 \
    55  	VEOR	T0.B16, ACC0.B16, ACC0.B16       \
    56  	VPMULL	POLY.D1, ACC0.D1, T0.Q1          \
    57  	VEOR	T0.B16, ACC1.B16, ACC1.B16       \
    58  	VEXT	$8, ACC1.B16, ACC1.B16, ACC1.B16 \
    59  	VEOR	ACC1.B16, ACC0.B16, ACC0.B16     \
    60  
    61  // func gcmAesFinish(productTable *[256]byte, tagMask, T *[16]byte, pLen, dLen uint64)
    62  TEXT ·gcmAesFinish(SB),NOSPLIT,$0
    63  #define pTbl R0
    64  #define tMsk R1
    65  #define tPtr R2
    66  #define plen R3
    67  #define dlen R4
    68  
    69  	MOVD	$0xC2, R1
    70  	LSL	$56, R1
    71  	MOVD	$1, R0
    72  	VMOV	R1, POLY.D[0]
    73  	VMOV	R0, POLY.D[1]
    74  	VEOR	ZERO.B16, ZERO.B16, ZERO.B16
    75  
    76  	MOVD	productTable+0(FP), pTbl
    77  	MOVD	tagMask+8(FP), tMsk
    78  	MOVD	T+16(FP), tPtr
    79  	MOVD	pLen+24(FP), plen
    80  	MOVD	dLen+32(FP), dlen
    81  
    82  	VLD1	(tPtr), [ACC0.B16]
    83  	VLD1	(tMsk), [B1.B16]
    84  
    85  	LSL	$3, plen
    86  	LSL	$3, dlen
    87  
    88  	VMOV	dlen, B0.D[0]
    89  	VMOV	plen, B0.D[1]
    90  
    91  	ADD	$14*16, pTbl
    92  	VLD1.P	(pTbl), [T1.B16, T2.B16]
    93  
    94  	VEOR	ACC0.B16, B0.B16, B0.B16
    95  
    96  	VEXT	$8, B0.B16, B0.B16, T0.B16
    97  	VEOR	B0.B16, T0.B16, T0.B16
    98  	VPMULL	B0.D1, T1.D1, ACC1.Q1
    99  	VPMULL2	B0.D2, T1.D2, ACC0.Q1
   100  	VPMULL	T0.D1, T2.D1, ACCM.Q1
   101  
   102  	reduce()
   103  
   104  	VREV64	ACC0.B16, ACC0.B16
   105  	VEOR	B1.B16, ACC0.B16, ACC0.B16
   106  
   107  	VST1	[ACC0.B16], (tPtr)
   108  	RET
   109  #undef pTbl
   110  #undef tMsk
   111  #undef tPtr
   112  #undef plen
   113  #undef dlen
   114  
   115  // func gcmAesInit(productTable *[256]byte, ks []uint32)
   116  TEXT ·gcmAesInit(SB),NOSPLIT,$0
   117  #define pTbl R0
   118  #define KS R1
   119  #define NR R2
   120  #define I R3
   121  	MOVD	productTable+0(FP), pTbl
   122  	MOVD	ks_base+8(FP), KS
   123  	MOVD	ks_len+16(FP), NR
   124  
   125  	MOVD	$0xC2, I
   126  	LSL	$56, I
   127  	VMOV	I, POLY.D[0]
   128  	MOVD	$1, I
   129  	VMOV	I, POLY.D[1]
   130  	VEOR	ZERO.B16, ZERO.B16, ZERO.B16
   131  
   132  	// Encrypt block 0 with the AES key to generate the hash key H
   133  	VLD1.P	64(KS), [T0.B16, T1.B16, T2.B16, T3.B16]
   134  	VEOR	B0.B16, B0.B16, B0.B16
   135  	AESE	T0.B16, B0.B16
   136  	AESMC	B0.B16, B0.B16
   137  	AESE	T1.B16, B0.B16
   138  	AESMC	B0.B16, B0.B16
   139  	AESE	T2.B16, B0.B16
   140  	AESMC	B0.B16, B0.B16
   141  	AESE	T3.B16, B0.B16
   142  	AESMC	B0.B16, B0.B16
   143  	VLD1.P	64(KS), [T0.B16, T1.B16, T2.B16, T3.B16]
   144  	AESE	T0.B16, B0.B16
   145  	AESMC	B0.B16, B0.B16
   146  	AESE	T1.B16, B0.B16
   147  	AESMC	B0.B16, B0.B16
   148  	AESE	T2.B16, B0.B16
   149  	AESMC	B0.B16, B0.B16
   150  	AESE	T3.B16, B0.B16
   151  	AESMC	B0.B16, B0.B16
   152  	TBZ	$4, NR, initEncFinish
   153  	VLD1.P	32(KS), [T0.B16, T1.B16]
   154  	AESE	T0.B16, B0.B16
   155  	AESMC	B0.B16, B0.B16
   156  	AESE	T1.B16, B0.B16
   157  	AESMC	B0.B16, B0.B16
   158  	TBZ	$3, NR, initEncFinish
   159  	VLD1.P	32(KS), [T0.B16, T1.B16]
   160  	AESE	T0.B16, B0.B16
   161  	AESMC	B0.B16, B0.B16
   162  	AESE	T1.B16, B0.B16
   163  	AESMC	B0.B16, B0.B16
   164  initEncFinish:
   165  	VLD1	(KS), [T0.B16, T1.B16, T2.B16]
   166  	AESE	T0.B16, B0.B16
   167  	AESMC	B0.B16, B0.B16
   168  	AESE	T1.B16, B0.B16
   169  	VEOR	T2.B16, B0.B16, B0.B16
   170  
   171  	VREV64	B0.B16, B0.B16
   172  
   173  	// Multiply by 2 modulo P
   174  	VMOV	B0.D[0], I
   175  	ASR	$63, I
   176  	VMOV	I, T1.D[0]
   177  	VMOV	I, T1.D[1]
   178  	VAND	POLY.B16, T1.B16, T1.B16
   179  	VUSHR	$63, B0.D2, T2.D2
   180  	VEXT	$8, ZERO.B16, T2.B16, T2.B16
   181  	VSHL	$1, B0.D2, B0.D2
   182  	VEOR	T1.B16, B0.B16, B0.B16
   183  	VEOR	T2.B16, B0.B16, B0.B16 // Can avoid this when VSLI is available
   184  
   185  	// Karatsuba pre-computation
   186  	VEXT	$8, B0.B16, B0.B16, B1.B16
   187  	VEOR	B0.B16, B1.B16, B1.B16
   188  
   189  	ADD	$14*16, pTbl
   190  	VST1	[B0.B16, B1.B16], (pTbl)
   191  	SUB	$2*16, pTbl
   192  
   193  	VMOV	B0.B16, B2.B16
   194  	VMOV	B1.B16, B3.B16
   195  
   196  	MOVD	$7, I
   197  
   198  initLoop:
   199  	// Compute powers of H
   200  	SUBS	$1, I
   201  
   202  	VPMULL	B0.D1, B2.D1, T1.Q1
   203  	VPMULL2	B0.D2, B2.D2, T0.Q1
   204  	VPMULL	B1.D1, B3.D1, T2.Q1
   205  	VEOR	T0.B16, T2.B16, T2.B16
   206  	VEOR	T1.B16, T2.B16, T2.B16
   207  	VEXT	$8, ZERO.B16, T2.B16, T3.B16
   208  	VEXT	$8, T2.B16, ZERO.B16, T2.B16
   209  	VEOR	T2.B16, T0.B16, T0.B16
   210  	VEOR	T3.B16, T1.B16, T1.B16
   211  	VPMULL	POLY.D1, T0.D1, T2.Q1
   212  	VEXT	$8, T0.B16, T0.B16, T0.B16
   213  	VEOR	T2.B16, T0.B16, T0.B16
   214  	VPMULL	POLY.D1, T0.D1, T2.Q1
   215  	VEXT	$8, T0.B16, T0.B16, T0.B16
   216  	VEOR	T2.B16, T0.B16, T0.B16
   217  	VEOR	T1.B16, T0.B16, B2.B16
   218  	VMOV	B2.B16, B3.B16
   219  	VEXT	$8, B2.B16, B2.B16, B2.B16
   220  	VEOR	B2.B16, B3.B16, B3.B16
   221  
   222  	VST1	[B2.B16, B3.B16], (pTbl)
   223  	SUB	$2*16, pTbl
   224  
   225  	BNE	initLoop
   226  	RET
   227  #undef I
   228  #undef NR
   229  #undef KS
   230  #undef pTbl
   231  
   232  // func gcmAesData(productTable *[256]byte, data []byte, T *[16]byte)
   233  TEXT ·gcmAesData(SB),NOSPLIT,$0
   234  #define pTbl R0
   235  #define aut R1
   236  #define tPtr R2
   237  #define autLen R3
   238  #define H0 R4
   239  #define pTblSave R5
   240  
   241  #define mulRound(X) \
   242  	VLD1.P	32(pTbl), [T1.B16, T2.B16] \
   243  	VREV64	X.B16, X.B16               \
   244  	VEXT	$8, X.B16, X.B16, T0.B16   \
   245  	VEOR	X.B16, T0.B16, T0.B16      \
   246  	VPMULL	X.D1, T1.D1, T3.Q1         \
   247  	VEOR	T3.B16, ACC1.B16, ACC1.B16 \
   248  	VPMULL2	X.D2, T1.D2, T3.Q1         \
   249  	VEOR	T3.B16, ACC0.B16, ACC0.B16 \
   250  	VPMULL	T0.D1, T2.D1, T3.Q1        \
   251  	VEOR	T3.B16, ACCM.B16, ACCM.B16
   252  
   253  	MOVD	productTable+0(FP), pTbl
   254  	MOVD	data_base+8(FP), aut
   255  	MOVD	data_len+16(FP), autLen
   256  	MOVD	T+32(FP), tPtr
   257  
   258  	VEOR	ACC0.B16, ACC0.B16, ACC0.B16
   259  	CBZ	autLen, dataBail
   260  
   261  	MOVD	$0xC2, H0
   262  	LSL	$56, H0
   263  	VMOV	H0, POLY.D[0]
   264  	MOVD	$1, H0
   265  	VMOV	H0, POLY.D[1]
   266  	VEOR	ZERO.B16, ZERO.B16, ZERO.B16
   267  	MOVD	pTbl, pTblSave
   268  
   269  	CMP	$13, autLen
   270  	BEQ	dataTLS
   271  	CMP	$128, autLen
   272  	BLT	startSinglesLoop
   273  	B	octetsLoop
   274  
   275  dataTLS:
   276  	ADD	$14*16, pTbl
   277  	VLD1.P	(pTbl), [T1.B16, T2.B16]
   278  	VEOR	B0.B16, B0.B16, B0.B16
   279  
   280  	MOVD	(aut), H0
   281  	VMOV	H0, B0.D[0]
   282  	MOVW	8(aut), H0
   283  	VMOV	H0, B0.S[2]
   284  	MOVB	12(aut), H0
   285  	VMOV	H0, B0.B[12]
   286  
   287  	MOVD	$0, autLen
   288  	B	dataMul
   289  
   290  octetsLoop:
   291  		CMP	$128, autLen
   292  		BLT	startSinglesLoop
   293  		SUB	$128, autLen
   294  
   295  		VLD1.P	32(aut), [B0.B16, B1.B16]
   296  
   297  		VLD1.P	32(pTbl), [T1.B16, T2.B16]
   298  		VREV64	B0.B16, B0.B16
   299  		VEOR	ACC0.B16, B0.B16, B0.B16
   300  		VEXT	$8, B0.B16, B0.B16, T0.B16
   301  		VEOR	B0.B16, T0.B16, T0.B16
   302  		VPMULL	B0.D1, T1.D1, ACC1.Q1
   303  		VPMULL2	B0.D2, T1.D2, ACC0.Q1
   304  		VPMULL	T0.D1, T2.D1, ACCM.Q1
   305  
   306  		mulRound(B1)
   307  		VLD1.P  32(aut), [B2.B16, B3.B16]
   308  		mulRound(B2)
   309  		mulRound(B3)
   310  		VLD1.P  32(aut), [B4.B16, B5.B16]
   311  		mulRound(B4)
   312  		mulRound(B5)
   313  		VLD1.P  32(aut), [B6.B16, B7.B16]
   314  		mulRound(B6)
   315  		mulRound(B7)
   316  
   317  		MOVD	pTblSave, pTbl
   318  		reduce()
   319  	B	octetsLoop
   320  
   321  startSinglesLoop:
   322  
   323  	ADD	$14*16, pTbl
   324  	VLD1.P	(pTbl), [T1.B16, T2.B16]
   325  
   326  singlesLoop:
   327  
   328  		CMP	$16, autLen
   329  		BLT	dataEnd
   330  		SUB	$16, autLen
   331  
   332  		VLD1.P	16(aut), [B0.B16]
   333  dataMul:
   334  		VREV64	B0.B16, B0.B16
   335  		VEOR	ACC0.B16, B0.B16, B0.B16
   336  
   337  		VEXT	$8, B0.B16, B0.B16, T0.B16
   338  		VEOR	B0.B16, T0.B16, T0.B16
   339  		VPMULL	B0.D1, T1.D1, ACC1.Q1
   340  		VPMULL2	B0.D2, T1.D2, ACC0.Q1
   341  		VPMULL	T0.D1, T2.D1, ACCM.Q1
   342  
   343  		reduce()
   344  
   345  	B	singlesLoop
   346  
   347  dataEnd:
   348  
   349  	CBZ	autLen, dataBail
   350  	VEOR	B0.B16, B0.B16, B0.B16
   351  	ADD	autLen, aut
   352  
   353  dataLoadLoop:
   354  		MOVB.W	-1(aut), H0
   355  		VEXT	$15, B0.B16, ZERO.B16, B0.B16
   356  		VMOV	H0, B0.B[0]
   357  		SUBS	$1, autLen
   358  		BNE	dataLoadLoop
   359  	B	dataMul
   360  
   361  dataBail:
   362  	VST1	[ACC0.B16], (tPtr)
   363  	RET
   364  
   365  #undef pTbl
   366  #undef aut
   367  #undef tPtr
   368  #undef autLen
   369  #undef H0
   370  #undef pTblSave
   371  
   372  // func gcmAesEnc(productTable *[256]byte, dst, src []byte, ctr, T *[16]byte, ks []uint32)
   373  TEXT ·gcmAesEnc(SB),NOSPLIT,$0
   374  #define pTbl R0
   375  #define dstPtr R1
   376  #define ctrPtr R2
   377  #define srcPtr R3
   378  #define ks R4
   379  #define tPtr R5
   380  #define srcPtrLen R6
   381  #define NR R10
   382  #define H0 R11
   383  #define H1 R12
   384  #define curK R13
   385  #define pTblSave R14
   386  
   387  #define aesrndx8(K) \
   388  	AESE	K.B16, B0.B16    \
   389  	AESMC	B0.B16, B0.B16   \
   390  	AESE	K.B16, B1.B16    \
   391  	AESMC	B1.B16, B1.B16   \
   392  	AESE	K.B16, B2.B16    \
   393  	AESMC	B2.B16, B2.B16   \
   394  	AESE	K.B16, B3.B16    \
   395  	AESMC	B3.B16, B3.B16   \
   396  	AESE	K.B16, B4.B16    \
   397  	AESMC	B4.B16, B4.B16   \
   398  	AESE	K.B16, B5.B16    \
   399  	AESMC	B5.B16, B5.B16   \
   400  	AESE	K.B16, B6.B16    \
   401  	AESMC	B6.B16, B6.B16   \
   402  	AESE	K.B16, B7.B16    \
   403  	AESMC	B7.B16, B7.B16
   404  
   405  #define aesrndlastx8(K) \
   406  	AESE	K.B16, B0.B16    \
   407  	AESE	K.B16, B1.B16    \
   408  	AESE	K.B16, B2.B16    \
   409  	AESE	K.B16, B3.B16    \
   410  	AESE	K.B16, B4.B16    \
   411  	AESE	K.B16, B5.B16    \
   412  	AESE	K.B16, B6.B16    \
   413  	AESE	K.B16, B7.B16
   414  
   415  // tailLoad reads the srcPtrLen bytes at srcPtr into the low bytes of
   416  // X, zero-padded, loading 8, 4, 2, and 1 bytes at a time to avoid
   417  // reading past the end of the source buffer. It also builds in T3 a
   418  // mask of the bytes within the source length. It clobbers H0 and H1.
   419  #define tailLoad(X) \
   420  	VEOR	X.B16, X.B16, X.B16           \
   421  	VEOR	T3.B16, T3.B16, T3.B16        \
   422  	MOVD	$-1, H1                       \
   423  	ADD	srcPtrLen, srcPtr             \
   424  	TBZ	$3, srcPtrLen, tailLoad4      \
   425  	MOVD.W	-8(srcPtr), H0                \
   426  	VMOV	H0, X.D[0]                    \
   427  	VMOV	H1, T3.D[0]                   \
   428  tailLoad4: \
   429  	TBZ	$2, srcPtrLen, tailLoad2      \
   430  	MOVW.W	-4(srcPtr), H0                \
   431  	VEXT	$12, X.B16, ZERO.B16, X.B16   \
   432  	VEXT	$12, T3.B16, ZERO.B16, T3.B16 \
   433  	VMOV	H0, X.S[0]                    \
   434  	VMOV	H1, T3.S[0]                   \
   435  tailLoad2: \
   436  	TBZ	$1, srcPtrLen, tailLoad1      \
   437  	MOVH.W	-2(srcPtr), H0                \
   438  	VEXT	$14, X.B16, ZERO.B16, X.B16   \
   439  	VEXT	$14, T3.B16, ZERO.B16, T3.B16 \
   440  	VMOV	H0, X.H[0]                    \
   441  	VMOV	H1, T3.H[0]                   \
   442  tailLoad1: \
   443  	TBZ	$0, srcPtrLen, tailLoad0      \
   444  	MOVB.W	-1(srcPtr), H0                \
   445  	VEXT	$15, X.B16, ZERO.B16, X.B16   \
   446  	VEXT	$15, T3.B16, ZERO.B16, T3.B16 \
   447  	VMOV	H0, X.B[0]                    \
   448  	VMOV	H1, T3.B[0]                   \
   449  tailLoad0:
   450  
   451  // tailStore writes the low srcPtrLen bytes of X to dstPtr, storing 8,
   452  // 4, 2, and 1 bytes at a time to avoid writing past the end of the
   453  // destination buffer. It clobbers X and H0.
   454  #define tailStore(X) \
   455  	TBZ	$3, srcPtrLen, tailStore4  \
   456  	VMOV	X.D[0], H0                 \
   457  	MOVD.P	H0, 8(dstPtr)              \
   458  	VEXT	$8, ZERO.B16, X.B16, X.B16 \
   459  tailStore4: \
   460  	TBZ	$2, srcPtrLen, tailStore2  \
   461  	VMOV	X.S[0], H0                 \
   462  	MOVW.P	H0, 4(dstPtr)              \
   463  	VEXT	$4, ZERO.B16, X.B16, X.B16 \
   464  tailStore2: \
   465  	TBZ	$1, srcPtrLen, tailStore1  \
   466  	VMOV	X.H[0], H0                 \
   467  	MOVH.P	H0, 2(dstPtr)              \
   468  	VEXT	$2, ZERO.B16, X.B16, X.B16 \
   469  tailStore1: \
   470  	TBZ	$0, srcPtrLen, tailStore0  \
   471  	VMOV	X.B[0], H0                 \
   472  	MOVB.P	H0, 1(dstPtr)              \
   473  tailStore0:
   474  
   475  	MOVD	productTable+0(FP), pTbl
   476  	MOVD	dst+8(FP), dstPtr
   477  	MOVD	src_base+32(FP), srcPtr
   478  	MOVD	src_len+40(FP), srcPtrLen
   479  	MOVD	ctr+56(FP), ctrPtr
   480  	MOVD	T+64(FP), tPtr
   481  	MOVD	ks_base+72(FP), ks
   482  	MOVD	ks_len+80(FP), NR
   483  
   484  	MOVD	$0xC2, H1
   485  	LSL	$56, H1
   486  	MOVD	$1, H0
   487  	VMOV	H1, POLY.D[0]
   488  	VMOV	H0, POLY.D[1]
   489  	VEOR	ZERO.B16, ZERO.B16, ZERO.B16
   490  	// Compute NR from len(ks)
   491  	MOVD	pTbl, pTblSave
   492  	// Current tag, after AAD
   493  	VLD1	(tPtr), [ACC0.B16]
   494  	VEOR	ACC1.B16, ACC1.B16, ACC1.B16
   495  	VEOR	ACCM.B16, ACCM.B16, ACCM.B16
   496  	// Prepare initial counter, and the increment vector
   497  	VLD1	(ctrPtr), [CTR.B16]
   498  	VEOR	INC.B16, INC.B16, INC.B16
   499  	MOVD	$1, H0
   500  	VMOV	H0, INC.S[3]
   501  	VREV32	CTR.B16, CTR.B16
   502  	VADD	CTR.S4, INC.S4, CTR.S4
   503  	// Skip to <8 blocks loop
   504  	CMP	$128, srcPtrLen
   505  
   506  	MOVD	ks, H0
   507  	// For AES-128 round keys are stored in: K0 .. K10, KLAST
   508  	VLD1.P	64(H0), [K0.B16, K1.B16, K2.B16, K3.B16]
   509  	VLD1.P	64(H0), [K4.B16, K5.B16, K6.B16, K7.B16]
   510  	VLD1.P	48(H0), [K8.B16, K9.B16, K10.B16]
   511  	VMOV	K10.B16, KLAST.B16
   512  
   513  	BLT	startSingles
   514  	// There are at least 8 blocks to encrypt
   515  	TBZ	$4, NR, octetsLoop
   516  
   517  	// For AES-192 round keys occupy: K0 .. K7, K10, K11, K8, K9, KLAST
   518  	VMOV	K8.B16, K10.B16
   519  	VMOV	K9.B16, K11.B16
   520  	VMOV	KLAST.B16, K8.B16
   521  	VLD1.P	16(H0), [K9.B16]
   522  	VLD1.P  16(H0), [KLAST.B16]
   523  	TBZ	$3, NR, octetsLoop
   524  	// For AES-256 round keys occupy: K0 .. K7, K10, K11, mem, mem, K8, K9, KLAST
   525  	VMOV	KLAST.B16, K8.B16
   526  	VLD1.P	16(H0), [K9.B16]
   527  	VLD1.P  16(H0), [KLAST.B16]
   528  	ADD	$10*16, ks, H0
   529  	MOVD	H0, curK
   530  
   531  octetsLoop:
   532  		SUB	$128, srcPtrLen
   533  
   534  		VMOV	CTR.B16, B0.B16
   535  		VADD	B0.S4, INC.S4, B1.S4
   536  		VREV32	B0.B16, B0.B16
   537  		VADD	B1.S4, INC.S4, B2.S4
   538  		VREV32	B1.B16, B1.B16
   539  		VADD	B2.S4, INC.S4, B3.S4
   540  		VREV32	B2.B16, B2.B16
   541  		VADD	B3.S4, INC.S4, B4.S4
   542  		VREV32	B3.B16, B3.B16
   543  		VADD	B4.S4, INC.S4, B5.S4
   544  		VREV32	B4.B16, B4.B16
   545  		VADD	B5.S4, INC.S4, B6.S4
   546  		VREV32	B5.B16, B5.B16
   547  		VADD	B6.S4, INC.S4, B7.S4
   548  		VREV32	B6.B16, B6.B16
   549  		VADD	B7.S4, INC.S4, CTR.S4
   550  		VREV32	B7.B16, B7.B16
   551  
   552  		aesrndx8(K0)
   553  		aesrndx8(K1)
   554  		aesrndx8(K2)
   555  		aesrndx8(K3)
   556  		aesrndx8(K4)
   557  		aesrndx8(K5)
   558  		aesrndx8(K6)
   559  		aesrndx8(K7)
   560  		TBZ	$4, NR, octetsFinish
   561  		aesrndx8(K10)
   562  		aesrndx8(K11)
   563  		TBZ	$3, NR, octetsFinish
   564  		VLD1.P	32(curK), [T1.B16, T2.B16]
   565  		aesrndx8(T1)
   566  		aesrndx8(T2)
   567  		MOVD	H0, curK
   568  octetsFinish:
   569  		aesrndx8(K8)
   570  		aesrndlastx8(K9)
   571  
   572  		VEOR	KLAST.B16, B0.B16, B0.B16
   573  		VEOR	KLAST.B16, B1.B16, B1.B16
   574  		VEOR	KLAST.B16, B2.B16, B2.B16
   575  		VEOR	KLAST.B16, B3.B16, B3.B16
   576  		VEOR	KLAST.B16, B4.B16, B4.B16
   577  		VEOR	KLAST.B16, B5.B16, B5.B16
   578  		VEOR	KLAST.B16, B6.B16, B6.B16
   579  		VEOR	KLAST.B16, B7.B16, B7.B16
   580  
   581  		VLD1.P	32(srcPtr), [T1.B16, T2.B16]
   582  		VEOR	B0.B16, T1.B16, B0.B16
   583  		VEOR	B1.B16, T2.B16, B1.B16
   584  		VST1.P  [B0.B16, B1.B16], 32(dstPtr)
   585  		VLD1.P	32(srcPtr), [T1.B16, T2.B16]
   586  		VEOR	B2.B16, T1.B16, B2.B16
   587  		VEOR	B3.B16, T2.B16, B3.B16
   588  		VST1.P  [B2.B16, B3.B16], 32(dstPtr)
   589  		VLD1.P	32(srcPtr), [T1.B16, T2.B16]
   590  		VEOR	B4.B16, T1.B16, B4.B16
   591  		VEOR	B5.B16, T2.B16, B5.B16
   592  		VST1.P  [B4.B16, B5.B16], 32(dstPtr)
   593  		VLD1.P	32(srcPtr), [T1.B16, T2.B16]
   594  		VEOR	B6.B16, T1.B16, B6.B16
   595  		VEOR	B7.B16, T2.B16, B7.B16
   596  		VST1.P  [B6.B16, B7.B16], 32(dstPtr)
   597  
   598  		VLD1.P	32(pTbl), [T1.B16, T2.B16]
   599  		VREV64	B0.B16, B0.B16
   600  		VEOR	ACC0.B16, B0.B16, B0.B16
   601  		VEXT	$8, B0.B16, B0.B16, T0.B16
   602  		VEOR	B0.B16, T0.B16, T0.B16
   603  		VPMULL	B0.D1, T1.D1, ACC1.Q1
   604  		VPMULL2	B0.D2, T1.D2, ACC0.Q1
   605  		VPMULL	T0.D1, T2.D1, ACCM.Q1
   606  
   607  		mulRound(B1)
   608  		mulRound(B2)
   609  		mulRound(B3)
   610  		mulRound(B4)
   611  		mulRound(B5)
   612  		mulRound(B6)
   613  		mulRound(B7)
   614  		MOVD	pTblSave, pTbl
   615  		reduce()
   616  
   617  		CMP	$128, srcPtrLen
   618  		BGE	octetsLoop
   619  
   620  startSingles:
   621  	CBZ	srcPtrLen, done
   622  	ADD	$14*16, pTbl
   623  	// Preload H and its Karatsuba precomp
   624  	VLD1.P	(pTbl), [T1.B16, T2.B16]
   625  	// Preload AES round keys
   626  	ADD	$128, ks
   627  	VLD1.P	48(ks), [K8.B16, K9.B16, K10.B16]
   628  	VMOV	K10.B16, KLAST.B16
   629  	TBZ	$4, NR, singlesLoop
   630  	VLD1.P	32(ks), [B1.B16, B2.B16]
   631  	VMOV	B2.B16, KLAST.B16
   632  	TBZ	$3, NR, singlesLoop
   633  	VLD1.P	32(ks), [B3.B16, B4.B16]
   634  	VMOV	B4.B16, KLAST.B16
   635  
   636  singlesLoop:
   637  		CMP	$16, srcPtrLen
   638  		BLT	tail
   639  		SUB	$16, srcPtrLen
   640  
   641  		VLD1.P	16(srcPtr), [T0.B16]
   642  		VEOR	KLAST.B16, T0.B16, T0.B16
   643  
   644  		VREV32	CTR.B16, B0.B16
   645  		VADD	CTR.S4, INC.S4, CTR.S4
   646  
   647  		AESE	K0.B16, B0.B16
   648  		AESMC	B0.B16, B0.B16
   649  		AESE	K1.B16, B0.B16
   650  		AESMC	B0.B16, B0.B16
   651  		AESE	K2.B16, B0.B16
   652  		AESMC	B0.B16, B0.B16
   653  		AESE	K3.B16, B0.B16
   654  		AESMC	B0.B16, B0.B16
   655  		AESE	K4.B16, B0.B16
   656  		AESMC	B0.B16, B0.B16
   657  		AESE	K5.B16, B0.B16
   658  		AESMC	B0.B16, B0.B16
   659  		AESE	K6.B16, B0.B16
   660  		AESMC	B0.B16, B0.B16
   661  		AESE	K7.B16, B0.B16
   662  		AESMC	B0.B16, B0.B16
   663  		AESE	K8.B16, B0.B16
   664  		AESMC	B0.B16, B0.B16
   665  		AESE	K9.B16, B0.B16
   666  		TBZ	$4, NR, singlesLast
   667  		AESMC	B0.B16, B0.B16
   668  		AESE	K10.B16, B0.B16
   669  		AESMC	B0.B16, B0.B16
   670  		AESE	B1.B16, B0.B16
   671  		TBZ	$3, NR, singlesLast
   672  		AESMC	B0.B16, B0.B16
   673  		AESE	B2.B16, B0.B16
   674  		AESMC	B0.B16, B0.B16
   675  		AESE	B3.B16, B0.B16
   676  singlesLast:
   677  		VEOR	T0.B16, B0.B16, B0.B16
   678  
   679  		VST1.P	[B0.B16], 16(dstPtr)
   680  encReduce:
   681  		VREV64	B0.B16, B0.B16
   682  		VEOR	ACC0.B16, B0.B16, B0.B16
   683  
   684  		VEXT	$8, B0.B16, B0.B16, T0.B16
   685  		VEOR	B0.B16, T0.B16, T0.B16
   686  		VPMULL	B0.D1, T1.D1, ACC1.Q1
   687  		VPMULL2	B0.D2, T1.D2, ACC0.Q1
   688  		VPMULL	T0.D1, T2.D1, ACCM.Q1
   689  
   690  		reduce()
   691  
   692  	B	singlesLoop
   693  tail:
   694  	CBZ	srcPtrLen, done
   695  
   696  	tailLoad(T0)
   697  
   698  	VEOR	KLAST.B16, T0.B16, T0.B16
   699  	VREV32	CTR.B16, B0.B16
   700  
   701  	AESE	K0.B16, B0.B16
   702  	AESMC	B0.B16, B0.B16
   703  	AESE	K1.B16, B0.B16
   704  	AESMC	B0.B16, B0.B16
   705  	AESE	K2.B16, B0.B16
   706  	AESMC	B0.B16, B0.B16
   707  	AESE	K3.B16, B0.B16
   708  	AESMC	B0.B16, B0.B16
   709  	AESE	K4.B16, B0.B16
   710  	AESMC	B0.B16, B0.B16
   711  	AESE	K5.B16, B0.B16
   712  	AESMC	B0.B16, B0.B16
   713  	AESE	K6.B16, B0.B16
   714  	AESMC	B0.B16, B0.B16
   715  	AESE	K7.B16, B0.B16
   716  	AESMC	B0.B16, B0.B16
   717  	AESE	K8.B16, B0.B16
   718  	AESMC	B0.B16, B0.B16
   719  	AESE	K9.B16, B0.B16
   720  	TBZ	$4, NR, tailLast
   721  	AESMC	B0.B16, B0.B16
   722  	AESE	K10.B16, B0.B16
   723  	AESMC	B0.B16, B0.B16
   724  	AESE	B1.B16, B0.B16
   725  	TBZ	$3, NR, tailLast
   726  	AESMC	B0.B16, B0.B16
   727  	AESE	B2.B16, B0.B16
   728  	AESMC	B0.B16, B0.B16
   729  	AESE	B3.B16, B0.B16
   730  
   731  tailLast:
   732  	VEOR	T0.B16, B0.B16, B0.B16
   733  	VAND	T3.B16, B0.B16, B0.B16
   734  
   735  	// Store from a copy, since tailStore clobbers its argument and
   736  	// B0 is the GHASH input of encReduce.
   737  	VMOV	B0.B16, T0.B16
   738  	tailStore(T0)
   739  	MOVD	ZR, srcPtrLen
   740  
   741  	B	encReduce
   742  
   743  done:
   744  	VST1	[ACC0.B16], (tPtr)
   745  	RET
   746  
   747  // func gcmAesDec(productTable *[256]byte, dst, src []byte, ctr, T *[16]byte, ks []uint32)
   748  TEXT ·gcmAesDec(SB),NOSPLIT,$0
   749  	MOVD	productTable+0(FP), pTbl
   750  	MOVD	dst+8(FP), dstPtr
   751  	MOVD	src_base+32(FP), srcPtr
   752  	MOVD	src_len+40(FP), srcPtrLen
   753  	MOVD	ctr+56(FP), ctrPtr
   754  	MOVD	T+64(FP), tPtr
   755  	MOVD	ks_base+72(FP), ks
   756  	MOVD	ks_len+80(FP), NR
   757  
   758  	MOVD	$0xC2, H1
   759  	LSL	$56, H1
   760  	MOVD	$1, H0
   761  	VMOV	H1, POLY.D[0]
   762  	VMOV	H0, POLY.D[1]
   763  	VEOR	ZERO.B16, ZERO.B16, ZERO.B16
   764  	// Compute NR from len(ks)
   765  	MOVD	pTbl, pTblSave
   766  	// Current tag, after AAD
   767  	VLD1	(tPtr), [ACC0.B16]
   768  	VEOR	ACC1.B16, ACC1.B16, ACC1.B16
   769  	VEOR	ACCM.B16, ACCM.B16, ACCM.B16
   770  	// Prepare initial counter, and the increment vector
   771  	VLD1	(ctrPtr), [CTR.B16]
   772  	VEOR	INC.B16, INC.B16, INC.B16
   773  	MOVD	$1, H0
   774  	VMOV	H0, INC.S[3]
   775  	VREV32	CTR.B16, CTR.B16
   776  	VADD	CTR.S4, INC.S4, CTR.S4
   777  
   778  	MOVD	ks, H0
   779  	// For AES-128 round keys are stored in: K0 .. K10, KLAST
   780  	VLD1.P	64(H0), [K0.B16, K1.B16, K2.B16, K3.B16]
   781  	VLD1.P	64(H0), [K4.B16, K5.B16, K6.B16, K7.B16]
   782  	VLD1.P	48(H0), [K8.B16, K9.B16, K10.B16]
   783  	VMOV	K10.B16, KLAST.B16
   784  
   785  	// Skip to <8 blocks loop
   786  	CMP	$128, srcPtrLen
   787  	BLT	startSingles
   788  	// There are at least 8 blocks to encrypt
   789  	TBZ	$4, NR, octetsLoop
   790  
   791  	// For AES-192 round keys occupy: K0 .. K7, K10, K11, K8, K9, KLAST
   792  	VMOV	K8.B16, K10.B16
   793  	VMOV	K9.B16, K11.B16
   794  	VMOV	KLAST.B16, K8.B16
   795  	VLD1.P	16(H0), [K9.B16]
   796  	VLD1.P  16(H0), [KLAST.B16]
   797  	TBZ	$3, NR, octetsLoop
   798  	// For AES-256 round keys occupy: K0 .. K7, K10, K11, mem, mem, K8, K9, KLAST
   799  	VMOV	KLAST.B16, K8.B16
   800  	VLD1.P	16(H0), [K9.B16]
   801  	VLD1.P  16(H0), [KLAST.B16]
   802  	ADD	$10*16, ks, H0
   803  	MOVD	H0, curK
   804  
   805  octetsLoop:
   806  		SUB	$128, srcPtrLen
   807  
   808  		VMOV	CTR.B16, B0.B16
   809  		VADD	B0.S4, INC.S4, B1.S4
   810  		VREV32	B0.B16, B0.B16
   811  		VADD	B1.S4, INC.S4, B2.S4
   812  		VREV32	B1.B16, B1.B16
   813  		VADD	B2.S4, INC.S4, B3.S4
   814  		VREV32	B2.B16, B2.B16
   815  		VADD	B3.S4, INC.S4, B4.S4
   816  		VREV32	B3.B16, B3.B16
   817  		VADD	B4.S4, INC.S4, B5.S4
   818  		VREV32	B4.B16, B4.B16
   819  		VADD	B5.S4, INC.S4, B6.S4
   820  		VREV32	B5.B16, B5.B16
   821  		VADD	B6.S4, INC.S4, B7.S4
   822  		VREV32	B6.B16, B6.B16
   823  		VADD	B7.S4, INC.S4, CTR.S4
   824  		VREV32	B7.B16, B7.B16
   825  
   826  		aesrndx8(K0)
   827  		aesrndx8(K1)
   828  		aesrndx8(K2)
   829  		aesrndx8(K3)
   830  		aesrndx8(K4)
   831  		aesrndx8(K5)
   832  		aesrndx8(K6)
   833  		aesrndx8(K7)
   834  		TBZ	$4, NR, octetsFinish
   835  		aesrndx8(K10)
   836  		aesrndx8(K11)
   837  		TBZ	$3, NR, octetsFinish
   838  		VLD1.P	32(curK), [T1.B16, T2.B16]
   839  		aesrndx8(T1)
   840  		aesrndx8(T2)
   841  		MOVD	H0, curK
   842  octetsFinish:
   843  		aesrndx8(K8)
   844  		aesrndlastx8(K9)
   845  
   846  		VEOR	KLAST.B16, B0.B16, T1.B16
   847  		VEOR	KLAST.B16, B1.B16, T2.B16
   848  		VEOR	KLAST.B16, B2.B16, B2.B16
   849  		VEOR	KLAST.B16, B3.B16, B3.B16
   850  		VEOR	KLAST.B16, B4.B16, B4.B16
   851  		VEOR	KLAST.B16, B5.B16, B5.B16
   852  		VEOR	KLAST.B16, B6.B16, B6.B16
   853  		VEOR	KLAST.B16, B7.B16, B7.B16
   854  
   855  		VLD1.P	32(srcPtr), [B0.B16, B1.B16]
   856  		VEOR	B0.B16, T1.B16, T1.B16
   857  		VEOR	B1.B16, T2.B16, T2.B16
   858  		VST1.P  [T1.B16, T2.B16], 32(dstPtr)
   859  
   860  		VLD1.P	32(pTbl), [T1.B16, T2.B16]
   861  		VREV64	B0.B16, B0.B16
   862  		VEOR	ACC0.B16, B0.B16, B0.B16
   863  		VEXT	$8, B0.B16, B0.B16, T0.B16
   864  		VEOR	B0.B16, T0.B16, T0.B16
   865  		VPMULL	B0.D1, T1.D1, ACC1.Q1
   866  		VPMULL2	B0.D2, T1.D2, ACC0.Q1
   867  		VPMULL	T0.D1, T2.D1, ACCM.Q1
   868  		mulRound(B1)
   869  
   870  		VLD1.P	32(srcPtr), [B0.B16, B1.B16]
   871  		VEOR	B2.B16, B0.B16, T1.B16
   872  		VEOR	B3.B16, B1.B16, T2.B16
   873  		VST1.P  [T1.B16, T2.B16], 32(dstPtr)
   874  		mulRound(B0)
   875  		mulRound(B1)
   876  
   877  		VLD1.P	32(srcPtr), [B0.B16, B1.B16]
   878  		VEOR	B4.B16, B0.B16, T1.B16
   879  		VEOR	B5.B16, B1.B16, T2.B16
   880  		VST1.P  [T1.B16, T2.B16], 32(dstPtr)
   881  		mulRound(B0)
   882  		mulRound(B1)
   883  
   884  		VLD1.P	32(srcPtr), [B0.B16, B1.B16]
   885  		VEOR	B6.B16, B0.B16, T1.B16
   886  		VEOR	B7.B16, B1.B16, T2.B16
   887  		VST1.P  [T1.B16, T2.B16], 32(dstPtr)
   888  		mulRound(B0)
   889  		mulRound(B1)
   890  
   891  		MOVD	pTblSave, pTbl
   892  		reduce()
   893  
   894  		CMP	$128, srcPtrLen
   895  		BGE	octetsLoop
   896  
   897  startSingles:
   898  	CBZ	srcPtrLen, done
   899  	ADD	$14*16, pTbl
   900  	// Preload H and its Karatsuba precomp
   901  	VLD1.P	(pTbl), [T1.B16, T2.B16]
   902  	// Preload AES round keys
   903  	ADD	$128, ks
   904  	VLD1.P	48(ks), [K8.B16, K9.B16, K10.B16]
   905  	VMOV	K10.B16, KLAST.B16
   906  	TBZ	$4, NR, singlesLoop
   907  	VLD1.P	32(ks), [B1.B16, B2.B16]
   908  	VMOV	B2.B16, KLAST.B16
   909  	TBZ	$3, NR, singlesLoop
   910  	VLD1.P	32(ks), [B3.B16, B4.B16]
   911  	VMOV	B4.B16, KLAST.B16
   912  
   913  singlesLoop:
   914  		CMP	$16, srcPtrLen
   915  		BLT	tail
   916  		SUB	$16, srcPtrLen
   917  
   918  		VLD1.P	16(srcPtr), [T0.B16]
   919  		VREV64	T0.B16, B5.B16
   920  		VEOR	KLAST.B16, T0.B16, T0.B16
   921  
   922  		VREV32	CTR.B16, B0.B16
   923  		VADD	CTR.S4, INC.S4, CTR.S4
   924  
   925  		AESE	K0.B16, B0.B16
   926  		AESMC	B0.B16, B0.B16
   927  		AESE	K1.B16, B0.B16
   928  		AESMC	B0.B16, B0.B16
   929  		AESE	K2.B16, B0.B16
   930  		AESMC	B0.B16, B0.B16
   931  		AESE	K3.B16, B0.B16
   932  		AESMC	B0.B16, B0.B16
   933  		AESE	K4.B16, B0.B16
   934  		AESMC	B0.B16, B0.B16
   935  		AESE	K5.B16, B0.B16
   936  		AESMC	B0.B16, B0.B16
   937  		AESE	K6.B16, B0.B16
   938  		AESMC	B0.B16, B0.B16
   939  		AESE	K7.B16, B0.B16
   940  		AESMC	B0.B16, B0.B16
   941  		AESE	K8.B16, B0.B16
   942  		AESMC	B0.B16, B0.B16
   943  		AESE	K9.B16, B0.B16
   944  		TBZ	$4, NR, singlesLast
   945  		AESMC	B0.B16, B0.B16
   946  		AESE	K10.B16, B0.B16
   947  		AESMC	B0.B16, B0.B16
   948  		AESE	B1.B16, B0.B16
   949  		TBZ	$3, NR, singlesLast
   950  		AESMC	B0.B16, B0.B16
   951  		AESE	B2.B16, B0.B16
   952  		AESMC	B0.B16, B0.B16
   953  		AESE	B3.B16, B0.B16
   954  singlesLast:
   955  		VEOR	T0.B16, B0.B16, B0.B16
   956  
   957  		VST1.P	[B0.B16], 16(dstPtr)
   958  
   959  		VEOR	ACC0.B16, B5.B16, B5.B16
   960  		VEXT	$8, B5.B16, B5.B16, T0.B16
   961  		VEOR	B5.B16, T0.B16, T0.B16
   962  		VPMULL	B5.D1, T1.D1, ACC1.Q1
   963  		VPMULL2	B5.D2, T1.D2, ACC0.Q1
   964  		VPMULL	T0.D1, T2.D1, ACCM.Q1
   965  		reduce()
   966  
   967  	B	singlesLoop
   968  tail:
   969  	CBZ	srcPtrLen, done
   970  
   971  	VREV32	CTR.B16, B0.B16
   972  	VADD	CTR.S4, INC.S4, CTR.S4
   973  
   974  	AESE	K0.B16, B0.B16
   975  	AESMC	B0.B16, B0.B16
   976  	AESE	K1.B16, B0.B16
   977  	AESMC	B0.B16, B0.B16
   978  	AESE	K2.B16, B0.B16
   979  	AESMC	B0.B16, B0.B16
   980  	AESE	K3.B16, B0.B16
   981  	AESMC	B0.B16, B0.B16
   982  	AESE	K4.B16, B0.B16
   983  	AESMC	B0.B16, B0.B16
   984  	AESE	K5.B16, B0.B16
   985  	AESMC	B0.B16, B0.B16
   986  	AESE	K6.B16, B0.B16
   987  	AESMC	B0.B16, B0.B16
   988  	AESE	K7.B16, B0.B16
   989  	AESMC	B0.B16, B0.B16
   990  	AESE	K8.B16, B0.B16
   991  	AESMC	B0.B16, B0.B16
   992  	AESE	K9.B16, B0.B16
   993  	TBZ	$4, NR, tailLast
   994  	AESMC	B0.B16, B0.B16
   995  	AESE	K10.B16, B0.B16
   996  	AESMC	B0.B16, B0.B16
   997  	AESE	B1.B16, B0.B16
   998  	TBZ	$3, NR, tailLast
   999  	AESMC	B0.B16, B0.B16
  1000  	AESE	B2.B16, B0.B16
  1001  	AESMC	B0.B16, B0.B16
  1002  	AESE	B3.B16, B0.B16
  1003  tailLast:
  1004  	VEOR	KLAST.B16, B0.B16, B0.B16
  1005  
  1006  	tailLoad(B5)
  1007  
  1008  	VEOR	B5.B16, B0.B16, B0.B16
  1009  
  1010  	tailStore(B0)
  1011  
  1012  	VREV64	B5.B16, B5.B16
  1013  
  1014  	VEOR	ACC0.B16, B5.B16, B5.B16
  1015  	VEXT	$8, B5.B16, B5.B16, T0.B16
  1016  	VEOR	B5.B16, T0.B16, T0.B16
  1017  	VPMULL	B5.D1, T1.D1, ACC1.Q1
  1018  	VPMULL2	B5.D2, T1.D2, ACC0.Q1
  1019  	VPMULL	T0.D1, T2.D1, ACCM.Q1
  1020  	reduce()
  1021  done:
  1022  	VST1	[ACC0.B16], (tPtr)
  1023  
  1024  	RET
  1025  

View as plain text