1 // Copyright 2018 The Go Authors. All rights reserved.
2 // Use of this source code is governed by a BSD-style
3 // license that can be found in the LICENSE file.
4
5 //go:build !purego
6
7 #include "textflag.h"
8
9 #define B0 V0
10 #define B1 V1
11 #define B2 V2
12 #define B3 V3
13 #define B4 V4
14 #define B5 V5
15 #define B6 V6
16 #define B7 V7
17
18 #define ACC0 V8
19 #define ACC1 V9
20 #define ACCM V10
21
22 #define T0 V11
23 #define T1 V12
24 #define T2 V13
25 #define T3 V14
26
27 #define POLY V15
28 #define ZERO V16
29 #define INC V17
30 #define CTR V18
31
32 #define K0 V19
33 #define K1 V20
34 #define K2 V21
35 #define K3 V22
36 #define K4 V23
37 #define K5 V24
38 #define K6 V25
39 #define K7 V26
40 #define K8 V27
41 #define K9 V28
42 #define K10 V29
43 #define K11 V30
44 #define KLAST V31
45
46 #define reduce() \
47 VEOR ACC0.B16, ACCM.B16, ACCM.B16 \
48 VEOR ACC1.B16, ACCM.B16, ACCM.B16 \
49 VEXT $8, ZERO.B16, ACCM.B16, T0.B16 \
50 VEXT $8, ACCM.B16, ZERO.B16, ACCM.B16 \
51 VEOR ACCM.B16, ACC0.B16, ACC0.B16 \
52 VEOR T0.B16, ACC1.B16, ACC1.B16 \
53 VPMULL POLY.D1, ACC0.D1, T0.Q1 \
54 VEXT $8, ACC0.B16, ACC0.B16, ACC0.B16 \
55 VEOR T0.B16, ACC0.B16, ACC0.B16 \
56 VPMULL POLY.D1, ACC0.D1, T0.Q1 \
57 VEOR T0.B16, ACC1.B16, ACC1.B16 \
58 VEXT $8, ACC1.B16, ACC1.B16, ACC1.B16 \
59 VEOR ACC1.B16, ACC0.B16, ACC0.B16 \
60
61 // func gcmAesFinish(productTable *[256]byte, tagMask, T *[16]byte, pLen, dLen uint64)
62 TEXT ·gcmAesFinish(SB),NOSPLIT,$0
63 #define pTbl R0
64 #define tMsk R1
65 #define tPtr R2
66 #define plen R3
67 #define dlen R4
68
69 MOVD $0xC2, R1
70 LSL $56, R1
71 MOVD $1, R0
72 VMOV R1, POLY.D[0]
73 VMOV R0, POLY.D[1]
74 VEOR ZERO.B16, ZERO.B16, ZERO.B16
75
76 MOVD productTable+0(FP), pTbl
77 MOVD tagMask+8(FP), tMsk
78 MOVD T+16(FP), tPtr
79 MOVD pLen+24(FP), plen
80 MOVD dLen+32(FP), dlen
81
82 VLD1 (tPtr), [ACC0.B16]
83 VLD1 (tMsk), [B1.B16]
84
85 LSL $3, plen
86 LSL $3, dlen
87
88 VMOV dlen, B0.D[0]
89 VMOV plen, B0.D[1]
90
91 ADD $14*16, pTbl
92 VLD1.P (pTbl), [T1.B16, T2.B16]
93
94 VEOR ACC0.B16, B0.B16, B0.B16
95
96 VEXT $8, B0.B16, B0.B16, T0.B16
97 VEOR B0.B16, T0.B16, T0.B16
98 VPMULL B0.D1, T1.D1, ACC1.Q1
99 VPMULL2 B0.D2, T1.D2, ACC0.Q1
100 VPMULL T0.D1, T2.D1, ACCM.Q1
101
102 reduce()
103
104 VREV64 ACC0.B16, ACC0.B16
105 VEOR B1.B16, ACC0.B16, ACC0.B16
106
107 VST1 [ACC0.B16], (tPtr)
108 RET
109 #undef pTbl
110 #undef tMsk
111 #undef tPtr
112 #undef plen
113 #undef dlen
114
115 // func gcmAesInit(productTable *[256]byte, ks []uint32)
116 TEXT ·gcmAesInit(SB),NOSPLIT,$0
117 #define pTbl R0
118 #define KS R1
119 #define NR R2
120 #define I R3
121 MOVD productTable+0(FP), pTbl
122 MOVD ks_base+8(FP), KS
123 MOVD ks_len+16(FP), NR
124
125 MOVD $0xC2, I
126 LSL $56, I
127 VMOV I, POLY.D[0]
128 MOVD $1, I
129 VMOV I, POLY.D[1]
130 VEOR ZERO.B16, ZERO.B16, ZERO.B16
131
132 // Encrypt block 0 with the AES key to generate the hash key H
133 VLD1.P 64(KS), [T0.B16, T1.B16, T2.B16, T3.B16]
134 VEOR B0.B16, B0.B16, B0.B16
135 AESE T0.B16, B0.B16
136 AESMC B0.B16, B0.B16
137 AESE T1.B16, B0.B16
138 AESMC B0.B16, B0.B16
139 AESE T2.B16, B0.B16
140 AESMC B0.B16, B0.B16
141 AESE T3.B16, B0.B16
142 AESMC B0.B16, B0.B16
143 VLD1.P 64(KS), [T0.B16, T1.B16, T2.B16, T3.B16]
144 AESE T0.B16, B0.B16
145 AESMC B0.B16, B0.B16
146 AESE T1.B16, B0.B16
147 AESMC B0.B16, B0.B16
148 AESE T2.B16, B0.B16
149 AESMC B0.B16, B0.B16
150 AESE T3.B16, B0.B16
151 AESMC B0.B16, B0.B16
152 TBZ $4, NR, initEncFinish
153 VLD1.P 32(KS), [T0.B16, T1.B16]
154 AESE T0.B16, B0.B16
155 AESMC B0.B16, B0.B16
156 AESE T1.B16, B0.B16
157 AESMC B0.B16, B0.B16
158 TBZ $3, NR, initEncFinish
159 VLD1.P 32(KS), [T0.B16, T1.B16]
160 AESE T0.B16, B0.B16
161 AESMC B0.B16, B0.B16
162 AESE T1.B16, B0.B16
163 AESMC B0.B16, B0.B16
164 initEncFinish:
165 VLD1 (KS), [T0.B16, T1.B16, T2.B16]
166 AESE T0.B16, B0.B16
167 AESMC B0.B16, B0.B16
168 AESE T1.B16, B0.B16
169 VEOR T2.B16, B0.B16, B0.B16
170
171 VREV64 B0.B16, B0.B16
172
173 // Multiply by 2 modulo P
174 VMOV B0.D[0], I
175 ASR $63, I
176 VMOV I, T1.D[0]
177 VMOV I, T1.D[1]
178 VAND POLY.B16, T1.B16, T1.B16
179 VUSHR $63, B0.D2, T2.D2
180 VEXT $8, ZERO.B16, T2.B16, T2.B16
181 VSHL $1, B0.D2, B0.D2
182 VEOR T1.B16, B0.B16, B0.B16
183 VEOR T2.B16, B0.B16, B0.B16 // Can avoid this when VSLI is available
184
185 // Karatsuba pre-computation
186 VEXT $8, B0.B16, B0.B16, B1.B16
187 VEOR B0.B16, B1.B16, B1.B16
188
189 ADD $14*16, pTbl
190 VST1 [B0.B16, B1.B16], (pTbl)
191 SUB $2*16, pTbl
192
193 VMOV B0.B16, B2.B16
194 VMOV B1.B16, B3.B16
195
196 MOVD $7, I
197
198 initLoop:
199 // Compute powers of H
200 SUBS $1, I
201
202 VPMULL B0.D1, B2.D1, T1.Q1
203 VPMULL2 B0.D2, B2.D2, T0.Q1
204 VPMULL B1.D1, B3.D1, T2.Q1
205 VEOR T0.B16, T2.B16, T2.B16
206 VEOR T1.B16, T2.B16, T2.B16
207 VEXT $8, ZERO.B16, T2.B16, T3.B16
208 VEXT $8, T2.B16, ZERO.B16, T2.B16
209 VEOR T2.B16, T0.B16, T0.B16
210 VEOR T3.B16, T1.B16, T1.B16
211 VPMULL POLY.D1, T0.D1, T2.Q1
212 VEXT $8, T0.B16, T0.B16, T0.B16
213 VEOR T2.B16, T0.B16, T0.B16
214 VPMULL POLY.D1, T0.D1, T2.Q1
215 VEXT $8, T0.B16, T0.B16, T0.B16
216 VEOR T2.B16, T0.B16, T0.B16
217 VEOR T1.B16, T0.B16, B2.B16
218 VMOV B2.B16, B3.B16
219 VEXT $8, B2.B16, B2.B16, B2.B16
220 VEOR B2.B16, B3.B16, B3.B16
221
222 VST1 [B2.B16, B3.B16], (pTbl)
223 SUB $2*16, pTbl
224
225 BNE initLoop
226 RET
227 #undef I
228 #undef NR
229 #undef KS
230 #undef pTbl
231
232 // func gcmAesData(productTable *[256]byte, data []byte, T *[16]byte)
233 TEXT ·gcmAesData(SB),NOSPLIT,$0
234 #define pTbl R0
235 #define aut R1
236 #define tPtr R2
237 #define autLen R3
238 #define H0 R4
239 #define pTblSave R5
240
241 #define mulRound(X) \
242 VLD1.P 32(pTbl), [T1.B16, T2.B16] \
243 VREV64 X.B16, X.B16 \
244 VEXT $8, X.B16, X.B16, T0.B16 \
245 VEOR X.B16, T0.B16, T0.B16 \
246 VPMULL X.D1, T1.D1, T3.Q1 \
247 VEOR T3.B16, ACC1.B16, ACC1.B16 \
248 VPMULL2 X.D2, T1.D2, T3.Q1 \
249 VEOR T3.B16, ACC0.B16, ACC0.B16 \
250 VPMULL T0.D1, T2.D1, T3.Q1 \
251 VEOR T3.B16, ACCM.B16, ACCM.B16
252
253 MOVD productTable+0(FP), pTbl
254 MOVD data_base+8(FP), aut
255 MOVD data_len+16(FP), autLen
256 MOVD T+32(FP), tPtr
257
258 VEOR ACC0.B16, ACC0.B16, ACC0.B16
259 CBZ autLen, dataBail
260
261 MOVD $0xC2, H0
262 LSL $56, H0
263 VMOV H0, POLY.D[0]
264 MOVD $1, H0
265 VMOV H0, POLY.D[1]
266 VEOR ZERO.B16, ZERO.B16, ZERO.B16
267 MOVD pTbl, pTblSave
268
269 CMP $13, autLen
270 BEQ dataTLS
271 CMP $128, autLen
272 BLT startSinglesLoop
273 B octetsLoop
274
275 dataTLS:
276 ADD $14*16, pTbl
277 VLD1.P (pTbl), [T1.B16, T2.B16]
278 VEOR B0.B16, B0.B16, B0.B16
279
280 MOVD (aut), H0
281 VMOV H0, B0.D[0]
282 MOVW 8(aut), H0
283 VMOV H0, B0.S[2]
284 MOVB 12(aut), H0
285 VMOV H0, B0.B[12]
286
287 MOVD $0, autLen
288 B dataMul
289
290 octetsLoop:
291 CMP $128, autLen
292 BLT startSinglesLoop
293 SUB $128, autLen
294
295 VLD1.P 32(aut), [B0.B16, B1.B16]
296
297 VLD1.P 32(pTbl), [T1.B16, T2.B16]
298 VREV64 B0.B16, B0.B16
299 VEOR ACC0.B16, B0.B16, B0.B16
300 VEXT $8, B0.B16, B0.B16, T0.B16
301 VEOR B0.B16, T0.B16, T0.B16
302 VPMULL B0.D1, T1.D1, ACC1.Q1
303 VPMULL2 B0.D2, T1.D2, ACC0.Q1
304 VPMULL T0.D1, T2.D1, ACCM.Q1
305
306 mulRound(B1)
307 VLD1.P 32(aut), [B2.B16, B3.B16]
308 mulRound(B2)
309 mulRound(B3)
310 VLD1.P 32(aut), [B4.B16, B5.B16]
311 mulRound(B4)
312 mulRound(B5)
313 VLD1.P 32(aut), [B6.B16, B7.B16]
314 mulRound(B6)
315 mulRound(B7)
316
317 MOVD pTblSave, pTbl
318 reduce()
319 B octetsLoop
320
321 startSinglesLoop:
322
323 ADD $14*16, pTbl
324 VLD1.P (pTbl), [T1.B16, T2.B16]
325
326 singlesLoop:
327
328 CMP $16, autLen
329 BLT dataEnd
330 SUB $16, autLen
331
332 VLD1.P 16(aut), [B0.B16]
333 dataMul:
334 VREV64 B0.B16, B0.B16
335 VEOR ACC0.B16, B0.B16, B0.B16
336
337 VEXT $8, B0.B16, B0.B16, T0.B16
338 VEOR B0.B16, T0.B16, T0.B16
339 VPMULL B0.D1, T1.D1, ACC1.Q1
340 VPMULL2 B0.D2, T1.D2, ACC0.Q1
341 VPMULL T0.D1, T2.D1, ACCM.Q1
342
343 reduce()
344
345 B singlesLoop
346
347 dataEnd:
348
349 CBZ autLen, dataBail
350 VEOR B0.B16, B0.B16, B0.B16
351 ADD autLen, aut
352
353 dataLoadLoop:
354 MOVB.W -1(aut), H0
355 VEXT $15, B0.B16, ZERO.B16, B0.B16
356 VMOV H0, B0.B[0]
357 SUBS $1, autLen
358 BNE dataLoadLoop
359 B dataMul
360
361 dataBail:
362 VST1 [ACC0.B16], (tPtr)
363 RET
364
365 #undef pTbl
366 #undef aut
367 #undef tPtr
368 #undef autLen
369 #undef H0
370 #undef pTblSave
371
372 // func gcmAesEnc(productTable *[256]byte, dst, src []byte, ctr, T *[16]byte, ks []uint32)
373 TEXT ·gcmAesEnc(SB),NOSPLIT,$0
374 #define pTbl R0
375 #define dstPtr R1
376 #define ctrPtr R2
377 #define srcPtr R3
378 #define ks R4
379 #define tPtr R5
380 #define srcPtrLen R6
381 #define NR R10
382 #define H0 R11
383 #define H1 R12
384 #define curK R13
385 #define pTblSave R14
386
387 #define aesrndx8(K) \
388 AESE K.B16, B0.B16 \
389 AESMC B0.B16, B0.B16 \
390 AESE K.B16, B1.B16 \
391 AESMC B1.B16, B1.B16 \
392 AESE K.B16, B2.B16 \
393 AESMC B2.B16, B2.B16 \
394 AESE K.B16, B3.B16 \
395 AESMC B3.B16, B3.B16 \
396 AESE K.B16, B4.B16 \
397 AESMC B4.B16, B4.B16 \
398 AESE K.B16, B5.B16 \
399 AESMC B5.B16, B5.B16 \
400 AESE K.B16, B6.B16 \
401 AESMC B6.B16, B6.B16 \
402 AESE K.B16, B7.B16 \
403 AESMC B7.B16, B7.B16
404
405 #define aesrndlastx8(K) \
406 AESE K.B16, B0.B16 \
407 AESE K.B16, B1.B16 \
408 AESE K.B16, B2.B16 \
409 AESE K.B16, B3.B16 \
410 AESE K.B16, B4.B16 \
411 AESE K.B16, B5.B16 \
412 AESE K.B16, B6.B16 \
413 AESE K.B16, B7.B16
414
415 // tailLoad reads the srcPtrLen bytes at srcPtr into the low bytes of
416 // X, zero-padded, loading 8, 4, 2, and 1 bytes at a time to avoid
417 // reading past the end of the source buffer. It also builds in T3 a
418 // mask of the bytes within the source length. It clobbers H0 and H1.
419 #define tailLoad(X) \
420 VEOR X.B16, X.B16, X.B16 \
421 VEOR T3.B16, T3.B16, T3.B16 \
422 MOVD $-1, H1 \
423 ADD srcPtrLen, srcPtr \
424 TBZ $3, srcPtrLen, tailLoad4 \
425 MOVD.W -8(srcPtr), H0 \
426 VMOV H0, X.D[0] \
427 VMOV H1, T3.D[0] \
428 tailLoad4: \
429 TBZ $2, srcPtrLen, tailLoad2 \
430 MOVW.W -4(srcPtr), H0 \
431 VEXT $12, X.B16, ZERO.B16, X.B16 \
432 VEXT $12, T3.B16, ZERO.B16, T3.B16 \
433 VMOV H0, X.S[0] \
434 VMOV H1, T3.S[0] \
435 tailLoad2: \
436 TBZ $1, srcPtrLen, tailLoad1 \
437 MOVH.W -2(srcPtr), H0 \
438 VEXT $14, X.B16, ZERO.B16, X.B16 \
439 VEXT $14, T3.B16, ZERO.B16, T3.B16 \
440 VMOV H0, X.H[0] \
441 VMOV H1, T3.H[0] \
442 tailLoad1: \
443 TBZ $0, srcPtrLen, tailLoad0 \
444 MOVB.W -1(srcPtr), H0 \
445 VEXT $15, X.B16, ZERO.B16, X.B16 \
446 VEXT $15, T3.B16, ZERO.B16, T3.B16 \
447 VMOV H0, X.B[0] \
448 VMOV H1, T3.B[0] \
449 tailLoad0:
450
451 // tailStore writes the low srcPtrLen bytes of X to dstPtr, storing 8,
452 // 4, 2, and 1 bytes at a time to avoid writing past the end of the
453 // destination buffer. It clobbers X and H0.
454 #define tailStore(X) \
455 TBZ $3, srcPtrLen, tailStore4 \
456 VMOV X.D[0], H0 \
457 MOVD.P H0, 8(dstPtr) \
458 VEXT $8, ZERO.B16, X.B16, X.B16 \
459 tailStore4: \
460 TBZ $2, srcPtrLen, tailStore2 \
461 VMOV X.S[0], H0 \
462 MOVW.P H0, 4(dstPtr) \
463 VEXT $4, ZERO.B16, X.B16, X.B16 \
464 tailStore2: \
465 TBZ $1, srcPtrLen, tailStore1 \
466 VMOV X.H[0], H0 \
467 MOVH.P H0, 2(dstPtr) \
468 VEXT $2, ZERO.B16, X.B16, X.B16 \
469 tailStore1: \
470 TBZ $0, srcPtrLen, tailStore0 \
471 VMOV X.B[0], H0 \
472 MOVB.P H0, 1(dstPtr) \
473 tailStore0:
474
475 MOVD productTable+0(FP), pTbl
476 MOVD dst+8(FP), dstPtr
477 MOVD src_base+32(FP), srcPtr
478 MOVD src_len+40(FP), srcPtrLen
479 MOVD ctr+56(FP), ctrPtr
480 MOVD T+64(FP), tPtr
481 MOVD ks_base+72(FP), ks
482 MOVD ks_len+80(FP), NR
483
484 MOVD $0xC2, H1
485 LSL $56, H1
486 MOVD $1, H0
487 VMOV H1, POLY.D[0]
488 VMOV H0, POLY.D[1]
489 VEOR ZERO.B16, ZERO.B16, ZERO.B16
490 // Compute NR from len(ks)
491 MOVD pTbl, pTblSave
492 // Current tag, after AAD
493 VLD1 (tPtr), [ACC0.B16]
494 VEOR ACC1.B16, ACC1.B16, ACC1.B16
495 VEOR ACCM.B16, ACCM.B16, ACCM.B16
496 // Prepare initial counter, and the increment vector
497 VLD1 (ctrPtr), [CTR.B16]
498 VEOR INC.B16, INC.B16, INC.B16
499 MOVD $1, H0
500 VMOV H0, INC.S[3]
501 VREV32 CTR.B16, CTR.B16
502 VADD CTR.S4, INC.S4, CTR.S4
503 // Skip to <8 blocks loop
504 CMP $128, srcPtrLen
505
506 MOVD ks, H0
507 // For AES-128 round keys are stored in: K0 .. K10, KLAST
508 VLD1.P 64(H0), [K0.B16, K1.B16, K2.B16, K3.B16]
509 VLD1.P 64(H0), [K4.B16, K5.B16, K6.B16, K7.B16]
510 VLD1.P 48(H0), [K8.B16, K9.B16, K10.B16]
511 VMOV K10.B16, KLAST.B16
512
513 BLT startSingles
514 // There are at least 8 blocks to encrypt
515 TBZ $4, NR, octetsLoop
516
517 // For AES-192 round keys occupy: K0 .. K7, K10, K11, K8, K9, KLAST
518 VMOV K8.B16, K10.B16
519 VMOV K9.B16, K11.B16
520 VMOV KLAST.B16, K8.B16
521 VLD1.P 16(H0), [K9.B16]
522 VLD1.P 16(H0), [KLAST.B16]
523 TBZ $3, NR, octetsLoop
524 // For AES-256 round keys occupy: K0 .. K7, K10, K11, mem, mem, K8, K9, KLAST
525 VMOV KLAST.B16, K8.B16
526 VLD1.P 16(H0), [K9.B16]
527 VLD1.P 16(H0), [KLAST.B16]
528 ADD $10*16, ks, H0
529 MOVD H0, curK
530
531 octetsLoop:
532 SUB $128, srcPtrLen
533
534 VMOV CTR.B16, B0.B16
535 VADD B0.S4, INC.S4, B1.S4
536 VREV32 B0.B16, B0.B16
537 VADD B1.S4, INC.S4, B2.S4
538 VREV32 B1.B16, B1.B16
539 VADD B2.S4, INC.S4, B3.S4
540 VREV32 B2.B16, B2.B16
541 VADD B3.S4, INC.S4, B4.S4
542 VREV32 B3.B16, B3.B16
543 VADD B4.S4, INC.S4, B5.S4
544 VREV32 B4.B16, B4.B16
545 VADD B5.S4, INC.S4, B6.S4
546 VREV32 B5.B16, B5.B16
547 VADD B6.S4, INC.S4, B7.S4
548 VREV32 B6.B16, B6.B16
549 VADD B7.S4, INC.S4, CTR.S4
550 VREV32 B7.B16, B7.B16
551
552 aesrndx8(K0)
553 aesrndx8(K1)
554 aesrndx8(K2)
555 aesrndx8(K3)
556 aesrndx8(K4)
557 aesrndx8(K5)
558 aesrndx8(K6)
559 aesrndx8(K7)
560 TBZ $4, NR, octetsFinish
561 aesrndx8(K10)
562 aesrndx8(K11)
563 TBZ $3, NR, octetsFinish
564 VLD1.P 32(curK), [T1.B16, T2.B16]
565 aesrndx8(T1)
566 aesrndx8(T2)
567 MOVD H0, curK
568 octetsFinish:
569 aesrndx8(K8)
570 aesrndlastx8(K9)
571
572 VEOR KLAST.B16, B0.B16, B0.B16
573 VEOR KLAST.B16, B1.B16, B1.B16
574 VEOR KLAST.B16, B2.B16, B2.B16
575 VEOR KLAST.B16, B3.B16, B3.B16
576 VEOR KLAST.B16, B4.B16, B4.B16
577 VEOR KLAST.B16, B5.B16, B5.B16
578 VEOR KLAST.B16, B6.B16, B6.B16
579 VEOR KLAST.B16, B7.B16, B7.B16
580
581 VLD1.P 32(srcPtr), [T1.B16, T2.B16]
582 VEOR B0.B16, T1.B16, B0.B16
583 VEOR B1.B16, T2.B16, B1.B16
584 VST1.P [B0.B16, B1.B16], 32(dstPtr)
585 VLD1.P 32(srcPtr), [T1.B16, T2.B16]
586 VEOR B2.B16, T1.B16, B2.B16
587 VEOR B3.B16, T2.B16, B3.B16
588 VST1.P [B2.B16, B3.B16], 32(dstPtr)
589 VLD1.P 32(srcPtr), [T1.B16, T2.B16]
590 VEOR B4.B16, T1.B16, B4.B16
591 VEOR B5.B16, T2.B16, B5.B16
592 VST1.P [B4.B16, B5.B16], 32(dstPtr)
593 VLD1.P 32(srcPtr), [T1.B16, T2.B16]
594 VEOR B6.B16, T1.B16, B6.B16
595 VEOR B7.B16, T2.B16, B7.B16
596 VST1.P [B6.B16, B7.B16], 32(dstPtr)
597
598 VLD1.P 32(pTbl), [T1.B16, T2.B16]
599 VREV64 B0.B16, B0.B16
600 VEOR ACC0.B16, B0.B16, B0.B16
601 VEXT $8, B0.B16, B0.B16, T0.B16
602 VEOR B0.B16, T0.B16, T0.B16
603 VPMULL B0.D1, T1.D1, ACC1.Q1
604 VPMULL2 B0.D2, T1.D2, ACC0.Q1
605 VPMULL T0.D1, T2.D1, ACCM.Q1
606
607 mulRound(B1)
608 mulRound(B2)
609 mulRound(B3)
610 mulRound(B4)
611 mulRound(B5)
612 mulRound(B6)
613 mulRound(B7)
614 MOVD pTblSave, pTbl
615 reduce()
616
617 CMP $128, srcPtrLen
618 BGE octetsLoop
619
620 startSingles:
621 CBZ srcPtrLen, done
622 ADD $14*16, pTbl
623 // Preload H and its Karatsuba precomp
624 VLD1.P (pTbl), [T1.B16, T2.B16]
625 // Preload AES round keys
626 ADD $128, ks
627 VLD1.P 48(ks), [K8.B16, K9.B16, K10.B16]
628 VMOV K10.B16, KLAST.B16
629 TBZ $4, NR, singlesLoop
630 VLD1.P 32(ks), [B1.B16, B2.B16]
631 VMOV B2.B16, KLAST.B16
632 TBZ $3, NR, singlesLoop
633 VLD1.P 32(ks), [B3.B16, B4.B16]
634 VMOV B4.B16, KLAST.B16
635
636 singlesLoop:
637 CMP $16, srcPtrLen
638 BLT tail
639 SUB $16, srcPtrLen
640
641 VLD1.P 16(srcPtr), [T0.B16]
642 VEOR KLAST.B16, T0.B16, T0.B16
643
644 VREV32 CTR.B16, B0.B16
645 VADD CTR.S4, INC.S4, CTR.S4
646
647 AESE K0.B16, B0.B16
648 AESMC B0.B16, B0.B16
649 AESE K1.B16, B0.B16
650 AESMC B0.B16, B0.B16
651 AESE K2.B16, B0.B16
652 AESMC B0.B16, B0.B16
653 AESE K3.B16, B0.B16
654 AESMC B0.B16, B0.B16
655 AESE K4.B16, B0.B16
656 AESMC B0.B16, B0.B16
657 AESE K5.B16, B0.B16
658 AESMC B0.B16, B0.B16
659 AESE K6.B16, B0.B16
660 AESMC B0.B16, B0.B16
661 AESE K7.B16, B0.B16
662 AESMC B0.B16, B0.B16
663 AESE K8.B16, B0.B16
664 AESMC B0.B16, B0.B16
665 AESE K9.B16, B0.B16
666 TBZ $4, NR, singlesLast
667 AESMC B0.B16, B0.B16
668 AESE K10.B16, B0.B16
669 AESMC B0.B16, B0.B16
670 AESE B1.B16, B0.B16
671 TBZ $3, NR, singlesLast
672 AESMC B0.B16, B0.B16
673 AESE B2.B16, B0.B16
674 AESMC B0.B16, B0.B16
675 AESE B3.B16, B0.B16
676 singlesLast:
677 VEOR T0.B16, B0.B16, B0.B16
678
679 VST1.P [B0.B16], 16(dstPtr)
680 encReduce:
681 VREV64 B0.B16, B0.B16
682 VEOR ACC0.B16, B0.B16, B0.B16
683
684 VEXT $8, B0.B16, B0.B16, T0.B16
685 VEOR B0.B16, T0.B16, T0.B16
686 VPMULL B0.D1, T1.D1, ACC1.Q1
687 VPMULL2 B0.D2, T1.D2, ACC0.Q1
688 VPMULL T0.D1, T2.D1, ACCM.Q1
689
690 reduce()
691
692 B singlesLoop
693 tail:
694 CBZ srcPtrLen, done
695
696 tailLoad(T0)
697
698 VEOR KLAST.B16, T0.B16, T0.B16
699 VREV32 CTR.B16, B0.B16
700
701 AESE K0.B16, B0.B16
702 AESMC B0.B16, B0.B16
703 AESE K1.B16, B0.B16
704 AESMC B0.B16, B0.B16
705 AESE K2.B16, B0.B16
706 AESMC B0.B16, B0.B16
707 AESE K3.B16, B0.B16
708 AESMC B0.B16, B0.B16
709 AESE K4.B16, B0.B16
710 AESMC B0.B16, B0.B16
711 AESE K5.B16, B0.B16
712 AESMC B0.B16, B0.B16
713 AESE K6.B16, B0.B16
714 AESMC B0.B16, B0.B16
715 AESE K7.B16, B0.B16
716 AESMC B0.B16, B0.B16
717 AESE K8.B16, B0.B16
718 AESMC B0.B16, B0.B16
719 AESE K9.B16, B0.B16
720 TBZ $4, NR, tailLast
721 AESMC B0.B16, B0.B16
722 AESE K10.B16, B0.B16
723 AESMC B0.B16, B0.B16
724 AESE B1.B16, B0.B16
725 TBZ $3, NR, tailLast
726 AESMC B0.B16, B0.B16
727 AESE B2.B16, B0.B16
728 AESMC B0.B16, B0.B16
729 AESE B3.B16, B0.B16
730
731 tailLast:
732 VEOR T0.B16, B0.B16, B0.B16
733 VAND T3.B16, B0.B16, B0.B16
734
735 // Store from a copy, since tailStore clobbers its argument and
736 // B0 is the GHASH input of encReduce.
737 VMOV B0.B16, T0.B16
738 tailStore(T0)
739 MOVD ZR, srcPtrLen
740
741 B encReduce
742
743 done:
744 VST1 [ACC0.B16], (tPtr)
745 RET
746
747 // func gcmAesDec(productTable *[256]byte, dst, src []byte, ctr, T *[16]byte, ks []uint32)
748 TEXT ·gcmAesDec(SB),NOSPLIT,$0
749 MOVD productTable+0(FP), pTbl
750 MOVD dst+8(FP), dstPtr
751 MOVD src_base+32(FP), srcPtr
752 MOVD src_len+40(FP), srcPtrLen
753 MOVD ctr+56(FP), ctrPtr
754 MOVD T+64(FP), tPtr
755 MOVD ks_base+72(FP), ks
756 MOVD ks_len+80(FP), NR
757
758 MOVD $0xC2, H1
759 LSL $56, H1
760 MOVD $1, H0
761 VMOV H1, POLY.D[0]
762 VMOV H0, POLY.D[1]
763 VEOR ZERO.B16, ZERO.B16, ZERO.B16
764 // Compute NR from len(ks)
765 MOVD pTbl, pTblSave
766 // Current tag, after AAD
767 VLD1 (tPtr), [ACC0.B16]
768 VEOR ACC1.B16, ACC1.B16, ACC1.B16
769 VEOR ACCM.B16, ACCM.B16, ACCM.B16
770 // Prepare initial counter, and the increment vector
771 VLD1 (ctrPtr), [CTR.B16]
772 VEOR INC.B16, INC.B16, INC.B16
773 MOVD $1, H0
774 VMOV H0, INC.S[3]
775 VREV32 CTR.B16, CTR.B16
776 VADD CTR.S4, INC.S4, CTR.S4
777
778 MOVD ks, H0
779 // For AES-128 round keys are stored in: K0 .. K10, KLAST
780 VLD1.P 64(H0), [K0.B16, K1.B16, K2.B16, K3.B16]
781 VLD1.P 64(H0), [K4.B16, K5.B16, K6.B16, K7.B16]
782 VLD1.P 48(H0), [K8.B16, K9.B16, K10.B16]
783 VMOV K10.B16, KLAST.B16
784
785 // Skip to <8 blocks loop
786 CMP $128, srcPtrLen
787 BLT startSingles
788 // There are at least 8 blocks to encrypt
789 TBZ $4, NR, octetsLoop
790
791 // For AES-192 round keys occupy: K0 .. K7, K10, K11, K8, K9, KLAST
792 VMOV K8.B16, K10.B16
793 VMOV K9.B16, K11.B16
794 VMOV KLAST.B16, K8.B16
795 VLD1.P 16(H0), [K9.B16]
796 VLD1.P 16(H0), [KLAST.B16]
797 TBZ $3, NR, octetsLoop
798 // For AES-256 round keys occupy: K0 .. K7, K10, K11, mem, mem, K8, K9, KLAST
799 VMOV KLAST.B16, K8.B16
800 VLD1.P 16(H0), [K9.B16]
801 VLD1.P 16(H0), [KLAST.B16]
802 ADD $10*16, ks, H0
803 MOVD H0, curK
804
805 octetsLoop:
806 SUB $128, srcPtrLen
807
808 VMOV CTR.B16, B0.B16
809 VADD B0.S4, INC.S4, B1.S4
810 VREV32 B0.B16, B0.B16
811 VADD B1.S4, INC.S4, B2.S4
812 VREV32 B1.B16, B1.B16
813 VADD B2.S4, INC.S4, B3.S4
814 VREV32 B2.B16, B2.B16
815 VADD B3.S4, INC.S4, B4.S4
816 VREV32 B3.B16, B3.B16
817 VADD B4.S4, INC.S4, B5.S4
818 VREV32 B4.B16, B4.B16
819 VADD B5.S4, INC.S4, B6.S4
820 VREV32 B5.B16, B5.B16
821 VADD B6.S4, INC.S4, B7.S4
822 VREV32 B6.B16, B6.B16
823 VADD B7.S4, INC.S4, CTR.S4
824 VREV32 B7.B16, B7.B16
825
826 aesrndx8(K0)
827 aesrndx8(K1)
828 aesrndx8(K2)
829 aesrndx8(K3)
830 aesrndx8(K4)
831 aesrndx8(K5)
832 aesrndx8(K6)
833 aesrndx8(K7)
834 TBZ $4, NR, octetsFinish
835 aesrndx8(K10)
836 aesrndx8(K11)
837 TBZ $3, NR, octetsFinish
838 VLD1.P 32(curK), [T1.B16, T2.B16]
839 aesrndx8(T1)
840 aesrndx8(T2)
841 MOVD H0, curK
842 octetsFinish:
843 aesrndx8(K8)
844 aesrndlastx8(K9)
845
846 VEOR KLAST.B16, B0.B16, T1.B16
847 VEOR KLAST.B16, B1.B16, T2.B16
848 VEOR KLAST.B16, B2.B16, B2.B16
849 VEOR KLAST.B16, B3.B16, B3.B16
850 VEOR KLAST.B16, B4.B16, B4.B16
851 VEOR KLAST.B16, B5.B16, B5.B16
852 VEOR KLAST.B16, B6.B16, B6.B16
853 VEOR KLAST.B16, B7.B16, B7.B16
854
855 VLD1.P 32(srcPtr), [B0.B16, B1.B16]
856 VEOR B0.B16, T1.B16, T1.B16
857 VEOR B1.B16, T2.B16, T2.B16
858 VST1.P [T1.B16, T2.B16], 32(dstPtr)
859
860 VLD1.P 32(pTbl), [T1.B16, T2.B16]
861 VREV64 B0.B16, B0.B16
862 VEOR ACC0.B16, B0.B16, B0.B16
863 VEXT $8, B0.B16, B0.B16, T0.B16
864 VEOR B0.B16, T0.B16, T0.B16
865 VPMULL B0.D1, T1.D1, ACC1.Q1
866 VPMULL2 B0.D2, T1.D2, ACC0.Q1
867 VPMULL T0.D1, T2.D1, ACCM.Q1
868 mulRound(B1)
869
870 VLD1.P 32(srcPtr), [B0.B16, B1.B16]
871 VEOR B2.B16, B0.B16, T1.B16
872 VEOR B3.B16, B1.B16, T2.B16
873 VST1.P [T1.B16, T2.B16], 32(dstPtr)
874 mulRound(B0)
875 mulRound(B1)
876
877 VLD1.P 32(srcPtr), [B0.B16, B1.B16]
878 VEOR B4.B16, B0.B16, T1.B16
879 VEOR B5.B16, B1.B16, T2.B16
880 VST1.P [T1.B16, T2.B16], 32(dstPtr)
881 mulRound(B0)
882 mulRound(B1)
883
884 VLD1.P 32(srcPtr), [B0.B16, B1.B16]
885 VEOR B6.B16, B0.B16, T1.B16
886 VEOR B7.B16, B1.B16, T2.B16
887 VST1.P [T1.B16, T2.B16], 32(dstPtr)
888 mulRound(B0)
889 mulRound(B1)
890
891 MOVD pTblSave, pTbl
892 reduce()
893
894 CMP $128, srcPtrLen
895 BGE octetsLoop
896
897 startSingles:
898 CBZ srcPtrLen, done
899 ADD $14*16, pTbl
900 // Preload H and its Karatsuba precomp
901 VLD1.P (pTbl), [T1.B16, T2.B16]
902 // Preload AES round keys
903 ADD $128, ks
904 VLD1.P 48(ks), [K8.B16, K9.B16, K10.B16]
905 VMOV K10.B16, KLAST.B16
906 TBZ $4, NR, singlesLoop
907 VLD1.P 32(ks), [B1.B16, B2.B16]
908 VMOV B2.B16, KLAST.B16
909 TBZ $3, NR, singlesLoop
910 VLD1.P 32(ks), [B3.B16, B4.B16]
911 VMOV B4.B16, KLAST.B16
912
913 singlesLoop:
914 CMP $16, srcPtrLen
915 BLT tail
916 SUB $16, srcPtrLen
917
918 VLD1.P 16(srcPtr), [T0.B16]
919 VREV64 T0.B16, B5.B16
920 VEOR KLAST.B16, T0.B16, T0.B16
921
922 VREV32 CTR.B16, B0.B16
923 VADD CTR.S4, INC.S4, CTR.S4
924
925 AESE K0.B16, B0.B16
926 AESMC B0.B16, B0.B16
927 AESE K1.B16, B0.B16
928 AESMC B0.B16, B0.B16
929 AESE K2.B16, B0.B16
930 AESMC B0.B16, B0.B16
931 AESE K3.B16, B0.B16
932 AESMC B0.B16, B0.B16
933 AESE K4.B16, B0.B16
934 AESMC B0.B16, B0.B16
935 AESE K5.B16, B0.B16
936 AESMC B0.B16, B0.B16
937 AESE K6.B16, B0.B16
938 AESMC B0.B16, B0.B16
939 AESE K7.B16, B0.B16
940 AESMC B0.B16, B0.B16
941 AESE K8.B16, B0.B16
942 AESMC B0.B16, B0.B16
943 AESE K9.B16, B0.B16
944 TBZ $4, NR, singlesLast
945 AESMC B0.B16, B0.B16
946 AESE K10.B16, B0.B16
947 AESMC B0.B16, B0.B16
948 AESE B1.B16, B0.B16
949 TBZ $3, NR, singlesLast
950 AESMC B0.B16, B0.B16
951 AESE B2.B16, B0.B16
952 AESMC B0.B16, B0.B16
953 AESE B3.B16, B0.B16
954 singlesLast:
955 VEOR T0.B16, B0.B16, B0.B16
956
957 VST1.P [B0.B16], 16(dstPtr)
958
959 VEOR ACC0.B16, B5.B16, B5.B16
960 VEXT $8, B5.B16, B5.B16, T0.B16
961 VEOR B5.B16, T0.B16, T0.B16
962 VPMULL B5.D1, T1.D1, ACC1.Q1
963 VPMULL2 B5.D2, T1.D2, ACC0.Q1
964 VPMULL T0.D1, T2.D1, ACCM.Q1
965 reduce()
966
967 B singlesLoop
968 tail:
969 CBZ srcPtrLen, done
970
971 VREV32 CTR.B16, B0.B16
972 VADD CTR.S4, INC.S4, CTR.S4
973
974 AESE K0.B16, B0.B16
975 AESMC B0.B16, B0.B16
976 AESE K1.B16, B0.B16
977 AESMC B0.B16, B0.B16
978 AESE K2.B16, B0.B16
979 AESMC B0.B16, B0.B16
980 AESE K3.B16, B0.B16
981 AESMC B0.B16, B0.B16
982 AESE K4.B16, B0.B16
983 AESMC B0.B16, B0.B16
984 AESE K5.B16, B0.B16
985 AESMC B0.B16, B0.B16
986 AESE K6.B16, B0.B16
987 AESMC B0.B16, B0.B16
988 AESE K7.B16, B0.B16
989 AESMC B0.B16, B0.B16
990 AESE K8.B16, B0.B16
991 AESMC B0.B16, B0.B16
992 AESE K9.B16, B0.B16
993 TBZ $4, NR, tailLast
994 AESMC B0.B16, B0.B16
995 AESE K10.B16, B0.B16
996 AESMC B0.B16, B0.B16
997 AESE B1.B16, B0.B16
998 TBZ $3, NR, tailLast
999 AESMC B0.B16, B0.B16
1000 AESE B2.B16, B0.B16
1001 AESMC B0.B16, B0.B16
1002 AESE B3.B16, B0.B16
1003 tailLast:
1004 VEOR KLAST.B16, B0.B16, B0.B16
1005
1006 tailLoad(B5)
1007
1008 VEOR B5.B16, B0.B16, B0.B16
1009
1010 tailStore(B0)
1011
1012 VREV64 B5.B16, B5.B16
1013
1014 VEOR ACC0.B16, B5.B16, B5.B16
1015 VEXT $8, B5.B16, B5.B16, T0.B16
1016 VEOR B5.B16, T0.B16, T0.B16
1017 VPMULL B5.D1, T1.D1, ACC1.Q1
1018 VPMULL2 B5.D2, T1.D2, ACC0.Q1
1019 VPMULL T0.D1, T2.D1, ACCM.Q1
1020 reduce()
1021 done:
1022 VST1 [ACC0.B16], (tPtr)
1023
1024 RET
1025
View as plain text