...
▽

Text file src/crypto/internal/fips140/nistec/p256_asm_arm64.s

Documentation: crypto/internal/fips140/nistec

     1// Copyright 2018 The Go Authors. All rights reserved.
     2// Use of this source code is governed by a BSD-style
     3// license that can be found in the LICENSE file.
     4
     5//go:build !purego
     6
     7// This file contains constant-time, 64-bit assembly implementation of
     8// P256. The optimizations performed here are described in detail in:
     9// S.Gueron and V.Krasnov, "Fast prime field elliptic-curve cryptography with
    10//                          256-bit primes"
    11// http://link.springer.com/article/10.1007%2Fs13389-014-0090-x
    12// https://eprint.iacr.org/2013/816.pdf
    13
    14#include "textflag.h"
    15
    16#define res_ptr R0
    17#define a_ptr R1
    18#define b_ptr R2
    19
    20#define acc0 R3
    21#define acc1 R4
    22#define acc2 R5
    23#define acc3 R6
    24
    25#define acc4 R7
    26#define acc5 R8
    27#define acc6 R9
    28#define acc7 R10
    29#define t0 R11
    30#define t1 R12
    31#define t2 R13
    32#define t3 R14
    33#define const0 R15
    34#define const1 R16
    35
    36#define hlp0 R17
    37#define hlp1 res_ptr
    38
    39#define x0 R19
    40#define x1 R20
    41#define x2 R21
    42#define x3 R22
    43#define y0 R23
    44#define y1 R24
    45#define y2 R25
    46#define y3 R26
    47
    48DATA p256const0<>+0x00(SB)/8, $0x00000000ffffffff
    49DATA p256const1<>+0x00(SB)/8, $0xffffffff00000001
    50DATA p256one<>+0x00(SB)/8, $0x0000000000000001
    51DATA p256one<>+0x08(SB)/8, $0xffffffff00000000
    52DATA p256one<>+0x10(SB)/8, $0xffffffffffffffff
    53DATA p256one<>+0x18(SB)/8, $0x00000000fffffffe
    54GLOBL p256const0<>(SB), 8, $8
    55GLOBL p256const1<>(SB), 8, $8
    56GLOBL p256one<>(SB), 8, $32
    57
    58/* ---------------------------------------*/
    59// func p256MovCond(res, a, b *P256Point, cond int)
    60// If cond == 0 res=b, else res=a
    61TEXT ·p256MovCond(SB),NOSPLIT,$0
    62	MOVD	res+0(FP), res_ptr
    63	MOVD	a+8(FP), a_ptr
    64	MOVD	b+16(FP), b_ptr
    65	MOVD	cond+24(FP), R3
    66
    67	CMP	$0, R3
    68	// Two remarks:
    69	// 1) Will want to revisit NEON, when support is better
    70	// 2) CSEL might not be constant time on all ARM processors
    71	LDP	0*16(a_ptr), (R4, R5)
    72	LDP	1*16(a_ptr), (R6, R7)
    73	LDP	2*16(a_ptr), (R8, R9)
    74	LDP	0*16(b_ptr), (R16, R17)
    75	LDP	1*16(b_ptr), (R19, R20)
    76	LDP	2*16(b_ptr), (R21, R22)
    77	CSEL	EQ, R16, R4, R4
    78	CSEL	EQ, R17, R5, R5
    79	CSEL	EQ, R19, R6, R6
    80	CSEL	EQ, R20, R7, R7
    81	CSEL	EQ, R21, R8, R8
    82	CSEL	EQ, R22, R9, R9
    83	STP	(R4, R5), 0*16(res_ptr)
    84	STP	(R6, R7), 1*16(res_ptr)
    85	STP	(R8, R9), 2*16(res_ptr)
    86
    87	LDP	3*16(a_ptr), (R4, R5)
    88	LDP	4*16(a_ptr), (R6, R7)
    89	LDP	5*16(a_ptr), (R8, R9)
    90	LDP	3*16(b_ptr), (R16, R17)
    91	LDP	4*16(b_ptr), (R19, R20)
    92	LDP	5*16(b_ptr), (R21, R22)
    93	CSEL	EQ, R16, R4, R4
    94	CSEL	EQ, R17, R5, R5
    95	CSEL	EQ, R19, R6, R6
    96	CSEL	EQ, R20, R7, R7
    97	CSEL	EQ, R21, R8, R8
    98	CSEL	EQ, R22, R9, R9
    99	STP	(R4, R5), 3*16(res_ptr)
   100	STP	(R6, R7), 4*16(res_ptr)
   101	STP	(R8, R9), 5*16(res_ptr)
   102
   103	RET
   104/* ---------------------------------------*/
   105// func p256NegCond(val *p256Element, cond int)
   106TEXT ·p256NegCond(SB),NOSPLIT,$0
   107	MOVD	val+0(FP), a_ptr
   108	MOVD	cond+8(FP), hlp0
   109	MOVD	a_ptr, res_ptr
   110	// acc = poly
   111	MOVD	$-1, acc0
   112	MOVD	p256const0<>(SB), acc1
   113	MOVD	$0, acc2
   114	MOVD	p256const1<>(SB), acc3
   115	// Load the original value
   116	LDP	0*16(a_ptr), (t0, t1)
   117	LDP	1*16(a_ptr), (t2, t3)
   118	// Speculatively subtract
   119	SUBS	t0, acc0
   120	SBCS	t1, acc1
   121	SBCS	t2, acc2
   122	SBC	t3, acc3
   123	// If condition is 0, keep original value
   124	CMP	$0, hlp0
   125	CSEL	EQ, t0, acc0, acc0
   126	CSEL	EQ, t1, acc1, acc1
   127	CSEL	EQ, t2, acc2, acc2
   128	CSEL	EQ, t3, acc3, acc3
   129	// Store result
   130	STP	(acc0, acc1), 0*16(res_ptr)
   131	STP	(acc2, acc3), 1*16(res_ptr)
   132
   133	RET
   134/* ---------------------------------------*/
   135// func p256Sqr(res, in *p256Element, n int)
   136TEXT ·p256Sqr(SB),NOSPLIT,$0
   137	MOVD	res+0(FP), res_ptr
   138	MOVD	in+8(FP), a_ptr
   139	MOVD	n+16(FP), b_ptr
   140
   141	MOVD	p256const0<>(SB), const0
   142	MOVD	p256const1<>(SB), const1
   143
   144	LDP	0*16(a_ptr), (x0, x1)
   145	LDP	1*16(a_ptr), (x2, x3)
   146
   147sqrLoop:
   148	SUB	$1, b_ptr
   149	CALL	p256SqrInternal<>(SB)
   150	MOVD	y0, x0
   151	MOVD	y1, x1
   152	MOVD	y2, x2
   153	MOVD	y3, x3
   154	CBNZ	b_ptr, sqrLoop
   155
   156	STP	(y0, y1), 0*16(res_ptr)
   157	STP	(y2, y3), 1*16(res_ptr)
   158	RET
   159/* ---------------------------------------*/
   160// func p256Mul(res, in1, in2 *p256Element)
   161TEXT ·p256Mul(SB),NOSPLIT,$0
   162	MOVD	res+0(FP), res_ptr
   163	MOVD	in1+8(FP), a_ptr
   164	MOVD	in2+16(FP), b_ptr
   165
   166	MOVD	p256const0<>(SB), const0
   167	MOVD	p256const1<>(SB), const1
   168
   169	LDP	0*16(a_ptr), (x0, x1)
   170	LDP	1*16(a_ptr), (x2, x3)
   171
   172	LDP	0*16(b_ptr), (y0, y1)
   173	LDP	1*16(b_ptr), (y2, y3)
   174
   175	CALL	p256MulInternal<>(SB)
   176
   177	STP	(y0, y1), 0*16(res_ptr)
   178	STP	(y2, y3), 1*16(res_ptr)
   179	RET
   180/* ---------------------------------------*/
   181// func p256FromMont(res, in *p256Element)
   182TEXT ·p256FromMont(SB),NOSPLIT,$0
   183	MOVD	res+0(FP), res_ptr
   184	MOVD	in+8(FP), a_ptr
   185
   186	MOVD	p256const0<>(SB), const0
   187	MOVD	p256const1<>(SB), const1
   188
   189	LDP	0*16(a_ptr), (acc0, acc1)
   190	LDP	1*16(a_ptr), (acc2, acc3)
   191	// Only reduce, no multiplications are needed
   192	// First reduction step
   193	ADDS	acc0<<32, acc1, acc1
   194	LSR	$32, acc0, t0
   195	MUL	acc0, const1, t1
   196	UMULH	acc0, const1, acc0
   197	ADCS	t0, acc2
   198	ADCS	t1, acc3
   199	ADC	$0, acc0
   200	// Second reduction step
   201	ADDS	acc1<<32, acc2, acc2
   202	LSR	$32, acc1, t0
   203	MUL	acc1, const1, t1
   204	UMULH	acc1, const1, acc1
   205	ADCS	t0, acc3
   206	ADCS	t1, acc0
   207	ADC	$0, acc1
   208	// Third reduction step
   209	ADDS	acc2<<32, acc3, acc3
   210	LSR	$32, acc2, t0
   211	MUL	acc2, const1, t1
   212	UMULH	acc2, const1, acc2
   213	ADCS	t0, acc0
   214	ADCS	t1, acc1
   215	ADC	$0, acc2
   216	// Last reduction step
   217	ADDS	acc3<<32, acc0, acc0
   218	LSR	$32, acc3, t0
   219	MUL	acc3, const1, t1
   220	UMULH	acc3, const1, acc3
   221	ADCS	t0, acc1
   222	ADCS	t1, acc2
   223	ADC	$0, acc3
   224
   225	SUBS	$-1, acc0, t0
   226	SBCS	const0, acc1, t1
   227	SBCS	$0, acc2, t2
   228	SBCS	const1, acc3, t3
   229
   230	CSEL	CS, t0, acc0, acc0
   231	CSEL	CS, t1, acc1, acc1
   232	CSEL	CS, t2, acc2, acc2
   233	CSEL	CS, t3, acc3, acc3
   234
   235	STP	(acc0, acc1), 0*16(res_ptr)
   236	STP	(acc2, acc3), 1*16(res_ptr)
   237
   238	RET
   239/* ---------------------------------------*/
   240// func p256Select(res *P256Point, table *p256Table, idx int)
   241TEXT ·p256Select(SB),NOSPLIT,$0
   242	MOVD	idx+16(FP), const0
   243	MOVD	table+8(FP), b_ptr
   244	MOVD	res+0(FP), res_ptr
   245
   246	EOR	x0, x0, x0
   247	EOR	x1, x1, x1
   248	EOR	x2, x2, x2
   249	EOR	x3, x3, x3
   250	EOR	y0, y0, y0
   251	EOR	y1, y1, y1
   252	EOR	y2, y2, y2
   253	EOR	y3, y3, y3
   254	EOR	t0, t0, t0
   255	EOR	t1, t1, t1
   256	EOR	t2, t2, t2
   257	EOR	t3, t3, t3
   258
   259	MOVD	$0, const1
   260
   261loop_select:
   262		ADD	$1, const1
   263		CMP	const0, const1
   264		LDP.P	16(b_ptr), (acc0, acc1)
   265		CSEL	EQ, acc0, x0, x0
   266		CSEL	EQ, acc1, x1, x1
   267		LDP.P	16(b_ptr), (acc2, acc3)
   268		CSEL	EQ, acc2, x2, x2
   269		CSEL	EQ, acc3, x3, x3
   270		LDP.P	16(b_ptr), (acc4, acc5)
   271		CSEL	EQ, acc4, y0, y0
   272		CSEL	EQ, acc5, y1, y1
   273		LDP.P	16(b_ptr), (acc6, acc7)
   274		CSEL	EQ, acc6, y2, y2
   275		CSEL	EQ, acc7, y3, y3
   276		LDP.P	16(b_ptr), (acc0, acc1)
   277		CSEL	EQ, acc0, t0, t0
   278		CSEL	EQ, acc1, t1, t1
   279		LDP.P	16(b_ptr), (acc2, acc3)
   280		CSEL	EQ, acc2, t2, t2
   281		CSEL	EQ, acc3, t3, t3
   282
   283		CMP	$16, const1
   284		BNE	loop_select
   285
   286	STP	(x0, x1), 0*16(res_ptr)
   287	STP	(x2, x3), 1*16(res_ptr)
   288	STP	(y0, y1), 2*16(res_ptr)
   289	STP	(y2, y3), 3*16(res_ptr)
   290	STP	(t0, t1), 4*16(res_ptr)
   291	STP	(t2, t3), 5*16(res_ptr)
   292	RET
   293/* ---------------------------------------*/
   294// func p256SelectAffine(res *p256AffinePoint, table *p256AffineTable, idx int)
   295TEXT ·p256SelectAffine(SB),NOSPLIT,$0
   296	MOVD	idx+16(FP), t0
   297	MOVD	table+8(FP), t1
   298	MOVD	res+0(FP), res_ptr
   299
   300	EOR	x0, x0, x0
   301	EOR	x1, x1, x1
   302	EOR	x2, x2, x2
   303	EOR	x3, x3, x3
   304	EOR	y0, y0, y0
   305	EOR	y1, y1, y1
   306	EOR	y2, y2, y2
   307	EOR	y3, y3, y3
   308
   309	MOVD	$0, t2
   310
   311loop_select:
   312		ADD	$1, t2
   313		CMP	t0, t2
   314		LDP.P	16(t1), (acc0, acc1)
   315		CSEL	EQ, acc0, x0, x0
   316		CSEL	EQ, acc1, x1, x1
   317		LDP.P	16(t1), (acc2, acc3)
   318		CSEL	EQ, acc2, x2, x2
   319		CSEL	EQ, acc3, x3, x3
   320		LDP.P	16(t1), (acc4, acc5)
   321		CSEL	EQ, acc4, y0, y0
   322		CSEL	EQ, acc5, y1, y1
   323		LDP.P	16(t1), (acc6, acc7)
   324		CSEL	EQ, acc6, y2, y2
   325		CSEL	EQ, acc7, y3, y3
   326
   327		CMP	$32, t2
   328		BNE	loop_select
   329
   330	STP	(x0, x1), 0*16(res_ptr)
   331	STP	(x2, x3), 1*16(res_ptr)
   332	STP	(y0, y1), 2*16(res_ptr)
   333	STP	(y2, y3), 3*16(res_ptr)
   334	RET
   335/* ---------------------------------------*/
   336// input: x0-x3, y0-y3
   337// output: x0-x3
   338// uses: const0, const1
   339// clobbers: y0-y3, hlp0
   340TEXT p256SubInternal<>(SB),NOSPLIT,$0
   341	SUBS	x0, y0, acc0
   342	SBCS	x1, y1, acc1
   343	SBCS	x2, y2, acc2
   344	SBCS	x3, y3, acc3
   345	SBC	$0, ZR, t0
   346
   347	ADDS	$-1, acc0, acc4
   348	ADCS	const0, acc1, acc5
   349	ADCS	$0, acc2, acc6
   350	ADC	const1, acc3, acc7
   351
   352	ANDS	$1, t0
   353	CSEL	EQ, acc0, acc4, x0
   354	CSEL	EQ, acc1, acc5, x1
   355	CSEL	EQ, acc2, acc6, x2
   356	CSEL	EQ, acc3, acc7, x3
   357
   358	RET
   359/* ---------------------------------------*/
   360TEXT p256SqrInternal<>(SB),NOSPLIT,$0
   361	// x[1:] * x[0]
   362	MUL	x0, x1, acc1
   363	UMULH	x0, x1, acc2
   364
   365	MUL	x0, x2, t0
   366	ADDS	t0, acc2, acc2
   367	UMULH	x0, x2, acc3
   368
   369	MUL	x0, x3, t0
   370	ADCS	t0, acc3, acc3
   371	UMULH	x0, x3, acc4
   372	ADC	$0, acc4, acc4
   373	// x[2:] * x[1]
   374	MUL	x1, x2, t0
   375	ADDS	t0, acc3
   376	UMULH	x1, x2, t1
   377	ADCS	t1, acc4
   378	ADC	$0, ZR, acc5
   379
   380	MUL	x1, x3, t0
   381	ADDS	t0, acc4
   382	UMULH	x1, x3, t1
   383	ADC	t1, acc5
   384	// x[3] * x[2]
   385	MUL	x2, x3, t0
   386	ADDS	t0, acc5
   387	UMULH	x2, x3, acc6
   388	ADC	$0, acc6
   389
   390	MOVD	$0, acc7
   391	// *2
   392	ADDS	acc1, acc1
   393	ADCS	acc2, acc2
   394	ADCS	acc3, acc3
   395	ADCS	acc4, acc4
   396	ADCS	acc5, acc5
   397	ADCS	acc6, acc6
   398	ADC	$0, acc7
   399	// Missing products
   400	MUL	x0, x0, acc0
   401	UMULH	x0, x0, t0
   402	ADDS	t0, acc1, acc1
   403
   404	MUL	x1, x1, t0
   405	ADCS	t0, acc2, acc2
   406	UMULH	x1, x1, t1
   407	ADCS	t1, acc3, acc3
   408
   409	MUL	x2, x2, t0
   410	ADCS	t0, acc4, acc4
   411	UMULH	x2, x2, t1
   412	ADCS	t1, acc5, acc5
   413
   414	MUL	x3, x3, t0
   415	ADCS	t0, acc6, acc6
   416	UMULH	x3, x3, t1
   417	ADCS	t1, acc7, acc7
   418	// First reduction step
   419	ADDS	acc0<<32, acc1, acc1
   420	LSR	$32, acc0, t0
   421	MUL	acc0, const1, t1
   422	UMULH	acc0, const1, acc0
   423	ADCS	t0, acc2, acc2
   424	ADCS	t1, acc3, acc3
   425	ADC	$0, acc0, acc0
   426	// Second reduction step
   427	ADDS	acc1<<32, acc2, acc2
   428	LSR	$32, acc1, t0
   429	MUL	acc1, const1, t1
   430	UMULH	acc1, const1, acc1
   431	ADCS	t0, acc3, acc3
   432	ADCS	t1, acc0, acc0
   433	ADC	$0, acc1, acc1
   434	// Third reduction step
   435	ADDS	acc2<<32, acc3, acc3
   436	LSR	$32, acc2, t0
   437	MUL	acc2, const1, t1
   438	UMULH	acc2, const1, acc2
   439	ADCS	t0, acc0, acc0
   440	ADCS	t1, acc1, acc1
   441	ADC	$0, acc2, acc2
   442	// Last reduction step
   443	ADDS	acc3<<32, acc0, acc0
   444	LSR	$32, acc3, t0
   445	MUL	acc3, const1, t1
   446	UMULH	acc3, const1, acc3
   447	ADCS	t0, acc1, acc1
   448	ADCS	t1, acc2, acc2
   449	ADC	$0, acc3, acc3
   450	// Add bits [511:256] of the sqr result
   451	ADDS	acc4, acc0, acc0
   452	ADCS	acc5, acc1, acc1
   453	ADCS	acc6, acc2, acc2
   454	ADCS	acc7, acc3, acc3
   455	ADC	$0, ZR, acc4
   456
   457	SUBS	$-1, acc0, t0
   458	SBCS	const0, acc1, t1
   459	SBCS	$0, acc2, t2
   460	SBCS	const1, acc3, t3
   461	SBCS	$0, acc4, acc4
   462
   463	CSEL	CS, t0, acc0, y0
   464	CSEL	CS, t1, acc1, y1
   465	CSEL	CS, t2, acc2, y2
   466	CSEL	CS, t3, acc3, y3
   467	RET
   468/* ---------------------------------------*/
   469TEXT p256MulInternal<>(SB),NOSPLIT,$0
   470	// y[0] * x
   471	MUL	y0, x0, acc0
   472	UMULH	y0, x0, acc1
   473
   474	MUL	y0, x1, t0
   475	ADDS	t0, acc1
   476	UMULH	y0, x1, acc2
   477
   478	MUL	y0, x2, t0
   479	ADCS	t0, acc2
   480	UMULH	y0, x2, acc3
   481
   482	MUL	y0, x3, t0
   483	ADCS	t0, acc3
   484	UMULH	y0, x3, acc4
   485	ADC	$0, acc4
   486	// First reduction step
   487	ADDS	acc0<<32, acc1, acc1
   488	LSR	$32, acc0, t0
   489	MUL	acc0, const1, t1
   490	UMULH	acc0, const1, acc0
   491	ADCS	t0, acc2
   492	ADCS	t1, acc3
   493	ADC	$0, acc0
   494	// y[1] * x
   495	MUL	y1, x0, t0
   496	ADDS	t0, acc1
   497	UMULH	y1, x0, t1
   498
   499	MUL	y1, x1, t0
   500	ADCS	t0, acc2
   501	UMULH	y1, x1, t2
   502
   503	MUL	y1, x2, t0
   504	ADCS	t0, acc3
   505	UMULH	y1, x2, t3
   506
   507	MUL	y1, x3, t0
   508	ADCS	t0, acc4
   509	UMULH	y1, x3, hlp0
   510	ADC	$0, ZR, acc5
   511
   512	ADDS	t1, acc2
   513	ADCS	t2, acc3
   514	ADCS	t3, acc4
   515	ADC	hlp0, acc5
   516	// Second reduction step
   517	ADDS	acc1<<32, acc2, acc2
   518	LSR	$32, acc1, t0
   519	MUL	acc1, const1, t1
   520	UMULH	acc1, const1, acc1
   521	ADCS	t0, acc3
   522	ADCS	t1, acc0
   523	ADC	$0, acc1
   524	// y[2] * x
   525	MUL	y2, x0, t0
   526	ADDS	t0, acc2
   527	UMULH	y2, x0, t1
   528
   529	MUL	y2, x1, t0
   530	ADCS	t0, acc3
   531	UMULH	y2, x1, t2
   532
   533	MUL	y2, x2, t0
   534	ADCS	t0, acc4
   535	UMULH	y2, x2, t3
   536
   537	MUL	y2, x3, t0
   538	ADCS	t0, acc5
   539	UMULH	y2, x3, hlp0
   540	ADC	$0, ZR, acc6
   541
   542	ADDS	t1, acc3
   543	ADCS	t2, acc4
   544	ADCS	t3, acc5
   545	ADC	hlp0, acc6
   546	// Third reduction step
   547	ADDS	acc2<<32, acc3, acc3
   548	LSR	$32, acc2, t0
   549	MUL	acc2, const1, t1
   550	UMULH	acc2, const1, acc2
   551	ADCS	t0, acc0
   552	ADCS	t1, acc1
   553	ADC	$0, acc2
   554	// y[3] * x
   555	MUL	y3, x0, t0
   556	ADDS	t0, acc3
   557	UMULH	y3, x0, t1
   558
   559	MUL	y3, x1, t0
   560	ADCS	t0, acc4
   561	UMULH	y3, x1, t2
   562
   563	MUL	y3, x2, t0
   564	ADCS	t0, acc5
   565	UMULH	y3, x2, t3
   566
   567	MUL	y3, x3, t0
   568	ADCS	t0, acc6
   569	UMULH	y3, x3, hlp0
   570	ADC	$0, ZR, acc7
   571
   572	ADDS	t1, acc4
   573	ADCS	t2, acc5
   574	ADCS	t3, acc6
   575	ADC	hlp0, acc7
   576	// Last reduction step
   577	ADDS	acc3<<32, acc0, acc0
   578	LSR	$32, acc3, t0
   579	MUL	acc3, const1, t1
   580	UMULH	acc3, const1, acc3
   581	ADCS	t0, acc1
   582	ADCS	t1, acc2
   583	ADC	$0, acc3
   584	// Add bits [511:256] of the mul result
   585	ADDS	acc4, acc0, acc0
   586	ADCS	acc5, acc1, acc1
   587	ADCS	acc6, acc2, acc2
   588	ADCS	acc7, acc3, acc3
   589	ADC	$0, ZR, acc4
   590
   591	SUBS	$-1, acc0, t0
   592	SBCS	const0, acc1, t1
   593	SBCS	$0, acc2, t2
   594	SBCS	const1, acc3, t3
   595	SBCS	$0, acc4, acc4
   596
   597	CSEL	CS, t0, acc0, y0
   598	CSEL	CS, t1, acc1, y1
   599	CSEL	CS, t2, acc2, y2
   600	CSEL	CS, t3, acc3, y3
   601	RET
   602/* ---------------------------------------*/
   603#define p256MulBy2Inline       \
   604	ADDS	y0, y0, x0;    \
   605	ADCS	y1, y1, x1;    \
   606	ADCS	y2, y2, x2;    \
   607	ADCS	y3, y3, x3;    \
   608	ADC	$0, ZR, hlp0;  \
   609	SUBS	$-1, x0, t0;   \
   610	SBCS	const0, x1, t1;\
   611	SBCS	$0, x2, t2;    \
   612	SBCS	const1, x3, t3;\
   613	SBCS	$0, hlp0, hlp0;\
   614	CSEL	CC, x0, t0, x0;\
   615	CSEL	CC, x1, t1, x1;\
   616	CSEL	CC, x2, t2, x2;\
   617	CSEL	CC, x3, t3, x3;
   618/* ---------------------------------------*/
   619#define x1in(off) (off)(a_ptr)
   620#define y1in(off) (off + 32)(a_ptr)
   621#define z1in(off) (off + 64)(a_ptr)
   622#define x2in(off) (off)(b_ptr)
   623#define z2in(off) (off + 64)(b_ptr)
   624#define x3out(off) (off)(res_ptr)
   625#define y3out(off) (off + 32)(res_ptr)
   626#define z3out(off) (off + 64)(res_ptr)
   627#define LDx(src) LDP src(0), (x0, x1); LDP src(16), (x2, x3)
   628#define LDy(src) LDP src(0), (y0, y1); LDP src(16), (y2, y3)
   629#define STx(src) STP (x0, x1), src(0); STP (x2, x3), src(16)
   630#define STy(src) STP (y0, y1), src(0); STP (y2, y3), src(16)
   631/* ---------------------------------------*/
   632#define y2in(off)  (32*0 + 8 + off)(RSP)
   633#define s2(off)    (32*1 + 8 + off)(RSP)
   634#define z1sqr(off) (32*2 + 8 + off)(RSP)
   635#define h(off)	   (32*3 + 8 + off)(RSP)
   636#define r(off)	   (32*4 + 8 + off)(RSP)
   637#define hsqr(off)  (32*5 + 8 + off)(RSP)
   638#define rsqr(off)  (32*6 + 8 + off)(RSP)
   639#define hcub(off)  (32*7 + 8 + off)(RSP)
   640
   641#define z2sqr(off) (32*8 + 8 + off)(RSP)
   642#define s1(off) (32*9 + 8 + off)(RSP)
   643#define u1(off) (32*10 + 8 + off)(RSP)
   644#define u2(off) (32*11 + 8 + off)(RSP)
   645
   646// func p256PointAddAffineAsm(res, in1 *P256Point, in2 *p256AffinePoint, sign, sel, zero int)
   647TEXT ·p256PointAddAffineAsm(SB),0,$264-48
   648	MOVD	in1+8(FP), a_ptr
   649	MOVD	in2+16(FP), b_ptr
   650	MOVD	sign+24(FP), hlp0
   651	MOVD	sel+32(FP), hlp1
   652	MOVD	zero+40(FP), t2
   653
   654	MOVD	$1, t0
   655	CMP	$0, t2
   656	CSEL	EQ, ZR, t0, t2
   657	CMP	$0, hlp1
   658	CSEL	EQ, ZR, t0, hlp1
   659
   660	MOVD	p256const0<>(SB), const0
   661	MOVD	p256const1<>(SB), const1
   662	EOR	t2<<1, hlp1
   663
   664	// Negate y2in based on sign
   665	LDP	2*16(b_ptr), (y0, y1)
   666	LDP	3*16(b_ptr), (y2, y3)
   667	MOVD	$-1, acc0
   668
   669	SUBS	y0, acc0, acc0
   670	SBCS	y1, const0, acc1
   671	SBCS	y2, ZR, acc2
   672	SBCS	y3, const1, acc3
   673	SBC	$0, ZR, t0
   674
   675	ADDS	$-1, acc0, acc4
   676	ADCS	const0, acc1, acc5
   677	ADCS	$0, acc2, acc6
   678	ADCS	const1, acc3, acc7
   679	ADC	$0, t0, t0
   680
   681	CMP	$0, t0
   682	CSEL	EQ, acc4, acc0, acc0
   683	CSEL	EQ, acc5, acc1, acc1
   684	CSEL	EQ, acc6, acc2, acc2
   685	CSEL	EQ, acc7, acc3, acc3
   686	// If condition is 0, keep original value
   687	CMP	$0, hlp0
   688	CSEL	EQ, y0, acc0, y0
   689	CSEL	EQ, y1, acc1, y1
   690	CSEL	EQ, y2, acc2, y2
   691	CSEL	EQ, y3, acc3, y3
   692	// Store result
   693	STy(y2in)
   694	// Begin point add
   695	LDx(z1in)
   696	CALL	p256SqrInternal<>(SB)    // z1ˆ2
   697	STy(z1sqr)
   698
   699	LDx(x2in)
   700	CALL	p256MulInternal<>(SB)    // x2 * z1ˆ2
   701
   702	LDx(x1in)
   703	CALL	p256SubInternal<>(SB)    // h = u2 - u1
   704	STx(h)
   705
   706	LDy(z1in)
   707	CALL	p256MulInternal<>(SB)    // z3 = h * z1
   708
   709	LDP	4*16(a_ptr), (acc0, acc1)// iff select[0] == 0, z3 = z1
   710	LDP	5*16(a_ptr), (acc2, acc3)
   711	ANDS	$1, hlp1, ZR
   712	CSEL	EQ, acc0, y0, y0
   713	CSEL	EQ, acc1, y1, y1
   714	CSEL	EQ, acc2, y2, y2
   715	CSEL	EQ, acc3, y3, y3
   716	LDP	p256one<>+0x00(SB), (acc0, acc1)
   717	LDP	p256one<>+0x10(SB), (acc2, acc3)
   718	ANDS	$2, hlp1, ZR            // iff select[1] == 0, z3 = 1
   719	CSEL	EQ, acc0, y0, y0
   720	CSEL	EQ, acc1, y1, y1
   721	CSEL	EQ, acc2, y2, y2
   722	CSEL	EQ, acc3, y3, y3
   723	LDx(z1in)
   724	MOVD	res+0(FP), t0
   725	STP	(y0, y1), 4*16(t0)
   726	STP	(y2, y3), 5*16(t0)
   727
   728	LDy(z1sqr)
   729	CALL	p256MulInternal<>(SB)    // z1 ^ 3
   730
   731	LDx(y2in)
   732	CALL	p256MulInternal<>(SB)    // s2 = y2 * z1ˆ3
   733	STy(s2)
   734
   735	LDx(y1in)
   736	CALL	p256SubInternal<>(SB)    // r = s2 - s1
   737	STx(r)
   738
   739	CALL	p256SqrInternal<>(SB)    // rsqr = rˆ2
   740	STy	(rsqr)
   741
   742	LDx(h)
   743	CALL	p256SqrInternal<>(SB)    // hsqr = hˆ2
   744	STy(hsqr)
   745
   746	CALL	p256MulInternal<>(SB)    // hcub = hˆ3
   747	STy(hcub)
   748
   749	LDx(y1in)
   750	CALL	p256MulInternal<>(SB)    // y1 * hˆ3
   751	STy(s2)
   752
   753	LDP	hsqr(0*8), (x0, x1)
   754	LDP	hsqr(2*8), (x2, x3)
   755	LDP	0*16(a_ptr), (y0, y1)
   756	LDP	1*16(a_ptr), (y2, y3)
   757	CALL	p256MulInternal<>(SB)    // u1 * hˆ2
   758	STP	(y0, y1), h(0*8)
   759	STP	(y2, y3), h(2*8)
   760
   761	p256MulBy2Inline               // u1 * hˆ2 * 2, inline
   762
   763	LDy(rsqr)
   764	CALL	p256SubInternal<>(SB)    // rˆ2 - u1 * hˆ2 * 2
   765
   766	MOVD	x0, y0
   767	MOVD	x1, y1
   768	MOVD	x2, y2
   769	MOVD	x3, y3
   770	LDx(hcub)
   771	CALL	p256SubInternal<>(SB)
   772
   773	LDP	0*16(a_ptr), (acc0, acc1)
   774	LDP	1*16(a_ptr), (acc2, acc3)
   775	ANDS	$1, hlp1, ZR           // iff select[0] == 0, x3 = x1
   776	CSEL	EQ, acc0, x0, x0
   777	CSEL	EQ, acc1, x1, x1
   778	CSEL	EQ, acc2, x2, x2
   779	CSEL	EQ, acc3, x3, x3
   780	LDP	0*16(b_ptr), (acc0, acc1)
   781	LDP	1*16(b_ptr), (acc2, acc3)
   782	ANDS	$2, hlp1, ZR           // iff select[1] == 0, x3 = x2
   783	CSEL	EQ, acc0, x0, x0
   784	CSEL	EQ, acc1, x1, x1
   785	CSEL	EQ, acc2, x2, x2
   786	CSEL	EQ, acc3, x3, x3
   787	MOVD	res+0(FP), t0
   788	STP	(x0, x1), 0*16(t0)
   789	STP	(x2, x3), 1*16(t0)
   790
   791	LDP	h(0*8), (y0, y1)
   792	LDP	h(2*8), (y2, y3)
   793	CALL	p256SubInternal<>(SB)
   794
   795	LDP	r(0*8), (y0, y1)
   796	LDP	r(2*8), (y2, y3)
   797	CALL	p256MulInternal<>(SB)
   798
   799	LDP	s2(0*8), (x0, x1)
   800	LDP	s2(2*8), (x2, x3)
   801	CALL	p256SubInternal<>(SB)
   802	LDP	2*16(a_ptr), (acc0, acc1)
   803	LDP	3*16(a_ptr), (acc2, acc3)
   804	ANDS	$1, hlp1, ZR           // iff select[0] == 0, y3 = y1
   805	CSEL	EQ, acc0, x0, x0
   806	CSEL	EQ, acc1, x1, x1
   807	CSEL	EQ, acc2, x2, x2
   808	CSEL	EQ, acc3, x3, x3
   809	LDP	y2in(0*8), (acc0, acc1)
   810	LDP	y2in(2*8), (acc2, acc3)
   811	ANDS	$2, hlp1, ZR            // iff select[1] == 0, y3 = y2
   812	CSEL	EQ, acc0, x0, x0
   813	CSEL	EQ, acc1, x1, x1
   814	CSEL	EQ, acc2, x2, x2
   815	CSEL	EQ, acc3, x3, x3
   816	MOVD	res+0(FP), t0
   817	STP	(x0, x1), 2*16(t0)
   818	STP	(x2, x3), 3*16(t0)
   819
   820	RET
   821
   822#define p256AddInline          \
   823	ADDS	y0, x0, x0;    \
   824	ADCS	y1, x1, x1;    \
   825	ADCS	y2, x2, x2;    \
   826	ADCS	y3, x3, x3;    \
   827	ADC	$0, ZR, hlp0;  \
   828	SUBS	$-1, x0, t0;   \
   829	SBCS	const0, x1, t1;\
   830	SBCS	$0, x2, t2;    \
   831	SBCS	const1, x3, t3;\
   832	SBCS	$0, hlp0, hlp0;\
   833	CSEL	CC, x0, t0, x0;\
   834	CSEL	CC, x1, t1, x1;\
   835	CSEL	CC, x2, t2, x2;\
   836	CSEL	CC, x3, t3, x3;
   837
   838#define s(off)	(32*0 + 8 + off)(RSP)
   839#define m(off)	(32*1 + 8 + off)(RSP)
   840#define zsqr(off) (32*2 + 8 + off)(RSP)
   841#define tmp(off)  (32*3 + 8 + off)(RSP)
   842
   843//func p256PointDoubleAsm(res, in *P256Point)
   844TEXT ·p256PointDoubleAsm(SB),NOSPLIT,$136-16
   845	MOVD	res+0(FP), res_ptr
   846	MOVD	in+8(FP), a_ptr
   847
   848	MOVD	p256const0<>(SB), const0
   849	MOVD	p256const1<>(SB), const1
   850
   851	// Begin point double
   852	LDP	4*16(a_ptr), (x0, x1)
   853	LDP	5*16(a_ptr), (x2, x3)
   854	CALL	p256SqrInternal<>(SB)
   855	STP	(y0, y1), zsqr(0*8)
   856	STP	(y2, y3), zsqr(2*8)
   857
   858	LDP	0*16(a_ptr), (x0, x1)
   859	LDP	1*16(a_ptr), (x2, x3)
   860	p256AddInline
   861	STx(m)
   862
   863	LDx(z1in)
   864	LDy(y1in)
   865	CALL	p256MulInternal<>(SB)
   866	p256MulBy2Inline
   867	STx(z3out)
   868
   869	LDy(x1in)
   870	LDx(zsqr)
   871	CALL	p256SubInternal<>(SB)
   872	LDy(m)
   873	CALL	p256MulInternal<>(SB)
   874
   875	// Multiply by 3
   876	p256MulBy2Inline
   877	p256AddInline
   878	STx(m)
   879
   880	LDy(y1in)
   881	p256MulBy2Inline
   882	CALL	p256SqrInternal<>(SB)
   883	STy(s)
   884	MOVD	y0, x0
   885	MOVD	y1, x1
   886	MOVD	y2, x2
   887	MOVD	y3, x3
   888	CALL	p256SqrInternal<>(SB)
   889
   890	// Divide by 2
   891	ADDS	$-1, y0, t0
   892	ADCS	const0, y1, t1
   893	ADCS	$0, y2, t2
   894	ADCS	const1, y3, t3
   895	ADC	$0, ZR, hlp0
   896
   897	ANDS	$1, y0, ZR
   898	CSEL	EQ, y0, t0, t0
   899	CSEL	EQ, y1, t1, t1
   900	CSEL	EQ, y2, t2, t2
   901	CSEL	EQ, y3, t3, t3
   902	AND	y0, hlp0, hlp0
   903
   904	EXTR	$1, t0, t1, y0
   905	EXTR	$1, t1, t2, y1
   906	EXTR	$1, t2, t3, y2
   907	EXTR	$1, t3, hlp0, y3
   908	STy(y3out)
   909
   910	LDx(x1in)
   911	LDy(s)
   912	CALL	p256MulInternal<>(SB)
   913	STy(s)
   914	p256MulBy2Inline
   915	STx(tmp)
   916
   917	LDx(m)
   918	CALL	p256SqrInternal<>(SB)
   919	LDx(tmp)
   920	CALL	p256SubInternal<>(SB)
   921
   922	STx(x3out)
   923
   924	LDy(s)
   925	CALL	p256SubInternal<>(SB)
   926
   927	LDy(m)
   928	CALL	p256MulInternal<>(SB)
   929
   930	LDx(y3out)
   931	CALL	p256SubInternal<>(SB)
   932	STx(y3out)
   933	RET
   934/* ---------------------------------------*/
   935#undef y2in
   936#undef x3out
   937#undef y3out
   938#undef z3out
   939#define y2in(off) (off + 32)(b_ptr)
   940#define x3out(off) (off)(b_ptr)
   941#define y3out(off) (off + 32)(b_ptr)
   942#define z3out(off) (off + 64)(b_ptr)
   943// func p256PointAddAsm(res, in1, in2 *P256Point) int
   944TEXT ·p256PointAddAsm(SB),0,$392-32
   945	// See https://hyperelliptic.org/EFD/g1p/auto-shortw-jacobian-3.html#addition-add-2007-bl
   946	// Move input to stack in order to free registers
   947	MOVD	in1+8(FP), a_ptr
   948	MOVD	in2+16(FP), b_ptr
   949
   950	MOVD	p256const0<>(SB), const0
   951	MOVD	p256const1<>(SB), const1
   952
   953	// Begin point add
   954	LDx(z2in)
   955	CALL	p256SqrInternal<>(SB)    // z2^2
   956	STy(z2sqr)
   957
   958	CALL	p256MulInternal<>(SB)    // z2^3
   959
   960	LDx(y1in)
   961	CALL	p256MulInternal<>(SB)    // s1 = z2ˆ3*y1
   962	STy(s1)
   963
   964	LDx(z1in)
   965	CALL	p256SqrInternal<>(SB)    // z1^2
   966	STy(z1sqr)
   967
   968	CALL	p256MulInternal<>(SB)    // z1^3
   969
   970	LDx(y2in)
   971	CALL	p256MulInternal<>(SB)    // s2 = z1ˆ3*y2
   972
   973	LDx(s1)
   974	CALL	p256SubInternal<>(SB)    // r = s2 - s1
   975	STx(r)
   976
   977	MOVD	$1, t2
   978	ORR	x0, x1, t0             // Check if zero mod p256
   979	ORR	x2, x3, t1
   980	ORR	t1, t0, t0
   981	CMP	$0, t0
   982	CSEL	EQ, t2, ZR, hlp1
   983
   984	EOR	$-1, x0, t0
   985	EOR	const0, x1, t1
   986	EOR	const1, x3, t3
   987
   988	ORR	t0, t1, t0
   989	ORR	x2, t3, t1
   990	ORR	t1, t0, t0
   991	CMP	$0, t0
   992	CSEL	EQ, t2, hlp1, hlp1
   993
   994	LDx(z2sqr)
   995	LDy(x1in)
   996	CALL	p256MulInternal<>(SB)    // u1 = x1 * z2ˆ2
   997	STy(u1)
   998
   999	LDx(z1sqr)
  1000	LDy(x2in)
  1001	CALL	p256MulInternal<>(SB)    // u2 = x2 * z1ˆ2
  1002	STy(u2)
  1003
  1004	LDx(u1)
  1005	CALL	p256SubInternal<>(SB)    // h = u2 - u1
  1006	STx(h)
  1007
  1008	MOVD	$1, t2
  1009	ORR	x0, x1, t0             // Check if zero mod p256
  1010	ORR	x2, x3, t1
  1011	ORR	t1, t0, t0
  1012	CMP	$0, t0
  1013	CSEL	EQ, t2, ZR, hlp0
  1014
  1015	EOR	$-1, x0, t0
  1016	EOR	const0, x1, t1
  1017	EOR	const1, x3, t3
  1018
  1019	ORR	t0, t1, t0
  1020	ORR	x2, t3, t1
  1021	ORR	t1, t0, t0
  1022	CMP	$0, t0
  1023	CSEL	EQ, t2, hlp0, hlp0
  1024
  1025	AND	hlp0, hlp1, hlp1
  1026
  1027	LDx(r)
  1028	CALL	p256SqrInternal<>(SB)    // rsqr = rˆ2
  1029	STy(rsqr)
  1030
  1031	LDx(h)
  1032	CALL	p256SqrInternal<>(SB)    // hsqr = hˆ2
  1033	STy(hsqr)
  1034
  1035	LDx(h)
  1036	CALL	p256MulInternal<>(SB)    // hcub = hˆ3
  1037	STy(hcub)
  1038
  1039	LDx(s1)
  1040	CALL	p256MulInternal<>(SB)
  1041	STy(s2)
  1042
  1043	LDx(z1in)
  1044	LDy(z2in)
  1045	CALL	p256MulInternal<>(SB)    // z1 * z2
  1046	LDx(h)
  1047	CALL	p256MulInternal<>(SB)    // z1 * z2 * h
  1048	MOVD	res+0(FP), b_ptr
  1049	STy(z3out)
  1050
  1051	LDx(hsqr)
  1052	LDy(u1)
  1053	CALL	p256MulInternal<>(SB)    // hˆ2 * u1
  1054	STy(u2)
  1055
  1056	p256MulBy2Inline               // u1 * hˆ2 * 2, inline
  1057	LDy(rsqr)
  1058	CALL	p256SubInternal<>(SB)    // rˆ2 - u1 * hˆ2 * 2
  1059
  1060	MOVD	x0, y0
  1061	MOVD	x1, y1
  1062	MOVD	x2, y2
  1063	MOVD	x3, y3
  1064	LDx(hcub)
  1065	CALL	p256SubInternal<>(SB)
  1066	STx(x3out)
  1067
  1068	LDy(u2)
  1069	CALL	p256SubInternal<>(SB)
  1070
  1071	LDy(r)
  1072	CALL	p256MulInternal<>(SB)
  1073
  1074	LDx(s2)
  1075	CALL	p256SubInternal<>(SB)
  1076	STx(y3out)
  1077
  1078	MOVD	hlp1, R0
  1079	MOVD	R0, ret+24(FP)
  1080
  1081	RET

View as plain text