Blame mpi/alpha/mpih-mul3.S

Packit 0680ba
/* Alpha 21064	submul_1 -- Multiply a limb vector with a limb and
Packit 0680ba
 *			    subtract the result from a second limb vector.
Packit 0680ba
 *      Copyright (C) 1992, 1994, 1995, 1998, 
Packit 0680ba
 *                    2001, 2002 Free Software Foundation, Inc.
Packit 0680ba
 *
Packit 0680ba
 * This file is part of Libgcrypt.
Packit 0680ba
 *
Packit 0680ba
 * Libgcrypt is free software; you can redistribute it and/or modify
Packit 0680ba
 * it under the terms of the GNU Lesser General Public License as
Packit 0680ba
 * published by the Free Software Foundation; either version 2.1 of
Packit 0680ba
 * the License, or (at your option) any later version.
Packit 0680ba
 *
Packit 0680ba
 * Libgcrypt is distributed in the hope that it will be useful,
Packit 0680ba
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
Packit 0680ba
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
Packit 0680ba
 * GNU Lesser General Public License for more details.
Packit 0680ba
 *
Packit 0680ba
 * You should have received a copy of the GNU Lesser General Public
Packit 0680ba
 * License along with this program; if not, write to the Free Software
Packit 0680ba
 * Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA
Packit 0680ba
 */
Packit 0680ba
Packit 0680ba
Packit 0680ba
/*******************
Packit 0680ba
 * mpi_limb_t
Packit 0680ba
 * _gcry_mpih_submul_1( mpi_ptr_t res_ptr,      (r16   )
Packit 0680ba
 *		     mpi_ptr_t s1_ptr,	     (r17   )
Packit 0680ba
 *		     mpi_size_t s1_size,     (r18   )
Packit 0680ba
 *		     mpi_limb_t s2_limb)     (r19   )
Packit 0680ba
 *
Packit 0680ba
 * This code runs at 42 cycles/limb on EV4 and 18 cycles/limb on EV5.
Packit 0680ba
 */
Packit 0680ba
Packit 0680ba
	.set	noreorder
Packit 0680ba
	.set	noat
Packit 0680ba
.text
Packit 0680ba
	.align	3
Packit 0680ba
	.globl	_gcry_mpih_submul_1
Packit 0680ba
	.ent	_gcry_mpih_submul_1 2
Packit 0680ba
_gcry_mpih_submul_1:
Packit 0680ba
	.frame	$30,0,$26
Packit 0680ba
Packit 0680ba
	ldq	$2,0($17)	# $2 = s1_limb
Packit 0680ba
	addq	$17,8,$17	# s1_ptr++
Packit 0680ba
	subq	$18,1,$18	# size--
Packit 0680ba
	mulq	$2,$19,$3	# $3 = prod_low
Packit 0680ba
	ldq	$5,0($16)	# $5 = *res_ptr
Packit 0680ba
	umulh	$2,$19,$0	# $0 = prod_high
Packit 0680ba
	beq	$18,.Lend1	# jump if size was == 1
Packit 0680ba
	ldq	$2,0($17)	# $2 = s1_limb
Packit 0680ba
	addq	$17,8,$17	# s1_ptr++
Packit 0680ba
	subq	$18,1,$18	# size--
Packit 0680ba
	subq	$5,$3,$3
Packit 0680ba
	cmpult	$5,$3,$4
Packit 0680ba
	stq	$3,0($16)
Packit 0680ba
	addq	$16,8,$16	# res_ptr++
Packit 0680ba
	beq	$18,.Lend2	# jump if size was == 2
Packit 0680ba
Packit 0680ba
	.align	3
Packit 0680ba
.Loop:	mulq	$2,$19,$3	# $3 = prod_low
Packit 0680ba
	ldq	$5,0($16)	# $5 = *res_ptr
Packit 0680ba
	addq	$4,$0,$0	# cy_limb = cy_limb + 'cy'
Packit 0680ba
	subq	$18,1,$18	# size--
Packit 0680ba
	umulh	$2,$19,$4	# $4 = cy_limb
Packit 0680ba
	ldq	$2,0($17)	# $2 = s1_limb
Packit 0680ba
	addq	$17,8,$17	# s1_ptr++
Packit 0680ba
	addq	$3,$0,$3	# $3 = cy_limb + prod_low
Packit 0680ba
	cmpult	$3,$0,$0	# $0 = carry from (cy_limb + prod_low)
Packit 0680ba
	subq	$5,$3,$3
Packit 0680ba
	cmpult	$5,$3,$5
Packit 0680ba
	stq	$3,0($16)
Packit 0680ba
	addq	$16,8,$16	# res_ptr++
Packit 0680ba
	addq	$5,$0,$0	# combine carries
Packit 0680ba
	bne	$18,.Loop
Packit 0680ba
Packit 0680ba
.Lend2: mulq	$2,$19,$3	# $3 = prod_low
Packit 0680ba
	ldq	$5,0($16)	# $5 = *res_ptr
Packit 0680ba
	addq	$4,$0,$0	# cy_limb = cy_limb + 'cy'
Packit 0680ba
	umulh	$2,$19,$4	# $4 = cy_limb
Packit 0680ba
	addq	$3,$0,$3	# $3 = cy_limb + prod_low
Packit 0680ba
	cmpult	$3,$0,$0	# $0 = carry from (cy_limb + prod_low)
Packit 0680ba
	subq	$5,$3,$3
Packit 0680ba
	cmpult	$5,$3,$5
Packit 0680ba
	stq	$3,0($16)
Packit 0680ba
	addq	$5,$0,$0	# combine carries
Packit 0680ba
	addq	$4,$0,$0	# cy_limb = prod_high + cy
Packit 0680ba
	ret	$31,($26),1
Packit 0680ba
.Lend1: subq	$5,$3,$3
Packit 0680ba
	cmpult	$5,$3,$5
Packit 0680ba
	stq	$3,0($16)
Packit 0680ba
	addq	$0,$5,$0
Packit 0680ba
	ret	$31,($26),1
Packit 0680ba
Packit 0680ba
	.end	_gcry_mpih_submul_1
Packit 0680ba