1 dnl AMD K7 mpn_mul_1 -- mpn by limb multiply.
3 dnl K7: 3.4 cycles/limb (at 16 limbs/loop).
6 dnl Copyright (C) 1999, 2000 Free Software Foundation, Inc.
8 dnl This file is part of the GNU MP Library.
10 dnl The GNU MP Library is free software; you can redistribute it and/or
11 dnl modify it under the terms of the GNU Lesser General Public License as
12 dnl published by the Free Software Foundation; either version 2.1 of the
13 dnl License, or (at your option) any later version.
15 dnl The GNU MP Library is distributed in the hope that it will be useful,
16 dnl but WITHOUT ANY WARRANTY; without even the implied warranty of
17 dnl MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
18 dnl Lesser General Public License for more details.
20 dnl You should have received a copy of the GNU Lesser General Public
21 dnl License along with the GNU MP Library; see the file COPYING.LIB. If
22 dnl not, write to the Free Software Foundation, Inc., 59 Temple Place -
23 dnl Suite 330, Boston, MA 02111-1307, USA.
26 include(`../config.m4')
29 dnl K7: UNROLL_COUNT cycles/limb
34 dnl Maximum possible with the current code is 64.
36 deflit(UNROLL_COUNT, 16)
39 C mp_limb_t mpn_mul_1 (mp_ptr dst, mp_srcptr src, mp_size_t size,
40 C mp_limb_t multiplier);
41 C mp_limb_t mpn_mul_1c (mp_ptr dst, mp_srcptr src, mp_size_t size,
42 C mp_limb_t multiplier, mp_limb_t carry);
44 C Multiply src,size by mult and store the result in dst,size.
45 C Return the carry limb from the top of the result.
47 C mpn_mul_1c() accepts an initial carry for the calculation, it's added into
48 C the low limb of the destination.
50 C Variations on the unrolled loop have been tried, with the current
51 C registers or with the counter on the stack to free up ecx. The current
52 C code is the fastest found.
54 C An interesting effect is that removing the stores "movl %ebx, disp0(%edi)"
55 C from the unrolled loop actually slows it down to 5.0 cycles/limb. Code
56 C with this change can be tested on sizes of the form UNROLL_COUNT*n+1
57 C without having to change the computed jump. There's obviously something
58 C fishy going on, perhaps with what execution units the mul needs.
60 defframe(PARAM_CARRY, 20)
61 defframe(PARAM_MULTIPLIER,16)
62 defframe(PARAM_SIZE, 12)
63 defframe(PARAM_SRC, 8)
64 defframe(PARAM_DST, 4)
66 defframe(SAVE_EBP, -4)
67 defframe(SAVE_EDI, -8)
68 defframe(SAVE_ESI, -12)
69 defframe(SAVE_EBX, -16)
70 deflit(STACK_SPACE, 16)
72 dnl Must have UNROLL_THRESHOLD >= 2, since the unrolled loop can't handle 1.
74 deflit(UNROLL_THRESHOLD, 7)
76 deflit(UNROLL_THRESHOLD, 5)
83 movl PARAM_CARRY, %edx
84 jmp LF(mpn_mul_1,start_nc)
90 xorl %edx, %edx C initial carry
93 subl $STACK_SPACE, %esp
94 deflit(`FRAME', STACK_SPACE)
102 cmpl $UNROLL_THRESHOLD, %ecx
108 leal (%esi,%ecx,4), %esi
109 leal (%edi,%ecx,4), %edi
112 movl PARAM_MULTIPLIER, %ebp
117 C ecx counter (negative)
123 movl (%esi,%ecx,4), %eax
128 movl %eax, (%edi,%ecx,4)
141 addl $STACK_SPACE, %esp
146 C -----------------------------------------------------------------------------
147 C The mov to load the next source limb is done well ahead of the mul, this
148 C is necessary for full speed. It leads to one limb handled separately
151 C When unrolling to 32 or more, an offset of +4 is used on the src pointer,
152 C to avoid having an 0x80 displacement in the code for the last limb in the
153 C unrolled loop. This is for a fair comparison between 16 and 32 unrolling.
155 ifelse(eval(UNROLL_COUNT >= 32),1,`
161 C this is offset 0x62, so close enough to aligned
170 deflit(`FRAME', STACK_SPACE)
172 leal -1(%ecx), %edx C one limb handled at end
173 leal -2(%ecx), %ecx C and ecx is one less than edx
177 shrl $UNROLL_LOG2, %ecx C unrolled loop counter
178 movl (%esi), %eax C src low limb
180 andl $UNROLL_MASK, %edx
186 C 17 code bytes per limb
188 call L(add_eip_to_edx)
191 leal L(entry) (%edx,%ebp), %edx
195 leal ifelse(UNROLL_BYTES,256,128+) SRC_OFFSET(%esi,%ebp,4), %esi
196 leal ifelse(UNROLL_BYTES,256,128) (%edi,%ebp,4), %edi
197 movl PARAM_MULTIPLIER, %ebp
204 C See README.family about old gas bugs
205 leal (%edx,%ebp), %edx
206 addl $L(entry)-L(here), %edx
212 C ----------------------------------------------------------------------------
223 C 17 code bytes per limb processed
226 forloop(i, 0, UNROLL_COUNT-1, `
227 deflit(`disp_dst', eval(i*4 ifelse(UNROLL_BYTES,256,-128)))
228 deflit(`disp_src', eval(disp_dst + 4-(SRC_OFFSET-0)))
233 Zdisp( movl, disp_src,(%esi), %eax)
234 Zdisp( movl, %ebx, disp_dst,(%edi))
242 leal UNROLL_BYTES(%esi), %esi
243 leal UNROLL_BYTES(%edi), %edi
247 deflit(`disp0', ifelse(UNROLL_BYTES,256,-128))
255 movl %ebx, disp0(%edi)
261 addl $STACK_SPACE, %esp