[v1,25/27] x86/fpu: Optimize svml_s_logf4_core_sse4.S

Message ID 20221207085236.1424424-25-goldstein.w.n@gmail.com (mailing list archive)
State New
Series [v1,01/27] x86/fpu: Create helper file for common data macros |


Context Check Description
dj/TryBot-apply_patch success Patch applied to master at the time it was sent

Commit Message

Noah Goldstein Dec. 7, 2022, 8:52 a.m. UTC
  1. Improve special values case which ends up covering ~half of all
   float bit patterns.
2. Cleanup some missed optimizations in instruction selection /
   unnecissary repeated rodata references.
3. Remove unused rodata.
4. Use common data definitions where possible.

As well instead of using the shared `__svml_slogf_data` just define
the data locally.  This is because; 1) Its not really ideal for
the sse4/avx2 to reuse the avx512 tables as it pollute the cache
which unnecessarily large blocks.  2) Really only one of the
versions is ever expected to be used by a given process so there
isn't any constructive caching between them. And 3) there is not
enough data shared to make up for the first two reasons.

Code Size Change: -286 Bytes (235 - 521)

Input                                 New Time / Old Time
0F          (0x00000000)           -> 0.5750
0F          (0x0000ffff, Denorm)   -> 0.9206
.1F         (0x3dcccccd)           -> 0.9389
5F          (0x40a00000)           -> 0.9361
2315255808F (0x4f0a0000)           -> 0.9306
-NaN        (0xffffffff)           -> 0.5226
 .../fpu/multiarch/svml_s_logf4_core_sse4.S    | 327 +++++++++---------
 1 file changed, 156 insertions(+), 171 deletions(-)


diff --git a/sysdeps/x86_64/fpu/multiarch/svml_s_logf4_core_sse4.S b/sysdeps/x86_64/fpu/multiarch/svml_s_logf4_core_sse4.S
index 20ad054eac..42d09db8df 100644
--- a/sysdeps/x86_64/fpu/multiarch/svml_s_logf4_core_sse4.S
+++ b/sysdeps/x86_64/fpu/multiarch/svml_s_logf4_core_sse4.S
@@ -16,179 +16,164 @@ 
    License along with the GNU C Library; if not, see
    <https://www.gnu.org/licenses/>.  */
+#define LOCAL_DATA_NAME	__svml_slog_data_internal
+#include "svml_s_common_sse4_rodata_offsets.h"
+#define _sPoly_7	0
+#define _sPoly_6	16
+#define _sPoly_5	32
+#define _sPoly_4	48
+#define _sPoly_3	64
+#define _sPoly_2	80
 #include <sysdep.h>
-#include "svml_s_logf_data.h"
 	.section .text.sse4, "ax", @progbits
 ENTRY (_ZGVbN4v_logf_sse4)
-     log(x) = exponent_x*log(2) + log(mantissa_x),         if mantissa_x<4/3
-     log(x) = (exponent_x+1)*log(2) + log(0.5*mantissa_x), if mantissa_x>4/3
-     R = mantissa_x - 1,     if mantissa_x<4/3
-     R = 0.5*mantissa_x - 1, if mantissa_x>4/3
-     |R|< 1/3
-     log(1+R) is approximated as a polynomial: degree 9 for 1-ulp,
-     degree 7 for 4-ulp, degree 3 for half-precision.  */
-        pushq     %rbp
-        cfi_adjust_cfa_offset (8)
-        cfi_rel_offset (%rbp, 0)
-        movq      %rsp, %rbp
-        cfi_def_cfa_register (%rbp)
-        andq      $-64, %rsp
-        subq      $320, %rsp
-/* reduction: compute r,n */
-        movaps    %xmm0, %xmm2
-/* check for working range,
-   set special argument mask (denormals/zero/Inf/NaN) */
-        movq      __svml_slog_data@GOTPCREL(%rip), %rax
-        movdqu _iHiDelta(%rax), %xmm1
-        movdqu _iLoRange(%rax), %xmm4
-        paddd     %xmm0, %xmm1
-        movdqu _iBrkValue(%rax), %xmm3
-        pcmpgtd   %xmm1, %xmm4
-        movdqu _iOffExpoMask(%rax), %xmm1
-        psubd     %xmm3, %xmm2
-        pand      %xmm2, %xmm1
-/* exponent_x (mantissa_x<4/3) or exponent_x+1 (mantissa_x>4/3) */
-        psrad     $23, %xmm2
-        paddd     %xmm3, %xmm1
-        movups _sPoly_7(%rax), %xmm5
-/* mantissa_x (mantissa_x<4/3), or 0.5*mantissa_x (mantissa_x>4/3) */
-        cvtdq2ps  %xmm2, %xmm6
-/* reduced argument R */
-        subps _sOne(%rax), %xmm1
-        movmskps  %xmm4, %ecx
-/* final reconstruction:
-   add exponent_value*log2 to polynomial result */
-        mulps _sLn2(%rax), %xmm6
-/* polynomial evaluation starts here */
-        mulps     %xmm1, %xmm5
-        addps _sPoly_6(%rax), %xmm5
-        mulps     %xmm1, %xmm5
-        addps _sPoly_5(%rax), %xmm5
-        mulps     %xmm1, %xmm5
-        addps _sPoly_4(%rax), %xmm5
-        mulps     %xmm1, %xmm5
-        addps _sPoly_3(%rax), %xmm5
-        mulps     %xmm1, %xmm5
-        addps _sPoly_2(%rax), %xmm5
-        mulps     %xmm1, %xmm5
-        addps _sPoly_1(%rax), %xmm5
-        mulps     %xmm1, %xmm5
-/* polynomial evaluation end */
-        mulps     %xmm1, %xmm5
-        addps     %xmm5, %xmm1
-        addps     %xmm6, %xmm1
-        testl     %ecx, %ecx
-        jne       .LBL_1_3
-        cfi_remember_state
-        movdqa    %xmm1, %xmm0
-        movq      %rbp, %rsp
-        cfi_def_cfa_register (%rsp)
-        popq      %rbp
-        cfi_adjust_cfa_offset (-8)
-        cfi_restore (%rbp)
-        ret
-        cfi_restore_state
-        movups    %xmm0, 192(%rsp)
-        movups    %xmm1, 256(%rsp)
-        je        .LBL_1_2
-        xorb      %dl, %dl
-        xorl      %eax, %eax
-        movups    %xmm8, 112(%rsp)
-        movups    %xmm9, 96(%rsp)
-        movups    %xmm10, 80(%rsp)
-        movups    %xmm11, 64(%rsp)
-        movups    %xmm12, 48(%rsp)
-        movups    %xmm13, 32(%rsp)
-        movups    %xmm14, 16(%rsp)
-        movups    %xmm15, (%rsp)
-        movq      %rsi, 136(%rsp)
-        movq      %rdi, 128(%rsp)
-        movq      %r12, 168(%rsp)
-        cfi_offset_rel_rsp (12, 168)
-        movb      %dl, %r12b
-        movq      %r13, 160(%rsp)
-        cfi_offset_rel_rsp (13, 160)
-        movl      %ecx, %r13d
-        movq      %r14, 152(%rsp)
-        cfi_offset_rel_rsp (14, 152)
-        movl      %eax, %r14d
-        movq      %r15, 144(%rsp)
-        cfi_offset_rel_rsp (15, 144)
-        cfi_remember_state
-        btl       %r14d, %r13d
-        jc        .LBL_1_12
-        lea       1(%r14), %esi
-        btl       %esi, %r13d
-        jc        .LBL_1_10
-        incb      %r12b
-        addl      $2, %r14d
-        cmpb      $16, %r12b
-        jb        .LBL_1_6
-        movups    112(%rsp), %xmm8
-        movups    96(%rsp), %xmm9
-        movups    80(%rsp), %xmm10
-        movups    64(%rsp), %xmm11
-        movups    48(%rsp), %xmm12
-        movups    32(%rsp), %xmm13
-        movups    16(%rsp), %xmm14
-        movups    (%rsp), %xmm15
-        movq      136(%rsp), %rsi
-        movq      128(%rsp), %rdi
-        movq      168(%rsp), %r12
-        cfi_restore (%r12)
-        movq      160(%rsp), %r13
-        cfi_restore (%r13)
-        movq      152(%rsp), %r14
-        cfi_restore (%r14)
-        movq      144(%rsp), %r15
-        cfi_restore (%r15)
-        movups    256(%rsp), %xmm1
-        jmp       .LBL_1_2
-        cfi_restore_state
-        movzbl    %r12b, %r15d
-        movss     196(%rsp,%r15,8), %xmm0
-        call      JUMPTARGET(logf)
-        movss     %xmm0, 260(%rsp,%r15,8)
-        jmp       .LBL_1_8
-        movzbl    %r12b, %r15d
-        movss     192(%rsp,%r15,8), %xmm0
-        call      JUMPTARGET(logf)
-        movss     %xmm0, 256(%rsp,%r15,8)
-        jmp       .LBL_1_7
+	movdqu	COMMON_DATA(_ILoRange)(%rip), %xmm1
+	   if mantissa_x<4/3
+        log(x) = exponent_x*log(2) + log(mantissa_x)
+	   if mantissa_x>4/3
+        log(x) = (exponent_x+1)*log(2) + log(0.5*mantissa_x)
+	   R = mantissa_x - 1,     if mantissa_x<4/3
+	   R = 0.5*mantissa_x - 1, if mantissa_x>4/3
+	   |R|< 1/3
+	   log(1+R) is approximated as a polynomial: degree 9 for
+	   1-ulp, degree 7 for 4-ulp, degree 3 for half-precision.  */
+	/* check for working range, set special argument mask
+	   (denormals/zero/Inf/NaN).  */
+	movdqu	COMMON_DATA(_NotiOffExpoMask)(%rip), %xmm2
+	movaps	%xmm0, %xmm3
+	psubd	%xmm2, %xmm3
+	pcmpgtd	%xmm3, %xmm1
+	movmskps %xmm1, %eax
+	movdqu	COMMON_DATA(_IBrkValue)(%rip), %xmm1
+	movaps	%xmm0, %xmm3
+	psubd	%xmm1, %xmm0
+	pandn	%xmm0, %xmm2
+	paddd	%xmm1, %xmm2
+	/* reduced argument R.  */
+	subps	COMMON_DATA(_OneF)(%rip), %xmm2
+	/* exponent_x (mantissa_x<4/3),
+	   or exponent_x+1 (mantissa_x>4/3).  */
+	psrad	$0x17, %xmm0
+	/* mantissa_x (mantissa_x<4/3),
+	   or 0.5 mantissa_x (mantissa_x>4/3).  */
+	cvtdq2ps %xmm0, %xmm0
+	/* final reconstruction: add exponent_value * log2 to polynomial
+	   result.  */
+	mulps	COMMON_DATA(_Ln2)(%rip), %xmm0
+	movups	LOCAL_DATA(_sPoly_7)(%rip), %xmm1
+	/* polynomial evaluation starts here.  */
+	mulps	%xmm2, %xmm1
+	addps	LOCAL_DATA(_sPoly_6)(%rip), %xmm1
+	mulps	%xmm2, %xmm1
+	addps	LOCAL_DATA(_sPoly_5)(%rip), %xmm1
+	mulps	%xmm2, %xmm1
+	addps	LOCAL_DATA(_sPoly_4)(%rip), %xmm1
+	mulps	%xmm2, %xmm1
+	addps	LOCAL_DATA(_sPoly_3)(%rip), %xmm1
+	mulps	%xmm2, %xmm1
+	addps	LOCAL_DATA(_sPoly_2)(%rip), %xmm1
+	mulps	%xmm2, %xmm1
+	addps	COMMON_DATA(_Neg5F)(%rip), %xmm1
+	mulps	%xmm2, %xmm1
+	/* polynomial evaluation end.  */
+	mulps	%xmm2, %xmm1
+	addps	%xmm1, %xmm2
+	addps	%xmm2, %xmm0
+	testl	%eax, %eax
+	ret
+	/* Cold case. edx has 1s where there was a special value that
+	   more so than speed here.  */
+	/* Stack coming in 16-byte aligned. Set 8-byte misaligned so on
+	   call entry will be 16-byte aligned.  */
+	subq	$0x38, %rsp
+	movups	%xmm0, 24(%rsp)
+	movups	%xmm3, 40(%rsp)
+	/* Use rbx/rbp for callee save registers as they get short
+	   encoding for many instructions (as compared with r12/r13).  */
+	movq	%rbx, (%rsp)
+	cfi_offset (rbx, -64)
+	movq	%rbp, 8(%rsp)
+	cfi_offset (rbp, -56)
+	/* edx has 1s where there was a special value that needs to be
+	   handled by a tanhf call.  */
+	movl	%eax, %ebx
+	/* use rbp as index for special value that is saved across calls
+	   to tanhf. We technically don't need a callee save register
+	   here as offset to rsp is always [0, 12] so we can restore
+	   rsp by realigning to 64. Essentially the tradeoff is 1 extra
+	   save/restore vs 2 extra instructions in the loop.  */
+	xorl	%ebp, %ebp
+	bsfl	%ebx, %ebp
+	/* Scalar math fucntion call to process special input.  */
+	movss	40(%rsp, %rbp, 4), %xmm0
+	call	logf@PLT
+	/* No good way to avoid the store-forwarding fault this will
+	   cause on return. `lfence` avoids the SF fault but at greater
+	   cost as it serialized stack/callee save restoration.  */
+	movss	%xmm0, 24(%rsp, %rbp, 4)
+	leal	-1(%rbx), %eax
+	andl	%eax, %ebx
+	/* All results have been written to 24(%rsp).  */
+	movups	24(%rsp), %xmm0
+	movq	(%rsp), %rbx
+	cfi_restore (rbx)
+	movq	8(%rsp), %rbp
+	cfi_restore (rbp)
+	addq	$56, %rsp
+	cfi_def_cfa_offset (8)
+	ret
 END (_ZGVbN4v_logf_sse4)
+	.section .rodata.sse4, "a"
+	.align	16
+	/* Data table for vector implementations of function logf. The
+	   table may contain polynomial, reduction, lookup coefficients
+	   and other coefficients obtained through different methods of
+	   research and experimental work.  */
+	/* Polynomial sPoly[] coefficients:.  */
+	/* -1.5177205204963684082031250e-01.  */
+	DATA_VEC (LOCAL_DATA_NAME, _sPoly_7, 0xbe1b6a22)
+	/* 1.6964881122112274169921875e-01.  */
+	DATA_VEC (LOCAL_DATA_NAME, _sPoly_6, 0x3e2db86b)
+	/* -1.6462457180023193359375000e-01.  */
+	DATA_VEC (LOCAL_DATA_NAME, _sPoly_5, 0xbe289358)
+	/* 1.9822503626346588134765625e-01.  */
+	DATA_VEC (LOCAL_DATA_NAME, _sPoly_4, 0x3e4afb81)
+	/* -2.5004664063453674316406250e-01.  */
+	DATA_VEC (LOCAL_DATA_NAME, _sPoly_3, 0xbe80061d)
+	/* 3.3336564898490905761718750e-01.  */
+	DATA_VEC (LOCAL_DATA_NAME, _sPoly_2, 0x3eaaaee7)
+	.type	LOCAL_DATA_NAME, @object