diff options
author | H.J. Lu <hjl.tools@gmail.com> | 2017-08-07 08:19:59 -0700 |
---|---|---|
committer | H.J. Lu <hjl.tools@gmail.com> | 2017-08-07 08:20:56 -0700 |
commit | 57a72fa3502673754d14707da02c7c44e83b8d20 (patch) | |
tree | 1dc9c5d4334079fc4245eec8b253d01732c6cf46 | |
parent | 219dd320d69deb9068f6b2ce46034d0eb4db888a (diff) | |
download | glibc-57a72fa3502673754d14707da02c7c44e83b8d20.zip glibc-57a72fa3502673754d14707da02c7c44e83b8d20.tar.gz glibc-57a72fa3502673754d14707da02c7c44e83b8d20.tar.bz2 |
x86-64: Add FMA multiarch functions to libm
This patch adds multiarch functions optimized with -mfma -mavx2 to libm.
e_pow-fma.c is compiled with $(config-cflags-nofma) due to PR 19003.
* sysdeps/x86_64/fpu/multiarch/Makefile (libm-sysdep_routines):
Add e_exp-fma, e_log-fma, e_pow-fma, s_atan-fma, e_asin-fma,
e_atan2-fma, s_sin-fma, s_tan-fma, mplog-fma, mpa-fma,
slowexp-fma, slowpow-fma, sincos32-fma, doasin-fma, dosincos-fma,
halfulp-fma, mpexp-fma, mpatan2-fma, mpatan-fma, mpsqrt-fma,
and mptan-fma.
(CFLAGS-doasin-fma.c): New.
(CFLAGS-dosincos-fma.c): Likewise.
(CFLAGS-e_asin-fma.c): Likewise.
(CFLAGS-e_atan2-fma.c): Likewise.
(CFLAGS-e_exp-fma.c): Likewise.
(CFLAGS-e_log-fma.c): Likewise.
(CFLAGS-e_pow-fma.c): Likewise.
(CFLAGS-halfulp-fma.c): Likewise.
(CFLAGS-mpa-fma.c): Likewise.
(CFLAGS-mpatan-fma.c): Likewise.
(CFLAGS-mpatan2-fma.c): Likewise.
(CFLAGS-mpexp-fma.c): Likewise.
(CFLAGS-mplog-fma.c): Likewise.
(CFLAGS-mpsqrt-fma.c): Likewise.
(CFLAGS-mptan-fma.c): Likewise.
(CFLAGS-s_atan-fma.c): Likewise.
(CFLAGS-sincos32-fma.c): Likewise.
(CFLAGS-slowexp-fma.c): Likewise.
(CFLAGS-slowpow-fma.c): Likewise.
(CFLAGS-s_sin-fma.c): Likewise.
(CFLAGS-s_tan-fma.c): Likewise.
* sysdeps/x86_64/fpu/multiarch/doasin-fma.c: New file.
* sysdeps/x86_64/fpu/multiarch/dosincos-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_asin-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_atan2-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_exp-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_log-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_pow-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/halfulp-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/ifunc-avx-fma4.h: Likewise.
* sysdeps/x86_64/fpu/multiarch/ifunc-fma4.h: Likewise.
* sysdeps/x86_64/fpu/multiarch/mpa-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/mpatan-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/mpatan2-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/mpexp-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/mplog-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/mpsqrt-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/mptan-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/s_atan-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/s_sin-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/s_tan-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/sincos32-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/slowexp-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/slowpow-fma.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_asin.c: Rewrite.
* sysdeps/x86_64/fpu/multiarch/e_atan2.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_exp.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_log.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/e_pow.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/s_atan.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/s_sin.c: Likewise.
* sysdeps/x86_64/fpu/multiarch/s_tan.c: Likewise.
33 files changed, 554 insertions, 105 deletions
@@ -1,3 +1,64 @@ +2017-08-07 H.J. Lu <hongjiu.lu@intel.com> + + * sysdeps/x86_64/fpu/multiarch/Makefile (libm-sysdep_routines): + Add e_exp-fma, e_log-fma, e_pow-fma, s_atan-fma, e_asin-fma, + e_atan2-fma, s_sin-fma, s_tan-fma, mplog-fma, mpa-fma, + slowexp-fma, slowpow-fma, sincos32-fma, doasin-fma, dosincos-fma, + halfulp-fma, mpexp-fma, mpatan2-fma, mpatan-fma, mpsqrt-fma, + and mptan-fma. + (CFLAGS-doasin-fma.c): New. + (CFLAGS-dosincos-fma.c): Likewise. + (CFLAGS-e_asin-fma.c): Likewise. + (CFLAGS-e_atan2-fma.c): Likewise. + (CFLAGS-e_exp-fma.c): Likewise. + (CFLAGS-e_log-fma.c): Likewise. + (CFLAGS-e_pow-fma.c): Likewise. + (CFLAGS-halfulp-fma.c): Likewise. + (CFLAGS-mpa-fma.c): Likewise. + (CFLAGS-mpatan-fma.c): Likewise. + (CFLAGS-mpatan2-fma.c): Likewise. + (CFLAGS-mpexp-fma.c): Likewise. + (CFLAGS-mplog-fma.c): Likewise. + (CFLAGS-mpsqrt-fma.c): Likewise. + (CFLAGS-mptan-fma.c): Likewise. + (CFLAGS-s_atan-fma.c): Likewise. + (CFLAGS-sincos32-fma.c): Likewise. + (CFLAGS-slowexp-fma.c): Likewise. + (CFLAGS-slowpow-fma.c): Likewise. + (CFLAGS-s_sin-fma.c): Likewise. + (CFLAGS-s_tan-fma.c): Likewise. + * sysdeps/x86_64/fpu/multiarch/doasin-fma.c: New file. + * sysdeps/x86_64/fpu/multiarch/dosincos-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_asin-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_atan2-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_exp-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_log-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_pow-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/halfulp-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/ifunc-avx-fma4.h: Likewise. + * sysdeps/x86_64/fpu/multiarch/ifunc-fma4.h: Likewise. + * sysdeps/x86_64/fpu/multiarch/mpa-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/mpatan-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/mpatan2-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/mpexp-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/mplog-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/mpsqrt-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/mptan-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/s_atan-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/s_sin-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/s_tan-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/sincos32-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/slowexp-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/slowpow-fma.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_asin.c: Rewrite. + * sysdeps/x86_64/fpu/multiarch/e_atan2.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_exp.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_log.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/e_pow.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/s_atan.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/s_sin.c: Likewise. + * sysdeps/x86_64/fpu/multiarch/s_tan.c: Likewise. + 2017-08-04 Joseph Myers <joseph@codesourcery.com> * sysdeps/generic/math_private.h (__EXPR_FLT128): Remove macro. diff --git a/sysdeps/x86_64/fpu/multiarch/Makefile b/sysdeps/x86_64/fpu/multiarch/Makefile index f9ceb09..9daf2cf 100644 --- a/sysdeps/x86_64/fpu/multiarch/Makefile +++ b/sysdeps/x86_64/fpu/multiarch/Makefile @@ -6,6 +6,35 @@ libm-sysdep_routines += s_ceil-sse4_1 s_ceilf-sse4_1 s_floor-sse4_1 \ s_floorf-sse4_1 s_nearbyint-sse4_1 \ s_nearbyintf-sse4_1 s_rint-sse4_1 s_rintf-sse4_1 +libm-sysdep_routines += e_exp-fma e_log-fma e_pow-fma s_atan-fma \ + e_asin-fma e_atan2-fma s_sin-fma s_tan-fma \ + mplog-fma mpa-fma slowexp-fma slowpow-fma \ + sincos32-fma doasin-fma dosincos-fma \ + halfulp-fma mpexp-fma \ + mpatan2-fma mpatan-fma mpsqrt-fma mptan-fma + +CFLAGS-doasin-fma.c = -mfma -mavx2 +CFLAGS-dosincos-fma.c = -mfma -mavx2 +CFLAGS-e_asin-fma.c = -mfma -mavx2 +CFLAGS-e_atan2-fma.c = -mfma -mavx2 +CFLAGS-e_exp-fma.c = -mfma -mavx2 +CFLAGS-e_log-fma.c = -mfma -mavx2 +CFLAGS-e_pow-fma.c = -mfma -mavx2 $(config-cflags-nofma) +CFLAGS-halfulp-fma.c = -mfma -mavx2 +CFLAGS-mpa-fma.c = -mfma -mavx2 +CFLAGS-mpatan-fma.c = -mfma -mavx2 +CFLAGS-mpatan2-fma.c = -mfma -mavx2 +CFLAGS-mpexp-fma.c = -mfma -mavx2 +CFLAGS-mplog-fma.c = -mfma -mavx2 +CFLAGS-mpsqrt-fma.c = -mfma -mavx2 +CFLAGS-mptan-fma.c = -mfma -mavx2 +CFLAGS-s_atan-fma.c = -mfma -mavx2 +CFLAGS-sincos32-fma.c = -mfma -mavx2 +CFLAGS-slowexp-fma.c = -mfma -mavx2 +CFLAGS-slowpow-fma.c = -mfma -mavx2 +CFLAGS-s_sin-fma.c = -mfma -mavx2 +CFLAGS-s_tan-fma.c = -mfma -mavx2 + libm-sysdep_routines += e_exp-fma4 e_log-fma4 e_pow-fma4 s_atan-fma4 \ e_asin-fma4 e_atan2-fma4 s_sin-fma4 s_tan-fma4 \ mplog-fma4 mpa-fma4 slowexp-fma4 slowpow-fma4 \ diff --git a/sysdeps/x86_64/fpu/multiarch/doasin-fma.c b/sysdeps/x86_64/fpu/multiarch/doasin-fma.c new file mode 100644 index 0000000..7a09865 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/doasin-fma.c @@ -0,0 +1,4 @@ +#define __doasin __doasin_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/doasin.c> diff --git a/sysdeps/x86_64/fpu/multiarch/dosincos-fma.c b/sysdeps/x86_64/fpu/multiarch/dosincos-fma.c new file mode 100644 index 0000000..5744586 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/dosincos-fma.c @@ -0,0 +1,6 @@ +#define __docos __docos_fma +#define __dubcos __dubcos_fma +#define __dubsin __dubsin_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/dosincos.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_asin-fma.c b/sysdeps/x86_64/fpu/multiarch/e_asin-fma.c new file mode 100644 index 0000000..50e9c64 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/e_asin-fma.c @@ -0,0 +1,11 @@ +#define __ieee754_acos __ieee754_acos_fma +#define __ieee754_asin __ieee754_asin_fma +#define __cos32 __cos32_fma +#define __doasin __doasin_fma +#define __docos __docos_fma +#define __dubcos __dubcos_fma +#define __dubsin __dubsin_fma +#define __sin32 __sin32_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/e_asin.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_asin.c b/sysdeps/x86_64/fpu/multiarch/e_asin.c index 111a5b9..37a44b2 100644 --- a/sysdeps/x86_64/fpu/multiarch/e_asin.c +++ b/sysdeps/x86_64/fpu/multiarch/e_asin.c @@ -1,26 +1,40 @@ -#include <init-arch.h> -#include <math.h> -#include <math_private.h> - -extern double __ieee754_acos_sse2 (double); -extern double __ieee754_asin_sse2 (double); -extern double __ieee754_acos_fma4 (double); -extern double __ieee754_asin_fma4 (double); - -libm_ifunc (__ieee754_acos, - HAS_ARCH_FEATURE (FMA4_Usable) - ? __ieee754_acos_fma4 - : __ieee754_acos_sse2); -strong_alias (__ieee754_acos, __acos_finite) +/* Multiple versions of IEEE 754 asin and acos. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_ieee754_asin (double); +extern double __redirect_ieee754_acos (double); + +#define SYMBOL_NAME ieee754_asin +#include "ifunc-fma4.h" -libm_ifunc (__ieee754_asin, - HAS_ARCH_FEATURE (FMA4_Usable) - ? __ieee754_asin_fma4 - : __ieee754_asin_sse2); +libc_ifunc_redirected (__redirect_ieee754_asin, __ieee754_asin, + IFUNC_SELECTOR ()); strong_alias (__ieee754_asin, __asin_finite) -#define __ieee754_acos __ieee754_acos_sse2 -#define __ieee754_asin __ieee754_asin_sse2 +#undef SYMBOL_NAME +#define SYMBOL_NAME ieee754_acos +#include "ifunc-fma4.h" + +libc_ifunc_redirected (__redirect_ieee754_acos, __ieee754_acos, + IFUNC_SELECTOR ()); +strong_alias (__ieee754_acos, __acos_finite) +#define __ieee754_acos __ieee754_acos_sse2 +#define __ieee754_asin __ieee754_asin_sse2 #include <sysdeps/ieee754/dbl-64/e_asin.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_atan2-fma.c b/sysdeps/x86_64/fpu/multiarch/e_atan2-fma.c new file mode 100644 index 0000000..caba686 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/e_atan2-fma.c @@ -0,0 +1,10 @@ +#define __ieee754_atan2 __ieee754_atan2_fma +#define __add __add_fma +#define __dbl_mp __dbl_mp_fma +#define __dvd __dvd_fma +#define __mpatan2 __mpatan2_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/e_atan2.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_atan2.c b/sysdeps/x86_64/fpu/multiarch/e_atan2.c index 9ca3c02..a2c8cfc 100644 --- a/sysdeps/x86_64/fpu/multiarch/e_atan2.c +++ b/sysdeps/x86_64/fpu/multiarch/e_atan2.c @@ -1,18 +1,29 @@ -#include <init-arch.h> -#include <math.h> -#include <math_private.h> +/* Multiple versions of IEEE 754 atan. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. -extern double __ieee754_atan2_sse2 (double, double); -extern double __ieee754_atan2_avx (double, double); -extern double __ieee754_atan2_fma4 (double, double); + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. -libm_ifunc (__ieee754_atan2, - HAS_ARCH_FEATURE (FMA4_Usable) ? __ieee754_atan2_fma4 - : (HAS_ARCH_FEATURE (AVX_Usable) - ? __ieee754_atan2_avx : __ieee754_atan2_sse2)); -strong_alias (__ieee754_atan2, __atan2_finite) + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. -#define __ieee754_atan2 __ieee754_atan2_sse2 + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_ieee754_atan2 (double, double); +#define SYMBOL_NAME ieee754_atan2 +#include "ifunc-avx-fma4.h" +libc_ifunc_redirected (__redirect_ieee754_atan2, + __ieee754_atan2, IFUNC_SELECTOR ()); +strong_alias (__ieee754_atan2, __atan2_finite) + +#define __ieee754_atan2 __ieee754_atan2_sse2 #include <sysdeps/ieee754/dbl-64/e_atan2.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_exp-fma.c b/sysdeps/x86_64/fpu/multiarch/e_exp-fma.c new file mode 100644 index 0000000..6e0fdb7 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/e_exp-fma.c @@ -0,0 +1,6 @@ +#define __ieee754_exp __ieee754_exp_fma +#define __exp1 __exp1_fma +#define __slowexp __slowexp_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/e_exp.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_exp.c b/sysdeps/x86_64/fpu/multiarch/e_exp.c index b7d7b5f..81211f4 100644 --- a/sysdeps/x86_64/fpu/multiarch/e_exp.c +++ b/sysdeps/x86_64/fpu/multiarch/e_exp.c @@ -1,18 +1,29 @@ -#include <init-arch.h> -#include <math.h> -#include <math_private.h> +/* Multiple versions of IEEE 754 exp. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. -extern double __ieee754_exp_sse2 (double); -extern double __ieee754_exp_avx (double); -extern double __ieee754_exp_fma4 (double); + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. -libm_ifunc (__ieee754_exp, - HAS_ARCH_FEATURE (FMA4_Usable) ? __ieee754_exp_fma4 - : (HAS_ARCH_FEATURE (AVX_Usable) - ? __ieee754_exp_avx : __ieee754_exp_sse2)); -strong_alias (__ieee754_exp, __exp_finite) + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. -#define __ieee754_exp __ieee754_exp_sse2 + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_ieee754_exp (double); +#define SYMBOL_NAME ieee754_exp +#include "ifunc-avx-fma4.h" +libc_ifunc_redirected (__redirect_ieee754_exp, __ieee754_exp, + IFUNC_SELECTOR ()); +strong_alias (__ieee754_exp, __exp_finite) + +#define __ieee754_exp __ieee754_exp_sse2 #include <sysdeps/ieee754/dbl-64/e_exp.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_log-fma.c b/sysdeps/x86_64/fpu/multiarch/e_log-fma.c new file mode 100644 index 0000000..a7123b1 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/e_log-fma.c @@ -0,0 +1,8 @@ +#define __ieee754_log __ieee754_log_fma +#define __mplog __mplog_fma +#define __add __add_fma +#define __dbl_mp __dbl_mp_fma +#define __sub __sub_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/e_log.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_log.c b/sysdeps/x86_64/fpu/multiarch/e_log.c index cf9533d..de1fbf1 100644 --- a/sysdeps/x86_64/fpu/multiarch/e_log.c +++ b/sysdeps/x86_64/fpu/multiarch/e_log.c @@ -1,18 +1,29 @@ -#include <init-arch.h> -#include <math.h> -#include <math_private.h> +/* Multiple versions of IEEE 754 log. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. -extern double __ieee754_log_sse2 (double); -extern double __ieee754_log_avx (double); -extern double __ieee754_log_fma4 (double); + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. -libm_ifunc (__ieee754_log, - HAS_ARCH_FEATURE (FMA4_Usable) ? __ieee754_log_fma4 - : (HAS_ARCH_FEATURE (AVX_Usable) - ? __ieee754_log_avx : __ieee754_log_sse2)); -strong_alias (__ieee754_log, __log_finite) + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. -#define __ieee754_log __ieee754_log_sse2 + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_ieee754_log (double); +#define SYMBOL_NAME ieee754_log +#include "ifunc-avx-fma4.h" +libc_ifunc_redirected (__redirect_ieee754_log, __ieee754_log, + IFUNC_SELECTOR ()); +strong_alias (__ieee754_log, __log_finite) + +#define __ieee754_log __ieee754_log_sse2 #include <sysdeps/ieee754/dbl-64/e_log.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_pow-fma.c b/sysdeps/x86_64/fpu/multiarch/e_pow-fma.c new file mode 100644 index 0000000..6fd4083 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/e_pow-fma.c @@ -0,0 +1,6 @@ +#define __ieee754_pow __ieee754_pow_fma +#define __exp1 __exp1_fma +#define __slowpow __slowpow_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/e_pow.c> diff --git a/sysdeps/x86_64/fpu/multiarch/e_pow.c b/sysdeps/x86_64/fpu/multiarch/e_pow.c index a5c5d89..07a7929 100644 --- a/sysdeps/x86_64/fpu/multiarch/e_pow.c +++ b/sysdeps/x86_64/fpu/multiarch/e_pow.c @@ -1,17 +1,29 @@ -#include <init-arch.h> -#include <math.h> -#include <math_private.h> +/* Multiple versions of IEEE 754 pow. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. -extern double __ieee754_pow_sse2 (double, double); -extern double __ieee754_pow_fma4 (double, double); + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. -libm_ifunc (__ieee754_pow, - HAS_ARCH_FEATURE (FMA4_Usable) - ? __ieee754_pow_fma4 - : __ieee754_pow_sse2); -strong_alias (__ieee754_pow, __pow_finite) + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. -#define __ieee754_pow __ieee754_pow_sse2 + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_ieee754_pow (double, double); +#define SYMBOL_NAME ieee754_pow +#include "ifunc-fma4.h" +libc_ifunc_redirected (__redirect_ieee754_pow, + __ieee754_pow, IFUNC_SELECTOR ()); +strong_alias (__ieee754_pow, __pow_finite) + +#define __ieee754_pow __ieee754_pow_sse2 #include <sysdeps/ieee754/dbl-64/e_pow.c> diff --git a/sysdeps/x86_64/fpu/multiarch/halfulp-fma.c b/sysdeps/x86_64/fpu/multiarch/halfulp-fma.c new file mode 100644 index 0000000..6ca7046 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/halfulp-fma.c @@ -0,0 +1,4 @@ +#define __halfulp __halfulp_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/halfulp.c> diff --git a/sysdeps/x86_64/fpu/multiarch/ifunc-avx-fma4.h b/sysdeps/x86_64/fpu/multiarch/ifunc-avx-fma4.h new file mode 100644 index 0000000..2277c0a --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/ifunc-avx-fma4.h @@ -0,0 +1,43 @@ +/* Common definition for ifunc selections optimized with AVX, AVX2/FMA + and FMA4. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +#include <init-arch.h> + +extern __typeof (REDIRECT_NAME) OPTIMIZE (sse2) attribute_hidden; +extern __typeof (REDIRECT_NAME) OPTIMIZE (avx) attribute_hidden; +extern __typeof (REDIRECT_NAME) OPTIMIZE (fma) attribute_hidden; +extern __typeof (REDIRECT_NAME) OPTIMIZE (fma4) attribute_hidden; + +static inline void * +IFUNC_SELECTOR (void) +{ + const struct cpu_features* cpu_features = __get_cpu_features (); + + if (CPU_FEATURES_ARCH_P (cpu_features, FMA_Usable) + && CPU_FEATURES_ARCH_P (cpu_features, AVX2_Usable)) + return OPTIMIZE (fma); + + if (CPU_FEATURES_ARCH_P (cpu_features, FMA4_Usable)) + return OPTIMIZE (fma4); + + if (CPU_FEATURES_ARCH_P (cpu_features, AVX_Usable)) + return OPTIMIZE (avx); + + return OPTIMIZE (sse2); +} diff --git a/sysdeps/x86_64/fpu/multiarch/ifunc-fma4.h b/sysdeps/x86_64/fpu/multiarch/ifunc-fma4.h new file mode 100644 index 0000000..5928984 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/ifunc-fma4.h @@ -0,0 +1,39 @@ +/* Common definition for ifunc selections optimized with AVX2/FMA and + FMA4. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +#include <init-arch.h> + +extern __typeof (REDIRECT_NAME) OPTIMIZE (sse2) attribute_hidden; +extern __typeof (REDIRECT_NAME) OPTIMIZE (fma) attribute_hidden; +extern __typeof (REDIRECT_NAME) OPTIMIZE (fma4) attribute_hidden; + +static inline void * +IFUNC_SELECTOR (void) +{ + const struct cpu_features* cpu_features = __get_cpu_features (); + + if (CPU_FEATURES_ARCH_P (cpu_features, FMA_Usable) + && CPU_FEATURES_ARCH_P (cpu_features, AVX2_Usable)) + return OPTIMIZE (fma); + + if (CPU_FEATURES_ARCH_P (cpu_features, FMA4_Usable)) + return OPTIMIZE (fma4); + + return OPTIMIZE (sse2); +} diff --git a/sysdeps/x86_64/fpu/multiarch/mpa-fma.c b/sysdeps/x86_64/fpu/multiarch/mpa-fma.c new file mode 100644 index 0000000..177cc25 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/mpa-fma.c @@ -0,0 +1,14 @@ +#define __add __add_fma +#define __mul __mul_fma +#define __sqr __sqr_fma +#define __sub __sub_fma +#define __dbl_mp __dbl_mp_fma +#define __dvd __dvd_fma + +#define NO___CPY 1 +#define NO___MP_DBL 1 +#define NO___ACR 1 +#define NO__CONST 1 +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/mpa.c> diff --git a/sysdeps/x86_64/fpu/multiarch/mpatan-fma.c b/sysdeps/x86_64/fpu/multiarch/mpatan-fma.c new file mode 100644 index 0000000..d216f91 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/mpatan-fma.c @@ -0,0 +1,10 @@ +#define __mpatan __mpatan_fma +#define __add __add_fma +#define __dvd __dvd_fma +#define __mpsqrt __mpsqrt_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define AVOID_MPATAN_H 1 +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/mpatan.c> diff --git a/sysdeps/x86_64/fpu/multiarch/mpatan2-fma.c b/sysdeps/x86_64/fpu/multiarch/mpatan2-fma.c new file mode 100644 index 0000000..98df336 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/mpatan2-fma.c @@ -0,0 +1,9 @@ +#define __mpatan2 __mpatan2_fma +#define __add __add_fma +#define __dvd __dvd_fma +#define __mpatan __mpatan_fma +#define __mpsqrt __mpsqrt_fma +#define __mul __mul_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/mpatan2.c> diff --git a/sysdeps/x86_64/fpu/multiarch/mpexp-fma.c b/sysdeps/x86_64/fpu/multiarch/mpexp-fma.c new file mode 100644 index 0000000..637631b --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/mpexp-fma.c @@ -0,0 +1,9 @@ +#define __mpexp __mpexp_fma +#define __add __add_fma +#define __dbl_mp __dbl_mp_fma +#define __dvd __dvd_fma +#define __mul __mul_fma +#define AVOID_MPEXP_H 1 +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/mpexp.c> diff --git a/sysdeps/x86_64/fpu/multiarch/mplog-fma.c b/sysdeps/x86_64/fpu/multiarch/mplog-fma.c new file mode 100644 index 0000000..645b6b7 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/mplog-fma.c @@ -0,0 +1,8 @@ +#define __mplog __mplog_fma +#define __add __add_fma +#define __mpexp __mpexp_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/mplog.c> diff --git a/sysdeps/x86_64/fpu/multiarch/mpsqrt-fma.c b/sysdeps/x86_64/fpu/multiarch/mpsqrt-fma.c new file mode 100644 index 0000000..44d7a23 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/mpsqrt-fma.c @@ -0,0 +1,8 @@ +#define __mpsqrt __mpsqrt_fma +#define __dbl_mp __dbl_mp_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define AVOID_MPSQRT_H 1 +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/mpsqrt.c> diff --git a/sysdeps/x86_64/fpu/multiarch/mptan-fma.c b/sysdeps/x86_64/fpu/multiarch/mptan-fma.c new file mode 100644 index 0000000..d1a6914 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/mptan-fma.c @@ -0,0 +1,7 @@ +#define __mptan __mptan_fma +#define __c32 __c32_fma +#define __dvd __dvd_fma +#define __mpranred __mpranred_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/mptan.c> diff --git a/sysdeps/x86_64/fpu/multiarch/s_atan-fma.c b/sysdeps/x86_64/fpu/multiarch/s_atan-fma.c new file mode 100644 index 0000000..bedb3f2 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/s_atan-fma.c @@ -0,0 +1,9 @@ +#define atan __atan_fma +#define __add __add_fma +#define __dbl_mp __dbl_mp_fma +#define __mpatan __mpatan_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/s_atan.c> diff --git a/sysdeps/x86_64/fpu/multiarch/s_atan.c b/sysdeps/x86_64/fpu/multiarch/s_atan.c index 742e95c..f81919c 100644 --- a/sysdeps/x86_64/fpu/multiarch/s_atan.c +++ b/sysdeps/x86_64/fpu/multiarch/s_atan.c @@ -1,15 +1,27 @@ -#include <init-arch.h> -#include <math.h> +/* Multiple versions of atan. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. -extern double __atan_sse2 (double); -extern double __atan_avx (double); -extern double __atan_fma4 (double); + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. -libm_ifunc (atan, (HAS_ARCH_FEATURE (FMA4_Usable) ? __atan_fma4 : - HAS_ARCH_FEATURE (AVX_Usable) - ? __atan_avx : __atan_sse2)); + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. -#define atan __atan_sse2 + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_atan (double); +#define SYMBOL_NAME atan +#include "ifunc-avx-fma4.h" +libc_ifunc_redirected (__redirect_atan, atan, IFUNC_SELECTOR ()); + +#define atan __atan_sse2 #include <sysdeps/ieee754/dbl-64/s_atan.c> diff --git a/sysdeps/x86_64/fpu/multiarch/s_sin-fma.c b/sysdeps/x86_64/fpu/multiarch/s_sin-fma.c new file mode 100644 index 0000000..15f3c39 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/s_sin-fma.c @@ -0,0 +1,11 @@ +#define __cos __cos_fma +#define __sin __sin_fma +#define __docos __docos_fma +#define __dubsin __dubsin_fma +#define __mpcos __mpcos_fma +#define __mpcos1 __mpcos1_fma +#define __mpsin __mpsin_fma +#define __mpsin1 __mpsin1_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/s_sin.c> diff --git a/sysdeps/x86_64/fpu/multiarch/s_sin.c b/sysdeps/x86_64/fpu/multiarch/s_sin.c index 8ffd3e7..eafc063 100644 --- a/sysdeps/x86_64/fpu/multiarch/s_sin.c +++ b/sysdeps/x86_64/fpu/multiarch/s_sin.c @@ -1,26 +1,37 @@ -#include <init-arch.h> -#include <math.h> -#undef NAN - -extern double __cos_sse2 (double); -extern double __sin_sse2 (double); -extern double __cos_avx (double); -extern double __sin_avx (double); -extern double __cos_fma4 (double); -extern double __sin_fma4 (double); - -libm_ifunc (__cos, (HAS_ARCH_FEATURE (FMA4_Usable) ? __cos_fma4 : - HAS_ARCH_FEATURE (AVX_Usable) - ? __cos_avx : __cos_sse2)); -weak_alias (__cos, cos) +/* Multiple versions of sin and cos. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. + + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. + + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. + + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_sin (double); +extern double __redirect_cos (double); + +#define SYMBOL_NAME sin +#include "ifunc-avx-fma4.h" -libm_ifunc (__sin, (HAS_ARCH_FEATURE (FMA4_Usable) ? __sin_fma4 : - HAS_ARCH_FEATURE (AVX_Usable) - ? __sin_avx : __sin_sse2)); +libc_ifunc_redirected (__redirect_sin, __sin, IFUNC_SELECTOR ()); weak_alias (__sin, sin) -#define __cos __cos_sse2 -#define __sin __sin_sse2 +#undef SYMBOL_NAME +#define SYMBOL_NAME cos +#include "ifunc-avx-fma4.h" +libc_ifunc_redirected (__redirect_cos, __cos, IFUNC_SELECTOR ()); +weak_alias (__cos, cos) +#define __cos __cos_sse2 +#define __sin __sin_sse2 #include <sysdeps/ieee754/dbl-64/s_sin.c> diff --git a/sysdeps/x86_64/fpu/multiarch/s_tan-fma.c b/sysdeps/x86_64/fpu/multiarch/s_tan-fma.c new file mode 100644 index 0000000..c85f8bc --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/s_tan-fma.c @@ -0,0 +1,8 @@ +#define tan __tan_fma +#define __dbl_mp __dbl_mp_fma +#define __mpranred __mpranred_fma +#define __mptan __mptan_fma +#define __sub __sub_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/s_tan.c> diff --git a/sysdeps/x86_64/fpu/multiarch/s_tan.c b/sysdeps/x86_64/fpu/multiarch/s_tan.c index 25f3bca..96a7381 100644 --- a/sysdeps/x86_64/fpu/multiarch/s_tan.c +++ b/sysdeps/x86_64/fpu/multiarch/s_tan.c @@ -1,15 +1,27 @@ -#include <init-arch.h> -#include <math.h> +/* Multiple versions of tan. + Copyright (C) 2017 Free Software Foundation, Inc. + This file is part of the GNU C Library. -extern double __tan_sse2 (double); -extern double __tan_avx (double); -extern double __tan_fma4 (double); + The GNU C Library is free software; you can redistribute it and/or + modify it under the terms of the GNU Lesser General Public + License as published by the Free Software Foundation; either + version 2.1 of the License, or (at your option) any later version. -libm_ifunc (tan, (HAS_ARCH_FEATURE (FMA4_Usable) ? __tan_fma4 : - HAS_ARCH_FEATURE (AVX_Usable) - ? __tan_avx : __tan_sse2)); + The GNU C Library is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + Lesser General Public License for more details. -#define tan __tan_sse2 + You should have received a copy of the GNU Lesser General Public + License along with the GNU C Library; if not, see + <http://www.gnu.org/licenses/>. */ + +extern double __redirect_tan (double); +#define SYMBOL_NAME tan +#include "ifunc-avx-fma4.h" +libc_ifunc_redirected (__redirect_tan, tan, IFUNC_SELECTOR ()); + +#define tan __tan_sse2 #include <sysdeps/ieee754/dbl-64/s_tan.c> diff --git a/sysdeps/x86_64/fpu/multiarch/sincos32-fma.c b/sysdeps/x86_64/fpu/multiarch/sincos32-fma.c new file mode 100644 index 0000000..dcd44bc --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/sincos32-fma.c @@ -0,0 +1,15 @@ +#define __cos32 __cos32_fma +#define __sin32 __sin32_fma +#define __c32 __c32_fma +#define __mpsin __mpsin_fma +#define __mpsin1 __mpsin1_fma +#define __mpcos __mpcos_fma +#define __mpcos1 __mpcos1_fma +#define __mpranred __mpranred_fma +#define __add __add_fma +#define __dbl_mp __dbl_mp_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/sincos32.c> diff --git a/sysdeps/x86_64/fpu/multiarch/slowexp-fma.c b/sysdeps/x86_64/fpu/multiarch/slowexp-fma.c new file mode 100644 index 0000000..6fffca1 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/slowexp-fma.c @@ -0,0 +1,9 @@ +#define __slowexp __slowexp_fma +#define __add __add_fma +#define __dbl_mp __dbl_mp_fma +#define __mpexp __mpexp_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/slowexp.c> diff --git a/sysdeps/x86_64/fpu/multiarch/slowpow-fma.c b/sysdeps/x86_64/fpu/multiarch/slowpow-fma.c new file mode 100644 index 0000000..160ed68 --- /dev/null +++ b/sysdeps/x86_64/fpu/multiarch/slowpow-fma.c @@ -0,0 +1,11 @@ +#define __slowpow __slowpow_fma +#define __add __add_fma +#define __dbl_mp __dbl_mp_fma +#define __mpexp __mpexp_fma +#define __mplog __mplog_fma +#define __mul __mul_fma +#define __sub __sub_fma +#define __halfulp __halfulp_fma +#define SECTION __attribute__ ((section (".text.fma"))) + +#include <sysdeps/ieee754/dbl-64/slowpow.c> |