aboutsummaryrefslogtreecommitdiff
path: root/gcc
diff options
context:
space:
mode:
authorliuhongt <hongtao.liu@intel.com>2020-07-10 15:22:30 +0800
committerliuhongt <hongtao.liu@intel.com>2021-09-22 12:56:31 +0800
commit59e9c4cbe665951f3f3714c966ecf7776b36894c (patch)
tree1869eab977bc038657754b334d622b4567a92b97 /gcc
parent8a5837cfb714d6a677332ee946b02f7f3558fca4 (diff)
downloadgcc-59e9c4cbe665951f3f3714c966ecf7776b36894c.zip
gcc-59e9c4cbe665951f3f3714c966ecf7776b36894c.tar.gz
gcc-59e9c4cbe665951f3f3714c966ecf7776b36894c.tar.bz2
AVX512FP16: Add expander for sqrthf2.
gcc/ChangeLog: * config/i386/i386-features.c (i386-features.c): Handle E_HFmode. * config/i386/i386.md (sqrthf2): New expander. (*sqrthf2): New define_insn. * config/i386/sse.md (*<sse>_vmsqrt<mode>2<mask_scalar_name><round_scalar_name>): Extend to VFH_128. gcc/testsuite/ChangeLog: * gcc.target/i386/avx512fp16-builtin-sqrt-1.c: New test. * gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c: New test.
Diffstat (limited to 'gcc')
-rw-r--r--gcc/config/i386/i386-features.c15
-rw-r--r--gcc/config/i386/i386.md13
-rw-r--r--gcc/config/i386/sse.md8
-rw-r--r--gcc/testsuite/gcc.target/i386/avx512fp16-builtin-sqrt-1.c18
-rw-r--r--gcc/testsuite/gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c19
5 files changed, 65 insertions, 8 deletions
diff --git a/gcc/config/i386/i386-features.c b/gcc/config/i386/i386-features.c
index 14f816f..43bb676 100644
--- a/gcc/config/i386/i386-features.c
+++ b/gcc/config/i386/i386-features.c
@@ -2258,15 +2258,22 @@ remove_partial_avx_dependency (void)
rtx zero;
machine_mode dest_vecmode;
- if (dest_mode == E_SFmode)
+ switch (dest_mode)
{
+ case E_HFmode:
+ dest_vecmode = V8HFmode;
+ zero = gen_rtx_SUBREG (V8HFmode, v4sf_const0, 0);
+ break;
+ case E_SFmode:
dest_vecmode = V4SFmode;
zero = v4sf_const0;
- }
- else
- {
+ break;
+ case E_DFmode:
dest_vecmode = V2DFmode;
zero = gen_rtx_SUBREG (V2DFmode, v4sf_const0, 0);
+ break;
+ default:
+ gcc_unreachable ();
}
/* Change source to vector mode. */
diff --git a/gcc/config/i386/i386.md b/gcc/config/i386/i386.md
index 188f431..ae1a81c 100644
--- a/gcc/config/i386/i386.md
+++ b/gcc/config/i386/i386.md
@@ -17041,6 +17041,19 @@
DONE;
})
+(define_insn "sqrthf2"
+ [(set (match_operand:HF 0 "register_operand" "=v,v")
+ (sqrt:HF
+ (match_operand:HF 1 "nonimmediate_operand" "v,m")))]
+ "TARGET_AVX512FP16"
+ "@
+ vsqrtsh\t{%d1, %0|%0, %d1}
+ vsqrtsh\t{%1, %d0|%d0, %1}"
+ [(set_attr "type" "sse")
+ (set_attr "prefix" "evex")
+ (set_attr "avx_partial_xmm_update" "false,true")
+ (set_attr "mode" "HF")])
+
(define_insn "*sqrt<mode>2_sse"
[(set (match_operand:MODEF 0 "register_operand" "=v,v,v")
(sqrt:MODEF
diff --git a/gcc/config/i386/sse.md b/gcc/config/i386/sse.md
index b08a9d3..e8aef0d 100644
--- a/gcc/config/i386/sse.md
+++ b/gcc/config/i386/sse.md
@@ -2522,12 +2522,12 @@
(set_attr "mode" "<ssescalarmode>")])
(define_insn "*<sse>_vmsqrt<mode>2<mask_scalar_name><round_scalar_name>"
- [(set (match_operand:VF_128 0 "register_operand" "=x,v")
- (vec_merge:VF_128
- (vec_duplicate:VF_128
+ [(set (match_operand:VFH_128 0 "register_operand" "=x,v")
+ (vec_merge:VFH_128
+ (vec_duplicate:VFH_128
(sqrt:<ssescalarmode>
(match_operand:<ssescalarmode> 1 "nonimmediate_operand" "xm,<round_scalar_constraint>")))
- (match_operand:VF_128 2 "register_operand" "0,v")
+ (match_operand:VFH_128 2 "register_operand" "0,v")
(const_int 1)))]
"TARGET_SSE"
"@
diff --git a/gcc/testsuite/gcc.target/i386/avx512fp16-builtin-sqrt-1.c b/gcc/testsuite/gcc.target/i386/avx512fp16-builtin-sqrt-1.c
new file mode 100644
index 0000000..b4efc84
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/avx512fp16-builtin-sqrt-1.c
@@ -0,0 +1,18 @@
+/* { dg-do compile } */
+/* { dg-options "-Ofast -mavx512fp16 -mprefer-vector-width=512" } */
+
+_Float16
+f1 (_Float16 x)
+{
+ return __builtin_sqrtf16 (x);
+}
+
+void
+f2 (_Float16* __restrict psrc, _Float16* __restrict pdst)
+{
+ for (int i = 0; i != 32; i++)
+ pdst[i] = __builtin_sqrtf16 (psrc[i]);
+}
+
+/* { dg-final { scan-assembler-times "vsqrtsh\[^\n\r\]*xmm\[0-9\]" 1 } } */
+/* { dg-final { scan-assembler-times "vsqrtph\[^\n\r\]*zmm\[0-9\]" 1 } } */
diff --git a/gcc/testsuite/gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c b/gcc/testsuite/gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c
new file mode 100644
index 0000000..08deb3e
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c
@@ -0,0 +1,19 @@
+/* { dg-do compile } */
+/* { dg-options "-Ofast -mavx512fp16 -mavx512vl" } */
+
+void
+f1 (_Float16* __restrict psrc, _Float16* __restrict pdst)
+{
+ for (int i = 0; i != 8; i++)
+ pdst[i] = __builtin_sqrtf16 (psrc[i]);
+}
+
+void
+f2 (_Float16* __restrict psrc, _Float16* __restrict pdst)
+{
+ for (int i = 0; i != 16; i++)
+ pdst[i] = __builtin_sqrtf16 (psrc[i]);
+}
+
+/* { dg-final { scan-assembler-times "vsqrtph\[^\n\r\]*xmm\[0-9\]" 1 } } */
+/* { dg-final { scan-assembler-times "vsqrtph\[^\n\r\]*ymm\[0-9\]" 1 } } */