AVX512FP16: Add expander for sqrthf2.
authorliuhongt <hongtao.liu@intel.com>
Fri, 10 Jul 2020 07:22:30 +0000 (15:22 +0800)
committerliuhongt <hongtao.liu@intel.com>
Wed, 22 Sep 2021 04:56:31 +0000 (12:56 +0800)
gcc/ChangeLog:

* config/i386/i386-features.c (i386-features.c): Handle
E_HFmode.
* config/i386/i386.md (sqrthf2): New expander.
(*sqrthf2): New define_insn.
* config/i386/sse.md
(*<sse>_vmsqrt<mode>2<mask_scalar_name><round_scalar_name>):
Extend to VFH_128.

gcc/testsuite/ChangeLog:

* gcc.target/i386/avx512fp16-builtin-sqrt-1.c: New test.
* gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c: New test.

gcc/config/i386/i386-features.c
gcc/config/i386/i386.md
gcc/config/i386/sse.md
gcc/testsuite/gcc.target/i386/avx512fp16-builtin-sqrt-1.c [new file with mode: 0644]
gcc/testsuite/gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c [new file with mode: 0644]

index 14f816f..43bb676 100644 (file)
@@ -2258,15 +2258,22 @@ remove_partial_avx_dependency (void)
 
          rtx zero;
          machine_mode dest_vecmode;
-         if (dest_mode == E_SFmode)
+         switch (dest_mode)
            {
+           case E_HFmode:
+             dest_vecmode = V8HFmode;
+             zero = gen_rtx_SUBREG (V8HFmode, v4sf_const0, 0);
+             break;
+           case E_SFmode:
              dest_vecmode = V4SFmode;
              zero = v4sf_const0;
-           }
-         else
-           {
+             break;
+           case E_DFmode:
              dest_vecmode = V2DFmode;
              zero = gen_rtx_SUBREG (V2DFmode, v4sf_const0, 0);
+             break;
+           default:
+             gcc_unreachable ();
            }
 
          /* Change source to vector mode.  */
index 188f431..ae1a81c 100644 (file)
   DONE;
 })
 
+(define_insn "sqrthf2"
+  [(set (match_operand:HF 0 "register_operand" "=v,v")
+       (sqrt:HF
+         (match_operand:HF 1 "nonimmediate_operand" "v,m")))]
+  "TARGET_AVX512FP16"
+  "@
+   vsqrtsh\t{%d1, %0|%0, %d1}
+   vsqrtsh\t{%1, %d0|%d0, %1}"
+  [(set_attr "type" "sse")
+   (set_attr "prefix" "evex")
+   (set_attr "avx_partial_xmm_update" "false,true")
+   (set_attr "mode" "HF")])
+
 (define_insn "*sqrt<mode>2_sse"
   [(set (match_operand:MODEF 0 "register_operand" "=v,v,v")
        (sqrt:MODEF
index b08a9d3..e8aef0d 100644 (file)
    (set_attr "mode" "<ssescalarmode>")])
 
 (define_insn "*<sse>_vmsqrt<mode>2<mask_scalar_name><round_scalar_name>"
-  [(set (match_operand:VF_128 0 "register_operand" "=x,v")
-       (vec_merge:VF_128
-         (vec_duplicate:VF_128
+  [(set (match_operand:VFH_128 0 "register_operand" "=x,v")
+       (vec_merge:VFH_128
+         (vec_duplicate:VFH_128
            (sqrt:<ssescalarmode>
              (match_operand:<ssescalarmode> 1 "nonimmediate_operand" "xm,<round_scalar_constraint>")))
-         (match_operand:VF_128 2 "register_operand" "0,v")
+         (match_operand:VFH_128 2 "register_operand" "0,v")
          (const_int 1)))]
   "TARGET_SSE"
   "@
diff --git a/gcc/testsuite/gcc.target/i386/avx512fp16-builtin-sqrt-1.c b/gcc/testsuite/gcc.target/i386/avx512fp16-builtin-sqrt-1.c
new file mode 100644 (file)
index 0000000..b4efc84
--- /dev/null
@@ -0,0 +1,18 @@
+/* { dg-do compile } */
+/* { dg-options "-Ofast -mavx512fp16 -mprefer-vector-width=512" } */
+
+_Float16
+f1 (_Float16 x)
+{
+  return __builtin_sqrtf16 (x);
+}
+
+void
+f2 (_Float16* __restrict psrc, _Float16* __restrict pdst)
+{
+  for (int i = 0; i != 32; i++)
+    pdst[i] = __builtin_sqrtf16 (psrc[i]);
+}
+
+/* { dg-final { scan-assembler-times "vsqrtsh\[^\n\r\]*xmm\[0-9\]" 1 } } */
+/* { dg-final { scan-assembler-times "vsqrtph\[^\n\r\]*zmm\[0-9\]" 1 } } */
diff --git a/gcc/testsuite/gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c b/gcc/testsuite/gcc.target/i386/avx512fp16vl-builtin-sqrt-1.c
new file mode 100644 (file)
index 0000000..08deb3e
--- /dev/null
@@ -0,0 +1,19 @@
+/* { dg-do compile } */
+/* { dg-options "-Ofast -mavx512fp16 -mavx512vl" } */
+
+void
+f1 (_Float16* __restrict psrc, _Float16* __restrict pdst)
+{
+  for (int i = 0; i != 8; i++)
+    pdst[i] = __builtin_sqrtf16 (psrc[i]);
+}
+
+void
+f2 (_Float16* __restrict psrc, _Float16* __restrict pdst)
+{
+  for (int i = 0; i != 16; i++)
+    pdst[i] = __builtin_sqrtf16 (psrc[i]);
+}
+
+/* { dg-final { scan-assembler-times "vsqrtph\[^\n\r\]*xmm\[0-9\]" 1 } } */
+/* { dg-final { scan-assembler-times "vsqrtph\[^\n\r\]*ymm\[0-9\]" 1 } } */