From: Kyrylo Tkachov <[email protected]>
FEAT_FP16 provides half-precision forms of FCCMP and FCCMPE. The
conditional-compare patterns and expansion hooks currently accept only
SFmode and DFmode comparisons. For example, with -O2 -fno-trapping-math
-march=armv8.2-a+fp16:
int
f (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
{
return (a == b) && (c == d);
}
AArch64 emitted:
fcmp h0, h1
cset w1, eq
fcmp h2, h3
cset w0, eq
and w0, w1, w0
After this patch it emits:
fcmp h0, h1
fccmp h2, h3, 0, eq
cset w0, eq
Use GPF_F16 for both conditional-compare patterns. Use the stype attribute
so HFmode keeps the single-precision scheduling type. Teach both expansion
hooks to select HFmode, and reject it when FEAT_FP16 is not available.
The tests also cover FCCMPE, the reverse pattern used by OR expressions, and
+nofp16.
Bootstrapped and tested on aarch64-none-linux-gnu.
Ok for trunk?
Thanks,
Kyrill
gcc/
* config/aarch64/aarch64.cc (aarch64_gen_ccmp_first): Handle HFmode.
(aarch64_gen_ccmp_next): Likewise. Reject unsupported HFmode early.
* config/aarch64/aarch64.md
(@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>): Use GPF_F16 and stype.
(@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>_rev): Likewise.
gcc/testsuite/
* gcc.target/aarch64/fccmp_1.c: New test.
* gcc.target/aarch64/fccmp_2.c: Likewise.
Signed-off-by: Kyrylo Tkachov <[email protected]>
---
gcc/config/aarch64/aarch64.cc | 19 ++++++---
gcc/config/aarch64/aarch64.md | 16 ++++----
gcc/testsuite/gcc.target/aarch64/fccmp_1.c | 47 ++++++++++++++++++++++
gcc/testsuite/gcc.target/aarch64/fccmp_2.c | 12 ++++++
4 files changed, 81 insertions(+), 13 deletions(-)
create mode 100644 gcc/testsuite/gcc.target/aarch64/fccmp_1.c
create mode 100644 gcc/testsuite/gcc.target/aarch64/fccmp_2.c
diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
index d19ca305d82..8e1eb2d7e33 100644
--- a/gcc/config/aarch64/aarch64.cc
+++ b/gcc/config/aarch64/aarch64.cc
@@ -29621,6 +29621,17 @@ aarch64_gen_ccmp_first (rtx_insn **prep_seq, rtx_insn **gen_seq,
icode = CODE_FOR_cmpdi;
break;
+ case E_HFmode:
+ if (!TARGET_FP_F16INST)
+ {
+ end_sequence ();
+ return NULL_RTX;
+ }
+ cmp_mode = HFmode;
+ cc_mode = aarch64_select_cc_mode (code, op0, op1);
+ icode = cc_mode == CCFPEmode ? CODE_FOR_fcmpehf : CODE_FOR_fcmphf;
+ break;
+
case E_SFmode:
cmp_mode = SFmode;
cc_mode = aarch64_select_cc_mode (code, op0, op1);
@@ -29677,6 +29688,7 @@ aarch64_gen_ccmp_next (rtx_insn **prep_seq, rtx_insn **gen_seq, rtx prev,
/* Exit early for modes that are ot handled to avoid O(n^2) part of expand_operands. */
op_mode = TYPE_MODE (TREE_TYPE (treeop0));
if (!(op_mode == QImode || op_mode == HImode || op_mode == SImode || op_mode == DImode
+ || (op_mode == HFmode && TARGET_FP_F16INST)
|| op_mode == SFmode || op_mode == DFmode))
return NULL_RTX;
@@ -29702,13 +29714,10 @@ aarch64_gen_ccmp_next (rtx_insn **prep_seq, rtx_insn **gen_seq, rtx prev,
cmp_mode = DImode;
break;
+ case E_HFmode:
case E_SFmode:
- cmp_mode = SFmode;
- cc_mode = aarch64_select_cc_mode (cmp_code, op0, op1);
- break;
-
case E_DFmode:
- cmp_mode = DFmode;
+ cmp_mode = op_mode;
cc_mode = aarch64_select_cc_mode (cmp_code, op0, op1);
break;
diff --git a/gcc/config/aarch64/aarch64.md b/gcc/config/aarch64/aarch64.md
index b2185c63819..9da5de8429f 100644
--- a/gcc/config/aarch64/aarch64.md
+++ b/gcc/config/aarch64/aarch64.md
@@ -1134,21 +1134,21 @@
}
)
-(define_insn "@ccmp<CCFP_CCFPE:mode><GPF:mode>"
+(define_insn "@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>"
[(set (match_operand:CCFP_CCFPE 1 "cc_register" "")
(if_then_else:CCFP_CCFPE
(match_operator 4 "aarch64_comparison_operator"
[(match_operand 0 "cc_register" "")
(const_int 0)])
(compare:CCFP_CCFPE
- (match_operand:GPF 2 "register_operand" "w")
- (match_operand:GPF 3 "register_operand" "w"))
+ (match_operand:GPF_F16 2 "register_operand" "w")
+ (match_operand:GPF_F16 3 "register_operand" "w"))
(unspec:CCFP_CCFPE
[(match_operand 5 "immediate_operand")]
UNSPEC_NZCV)))]
"TARGET_FLOAT"
"fccmp<e>\\t%<s>2, %<s>3, %k5, %m4"
- [(set_attr "type" "fccmp<s>")]
+ [(set_attr "type" "fccmp<stype>")]
)
(define_insn "@ccmp<CC_ONLY:mode><GPI:mode>_rev"
@@ -1171,7 +1171,7 @@
}
)
-(define_insn "@ccmp<CCFP_CCFPE:mode><GPF:mode>_rev"
+(define_insn "@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>_rev"
[(set (match_operand:CCFP_CCFPE 1 "cc_register" "")
(if_then_else:CCFP_CCFPE
(match_operator 4 "aarch64_comparison_operator"
@@ -1181,11 +1181,11 @@
[(match_operand 5 "immediate_operand")]
UNSPEC_NZCV)
(compare:CCFP_CCFPE
- (match_operand:GPF 2 "register_operand" "w")
- (match_operand:GPF 3 "register_operand" "w"))))]
+ (match_operand:GPF_F16 2 "register_operand" "w")
+ (match_operand:GPF_F16 3 "register_operand" "w"))))]
"TARGET_FLOAT"
"fccmp<e>\\t%<s>2, %<s>3, %k5, %M4"
- [(set_attr "type" "fccmp<s>")]
+ [(set_attr "type" "fccmp<stype>")]
)
;; Expansion of signed mod by a power of 2 using CSNEG.
diff --git a/gcc/testsuite/gcc.target/aarch64/fccmp_1.c b/gcc/testsuite/gcc.target/aarch64/fccmp_1.c
new file mode 100644
index 00000000000..96d6f717136
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/fccmp_1.c
@@ -0,0 +1,47 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -fno-trapping-math -march=armv8.2-a+fp16" } */
+/* { dg-final { check-function-bodies "**" "" } } */
+
+/* FCCMP and FCCMPE have half precision forms when FEAT_FP16 is available.
+ Trapping behaviour is disabled so each conditional comparison is safe. */
+
+/*
+** hf_lt:
+** fcmpe h0, h1
+** fccmpe h2, h3, 0, mi
+** cset w0, mi
+** ret
+*/
+int
+hf_lt (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
+{
+ return (a < b) && (c < d);
+}
+
+/*
+** hf_eq:
+** fcmp h0, h1
+** fccmp h2, h3, 0, eq
+** cset w0, eq
+** ret
+*/
+int
+hf_eq (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
+{
+ return (a == b) && (c == d);
+}
+
+/* Exercise the reverse conditional-compare pattern. */
+
+/*
+** hf_ior:
+** fcmpe h0, h1
+** fccmpe h2, h3, 8, pl
+** cset w0, mi
+** ret
+*/
+int
+hf_ior (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
+{
+ return (a < b) || (c < d);
+}
diff --git a/gcc/testsuite/gcc.target/aarch64/fccmp_2.c b/gcc/testsuite/gcc.target/aarch64/fccmp_2.c
new file mode 100644
index 00000000000..c54a83b2eef
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/fccmp_2.c
@@ -0,0 +1,12 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -march=armv8.2-a+nofp16" } */
+
+int
+hf_and (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
+{
+ return (a == b) && (c == d);
+}
+
+/* FEAT_FP16 is required for half precision compare instructions. */
+/* { dg-final { scan-assembler-not {\tfcmpe?\th} } } */
+/* { dg-final { scan-assembler-not {\tfccmpe?\th} } } */
--
2.50.1 (Apple Git-155)
lmpx.com only provides a reader for public news (NNTP) servers. It is not
affiliated with the servers or forums shown here and is not responsible for
the content of articles, which is written by their respective authors.