From: Kyrylo Tkachov <[email protected]>
vect_recog_sat_trunc_pattern only forms .SAT_TRUNC when the target has the
sstrunc or ustrunc optab for the mode pair, and the backend had neither, even
though aarch64_<su>qmovn<mode> already matches exactly the RTL those optabs
describe. A saturating narrowing loop was therefore vectorised as a pair of
clamps and a permute:
static inline unsigned char clip (unsigned short x)
{ return x & ~255 ? 255 : x; }
for (i) r[i] = clip (x[i]);
before after
movi v29.8h, 0xff ldp q30, q31, [x1], 32
ldp q31, q30, [x1], 32 uqxtn v30.8b, v30.8h
umin v31.8h, v31.8h, v29.8h uqxtn v31.8b, v31.8h
umin v30.8h, v30.8h, v29.8h stp d30, d31, [x0], 16
uzp1 v31.16b, v31.16b, v30.16b
str q31, [x0], 16
and the signed form was worse still, needing two constants and a pair of
compares per half. Only the standard names were missing, so this adds the
expander and leaves the existing insn to match it.
Bootstrapped and tested on aarch64-none-linux-gnu.
Ok for trunk?
Thanks,
Kyrill
gcc/ChangeLog:
* config/aarch64/iterators.md (vnarrowq): New mode attribute.
(sat_trunc_op): New code attribute.
* config/aarch64/aarch64-simd.md (<sat_trunc_op>trunc<mode><vnarrowq>2):
New expander.
gcc/testsuite/ChangeLog:
* gcc.target/aarch64/vect-sat-trunc-1.c: New test.
* gcc.target/aarch64/vect-sat-trunc-2.c: New test.
Signed-off-by: Kyrylo Tkachov <[email protected]>
---
gcc/config/aarch64/aarch64-simd.md | 8 ++
gcc/config/aarch64/iterators.md | 6 ++
.../gcc.target/aarch64/vect-sat-trunc-1.c | 73 +++++++++++++++++++
.../gcc.target/aarch64/vect-sat-trunc-2.c | 57 +++++++++++++++
4 files changed, 144 insertions(+)
create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c
create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c
diff --git a/gcc/config/aarch64/aarch64-simd.md b/gcc/config/aarch64/aarch64-simd.md
index 12861333b22..527efe94084 100644
--- a/gcc/config/aarch64/aarch64-simd.md
+++ b/gcc/config/aarch64/aarch64-simd.md
@@ -6122,6 +6122,14 @@
[(set_attr "type" "neon_qadd<q>")]
)
+;; The saturating narrowing conversion that the vectoriser asks for through
+;; IFN_SAT_TRUNC. The RTL is what aarch64_<su>qmovn<mode> below matches.
+(define_expand "<sat_trunc_op>trunc<mode><vnarrowq>2"
+ [(set (match_operand:<VNARROWQ> 0 "register_operand")
+ (SAT_TRUNC:<VNARROWQ> (match_operand:VQN 1 "register_operand")))]
+ "TARGET_SIMD"
+)
+
;; sqmovn and uqmovn
(define_insn "aarch64_<su>qmovn<mode><vczle><vczbe>"
diff --git a/gcc/config/aarch64/iterators.md b/gcc/config/aarch64/iterators.md
index 12e65c0f230..7dbc19b7e05 100644
--- a/gcc/config/aarch64/iterators.md
+++ b/gcc/config/aarch64/iterators.md
@@ -2134,6 +2134,9 @@
(DI "v2si")])
;; Narrowed double-modes for VQN (Used for XTN).
+;; Lower case VNARROWQ, for the sstrunc and ustrunc optab names.
+(define_mode_attr vnarrowq [(V8HI "v8qi") (V4SI "v4hi") (V2DI "v2si")])
+
(define_mode_attr VNARROWQ [(V8HI "V8QI") (V4SI "V4HI")
(V2DI "V2SI")
(DI "SI") (SI "HI")
@@ -3407,6 +3410,9 @@
(define_code_attr TRUNC_SHIFT [(ss_truncate "ashiftrt")
(us_truncate "lshiftrt") (truncate "lshiftrt")])
+;; The optab prefix of a saturating truncation.
+(define_code_attr sat_trunc_op [(ss_truncate "ss") (us_truncate "us")])
+
(define_code_attr shrn_op [(ss_truncate "sq")
(us_truncate "uq") (truncate "")])
diff --git a/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c
new file mode 100644
index 00000000000..b3128971356
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-1.c
@@ -0,0 +1,73 @@
+/* { dg-do compile } */
+/* { dg-options "-O3 -march=armv8-a" } */
+/* { dg-additional-options "-fno-schedule-insns -fno-schedule-insns2" } */
+/* { dg-final { check-function-bodies "**" "" } } */
+
+typedef __UINT16_TYPE__ u16;
+typedef __UINT8_TYPE__ u8;
+typedef __INT16_TYPE__ i16;
+typedef __INT8_TYPE__ i8;
+
+static inline u8
+clip_u8 (u16 x)
+{
+ return x & (u16) ~(u16) 255 ? (u8) 255 : (u8) x;
+}
+
+/*
+** clu:
+** ...
+** ldp q[0-9]+, q[0-9]+, \[x[0-9]+\]
+** uqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** uqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** stp d[0-9]+, d[0-9]+, \[x[0-9]+\]
+** ret
+*/
+void
+clu (u8 *__restrict r, u16 *__restrict x)
+{
+ for (int i = 0; i < 16; i++)
+ r[i] = clip_u8 (x[i]);
+}
+
+static inline i8
+clip_i8 (i16 x)
+{
+ i8 t = (i8) x;
+ return (i16) -128 <= x && x <= (i16) 127 ? t : x < 0 ? -128 : 127;
+}
+
+/*
+** cls:
+** ...
+** ldp q[0-9]+, q[0-9]+, \[x[0-9]+\]
+** sqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** sqxtn v[0-9]+\.8b, v[0-9]+\.8h
+** stp d[0-9]+, d[0-9]+, \[x[0-9]+\]
+** ret
+*/
+void
+cls (i8 *__restrict r, i16 *__restrict x)
+{
+ for (int i = 0; i < 16; i++)
+ r[i] = clip_i8 (x[i]);
+}
+
+/* The same in variable-length loops, and at the other two element widths. */
+void
+clu_n (u8 *__restrict r, u16 *__restrict x, int n)
+{
+ for (int i = 0; i < n; i++)
+ r[i] = clip_u8 (x[i]);
+}
+
+void
+cls_n (i8 *__restrict r, i16 *__restrict x, int n)
+{
+ for (int i = 0; i < n; i++)
+ r[i] = clip_i8 (x[i]);
+}
+
+/* { dg-final { scan-assembler-times {\tuqxtn\tv[0-9]+\.8b, v[0-9]+\.8h} 3 } } */
+/* { dg-final { scan-assembler-times {\tsqxtn\tv[0-9]+\.8b, v[0-9]+\.8h} 3 } } */
+/* { dg-final { scan-assembler-not {\tuzp1\t} } } */
diff --git a/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c
new file mode 100644
index 00000000000..76d5d6e632f
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/vect-sat-trunc-2.c
@@ -0,0 +1,57 @@
+/* { dg-do run } */
+/* { dg-options "-O3" } */
+
+typedef __UINT16_TYPE__ u16; typedef __UINT8_TYPE__ u8;
+typedef __UINT32_TYPE__ u32; typedef __UINT64_TYPE__ u64;
+typedef __INT16_TYPE__ i16; typedef __INT8_TYPE__ i8;
+typedef __INT32_TYPE__ i32; typedef __INT64_TYPE__ i64;
+#define N 137
+
+#define DEFU(name, WT, NT, NMAX) \
+ static inline NT clip_##name (WT x) \
+ { return x & (WT) ~(WT) NMAX ? (NT) NMAX : (NT) x; } \
+ __attribute__((noipa)) void name (NT *__restrict r, WT *__restrict x, int n) \
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); } \
+ __attribute__((noipa, optimize ("O0"))) \
+ void name##_ref (NT *__restrict r, WT *__restrict x, int n) \
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); }
+
+#define DEFS(name, WT, NT, NMIN, NMAX) \
+ static inline NT clip_##name (WT x) \
+ { NT t = (NT) x; \
+ return (WT) NMIN <= x && x <= (WT) NMAX ? t : x < 0 ? NMIN : NMAX; } \
+ __attribute__((noipa)) void name (NT *__restrict r, WT *__restrict x, int n) \
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); } \
+ __attribute__((noipa, optimize ("O0"))) \
+ void name##_ref (NT *__restrict r, WT *__restrict x, int n) \
+ { for (int i = 0; i < n; i++) r[i] = clip_##name (x[i]); }
+
+DEFU (u16to8, u16, u8, 255)
+DEFU (u32to16, u32, u16, 65535)
+DEFU (u64to32, u64, u32, 0xffffffffu)
+DEFS (i16to8, i16, i8, -128, 127)
+DEFS (i32to16, i32, i16, -32768, 32767)
+DEFS (i64to32, i64, i32, (i32) 0x80000000, 0x7fffffff)
+
+static u16 a16[N]; static u32 a32[N]; static u64 a64[N];
+static u8 d8[N], e8[N]; static u16 d16[N], e16[N]; static u32 d32[N], e32[N];
+static unsigned long seed = 7;
+static unsigned rnd (void) { seed = seed * 6364136223846793005UL + 1; return (unsigned)(seed >> 33); }
+#define CHK(d, e, n) for (int i = 0; i < n; i++) if (d[i] != e[i]) __builtin_abort ();
+int main (void)
+{
+ for (int r = 0; r < 200; r++)
+ {
+ for (int i = 0; i < N; i++)
+ { unsigned v = rnd ();
+ a16[i] = (u16) v; a32[i] = v; a64[i] = ((u64) v << 32) | rnd ();
+ if ((i & 7) == 0) { a16[i] = 0xffff; a32[i] = 0xffffffffu; a64[i] = ~0UL; } }
+ u16to8 (d8, a16, N); u16to8_ref (e8, a16, N); CHK (d8, e8, N)
+ u32to16 (d16, a32, N); u32to16_ref (e16, a32, N); CHK (d16, e16, N)
+ u64to32 (d32, a64, N); u64to32_ref (e32, a64, N); CHK (d32, e32, N)
+ i16to8 ((i8 *) d8, (i16 *) a16, N); i16to8_ref ((i8 *) e8, (i16 *) a16, N); CHK (d8, e8, N)
+ i32to16 ((i16 *) d16, (i32 *) a32, N); i32to16_ref ((i16 *) e16, (i32 *) a32, N); CHK (d16, e16, N)
+ i64to32 ((i32 *) d32, (i64 *) a64, N); i64to32_ref ((i32 *) e32, (i64 *) a64, N); CHK (d32, e32, N)
+ }
+ return 0;
+}
--
2.50.1 (Apple Git-155)
lmpx.com only provides a reader for public news (NNTP) servers. It is not
affiliated with the servers or forums shown here and is not responsible for
the content of articles, which is written by their respective authors.