[gcc r17-3477] aarch64: Use SVE SPLICE for scalable recurrences

Kyrylo Tkachov via Gcc-cvs <[email protected]>
Newsgroups gmane.comp.gcc.cvs
Message-ID <[email protected]>
https://gcc.gnu.org/g:70ebcf89f8534fe410163415f18c527b7b694b0b

commit r17-3477-g70ebcf89f8534fe410163415f18c527b7b694b0b
Author: Kyrylo Tkachov <[email protected]>
Date:   Thu Aug 13 10:13:02 2026 +0200

    aarch64: Use SVE SPLICE for scalable recurrences
    
    The vectorizer represents a first-order recurrence as a scalable
    permutation that takes the final D elements from one vector and the
    remaining elements from the next vector.  The AArch64 permutation
    expander did not recognize this variable-length shape.
    
    Recognize the exact two-input permutation and build a predicate for D
    elements.  Reverse that predicate and use the SVE SPLICE pattern.  Reject
    fixed-length vectors, rotations, and distances that cannot be represented
    at the minimum vector length.
    
    The compile test checks seven direct SPLICE sequences across byte,
    halfword, word, and doubleword modes.  A 67-element runtime test covers
    distances 1 and 2 with two seed values, and partial final vectors.
    
    On an example testcase:
    
            typedef __UINT32_TYPE__ uint32_t;
    
            void
            recur (const uint32_t *__restrict a, uint32_t *__restrict b,
            const uint32_t *__restrict init)
            {
              uint32_t prev = *init;
              for (int i = 0; i < 67; ++i)
                {
                  b[i] = a[i] - prev;
                  prev = a[i];
                }
            }
    
    With -O3 -march=armv8.2-a+sve
    We emitted before:
    
            ldr     w4, [x2]
            mov     w3, 0
            .p2align 5,,15
            .L2:
            mov     w2, w4
            ldr     w4, [x0, x3]
            sub     w2, w4, w2
            str     w2, [x1, x3]
            add     x3, x3, 4
            cmp     x3, 268
            bne     .L2
    
    And after this patch:
            mov     w3, 0
            cntw    x5
            ptrue   p5.b, all
            mov     w4, 67
            ptrue   p6.b, vl1
            mov     p7.b, p5.b
            ld1rw   z31.s, p5/z, [x2]
            rev     p6.s, p6.s
            .p2align 5,,15
            .L2:
            mov     z30.d, z31.d
            ld1w    z31.s, p7/z, [x0, x3, lsl 2]
            splice  z30.s, p6, z30.s, z31.s
            sub     z30.s, z31.s, z30.s
            st1w    z30.s, p7, [x1, x3, lsl 2]
            add     x3, x3, x5
            whilelo p7.s, w3, w4
            b.any   .L2
    
    This also vectorizes some loops in OpenEXR 4.0.0.
    
    Upstream PXR24, RLE, ZIP, and ZIPS tests pass, and the baseline and
    patched encoders produced identical files.  A paired exrmetrics run on an
    NVIDIA Vera system measured the following write-time improvements.
    
                                    Flowers.exr   multipart.0001.exr
            PXR24, float                   2.28%                 4.58%
            PXR24, original                3.22%                 8.07%
            RLE                           13.52%                23.27%
            ZIP                            3.67%                 9.59%
            ZIPS                           3.15%                 5.62%
    
    Bootstrapped and tested on aarch64-none-linux-gnu.
    Ok for trunk?
    Thanks,
    Kyrill
    
    gcc/ChangeLog:
    
            * config/aarch64/aarch64.cc (aarch64_evpc_splice): New function.
            (aarch64_expand_vec_perm_const_1): Use it.
    
    gcc/testsuite/ChangeLog:
    
            * gcc.target/aarch64/sve/perm_splice_1.c: New test.
            * gcc.target/aarch64/sve/perm_splice_1-run.c: Likewise.
    
    Signed-off-by: Kyrylo Tkachov <[email protected]>

Diff:
---
 gcc/config/aarch64/aarch64.cc                      | 42 ++++++++++
 .../gcc.target/aarch64/sve/perm_splice_1-run.c     | 94 ++++++++++++++++++++++
 .../gcc.target/aarch64/sve/perm_splice_1.c         | 63 +++++++++++++++
 3 files changed, 199 insertions(+)

diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
index 8a788a5c7e41..f36864a10da3 100644
--- a/gcc/config/aarch64/aarch64.cc
+++ b/gcc/config/aarch64/aarch64.cc
@@ -28261,6 +28261,46 @@ aarch64_evpc_ext (struct expand_vec_perm_d *d)
   return true;
 }
 
+/* Return true if D describes a scalable-vector permutation that takes the
+   last DIST elements from the first input and the remaining elements from
+   the second input.  */
+
+static bool
+aarch64_evpc_splice (struct expand_vec_perm_d *d)
+{
+  poly_int64 nelt = d->perm.length ();
+  HOST_WIDE_INT dist;
+
+  if (d->vec_flags != VEC_SVE_DATA
+      || d->one_vector_p
+      || nelt.is_constant ()
+      || !(nelt - d->perm[0]).is_constant (&dist)
+      || !IN_RANGE (dist, 1, INT_MAX)
+      || !d->perm.series_p (0, 1, nelt - dist, 1))
+    return false;
+
+  machine_mode pred_mode = aarch64_sve_pred_mode (d->vmode);
+  if (aarch64_svpattern_for_vl (pred_mode, dist)
+      == AARCH64_NUM_SVPATTERNS)
+    return false;
+
+  if (d->testing_p)
+    return true;
+
+  rtx_vector_builder builder (pred_mode, dist, 2);
+  for (HOST_WIDE_INT i = 0; i < dist; ++i)
+    builder.quick_push (CONST1_RTX (BImode));
+  for (HOST_WIDE_INT i = 0; i < dist; ++i)
+    builder.quick_push (CONST0_RTX (BImode));
+
+  rtx head = force_reg (pred_mode, builder.build ());
+  rtx pred = gen_reg_rtx (pred_mode);
+  emit_insn (gen_aarch64_sve_rev (pred_mode, pred, head));
+  emit_insn (gen_aarch64_sve_splice (d->vmode, d->target, pred,
+				     d->op0, d->op1));
+  return true;
+}
+
 /* Recognize patterns for the REV{64,32,16} insns, which reverse elements
    within each 64-bit, 32-bit or 16-bit granule.  */
 
@@ -28787,6 +28827,8 @@ aarch64_expand_vec_perm_const_1 (struct expand_vec_perm_d *d)
 	    return true;
 	  else if (aarch64_evpc_rev_global (d))
 	    return true;
+	  else if (aarch64_evpc_splice (d))
+	    return true;
 	  else if (aarch64_evpc_ext (d))
 	    return true;
 	  else if (aarch64_evpc_dup (d))
diff --git a/gcc/testsuite/gcc.target/aarch64/sve/perm_splice_1-run.c b/gcc/testsuite/gcc.target/aarch64/sve/perm_splice_1-run.c
new file mode 100644
index 000000000000..b2c13120c539
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/sve/perm_splice_1-run.c
@@ -0,0 +1,94 @@
+/* { dg-do run } */
+/* { dg-require-effective-target aarch64_sve_hw } */
+/* { dg-options "-O3 -march=armv8.2-a+sve -mautovec-preference=sve-only" } */
+/* { dg-additional-options "-msve-vector-bits=scalable" } */
+/* { dg-additional-options "-fvect-cost-model=unlimited" } */
+
+typedef __UINT8_TYPE__ uint8_t;
+typedef __UINT16_TYPE__ uint16_t;
+typedef __UINT32_TYPE__ uint32_t;
+typedef __UINT64_TYPE__ uint64_t;
+
+#define N 67
+
+#define DEFINE_RECUR(TYPE, SUFFIX) \
+  __attribute__((noipa)) \
+  static void \
+  recur1_##SUFFIX (TYPE *__restrict a, TYPE *__restrict b, \
+		   const TYPE *__restrict init) \
+  { \
+    TYPE prev = init[0]; \
+    for (int i = 0; i < N; ++i) \
+      { \
+	b[i] = a[i] - prev; \
+	prev = a[i]; \
+      } \
+  } \
+  \
+  __attribute__((noipa)) \
+  static void \
+  recur2_##SUFFIX (TYPE *__restrict a, TYPE *__restrict b, \
+		   const TYPE *__restrict init) \
+  { \
+    TYPE prev0 = init[0]; \
+    TYPE prev1 = init[0]; \
+    for (int i = 0; i + 1 < N; i += 2) \
+      { \
+	b[i] = a[i] - prev0; \
+	prev0 = a[i]; \
+	b[i + 1] = a[i + 1] - prev1; \
+	prev1 = a[i + 1]; \
+      } \
+    b[N - 1] = 0; \
+  }
+
+DEFINE_RECUR (uint8_t, u8)
+DEFINE_RECUR (uint16_t, u16)
+DEFINE_RECUR (uint32_t, u32)
+DEFINE_RECUR (uint64_t, u64)
+
+#define CHECK_RECUR(TYPE, SUFFIX) \
+  do \
+    { \
+      TYPE a[N]; \
+      TYPE b[N]; \
+      const TYPE init[2] = { (TYPE) 19, (TYPE) 43 }; \
+      \
+      for (int i = 0; i < N; ++i) \
+	a[i] = (TYPE) ((unsigned int) i * (unsigned int) i \
+			  + 5U * (unsigned int) i + 11U); \
+      \
+      for (int seed = 0; seed < 2; ++seed) \
+	{ \
+	  recur1_##SUFFIX (a, b, &init[seed]); \
+	  if (b[0] != (TYPE) (a[0] - init[seed])) \
+	    __builtin_abort (); \
+	  _Pragma ("GCC novector") \
+	  for (int i = 1; i < N; ++i) \
+	    if (b[i] != (TYPE) (a[i] - a[i - 1])) \
+	      __builtin_abort (); \
+	  \
+	  recur2_##SUFFIX (a, b, &init[seed]); \
+	  if (b[0] != (TYPE) (a[0] - init[seed]) \
+	      || b[1] != (TYPE) (a[1] - init[seed])) \
+	    __builtin_abort (); \
+	  _Pragma ("GCC novector") \
+	  for (int i = 2; i < N - 1; ++i) \
+	    if (b[i] != (TYPE) (a[i] - a[i - 2])) \
+	      __builtin_abort (); \
+	  if (b[N - 1] != 0) \
+	    __builtin_abort (); \
+	} \
+    } \
+  while (0)
+
+int
+main (void)
+{
+  CHECK_RECUR (uint8_t, u8);
+  CHECK_RECUR (uint16_t, u16);
+  CHECK_RECUR (uint32_t, u32);
+  CHECK_RECUR (uint64_t, u64);
+
+  return 0;
+}
diff --git a/gcc/testsuite/gcc.target/aarch64/sve/perm_splice_1.c b/gcc/testsuite/gcc.target/aarch64/sve/perm_splice_1.c
new file mode 100644
index 000000000000..c2a9879008b7
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/sve/perm_splice_1.c
@@ -0,0 +1,63 @@
+/* { dg-do compile } */
+/* { dg-options "-O3 -march=armv8.2-a+sve -mautovec-preference=sve-only" } */
+/* { dg-additional-options "-msve-vector-bits=scalable" } */
+/* { dg-additional-options "-fvect-cost-model=unlimited -fdump-tree-vect-details" } */
+
+typedef __UINT8_TYPE__ uint8_t;
+typedef __UINT16_TYPE__ uint16_t;
+typedef __UINT32_TYPE__ uint32_t;
+typedef __UINT64_TYPE__ uint64_t;
+
+#define DEF_RECUR1(TYPE, SUFFIX)                                     \
+  __attribute__((noipa))                                             \
+  void                                                               \
+  recur1_##SUFFIX (TYPE *__restrict a, TYPE *__restrict b,           \
+		   const TYPE *__restrict init)                         \
+  {                                                                  \
+    TYPE prev = *init;                                               \
+    for (int i = 0; i < 67; ++i)                                    \
+      {                                                              \
+	b[i] = a[i] - prev;                                           \
+	prev = a[i];                                                  \
+      }                                                              \
+  }
+
+#define DEF_RECUR2(TYPE, SUFFIX)                                     \
+  __attribute__((noipa))                                             \
+  void                                                               \
+  recur2_##SUFFIX (TYPE *__restrict a, TYPE *__restrict b,           \
+		   const TYPE *__restrict init)                         \
+  {                                                                  \
+    TYPE prev0 = *init;                                              \
+    TYPE prev1 = *init;                                              \
+    for (int i = 0; i < 66; i += 2)                                 \
+      {                                                              \
+	b[i] = a[i] - prev0;                                          \
+	prev0 = a[i];                                                 \
+	b[i + 1] = a[i + 1] - prev1;                                 \
+	prev1 = a[i + 1];                                             \
+      }                                                              \
+  }
+
+DEF_RECUR1 (uint8_t, u8)
+DEF_RECUR1 (uint16_t, u16)
+DEF_RECUR1 (uint32_t, u32)
+DEF_RECUR1 (uint64_t, u64)
+
+DEF_RECUR2 (uint8_t, u8)
+DEF_RECUR2 (uint16_t, u16)
+DEF_RECUR2 (uint32_t, u32)
+
+/* { dg-final { scan-assembler-times {\tptrue\tp[0-9]+\.b, vl1} 4 } } */
+/* { dg-final { scan-assembler-times {\tptrue\tp[0-9]+\.b, vl2} 1 } } */
+/* { dg-final { scan-assembler-times {\tptrue\tp[0-9]+\.h, vl2} 1 } } */
+/* { dg-final { scan-assembler-times {\tptrue\tp[0-9]+\.s, vl2} 1 } } */
+/* { dg-final { scan-assembler-times {\trev\tp[0-9]+\.b, p[0-9]+\.b} 2 } } */
+/* { dg-final { scan-assembler-times {\trev\tp[0-9]+\.h, p[0-9]+\.h} 2 } } */
+/* { dg-final { scan-assembler-times {\trev\tp[0-9]+\.s, p[0-9]+\.s} 2 } } */
+/* { dg-final { scan-assembler-times {\trev\tp[0-9]+\.d, p[0-9]+\.d} 1 } } */
+/* { dg-final { scan-assembler-times {\tsplice\tz[0-9]+\.b, p[0-9]+, z[0-9]+\.b, z[0-9]+\.b} 2 } } */
+/* { dg-final { scan-assembler-times {\tsplice\tz[0-9]+\.h, p[0-9]+, z[0-9]+\.h, z[0-9]+\.h} 2 } } */
+/* { dg-final { scan-assembler-times {\tsplice\tz[0-9]+\.s, p[0-9]+, z[0-9]+\.s, z[0-9]+\.s} 2 } } */
+/* { dg-final { scan-assembler-times {\tsplice\tz[0-9]+\.d, p[0-9]+, z[0-9]+\.d, z[0-9]+\.d} 1 } } */
+/* { dg-final { scan-tree-dump-times "vectorized 1 loops in function" 7 "vect" } } */
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.