[gcc r17-3472] aarch64: implement ctz2 for the Advanced SIMD byte and halfword modes
Kyrylo Tkachov via Gcc-cvs <[email protected]>
| Newsgroups | gmane.comp.gcc.cvs |
|---|---|
| Message-ID | <[email protected]> |
https://gcc.gnu.org/g:48425ec4ef2ad3697f07a3075ff52d2d9dea91f4 commit r17-3472-g48425ec4ef2ad3697f07a3075ff52d2d9dea91f4 Author: Kyrylo Tkachov <[email protected]> Date: Mon Aug 3 09:10:24 2026 -0700 aarch64: implement ctz2 for the Advanced SIMD byte and halfword modes Reversing the bits of an element turns its trailing zeros into leading ones, so a count of trailing zeros is a bit reversal followed by a CLZ. ctz<mode>2 only covered V2SI and V4SI, so the byte and halfword loops were expanded by the middle end into the generic negate/and/clz/subtract sequence, which needs two vector constants as well as four instructions. RBIT reverses the bits within each byte, so a byte element needs nothing else and a halfword element needs REV16 to put its two bytes in the opposite order. That is the same shape the V2SI and V4SI expander already had, so fold all of them into one expander over VDQ_BHSI and give bitreverse<mode>2 the wider modes it builds on. For a halfword loop the inner loop changes from ldr q0, [x1] orr v0.8h, #128, lsl #8 add v30.8h, v0.8h, v30.8h bic v30.16b, v30.16b, v0.16b clz v30.8h, v30.8h sub v31.8h, v31.8h, v30.8h str q31, [x0] to ldr q31, [x1] orr v31.8h, #128, lsl #8 rev16 v31.16b, v31.16b rbit v31.16b, v31.16b clz v31.8h, v31.8h str q31, [x0] with the two constants no longer needed. A byte loop loses the REV16 as well and counts in two instructions. The 64-bit elements have no CLZ to pair with, so they keep the generic expansion through popcount, which is what LLVM emits for them too. A fixed-length loop keeps this Advanced SIMD form even when SVE is available, while a variable-length one is vectorised with SVE and counts there, so the two cases get a test each over a common source. Bootstrapped and tested on aarch64-none-linux-gnu. Ok for trunk? Thanks, Kyrill gcc/ChangeLog: * config/aarch64/aarch64-simd.md (bitreverse<mode>2): New expander for VDQHS. (ctz<mode>2): Replace the VB and VS expanders with one for VDQ_BHSI. gcc/testsuite/ChangeLog: * gcc.target/aarch64/vect-ctz.h: New file. * gcc.target/aarch64/vect-ctz-1.c: New test. * gcc.target/aarch64/vect-ctz-2.c: New test. * gcc.target/aarch64/vect-ctz-3.c: New test. Signed-off-by: Kyrylo Tkachov <[email protected]> Diff: --- gcc/config/aarch64/aarch64-simd.md | 29 ++++++++--- gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c | 39 +++++++++++++++ gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c | 40 +++++++++++++++ gcc/testsuite/gcc.target/aarch64/vect-ctz-3.c | 72 +++++++++++++++++++++++++++ gcc/testsuite/gcc.target/aarch64/vect-ctz.h | 34 +++++++++++++ 5 files changed, 208 insertions(+), 6 deletions(-) diff --git a/gcc/config/aarch64/aarch64-simd.md b/gcc/config/aarch64/aarch64-simd.md index 66369a5075a6..e91692ce4867 100644 --- a/gcc/config/aarch64/aarch64-simd.md +++ b/gcc/config/aarch64/aarch64-simd.md @@ -509,6 +509,23 @@ "TARGET_SIMD" "") +;; RBIT reverses the bits within each byte, so a wider element needs its bytes +;; put in the opposite order as well. The 64-bit elements are left out because +;; nothing reaches them: they have no CLZ to pair with in ctz<mode>2, and the +;; vectorizer does not yet handle IFN_BITREVERSE. +(define_expand "bitreverse<mode>2" + [(set (match_operand:VDQHS 0 "register_operand") + (bitreverse:VDQHS (match_operand:VDQHS 1 "register_operand")))] + "TARGET_SIMD" + { + emit_insn (gen_bswap<mode>2 (operands[0], operands[1])); + machine_mode qimode = <bitsize> == 64 ? V8QImode : V16QImode; + rtx bytes = force_subreg (qimode, operands[0], <MODE>mode, 0); + emit_insn (gen_aarch64_rbit (qimode, bytes, bytes)); + DONE; + } +) + (define_insn "@aarch64_rbit<mode><vczle><vczbe>" [(set (match_operand:VB 0 "register_operand" "=w") (bitreverse:VB (match_operand:VB 1 "register_operand" "w")))] @@ -517,15 +534,15 @@ [(set_attr "type" "neon_rbit")] ) +;; Reversing the bits of an element turns its trailing zeros into leading ones, +;; so counting them is a bit reversal followed by a CLZ. The 64-bit elements +;; have no CLZ and are left to the generic expansion. (define_expand "ctz<mode>2" - [(set (match_operand:VS 0 "register_operand") - (ctz:VS (match_operand:VS 1 "register_operand")))] + [(set (match_operand:VDQ_BHSI 0 "register_operand") + (ctz:VDQ_BHSI (match_operand:VDQ_BHSI 1 "register_operand")))] "TARGET_SIMD" { - emit_insn (gen_bswap<mode>2 (operands[0], operands[1])); - rtx op0_castsi2qi = force_subreg (<VS:VSI2QI>mode, operands[0], - <MODE>mode, 0); - emit_insn (gen_aarch64_rbit<VS:vsi2qi> (op0_castsi2qi, op0_castsi2qi)); + emit_insn (gen_bitreverse<mode>2 (operands[0], operands[1])); emit_insn (gen_clz<mode>2 (operands[0], operands[0])); DONE; } diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c new file mode 100644 index 000000000000..fce38c626690 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c @@ -0,0 +1,39 @@ +/* { dg-do compile } */ +/* { dg-options "-O3 -march=armv8-a -fno-schedule-insns -fno-schedule-insns2" } */ +/* { dg-final { check-function-bodies "**" "" } } */ + +#include "vect-ctz.h" + +/* Without SVE the 0x80 needs a vector constant of its own. */ +/* +** ctzb: +** ... +** movi v([0-9]+)\.16b, 0xffffffffffffff80 +** ldr q([0-9]+), \[x[0-9]+\] +** orr v([0-9]+)\.16b, v\2\.16b, v\1\.16b +** rbit v([0-9]+)\.16b, v\3\.16b +** clz v([0-9]+)\.16b, v\4\.16b +** str q\5, \[x[0-9]+\] +** ret +*/ + +/* A halfword element needs REV16 as well as RBIT. The 0x8000 fits the + immediate form of ORR here. */ +/* +** ctzh: +** ... +** ldr q([0-9]+), \[x[0-9]+\] +** orr v\1\.8h, #128, lsl #8 +** rev16 v([0-9]+)\.16b, v\1\.16b +** rbit v([0-9]+)\.16b, v\2\.16b +** clz v([0-9]+)\.8h, v\3\.8h +** str q\4, \[x[0-9]+\] +** ret +*/ + +/* Both the fixed-length and the variable-length loop count this way, for each + of the two element sizes. */ +/* { dg-final { scan-assembler-times {\trbit\tv[0-9]+\.16b, v[0-9]+\.16b} 4 } } */ +/* { dg-final { scan-assembler-times {\tclz\tv[0-9]+\.16b, v[0-9]+\.16b} 2 } } */ +/* { dg-final { scan-assembler-times {\trev16\tv[0-9]+\.16b, v[0-9]+\.16b} 2 } } */ +/* { dg-final { scan-assembler-times {\tclz\tv[0-9]+\.8h, v[0-9]+\.8h} 2 } } */ diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c new file mode 100644 index 000000000000..69b048f75209 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c @@ -0,0 +1,40 @@ +/* { dg-do compile } */ +/* { dg-options "-O3 -march=armv8.2-a+sve -fno-schedule-insns -fno-schedule-insns2" } */ +/* { dg-final { check-function-bodies "**" "" } } */ + +#include "vect-ctz.h" + +/* With SVE the 0x80 folds into an immediate ORR, but the fixed-length loop + still counts with the Advanced SIMD RBIT and CLZ. */ +/* +** ctzb: +** ... +** ldr q([0-9]+), \[x[0-9]+\] +** orr z([0-9]+)\.b, z\1\.b, -128 +** rbit v([0-9]+)\.16b, v\2\.16b +** clz v([0-9]+)\.16b, v\3\.16b +** str q\4, \[x[0-9]+\] +** ret +*/ + +/* +** ctzh: +** ... +** ldr q([0-9]+), \[x[0-9]+\] +** orr v\1\.8h, #128, lsl #8 +** rev16 v([0-9]+)\.16b, v\1\.16b +** rbit v([0-9]+)\.16b, v\2\.16b +** clz v([0-9]+)\.8h, v\3\.8h +** str q\4, \[x[0-9]+\] +** ret +*/ + +/* The variable-length loops are vectorised with SVE and count there instead. */ +/* { dg-final { scan-assembler-times {\trbit\tv[0-9]+\.16b, v[0-9]+\.16b} 2 } } */ +/* { dg-final { scan-assembler-times {\tclz\tv[0-9]+\.16b, v[0-9]+\.16b} 1 } } */ +/* { dg-final { scan-assembler-times {\trev16\tv[0-9]+\.16b, v[0-9]+\.16b} 1 } } */ +/* { dg-final { scan-assembler-times {\tclz\tv[0-9]+\.8h, v[0-9]+\.8h} 1 } } */ +/* { dg-final { scan-assembler-times {\trbit\tz[0-9]+\.b, p[0-9]+/m, z[0-9]+\.b} 1 } } */ +/* { dg-final { scan-assembler-times {\tclz\tz[0-9]+\.b, p[0-9]+/m, z[0-9]+\.b} 1 } } */ +/* { dg-final { scan-assembler-times {\trbit\tz[0-9]+\.h, p[0-9]+/m, z[0-9]+\.h} 1 } } */ +/* { dg-final { scan-assembler-times {\tclz\tz[0-9]+\.h, p[0-9]+/m, z[0-9]+\.h} 1 } } */ diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-3.c b/gcc/testsuite/gcc.target/aarch64/vect-ctz-3.c new file mode 100644 index 000000000000..1df6a3581551 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-3.c @@ -0,0 +1,72 @@ +/* { dg-do run } */ +/* { dg-options "-O3" } */ + +#include "vect-ctz.h" + +#define N 61 +static u8 a[N], d[N], e[N]; +static u16 ah[N], dh[N], eh[N]; + +__attribute__((noipa, optimize ("O0"))) void +ctzb_ref (u8 *__restrict d, u8 *__restrict a) +{ + for (int i = 0; i < 16; i++) + d[i] = __builtin_ctzg ((u8) (a[i] | 0x80)); +} + +__attribute__((noipa, optimize ("O0"))) void +ctzb_n_ref (u8 *__restrict d, u8 *__restrict a, int n) +{ + for (int i = 0; i < n; i++) + d[i] = __builtin_ctzg (a[i], 8); +} + +__attribute__((noipa, optimize ("O0"))) void +ctzh_ref (u16 *__restrict d, u16 *__restrict a) +{ + for (int i = 0; i < 8; i++) + d[i] = __builtin_ctzg ((u16) (a[i] | 0x8000)); +} + +__attribute__((noipa, optimize ("O0"))) void +ctzh_n_ref (u16 *__restrict d, u16 *__restrict a, int n) +{ + for (int i = 0; i < n; i++) + d[i] = __builtin_ctzg (a[i], 16); +} + +int +main (void) +{ + for (int i = 0; i < N; i++) + { + a[i] = (u8) (i * 37 + (i & 7)); + ah[i] = (u16) (i * 9973 + (i & 15)); + } + + ctzb (d, a); + ctzb_ref (e, a); + for (int i = 0; i < 16; i++) + if (d[i] != e[i]) + __builtin_abort (); + + ctzb_n (d, a, N); + ctzb_n_ref (e, a, N); + for (int i = 0; i < N; i++) + if (d[i] != e[i]) + __builtin_abort (); + + ctzh (dh, ah); + ctzh_ref (eh, ah); + for (int i = 0; i < 8; i++) + if (dh[i] != eh[i]) + __builtin_abort (); + + ctzh_n (dh, ah, N); + ctzh_n_ref (eh, ah, N); + for (int i = 0; i < N; i++) + if (dh[i] != eh[i]) + __builtin_abort (); + + return 0; +} diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz.h b/gcc/testsuite/gcc.target/aarch64/vect-ctz.h new file mode 100644 index 000000000000..04f1a8452a44 --- /dev/null +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz.h @@ -0,0 +1,34 @@ +/* Shared source for the Advanced SIMD and the SVE ctz code-quality tests + and for the execution test. */ + +typedef __UINT8_TYPE__ u8; +typedef __UINT16_TYPE__ u16; + +/* The OR keeps the input nonzero so that the loop is just the count. */ +__attribute__((noipa)) void +ctzb (u8 *__restrict d, u8 *__restrict a) +{ + for (int i = 0; i < 16; i++) + d[i] = __builtin_ctzg ((u8) (a[i] | 0x80)); +} + +__attribute__((noipa)) void +ctzb_n (u8 *__restrict d, u8 *__restrict a, int n) +{ + for (int i = 0; i < n; i++) + d[i] = __builtin_ctzg (a[i], 8); +} + +__attribute__((noipa)) void +ctzh (u16 *__restrict d, u16 *__restrict a) +{ + for (int i = 0; i < 8; i++) + d[i] = __builtin_ctzg ((u16) (a[i] | 0x8000)); +} + +__attribute__((noipa)) void +ctzh_n (u16 *__restrict d, u16 *__restrict a, int n) +{ + for (int i = 0; i < n; i++) + d[i] = __builtin_ctzg (a[i], 16); +}