[gcc r17-2648] i386: Add vector lrint expanders for conversions
hongtao Liu via Gcc-cvs <[email protected]>
| Newsgroups | gmane.comp.gcc.cvs |
|---|---|
| Message-ID | <[email protected]> |
https://gcc.gnu.org/g:e870be66c6adb856b2ee8869801f8c4542b3a914 commit r17-2648-ge870be66c6adb856b2ee8869801f8c4542b3a914 Author: liuhongt <[email protected]> Date: Wed Jul 8 20:28:00 2026 -0700 i386: Add vector lrint expanders for conversions Expose existing vector conversion instructions through lrint optabs so the vectorizer can use vcvtps2qq and vcvtpd2dq instead of scalar conversion sequences. gcc/ChangeLog: PR target/126160 * config/i386/sse.md (lrint<ssePSmode2lower><mode>2): New expander. (lrintv2sfv2di2): Likewise. (ssedfsimode): New mode attribute. (ssedfsimodelower): Likewise. (lrint<mode><ssedfsimodelower>2): New expander. gcc/testsuite/ChangeLog: PR target/126160 * gcc.target/i386/pr126160.c: New test. * gcc.target/i386/pr126160-2.c: New test. Diff: --- gcc/config/i386/sse.md | 39 ++++++++++++++++++++++++++++++ gcc/testsuite/gcc.target/i386/pr126160-2.c | 37 ++++++++++++++++++++++++++++ gcc/testsuite/gcc.target/i386/pr126160.c | 24 ++++++++++++++++++ 3 files changed, 100 insertions(+) diff --git a/gcc/config/i386/sse.md b/gcc/config/i386/sse.md index 122dfb367646..7521034e128a 100644 --- a/gcc/config/i386/sse.md +++ b/gcc/config/i386/sse.md @@ -9205,6 +9205,13 @@ (set_attr "prefix" "evex") (set_attr "mode" "<sseinsnmode>")]) +;; vcvtps2qq widens V4SF/V8SF directly to V4DI/V8DI (float -> long). +(define_expand "lrint<ssePSmode2lower><mode>2" + [(set (match_operand:VI8_256_512 0 "register_operand") + (unspec:VI8_256_512 [(match_operand:<ssePSmode2> 1 "nonimmediate_operand")] + UNSPEC_FIX_NOTRUNC))] + "TARGET_AVX512DQ") + (define_insn "<mask_codefor>avx512dq_cvtps2qq<mode><mask_name><round_name>" [(set (match_operand:VI8_256_512 0 "register_operand" "=v") (unspec:VI8_256_512 [(match_operand:<ssePSmode2> 1 "nonimmediate_operand" "<round_constraint>")] @@ -9215,6 +9222,24 @@ (set_attr "prefix" "evex") (set_attr "mode" "<sseinsnmode>")]) +;; vcvtps2qq xmm converts the low 2 elements of V4SF to V2DI (float -> long). +(define_expand "lrintv2sfv2di2" + [(match_operand:V2DI 0 "register_operand") + (match_operand:V2SF 1 "nonimmediate_operand")] + "TARGET_AVX512DQ && TARGET_AVX512VL" +{ + rtx op1 = gen_reg_rtx (V4SFmode); + emit_insn (gen_movq_v2sf_to_sse (op1, operands[1])); + + rtx sel = gen_rtx_PARALLEL (VOIDmode, + gen_rtvec (2, const0_rtx, const1_rtx)); + rtx src = gen_rtx_VEC_SELECT (V2SFmode, op1, sel); + emit_insn (gen_rtx_SET (operands[0], + gen_rtx_UNSPEC (V2DImode, gen_rtvec (1, src), + UNSPEC_FIX_NOTRUNC))); + DONE; +}) + (define_insn "<mask_codefor>avx512dq_cvtps2qqv2di<mask_name>" [(set (match_operand:V2DI 0 "register_operand" "=v") (unspec:V2DI @@ -10032,6 +10057,20 @@ (set_attr "prefix" "maybe_vex") (set_attr "mode" "V2DF")]) +;; Mapping of a DFmode vector to the SImode vector of the same length. +(define_mode_attr ssedfsimode + [(V8DF "V8SI") (V4DF "V4SI") (V2DF "V2SI")]) +(define_mode_attr ssedfsimodelower + [(V8DF "v8si") (V4DF "v4si") (V2DF "v2si")]) + +;; vcvtpd2dq narrows V2DF/V4DF/V8DF to V2SI/V4SI/V8SI (double -> int). +(define_expand "lrint<mode><ssedfsimodelower>2" + [(set (match_operand:<ssedfsimode> 0 "register_operand") + (unspec:<ssedfsimode> + [(match_operand:VF2 1 "register_operand")] + UNSPEC_FIX_NOTRUNC))] + "TARGET_SSE2") + (define_insn "avx512f_cvtpd2dq512<mask_name><round_name>" [(set (match_operand:V8SI 0 "register_operand" "=v") (unspec:V8SI diff --git a/gcc/testsuite/gcc.target/i386/pr126160-2.c b/gcc/testsuite/gcc.target/i386/pr126160-2.c new file mode 100644 index 000000000000..0fd5506ced28 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr126160-2.c @@ -0,0 +1,37 @@ +/* PR target/126160 */ +/* { dg-do compile } */ +/* { dg-options "-Ofast -mavx512f -mavx512vl -mavx512dq -mprefer-vector-width=512" } */ +/* { dg-final { scan-assembler-times "vcvtpd2dq" 3 } } */ +/* { dg-final { scan-assembler-not "vcvtsd2si" } } */ + +/* __builtin_irint (double -> int) should vectorize into vcvtpd2dq rather + than scalar vcvtsd2sil: V2DF -> V2SI, V4DF -> V4SI and V8DF -> V8SI. */ + +void +f2 (int *a, double *b) +{ + a[0] = __builtin_irint (b[0]); + a[1] = __builtin_irint (b[1]); +} + +void +f4 (int *a, double *b) +{ + a[0] = __builtin_irint (b[0]); + a[1] = __builtin_irint (b[1]); + a[2] = __builtin_irint (b[2]); + a[3] = __builtin_irint (b[3]); +} + +void +f8 (int *a, double *b) +{ + a[0] = __builtin_irint (b[0]); + a[1] = __builtin_irint (b[1]); + a[2] = __builtin_irint (b[2]); + a[3] = __builtin_irint (b[3]); + a[4] = __builtin_irint (b[4]); + a[5] = __builtin_irint (b[5]); + a[6] = __builtin_irint (b[6]); + a[7] = __builtin_irint (b[7]); +} diff --git a/gcc/testsuite/gcc.target/i386/pr126160.c b/gcc/testsuite/gcc.target/i386/pr126160.c new file mode 100644 index 000000000000..8c44727c1fc0 --- /dev/null +++ b/gcc/testsuite/gcc.target/i386/pr126160.c @@ -0,0 +1,24 @@ +/* PR target/126160 */ +/* { dg-do compile { target { ! ia32 } } } */ +/* { dg-options "-Ofast -mavx512dq -mavx512vl" } */ +/* { dg-final { scan-assembler-times "vcvtps2qq" 2 } } */ +/* { dg-final { scan-assembler-not "vcvtss2si" } } */ + +/* The float -> long conversions should vectorize into vcvtps2qq rather + than scalar vcvtss2siq: V4SF -> V4DI for foo, V2SF -> V2DI for bar. */ + +void +foo (long *a, float *b) +{ + a[0] = __builtin_lrint (b[0]); + a[1] = __builtin_lrint (b[1]); + a[2] = __builtin_lrint (b[2]); + a[3] = __builtin_lrint (b[3]); +} + +void +bar (long *a, float *b) +{ + a[0] = __builtin_lrint (b[0]); + a[1] = __builtin_lrint (b[1]); +}