Hi, ping for this patch:

https://gcc.gnu.org/pipermail/gcc-patches/2026-August/726330.html

Thanks,
huzife


On Tue, Aug 4, 2026 at 8:45 PM huzife <[email protected]> wrote:
>
> The existing SWAR sign-mask simplification handles a shifted and masked
> value multiplied by an all-ones value.  The equivalent (x << N) - x form
> is not recognized.
>
> Canonicalize the shift-and-subtract form to multiplication so that it can
> reuse the existing simplification.  Require unsigned vectors, exact shift
> and mask constants, valid vector geometry, and target support for the
> resulting comparison and conditional operation.
>
> gcc/ChangeLog:
>
>         * match.pd: Canonicalize shift-and-subtract SWAR sign masks to
>         multiplication.
>
> gcc/testsuite/ChangeLog:
>
>         * gcc.target/aarch64/swar_to_vec_cmp_2.c: New test.
>         * gcc.target/aarch64/swar_to_vec_cmp_3.c: New test.
>
> Signed-off-by: huzife <[email protected]>
> ---
>  gcc/match.pd                                  | 60 +++++++++++++++++
>  .../gcc.target/aarch64/swar_to_vec_cmp_2.c    | 65 +++++++++++++++++++
>  .../gcc.target/aarch64/swar_to_vec_cmp_3.c    | 60 +++++++++++++++++
>  3 files changed, 185 insertions(+)
>  create mode 100644 gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_2.c
>  create mode 100644 gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_3.c
>
> diff --git a/gcc/match.pd b/gcc/match.pd
> index 293650760e2..087f7636867 100644
> --- a/gcc/match.pd
> +++ b/gcc/match.pd
> @@ -408,6 +408,66 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT)
>      (view_convert (bit_and:itype (view_convert @0)
>                                  (ne @1 { build_zero_cst (type); })))))))
>
> +/* Canonicalize the shift-and-subtract form of a SWAR sign mask to the
> +   equivalent multiplication form below.  Check the same shape and target
> +   requirements here, so that this cannot introduce an unsupported vector
> +   multiply.  */
> +#if GIMPLE
> +(simplify
> + (minus
> +  (lshift (bit_and@4
> +            (rshift @0 uniform_integer_cst_p@1)
> +            uniform_integer_cst_p@2)
> +          uniform_integer_cst_p@3)
> +  @4)
> + (with {
> +   tree rshift_cst = uniform_integer_cst_p (@1);
> +   tree bit_and_cst = uniform_integer_cst_p (@2);
> +   tree lshift_cst = uniform_integer_cst_p (@3);
> +  }
> +  (if (VECTOR_TYPE_P (type)
> +       && TYPE_UNSIGNED (TREE_TYPE (type))
> +       && tree_fits_uhwi_p (rshift_cst)
> +       && tree_fits_uhwi_p (bit_and_cst)
> +       && tree_fits_uhwi_p (lshift_cst)
> +       && tree_to_uhwi (rshift_cst) < HOST_BITS_PER_WIDE_INT - 1)
> +   (with {
> +     HOST_WIDE_INT vec_elem_bits = vector_element_bits (type);
> +     poly_int64 vec_nelts = TYPE_VECTOR_SUBPARTS (type);
> +     poly_int64 vec_bits = vec_elem_bits * vec_nelts;
> +     unsigned HOST_WIDE_INT cmp_bits_i, bit_and_i, target_bit_and_i;
> +     cmp_bits_i = tree_to_uhwi (rshift_cst) + 1;
> +     bit_and_i = tree_to_uhwi (bit_and_cst);
> +     target_bit_and_i = 0;
> +
> +     for (unsigned i = 0; i < vec_elem_bits / cmp_bits_i; i++)
> +       target_bit_and_i = (target_bit_and_i << cmp_bits_i) | 1U;
> +    }
> +    (if (exact_log2 (cmp_bits_i) >= 0
> +        && cmp_bits_i < HOST_BITS_PER_WIDE_INT
> +        && cmp_bits_i < (unsigned HOST_WIDE_INT) vec_elem_bits
> +        && tree_to_uhwi (lshift_cst) == cmp_bits_i
> +        && multiple_p (vec_bits, cmp_bits_i)
> +        && vec_elem_bits <= HOST_BITS_PER_WIDE_INT
> +        && target_bit_and_i == bit_and_i)
> +     (with {
> +       tree cmp_type = build_nonstandard_integer_type (cmp_bits_i, 0);
> +       poly_int64 vector_type_nelts = exact_div (vec_bits, cmp_bits_i);
> +       tree vec_cmp_type = build_vector_type (cmp_type, vector_type_nelts);
> +       tree vec_truth_type = truth_type_for (vec_cmp_type);
> +      }
> +      (if (expand_vec_cmp_expr_p (vec_cmp_type, vec_truth_type, LT_EXPR)
> +          && expand_vec_cond_expr_p (vec_cmp_type, vec_truth_type))
> +       (with {
> +        wide_int multiplier = wi::mask (cmp_bits_i, false,
> +                                       element_precision (type));
> +        tree multiplier_cst
> +          = wide_int_to_tree (TREE_TYPE (type), multiplier);
> +        multiplier_cst = build_uniform_cst (type, multiplier_cst);
> +       }
> +       (mult @4 { multiplier_cst; })))))))))
> +#endif
> +
>  /* In SWAR (SIMD within a register) code a signed comparison of packed data
>     can be constructed with a particular combination of shift, bitwise and,
>     and multiplication by constants.  If that code is vectorized we can
> diff --git a/gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_2.c 
> b/gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_2.c
> new file mode 100644
> index 00000000000..4722d3c1661
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_2.c
> @@ -0,0 +1,65 @@
> +/* { dg-do compile } */
> +/* { dg-options "-O2" } */
> +
> +typedef unsigned short v4hi __attribute__ ((vector_size (8)));
> +typedef unsigned short v8hi __attribute__ ((vector_size (16)));
> +typedef unsigned int v2si __attribute__ ((vector_size (8)));
> +typedef unsigned int v4si __attribute__ ((vector_size (16)));
> +typedef unsigned long long v2di __attribute__ ((vector_size (16)));
> +
> +v4hi
> +packed_cmp_v4hi (v4hi a)
> +{
> +  const v4hi right = { 7, 7, 7, 7 };
> +  const v4hi left = { 8, 8, 8, 8 };
> +  const v4hi mask = { 0x101, 0x101, 0x101, 0x101 };
> +  v4hi x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +v8hi
> +packed_cmp_v8hi (v8hi a)
> +{
> +  const v8hi right = { 7, 7, 7, 7, 7, 7, 7, 7 };
> +  const v8hi left = { 8, 8, 8, 8, 8, 8, 8, 8 };
> +  const v8hi mask = { 0x101, 0x101, 0x101, 0x101,
> +                     0x101, 0x101, 0x101, 0x101 };
> +  v8hi x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +v2si
> +packed_cmp_v2si (v2si a)
> +{
> +  const v2si right = { 15, 15 };
> +  const v2si left = { 16, 16 };
> +  const v2si mask = { 0x10001, 0x10001 };
> +  v2si x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +v4si
> +packed_cmp_v4si (v4si a)
> +{
> +  const v4si right = { 15, 15, 15, 15 };
> +  const v4si left = { 16, 16, 16, 16 };
> +  const v4si mask = { 0x10001, 0x10001, 0x10001, 0x10001 };
> +  v4si x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +v2di
> +packed_cmp_v2di (v2di a)
> +{
> +  const v2di right = { 31, 31 };
> +  const v2di left = { 32, 32 };
> +  const v2di mask = { 0x100000001ULL, 0x100000001ULL };
> +  v2di x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +/* { dg-final { scan-assembler-times {\tcmlt\tv[0-9]+\.8b,} 1 } } */
> +/* { dg-final { scan-assembler-times {\tcmlt\tv[0-9]+\.16b,} 1 } } */
> +/* { dg-final { scan-assembler-times {\tcmlt\tv[0-9]+\.4h,} 1 } } */
> +/* { dg-final { scan-assembler-times {\tcmlt\tv[0-9]+\.8h,} 1 } } */
> +/* { dg-final { scan-assembler-times {\tcmlt\tv[0-9]+\.4s,} 1 } } */
> diff --git a/gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_3.c 
> b/gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_3.c
> new file mode 100644
> index 00000000000..861ed3e58c5
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/swar_to_vec_cmp_3.c
> @@ -0,0 +1,60 @@
> +/* { dg-do compile } */
> +/* { dg-options "-O2" } */
> +
> +typedef unsigned long long v2di __attribute__ ((vector_size (16)));
> +typedef long long signed_v2di __attribute__ ((vector_size (16)));
> +
> +v2di
> +wrong_mask (v2di a)
> +{
> +  const v2di right = { 31, 31 };
> +  const v2di left = { 32, 32 };
> +  const v2di mask = { 0x100000000ULL, 0x100000000ULL };
> +  v2di x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +v2di
> +wrong_right_shift (v2di a)
> +{
> +  const v2di right = { 30, 30 };
> +  const v2di left = { 32, 32 };
> +  const v2di mask = { 0x100000001ULL, 0x100000001ULL };
> +  v2di x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +v2di
> +wrong_left_shift (v2di a)
> +{
> +  const v2di right = { 31, 31 };
> +  const v2di left = { 31, 31 };
> +  const v2di mask = { 0x100000001ULL, 0x100000001ULL };
> +  v2di x = (a >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +v2di
> +different_operands (v2di a, v2di b)
> +{
> +  const v2di right = { 31, 31 };
> +  const v2di left = { 32, 32 };
> +  const v2di mask = { 0x100000001ULL, 0x100000001ULL };
> +  v2di x = (a >> right) & mask;
> +  v2di y = (b >> right) & mask;
> +  return (x << left) - y;
> +}
> +
> +signed_v2di
> +signed_type (signed_v2di a)
> +{
> +  const signed_v2di right = { 31, 31 };
> +  const signed_v2di left = { 32, 32 };
> +  const signed_v2di input_mask = { 0x7fffffffffffffffLL,
> +                                   0x7fffffffffffffffLL };
> +  const signed_v2di mask = { 0x100000001LL, 0x100000001LL };
> +  signed_v2di x = ((a & input_mask) >> right) & mask;
> +  return (x << left) - x;
> +}
> +
> +/* { dg-final { scan-assembler-not {\tcmlt\t} } } */
> --
> 2.43.0
>

Reply via email to