On Tue, Aug 18, 2026 at 6:30 AM <[email protected]> wrote:
>
> From: Kyrylo Tkachov <[email protected]>
>
> RBIT has a .8B and .16B arrangement and CLZ has a .8B and .16B arrangement,
> so a byte-element count-trailing-zeros is two instructions.  ctz<mode>2 only
> covered V2SI and V4SI, so a loop over unsigned char was expanded by the
> middle end into the generic negate/and/clz/subtract sequence:
>
>   before                                after
>
>   mvni  v31.4s, 0                       ldr     q31, [x1, x3]
>   movi  v30.16b, 8                      rbit    v31.16b, v31.16b
> .L4:                                    clz     v31.16b, v31.16b
>   ldr   q29, [x1, x3]                   str     q31, [x0, x3]
>   add   v28.16b, v29.16b, v31.16b
>   bic   v29.16b, v28.16b, v29.16b
>   clz   v29.16b, v29.16b
>   sub   v29.16b, v30.16b, v29.16b
>   str   v29.16b, [x0, x3]
>
> The halfword modes are left alone.  Bit-reversing a halfword needs REV16 as
> well as RBIT, so the sequence would be three instructions against the four
> of the generic one, which is not enough of a difference to be worth the
> extra pattern.
>
> Bootstrapped and tested on aarch64-none-linux-gnu.
> Ok for trunk?
> Thanks,
> Kyrill
>
> gcc/ChangeLog:
>
>         * config/aarch64/aarch64-simd.md (ctz<mode>2): New expander for VB.
>
> gcc/testsuite/ChangeLog:
>
>         * gcc.target/aarch64/vect-ctz-1.c: New test.
>         * gcc.target/aarch64/vect-ctz-2.c: New test.
>
> Signed-off-by: Kyrylo Tkachov <[email protected]>
> ---
>  gcc/config/aarch64/aarch64-simd.md            | 11 ++++++
>  gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c | 35 +++++++++++++++++++
>  gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c | 32 +++++++++++++++++
>  3 files changed, 78 insertions(+)
>  create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
>  create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
>
> diff --git a/gcc/config/aarch64/aarch64-simd.md 
> b/gcc/config/aarch64/aarch64-simd.md
> index 527efe94084..1fd990be50f 100644
> --- a/gcc/config/aarch64/aarch64-simd.md
> +++ b/gcc/config/aarch64/aarch64-simd.md
> @@ -517,6 +517,17 @@
>    [(set_attr "type" "neon_rbit")]
>  )
>
> +(define_expand "ctz<mode>2"
> +  [(set (match_operand:VB 0 "register_operand")
> +       (ctz:VB (match_operand:VB 1 "register_operand")))]
> +  "TARGET_SIMD"
> +  {
> +     emit_insn (gen_aarch64_rbit<mode> (operands[0], operands[1]));
> +     emit_insn (gen_clz<mode>2 (operands[0], operands[0]));
> +     DONE;
> +  }
> +)
> +
>  (define_expand "ctz<mode>2"
>    [(set (match_operand:VS 0 "register_operand")
>          (ctz:VS (match_operand:VS 1 "register_operand")))]
> diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c 
> b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
> new file mode 100644
> index 00000000000..70d89d3dfa5
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
> @@ -0,0 +1,35 @@
> +/* { dg-do compile } */
> +/* { dg-options "-O3 -fno-schedule-insns -fno-schedule-insns2" } */
> +/* { dg-final { check-function-bodies "**" "" } } */
> +
> +typedef __UINT8_TYPE__ u8;
> +
> +/* The OR keeps the input nonzero so that the loop is just the count.  */
> +/*
> +** ctzb:
> +**     ...
> +**     movi    v[0-9]+\.16b, 0xffffffffffffff80
> +**     ldr     q[0-9]+, \[x[0-9]+\]
> +**     orr     v[0-9]+\.16b, v[0-9]+\.16b, v[0-9]+\.16b
> +**     rbit    v[0-9]+\.16b, v[0-9]+\.16b
> +**     clz     v[0-9]+\.16b, v[0-9]+\.16b
> +**     str     q[0-9]+, \[x[0-9]+\]

My only suggestion is to write this testcase so that the register
numbers for outputs are captured and then inputs then use that a
captured one.
Otherwise ok.


> +**     ret
> +*/
> +void
> +ctzb (u8 *__restrict d, u8 *__restrict a)
> +{
> +  for (int i = 0; i < 16; i++)
> +    d[i] = __builtin_ctzg ((u8) (a[i] | 0x80));
> +}
> +
> +/* The same count in a variable-length loop must use RBIT and CLZ too.  */
> +void
> +ctzb_n (u8 *__restrict d, u8 *__restrict a, int n)
> +{
> +  for (int i = 0; i < n; i++)
> +    d[i] = __builtin_ctzg (a[i], 8);
> +}
> +
> +/* { dg-final { scan-assembler-times {\trbit\tv[0-9]+\.16b, v[0-9]+\.16b} 2 
> } } */
> +/* { dg-final { scan-assembler-times {\tclz\tv[0-9]+\.16b, v[0-9]+\.16b} 2 } 
> } */
> diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c 
> b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
> new file mode 100644
> index 00000000000..a5be943f955
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
> @@ -0,0 +1,32 @@
> +/* { dg-do run } */
> +/* { dg-options "-O3" } */
> +
> +#define N 61
> +static unsigned char a[N], d[N], e[N];
> +
> +__attribute__((noipa)) void
> +ctzb (unsigned char *restrict r, unsigned char *restrict x, int n)
> +{
> +  for (int i = 0; i < n; i++)
> +    r[i] = __builtin_ctzg (x[i], 8);
> +}
> +
> +__attribute__((noipa, optimize ("O0"))) void
> +ctzb_ref (unsigned char *restrict r, unsigned char *restrict x, int n)
> +{
> +  for (int i = 0; i < n; i++)
> +    r[i] = __builtin_ctzg (x[i], 8);
> +}
> +
> +int
> +main (void)
> +{
> +  for (int i = 0; i < N; i++)
> +    a[i] = (unsigned char) (i * 37 + (i & 7));
> +  ctzb (d, a, N);
> +  ctzb_ref (e, a, N);
> +  for (int i = 0; i < N; i++)
> +    if (d[i] != e[i])
> +      __builtin_abort ();
> +  return 0;
> +}
> --
> 2.50.1 (Apple Git-155)
>

Reply via email to