On Tue, Aug 18, 2026 at 6:30 AM <[email protected]> wrote:
>
> From: Kyrylo Tkachov <[email protected]>
>
> RBIT has a .8B and .16B arrangement and CLZ has a .8B and .16B arrangement,
> so a byte-element count-trailing-zeros is two instructions. ctz<mode>2 only
> covered V2SI and V4SI, so a loop over unsigned char was expanded by the
> middle end into the generic negate/and/clz/subtract sequence:
>
> before after
>
> mvni v31.4s, 0 ldr q31, [x1, x3]
> movi v30.16b, 8 rbit v31.16b, v31.16b
> .L4: clz v31.16b, v31.16b
> ldr q29, [x1, x3] str q31, [x0, x3]
> add v28.16b, v29.16b, v31.16b
> bic v29.16b, v28.16b, v29.16b
> clz v29.16b, v29.16b
> sub v29.16b, v30.16b, v29.16b
> str v29.16b, [x0, x3]
>
> The halfword modes are left alone. Bit-reversing a halfword needs REV16 as
> well as RBIT, so the sequence would be three instructions against the four
> of the generic one, which is not enough of a difference to be worth the
> extra pattern.
>
> Bootstrapped and tested on aarch64-none-linux-gnu.
> Ok for trunk?
> Thanks,
> Kyrill
>
> gcc/ChangeLog:
>
> * config/aarch64/aarch64-simd.md (ctz<mode>2): New expander for VB.
>
> gcc/testsuite/ChangeLog:
>
> * gcc.target/aarch64/vect-ctz-1.c: New test.
> * gcc.target/aarch64/vect-ctz-2.c: New test.
>
> Signed-off-by: Kyrylo Tkachov <[email protected]>
> ---
> gcc/config/aarch64/aarch64-simd.md | 11 ++++++
> gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c | 35 +++++++++++++++++++
> gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c | 32 +++++++++++++++++
> 3 files changed, 78 insertions(+)
> create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
> create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
>
> diff --git a/gcc/config/aarch64/aarch64-simd.md
> b/gcc/config/aarch64/aarch64-simd.md
> index 527efe94084..1fd990be50f 100644
> --- a/gcc/config/aarch64/aarch64-simd.md
> +++ b/gcc/config/aarch64/aarch64-simd.md
> @@ -517,6 +517,17 @@
> [(set_attr "type" "neon_rbit")]
> )
>
> +(define_expand "ctz<mode>2"
> + [(set (match_operand:VB 0 "register_operand")
> + (ctz:VB (match_operand:VB 1 "register_operand")))]
> + "TARGET_SIMD"
> + {
> + emit_insn (gen_aarch64_rbit<mode> (operands[0], operands[1]));
> + emit_insn (gen_clz<mode>2 (operands[0], operands[0]));
> + DONE;
> + }
> +)
> +
> (define_expand "ctz<mode>2"
> [(set (match_operand:VS 0 "register_operand")
> (ctz:VS (match_operand:VS 1 "register_operand")))]
> diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
> b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
> new file mode 100644
> index 00000000000..70d89d3dfa5
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
> @@ -0,0 +1,35 @@
> +/* { dg-do compile } */
> +/* { dg-options "-O3 -fno-schedule-insns -fno-schedule-insns2" } */
> +/* { dg-final { check-function-bodies "**" "" } } */
> +
> +typedef __UINT8_TYPE__ u8;
> +
> +/* The OR keeps the input nonzero so that the loop is just the count. */
> +/*
> +** ctzb:
> +** ...
> +** movi v[0-9]+\.16b, 0xffffffffffffff80
> +** ldr q[0-9]+, \[x[0-9]+\]
> +** orr v[0-9]+\.16b, v[0-9]+\.16b, v[0-9]+\.16b
> +** rbit v[0-9]+\.16b, v[0-9]+\.16b
> +** clz v[0-9]+\.16b, v[0-9]+\.16b
> +** str q[0-9]+, \[x[0-9]+\]
My only suggestion is to write this testcase so that the register
numbers for outputs are captured and then inputs then use that a
captured one.
Otherwise ok.
> +** ret
> +*/
> +void
> +ctzb (u8 *__restrict d, u8 *__restrict a)
> +{
> + for (int i = 0; i < 16; i++)
> + d[i] = __builtin_ctzg ((u8) (a[i] | 0x80));
> +}
> +
> +/* The same count in a variable-length loop must use RBIT and CLZ too. */
> +void
> +ctzb_n (u8 *__restrict d, u8 *__restrict a, int n)
> +{
> + for (int i = 0; i < n; i++)
> + d[i] = __builtin_ctzg (a[i], 8);
> +}
> +
> +/* { dg-final { scan-assembler-times {\trbit\tv[0-9]+\.16b, v[0-9]+\.16b} 2
> } } */
> +/* { dg-final { scan-assembler-times {\tclz\tv[0-9]+\.16b, v[0-9]+\.16b} 2 }
> } */
> diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
> b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
> new file mode 100644
> index 00000000000..a5be943f955
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
> @@ -0,0 +1,32 @@
> +/* { dg-do run } */
> +/* { dg-options "-O3" } */
> +
> +#define N 61
> +static unsigned char a[N], d[N], e[N];
> +
> +__attribute__((noipa)) void
> +ctzb (unsigned char *restrict r, unsigned char *restrict x, int n)
> +{
> + for (int i = 0; i < n; i++)
> + r[i] = __builtin_ctzg (x[i], 8);
> +}
> +
> +__attribute__((noipa, optimize ("O0"))) void
> +ctzb_ref (unsigned char *restrict r, unsigned char *restrict x, int n)
> +{
> + for (int i = 0; i < n; i++)
> + r[i] = __builtin_ctzg (x[i], 8);
> +}
> +
> +int
> +main (void)
> +{
> + for (int i = 0; i < N; i++)
> + a[i] = (unsigned char) (i * 37 + (i & 7));
> + ctzb (d, a, N);
> + ctzb_ref (e, a, N);
> + for (int i = 0; i < N; i++)
> + if (d[i] != e[i])
> + __builtin_abort ();
> + return 0;
> +}
> --
> 2.50.1 (Apple Git-155)
>