From: Kyrylo Tkachov <[email protected]>

RBIT has a .8B and .16B arrangement and CLZ has a .8B and .16B arrangement,
so a byte-element count-trailing-zeros is two instructions.  ctz<mode>2 only
covered V2SI and V4SI, so a loop over unsigned char was expanded by the
middle end into the generic negate/and/clz/subtract sequence:

  before                                after

  mvni  v31.4s, 0                       ldr     q31, [x1, x3]
  movi  v30.16b, 8                      rbit    v31.16b, v31.16b
.L4:                                    clz     v31.16b, v31.16b
  ldr   q29, [x1, x3]                   str     q31, [x0, x3]
  add   v28.16b, v29.16b, v31.16b
  bic   v29.16b, v28.16b, v29.16b
  clz   v29.16b, v29.16b
  sub   v29.16b, v30.16b, v29.16b
  str   v29.16b, [x0, x3]

The halfword modes are left alone.  Bit-reversing a halfword needs REV16 as
well as RBIT, so the sequence would be three instructions against the four
of the generic one, which is not enough of a difference to be worth the
extra pattern.

Bootstrapped and tested on aarch64-none-linux-gnu.
Ok for trunk?
Thanks,
Kyrill

gcc/ChangeLog:

        * config/aarch64/aarch64-simd.md (ctz<mode>2): New expander for VB.

gcc/testsuite/ChangeLog:

        * gcc.target/aarch64/vect-ctz-1.c: New test.
        * gcc.target/aarch64/vect-ctz-2.c: New test.

Signed-off-by: Kyrylo Tkachov <[email protected]>
---
 gcc/config/aarch64/aarch64-simd.md            | 11 ++++++
 gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c | 35 +++++++++++++++++++
 gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c | 32 +++++++++++++++++
 3 files changed, 78 insertions(+)
 create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
 create mode 100644 gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c

diff --git a/gcc/config/aarch64/aarch64-simd.md 
b/gcc/config/aarch64/aarch64-simd.md
index 527efe94084..1fd990be50f 100644
--- a/gcc/config/aarch64/aarch64-simd.md
+++ b/gcc/config/aarch64/aarch64-simd.md
@@ -517,6 +517,17 @@
   [(set_attr "type" "neon_rbit")]
 )
 
+(define_expand "ctz<mode>2"
+  [(set (match_operand:VB 0 "register_operand")
+       (ctz:VB (match_operand:VB 1 "register_operand")))]
+  "TARGET_SIMD"
+  {
+     emit_insn (gen_aarch64_rbit<mode> (operands[0], operands[1]));
+     emit_insn (gen_clz<mode>2 (operands[0], operands[0]));
+     DONE;
+  }
+)
+
 (define_expand "ctz<mode>2"
   [(set (match_operand:VS 0 "register_operand")
         (ctz:VS (match_operand:VS 1 "register_operand")))]
diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c 
b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
new file mode 100644
index 00000000000..70d89d3dfa5
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-1.c
@@ -0,0 +1,35 @@
+/* { dg-do compile } */
+/* { dg-options "-O3 -fno-schedule-insns -fno-schedule-insns2" } */
+/* { dg-final { check-function-bodies "**" "" } } */
+
+typedef __UINT8_TYPE__ u8;
+
+/* The OR keeps the input nonzero so that the loop is just the count.  */
+/*
+** ctzb:
+**     ...
+**     movi    v[0-9]+\.16b, 0xffffffffffffff80
+**     ldr     q[0-9]+, \[x[0-9]+\]
+**     orr     v[0-9]+\.16b, v[0-9]+\.16b, v[0-9]+\.16b
+**     rbit    v[0-9]+\.16b, v[0-9]+\.16b
+**     clz     v[0-9]+\.16b, v[0-9]+\.16b
+**     str     q[0-9]+, \[x[0-9]+\]
+**     ret
+*/
+void
+ctzb (u8 *__restrict d, u8 *__restrict a)
+{
+  for (int i = 0; i < 16; i++)
+    d[i] = __builtin_ctzg ((u8) (a[i] | 0x80));
+}
+
+/* The same count in a variable-length loop must use RBIT and CLZ too.  */
+void
+ctzb_n (u8 *__restrict d, u8 *__restrict a, int n)
+{
+  for (int i = 0; i < n; i++)
+    d[i] = __builtin_ctzg (a[i], 8);
+}
+
+/* { dg-final { scan-assembler-times {\trbit\tv[0-9]+\.16b, v[0-9]+\.16b} 2 } 
} */
+/* { dg-final { scan-assembler-times {\tclz\tv[0-9]+\.16b, v[0-9]+\.16b} 2 } } 
*/
diff --git a/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c 
b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
new file mode 100644
index 00000000000..a5be943f955
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/vect-ctz-2.c
@@ -0,0 +1,32 @@
+/* { dg-do run } */
+/* { dg-options "-O3" } */
+
+#define N 61
+static unsigned char a[N], d[N], e[N];
+
+__attribute__((noipa)) void
+ctzb (unsigned char *restrict r, unsigned char *restrict x, int n)
+{
+  for (int i = 0; i < n; i++)
+    r[i] = __builtin_ctzg (x[i], 8);
+}
+
+__attribute__((noipa, optimize ("O0"))) void
+ctzb_ref (unsigned char *restrict r, unsigned char *restrict x, int n)
+{
+  for (int i = 0; i < n; i++)
+    r[i] = __builtin_ctzg (x[i], 8);
+}
+
+int
+main (void)
+{
+  for (int i = 0; i < N; i++)
+    a[i] = (unsigned char) (i * 37 + (i & 7));
+  ctzb (d, a, N);
+  ctzb_ref (e, a, N);
+  for (int i = 0; i < N; i++)
+    if (d[i] != e[i])
+      __builtin_abort ();
+  return 0;
+}
-- 
2.50.1 (Apple Git-155)

Reply via email to