From: Kyrylo Tkachov <[email protected]>
aarch64_expand_vec_perm_const_1 normalizes a permutation whose first index
selects the second operand by rotating the indices and swapping op0 and op1.
It left zero_op0_p and zero_op1_p pointing at the old operands, so the later
recognizers that consult them, aarch64_evpc_and and aarch64_evpc_tbl, read
the wrong vector.
Swap the two flags together with the operands.
For
typedef int v4si __attribute__ ((vector_size (16)));
v4si f (v4si x)
{
const v4si m = { 4, 1, 2, 3 };
return __builtin_shuffle (x, (v4si) { 0, 0, 0, 0 }, m);
}
at -O0 the AND was applied to the all-zero operand, so the function returned
{0,0,0,0} instead of {0,x1,x2,x3}:
sub sp, sp, #32
str q0, [sp]
adrp x0, .LC0
add x0, x0, :lo12:.LC0
ldr q31, [x0]
str q31, [sp, 16]
movi v31.4s, 0
fmov s31, s31
mov v0.16b, v31.16b
add sp, sp, 32
ret
With the fix the AND is applied to the incoming vector:
sub sp, sp, #32
str q0, [sp]
adrp x0, .LC0
add x0, x0, :lo12:.LC0
ldr q31, [x0]
str q31, [sp, 16]
ldr q30, [sp]
adrp x0, .LC1
add x0, x0, :lo12:.LC1
ldr q31, [x0]
and v31.16b, v30.16b, v31.16b
mov v0.16b, v31.16b
add sp, sp, 32
ret
Bootstrapped and tested on aarch64-none-linux-gnu.
Pusing to trunk.
Thanks,
Kyrill
gcc/ChangeLog:
PR target/126597
* config/aarch64/aarch64.cc (aarch64_expand_vec_perm_const_1): Swap
zero_op0_p and zero_op1_p along with the operands.
gcc/testsuite/ChangeLog:
PR target/126597
* gcc.target/aarch64/pr126597.c: New test.
Signed-off-by: Kyrylo Tkachov <[email protected]>
---
gcc/config/aarch64/aarch64.cc | 1 +
gcc/testsuite/gcc.target/aarch64/pr126597.c | 21 +++++++++++++++++++++
2 files changed, 22 insertions(+)
create mode 100644 gcc/testsuite/gcc.target/aarch64/pr126597.c
diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
index 2e440ce916d..f36864a10da 100644
--- a/gcc/config/aarch64/aarch64.cc
+++ b/gcc/config/aarch64/aarch64.cc
@@ -28812,6 +28812,7 @@ aarch64_expand_vec_perm_const_1 (struct
expand_vec_perm_d *d)
{
d->perm.rotate_inputs (1);
std::swap (d->op0, d->op1);
+ std::swap (d->zero_op0_p, d->zero_op1_p);
}
if (((d->vec_flags == VEC_ADVSIMD && TARGET_SIMD)
diff --git a/gcc/testsuite/gcc.target/aarch64/pr126597.c
b/gcc/testsuite/gcc.target/aarch64/pr126597.c
new file mode 100644
index 00000000000..922d673fa6a
--- /dev/null
+++ b/gcc/testsuite/gcc.target/aarch64/pr126597.c
@@ -0,0 +1,21 @@
+/* { dg-do run } */
+/* { dg-options "-O0" } */
+
+typedef int v4si __attribute__ ((vector_size (16)));
+
+__attribute__((noipa)) v4si
+f (v4si x)
+{
+ const v4si mask = { 4, 1, 2, 3 };
+ return __builtin_shuffle (x, (v4si) { 0, 0, 0, 0 }, mask);
+}
+
+int
+main (void)
+{
+ v4si x = { 1, 2, 3, 4 };
+ v4si r = f (x);
+ if (r[0] != 0 || r[1] != 2 || r[2] != 3 || r[3] != 4)
+ __builtin_abort ();
+ return 0;
+}
--
2.50.1 (Apple Git-155)