When VECTOR_BOOLEAN_TYPE_P have integer mode then we need the original
scalar type that was used to derive it for determining the mask
precision of a statement. append_pattern_def_seq already has means
to get this, but the build_mask_conversion and vect_convert_mask_for_vectype
lack that and thus suffer from bogus mask_precision values and in the
end failed vectorization.
Bootstrapped and tested on x86_64-unknown-linux-gnu, pushed.
PR tree-optimization/126789
* tree-vect-patterns.cc (build_mask_conversion): Add
scalar_type_for_mask parameter and pass it along.
(vect_convert_mask_for_vectype): Likewise.
(vect_recog_bool_pattern): Adjust.
(vect_recog_mask_conversion_pattern): Likewise.
* gcc.target/i386/vect-pr126789.c: New testcase.
---
gcc/testsuite/gcc.target/i386/vect-pr126789.c | 20 +++++++++++++
gcc/tree-vect-patterns.cc | 30 +++++++++++++------
2 files changed, 41 insertions(+), 9 deletions(-)
create mode 100644 gcc/testsuite/gcc.target/i386/vect-pr126789.c
diff --git a/gcc/testsuite/gcc.target/i386/vect-pr126789.c
b/gcc/testsuite/gcc.target/i386/vect-pr126789.c
new file mode 100644
index 00000000000..0378e4e8fc4
--- /dev/null
+++ b/gcc/testsuite/gcc.target/i386/vect-pr126789.c
@@ -0,0 +1,20 @@
+/* { dg-do compile } */
+/* { dg-options "-O2 -mavx512bw -mavx512vl -fno-vect-cost-model" } */
+
+int foo (double g, int f, double *r, int *s)
+{
+ int hu = 0;
+ bool test0 = r[0] < g;
+ bool test1 = r[1] < g;
+ bool test2 = r[2] < g;
+ bool test3 = r[3] < g;
+ bool test4 = s[0] < f;
+ bool test5 = s[1] < f;
+ bool test6 = s[2] < f;
+ bool test7 = s[3] < f;
+ hu += (test0 & test4) + (test1 & test5) + (test2 & test6) + (test3 & test7);
+ return hu;
+}
+
+/* { dg-final { scan-assembler "vcmppd" } } */
+/* { dg-final { scan-assembler "vpcmpd" } } */
diff --git a/gcc/tree-vect-patterns.cc b/gcc/tree-vect-patterns.cc
index b921ae94848..fd1232cc755 100644
--- a/gcc/tree-vect-patterns.cc
+++ b/gcc/tree-vect-patterns.cc
@@ -5917,21 +5917,28 @@ vect_recog_gcond_pattern (vec_info *vinfo,
conversion of MASK to a type suitable for masking VECTYPE.
Built statement gets required vectype and is appended to
a pattern sequence of STMT_VINFO.
+ If VECTYPE is a mask type, SCALAR_TYPE_FOR_MASK is the scalar type
+ from which it was derived.
Return converted mask. */
static tree
build_mask_conversion (vec_info *vinfo,
- tree mask, tree vectype, stmt_vec_info stmt_vinfo)
+ tree mask, tree vectype, stmt_vec_info stmt_vinfo,
+ tree scalar_type_for_mask = NULL_TREE)
{
gimple *stmt;
tree masktype, tmp;
+ gcc_assert (!scalar_type_for_mask == !VECTOR_BOOLEAN_TYPE_P (vectype));
+
masktype = truth_type_for (vectype);
tmp = vect_recog_temp_ssa_var (TREE_TYPE (masktype), NULL);
stmt = gimple_build_assign (tmp, CONVERT_EXPR, mask);
append_pattern_def_seq (vinfo, stmt_vinfo,
- stmt, masktype, TREE_TYPE (vectype));
+ stmt, masktype,
+ scalar_type_for_mask
+ ? scalar_type_for_mask : TREE_TYPE (vectype));
return tmp;
}
@@ -5940,11 +5947,13 @@ build_mask_conversion (vec_info *vinfo,
/* Return MASK if MASK is suitable for masking an operation on vectors
of type VECTYPE, otherwise convert it into such a form and return
the result. Associate any conversion statements with STMT_INFO's
- pattern. */
+ pattern. If VECTYPE is a mask type, SCALAR_TYPE_FOR_MASK is the scalar
+ type from which it was derived. */
static tree
vect_convert_mask_for_vectype (tree mask, tree vectype,
- stmt_vec_info stmt_info, vec_info *vinfo)
+ stmt_vec_info stmt_info, vec_info *vinfo,
+ tree scalar_type_for_mask = NULL_TREE)
{
tree mask_type = integer_type_for_mask (mask, vinfo);
if (mask_type)
@@ -5953,7 +5962,8 @@ vect_convert_mask_for_vectype (tree mask, tree vectype,
if (mask_vectype
&& maybe_ne (TYPE_VECTOR_SUBPARTS (vectype),
TYPE_VECTOR_SUBPARTS (mask_vectype)))
- mask = build_mask_conversion (vinfo, mask, vectype, stmt_info);
+ mask = build_mask_conversion (vinfo, mask, vectype, stmt_info,
+ scalar_type_for_mask);
}
return mask;
}
@@ -6191,7 +6201,7 @@ vect_recog_bool_pattern (vec_info *vinfo,
append_pattern_def_seq (vinfo, stmt_vinfo, pattern_stmt,
new_vectype, TREE_TYPE (new_vectype));
rhs2 = vect_convert_mask_for_vectype (tem, rhs1_vectype,
- stmt_vinfo, vinfo);
+ stmt_vinfo, vinfo, rhs1_type);
}
else if (!rhs1_type && rhs2_type)
{
@@ -6210,7 +6220,7 @@ vect_recog_bool_pattern (vec_info *vinfo,
append_pattern_def_seq (vinfo, stmt_vinfo, pattern_stmt,
new_vectype, TREE_TYPE (new_vectype));
var = vect_convert_mask_for_vectype (tem, rhs2_vectype,
- stmt_vinfo, vinfo);
+ stmt_vinfo, vinfo, rhs2_type);
}
lhs = vect_recog_temp_ssa_var (TREE_TYPE (lhs), NULL);
pattern_stmt = gimple_build_assign (lhs, rhs_code, var, rhs2);
@@ -6442,14 +6452,16 @@ vect_recog_mask_conversion_pattern (vec_info *vinfo,
vectype1 = get_mask_type_for_scalar_type (vinfo, rhs1_type);
if (!vectype1)
return NULL;
- rhs2 = build_mask_conversion (vinfo, rhs2, vectype1, stmt_vinfo);
+ rhs2 = build_mask_conversion (vinfo, rhs2, vectype1, stmt_vinfo,
+ rhs2_type);
}
else
{
vectype1 = get_mask_type_for_scalar_type (vinfo, rhs2_type);
if (!vectype1)
return NULL;
- rhs1 = build_mask_conversion (vinfo, rhs1, vectype1, stmt_vinfo);
+ rhs1 = build_mask_conversion (vinfo, rhs1, vectype1, stmt_vinfo,
+ rhs2_type);
}
lhs = vect_recog_temp_ssa_var (TREE_TYPE (lhs), NULL);
--
2.51.0