Account for the real Adv.SIMD cost of SAD and ABD reductions.
For SAD_EXPR, Adv.SIMD has to build a 128-bit result from two 64-bit
halves. Account for that extra work when costing 128-bit Adv.SIMD
vectors.
For SAD-style ABD reductions, SVE can keep the byte absolute difference as
uabd and then use udot for the widening sum. Without dot-product support,
the Adv.SIMD fallback is a longer dependent sequence, using uabdl/uabal
followed by the widening reduction. Charge that extra Adv.SIMD work in
both the statement cost and the issue model, so scalar profitability and
vector-loop comparison stay in sync.
This recovers costing fallouts from patch 1 in the series where the costs
makes us pick Adv. SIMD in cases where SVE would be much faster.
Bootstrapped Regtested on aarch64-none-linux-gnu and no issues.
OK for master?
Thanks,
Tamar
gcc/ChangeLog:
* config/aarch64/aarch64.cc (aarch64_ifn_vect_stmt_p): New function.
(aarch64_adjust_stmt_cost): Account for extra Adv.SIMD SAD and ABD
reduction costs.
(aarch64_vector_costs::count_ops): Likewise.
---
diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
index
c168941be4e173d29bb864d937b4f80b02931f62..3e1fc9bac5fe6f24773854db2b310421a3864bb5
100644
--- a/gcc/config/aarch64/aarch64.cc
+++ b/gcc/config/aarch64/aarch64.cc
@@ -18447,6 +18447,74 @@ aarch64_sve_adjust_stmt_cost (class vec_info *vinfo,
vect_cost_for_stmt kind,
return stmt_cost;
}
+/* Return > 0 if STMT_INFO is the instruction denoted by CODE and return the
+ number of instructions the given CODE is implemented using. */
+static unsigned int
+aarch64_ifn_vect_stmt_p (stmt_vec_info stmt_info, tree vectype,
+ unsigned int vec_flags)
+{
+ stmt_info = vect_stmt_to_vectorize (stmt_info);
+ gcall *call = dyn_cast<gcall *> (STMT_VINFO_STMT (stmt_info));
+ if (!vectype)
+ return 0;
+
+ gassign *assign = dyn_cast <gassign *> (STMT_VINFO_STMT (stmt_info));
+ if (!call && !assign)
+ return 0;
+
+ bool advsimd_p = vec_flags & VEC_ADVSIMD;
+ bool is_64bit_p = known_gt (GET_MODE_BITSIZE (TYPE_MODE (vectype)), 64);
+ if (assign)
+ {
+ switch (gimple_assign_rhs_code (assign))
+ {
+ /* SAD for Adv. SIMD is emulated using two instuctions per 64-bit
+ quantities. So 128-bit ADB requires 4 INSN. Account for that. */
+ case SAD_EXPR:
+ return advsimd_p && is_64bit_p ? 2 : 0;
+ case WIDEN_SUM_EXPR:
+ {
+ tree rhs = gimple_assign_rhs1 (assign);
+ if (!advsimd_p
+ || TARGET_DOTPROD
+ || !vect_is_reduction (stmt_info)
+ || TREE_CODE (rhs) != SSA_NAME)
+ return 0;
+
+ gimple *def_stmt = SSA_NAME_DEF_STMT (rhs);
+ gassign *def_assign = dyn_cast<gassign *> (def_stmt);
+ if (def_assign
+ && CONVERT_EXPR_CODE_P (gimple_assign_rhs_code (def_assign))
+ && TREE_CODE (gimple_assign_rhs1 (def_assign)) == SSA_NAME)
+ def_stmt = SSA_NAME_DEF_STMT (gimple_assign_rhs1 (def_assign));
+
+ gcall *def = dyn_cast<gcall *> (def_stmt);
+ if (!def || gimple_call_combined_fn (def) != CFN_ABD)
+ return 0;
+
+ for (unsigned int i = 0; i < 2; ++i)
+ {
+ tree arg = gimple_call_arg (def, i);
+ if (TREE_CODE (arg) != SSA_NAME
+ || !gimple_assign_load_p (SSA_NAME_DEF_STMT (arg)))
+ return 0;
+ }
+ return 2;
+ }
+ default:
+ break;
+ }
+ return 0;
+ }
+
+ switch (gimple_call_combined_fn (call))
+ {
+ default:
+ break;
+ }
+ return 0;
+}
+
/* STMT_COST is the cost calculated for STMT_INFO, which has cost kind KIND
and which when vectorized would operate on vector type VECTYPE. Add the
cost of any embedded operations. */
@@ -18572,6 +18640,14 @@ aarch64_vector_costs::count_ops (unsigned int count,
vect_cost_for_stmt kind,
return;
}
+ unsigned int n_insn = 0;
+ if (stmt_info
+ && kind == vector_stmt
+ && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info,
+ STMT_VINFO_VECTYPE (stmt_info),
+ m_vec_flags)))
+ ops->general_ops += n_insn * count;
+
/* Detect the case where we are using an emulated gather/scatter. When a
target does not support gathers and scatters directly the vectorizer
emulates these by constructing an index vector and then issuing an
@@ -19151,6 +19227,15 @@ aarch64_vector_costs::add_stmt_cost (int count,
vect_cost_for_stmt kind,
to the base cost calculated above. */
stmt_cost = aarch64_adjust_stmt_cost (m_vinfo, kind, stmt_info, node,
vectype, m_vec_flags, stmt_cost);
+ unsigned int n_insn = 0;
+ if (vectype
+ && kind == vector_stmt
+ && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info, vectype,
+ m_vec_flags)))
+ {
+ const simd_vec_cost *simd_costs = aarch64_simd_vec_costs (vectype);
+ stmt_cost += count * n_insn * simd_costs->int_stmt_cost;
+ }
/* If we're applying the SVE vs. Advanced SIMD unrolling heuristic,
estimate the number of statements in the unrolled Advanced SIMD
--
diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
index c168941be4e173d29bb864d937b4f80b02931f62..3e1fc9bac5fe6f24773854db2b310421a3864bb5 100644
--- a/gcc/config/aarch64/aarch64.cc
+++ b/gcc/config/aarch64/aarch64.cc
@@ -18447,6 +18447,74 @@ aarch64_sve_adjust_stmt_cost (class vec_info *vinfo, vect_cost_for_stmt kind,
return stmt_cost;
}
+/* Return > 0 if STMT_INFO is the instruction denoted by CODE and return the
+ number of instructions the given CODE is implemented using. */
+static unsigned int
+aarch64_ifn_vect_stmt_p (stmt_vec_info stmt_info, tree vectype,
+ unsigned int vec_flags)
+{
+ stmt_info = vect_stmt_to_vectorize (stmt_info);
+ gcall *call = dyn_cast<gcall *> (STMT_VINFO_STMT (stmt_info));
+ if (!vectype)
+ return 0;
+
+ gassign *assign = dyn_cast <gassign *> (STMT_VINFO_STMT (stmt_info));
+ if (!call && !assign)
+ return 0;
+
+ bool advsimd_p = vec_flags & VEC_ADVSIMD;
+ bool is_64bit_p = known_gt (GET_MODE_BITSIZE (TYPE_MODE (vectype)), 64);
+ if (assign)
+ {
+ switch (gimple_assign_rhs_code (assign))
+ {
+ /* SAD for Adv. SIMD is emulated using two instuctions per 64-bit
+ quantities. So 128-bit ADB requires 4 INSN. Account for that. */
+ case SAD_EXPR:
+ return advsimd_p && is_64bit_p ? 2 : 0;
+ case WIDEN_SUM_EXPR:
+ {
+ tree rhs = gimple_assign_rhs1 (assign);
+ if (!advsimd_p
+ || TARGET_DOTPROD
+ || !vect_is_reduction (stmt_info)
+ || TREE_CODE (rhs) != SSA_NAME)
+ return 0;
+
+ gimple *def_stmt = SSA_NAME_DEF_STMT (rhs);
+ gassign *def_assign = dyn_cast<gassign *> (def_stmt);
+ if (def_assign
+ && CONVERT_EXPR_CODE_P (gimple_assign_rhs_code (def_assign))
+ && TREE_CODE (gimple_assign_rhs1 (def_assign)) == SSA_NAME)
+ def_stmt = SSA_NAME_DEF_STMT (gimple_assign_rhs1 (def_assign));
+
+ gcall *def = dyn_cast<gcall *> (def_stmt);
+ if (!def || gimple_call_combined_fn (def) != CFN_ABD)
+ return 0;
+
+ for (unsigned int i = 0; i < 2; ++i)
+ {
+ tree arg = gimple_call_arg (def, i);
+ if (TREE_CODE (arg) != SSA_NAME
+ || !gimple_assign_load_p (SSA_NAME_DEF_STMT (arg)))
+ return 0;
+ }
+ return 2;
+ }
+ default:
+ break;
+ }
+ return 0;
+ }
+
+ switch (gimple_call_combined_fn (call))
+ {
+ default:
+ break;
+ }
+ return 0;
+}
+
/* STMT_COST is the cost calculated for STMT_INFO, which has cost kind KIND
and which when vectorized would operate on vector type VECTYPE. Add the
cost of any embedded operations. */
@@ -18572,6 +18640,14 @@ aarch64_vector_costs::count_ops (unsigned int count, vect_cost_for_stmt kind,
return;
}
+ unsigned int n_insn = 0;
+ if (stmt_info
+ && kind == vector_stmt
+ && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info,
+ STMT_VINFO_VECTYPE (stmt_info),
+ m_vec_flags)))
+ ops->general_ops += n_insn * count;
+
/* Detect the case where we are using an emulated gather/scatter. When a
target does not support gathers and scatters directly the vectorizer
emulates these by constructing an index vector and then issuing an
@@ -19151,6 +19227,15 @@ aarch64_vector_costs::add_stmt_cost (int count, vect_cost_for_stmt kind,
to the base cost calculated above. */
stmt_cost = aarch64_adjust_stmt_cost (m_vinfo, kind, stmt_info, node,
vectype, m_vec_flags, stmt_cost);
+ unsigned int n_insn = 0;
+ if (vectype
+ && kind == vector_stmt
+ && (n_insn = aarch64_ifn_vect_stmt_p (stmt_info, vectype,
+ m_vec_flags)))
+ {
+ const simd_vec_cost *simd_costs = aarch64_simd_vec_costs (vectype);
+ stmt_cost += count * n_insn * simd_costs->int_stmt_cost;
+ }
/* If we're applying the SVE vs. Advanced SIMD unrolling heuristic,
estimate the number of statements in the unrolled Advanced SIMD